Skip to main content

subc_daemon/
supervise.rs

1use std::{
2    collections::{HashMap, HashSet, VecDeque},
3    error::Error,
4    fmt, io,
5    path::PathBuf,
6    process::{ExitStatus, Stdio},
7    sync::{Arc, Mutex, OnceLock},
8    time::{Duration, SystemTime, UNIX_EPOCH},
9};
10
11use cortexkit_log::Retention;
12use serde_json::Value;
13use subc_control::{
14    ClientControlPush, LiveSpawn, ModuleProtocol, RouteCloseReason, SpawnCursor, SpawnEvent,
15    SpawnEventKind, SpawnSnapshot, SupervisorHealthStatus, TerminalDisposition, TerminalExitKind,
16};
17use subc_protocol::{
18    manifest::{SelfSignalKind, SignalAnchor},
19    session::{
20        HealthReport, HealthStatus, ModuleControlCommand, ModuleControlRequest,
21        MODULE_CONTROL_OP_HEALTH_CHECK,
22    },
23    Flags, FrameType, Priority, SUBC_LAUNCH_NONCE_ENV, SUBC_MODULE_ID_ENV,
24};
25use tokio::{
26    process::{Child, Command},
27    sync::{mpsc, oneshot, watch, Mutex as AsyncMutex},
28    task::JoinHandle,
29    time::{sleep, sleep_until, timeout, timeout_at, Instant},
30};
31use tracing::{debug, error, info, warn};
32
33use crate::{
34    child_roster::ChildRoster,
35    daemon_config::{
36        CAPTURE_KEEP_ENV, CAPTURE_MAX_AGE_DAYS_ENV, CAPTURE_MAX_FILE_MB_ENV, CK_LOG_ENV,
37    },
38    forwarding::{
39        CloseReason, ForwardingError, ForwardingTable, GoodbyeTarget, ModuleControlRpcOutcome,
40        ModuleDrainTarget, PendingModuleControlRpc,
41    },
42    provenance::{spawned_file_identity, ExecutableIdentityProbe, SpawnedFileIdentity},
43    registry::{ConnectionId, RegistryError},
44    stderr_tail::{
45        pump_stderr_to, pump_stdout_to, ChildOutputSink, StderrRing, StderrTailConfig,
46        StderrTailSnapshot,
47    },
48    terminal_ring::{TerminalHistorySnapshot, TerminalRecord, TerminalRing, TerminalRingConfig},
49    Frame, FrameSink, Registry,
50};
51
52#[path = "supervise_swap.rs"]
53mod swap;
54
55/// Command-line flag used by supervised modules to find subc.
56///
57/// subc launches module-mode children as `<module> --subc <connection-file-path>`.
58/// The path points at the TCP+key connection file; it is not an ambient signal and
59/// is never inherited by standalone children.
60pub const SUBC_ARG: &str = "--subc";
61
62const DEFAULT_MAX_RESTARTS: u32 = 3;
63const DEFAULT_BACKOFF: Duration = Duration::from_millis(100);
64const DEFAULT_MAX_BACKOFF: Duration = Duration::from_secs(30);
65/// The span `DEFAULT_MAX_RESTARTS` is counted over. Ten minutes is long enough
66/// to contain a real crash loop (which respawns in seconds) and short enough
67/// that unrelated crashes hours apart never accumulate into a permanent stop.
68const DEFAULT_RESTART_WINDOW: Duration = Duration::from_secs(600);
69/// How long a drain waits for already-dispatched requests to finalize before
70/// the child is torn down. Sized for TOOL-SCALE work (bash, inspect, builds),
71/// not RPC-scale: the original 2s value silently cut nearly every real tool
72/// call at the fence, making the wait-for-finalize design decorative for the
73/// workloads it existed for. Quiescence short-circuits, so an idle module
74/// restarts immediately regardless of this value; the budget is spent only
75/// when a genuine in-flight request is worth finishing. Per-module override:
76/// `drain_timeout_ms` in subc.jsonc; per-restart override: the operator's
77/// `supervisor.restart{drain_timeout_ms}` (0 = cut now, for wedge bounces
78/// where a stuck request will never settle).
79pub const DEFAULT_DRAIN_TIMEOUT: Duration = Duration::from_secs(30);
80const REGISTRY_RELEASE_TIMEOUT: Duration = Duration::from_secs(1);
81const REGISTRY_RELEASE_POLL: Duration = Duration::from_millis(10);
82/// How long a restart waits for the exited process's output readers.
83///
84/// The restart does not depend on the stderr reader finishing. A reader still
85/// running at this bound is left running, and whatever it delivers later goes
86/// into the exited process's own section of the stderr ring (see
87/// `StderrRing::push_line_from`), ending naturally at EOF on its pipe. So the
88/// bound no longer decides whether a crash's last lines are kept: under load
89/// the reader may simply not have been scheduled yet, and cutting it there
90/// discarded exactly the lines that explained the crash.
91///
92/// What the bound still decides is when the tail starts reporting
93/// `Incomplete`: a pipe open past it usually means a descendant of the exited
94/// process still holds it, and the tail cannot claim to be whole until that
95/// pipe closes. That is also the only case in which the wait costs the restart
96/// anything, because a pipe with no other holder reaches EOF when the process
97/// exits. Under load a slow reader can show `Incomplete` briefly; it returns to
98/// `Captured` at EOF with nothing lost.
99///
100/// The stdout reader carries no ring, only the capture file, and is still
101/// stopped at this bound so an old process's stdout cannot trail into the file
102/// after its successor starts.
103const STDERR_PUMP_DRAIN_TIMEOUT: Duration = Duration::from_millis(250);
104/// Maximum number of supervised process spawn/exit facts retained per daemon incarnation.
105pub const SPAWN_EVENT_RING_CAPACITY: usize = 4096;
106const SPAWN_SUBSCRIBER_BUFFER: usize = SPAWN_EVENT_RING_CAPACITY + 1;
107/// Terminal Error code for a `supervisor.spawn_subscribe` stream the daemon
108/// dropped because the subscriber stopped draining its frames. The detail's
109/// `first_undelivered_cursor` names the first event it did not receive; the
110/// client resubscribes from the last cursor it did receive.
111pub(crate) const SPAWN_SUBSCRIBER_LAGGED_CODE: &str = "spawn_subscriber_lagged";
112
113#[cfg(target_os = "macos")]
114struct PrivacyExec {
115    reader: tokio::io::unix::AsyncFd<std::io::PipeReader>,
116    deadline: tokio::time::Instant,
117    expected: Option<subc_os::FileIdentity>,
118    trampoline: Option<subc_os::FileIdentity>,
119    script: bool,
120    module_id: String,
121}
122
123#[cfg(all(test, target_os = "macos"))]
124pub(crate) fn test_privacy_trampoline() -> PathBuf {
125    // Cargo's unit-test executable lives in <profile>/deps; its fixture bin
126    // lives beside that directory. This honors custom CARGO_TARGET_DIR too.
127    let path = std::env::current_exe()
128        .unwrap()
129        .parent()
130        .unwrap()
131        .parent()
132        .unwrap()
133        .join("privacy-trampoline-fixture");
134    // Without the fixture every macOS spawn is refused, and the tests that
135    // spawn fail later as a module in state Failed, which names the wrong
136    // cause. `cargo test -p subc-daemon --lib` alone does not build it.
137    assert!(
138        path.exists(),
139        "privacy-trampoline-fixture not built at {}: run `cargo build -p subc-daemon \
140         --bins --features test-support` or `cargo test -p subc-daemon` first",
141        path.display()
142    );
143    path
144}
145
146#[cfg(target_os = "macos")]
147fn probe_privacy_trampoline(path: &std::path::Path) -> Result<(), String> {
148    use std::io::Read;
149    let mut probe = std::process::Command::new(path)
150        .args(["__disclaim-exec", "--probe"])
151        .stdin(Stdio::null())
152        .stdout(Stdio::piped())
153        .stderr(Stdio::piped())
154        .spawn()
155        .map_err(|error| {
156            format!(
157                "privacy trampoline probe failed for {}: {error}",
158                path.display()
159            )
160        })?;
161    let deadline = std::time::Instant::now() + Duration::from_secs(5);
162    let status = loop {
163        match probe.try_wait() {
164            Ok(Some(status)) => break status,
165            Ok(None) if std::time::Instant::now() < deadline => {
166                std::thread::sleep(Duration::from_millis(5))
167            }
168            result => {
169                let _ = probe.kill();
170                let _ = probe.wait();
171                return Err(format!(
172                    "privacy trampoline probe failed or timed out for {}: {result:?}",
173                    path.display()
174                ));
175            }
176        }
177    };
178    let mut answer = String::new();
179    if let Some(stdout) = probe.stdout.take() {
180        let _ = stdout.take(256).read_to_string(&mut answer);
181    }
182    if status.success() && answer.trim() == subc_os::privacy_identity::TRAMPOLINE_PROBE {
183        return Ok(());
184    }
185    let mut diagnostic = String::new();
186    if let Some(stderr) = probe.stderr.take() {
187        let _ = stderr.take(1024).read_to_string(&mut diagnostic);
188    }
189    let cause = diagnostic
190        .trim()
191        .strip_prefix("ck-subc: own privacy identity refused: ")
192        .unwrap_or("binary does not implement the privacy trampoline protocol");
193    Err(format!(
194        "{cause}: probe of {} exited {status}",
195        path.display()
196    ))
197}
198
199#[cfg(target_os = "macos")]
200fn privacy_command(
201    spec: &ModuleSpec,
202    roster: &ChildRoster,
203) -> Result<
204    (
205        Command,
206        Option<PrivacyExec>,
207        subc_os::privacy_identity::ExecAcknowledgement,
208    ),
209    SuperviseError,
210> {
211    let failure = |cause: String| {
212        warn!(module_id = %spec.module_id, %cause, "privacy identity trampoline refused module spawn");
213        SuperviseError::Spawn {
214            program: spec.program.clone(),
215            source: io::Error::other(cause),
216            cgroup_path: None,
217        }
218    };
219    let trampoline = roster.privacy_trampoline().map_err(failure)?;
220    // Resolve PATH with the same environment the Command will receive. For
221    // scripts retain the existing orphan-identity rule: the kernel chooses
222    // the interpreter, and its observed image is the one recorded. Do not
223    // duplicate the kernel's shebang/PATH interpreter resolution in Rust.
224    let program = if spec.program.components().count() == 1 && !spec.program.is_absolute() {
225        let path = spec
226            .env
227            .iter()
228            .find(|(key, _)| key == "PATH")
229            .map(|(_, value)| std::ffi::OsString::from(value))
230            .or_else(|| std::env::var_os("PATH"))
231            .unwrap_or_else(|| "/usr/bin:/bin".into());
232        std::env::split_paths(&path)
233            .map(|dir| dir.join(&spec.program))
234            .find(|path| path.is_file())
235            .unwrap_or_else(|| spec.program.clone())
236    } else {
237        spec.program.clone()
238    };
239    let expected = subc_os::file_identity(&program);
240    let trampoline_image = subc_os::file_identity(&trampoline);
241    let script = {
242        use std::io::Read;
243        let mut prefix = [0u8; 2];
244        std::fs::File::open(&program)
245            .is_ok_and(|mut file| file.read_exact(&mut prefix).is_ok() && &prefix == b"#!")
246    };
247    if expected.is_none() || expected == subc_os::file_identity(&trampoline) {
248        return Err(failure(
249            "privacy identity module executable is missing or is the trampoline itself".to_string(),
250        ));
251    }
252    let (reader, ack) = subc_os::privacy_identity::ExecAcknowledgement::pipe()
253        .map_err(|error| failure(error.to_string()))?;
254    let reader =
255        tokio::io::unix::AsyncFd::new(reader).map_err(|error| failure(error.to_string()))?;
256    let mut command = Command::new(&trampoline);
257    command
258        .arg("__disclaim-exec")
259        .arg(ack.fd().to_string())
260        .arg(&program);
261    ack.install(command.as_std_mut());
262    Ok((
263        command,
264        Some(PrivacyExec {
265            reader,
266            deadline: tokio::time::Instant::now() + Duration::from_secs(5),
267            expected,
268            trampoline: trampoline_image,
269            script,
270            module_id: spec.module_id.clone(),
271        }),
272        ack,
273    ))
274}
275
276struct SupervisedChild {
277    child: Child,
278    #[cfg(target_os = "macos")]
279    privacy_exec: Option<PrivacyExec>,
280    /// Set once this launch's exec acknowledgement confirms the module image.
281    /// On macOS the pid first runs the `ck-subc` launch trampoline (see
282    /// `subc_os::privacy_identity`), which then replaces itself with the
283    /// module. The supervisor owns and can kill that pid from spawn, but
284    /// status readers report it only after this latch is set, so nothing
285    /// reports the trampoline's image as the module's.
286    #[cfg(target_os = "macos")]
287    report_ready: Arc<OnceLock<()>>,
288    /// Refusal before the module image was accepted, retained for terminal records.
289    spawn_failure: Option<String>,
290    /// The protocol this process was launched with. A reload can store a new
291    /// launch spec with a different protocol, but that takes effect only at the
292    /// next spawn, so this process keeps being handled by the protocol it
293    /// actually speaks.
294    protocol: ModuleProtocol,
295    /// This process's cgroup name: a bounded module/slot label followed by a
296    /// spawn suffix unique to this process (when cgroup placement is on). A
297    /// retired process in a slot may still be draining when a later one is
298    /// spawned into that slot, so the suffix keeps the later process out of
299    /// the retired one's cgroup, which is the domain a kill applies to.
300    #[cfg(target_os = "linux")]
301    module_id: String,
302    #[cfg(target_os = "linux")]
303    cgroup_placement: Option<subc_cgroup::Placement>,
304    /// The job that contains this child and every process it spawns (issue #109).
305    ///
306    /// Dropping this handle is what reaps a surviving tree when no supervisor
307    /// code runs — a daemon crash — because the job carries
308    /// `JOB_OBJECT_LIMIT_KILL_ON_JOB_CLOSE`.
309    ///
310    /// That limit is not crash-only, and the difference is worth knowing: a
311    /// Windows daemon *stop* is `taskkill` or the scheduler's `/End` — the
312    /// SIGTERM handler is `#[cfg(unix)]` — so the daemon dies with no stop
313    /// notice and the kernel closes the job handle, `TerminateProcess`ing every
314    /// module at once. Before this change they survived that, saw EOF on the
315    /// control socket, and ran their own teardown; Unix keeps that path
316    /// deliberately, so a module can seal a WAL or close a capture rather than
317    /// be killed mid-write. So this trades graceful teardown on every Windows
318    /// daemon stop for containment on a crash, which is the right way round
319    /// today: orphaned GPU workers are a reported, recurring problem, and the
320    /// modules that write most heavily do not run on Windows.
321    ///
322    /// The fix is a real Windows stop path — the daemon draining before it
323    /// exits, the twin of the Unix SIGTERM handler. Once it exists, this limit
324    /// reaches only what the drain left behind, which is what it should reach.
325    #[cfg(windows)]
326    job: Option<subc_jobobject::JobObject>,
327    stdout_pump: Option<JoinHandle<()>>,
328    stderr_pump: Option<StderrPump>,
329    stderr_ring: Arc<Mutex<StderrRing>>,
330    spawned_at_ms: u64,
331    spawned_from: PathBuf,
332    spawned_file_identity: Option<SpawnedFileIdentity>,
333    process_start_time: Option<u64>,
334    process_identity: Option<ProcessIdentity>,
335    pid: u32,
336    /// This process's entry in the daemon's child roster, released when the
337    /// process is reaped or this handle is dropped.
338    roster_guard: Option<crate::child_roster::RosterGuard>,
339}
340
341impl SupervisedChild {
342    fn id(&self) -> Option<u32> {
343        Some(self.pid)
344    }
345
346    fn process_identity(&self) -> Option<ProcessIdentity> {
347        self.process_identity
348    }
349
350    async fn wait(&mut self) -> io::Result<ExitStatus> {
351        #[cfg(target_os = "macos")]
352        self.confirm_privacy_exec().await;
353        // The roster entry is NOT released here. A daemon shutdown waits for the
354        // roster to empty and then exits the process, so releasing at the reap
355        // let it exit before the exit handler wrote this child's terminal record
356        // (the stderr drain and snapshot update sit in between), and the
357        // shutdown's own `daemon_shutdown` record was intermittently lost. The
358        // caller releases it after recording the exit (`release_roster`), and
359        // dropping the handle releases it too.
360        let result = self.child.wait().await;
361        #[cfg(target_os = "linux")]
362        if result.is_ok() {
363            if let Some(placement) = self.cgroup_placement.as_ref() {
364                cleanup_reaped_cgroup(placement, &self.module_id).await;
365                // Keep ownership while awaiting kernel population changes: a
366                // drain timeout may cancel this wait and then escalate/reap.
367                self.cgroup_placement = None;
368            }
369        }
370        result
371    }
372
373    #[cfg(target_os = "macos")]
374    async fn confirm_privacy_exec(&mut self) {
375        let Some(pending) = &mut self.privacy_exec else {
376            return;
377        };
378        let result = tokio::time::timeout_at(pending.deadline, async {
379            let mut record = Vec::new();
380            loop {
381                let mut ready = pending.reader.readable().await?;
382                let read = ready.try_io(|reader| {
383                    use std::io::Read;
384                    let mut reader = reader.get_ref();
385                    let mut buffer = [0u8; 256];
386                    reader.read(&mut buffer).map(|count| (count, buffer))
387                });
388                match read {
389                    Ok(Ok((0, _))) => return Ok::<_, io::Error>(record),
390                    Ok(Ok((count, buffer))) => {
391                        if record.len() + count > 1024 {
392                            return Err(io::Error::other(
393                                "privacy exec refusal record is too long",
394                            ));
395                        }
396                        record.extend_from_slice(&buffer[..count]);
397                    }
398                    Ok(Err(error)) => return Err(error),
399                    Err(_) => continue,
400                }
401            }
402        })
403        .await;
404        // Keep the reader in self across await: select cancellation must not
405        // discard the handshake or reset its original five-second deadline.
406        let pending = self.privacy_exec.as_ref().expect("pending exec");
407        let cause = match result {
408            Err(_) => Some("privacy identity trampoline did not exec within 5s".to_string()),
409            Ok(Err(error)) => Some(format!(
410                "privacy identity exec acknowledgement failed: {error}"
411            )),
412            Ok(Ok(record)) if !record.is_empty() => Some(
413                std::str::from_utf8(&record)
414                    .ok()
415                    .and_then(|record| {
416                        record
417                            .trim()
418                            .strip_prefix(subc_os::privacy_identity::EXEC_REFUSAL_TAG)
419                    })
420                    .filter(|cause| !cause.is_empty())
421                    .unwrap_or("invalid privacy exec refusal record")
422                    .to_string(),
423            ),
424            Ok(Ok(_)) => match self.child.try_wait() {
425                // Empty EOF is the exec acknowledgement. A real module may exit
426                // immediately, including with a reserved trampoline status; no
427                // image is admitted, and its ordinary exit contract stays intact.
428                Ok(Some(_status)) => None,
429                Err(error) => Some(format!("privacy identity trampoline wait failed: {error}")),
430                Ok(None) => {
431                    let image = observe_spawned_image(self.pid);
432                    if let Some(image) = image.filter(|image| {
433                        image.executable.is_some()
434                            && image.executable != pending.trampoline
435                            && (image.executable == pending.expected || pending.script)
436                    }) {
437                        if let Some(guard) = &self.roster_guard {
438                            guard.confirm_executable(image);
439                        }
440                        let _ = self.report_ready.set(());
441                        info!(module_id = %pending.module_id, pid = self.pid,
442                            "module spawned with own privacy identity (responsibility disclaimed)");
443                        None
444                    } else if image.is_none()
445                        || image.is_some_and(|image| image.executable.is_none())
446                    {
447                        // A process can exit between try_wait and the kernel
448                        // image read. Empty EOF already acknowledged exec, so
449                        // preserve that module's ordinary exit rather than
450                        // mislabel a disappearing image as trampoline refusal.
451                        // Keep pending in self across await so cancellation does
452                        // not discard validation or reset its original deadline.
453                        match tokio::time::timeout_at(pending.deadline, self.child.wait()).await {
454                            Ok(Ok(_status)) => None,
455                            Ok(Err(error)) => Some(format!("privacy identity trampoline wait failed: {error}")),
456                            Err(_) => Some("privacy identity module executable remained unreadable until the 5s exec deadline".to_string()),
457                        }
458                    } else {
459                        Some("privacy identity trampoline executable mismatch: expected the module program, not ck-subc".to_string())
460                    }
461                }
462            },
463        };
464        let pending = self.privacy_exec.take().expect("pending exec");
465        if let Some(cause) = cause {
466            warn!(module_id = %pending.module_id, pid = self.pid, %cause, "privacy identity trampoline refused module spawn");
467            self.spawn_failure = Some(cause);
468            // No image is admitted on failure. Reach the entire fresh process
469            // group, including a module which spawned a helper before refusal.
470            if let Some(pid) = rustix::process::Pid::from_raw(self.pid as i32) {
471                let _ = rustix::process::kill_process_group(pid, rustix::process::Signal::KILL);
472            }
473            let _ = self.child.start_kill();
474        }
475    }
476
477    /// Releases this child's daemon-shutdown roster entry once its exit has
478    /// been recorded. The pid is already reaped and free for reuse, so the
479    /// entry must not outlive the record any longer than that.
480    fn release_roster(&mut self) {
481        self.roster_guard = None;
482    }
483
484    /// Kill the child and, where containment is available, its process tree.
485    ///
486    /// `Child::kill` is `TerminateProcess` scoped to one pid, so a module with a
487    /// helper process leaked the helper — the Synapse embedding module's CUDA
488    /// worker holds the GPU allocation, so the leak cost VRAM until the next
489    /// restart of something else. Terminating the job reaches grandchildren that
490    /// a tree walk cannot, including one whose parent has already exited and
491    /// been reparented away.
492    ///
493    /// Best-effort like `request_graceful_stop`: a job failure is logged and the
494    /// direct-child kill still decides the outcome, so containment can never
495    /// change whether a module is reported as stopped.
496    fn start_kill(&mut self) -> io::Result<()> {
497        #[cfg(windows)]
498        if let Some(job) = &self.job {
499            if let Err(error) = job.terminate() {
500                debug!(
501                    error = %error,
502                    "job termination failed; the direct-child kill still owns the outcome"
503                );
504            }
505        }
506        #[cfg(target_os = "linux")]
507        kill_module_cgroup(self.cgroup_placement.as_ref(), &self.module_id);
508        self.child.start_kill()
509    }
510
511    async fn drain_stderr(&mut self, module_id: &str) {
512        if let Some(mut pump) = self.stdout_pump.take() {
513            match timeout(STDERR_PUMP_DRAIN_TIMEOUT, &mut pump).await {
514                Ok(Ok(())) => {}
515                Ok(Err(error)) => {
516                    warn!(module_id, error = %error, "stdout pump ended unexpectedly");
517                }
518                Err(_) => {
519                    pump.abort();
520                    warn!(
521                        module_id,
522                        waited = ?STDERR_PUMP_DRAIN_TIMEOUT,
523                        "stdout pump did not drain before restart; stopped it before the next process"
524                    );
525                }
526            }
527        }
528
529        let Some(pump) = self.stderr_pump.take() else {
530            return;
531        };
532        settle_stderr_pump(
533            module_id,
534            &self.stderr_ring,
535            pump,
536            STDERR_PUMP_DRAIN_TIMEOUT,
537        )
538        .await;
539    }
540}
541
542/// The reader task for one process's stderr, with the ring generation its
543/// lines are attributed to.
544struct StderrPump {
545    task: JoinHandle<()>,
546    generation: u64,
547}
548
549/// Retire an exited process's stderr reader and wait up to `bound` for it to
550/// reach EOF. A reader still running at the bound is detached, not stopped: it
551/// keeps filling the exited process's section of the ring until its pipe
552/// closes, and the tail reads `Incomplete` until then. See
553/// [`STDERR_PUMP_DRAIN_TIMEOUT`] for why.
554async fn settle_stderr_pump(
555    module_id: &str,
556    ring: &Arc<Mutex<StderrRing>>,
557    pump: StderrPump,
558    bound: Duration,
559) {
560    let lock = || ring.lock().unwrap_or_else(|poisoned| poisoned.into_inner());
561    let StderrPump {
562        mut task,
563        generation,
564    } = pump;
565    lock().retire_pump(generation);
566    match timeout(bound, &mut task).await {
567        Ok(Ok(())) => {}
568        Ok(Err(err)) => {
569            let mut ring = lock();
570            ring.mark_incomplete_from(generation, format!("stderr pump ended unexpectedly: {err}"));
571            ring.finish_pump(generation);
572            warn!(module_id, error = %err, "stderr pump ended before clean EOF");
573        }
574        Err(_) => {
575            // Dropping the handle detaches the task; it ends at EOF on its pipe.
576            drop(task);
577            lock().mark_pump_late(
578                generation,
579                format!(
580                    "stderr of the exited process had not reached EOF {bound:?} after it was \
581                     retired (a descendant may still hold the pipe open); lines it still \
582                     writes are kept in that process's section"
583                ),
584            );
585            warn!(
586                module_id,
587                waited = ?bound,
588                "stderr pipe of the exited process is still open; its reader keeps running without delaying the restart"
589            );
590        }
591    }
592}
593
594fn registration_release_events() -> &'static watch::Sender<u64> {
595    static EVENTS: OnceLock<watch::Sender<u64>> = OnceLock::new();
596    EVENTS.get_or_init(|| {
597        let (sender, _receiver) = watch::channel(0);
598        sender
599    })
600}
601
602pub(crate) fn notify_registration_release() {
603    let events = registration_release_events();
604    let next_generation = (*events.borrow()).wrapping_add(1);
605    events.send_replace(next_generation);
606}
607
608/// How to launch one singleton module process.
609#[derive(Debug, Clone, PartialEq, Eq)]
610pub struct ModuleSpec {
611    pub module_id: String,
612    pub program: PathBuf,
613    pub args: Vec<String>,
614    pub env: Vec<(String, String)>,
615    /// When true this is a reserved module: each spawn gets a fresh one-time launch
616    /// nonce that the child must echo in its HELLO, so only the daemon-spawned
617    /// process can register this module_id (a security-boundary module like the
618    /// credential vault must not be impersonable while it is down/restarting).
619    pub reserved: bool,
620    /// Module-id prefixes this supervised module owns for reserved HELLO checks.
621    /// Prefixes come from daemon config and must end in `:` before they reach the
622    /// supervisor; the owner module's current spawn nonce authorizes claims under
623    /// each prefix.
624    pub reserved_prefixes: Vec<String>,
625    /// The wire protocol this module speaks, as DECLARED in daemon config.
626    ///
627    /// [`ModuleProtocol::None`] changes five things and nothing else: wire health
628    /// probing is suppressed (an optional HTTP probe can replace it), teardown sends SIGTERM before waiting,
629    /// `route.open` is refused, the spawn passes NO `--subc <path>` argument
630    /// and NO launch nonce, and a clean exit the daemon did not request is
631    /// restarted as a crash rather than recorded as a stop (see `on_child_exit`:
632    /// a stock program exits 0 on a stray SIGTERM, and a stop would leave it
633    /// down for good). `SUBC_MODULE_ID` still goes into the environment,
634    /// because a process ignores an environment variable it does not read.
635    ///
636    /// The argument is the part that cannot be "harmless to a process that
637    /// ignores it": a stock binary exits on an unknown flag before it listens
638    /// (`nats-server`: "flag provided but not defined: -subc"), which is how the
639    /// first conformance run against this mode found it. The nonce is withheld
640    /// because a process that will never present it gains nothing from holding
641    /// it, and a secret in the environment of a process that does not need it is
642    /// a leak surface for no benefit.
643    pub protocol: ModuleProtocol,
644    /// Whether two processes of this module may run at once, which is what a
645    /// blue/green swap does for the length of its overlap. Declared in daemon
646    /// config because the daemon must be able to answer it while the module is
647    /// down, and so a module cannot talk itself into it after registering.
648    pub overlap: ModuleOverlap,
649}
650
651/// Whether a module tolerates a second process of itself running alongside.
652///
653/// Most modules are single-writer on their store (a WAL, a capture log, a
654/// resident index behind a writer barrier), and two processes on one store
655/// corrupt it. So a swap, which overlaps the old and new process by design,
656/// is refused unless the module's config opts in with `overlap: "safe"`.
657#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
658pub enum ModuleOverlap {
659    /// Never run two processes of this module at once. The default.
660    #[default]
661    Exclusive,
662    /// The module has said a second process of itself is harmless for the
663    /// length of a swap.
664    ///
665    /// Declare it only if a second instance can run for a few seconds without
666    /// touching ANY single-writer store: every database, WAL, index, projector
667    /// and scheduled job the module owns. A lease on part of that state is not
668    /// enough. broca's session lease guards WAL appends while its run index, its
669    /// store projector and its archive fold timer (which unlinks live WAL files)
670    /// stay single-writer, so broca is exclusive despite holding a lease. The
671    /// refusal only fires after this has been decided, so the decision is the
672    /// check.
673    Safe,
674}
675
676impl ModuleOverlap {
677    pub fn as_str(self) -> &'static str {
678        match self {
679            Self::Exclusive => "exclusive",
680            Self::Safe => "safe",
681        }
682    }
683}
684
685/// Environment variable telling a spawned module which case it was started
686/// for, before it sends HELLO. Only a swap candidate carries it, as
687/// [`SPAWN_ROLE_SWAP_CANDIDATE`]; every other spawn has it removed.
688///
689/// It chooses a warm-up budget, nothing else: a swap candidate can warm for
690/// longer because nobody waits on it, while a plain restart must flip ready
691/// quickly because callers see `module_warming` until it does. Absence means
692/// plain restart, the safe reading. The daemon trusts nothing about it; the
693/// candidate is proven by its launch nonce at HELLO.
694pub const SUBC_SPAWN_ROLE_ENV: &str = "SUBC_SPAWN_ROLE";
695/// The one value of [`SUBC_SPAWN_ROLE_ENV`] the daemon sets.
696pub const SPAWN_ROLE_SWAP_CANDIDATE: &str = "swap_candidate";
697/// How long a swap waits for its candidate to register and declare itself
698/// ready when the operator does not say. A module warming as a swap candidate
699/// may take up to 90 s (aft's ceiling, the largest in the fleet), so the
700/// daemon allows that plus time to start the process and send HELLO.
701pub const DEFAULT_SWAP_READY_TIMEOUT: Duration = Duration::from_secs(100);
702
703/// Bounded restart policy for crash exits.
704///
705/// `max_restarts` is the number of replacement processes allowed after the
706/// initial spawn WITHIN `window`. After that many crash restarts inside one
707/// window the module enters [`ModuleState::Failed`] and the supervisor stops
708/// the crash loop.
709///
710/// The budget is a RATE, not a lifetime total. It used to be a lifetime total,
711/// and that only survived because crashes were rare: a module that crashed
712/// three times across a week was disabled forever by crashes that had nothing
713/// to do with each other. That stopped being survivable once modules began
714/// exiting non-zero whenever the daemon's connection to them drops, because
715/// then every daemon-side connection drop spends a unit of the same budget and
716/// one flappy hour permanently stops a healthy module. Restarts older than
717/// `window` release their slot, so a module that crashed twice yesterday has a
718/// full budget today, while a genuine crash loop -- which is fast by
719/// definition -- still reaches the cap and stops.
720#[derive(Debug, Clone, Copy, PartialEq, Eq)]
721pub struct RestartPolicy {
722    pub max_restarts: u32,
723    /// Base delay before a crash replacement. The actual delay escalates with
724    /// the number of recent crash replacements and is capped by `max_backoff`.
725    pub backoff: Duration,
726    /// Maximum delay before a crash replacement.
727    pub max_backoff: Duration,
728    /// The span `max_restarts` is counted over. `Duration::ZERO` makes the
729    /// budget effectively infinite (nothing is ever in-window), which is why
730    /// daemon config refuses `window_secs: 0` rather than quietly accepting it.
731    pub window: Duration,
732}
733
734impl RestartPolicy {
735    /// A policy with the default crash window. Callers that care about the
736    /// window say so with [`Self::with_window`]; the ones that do not are
737    /// asking for the standard rate limit, not for no limit.
738    pub fn new(max_restarts: u32, backoff: Duration) -> Self {
739        Self {
740            max_restarts,
741            backoff,
742            max_backoff: DEFAULT_MAX_BACKOFF,
743            window: DEFAULT_RESTART_WINDOW,
744        }
745    }
746
747    pub fn with_max_backoff(mut self, max_backoff: Duration) -> Self {
748        self.max_backoff = max_backoff;
749        self
750    }
751
752    pub fn with_window(mut self, window: Duration) -> Self {
753        self.window = window;
754        self
755    }
756
757    /// Calculate the capped exponential delay for the next crash replacement.
758    /// `restart_in_window` is zero for the first replacement after an operator
759    /// action (restart, reload, re-enable) cleared the crash ring, or after all
760    /// older crash replacements have aged out of the window.
761    fn delay_for_restart(&self, restart_in_window: u32) -> Duration {
762        if self.backoff.is_zero() || self.max_backoff.is_zero() {
763            return Duration::ZERO;
764        }
765
766        let mut delay = self.backoff;
767        for _ in 0..restart_in_window {
768            if delay >= self.max_backoff {
769                return self.max_backoff;
770            }
771            delay = delay
772                .checked_mul(10)
773                .unwrap_or(self.max_backoff)
774                .min(self.max_backoff);
775        }
776        delay.min(self.max_backoff)
777    }
778
779    /// The one sentence that explains a budget-exhausted stop, used for both the
780    /// log line and the terminal record so the two cannot drift. It names the
781    /// window because `max_restarts=3` alone reads as a lifetime cap, which is
782    /// exactly what this budget is not.
783    fn budget_exhausted_detail(&self) -> String {
784        format!(
785            "crash budget exhausted: max_restarts={} within window_secs={}",
786            self.max_restarts,
787            self.window.as_secs()
788        )
789    }
790}
791
792impl Default for RestartPolicy {
793    fn default() -> Self {
794        Self {
795            max_restarts: DEFAULT_MAX_RESTARTS,
796            backoff: DEFAULT_BACKOFF,
797            max_backoff: DEFAULT_MAX_BACKOFF,
798            window: DEFAULT_RESTART_WINDOW,
799        }
800    }
801}
802
803#[derive(Debug, Clone, Copy, PartialEq, Eq)]
804struct CrashRestartSchedule {
805    restart_in_window: u32,
806    delay: Duration,
807}
808
809/// Whether the daemon itself will bring this module back after the exit being
810/// handled: it is enabled AND its in-window crash restarts are below the cap.
811///
812/// Takes `&mut` because reading the budget prunes it. Instants that fell out of
813/// the window are dropped here rather than by a timer, so the count is right
814/// the moment somebody asks and no bookkeeping runs for idle modules.
815fn daemon_will_restart(
816    state: &mut SupervisorSnapshot,
817    policy: &RestartPolicy,
818    now: Instant,
819) -> bool {
820    state.enabled && state.crash_restarts_in_window(policy.window, now) < policy.max_restarts
821}
822
823const DEFAULT_HEALTH_CADENCE: Duration = Duration::from_secs(30);
824const DEFAULT_HEALTH_DEADLINE: Duration = Duration::from_secs(5);
825const DEFAULT_HEALTH_FAILURE_THRESHOLD: u32 = 3;
826const MAX_HEALTH_METRICS_BYTES: usize = 16 * 1024;
827
828#[derive(Debug, Clone, Copy, PartialEq, Eq)]
829pub enum HealthAction {
830    Report,
831    Restart,
832    Alert,
833}
834
835impl fmt::Display for HealthAction {
836    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
837        f.write_str(match self {
838            Self::Report => "report",
839            Self::Restart => "restart",
840            Self::Alert => "alert",
841        })
842    }
843}
844
845#[derive(Debug, Clone, PartialEq, Eq)]
846pub struct HealthConfig {
847    /// Optional loopback HTTP endpoint for a managed non-wire process.
848    /// Changing it applies live on rescan; the process protocol changes only
849    /// at its next spawn.
850    pub http: Option<String>,
851    pub cadence: Duration,
852    pub deadline: Duration,
853    pub failure_threshold: u32,
854    pub on_degraded: HealthAction,
855    pub on_failing: HealthAction,
856    pub critical: bool,
857}
858
859impl Default for HealthConfig {
860    fn default() -> Self {
861        Self {
862            http: None,
863            cadence: DEFAULT_HEALTH_CADENCE,
864            deadline: DEFAULT_HEALTH_DEADLINE,
865            failure_threshold: DEFAULT_HEALTH_FAILURE_THRESHOLD,
866            on_degraded: HealthAction::Report,
867            on_failing: HealthAction::Report,
868            critical: false,
869        }
870    }
871}
872
873/// The supervisor's view of one module's health, relayed to clients over
874/// channel-0 and rendered by `ck health`.
875///
876/// THIS TYPE IS WHERE THE ABSENCE MEANINGS ARE CREATED, which is why they are
877/// stated here rather than only at the wire type a consumer reads. A reader can
878/// look up what `None` means; only a writer can silently change it, and the
879/// writer has no reason to go looking at a downstream contract before editing.
880///
881/// `last_probe_ms: None` MEANS NEVER PROBED, not probed-long-ago. It is cleared
882/// back to `None` on re-registration precisely so a respawned module does not
883/// carry its predecessor's timestamp — so an old value and an absent one call for
884/// opposite readings, and anything that defaulted this to a number would make a
885/// never-probed module indistinguishable from one probed at the epoch.
886///
887/// `detail` and `metrics` are `None` when the module published none on this
888/// probe, which does not mean it reported nothing wrong — it is also the shape
889/// when the probe never reached it. `last_probe_ms` is what separates those.
890#[derive(Debug, Clone, PartialEq)]
891pub struct ModuleHealthStatus {
892    pub status: SupervisorHealthStatus,
893    pub last_probe_ms: Option<u64>,
894    pub detail: Option<String>,
895    pub metrics: Option<Value>,
896    pub consecutive_failures: u32,
897    /// Number of replies received after a recurring health probe's deadline.
898    /// Unlike a timeout, every increment proves the module was alive.
899    pub late_answer_count: u64,
900    /// End-to-end latency of the newest late reply, measured from probe start.
901    pub last_late_answer_latency_ms: Option<u64>,
902    pub last_action: Option<String>,
903    /// Set together with `last_action`; the pair moves as one, and both being
904    /// absent means no escalation has ever been taken rather than that the last
905    /// one succeeded.
906    pub last_action_ms: Option<u64>,
907}
908
909impl Default for ModuleHealthStatus {
910    fn default() -> Self {
911        Self {
912            status: SupervisorHealthStatus::Unknown,
913            last_probe_ms: None,
914            detail: None,
915            metrics: None,
916            consecutive_failures: 0,
917            late_answer_count: 0,
918            last_late_answer_latency_ms: None,
919            last_action: None,
920            last_action_ms: None,
921        }
922    }
923}
924
925/// Typed lifecycle state for a supervised module.
926#[derive(Debug, Clone, Copy, PartialEq, Eq)]
927pub enum ModuleState {
928    Starting,
929    Running,
930    Unresponsive,
931    Restarting,
932    Draining,
933    Stopped,
934    Failed,
935    Disabled,
936}
937
938impl fmt::Display for ModuleState {
939    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
940        f.write_str(match self {
941            Self::Starting => "starting",
942            Self::Running => "running",
943            Self::Unresponsive => "unresponsive",
944            Self::Restarting => "restarting",
945            Self::Draining => "draining",
946            Self::Stopped => "stopped",
947            Self::Failed => "failed",
948            Self::Disabled => "disabled",
949        })
950    }
951}
952
953/// Supervisor classification of a child-process exit.
954#[derive(Debug, Clone, Copy, PartialEq, Eq)]
955pub enum ExitKind {
956    Clean,
957    Crash,
958    DeliberateSeverance,
959}
960
961impl From<ExitKind> for TerminalExitKind {
962    fn from(kind: ExitKind) -> Self {
963        match kind {
964            ExitKind::Clean => Self::Clean,
965            ExitKind::Crash => Self::Crash,
966            ExitKind::DeliberateSeverance => Self::DeliberateSeverance,
967        }
968    }
969}
970
971/// Exact process identity retained when a supervised module registers its
972/// connection. PID reuse makes a PID alone insufficient evidence of ownership.
973#[derive(Debug, Clone, Copy, PartialEq, Eq)]
974pub(crate) struct ProcessIdentity {
975    pub(crate) pid: u32,
976    pub(crate) start_time: u64,
977}
978
979/// Last observed child exit, if any.
980#[derive(Debug, Clone, PartialEq, Eq)]
981pub struct ExitReport {
982    pub kind: ExitKind,
983    pub code: Option<i32>,
984    pub signal: Option<i32>,
985    pub at_ms: u64,
986}
987
988/// Point-in-time module status answerable by subc without forwarding to the
989/// module process.
990#[derive(Debug, Clone, PartialEq)]
991pub struct ModuleStatus {
992    pub module_id: String,
993    pub state: ModuleState,
994    pub enabled: bool,
995    pub process_alive: bool,
996    pub registration_active: bool,
997    /// The module's declared wire protocol, carried beside `live` because it is
998    /// what makes `live` readable: the two fields answer one question together.
999    /// While a process is alive this is its launch declaration, not a later
1000    /// pending-reload edit. When down it is the configured next launch protocol.
1001    pub protocol: ModuleProtocol,
1002    /// Whether the module is serving, under the strongest definition the daemon
1003    /// can assert for its protocol.
1004    ///
1005    /// A subc module must also be REGISTERED: its process being alive says
1006    /// nothing about whether it can take a request. A `protocol: "none"` module
1007    /// never registers, so that term is dropped and this falls back to "enabled,
1008    /// running, and the process the daemon launched is alive" -- which is all
1009    /// the daemon observes about a process that speaks no subc wire. It stays a
1010    /// `bool` on the wire for compatibility; renderers pair it with `protocol`
1011    /// rather than printing it bare.
1012    pub live: bool,
1013    /// Crash restarts spent INSIDE `restart_window` as of this read. Older
1014    /// restarts have already released their slot, so this count can go down
1015    /// without anybody touching the module.
1016    pub restart_count: u32,
1017    /// Replacement processes spawned over this module's entire supervisor lifetime;
1018    /// unlike `restart_count`, this value is never reset by an operator action
1019    /// and never falls out of a window.
1020    pub lifetime_restarts: u32,
1021    pub spawn_generation: u64,
1022    /// The budget `restart_count` is spent against. Carried alongside the count
1023    /// because the count alone does not say how close the module is to being
1024    /// disabled, and reporting one without the other is what makes an
1025    /// about-to-be-retired module look ordinary.
1026    pub max_restarts: u32,
1027    /// The span `restart_count` is counted over. Carried with the pair above for
1028    /// the same reason they are carried together: "2 of 3" means one thing for a
1029    /// ten-minute window and something else entirely for a lifetime.
1030    pub restart_window: Duration,
1031    /// Effective drain and restart timing policy used by this running module.
1032    /// These values are carried together with the restart budget so status
1033    /// readers can compare configured intent with what the supervisor applied.
1034    pub drain_timeout: Duration,
1035    pub restart_backoff: Duration,
1036    pub restart_max_backoff: Duration,
1037    /// The module's process. On macOS this stays absent while the `ck-subc`
1038    /// launch trampoline is still running in that pid, and appears once the
1039    /// exec acknowledgement confirms the module image has replaced it. Launch
1040    /// time and the supervisor's own hold on the process are unaffected.
1041    pub pid: Option<u32>,
1042    pub spawned_at_ms: Option<u64>,
1043    pub spawned_from: Option<PathBuf>,
1044    pub process_start_time: Option<u64>,
1045    pub last_exit: Option<ExitReport>,
1046    pub health: ModuleHealthStatus,
1047}
1048
1049#[derive(Debug, Clone, PartialEq)]
1050struct SupervisorSnapshot {
1051    state: ModuleState,
1052    enabled: bool,
1053    process_alive: bool,
1054    spawned_protocol: Option<ModuleProtocol>,
1055    spawn_failure: Option<String>,
1056    /// When each crash restart was spent, oldest first. This IS the crash
1057    /// budget: its in-window length is the count an operator sees and the count
1058    /// the restart decision is made against, so there is no second counter that
1059    /// can disagree with it. Bounded by `max_restarts`, and cleared by the same
1060    /// operator actions that used to zero the old lifetime counter.
1061    crash_restarts: VecDeque<Instant>,
1062    lifetime_restarts: u32,
1063    /// Successful child spawns in this daemon incarnation.
1064    ///
1065    /// `lifetime_restarts` was considered and rejected: it starts at zero
1066    /// (line 640), successful initial/operator spawns in `set_running` do not
1067    /// increment it (lines 5264-5274), and crash/deliberate retry bookkeeping
1068    /// increments before a successful replacement exists (lines 604, 3846,
1069    /// and 3921), so a failed spawn can consume it. This counter moves only
1070    /// when a live PID is accepted below.
1071    spawn_generation: u64,
1072    pid: Option<u32>,
1073    #[cfg(target_os = "macos")]
1074    report_ready: Option<Arc<OnceLock<()>>>,
1075    /// Last reaped child, retained after current process facts are cleared.
1076    reaped_pid: Option<u32>,
1077    /// Whether the command-serving supervision loop has a scheduled respawn.
1078    respawn_pending: bool,
1079    /// A second restart is waiting for the replacement already scheduled.
1080    coalesced_restart_pending: bool,
1081    spawned_at_ms: Option<u64>,
1082    spawned_from: Option<PathBuf>,
1083    spawned_file_identity: Option<SpawnedFileIdentity>,
1084    process_start_time: Option<u64>,
1085    deliberate_severance: Option<ProcessIdentity>,
1086    last_exit: Option<ExitReport>,
1087    /// Diagnostic attached to the next drain's terminal record, if any.
1088    drain_disposition_detail: Option<String>,
1089    health: ModuleHealthStatus,
1090    /// Whether the current process was started as a swap candidate and so
1091    /// lives in the module's alternate cgroup. The next swap's candidate takes
1092    /// the other one, so the two processes of a swap never share a cgroup. A
1093    /// plain spawn always uses the primary cgroup.
1094    in_alternate_slot: bool,
1095    /// Whether the current `Draining` state ends in a replacement process
1096    /// (restart, reload, health restart) rather than a stop. Only meaningful
1097    /// while `state` is `Draining`; every entry into that state rewrites it.
1098    /// It is what lets route.open answer the retryable `module_reloading` to a
1099    /// consumer that reaches a still-registered process mid-restart, instead of
1100    /// the `supervisor_not_live` a stop or disable deserves.
1101    draining_to_replace: bool,
1102    /// Whether a configuration update has been applied since the current
1103    /// process was spawned, so that process runs an older spec than the one
1104    /// the supervisor now holds. A queued restart is only coalesced into a
1105    /// fresher process when this is false: a restart requested to pick up a
1106    /// new configuration must not be satisfied by a process that predates it.
1107    configuration_updated_since_spawn: bool,
1108}
1109
1110impl SupervisorSnapshot {
1111    /// The pid that status, provenance and resource readings may report. While
1112    /// the launch trampoline still runs in the pid, reading its executable or
1113    /// resource use would describe `ck-subc`, not the module, so none is
1114    /// reported until the exec acknowledgement confirms the module image.
1115    fn reported_pid(&self) -> Option<u32> {
1116        #[cfg(target_os = "macos")]
1117        if self
1118            .report_ready
1119            .as_ref()
1120            .is_some_and(|ready| ready.get().is_none())
1121        {
1122            return None;
1123        }
1124        self.pid
1125    }
1126
1127    fn starting() -> Self {
1128        Self::new(ModuleState::Starting, true)
1129    }
1130
1131    fn disabled() -> Self {
1132        Self::new(ModuleState::Disabled, false)
1133    }
1134
1135    fn failed() -> Self {
1136        Self::new(ModuleState::Failed, true)
1137    }
1138
1139    /// Crash restarts still inside `window`, having dropped the ones that are
1140    /// not. Pruning on read is what makes the budget a rate: an instant older
1141    /// than the window stops holding a slot the moment anybody counts.
1142    fn crash_restarts_in_window(&mut self, window: Duration, now: Instant) -> u32 {
1143        while let Some(oldest) = self.crash_restarts.front() {
1144            if now.duration_since(*oldest) > window {
1145                self.crash_restarts.pop_front();
1146            } else {
1147                break;
1148            }
1149        }
1150        u32::try_from(self.crash_restarts.len()).unwrap_or(u32::MAX)
1151    }
1152
1153    /// Spend one unit of the crash budget and record the restart in the ledger.
1154    ///
1155    /// The ring is bounded by the cap because more than `max_restarts` in-window
1156    /// instants can never be reached (the caller refuses the restart first), so
1157    /// anything beyond that is an unbounded queue waiting to happen.
1158    fn record_crash_restart(&mut self, policy: &RestartPolicy, now: Instant) {
1159        self.crash_restarts.push_back(now);
1160        while self.crash_restarts.len() > policy.max_restarts as usize {
1161            self.crash_restarts.pop_front();
1162        }
1163        self.lifetime_restarts += 1;
1164    }
1165
1166    /// Reserve one crash-restart slot and calculate the delay before respawning.
1167    /// The count is captured before recording this restart, so the first retry
1168    /// uses the base delay and each later in-window retry escalates once.
1169    fn next_crash_restart(
1170        &mut self,
1171        policy: &RestartPolicy,
1172        now: Instant,
1173    ) -> Option<CrashRestartSchedule> {
1174        let restart_in_window = self.crash_restarts_in_window(policy.window, now);
1175        if restart_in_window >= policy.max_restarts {
1176            return None;
1177        }
1178        self.record_crash_restart(policy, now);
1179        Some(CrashRestartSchedule {
1180            restart_in_window,
1181            delay: policy.delay_for_restart(restart_in_window),
1182        })
1183    }
1184
1185    /// Give the module its full budget back, as an operator restart, reload, or
1186    /// re-enable does. `lifetime_restarts` deliberately does not move: it is the
1187    /// ledger of what actually happened, and an operator action does not unmake
1188    /// the crashes.
1189    fn clear_crash_restarts(&mut self) {
1190        self.crash_restarts.clear();
1191    }
1192
1193    fn new(state: ModuleState, enabled: bool) -> Self {
1194        Self {
1195            state,
1196            enabled,
1197            process_alive: false,
1198            spawned_protocol: None,
1199            spawn_failure: None,
1200            crash_restarts: VecDeque::new(),
1201            lifetime_restarts: 0,
1202            spawn_generation: 0,
1203            pid: None,
1204            #[cfg(target_os = "macos")]
1205            report_ready: None,
1206            reaped_pid: None,
1207            respawn_pending: false,
1208            coalesced_restart_pending: false,
1209            spawned_at_ms: None,
1210            spawned_from: None,
1211            spawned_file_identity: None,
1212            process_start_time: None,
1213            deliberate_severance: None,
1214            last_exit: None,
1215            drain_disposition_detail: None,
1216            health: ModuleHealthStatus::default(),
1217            in_alternate_slot: false,
1218            draining_to_replace: false,
1219            configuration_updated_since_spawn: false,
1220        }
1221    }
1222}
1223
1224type SharedSnapshot = Arc<Mutex<SupervisorSnapshot>>;
1225
1226type SpawnSubscriberKey = (ConnectionId, u64);
1227
1228#[derive(Debug)]
1229struct SpawnSubscriber {
1230    version: u8,
1231    frames: mpsc::Sender<Frame>,
1232    /// Tells this subscriber's forwarder that it was dropped for lagging, and
1233    /// from which event. The full frame channel cannot carry that news, so it
1234    /// travels beside it; see `SpawnEventFeed::subscribe`.
1235    lagged: Option<oneshot::Sender<SpawnCursor>>,
1236}
1237
1238#[derive(Debug)]
1239struct SpawnEventState {
1240    daemon_incarnation: String,
1241    seq: u64,
1242    capacity: usize,
1243    live: HashMap<String, LiveSpawn>,
1244    generations: HashMap<String, u64>,
1245    events: VecDeque<SpawnEvent>,
1246    subscribers: HashMap<SpawnSubscriberKey, SpawnSubscriber>,
1247}
1248
1249impl Default for SpawnEventState {
1250    fn default() -> Self {
1251        Self {
1252            daemon_incarnation: "unconfigured".to_string(),
1253            seq: 0,
1254            capacity: SPAWN_EVENT_RING_CAPACITY,
1255            live: HashMap::new(),
1256            generations: HashMap::new(),
1257            events: VecDeque::new(),
1258            subscribers: HashMap::new(),
1259        }
1260    }
1261}
1262
1263#[derive(Debug, Clone, Default)]
1264struct SpawnEventFeed(Arc<Mutex<SpawnEventState>>);
1265
1266#[derive(Debug, Clone, PartialEq, Eq)]
1267pub(crate) enum SpawnSubscribeRefusal {
1268    ForeignIncarnation { current: String },
1269    TooOld { oldest: SpawnCursor },
1270    Frame(String),
1271}
1272
1273impl SpawnEventFeed {
1274    fn configure_incarnation(&self, daemon_incarnation: String) {
1275        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1276        state.daemon_incarnation = daemon_incarnation;
1277        state.seq = 0;
1278        state.live.clear();
1279        state.generations.clear();
1280        state.events.clear();
1281        state.subscribers.clear();
1282    }
1283
1284    fn cursor(state: &SpawnEventState) -> SpawnCursor {
1285        SpawnCursor {
1286            daemon_incarnation: state.daemon_incarnation.clone(),
1287            seq: state.seq,
1288        }
1289    }
1290
1291    fn snapshot(&self) -> SpawnSnapshot {
1292        let state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1293        let mut live = state.live.values().cloned().collect::<Vec<_>>();
1294        live.sort_by(|left, right| left.module_id.cmp(&right.module_id));
1295        SpawnSnapshot {
1296            cursor: Self::cursor(&state),
1297            ring_bound: state.capacity as u64,
1298            live,
1299        }
1300    }
1301
1302    fn emit_spawned(&self, module_id: &str, pid: u32, spawned_at_ms: u64) -> u64 {
1303        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1304        let generation = state
1305            .generations
1306            .get(module_id)
1307            .copied()
1308            .unwrap_or(0)
1309            .checked_add(1)
1310            .expect("spawn generation exhausted");
1311        state.generations.insert(module_id.to_string(), generation);
1312        let live = LiveSpawn {
1313            module_id: module_id.to_string(),
1314            spawn_generation: generation,
1315            pid,
1316            spawned_at_ms,
1317        };
1318        state.live.insert(module_id.to_string(), live);
1319        Self::emit_locked(
1320            &mut state,
1321            SpawnEventKind::Spawned,
1322            module_id.to_string(),
1323            generation,
1324            pid,
1325            None,
1326            None,
1327        );
1328        generation
1329    }
1330
1331    fn emit_exited(&self, module_id: &str, exit_code: Option<i32>, exit_signal: Option<i32>) {
1332        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1333        let Some(live) = state.live.remove(module_id) else {
1334            warn!(
1335                module_id,
1336                "terminal record had no live spawn event identity"
1337            );
1338            return;
1339        };
1340        Self::emit_locked(
1341            &mut state,
1342            SpawnEventKind::Exited,
1343            module_id.to_string(),
1344            live.spawn_generation,
1345            live.pid,
1346            exit_code,
1347            exit_signal,
1348        );
1349    }
1350
1351    /// Report the exit of a process that a swap has already replaced.
1352    ///
1353    /// `emit_exited` removes the module's live entry, which after a swap's
1354    /// cutover describes the promoted candidate, not the old process now
1355    /// exiting. This emits the old generation's exit and leaves the live entry
1356    /// alone unless it still names that generation.
1357    fn emit_superseded_exited(
1358        &self,
1359        module_id: &str,
1360        spawn_generation: u64,
1361        pid: u32,
1362        exit_code: Option<i32>,
1363        exit_signal: Option<i32>,
1364    ) {
1365        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1366        if state
1367            .live
1368            .get(module_id)
1369            .is_some_and(|live| live.spawn_generation == spawn_generation)
1370        {
1371            state.live.remove(module_id);
1372        }
1373        Self::emit_locked(
1374            &mut state,
1375            SpawnEventKind::Exited,
1376            module_id.to_string(),
1377            spawn_generation,
1378            pid,
1379            exit_code,
1380            exit_signal,
1381        );
1382    }
1383
1384    #[allow(clippy::too_many_arguments)]
1385    fn emit_locked(
1386        state: &mut SpawnEventState,
1387        kind: SpawnEventKind,
1388        module_id: String,
1389        spawn_generation: u64,
1390        pid: u32,
1391        exit_code: Option<i32>,
1392        exit_signal: Option<i32>,
1393    ) {
1394        state.seq = state
1395            .seq
1396            .checked_add(1)
1397            .expect("spawn event sequence exhausted");
1398        let event = SpawnEvent {
1399            cursor: Self::cursor(state),
1400            kind,
1401            module_id,
1402            spawn_generation,
1403            pid,
1404            exit_code,
1405            exit_signal,
1406        };
1407        state.events.push_back(event.clone());
1408        while state.events.len() > state.capacity {
1409            state.events.pop_front();
1410        }
1411        let body = match serde_json::to_vec(&event) {
1412            Ok(body) => body,
1413            Err(error) => {
1414                error!(%error, "failed to serialize supervisor spawn event");
1415                return;
1416            }
1417        };
1418        state.subscribers.retain(|(connection_id, corr), subscriber| {
1419            let frame = Frame::build_with_version(
1420                subscriber.version,
1421                FrameType::StreamData,
1422                control_flags(),
1423                0,
1424                0,
1425                *corr,
1426                body.clone(),
1427            );
1428            match frame {
1429                Ok(frame) => {
1430                    if subscriber.frames.try_send(frame).is_ok() {
1431                        true
1432                    } else {
1433                        warn!(connection_id = connection_id.get(), corr, "dropping lagged supervisor spawn subscriber");
1434                        if let Some(lagged) = subscriber.lagged.take() {
1435                            let _ = lagged.send(event.cursor.clone());
1436                        }
1437                        false
1438                    }
1439                }
1440                Err(error) => {
1441                    warn!(connection_id = connection_id.get(), corr, %error, "dropping supervisor spawn subscriber after frame build failure");
1442                    false
1443                }
1444            }
1445        });
1446    }
1447
1448    fn subscribe(
1449        &self,
1450        connection_id: ConnectionId,
1451        corr: u64,
1452        version: u8,
1453        since: Option<SpawnCursor>,
1454        sink: FrameSink,
1455    ) -> Result<(), SpawnSubscribeRefusal> {
1456        let (frames, mut receiver) = mpsc::channel(SPAWN_SUBSCRIBER_BUFFER);
1457        let (lagged, mut lagged_rx) = oneshot::channel::<SpawnCursor>();
1458        {
1459            let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1460            let replay = if let Some(since) = since {
1461                if since.daemon_incarnation != state.daemon_incarnation {
1462                    return Err(SpawnSubscribeRefusal::ForeignIncarnation {
1463                        current: state.daemon_incarnation.clone(),
1464                    });
1465                }
1466                if let Some(oldest) = state.events.front().map(|event| event.cursor.clone()) {
1467                    if since.seq < oldest.seq.saturating_sub(1) {
1468                        return Err(SpawnSubscribeRefusal::TooOld { oldest });
1469                    }
1470                }
1471                state
1472                    .events
1473                    .iter()
1474                    .filter(|event| event.cursor.seq > since.seq)
1475                    .cloned()
1476                    .collect::<Vec<_>>()
1477            } else {
1478                Vec::new()
1479            };
1480            for event in replay {
1481                let body = serde_json::to_vec(&event)
1482                    .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1483                let frame = Frame::build_with_version(
1484                    version,
1485                    FrameType::StreamData,
1486                    control_flags(),
1487                    0,
1488                    0,
1489                    corr,
1490                    body,
1491                )
1492                .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1493                frames
1494                    .try_send(frame)
1495                    .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1496            }
1497            state.subscribers.insert(
1498                (connection_id, corr),
1499                SpawnSubscriber {
1500                    version,
1501                    frames,
1502                    lagged: Some(lagged),
1503                },
1504            );
1505        }
1506        // The lagged terminal is sent here, by the forwarder, rather than by
1507        // the emitter: at the moment of the drop the subscriber's own channel
1508        // is full, and writing to the connection sink directly from the emitter
1509        // would put the Error AHEAD of the events still queued in that channel
1510        // (and the emitter holds the feed lock, so it cannot await the sink).
1511        // Dropping the subscriber drops the only sender, so `recv` drains every
1512        // queued event and then returns `None`; only then is the Error sent, so
1513        // the client sees each event it can keep, then the reason it was cut.
1514        // Cancel and connection removal drop the oneshot unsent, so they end
1515        // the stream with no Error.
1516        tokio::spawn(async move {
1517            while let Some(frame) = receiver.recv().await {
1518                if sink.send(frame).await.is_err() {
1519                    return;
1520                }
1521            }
1522            let Ok(first_undelivered) = lagged_rx.try_recv() else {
1523                return;
1524            };
1525            match spawn_subscriber_lagged_frame(version, corr, first_undelivered) {
1526                Ok(frame) => {
1527                    let _ = sink.send(frame).await;
1528                }
1529                Err(error) => {
1530                    error!(%error, corr, "failed to build lagged spawn subscriber terminal frame");
1531                }
1532            }
1533        });
1534        Ok(())
1535    }
1536
1537    fn cancel(&self, connection_id: ConnectionId, corr: u64) -> bool {
1538        let Some(subscriber) = self
1539            .0
1540            .lock()
1541            .unwrap_or_else(|p| p.into_inner())
1542            .subscribers
1543            .remove(&(connection_id, corr))
1544        else {
1545            return false;
1546        };
1547        if let Ok(frame) = Frame::build_with_version(
1548            subscriber.version,
1549            FrameType::StreamEnd,
1550            control_flags(),
1551            0,
1552            0,
1553            corr,
1554            Vec::new(),
1555        ) {
1556            tokio::spawn(async move {
1557                let _ = subscriber.frames.send(frame).await;
1558            });
1559        }
1560        true
1561    }
1562
1563    fn remove_connection(&self, connection_id: ConnectionId) {
1564        self.0
1565            .lock()
1566            .unwrap_or_else(|p| p.into_inner())
1567            .subscribers
1568            .retain(|(subscriber_connection, _), _| *subscriber_connection != connection_id);
1569    }
1570
1571    #[cfg(any(test, feature = "test-support"))]
1572    fn set_capacity(&self, capacity: usize) {
1573        self.0.lock().unwrap_or_else(|p| p.into_inner()).capacity = capacity;
1574    }
1575
1576    #[cfg(any(test, feature = "test-support"))]
1577    fn subscriber_count(&self) -> usize {
1578        self.0
1579            .lock()
1580            .unwrap_or_else(|p| p.into_inner())
1581            .subscribers
1582            .len()
1583    }
1584}
1585
1586/// Narrow process-liveness signal published by supervisors and consumed by passive liveness polls.
1587/// The terminal Error a lagged spawn subscriber receives after its queued events.
1588fn spawn_subscriber_lagged_frame(
1589    version: u8,
1590    corr: u64,
1591    first_undelivered: SpawnCursor,
1592) -> Result<Frame, String> {
1593    let body = serde_json::to_vec(&subc_protocol::ErrorBody {
1594        code: SPAWN_SUBSCRIBER_LAGGED_CODE.to_string(),
1595        message: "spawn subscriber fell behind and was dropped; resubscribe from the last cursor received"
1596            .to_string(),
1597        detail: Some(serde_json::json!({
1598            "first_undelivered_cursor": first_undelivered
1599        })),
1600    })
1601    .map_err(|error| error.to_string())?;
1602    Frame::build_with_version(version, FrameType::Error, control_flags(), 0, 0, corr, body)
1603        .map_err(|error| error.to_string())
1604}
1605
1606pub trait ModuleProcessLiveness: Send + Sync {
1607    fn process_live(&self, module_id: &str) -> Option<bool>;
1608
1609    /// Whether the supervisor is replacing this module's process right now: an
1610    /// operator restart or reload, a health restart, or a crash respawn whose
1611    /// backoff is running. A module in that state is not live, but a consumer
1612    /// refused now should retry shortly rather than treat the target as gone.
1613    /// Stopped, failed, and disabled modules are not replacing.
1614    fn process_replacing(&self, _module_id: &str) -> bool {
1615        false
1616    }
1617}
1618
1619/// Shared process-liveness registry keyed by supervised `module_id`.
1620#[derive(Debug, Clone, Default)]
1621pub struct SupervisorProcessLiveness {
1622    snapshots: Arc<Mutex<HashMap<String, SharedSnapshot>>>,
1623}
1624
1625impl SupervisorProcessLiveness {
1626    pub fn new() -> Self {
1627        Self::default()
1628    }
1629
1630    fn track(&self, module_id: String, snapshot: SharedSnapshot) {
1631        let mut snapshots = self
1632            .snapshots
1633            .lock()
1634            .unwrap_or_else(|poisoned| poisoned.into_inner());
1635        snapshots.insert(module_id, snapshot);
1636    }
1637
1638    fn untrack_if_current(&self, module_id: &str, snapshot: &SharedSnapshot) {
1639        let mut snapshots = self
1640            .snapshots
1641            .lock()
1642            .unwrap_or_else(|poisoned| poisoned.into_inner());
1643        let is_current = snapshots
1644            .get(module_id)
1645            .map(|tracked| Arc::ptr_eq(tracked, snapshot))
1646            .unwrap_or(false);
1647        if is_current {
1648            snapshots.remove(module_id);
1649        }
1650    }
1651}
1652
1653impl ModuleProcessLiveness for SupervisorProcessLiveness {
1654    fn process_live(&self, module_id: &str) -> Option<bool> {
1655        let snapshot = {
1656            let snapshots = self
1657                .snapshots
1658                .lock()
1659                .unwrap_or_else(|poisoned| poisoned.into_inner());
1660            snapshots.get(module_id).cloned()
1661        }?;
1662        let snapshot = snapshot
1663            .lock()
1664            .unwrap_or_else(|poisoned| poisoned.into_inner());
1665        Some(snapshot.state == ModuleState::Running && snapshot.process_alive)
1666    }
1667
1668    fn process_replacing(&self, module_id: &str) -> bool {
1669        let Some(snapshot) = self
1670            .snapshots
1671            .lock()
1672            .unwrap_or_else(|poisoned| poisoned.into_inner())
1673            .get(module_id)
1674            .cloned()
1675        else {
1676            return false;
1677        };
1678        let snapshot = snapshot
1679            .lock()
1680            .unwrap_or_else(|poisoned| poisoned.into_inner());
1681        snapshot.enabled
1682            && match snapshot.state {
1683                ModuleState::Restarting => true,
1684                ModuleState::Draining => snapshot.draining_to_replace,
1685                ModuleState::Starting
1686                | ModuleState::Running
1687                | ModuleState::Unresponsive
1688                | ModuleState::Stopped
1689                | ModuleState::Failed
1690                | ModuleState::Disabled => false,
1691            }
1692    }
1693}
1694
1695#[cfg(test)]
1696#[derive(Debug, Default)]
1697struct ReloadExitRecordGate {
1698    reached: tokio::sync::Notify,
1699    resume: tokio::sync::Notify,
1700}
1701
1702#[derive(Debug, Clone, Copy)]
1703enum RespawnKind {
1704    Spawn,
1705    Reload,
1706}
1707
1708#[derive(Debug, Clone, Copy)]
1709struct PendingRespawn {
1710    deadline: Instant,
1711    kind: RespawnKind,
1712}
1713
1714type ReloadReply = oneshot::Sender<Result<(), SuperviseError>>;
1715
1716#[derive(Debug, Clone)]
1717struct SupervisorRuntimeConfig {
1718    /// Restart operations hand their backoff to the loop, which keeps serving commands.
1719    scheduled_respawn: Arc<Mutex<Option<PendingRespawn>>>,
1720    /// A reload acknowledges completion only after its replacement registers.
1721    deferred_reload_reply: Arc<Mutex<Option<ReloadReply>>>,
1722    restart_policy: RestartPolicy,
1723    /// This module's RESOLVED drain budget: per-module config when present,
1724    /// else `default_drain_timeout`.
1725    drain_timeout: Duration,
1726    /// Shared with the status handle so the attested value changes atomically
1727    /// when a rescan updates the running drain policy.
1728    effective_drain_timeout: Arc<Mutex<Duration>>,
1729    /// The supervisor-wide fallback, kept so a configuration update that
1730    /// REMOVES the per-module override can re-resolve to it.
1731    default_drain_timeout: Duration,
1732    health: HealthConfig,
1733    connection_file_path: Option<PathBuf>,
1734    capture_logs_dir: Option<PathBuf>,
1735    forwarding: Option<Arc<ForwardingTable>>,
1736    /// The shared handle, so every spawn path (initial, restart, reload) records the
1737    /// reserved-module launch nonce the HELLO verifier checks against.
1738    supervisor_handle: Option<SupervisorHandle>,
1739    /// This module's stderr tail, shared with the [`SupervisedModule`] that answers
1740    /// status queries.
1741    ///
1742    /// One ring per module, held across every respawn. The lines explaining an exit
1743    /// are written BEFORE that exit, so a ring recreated per process would be empty
1744    /// exactly when it is asked for.
1745    stderr_ring: Arc<Mutex<StderrRing>>,
1746    terminal_ring: Arc<Mutex<TerminalRing>>,
1747    spawn_events: SpawnEventFeed,
1748    child_roster: ChildRoster,
1749    #[cfg(target_os = "linux")]
1750    cgroup_placement: Option<subc_cgroup::Placement>,
1751    #[cfg(test)]
1752    test_seed_stale_facts_before_enable_spawn: bool,
1753    #[cfg(test)]
1754    test_reload_exit_record_gate: Option<Arc<ReloadExitRecordGate>>,
1755}
1756
1757#[derive(Debug, Clone, PartialEq, Eq)]
1758struct SupervisedConfiguration {
1759    spec: ModuleSpec,
1760    health: HealthConfig,
1761}
1762
1763/// Shared daemon lookup table for supervised module handles.
1764///
1765/// Shared by clone between the [`Supervisor`] (which spawns processes) and the
1766/// channel-0 control handler (which verifies HELLOs and consumer route opens), so
1767/// launch nonces recorded at spawn are checked by the same daemon instance.
1768#[derive(Debug, Clone, Default)]
1769pub struct SupervisorHandle {
1770    modules: Arc<Mutex<HashMap<String, SupervisedModule>>>,
1771    /// Module ids the supervisor has taken on. An id is added BEFORE the
1772    /// module's first process is spawned and removed only when the module
1773    /// leaves the roster (see [`Self::retire`]), so it is always a superset of
1774    /// the keys of `modules`.
1775    ///
1776    /// `modules` cannot answer "is this module configured?" on its own: a
1777    /// [`SupervisedModule`] only exists once its process has been spawned, and
1778    /// a fast child can connect, register, sync its scopes and ask about them
1779    /// before the supervisor has inserted it. Answering "not configured" in that
1780    /// gap makes scope admission refuse with the terminal "will never sync"
1781    /// instead of the retryable "has not synced yet".
1782    configured_ids: Arc<Mutex<HashSet<String>>>,
1783    spawn_events: SpawnEventFeed,
1784    /// The current expected launch nonce for each reserved module_id. Set when the
1785    /// supervisor spawns the reserved module; checked when a HELLO claims that id. A
1786    /// non-reserved module never has an entry here and is never nonce-checked.
1787    /// Reserved module ids and the nonce that authorizes their next HELLO.
1788    /// `None` means RESERVED WITH NO LEGITIMATE HOLDER — a reserved module that
1789    /// has never been spawned (e.g. configured `enabled: false`) — and refuses
1790    /// every HELLO. Before this was expressible, a reserved-but-never-spawned id
1791    /// had NO entry and admitted anyone: the reservation protected the nonce
1792    /// holder, not the NAME (found live by CKCRED's canary probe registering
1793    /// against a reserved scratch id).
1794    reserved_nonces: Arc<Mutex<HashMap<String, Option<String>>>>,
1795    /// Module ids removed by an executed rescan and the unix-millisecond removal time.
1796    ///
1797    /// This is deliberately in-memory only: subc is state-free across daemon
1798    /// restarts, and the tombstone only explains the hours-after-removal window
1799    /// while this executing daemon is still alive. Do not persist it in a store.
1800    removal_tombstones: Arc<Mutex<HashMap<String, u64>>>,
1801    /// The current launch nonce for every supervised spawn. This is separate from
1802    /// reserved_nonces because consumer route.open attestation applies to all spawned
1803    /// modules, while HELLO id-squatting protection remains opt-in via `reserved`.
1804    spawn_nonces: Arc<Mutex<HashMap<String, String>>>,
1805    /// Reserved namespace prefixes mapped to the supervised owner module whose
1806    /// current spawn nonce authorizes HELLO claims below the prefix.
1807    ///
1808    /// Per §2.6 this is not a same-user security barrier: a same-user process can
1809    /// read the key file and launch nonce env. Like exact reserved ids, it prevents
1810    /// accidental collisions and lower-trust processes from squatting protected
1811    /// namespaces.
1812    reserved_prefix_owners: Arc<Mutex<HashMap<String, String>>>,
1813    /// Blue/green swaps in progress, by module id. An entry exists from just
1814    /// before the candidate process is spawned until the swap has failed, or
1815    /// has cut over and the old process is gone. While it exists, HELLO for the
1816    /// id is gated on the swap token (see [`Self::swap_hello_admission`]) and
1817    /// consumer attestation accepts both processes' nonces.
1818    swaps: Arc<Mutex<HashMap<String, OpenSwap>>>,
1819    /// Told when a swap promotes its candidate; see [`SwapPromotionObserver`].
1820    promotion_observer: PromotionObserverSlot,
1821    /// Serializes module-set reconciliation with operator lifecycle commands. Without
1822    /// this daemon-wide ordering, a rescan could retire or update a module while a
1823    /// concurrent reload still held its old handle and launch specification.
1824    operation_lock: Arc<AsyncMutex<()>>,
1825}
1826
1827/// Told when a swap has promoted its candidate to be the module's active
1828/// registration.
1829///
1830/// An ordinary HELLO runs the control plane's registration side effects (the
1831/// capability cache, the deny census, the requirement recompute) as it
1832/// registers. A swap candidate's HELLO does not, because it is not routable;
1833/// promotion is when those must run instead, and promotion happens in the
1834/// supervisor, which has no other way into the control handler.
1835pub(crate) trait SwapPromotionObserver: Send + Sync {
1836    fn swap_promoted(&self, registration: &crate::registry::ModuleRegistration);
1837}
1838
1839/// The installed [`SwapPromotionObserver`], held weakly: the observer (the
1840/// control handler) owns this handle, so a strong reference back would be a
1841/// cycle that keeps both alive.
1842#[derive(Clone, Default)]
1843struct PromotionObserverSlot(Arc<Mutex<Option<std::sync::Weak<dyn SwapPromotionObserver>>>>);
1844
1845impl fmt::Debug for PromotionObserverSlot {
1846    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1847        f.write_str("PromotionObserverSlot")
1848    }
1849}
1850
1851/// The nonces of one open swap.
1852#[derive(Debug, Clone)]
1853struct OpenSwap {
1854    /// The launch nonce minted for the candidate process. It is the swap
1855    /// token: the only thing that admits a HELLO into the candidate slot.
1856    candidate_nonce: String,
1857    /// The incumbent's launch nonce, captured when the swap opened. It is kept
1858    /// here because cutover moves the module's recorded spawn nonce to the
1859    /// candidate while the incumbent is still draining and its consumers are
1860    /// still attesting with this one.
1861    incumbent_nonce: Option<String>,
1862    /// Set once a HELLO has been admitted with the swap token, so the token
1863    /// admits one registration and cannot be replayed after cutover empties
1864    /// the candidate slot.
1865    candidate_admitted: bool,
1866}
1867
1868/// What the swap gate says about a HELLO. See
1869/// [`SupervisorHandle::swap_hello_admission`].
1870#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1871pub(crate) enum SwapHelloAdmission {
1872    /// No swap is open for the id (or the HELLO carries the incumbent's own
1873    /// nonce); the ordinary gates decide.
1874    NotSwapping,
1875    /// The HELLO carries the swap token: register it into the candidate slot.
1876    Candidate,
1877    /// A swap is open and the HELLO carries a nonce the supervisor did not
1878    /// mint for this id, no nonce, or a token already used.
1879    Refused,
1880}
1881
1882#[derive(Debug, Clone, PartialEq, Eq)]
1883pub(crate) enum ReservedHelloRejection {
1884    Exact {
1885        module_id: String,
1886    },
1887    Prefix {
1888        prefix: String,
1889        owner_module_id: String,
1890    },
1891}
1892
1893impl SupervisorHandle {
1894    pub fn new() -> Self {
1895        Self::default()
1896    }
1897
1898    pub(crate) fn spawn_snapshot(&self) -> SpawnSnapshot {
1899        self.spawn_events.snapshot()
1900    }
1901
1902    pub(crate) fn subscribe_spawns(
1903        &self,
1904        connection_id: ConnectionId,
1905        corr: u64,
1906        version: u8,
1907        since: Option<SpawnCursor>,
1908        sink: FrameSink,
1909    ) -> Result<(), SpawnSubscribeRefusal> {
1910        self.spawn_events
1911            .subscribe(connection_id, corr, version, since, sink)
1912    }
1913
1914    pub(crate) fn cancel_spawn_subscription(&self, connection_id: ConnectionId, corr: u64) -> bool {
1915        self.spawn_events.cancel(connection_id, corr)
1916    }
1917
1918    pub(crate) fn remove_spawn_subscribers(&self, connection_id: ConnectionId) {
1919        self.spawn_events.remove_connection(connection_id);
1920    }
1921
1922    #[cfg(any(test, feature = "test-support"))]
1923    pub fn set_spawn_event_capacity_for_test(&self, capacity: usize) {
1924        assert!(capacity > 0, "spawn event capacity must be non-zero");
1925        self.spawn_events.set_capacity(capacity);
1926    }
1927
1928    #[cfg(any(test, feature = "test-support"))]
1929    pub fn spawn_subscriber_count_for_test(&self) -> usize {
1930        self.spawn_events.subscriber_count()
1931    }
1932
1933    /// Record the launch nonce from a supervised spawn, replacing any prior nonce so
1934    /// a respawn invalidates stale consumer identities.
1935    pub fn set_spawn_nonce(&self, module_id: &str, nonce: String) {
1936        self.spawn_nonces
1937            .lock()
1938            .unwrap_or_else(|poisoned| poisoned.into_inner())
1939            .insert(module_id.to_string(), nonce);
1940    }
1941
1942    /// Record the launch nonce expected from the next HELLO for a reserved module,
1943    /// replacing any prior nonce (a respawn invalidates the previous one).
1944    pub fn set_reserved_nonce(&self, module_id: &str, nonce: String) {
1945        self.reserved_nonces
1946            .lock()
1947            .unwrap_or_else(|poisoned| poisoned.into_inner())
1948            .insert(module_id.to_string(), Some(nonce));
1949    }
1950
1951    /// Record namespace prefixes owned by a supervised module.
1952    pub fn set_reserved_prefixes(&self, owner_module_id: &str, prefixes: &[String]) {
1953        let mut owners = self
1954            .reserved_prefix_owners
1955            .lock()
1956            .unwrap_or_else(|poisoned| poisoned.into_inner());
1957        owners.retain(|_, owner| owner != owner_module_id);
1958        for prefix in prefixes {
1959            owners.insert(prefix.clone(), owner_module_id.to_string());
1960        }
1961    }
1962
1963    /// The launch nonce most recently minted for a module's spawn, if any.
1964    #[cfg(test)]
1965    pub(crate) fn spawn_nonce(&self, module_id: &str) -> Option<String> {
1966        self.spawn_nonces
1967            .lock()
1968            .unwrap_or_else(|poisoned| poisoned.into_inner())
1969            .get(module_id)
1970            .cloned()
1971    }
1972
1973    fn apply_identity_configuration(&self, spec: &ModuleSpec) {
1974        self.set_reserved_prefixes(&spec.module_id, &spec.reserved_prefixes);
1975        let spawn_nonce = self
1976            .spawn_nonces
1977            .lock()
1978            .unwrap_or_else(|poisoned| poisoned.into_inner())
1979            .get(&spec.module_id)
1980            .cloned();
1981        let mut reserved_nonces = self
1982            .reserved_nonces
1983            .lock()
1984            .unwrap_or_else(|poisoned| poisoned.into_inner());
1985        if spec.reserved {
1986            // `None` (no spawn nonce minted) is INSERTED, not skipped: a
1987            // reserved name whose module has never spawned has no legitimate
1988            // holder, and the entry's absence is what used to leave the name
1989            // open to the first claimant.
1990            reserved_nonces.insert(spec.module_id.clone(), spawn_nonce);
1991        }
1992        drop(reserved_nonces);
1993        // A later unreserved declaration must not silently unreserve an id that
1994        // was retained after its reserved configuration was removed. The explicit
1995        // release ceremony is the only operation that retires that gate.
1996        self.removal_tombstones
1997            .lock()
1998            .unwrap_or_else(|poisoned| poisoned.into_inner())
1999            .remove(&spec.module_id);
2000    }
2001
2002    /// Whether a HELLO claiming `module_id` is authorized. An exact reserved id is
2003    /// authorized only by its expected nonce; otherwise a matching reserved prefix
2004    /// is authorized by the owner module's current spawn nonce. Non-reserved ids
2005    /// with no matching prefix are always authorized.
2006    pub fn reserved_hello_authorized(&self, module_id: &str, presented: Option<&str>) -> bool {
2007        self.reserved_hello_rejection(module_id, presented)
2008            .is_none()
2009    }
2010
2011    pub(crate) fn reserved_hello_rejection(
2012        &self,
2013        module_id: &str,
2014        presented: Option<&str>,
2015    ) -> Option<ReservedHelloRejection> {
2016        let nonces = self
2017            .reserved_nonces
2018            .lock()
2019            .unwrap_or_else(|poisoned| poisoned.into_inner());
2020        if let Some(expected) = nonces.get(module_id) {
2021            // `None` = reserved with no legitimate holder: refuse every
2022            // presentation, because no process can hold a nonce that was never
2023            // minted. Only a real minted nonce admits, in constant time.
2024            let authorized = match expected {
2025                Some(expected) => {
2026                    presented.is_some_and(|p| constant_time_eq(expected.as_bytes(), p.as_bytes()))
2027                }
2028                None => false,
2029            };
2030            if authorized {
2031                return None;
2032            }
2033            return Some(ReservedHelloRejection::Exact {
2034                module_id: module_id.to_string(),
2035            });
2036        }
2037        drop(nonces);
2038
2039        let matched_prefix = self
2040            .reserved_prefix_owners
2041            .lock()
2042            .unwrap_or_else(|poisoned| poisoned.into_inner())
2043            .iter()
2044            .filter(|(prefix, _)| module_id.starts_with(prefix.as_str()))
2045            .max_by_key(|(prefix, _)| prefix.len())
2046            .map(|(prefix, owner)| (prefix.clone(), owner.clone()));
2047        let (prefix, owner_module_id) = matched_prefix?;
2048
2049        let authorized = presented.is_some_and(|presented| {
2050            self.spawn_nonces
2051                .lock()
2052                .unwrap_or_else(|poisoned| poisoned.into_inner())
2053                .get(&owner_module_id)
2054                .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()))
2055                // While the owner is being swapped, children started by
2056                // either of its two processes hold that process's nonce.
2057                || self.swap_nonce_matches(&owner_module_id, presented)
2058        });
2059        if authorized {
2060            None
2061        } else {
2062            Some(ReservedHelloRejection::Prefix {
2063                prefix,
2064                owner_module_id,
2065            })
2066        }
2067    }
2068
2069    /// Whether a consumer connection proved it came from a daemon-spawned module.
2070    ///
2071    /// Absence of an expected spawn nonce is a hard failure: consumer_identity is
2072    /// accepted only for module ids the supervisor has spawned.
2073    pub fn spawned_consumer_authorized(&self, module_id: &str, presented: &str) -> bool {
2074        if presented.is_empty() {
2075            return false;
2076        }
2077        let nonces = self
2078            .spawn_nonces
2079            .lock()
2080            .unwrap_or_else(|poisoned| poisoned.into_inner());
2081        let current = nonces
2082            .get(module_id)
2083            .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()));
2084        drop(nonces);
2085        // During a swap two processes of the module are alive, and a consumer
2086        // started by either one presents that process's nonce. Accepting only
2087        // the recorded one would fail the incumbent's consumers for the whole
2088        // overlap once cutover moves the record to the candidate.
2089        current || self.swap_nonce_matches(module_id, presented)
2090    }
2091
2092    /// Whether `presented` is either nonce of an open swap for `module_id`.
2093    fn swap_nonce_matches(&self, module_id: &str, presented: &str) -> bool {
2094        let swaps = self
2095            .swaps
2096            .lock()
2097            .unwrap_or_else(|poisoned| poisoned.into_inner());
2098        swaps.get(module_id).is_some_and(|swap| {
2099            constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes())
2100                || swap.incumbent_nonce.as_deref().is_some_and(|incumbent| {
2101                    constant_time_eq(incumbent.as_bytes(), presented.as_bytes())
2102                })
2103        })
2104    }
2105
2106    /// Open a swap for `module_id` with the candidate's freshly minted nonce.
2107    /// Called before the candidate process exists.
2108    pub(crate) fn open_swap(&self, module_id: &str, candidate_nonce: String) {
2109        let incumbent_nonce = self
2110            .spawn_nonces
2111            .lock()
2112            .unwrap_or_else(|poisoned| poisoned.into_inner())
2113            .get(module_id)
2114            .cloned();
2115        self.swaps
2116            .lock()
2117            .unwrap_or_else(|poisoned| poisoned.into_inner())
2118            .insert(
2119                module_id.to_string(),
2120                OpenSwap {
2121                    candidate_nonce,
2122                    incumbent_nonce,
2123                    candidate_admitted: false,
2124                },
2125            );
2126    }
2127
2128    /// Close the swap for `module_id`, releasing whichever nonce is no longer
2129    /// the module's recorded one.
2130    pub(crate) fn close_swap(&self, module_id: &str) {
2131        self.swaps
2132            .lock()
2133            .unwrap_or_else(|poisoned| poisoned.into_inner())
2134            .remove(module_id);
2135    }
2136
2137    /// Install the observer told about swap promotions, replacing any earlier
2138    /// one.
2139    pub(crate) fn set_swap_promotion_observer(
2140        &self,
2141        observer: std::sync::Weak<dyn SwapPromotionObserver>,
2142    ) {
2143        *self
2144            .promotion_observer
2145            .0
2146            .lock()
2147            .unwrap_or_else(|poisoned| poisoned.into_inner()) = Some(observer);
2148    }
2149
2150    /// Tell the installed observer, if it is still alive, that a swap promoted
2151    /// `registration`.
2152    fn notify_swap_promoted(&self, registration: &crate::registry::ModuleRegistration) {
2153        let observer = self
2154            .promotion_observer
2155            .0
2156            .lock()
2157            .unwrap_or_else(|poisoned| poisoned.into_inner())
2158            .as_ref()
2159            .and_then(std::sync::Weak::upgrade);
2160        if let Some(observer) = observer {
2161            observer.swap_promoted(registration);
2162        }
2163    }
2164
2165    /// Whether a swap is open for `module_id`.
2166    pub(crate) fn swap_open(&self, module_id: &str) -> bool {
2167        self.swaps
2168            .lock()
2169            .unwrap_or_else(|poisoned| poisoned.into_inner())
2170            .contains_key(module_id)
2171    }
2172
2173    /// Make the candidate's nonce the module's recorded spawn nonce, as a plain
2174    /// respawn would, once cutover has made the candidate the module's process.
2175    /// The swap stays open so the incumbent's nonce keeps attesting until the
2176    /// incumbent has drained and exited.
2177    fn promote_swap_nonce(&self, module_id: &str, reserved: bool) {
2178        let candidate_nonce = self
2179            .swaps
2180            .lock()
2181            .unwrap_or_else(|poisoned| poisoned.into_inner())
2182            .get(module_id)
2183            .map(|swap| swap.candidate_nonce.clone());
2184        let Some(nonce) = candidate_nonce else {
2185            return;
2186        };
2187        self.set_spawn_nonce(module_id, nonce.clone());
2188        if reserved {
2189            self.set_reserved_nonce(module_id, nonce);
2190        }
2191    }
2192
2193    /// The swap gate for a HELLO claiming `module_id`.
2194    ///
2195    /// This runs BEFORE the reserved-module gate. A reserved module's candidate
2196    /// presents the candidate nonce, which the reserved gate (holding the
2197    /// incumbent's nonce) would refuse as `reserved_module` before swap
2198    /// admission was ever reached. And it applies to unreserved ids too: for an
2199    /// unreserved id the only thing that ever stopped a second process claiming
2200    /// a live id was the `duplicate_module_id` refusal, which is exactly the
2201    /// refusal a swap lifts for its candidate.
2202    ///
2203    /// The incumbent's own nonce falls through to the ordinary gates, which
2204    /// treat it as they always have (a live incumbent is refused as a
2205    /// duplicate). Anything else while a swap is open is refused, including an
2206    /// absent nonce.
2207    pub(crate) fn swap_hello_admission(
2208        &self,
2209        module_id: &str,
2210        presented: Option<&str>,
2211    ) -> SwapHelloAdmission {
2212        let swaps = self
2213            .swaps
2214            .lock()
2215            .unwrap_or_else(|poisoned| poisoned.into_inner());
2216        let Some(swap) = swaps.get(module_id) else {
2217            return SwapHelloAdmission::NotSwapping;
2218        };
2219        let Some(presented) = presented else {
2220            return SwapHelloAdmission::Refused;
2221        };
2222        if constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes()) {
2223            return if swap.candidate_admitted {
2224                SwapHelloAdmission::Refused
2225            } else {
2226                SwapHelloAdmission::Candidate
2227            };
2228        }
2229        if swap
2230            .incumbent_nonce
2231            .as_deref()
2232            .is_some_and(|incumbent| constant_time_eq(incumbent.as_bytes(), presented.as_bytes()))
2233        {
2234            return SwapHelloAdmission::NotSwapping;
2235        }
2236        SwapHelloAdmission::Refused
2237    }
2238
2239    /// Record that the swap token has registered a candidate, so it admits no
2240    /// second HELLO.
2241    pub(crate) fn mark_swap_candidate_admitted(&self, module_id: &str) {
2242        if let Some(swap) = self
2243            .swaps
2244            .lock()
2245            .unwrap_or_else(|poisoned| poisoned.into_inner())
2246            .get_mut(module_id)
2247        {
2248            swap.candidate_admitted = true;
2249        }
2250    }
2251
2252    /// Test/support lookup for the current launch nonce of a supervised spawn.
2253    pub fn spawn_launch_nonce_for(&self, module_id: &str) -> Option<String> {
2254        self.spawn_nonces
2255            .lock()
2256            .unwrap_or_else(|poisoned| poisoned.into_inner())
2257            .get(module_id)
2258            .cloned()
2259    }
2260
2261    /// Test/support lookup for the HELLO-gating nonce of a reserved module.
2262    pub fn reserved_launch_nonce_for(&self, module_id: &str) -> Option<String> {
2263        self.reserved_nonces
2264            .lock()
2265            .unwrap_or_else(|poisoned| poisoned.into_inner())
2266            .get(module_id)
2267            .cloned()
2268            .flatten()
2269    }
2270
2271    pub fn insert(&self, module: SupervisedModule) -> Option<SupervisedModule> {
2272        // Normally already marked before the process was spawned; marking here
2273        // too keeps `configured_ids` a superset of the roster for any caller
2274        // that inserts a module directly.
2275        self.mark_configured(module.module_id());
2276        let mut modules = self
2277            .modules
2278            .lock()
2279            .unwrap_or_else(|poisoned| poisoned.into_inner());
2280        modules.insert(module.module_id().to_string(), module)
2281    }
2282
2283    /// Record that the supervisor has taken on `module_id`. Called before the
2284    /// module's first process is spawned, so that by the time that process can
2285    /// register, [`Self::is_configured`] already answers true.
2286    fn mark_configured(&self, module_id: &str) {
2287        self.configured_ids
2288            .lock()
2289            .unwrap_or_else(|poisoned| poisoned.into_inner())
2290            .insert(module_id.to_string());
2291    }
2292
2293    /// Undo [`Self::mark_configured`] for a module whose first spawn failed
2294    /// before it was ever put on the roster. A module already on the roster
2295    /// keeps its mark: only [`Self::retire`] takes a rostered module off.
2296    fn unmark_configured_unless_rostered(&self, module_id: &str) {
2297        let modules = self
2298            .modules
2299            .lock()
2300            .unwrap_or_else(|poisoned| poisoned.into_inner());
2301        if !modules.contains_key(module_id) {
2302            self.configured_ids
2303                .lock()
2304                .unwrap_or_else(|poisoned| poisoned.into_inner())
2305                .remove(module_id);
2306        }
2307    }
2308
2309    /// Whether `module_id` is a module this daemon supervises: on the roster,
2310    /// or about to be (its process is being spawned right now).
2311    ///
2312    /// This, not `get(..).is_some()`, is what "the owner is configured" means
2313    /// for scopes: a supervised module's process can register and sync before
2314    /// [`Self::get`] can return it, and in that window it is still a module
2315    /// that will sync, not one that never will.
2316    pub(crate) fn is_configured(&self, module_id: &str) -> bool {
2317        self.configured_ids
2318            .lock()
2319            .unwrap_or_else(|poisoned| poisoned.into_inner())
2320            .contains(module_id)
2321    }
2322
2323    pub fn get(&self, module_id: &str) -> Option<SupervisedModule> {
2324        let modules = self
2325            .modules
2326            .lock()
2327            .unwrap_or_else(|poisoned| poisoned.into_inner());
2328        modules.get(module_id).cloned()
2329    }
2330
2331    pub(crate) fn record_late_health_answer(
2332        &self,
2333        module_id: &str,
2334        latency_ms: u64,
2335    ) -> Result<bool, SuperviseError> {
2336        let Some(module) = self.get(module_id) else {
2337            return Ok(false);
2338        };
2339        update_snapshot(&module.inner.snapshot, Some(module_id), |state| {
2340            state.health.late_answer_count = state.health.late_answer_count.saturating_add(1);
2341            state.health.last_late_answer_latency_ms = Some(latency_ms);
2342            // A late answer is an answer: the module served the probe, just past
2343            // the deadline. Leaving the miss streak in place while logging
2344            // "proves the module is alive" is how a CPU-starved module that
2345            // answers every probe a few seconds late still marches to the
2346            // threshold and gets killed — the exact kill class `NoAnswer` is
2347            // excluded from `is_proof_of_death` to prevent. Slow-but-answering
2348            // is degradation, and degradation reports; it does not restart.
2349            state.health.consecutive_failures = 0;
2350        })?;
2351        Ok(true)
2352    }
2353
2354    /// Arm the one-shot marker for the module process that this caller
2355    /// deliberately initiated severance against. Generic connection teardown
2356    /// must not call this:
2357    /// a surviving process would otherwise retain an exemption for a later
2358    /// genuine crash.
2359    pub fn record_deliberate_severance(&self, module_id: &str) -> Result<bool, SuperviseError> {
2360        let Some(module) = self.get(module_id) else {
2361            return Ok(false);
2362        };
2363        let snapshot = lock_snapshot(&module.inner.snapshot)?;
2364        let Some((pid, start_time)) = snapshot.pid.zip(snapshot.process_start_time) else {
2365            return Ok(false);
2366        };
2367        drop(snapshot);
2368        module.record_deliberate_severance(ProcessIdentity { pid, start_time })
2369    }
2370
2371    pub fn list(&self) -> Vec<SupervisedModule> {
2372        let modules = self
2373            .modules
2374            .lock()
2375            .unwrap_or_else(|poisoned| poisoned.into_inner());
2376        let mut modules = modules.values().cloned().collect::<Vec<_>>();
2377        modules.sort_by(|left, right| left.module_id().cmp(right.module_id()));
2378        modules
2379    }
2380
2381    pub(crate) fn retire(&self, module_id: &str) -> Option<SupervisedModule> {
2382        self.spawn_nonces
2383            .lock()
2384            .unwrap_or_else(|poisoned| poisoned.into_inner())
2385            .remove(module_id);
2386        self.close_swap(module_id);
2387        let mut reserved_nonces = self
2388            .reserved_nonces
2389            .lock()
2390            .unwrap_or_else(|poisoned| poisoned.into_inner());
2391        if reserved_nonces.contains_key(module_id) {
2392            // The old nonce must die with the removed process, but the exact-id
2393            // gate remains until an operator explicitly releases it.
2394            reserved_nonces.insert(module_id.to_string(), None);
2395        }
2396        drop(reserved_nonces);
2397        self.reserved_prefix_owners
2398            .lock()
2399            .unwrap_or_else(|poisoned| poisoned.into_inner())
2400            .retain(|_, owner| owner != module_id);
2401        let removed = self
2402            .modules
2403            .lock()
2404            .unwrap_or_else(|poisoned| poisoned.into_inner())
2405            .remove(module_id);
2406        self.configured_ids
2407            .lock()
2408            .unwrap_or_else(|poisoned| poisoned.into_inner())
2409            .remove(module_id);
2410        removed
2411    }
2412
2413    /// Remember a module removed by a non-preview rescan so route.open can
2414    /// distinguish that intentional removal from an unknown id.
2415    pub(crate) fn record_rescan_removal(&self, module_id: &str) {
2416        self.removal_tombstones
2417            .lock()
2418            .unwrap_or_else(|poisoned| poisoned.into_inner())
2419            .insert(module_id.to_string(), unix_ms_now());
2420    }
2421
2422    /// Return how long ago a rescan removed this module in milliseconds.
2423    pub(crate) fn removal_tombstone_age_ms(&self, module_id: &str) -> Option<u64> {
2424        self.removal_tombstones
2425            .lock()
2426            .unwrap_or_else(|poisoned| poisoned.into_inner())
2427            .get(module_id)
2428            .copied()
2429            .map(|removed_at_ms| unix_ms_now().saturating_sub(removed_at_ms))
2430    }
2431
2432    /// Retire a reserved-id gate only after its module has left supervision.
2433    ///
2434    /// A retained gate has no live nonce (`None`), so releasing any other entry
2435    /// would weaken a currently configured or otherwise active reservation.
2436    pub(crate) fn release_retained_reserved_gate(&self, module_id: &str) -> bool {
2437        if self.get(module_id).is_some() {
2438            return false;
2439        }
2440        let mut reserved_nonces = self
2441            .reserved_nonces
2442            .lock()
2443            .unwrap_or_else(|poisoned| poisoned.into_inner());
2444        if !matches!(reserved_nonces.get(module_id), Some(None)) {
2445            return false;
2446        }
2447        reserved_nonces.remove(module_id);
2448        true
2449    }
2450
2451    pub(crate) fn operation_lock(&self) -> Arc<AsyncMutex<()>> {
2452        Arc::clone(&self.operation_lock)
2453    }
2454}
2455
2456/// Process supervisor for subc-owned singleton modules.
2457#[derive(Debug, Clone)]
2458pub struct Supervisor {
2459    registry: Arc<Registry>,
2460    restart_policy: RestartPolicy,
2461    drain_timeout: Duration,
2462    connection_file_path: Option<PathBuf>,
2463    capture_logs_dir: Option<PathBuf>,
2464    forwarding: Option<Arc<ForwardingTable>>,
2465    process_liveness: Arc<SupervisorProcessLiveness>,
2466    supervisor_handle: Option<SupervisorHandle>,
2467    health: HealthConfig,
2468    daemon_start_clock: crate::clock::StartClock,
2469    terminal_journal: Option<Arc<crate::terminal_journal::TerminalJournal>>,
2470    spawn_events: SpawnEventFeed,
2471    provenance_probe: ExecutableIdentityProbe,
2472    /// Every process spawned through this supervisor (and its clones) and not
2473    /// yet reaped, so daemon shutdown can end them.
2474    child_roster: ChildRoster,
2475    #[cfg(target_os = "linux")]
2476    cgroup_placement: Option<subc_cgroup::Placement>,
2477    #[cfg(test)]
2478    test_after_first_spawn: AfterFirstSpawnHook,
2479}
2480
2481/// Test-only hook run on the path that takes on a new module, right after its
2482/// first `spawn_child` returns (the process exists and could already be
2483/// registering) and before that process is handed to the module's supervise
2484/// loop and put on the roster. Lets a test observe what a fast child would see
2485/// in that window without racing a real one.
2486#[cfg(test)]
2487#[derive(Clone, Default)]
2488struct AfterFirstSpawnHook(Option<AfterFirstSpawnFn>);
2489
2490#[cfg(test)]
2491type AfterFirstSpawnFn = Arc<dyn Fn(&str) + Send + Sync>;
2492
2493#[cfg(test)]
2494impl fmt::Debug for AfterFirstSpawnHook {
2495    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
2496        f.write_str("AfterFirstSpawnHook")
2497    }
2498}
2499
2500#[cfg(test)]
2501impl AfterFirstSpawnHook {
2502    fn run(&self, module_id: &str) {
2503        if let Some(hook) = &self.0 {
2504            hook(module_id);
2505        }
2506    }
2507}
2508
2509impl Supervisor {
2510    #[cfg(test)]
2511    pub(crate) fn new_for_test(registry: Arc<Registry>, policy: RestartPolicy) -> Self {
2512        let supervisor = Self::new(registry, policy);
2513        #[cfg(target_os = "macos")]
2514        let supervisor = supervisor.with_privacy_trampoline(test_privacy_trampoline());
2515        supervisor
2516    }
2517    /// Verify the trampoline once when configured. Missing private OS support
2518    /// refuses every macOS launch by name but does not stop the daemon's control
2519    /// server. Embedders must explicitly provide a binary with the subc-os hidden
2520    /// entry point; the library must not exec an arbitrary hosting program.
2521    pub fn with_privacy_trampoline(self, path: impl Into<PathBuf>) -> Self {
2522        let path = path.into();
2523        #[cfg(target_os = "macos")]
2524        {
2525            let result = probe_privacy_trampoline(&path).map(|()| path);
2526            if let Err(cause) = &result {
2527                error!(%cause, "macOS privacy identity unavailable; supervised launches will refuse");
2528            }
2529            self.child_roster.set_privacy_trampoline(result);
2530        }
2531        #[cfg(not(target_os = "macos"))]
2532        let _ = path;
2533        self
2534    }
2535    /// The first step of an announced daemon shutdown, before the notice and
2536    /// before any connection is closed.
2537    ///
2538    /// Sets the daemon-shutdown flag first: from here on no module is
2539    /// respawned (crash restart, operator restart, or swap), and every child
2540    /// exit is recorded as `daemon_shutdown` rather than as a crash, whether
2541    /// the module exits on the EOF this shutdown gives it or is signalled by a
2542    /// service manager that kills the whole cgroup. Then writes the journal's
2543    /// shutdown marker, which records the instant and closes this daemon
2544    /// incarnation's stretch of the journal.
2545    #[cfg(unix)]
2546    pub(crate) fn begin_daemon_shutdown(&self) {
2547        self.child_roster.close();
2548        if let Some(journal) = &self.terminal_journal {
2549            journal.stamp_shutdown();
2550        }
2551    }
2552
2553    /// Announce a cut while established connections can still carry replies.
2554    /// These budgets promise notice and a bounded wait, not child completion;
2555    /// they are local policy, not an estimate of launchd's unknown kill ceiling.
2556    #[cfg(unix)]
2557    pub(crate) async fn drain_for_daemon_shutdown(&self) -> Result<(), SuperviseError> {
2558        const NOTICE_BUDGET: Duration = Duration::from_millis(500);
2559        const DRAIN_BUDGET: Duration = Duration::from_secs(2);
2560        let Some(forwarding) = &self.forwarding else {
2561            return Ok(());
2562        };
2563        let module_ids = forwarding
2564            .begin_daemon_drain()
2565            .map_err(SuperviseError::Forwarding)?;
2566        let deadline_ms =
2567            unix_ms_now().saturating_add((NOTICE_BUDGET + DRAIN_BUDGET).as_millis() as u64);
2568        let mut notices = tokio::task::JoinSet::new();
2569        let mut drains = Vec::new();
2570        for module_id in module_ids {
2571            let Some(target) = forwarding
2572                .begin_module_drain(&module_id, RouteCloseReason::Restart)
2573                .map_err(SuperviseError::Forwarding)?
2574            else {
2575                continue;
2576            };
2577            let routes = forwarding
2578                .endpoint_routes(target.endpoint)
2579                .map_err(SuperviseError::Forwarding)?;
2580            // Restart allows deployed consumers to reopen after the new daemon
2581            // appears. The wire reason stays `restart`; what tells a daemon cut
2582            // apart from a module restart afterwards is the terminal record
2583            // itself, whose disposition is `daemon_shutdown` for every exit
2584            // observed once `begin_daemon_shutdown` has run.
2585            let command = serde_json::to_vec(&ModuleControlCommand::Draining {
2586                reason: RouteCloseReason::Restart,
2587                deadline_ms,
2588            })
2589            .expect("module draining serializes");
2590            let mut recipients = vec![(target.sink.clone(), target.negotiated_ver, command)];
2591            let mut clients: Vec<(crate::forwarding::GoodbyeTarget, Vec<u16>)> = Vec::new();
2592            for route in routes {
2593                let client = route.goodbye_target;
2594                if let Some((_, channels)) = clients
2595                    .iter_mut()
2596                    .find(|(existing, _)| existing.connection_id == client.connection_id)
2597                {
2598                    channels.push(client.channel);
2599                } else {
2600                    let channel = client.channel;
2601                    clients.push((client, vec![channel]));
2602                }
2603            }
2604            for (client, mut channels) in clients {
2605                channels.sort_unstable();
2606                channels.dedup();
2607                let closing = serde_json::to_vec(&ClientControlPush::RouteClosing {
2608                    module_id: module_id.clone(),
2609                    channels,
2610                    reason: RouteCloseReason::Restart,
2611                })
2612                .expect("route closing serializes");
2613                recipients.push((client.sink, client.negotiated_ver, closing));
2614            }
2615            for (sink, version, body) in recipients {
2616                notices.spawn(async move {
2617                    let frame = Frame::build_with_version(
2618                        version,
2619                        FrameType::Push,
2620                        control_flags(),
2621                        0,
2622                        0,
2623                        0,
2624                        body,
2625                    )
2626                    .expect("bounded lifecycle notice frame builds");
2627                    sink.send_flushed(frame).await
2628                });
2629            }
2630            let gauges = declared_busy_gauges(&self.registry, &module_id)?;
2631            drains.push((module_id, target.endpoint, gauges));
2632        }
2633        // A quiet forwarding table is not proof that queued notices reached the
2634        // socket. Wait for writer flush acknowledgements before testing quiescence.
2635        let notice_deadline = Instant::now() + NOTICE_BUDGET;
2636        while let Ok(Some(result)) = timeout_at(notice_deadline, notices.join_next()).await {
2637            if !matches!(result, Ok(Ok(()))) {
2638                warn!(?result, "daemon shutdown notice delivery failed");
2639            }
2640        }
2641        notices.abort_all();
2642        let deadline = Instant::now() + DRAIN_BUDGET;
2643        let mut waits = tokio::task::JoinSet::new();
2644        for (module_id, endpoint, gauges) in drains {
2645            let forwarding = Arc::clone(forwarding);
2646            let mut runtime = self.runtime_config();
2647            runtime.health.cadence = Duration::from_millis(100);
2648            waits.spawn(async move {
2649                wait_for_forwarding_quiescence(
2650                    &forwarding,
2651                    &module_id,
2652                    &runtime,
2653                    endpoint,
2654                    deadline,
2655                    &gauges,
2656                    DrainScope::Active,
2657                )
2658                .await
2659            });
2660        }
2661        while let Ok(Some(result)) = timeout_at(deadline, waits.join_next()).await {
2662            if !matches!(result, Ok(Ok(true))) {
2663                warn!(?result, "daemon shutdown drain did not reach quiescence");
2664            }
2665        }
2666        Ok(())
2667    }
2668
2669    /// The last step of an announced daemon shutdown, after the notice and the
2670    /// drain: send every registered module a module GOODBYE, the same planned
2671    /// stop signal `ck module stop` gives, then close every connection so each
2672    /// subc module sees EOF and starts its own teardown, then end every
2673    /// supervised child that has not exited
2674    /// by its own deadline (its drain budget, capped). Modules lead their own
2675    /// process groups, so a
2676    /// service manager's group kill no longer reaches them; without this a
2677    /// child that does not stop on EOF (every `protocol: "none"` child, which
2678    /// has no connection) would outlive the daemon. Every wait is bounded (see
2679    /// `child_roster`), and `escalate` resolving (a second SIGTERM) cuts them.
2680    #[cfg(unix)]
2681    pub(crate) async fn end_children_for_daemon_shutdown(
2682        &self,
2683        already_escalated: bool,
2684        escalate: impl std::future::Future<Output = ()>,
2685    ) {
2686        tokio::pin!(escalate);
2687        let mut escalated = already_escalated;
2688        if let Some(forwarding) = &self.forwarding {
2689            let reason = CloseReason::new(
2690                "daemon_shutdown",
2691                "the daemon is exiting after its shutdown notice and drain",
2692            );
2693            if escalated {
2694                // The operator asked to stop waiting: queue the GOODBYEs but
2695                // do not wait for them to be written.
2696                send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, false).await;
2697            } else {
2698                tokio::select! {
2699                    biased;
2700                    _ = escalate.as_mut() => {
2701                        info!("second SIGTERM: abandoning module GOODBYE delivery");
2702                        escalated = true;
2703                    }
2704                    _ = send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, true) => {}
2705                }
2706            }
2707            let closed = forwarding.close_all_connections(&reason);
2708            debug!(closed, "closed established connections for daemon shutdown");
2709        }
2710        // If the second SIGTERM arrived during GOODBYE delivery, `escalate` has
2711        // already completed and must not be polled again; the child shutdown
2712        // wait is told it is escalated and gets a future that never fires.
2713        let escalated_here = escalated && !already_escalated;
2714        let remaining_escalate = async move {
2715            if escalated_here {
2716                std::future::pending::<()>().await;
2717            } else {
2718                escalate.await;
2719            }
2720        };
2721        crate::child_roster::end_children_for_daemon_shutdown(
2722            &self.child_roster,
2723            escalated,
2724            remaining_escalate,
2725        )
2726        .await;
2727    }
2728
2729    pub fn new(registry: Arc<Registry>, restart_policy: RestartPolicy) -> Self {
2730        Self {
2731            registry,
2732            restart_policy,
2733            drain_timeout: DEFAULT_DRAIN_TIMEOUT,
2734            connection_file_path: None,
2735            capture_logs_dir: None,
2736            forwarding: None,
2737            process_liveness: Arc::new(SupervisorProcessLiveness::default()),
2738            supervisor_handle: None,
2739            health: HealthConfig::default(),
2740            daemon_start_clock: crate::clock::StartClock::capture(),
2741            terminal_journal: None,
2742            spawn_events: SpawnEventFeed::default(),
2743            provenance_probe: ExecutableIdentityProbe::default(),
2744            child_roster: ChildRoster::default(),
2745            #[cfg(target_os = "linux")]
2746            cgroup_placement: None,
2747            #[cfg(test)]
2748            test_after_first_spawn: AfterFirstSpawnHook::default(),
2749        }
2750    }
2751
2752    pub fn with_drain_timeout(mut self, drain_timeout: Duration) -> Self {
2753        self.drain_timeout = drain_timeout;
2754        self
2755    }
2756
2757    pub fn with_process_liveness(
2758        mut self,
2759        process_liveness: Arc<SupervisorProcessLiveness>,
2760    ) -> Self {
2761        self.process_liveness = process_liveness;
2762        self
2763    }
2764
2765    pub fn with_connection_file_path(mut self, connection_file_path: impl Into<PathBuf>) -> Self {
2766        self.connection_file_path = Some(connection_file_path.into());
2767        self
2768    }
2769
2770    /// Enables daemon-owned capture files for supervised stdout and stderr.
2771    pub fn with_capture_logs_dir(mut self, logs_dir: impl Into<PathBuf>) -> Self {
2772        self.capture_logs_dir = Some(logs_dir.into());
2773        self
2774    }
2775
2776    /// Names this daemon lifetime in spawn events, independently of whether a
2777    /// terminal journal is configured.
2778    pub fn with_daemon_incarnation(self, daemon_incarnation: String) -> Self {
2779        // A millisecond start stamp can repeat after clock rollback or a rapid
2780        // restart. Use the connection file's random daemon_id instead: it already
2781        // identifies this daemon lifetime independently of the wall clock.
2782        self.spawn_events.configure_incarnation(daemon_incarnation);
2783        self
2784    }
2785
2786    /// Enables best-effort history shared by every supervised module. Without
2787    /// it, terminal history is kept only in each module's in-memory ring.
2788    pub fn with_terminal_journal(self, path: PathBuf, daemon_incarnation: String) -> Self {
2789        let mut this = self.with_daemon_incarnation(daemon_incarnation.clone());
2790        this.terminal_journal = Some(Arc::new(crate::terminal_journal::TerminalJournal::open(
2791            path,
2792            daemon_incarnation,
2793        )));
2794        this
2795    }
2796
2797    pub fn with_forwarding(mut self, forwarding: Arc<ForwardingTable>) -> Self {
2798        self.forwarding = Some(forwarding);
2799        self
2800    }
2801
2802    pub fn with_handle(mut self, supervisor_handle: SupervisorHandle) -> Self {
2803        self.spawn_events = supervisor_handle.spawn_events.clone();
2804        self.supervisor_handle = Some(supervisor_handle);
2805        self
2806    }
2807
2808    pub fn with_health_config(mut self, health: HealthConfig) -> Self {
2809        self.health = health;
2810        self
2811    }
2812
2813    /// Keeps a record of every live child at `path`, rewritten on each spawn and
2814    /// reap, for the orphan sweep a later daemon runs at boot. Without it no
2815    /// record is kept.
2816    pub fn with_live_children_record(self, path: impl Into<PathBuf>) -> Self {
2817        self.child_roster.record_to(path.into());
2818        self
2819    }
2820
2821    #[cfg(target_os = "linux")]
2822    pub fn with_cgroup_placement(
2823        mut self,
2824        cgroup_placement: Option<subc_cgroup::Placement>,
2825    ) -> Self {
2826        self.cgroup_placement = cgroup_placement;
2827        self
2828    }
2829
2830    /// Spawn `spec.program` and start monitoring it.
2831    ///
2832    /// The child is expected to parse `--subc <connection-file-path>`, read the
2833    /// TCP+key connection file, authenticate to the already-running listener, and
2834    /// register with channel-0 `HELLO` using `spec.module_id` as its manifest id.
2835    pub fn spawn(&self, spec: ModuleSpec) -> Result<SupervisedModule, SuperviseError> {
2836        validate_spec(&spec)?;
2837        self.establish_identity(&spec);
2838
2839        let runtime = self.runtime_config();
2840        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2841        let spawned = spawn_child(
2842            &spec,
2843            runtime.connection_file_path.as_deref(),
2844            self.supervisor_handle.as_ref(),
2845            &runtime.stderr_ring,
2846            runtime.capture_logs_dir.as_deref(),
2847            &runtime.child_roster,
2848            #[cfg(target_os = "linux")]
2849            runtime.cgroup_placement.as_ref(),
2850        );
2851        #[cfg(test)]
2852        self.test_after_first_spawn.run(&spec.module_id);
2853        let child = match spawned {
2854            Ok(child) => child,
2855            Err(err) => {
2856                // Unlike the configured paths, a failed `spawn` leaves nothing
2857                // on the roster, so the module must not stay marked configured.
2858                self.abandon_unrostered(&spec.module_id);
2859                return Err(err);
2860            }
2861        };
2862        self.mark_first_process_running(&spec, &runtime, &snapshot, &child)?;
2863
2864        Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2865    }
2866
2867    /// Make `spec`'s module count as configured, with its identity gates
2868    /// (reserved id, reserved prefixes) in place, BEFORE any process of it
2869    /// exists.
2870    ///
2871    /// Every path that takes on a new module calls this before `spawn_child`.
2872    /// The order is the point: the child can connect, register, sync its
2873    /// scopes and ask about them as soon as it is spawned, and the module is
2874    /// only put on the roster after `spawn_child` returns. Were the mark set
2875    /// with the roster entry, a fast child would see its own owner reported
2876    /// as not configured, and a scoped `route.open` in that window would be
2877    /// refused as terminal `scope_not_live` ("will never sync") instead of
2878    /// retryable `scope_not_synced`.
2879    fn establish_identity(&self, spec: &ModuleSpec) {
2880        if let Some(supervisor_handle) = &self.supervisor_handle {
2881            supervisor_handle.apply_identity_configuration(spec);
2882            supervisor_handle.mark_configured(&spec.module_id);
2883        }
2884    }
2885
2886    /// Take back [`Self::establish_identity`]'s configured mark when the
2887    /// module will not be put on the roster after all.
2888    fn abandon_unrostered(&self, module_id: &str) {
2889        if let Some(supervisor_handle) = &self.supervisor_handle {
2890            supervisor_handle.unmark_configured_unless_rostered(module_id);
2891        }
2892    }
2893
2894    /// Record a freshly spawned first process as running. On failure the
2895    /// module never reaches the roster, so its configured mark is taken back.
2896    fn mark_first_process_running(
2897        &self,
2898        spec: &ModuleSpec,
2899        runtime: &SupervisorRuntimeConfig,
2900        snapshot: &SharedSnapshot,
2901        child: &SupervisedChild,
2902    ) -> Result<(), SuperviseError> {
2903        if let Err(err) = set_running(snapshot, child, &spec.module_id, &runtime.spawn_events) {
2904            self.abandon_unrostered(&spec.module_id);
2905            return Err(err);
2906        }
2907        self.process_liveness
2908            .track(spec.module_id.clone(), Arc::clone(snapshot));
2909        Ok(())
2910    }
2911
2912    /// Start supervising a module declared in daemon configuration.
2913    ///
2914    /// Unlike [`Self::spawn`], this records disabled modules and immediate spawn
2915    /// failures in the supervisor handle so operator-facing `supervisor.list`
2916    /// reflects every configured module while daemon startup continues.
2917    pub fn supervise_configured(
2918        &self,
2919        spec: ModuleSpec,
2920        enabled: bool,
2921    ) -> Result<SupervisedModule, SuperviseError> {
2922        validate_spec(&spec)?;
2923        self.establish_identity(&spec);
2924
2925        let runtime = self.runtime_config();
2926        if !enabled {
2927            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
2928            return Ok(self.supervised_module(spec, runtime, snapshot, None));
2929        }
2930
2931        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2932        let spawned = spawn_child(
2933            &spec,
2934            runtime.connection_file_path.as_deref(),
2935            self.supervisor_handle.as_ref(),
2936            &runtime.stderr_ring,
2937            runtime.capture_logs_dir.as_deref(),
2938            &runtime.child_roster,
2939            #[cfg(target_os = "linux")]
2940            runtime.cgroup_placement.as_ref(),
2941        );
2942        #[cfg(test)]
2943        self.test_after_first_spawn.run(&spec.module_id);
2944        match spawned {
2945            Ok(child) => {
2946                self.mark_first_process_running(&spec, &runtime, &snapshot, &child)?;
2947                Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2948            }
2949            Err(err) => {
2950                error!(
2951                    module_id = %spec.module_id,
2952                    program = %spec.program.display(),
2953                    error = %err,
2954                    "configured module failed to spawn; marking failed and continuing"
2955                );
2956                let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
2957                Ok(self.supervised_module(spec, runtime, snapshot, None))
2958            }
2959        }
2960    }
2961
2962    /// Supervise a configured module with its own health, drain, and crash
2963    /// budget. The restart policy is per-module because the config file is:
2964    /// `modules.<id>.restart` resolves to a full policy at parse time, and a
2965    /// module that is expensive to restart should not be forced onto the same
2966    /// budget as one that is cheap.
2967    pub fn supervise_configured_with_health(
2968        &self,
2969        spec: ModuleSpec,
2970        enabled: bool,
2971        health: HealthConfig,
2972        drain_timeout_ms: Option<u64>,
2973        restart_policy: RestartPolicy,
2974    ) -> Result<SupervisedModule, SuperviseError> {
2975        validate_spec(&spec)?;
2976        self.establish_identity(&spec);
2977
2978        let mut runtime = self.runtime_config();
2979        runtime.health = health.clone();
2980        runtime.restart_policy = restart_policy;
2981        if let Some(ms) = drain_timeout_ms {
2982            runtime.drain_timeout = Duration::from_millis(ms);
2983            *runtime
2984                .effective_drain_timeout
2985                .lock()
2986                .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
2987        }
2988        if !enabled {
2989            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
2990            return Ok(self.supervised_module(spec, runtime, snapshot, None));
2991        }
2992
2993        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2994        let spawned = spawn_child(
2995            &spec,
2996            runtime.connection_file_path.as_deref(),
2997            self.supervisor_handle.as_ref(),
2998            &runtime.stderr_ring,
2999            runtime.capture_logs_dir.as_deref(),
3000            &runtime.child_roster,
3001            #[cfg(target_os = "linux")]
3002            runtime.cgroup_placement.as_ref(),
3003        );
3004        #[cfg(test)]
3005        self.test_after_first_spawn.run(&spec.module_id);
3006        match spawned {
3007            Ok(child) => {
3008                self.mark_first_process_running(&spec, &runtime, &snapshot, &child)?;
3009                Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
3010            }
3011            Err(err) => {
3012                if health.critical {
3013                    error!(
3014                        module_id = %spec.module_id,
3015                        program = %spec.program.display(),
3016                        error = %err,
3017                        "critical configured module failed to spawn; marking failed and alerting"
3018                    );
3019                } else {
3020                    error!(
3021                        module_id = %spec.module_id,
3022                        program = %spec.program.display(),
3023                        error = %err,
3024                        "configured module failed to spawn; marking failed and continuing"
3025                    );
3026                }
3027                let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
3028                Ok(self.supervised_module(spec, runtime, snapshot, None))
3029            }
3030        }
3031    }
3032
3033    fn runtime_config(&self) -> SupervisorRuntimeConfig {
3034        let effective_drain_timeout = Arc::new(Mutex::new(self.drain_timeout));
3035        SupervisorRuntimeConfig {
3036            scheduled_respawn: Arc::default(),
3037            deferred_reload_reply: Arc::default(),
3038            restart_policy: self.restart_policy,
3039            drain_timeout: self.drain_timeout,
3040            // Shared with this module's roster copy: daemon shutdown waits on
3041            // each child for the module's own drain budget, as resolved now.
3042            child_roster: self
3043                .child_roster
3044                .for_module(Arc::clone(&effective_drain_timeout)),
3045            effective_drain_timeout,
3046            default_drain_timeout: self.drain_timeout,
3047            health: self.health.clone(),
3048            connection_file_path: self.connection_file_path.clone(),
3049            capture_logs_dir: self.capture_logs_dir.clone(),
3050            forwarding: self.forwarding.clone(),
3051            supervisor_handle: self.supervisor_handle.clone(),
3052            stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
3053            terminal_ring: Arc::new(Mutex::new(
3054                TerminalRing::new(
3055                    TerminalRingConfig::default(),
3056                    self.daemon_start_clock.started_at_ms(),
3057                )
3058                .with_start_clock(self.daemon_start_clock)
3059                .with_journal(self.terminal_journal.clone())
3060                .with_daemon_shutdown(self.child_roster.shutdown_flag()),
3061            )),
3062            spawn_events: self.spawn_events.clone(),
3063            #[cfg(target_os = "linux")]
3064            cgroup_placement: self.cgroup_placement.clone(),
3065            #[cfg(test)]
3066            test_seed_stale_facts_before_enable_spawn: false,
3067            #[cfg(test)]
3068            test_reload_exit_record_gate: None,
3069        }
3070    }
3071
3072    fn supervised_module(
3073        &self,
3074        spec: ModuleSpec,
3075        runtime: SupervisorRuntimeConfig,
3076        snapshot: SharedSnapshot,
3077        child: Option<SupervisedChild>,
3078    ) -> SupervisedModule {
3079        let configuration = Arc::new(Mutex::new(SupervisedConfiguration {
3080            spec: spec.clone(),
3081            health: runtime.health.clone(),
3082        }));
3083        let stderr_ring = Arc::clone(&runtime.stderr_ring);
3084        let terminal_ring = Arc::clone(&runtime.terminal_ring);
3085        // The module's OWN policy, which may be its per-module config rather than
3086        // the supervisor-wide one; status must report the budget the supervise
3087        // loop actually enforces.
3088        let restart_policy = runtime.restart_policy;
3089        let effective_drain_timeout = Arc::clone(&runtime.effective_drain_timeout);
3090        let (tx, rx) = mpsc::channel(4);
3091        let monitor = tokio::spawn(supervise_loop(
3092            spec.clone(),
3093            runtime,
3094            Arc::clone(&self.registry),
3095            Arc::clone(&self.process_liveness),
3096            Arc::clone(&snapshot),
3097            child,
3098            rx,
3099        ));
3100
3101        let module_id = spec.module_id.clone();
3102        let module = SupervisedModule {
3103            inner: Arc::new(SupervisedModuleInner {
3104                module_id: module_id.clone(),
3105                registry: Arc::clone(&self.registry),
3106                snapshot,
3107                configuration,
3108                stderr_ring,
3109                terminal_ring,
3110                commands: tx,
3111                monitor: Mutex::new(Some(monitor)),
3112                restart_policy,
3113                effective_drain_timeout,
3114                provenance_probe: self.provenance_probe.clone(),
3115            }),
3116        };
3117        // The identity gates and the configured mark were set by
3118        // `establish_identity` before any process was spawned; only the roster
3119        // entry waits for the module handle, which needs the spawned child.
3120        if let Some(supervisor_handle) = &self.supervisor_handle {
3121            supervisor_handle.insert(module.clone());
3122        }
3123        module
3124    }
3125}
3126
3127impl Default for Supervisor {
3128    fn default() -> Self {
3129        Self::new(Arc::new(Registry::default()), RestartPolicy::default())
3130    }
3131}
3132
3133/// Handle to one supervised child process.
3134#[derive(Clone)]
3135pub struct SupervisedModule {
3136    inner: Arc<SupervisedModuleInner>,
3137}
3138
3139struct SupervisedModuleInner {
3140    module_id: String,
3141    registry: Arc<Registry>,
3142    snapshot: SharedSnapshot,
3143    configuration: Arc<Mutex<SupervisedConfiguration>>,
3144    stderr_ring: Arc<Mutex<StderrRing>>,
3145    terminal_ring: Arc<Mutex<TerminalRing>>,
3146    commands: mpsc::Sender<SupervisorCommand>,
3147    monitor: Mutex<Option<JoinHandle<()>>>,
3148    /// Copied from the supervisor's runtime config at spawn so `status()` can
3149    /// report the restart budget without reaching back into the supervisor. The
3150    /// policy is fixed for the process's lifetime, so a copy cannot drift.
3151    restart_policy: RestartPolicy,
3152    effective_drain_timeout: Arc<Mutex<Duration>>,
3153    provenance_probe: ExecutableIdentityProbe,
3154}
3155
3156impl fmt::Debug for SupervisedModule {
3157    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3158        f.debug_struct("SupervisedModule")
3159            .field("module_id", &self.inner.module_id)
3160            .field("status", &self.status())
3161            .finish_non_exhaustive()
3162    }
3163}
3164
3165impl SupervisedModule {
3166    pub fn module_id(&self) -> &str {
3167        &self.inner.module_id
3168    }
3169
3170    /// Test-only: put one probe miss on the streak, the way
3171    /// `handle_health_probe_failure` does, so tests can assert what a later
3172    /// event does to the streak without driving the whole probe loop.
3173    #[cfg(test)]
3174    pub(crate) fn record_health_probe_failure_for_test(
3175        &self,
3176        detail: &str,
3177    ) -> Result<(), SuperviseError> {
3178        update_snapshot(&self.inner.snapshot, Some(&self.inner.module_id), |state| {
3179            state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
3180            state.health.detail = Some(detail.to_string());
3181        })
3182    }
3183
3184    pub fn state(&self) -> Result<ModuleState, SuperviseError> {
3185        Ok(lock_snapshot(&self.inner.snapshot)?.state)
3186    }
3187
3188    /// The module's retained stderr, newest lines last.
3189    ///
3190    /// Deliberately NOT on [`Self::status`]: a bounded tail is kilobytes per
3191    /// module, `supervisor.list` renders every module, and putting it in the
3192    /// shared snapshot would make each status read carry a payload almost nobody
3193    /// asked for. Callers that want the text ask for it.
3194    pub fn stderr_tail(
3195        &self,
3196        max_lines: Option<usize>,
3197        max_bytes: Option<usize>,
3198    ) -> StderrTailSnapshot {
3199        self.inner
3200            .stderr_ring
3201            .lock()
3202            .unwrap_or_else(|poisoned| poisoned.into_inner())
3203            .snapshot(max_lines, max_bytes)
3204    }
3205
3206    /// The module's bounded terminal history, oldest retained exit first.
3207    ///
3208    /// The daemon-start stamp distinguishes a quiet supervisor from a replacement
3209    /// daemon whose in-memory history was necessarily reset.
3210    pub fn terminal_history(&self) -> TerminalHistorySnapshot {
3211        self.inner
3212            .terminal_ring
3213            .lock()
3214            .unwrap_or_else(|poisoned| poisoned.into_inner())
3215            .snapshot()
3216    }
3217
3218    /// Retained observations from the current ring and all journal generations.
3219    ///
3220    /// Blocking: this reads the journal files. Async callers use
3221    /// [`Self::read_durable_terminal_history`].
3222    pub fn durable_terminal_history(&self) -> subc_control::TerminalHistory {
3223        durable_terminal_history_of(&self.inner.terminal_ring, &self.inner.module_id)
3224    }
3225
3226    /// [`Self::durable_terminal_history`] on a blocking thread, so the journal
3227    /// read (up to every retained generation) never occupies a runtime worker.
3228    /// Fails only if the blocking task could not finish (runtime shutdown or a
3229    /// panic in the read).
3230    pub(crate) async fn read_durable_terminal_history(
3231        &self,
3232    ) -> Result<subc_control::TerminalHistory, tokio::task::JoinError> {
3233        let terminal_ring = Arc::clone(&self.inner.terminal_ring);
3234        let module_id = self.inner.module_id.clone();
3235        tokio::task::spawn_blocking(move || durable_terminal_history_of(&terminal_ring, &module_id))
3236            .await
3237    }
3238
3239    pub fn status(&self) -> Result<ModuleStatus, SuperviseError> {
3240        self.status_with_snapshot_lock(&self.inner.snapshot, None)
3241            .map(|(status, _)| status)
3242    }
3243
3244    pub(crate) fn record_deliberate_severance(
3245        &self,
3246        identity: ProcessIdentity,
3247    ) -> Result<bool, SuperviseError> {
3248        let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
3249        if snapshot.pid != Some(identity.pid)
3250            || snapshot.process_start_time != Some(identity.start_time)
3251        {
3252            return Ok(false);
3253        }
3254        snapshot.deliberate_severance = Some(identity);
3255        Ok(true)
3256    }
3257
3258    /// Read status for a channel-0 renderer and report a contended snapshot lock.
3259    ///
3260    /// Internal supervision callers use [`Self::status`] so writer-side machinery
3261    /// does not produce reader-observability logs.
3262    pub(crate) fn status_for_control(
3263        &self,
3264        caller: &'static str,
3265    ) -> Result<ModuleStatus, SuperviseError> {
3266        self.status_with_snapshot_lock(&self.inner.snapshot, Some(caller))
3267            .map(|(status, _)| status)
3268    }
3269
3270    fn status_with_snapshot_lock(
3271        &self,
3272        snapshot: &SharedSnapshot,
3273        caller: Option<&'static str>,
3274    ) -> Result<(ModuleStatus, Option<SpawnedFileIdentity>), SuperviseError> {
3275        let mut guard = match caller {
3276            Some(caller) => lock_snapshot_for_control(snapshot, &self.inner.module_id, caller)?,
3277            None => lock_snapshot(snapshot)?,
3278        };
3279        // Read the budget through the pruning path so a reader sees the same
3280        // in-window count the restart decision would use, not a stale total.
3281        let restart_count =
3282            guard.crash_restarts_in_window(self.inner.restart_policy.window, Instant::now());
3283        let snapshot = guard.clone();
3284        drop(guard);
3285        let drain_timeout = *self.inner.effective_drain_timeout.lock().map_err(|_| {
3286            SuperviseError::StatePoisoned {
3287                module_id: Some(self.inner.module_id.clone()),
3288            }
3289        })?;
3290        let registration_active = self
3291            .inner
3292            .registry
3293            .get_module(&self.inner.module_id)
3294            .map_err(SuperviseError::Registry)?
3295            .is_some();
3296        let protocol = snapshot
3297            .spawned_protocol
3298            .unwrap_or(self.declared_protocol()?);
3299        let running_process =
3300            snapshot.enabled && snapshot.state == ModuleState::Running && snapshot.process_alive;
3301        // Registration is the difference between the two protocols and the only
3302        // one: a subc module that has not registered cannot serve a request even
3303        // though its process is up, and a `none` module never registers at all,
3304        // so requiring it there would pin `live` to false for the whole life of
3305        // a perfectly healthy process.
3306        let live = match protocol {
3307            ModuleProtocol::Subc => running_process && registration_active,
3308            ModuleProtocol::None => running_process,
3309        };
3310
3311        Ok((
3312            ModuleStatus {
3313                module_id: self.inner.module_id.clone(),
3314                state: snapshot.state,
3315                enabled: snapshot.enabled,
3316                process_alive: snapshot.process_alive,
3317                registration_active,
3318                protocol,
3319                live,
3320                restart_count,
3321                lifetime_restarts: snapshot.lifetime_restarts,
3322                spawn_generation: snapshot.spawn_generation,
3323                max_restarts: self.inner.restart_policy.max_restarts,
3324                restart_window: self.inner.restart_policy.window,
3325                drain_timeout,
3326                restart_backoff: self.inner.restart_policy.backoff,
3327                restart_max_backoff: self.inner.restart_policy.max_backoff,
3328                pid: snapshot.reported_pid(),
3329                spawned_at_ms: snapshot.spawned_at_ms,
3330                spawned_from: snapshot.spawned_from,
3331                process_start_time: snapshot.process_start_time,
3332                last_exit: snapshot.last_exit,
3333                health: snapshot.health,
3334            },
3335            snapshot.spawned_file_identity,
3336        ))
3337    }
3338
3339    #[cfg(test)]
3340    pub(crate) fn hold_snapshot_for_test(
3341        &self,
3342        acquired: std::sync::mpsc::Sender<()>,
3343        hold: Duration,
3344    ) -> std::thread::JoinHandle<()> {
3345        let snapshot = Arc::clone(&self.inner.snapshot);
3346        std::thread::spawn(move || {
3347            let _guard = snapshot.lock().expect("test snapshot lock is not poisoned");
3348            acquired
3349                .send(())
3350                .expect("test receiver waits for snapshot lock");
3351            std::thread::sleep(hold);
3352        })
3353    }
3354
3355    /// The status and the running-image check for `supervisor.provenance`,
3356    /// taken from one status read. The exec acknowledgement can land between
3357    /// two separate reads, and the reply would then pair "no pid yet" with an
3358    /// image observed after the module started, which describes no single
3359    /// moment.
3360    pub(crate) async fn status_and_running_image_agreement(
3361        &self,
3362    ) -> Result<(ModuleStatus, subc_control::RunningImageAgreement), SuperviseError> {
3363        let (status, identity) = self.status_with_snapshot_lock(&self.inner.snapshot, None)?;
3364        let image = self
3365            .inner
3366            .provenance_probe
3367            .observe(
3368                status.pid,
3369                status.spawned_from.as_deref(),
3370                identity,
3371                status.process_start_time,
3372            )
3373            .await;
3374        Ok((status, image))
3375    }
3376
3377    pub(crate) async fn running_image_agreement(&self) -> subc_control::RunningImageAgreement {
3378        let snapshot = match lock_snapshot(&self.inner.snapshot) {
3379            Ok(snapshot) => snapshot.clone(),
3380            Err(_) => {
3381                return subc_control::RunningImageAgreement::Unavailable {
3382                    reason: subc_control::RunningImageUnavailableReason::NotRunning,
3383                };
3384            }
3385        };
3386        self.inner
3387            .provenance_probe
3388            .observe(
3389                snapshot.reported_pid(),
3390                snapshot.spawned_from.as_deref(),
3391                snapshot.spawned_file_identity,
3392                snapshot.process_start_time,
3393            )
3394            .await
3395    }
3396
3397    /// Memory and CPU time of the module's current process, read now. Only the
3398    /// process the supervisor spawned is read, not processes it has started.
3399    pub(crate) fn child_resource_usage(&self) -> subc_control::ChildResourceUsage {
3400        let (pid, start_time) = match lock_snapshot(&self.inner.snapshot) {
3401            Ok(snapshot) => (snapshot.reported_pid(), snapshot.process_start_time),
3402            Err(_) => {
3403                return subc_control::ChildResourceUsage::Unavailable {
3404                    reason: subc_control::ChildResourceUnavailableReason::Unreadable,
3405                }
3406            }
3407        };
3408        crate::child_resources::read(pid, start_time)
3409    }
3410
3411    pub(crate) fn will_recover_after_connection_loss(&self) -> Result<bool, SuperviseError> {
3412        let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
3413        Ok(match snapshot.state {
3414            ModuleState::Restarting => true,
3415            ModuleState::Failed | ModuleState::Disabled => false,
3416            _ => daemon_will_restart(&mut snapshot, &self.inner.restart_policy, Instant::now()),
3417        })
3418    }
3419
3420    #[cfg(test)]
3421    pub(crate) fn is_warming(&self) -> Result<bool, SuperviseError> {
3422        self.is_warming_with_snapshot_lock(None)
3423    }
3424
3425    pub(crate) fn is_warming_for_control(
3426        &self,
3427        caller: &'static str,
3428    ) -> Result<bool, SuperviseError> {
3429        self.is_warming_with_snapshot_lock(Some(caller))
3430    }
3431
3432    fn is_warming_with_snapshot_lock(
3433        &self,
3434        caller: Option<&'static str>,
3435    ) -> Result<bool, SuperviseError> {
3436        let snapshot = match caller {
3437            Some(caller) => {
3438                lock_snapshot_for_control(&self.inner.snapshot, &self.inner.module_id, caller)?
3439            }
3440            None => lock_snapshot(&self.inner.snapshot)?,
3441        }
3442        .clone();
3443        Ok(matches!(
3444            snapshot.state,
3445            ModuleState::Starting | ModuleState::Running | ModuleState::Restarting
3446        ))
3447    }
3448
3449    /// Drain the module and stop monitoring it.
3450    pub async fn drain(&self) -> Result<(), SuperviseError> {
3451        self.stop().await
3452    }
3453
3454    pub(crate) async fn retire(&self) -> Result<(), SuperviseError> {
3455        match self.state()? {
3456            ModuleState::Stopped | ModuleState::Failed => return Ok(()),
3457            ModuleState::Starting
3458            | ModuleState::Running
3459            | ModuleState::Unresponsive
3460            | ModuleState::Restarting
3461            | ModuleState::Draining
3462            | ModuleState::Disabled => {}
3463        }
3464
3465        let (reply_tx, reply_rx) = oneshot::channel();
3466        self.inner
3467            .commands
3468            .send(SupervisorCommand::Retire { reply: reply_tx })
3469            .await
3470            .map_err(|_| SuperviseError::CommandClosed {
3471                module_id: self.inner.module_id.clone(),
3472            })?;
3473        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3474            module_id: self.inner.module_id.clone(),
3475        })?
3476    }
3477
3478    pub async fn stop(&self) -> Result<(), SuperviseError> {
3479        match self.state()? {
3480            ModuleState::Stopped | ModuleState::Failed => return Ok(()),
3481            ModuleState::Starting
3482            | ModuleState::Running
3483            | ModuleState::Unresponsive
3484            | ModuleState::Restarting
3485            | ModuleState::Draining
3486            | ModuleState::Disabled => {}
3487        }
3488
3489        let (reply_tx, reply_rx) = oneshot::channel();
3490        self.inner
3491            .commands
3492            .send(SupervisorCommand::Drain { reply: reply_tx })
3493            .await
3494            .map_err(|_| SuperviseError::CommandClosed {
3495                module_id: self.inner.module_id.clone(),
3496            })?;
3497        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3498            module_id: self.inner.module_id.clone(),
3499        })?
3500    }
3501
3502    pub async fn restart(&self, drain_timeout_ms: Option<u64>) -> Result<(), SuperviseError> {
3503        let received_at_generation = lock_snapshot(&self.inner.snapshot)?.spawn_generation;
3504        let (reply_tx, reply_rx) = oneshot::channel();
3505        self.inner
3506            .commands
3507            .send(SupervisorCommand::Restart {
3508                drain_timeout_ms,
3509                received_at_generation,
3510                queued_at: Instant::now(),
3511                reply: reply_tx,
3512            })
3513            .await
3514            .map_err(|_| SuperviseError::CommandClosed {
3515                module_id: self.inner.module_id.clone(),
3516            })?;
3517        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3518            module_id: self.inner.module_id.clone(),
3519        })?
3520    }
3521
3522    /// Blue/green restart: see [`SupervisorCommand::Swap`] and the
3523    /// `supervisor_swap` module. Returns once the swap has cut over (the old
3524    /// process then drains in the background of the supervise loop) or has
3525    /// failed, leaving the old process serving.
3526    pub async fn swap(&self, ready_timeout: Option<Duration>) -> Result<(), SuperviseError> {
3527        let (reply_tx, reply_rx) = oneshot::channel();
3528        self.inner
3529            .commands
3530            .send(SupervisorCommand::Swap {
3531                ready_timeout,
3532                reply: reply_tx,
3533            })
3534            .await
3535            .map_err(|_| SuperviseError::CommandClosed {
3536                module_id: self.inner.module_id.clone(),
3537            })?;
3538        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3539            module_id: self.inner.module_id.clone(),
3540        })?
3541    }
3542
3543    pub async fn reload(&self) -> Result<(), SuperviseError> {
3544        let (reply_tx, reply_rx) = oneshot::channel();
3545        self.inner
3546            .commands
3547            .send(SupervisorCommand::Reload { reply: reply_tx })
3548            .await
3549            .map_err(|_| SuperviseError::CommandClosed {
3550                module_id: self.inner.module_id.clone(),
3551            })?;
3552        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3553            module_id: self.inner.module_id.clone(),
3554        })?
3555    }
3556
3557    pub async fn set_enabled(&self, enabled: bool) -> Result<bool, SuperviseError> {
3558        let (reply_tx, reply_rx) = oneshot::channel();
3559        self.inner
3560            .commands
3561            .send(SupervisorCommand::SetEnabled {
3562                enabled,
3563                reply: reply_tx,
3564            })
3565            .await
3566            .map_err(|_| SuperviseError::CommandClosed {
3567                module_id: self.inner.module_id.clone(),
3568            })?;
3569        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3570            module_id: self.inner.module_id.clone(),
3571        })?
3572    }
3573
3574    /// The current process's protocol, or the configured protocol when down.
3575    /// A rescan stores the next launch spec without changing how an existing
3576    /// process registers, serves routes, is probed, or exits.
3577    pub(crate) fn declared_protocol(&self) -> Result<ModuleProtocol, SuperviseError> {
3578        let configured = self
3579            .inner
3580            .configuration
3581            .lock()
3582            .map_err(|_| SuperviseError::StatePoisoned {
3583                module_id: Some(self.inner.module_id.clone()),
3584            })?
3585            .spec
3586            .protocol;
3587        let state = lock_snapshot(&self.inner.snapshot)?;
3588        Ok(state.spawned_protocol.unwrap_or(configured))
3589    }
3590
3591    pub(crate) fn configuration(&self) -> Result<(ModuleSpec, HealthConfig), SuperviseError> {
3592        let configuration =
3593            self.inner
3594                .configuration
3595                .lock()
3596                .map_err(|_| SuperviseError::StatePoisoned {
3597                    module_id: Some(self.inner.module_id.clone()),
3598                })?;
3599        Ok((configuration.spec.clone(), configuration.health.clone()))
3600    }
3601
3602    /// Replace this module's launch spec, keeping its health and drain policy,
3603    /// the way a rescan does for a changed config entry. The running process is
3604    /// untouched; the next spawn (a restart, or a swap's candidate) uses it.
3605    #[cfg(any(test, feature = "test-support"))]
3606    pub async fn update_spec_for_test(&self, spec: ModuleSpec) -> Result<(), SuperviseError> {
3607        let (_, health) = self.configuration()?;
3608        let drain_timeout_ms = u64::try_from(
3609            self.inner
3610                .effective_drain_timeout
3611                .lock()
3612                .unwrap_or_else(|poisoned| poisoned.into_inner())
3613                .as_millis(),
3614        )
3615        .ok();
3616        self.update_configuration(spec, health, drain_timeout_ms)
3617            .await
3618    }
3619
3620    pub(crate) async fn update_configuration(
3621        &self,
3622        spec: ModuleSpec,
3623        health: HealthConfig,
3624        drain_timeout_ms: Option<u64>,
3625    ) -> Result<(), SuperviseError> {
3626        if spec.module_id != self.inner.module_id {
3627            return Err(SuperviseError::InvalidSpec {
3628                reason: "a supervised module's module_id cannot be changed".to_string(),
3629            });
3630        }
3631        validate_spec(&spec)?;
3632        let (reply_tx, reply_rx) = oneshot::channel();
3633        self.inner
3634            .commands
3635            .send(SupervisorCommand::UpdateConfiguration {
3636                spec: spec.clone(),
3637                health: health.clone(),
3638                drain_timeout_ms,
3639                reply: reply_tx,
3640            })
3641            .await
3642            .map_err(|_| SuperviseError::CommandClosed {
3643                module_id: self.inner.module_id.clone(),
3644            })?;
3645        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3646            module_id: self.inner.module_id.clone(),
3647        })?;
3648        let mut configuration =
3649            self.inner
3650                .configuration
3651                .lock()
3652                .map_err(|_| SuperviseError::StatePoisoned {
3653                    module_id: Some(self.inner.module_id.clone()),
3654                })?;
3655        configuration.spec = spec;
3656        configuration.health = health;
3657        Ok(())
3658    }
3659}
3660
3661impl Drop for SupervisedModuleInner {
3662    fn drop(&mut self) {
3663        let Ok(mut monitor) = self.monitor.lock() else {
3664            return;
3665        };
3666        if let Some(monitor) = monitor.as_ref().filter(|monitor| !monitor.is_finished()) {
3667            let _ = update_snapshot(&self.snapshot, Some(&self.module_id), |state| {
3668                state.state = ModuleState::Stopped;
3669                clear_current_process_facts(state);
3670            });
3671            monitor.abort();
3672        }
3673        let _ = monitor.take();
3674    }
3675}
3676
3677#[derive(Debug)]
3678enum SupervisorCommand {
3679    Drain {
3680        reply: oneshot::Sender<Result<(), SuperviseError>>,
3681    },
3682    Retire {
3683        reply: oneshot::Sender<Result<(), SuperviseError>>,
3684    },
3685    Restart {
3686        /// Operator override for this one restart's drain budget, in ms. `None`
3687        /// uses the module's configured/default budget; `Some(0)` cuts
3688        /// immediately (wedge bounce: a stuck request never settles, so
3689        /// waiting only delays recovery).
3690        drain_timeout_ms: Option<u64>,
3691        /// The module's `spawn_generation` when the request was received, before
3692        /// it waited in the command queue. A queued restart whose module has
3693        /// since spawned a newer process is already satisfied (see the handler).
3694        received_at_generation: u64,
3695        /// When the request entered the command queue, so the handler can log
3696        /// how long it waited behind the loop's other work.
3697        queued_at: Instant,
3698        reply: oneshot::Sender<Result<(), SuperviseError>>,
3699    },
3700    Reload {
3701        reply: oneshot::Sender<Result<(), SuperviseError>>,
3702    },
3703    SetEnabled {
3704        enabled: bool,
3705        reply: oneshot::Sender<Result<bool, SuperviseError>>,
3706    },
3707    UpdateConfiguration {
3708        spec: ModuleSpec,
3709        health: HealthConfig,
3710        /// Per-module drain override from the new config; `None` re-resolves to
3711        /// the supervisor-wide default.
3712        drain_timeout_ms: Option<u64>,
3713        reply: oneshot::Sender<()>,
3714    },
3715    Swap {
3716        /// How long the candidate may take to register and declare itself
3717        /// ready. `None` uses [`DEFAULT_SWAP_READY_TIMEOUT`].
3718        ready_timeout: Option<Duration>,
3719        /// Answered at cutover or failure; the incumbent's drain follows.
3720        reply: oneshot::Sender<Result<(), SuperviseError>>,
3721    },
3722}
3723
3724#[derive(Debug)]
3725pub enum SuperviseError {
3726    InvalidSpec {
3727        reason: String,
3728    },
3729    Spawn {
3730        program: PathBuf,
3731        source: io::Error,
3732        cgroup_path: Option<PathBuf>,
3733    },
3734    Cgroup {
3735        module_id: String,
3736        source: io::Error,
3737    },
3738    /// CSPRNG failure generating a reserved module's launch nonce. Fail loud rather
3739    /// than spawn a reserved module without its identity binding.
3740    LaunchNonce {
3741        reason: String,
3742    },
3743    Wait {
3744        module_id: String,
3745        source: io::Error,
3746    },
3747    Kill {
3748        module_id: String,
3749        source: io::Error,
3750    },
3751    Forwarding(ForwardingError),
3752    Registry(RegistryError),
3753    ReloadUnavailable {
3754        module_id: String,
3755        reason: String,
3756    },
3757    /// An operator restart/reload was requested for a module that is currently
3758    /// disabled. Restart/reload cycle a *running* module; a disabled module must
3759    /// be explicitly re-enabled (set_enabled(true)) rather than silently started
3760    /// by a restart, so these commands are rejected instead of re-enabling it.
3761    Disabled {
3762        module_id: String,
3763    },
3764    ReloadFailed {
3765        module_id: String,
3766        reason: String,
3767    },
3768    RegistrationStillActive {
3769        module_id: String,
3770        waited: Duration,
3771    },
3772    StatePoisoned {
3773        module_id: Option<String>,
3774    },
3775    CommandClosed {
3776        module_id: String,
3777    },
3778    /// A restart or reload arrived while a swap's candidate was warming. The
3779    /// swap owns the module until it cuts over or fails; a stop or disable
3780    /// would have aborted it instead.
3781    SwapInProgress {
3782        module_id: String,
3783    },
3784    /// A swap was refused before anything was spawned.
3785    SwapRefused {
3786        module_id: String,
3787        reason: SwapRefusal,
3788    },
3789    /// A swap spawned a candidate and gave up on it. The candidate has been
3790    /// killed and its slot freed; the incumbent was left serving and was never
3791    /// drained, except in the one `CutoverLost` case described on that arm.
3792    SwapFailed {
3793        module_id: String,
3794        arm: SwapFailureArm,
3795        detail: String,
3796        /// How the candidate exited, when it exited on its own before the
3797        /// supervisor gave up on it.
3798        candidate_exit: Option<ExitReport>,
3799    },
3800}
3801
3802/// Why a swap was refused before a candidate was spawned.
3803#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3804pub enum SwapRefusal {
3805    /// The module's config does not declare `overlap: "safe"`.
3806    OverlapExclusive,
3807    /// The module is not registered, so there is no incumbent to keep serving
3808    /// and nothing a swap would improve on; a plain restart is the tool.
3809    NotRegistered,
3810    /// The module does not speak the subc wire, so a candidate could never
3811    /// register or declare itself ready.
3812    ProtocolNone,
3813    /// The supervisor lacks the forwarding table (to cut routes over) or the
3814    /// shared handle (to admit the candidate's HELLO) that a swap needs.
3815    NotConfigured,
3816    /// A swap is already open for this module.
3817    AlreadySwapping,
3818}
3819
3820impl SwapRefusal {
3821    pub fn as_str(self) -> &'static str {
3822        match self {
3823            Self::OverlapExclusive => "overlap_exclusive",
3824            Self::NotRegistered => "not_registered",
3825            Self::ProtocolNone => "protocol_none",
3826            Self::NotConfigured => "not_configured",
3827            Self::AlreadySwapping => "already_swapping",
3828        }
3829    }
3830}
3831
3832/// Which failure arm ended a swap. Every arm but one leaves the incumbent
3833/// serving and undrained; see `CutoverLost`.
3834#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3835pub enum SwapFailureArm {
3836    /// The candidate process could not be started.
3837    SpawnFailed,
3838    /// The candidate did not register within the readiness budget.
3839    NeverRegistered,
3840    /// The candidate registered but did not declare itself ready in time.
3841    NeverReady,
3842    /// The candidate exited before cutover.
3843    CandidateExited,
3844    /// The candidate declared itself ready but failed its health probe.
3845    CandidateUnhealthy,
3846    /// An operator stop, disable or retire arrived while the candidate warmed.
3847    /// The candidate was killed and the operator's command then carried out on
3848    /// the incumbent.
3849    Interrupted,
3850    /// The candidate's connection closed at the moment of cutover. If it
3851    /// closed before forwarding moved, the incumbent is untouched. If it closed
3852    /// between the forwarding and registry halves of cutover, forwarding can no
3853    /// longer route to the incumbent, so the module is restarted plainly.
3854    CutoverLost,
3855}
3856
3857impl SwapFailureArm {
3858    pub fn as_str(self) -> &'static str {
3859        match self {
3860            Self::SpawnFailed => "spawn_failed",
3861            Self::NeverRegistered => "never_registered",
3862            Self::NeverReady => "never_ready",
3863            Self::CandidateExited => "candidate_exited",
3864            Self::CandidateUnhealthy => "candidate_unhealthy",
3865            Self::Interrupted => "interrupted",
3866            Self::CutoverLost => "cutover_lost",
3867        }
3868    }
3869}
3870
3871impl fmt::Display for SuperviseError {
3872    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3873        match self {
3874            Self::InvalidSpec { reason } => write!(f, "invalid module spec: {reason}"),
3875            Self::Spawn {
3876                program,
3877                source,
3878                cgroup_path: Some(cgroup_path),
3879            } => write!(
3880                f,
3881                "failed to place module in cgroup '{}' while spawning '{}': {source}",
3882                cgroup_path.display(),
3883                program.display()
3884            ),
3885            Self::Spawn {
3886                program,
3887                source,
3888                cgroup_path: None,
3889            } => write!(
3890                f,
3891                "failed to spawn module '{}': {source}",
3892                program.display()
3893            ),
3894            Self::Cgroup { module_id, source } => {
3895                write!(
3896                    f,
3897                    "failed to prepare cgroup for module '{module_id}': {source}"
3898                )
3899            }
3900            Self::LaunchNonce { reason } => {
3901                write!(
3902                    f,
3903                    "failed to generate reserved-module launch nonce: {reason}"
3904                )
3905            }
3906            Self::Wait { module_id, source } => {
3907                write!(f, "failed to wait for module '{module_id}': {source}")
3908            }
3909            Self::Kill { module_id, source } => {
3910                write!(f, "failed to kill module '{module_id}': {source}")
3911            }
3912            Self::Forwarding(err) => write!(f, "forwarding error: {err}"),
3913            Self::Registry(err) => write!(f, "registry error: {err}"),
3914            Self::ReloadUnavailable { module_id, reason } => {
3915                write!(f, "reload unavailable for module '{module_id}': {reason}")
3916            }
3917            Self::Disabled { module_id } => {
3918                write!(
3919                    f,
3920                    "module '{module_id}' is disabled; enable it before restart or reload"
3921                )
3922            }
3923            Self::ReloadFailed { module_id, reason } => {
3924                write!(f, "reload failed for module '{module_id}': {reason}")
3925            }
3926            Self::RegistrationStillActive { module_id, waited } => write!(
3927                f,
3928                "module '{module_id}' registration remained active after waiting {waited:?}"
3929            ),
3930            Self::StatePoisoned { module_id } => match module_id {
3931                Some(module_id) => {
3932                    write!(f, "supervisor state for module '{module_id}' was poisoned")
3933                }
3934                None => write!(f, "supervisor state was poisoned"),
3935            },
3936            Self::CommandClosed { module_id } => {
3937                write!(
3938                    f,
3939                    "supervisor command channel for module '{module_id}' is closed"
3940                )
3941            }
3942            Self::SwapInProgress { module_id } => write!(
3943                f,
3944                "module '{module_id}' is being swapped; retry once the swap has cut over or failed, or stop the module to abort the swap"
3945            ),
3946            Self::SwapRefused { module_id, reason } => match reason {
3947                SwapRefusal::OverlapExclusive => write!(
3948                    f,
3949                    "module '{module_id}' is declared overlap: \"exclusive\" (the default): two processes of it must not run at once, so it cannot be swapped; use a plain restart, or declare overlap: \"safe\" in its config if it really tolerates a second process"
3950                ),
3951                SwapRefusal::NotRegistered => write!(
3952                    f,
3953                    "module '{module_id}' is not registered, so there is no serving process to keep while a replacement warms; use a plain restart"
3954                ),
3955                SwapRefusal::ProtocolNone => write!(
3956                    f,
3957                    "module '{module_id}' is protocol: \"none\" and never registers, so a swap could never see its replacement become ready; use a plain restart"
3958                ),
3959                SwapRefusal::NotConfigured => write!(
3960                    f,
3961                    "module '{module_id}' cannot be swapped: the supervisor was built without the forwarding table or shared handle a swap needs"
3962                ),
3963                SwapRefusal::AlreadySwapping => {
3964                    write!(f, "module '{module_id}' is already being swapped")
3965                }
3966            },
3967            Self::SwapFailed {
3968                module_id,
3969                arm,
3970                detail,
3971                ..
3972            } => write!(
3973                f,
3974                "swap of module '{module_id}' failed ({}): {detail}; the running process was left serving",
3975                arm.as_str()
3976            ),
3977        }
3978    }
3979}
3980
3981impl Error for SuperviseError {
3982    fn source(&self) -> Option<&(dyn Error + 'static)> {
3983        match self {
3984            Self::Spawn { source, .. }
3985            | Self::Cgroup { source, .. }
3986            | Self::Wait { source, .. }
3987            | Self::Kill { source, .. } => Some(source),
3988            Self::Forwarding(err) => Some(err),
3989            Self::Registry(err) => Some(err),
3990            Self::LaunchNonce { .. }
3991            | Self::InvalidSpec { .. }
3992            | Self::ReloadUnavailable { .. }
3993            | Self::Disabled { .. }
3994            | Self::ReloadFailed { .. }
3995            | Self::RegistrationStillActive { .. }
3996            | Self::StatePoisoned { .. }
3997            | Self::CommandClosed { .. }
3998            | Self::SwapInProgress { .. }
3999            | Self::SwapRefused { .. }
4000            | Self::SwapFailed { .. } => None,
4001        }
4002    }
4003}
4004
4005pub(crate) fn validate_spec(spec: &ModuleSpec) -> Result<(), SuperviseError> {
4006    if spec.module_id.trim().is_empty() {
4007        return Err(SuperviseError::InvalidSpec {
4008            reason: "module_id must not be empty".to_string(),
4009        });
4010    }
4011
4012    Ok(())
4013}
4014
4015#[derive(Debug, Default)]
4016struct HealthProbeRuntime {
4017    configured_health: Option<HealthConfig>,
4018    registered_connection: Option<crate::ConnectionId>,
4019    advertised: bool,
4020    next_probe_at: Option<Instant>,
4021    probe_index: u64,
4022}
4023
4024fn running_protocol(spec: &ModuleSpec, snapshot: &SharedSnapshot) -> ModuleProtocol {
4025    lock_snapshot(snapshot)
4026        .ok()
4027        .and_then(|state| state.spawned_protocol)
4028        .unwrap_or(spec.protocol)
4029}
4030
4031impl HealthProbeRuntime {
4032    fn refresh_registration(
4033        &mut self,
4034        spec: &ModuleSpec,
4035        runtime: &SupervisorRuntimeConfig,
4036        registry: &Registry,
4037        snapshot: &SharedSnapshot,
4038    ) {
4039        if self.configured_health.as_ref() != Some(&runtime.health) {
4040            self.configured_health = Some(runtime.health.clone());
4041            self.next_probe_at = None;
4042            self.registered_connection = None;
4043            self.probe_index = 0;
4044        }
4045        // A non-wire process never registers. Only an explicitly configured
4046        // HTTP endpoint can arm its health probe; an absent HELLO is not a
4047        // health failure for that kind of process.
4048        if running_protocol(spec, snapshot) == ModuleProtocol::None {
4049            self.registered_connection = None;
4050            self.advertised = runtime.health.http.is_some();
4051            if !self.advertised {
4052                self.next_probe_at = None;
4053                let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4054                    state.health = ModuleHealthStatus::default();
4055                });
4056            } else if self.next_probe_at.is_none() {
4057                self.next_probe_at = Some(
4058                    Instant::now()
4059                        + jittered_health_delay(&spec.module_id, 0, runtime.health.cadence),
4060                );
4061            }
4062            return;
4063        }
4064
4065        let registration = match registry.get_module(&spec.module_id) {
4066            Ok(registration) => registration,
4067            Err(err) => {
4068                warn!(module_id = %spec.module_id, error = %err, "health prober could not read registry");
4069                self.advertised = false;
4070                self.next_probe_at = None;
4071                return;
4072            }
4073        };
4074
4075        let Some(registration) = registration else {
4076            self.registered_connection = None;
4077            self.advertised = false;
4078            self.next_probe_at = None;
4079            return;
4080        };
4081
4082        let advertised = registration
4083            .control_ops
4084            .iter()
4085            .any(|op| op == MODULE_CONTROL_OP_HEALTH_CHECK);
4086        if !advertised {
4087            self.registered_connection = Some(registration.connection_id);
4088            self.advertised = false;
4089            self.next_probe_at = None;
4090            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4091                state.health.status = SupervisorHealthStatus::Unknown;
4092                state.health.consecutive_failures = 0;
4093                state.health.last_probe_ms = None;
4094                state.health.detail = None;
4095                state.health.metrics = None;
4096            });
4097            return;
4098        }
4099
4100        let reregistered = self.registered_connection != Some(registration.connection_id);
4101        self.registered_connection = Some(registration.connection_id);
4102        self.advertised = true;
4103        if reregistered || self.next_probe_at.is_none() {
4104            self.probe_index = 0;
4105            self.next_probe_at = Some(
4106                Instant::now() + jittered_health_delay(&spec.module_id, 0, runtime.health.cadence),
4107            );
4108            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4109                state.health.status = SupervisorHealthStatus::Unknown;
4110                state.health.consecutive_failures = 0;
4111                state.health.detail = None;
4112                state.health.metrics = None;
4113            });
4114        }
4115    }
4116
4117    fn wake_after(&self) -> Duration {
4118        if !self.advertised {
4119            return REGISTRY_RELEASE_POLL;
4120        }
4121        self.next_probe_at
4122            .map(|next| next.saturating_duration_since(Instant::now()))
4123            .unwrap_or(REGISTRY_RELEASE_POLL)
4124    }
4125
4126    fn due(&self) -> bool {
4127        self.advertised
4128            && self
4129                .next_probe_at
4130                .is_some_and(|next| Instant::now() >= next)
4131    }
4132
4133    fn schedule_next(&mut self, spec: &ModuleSpec, cadence: Duration) {
4134        self.probe_index = self.probe_index.wrapping_add(1);
4135        self.next_probe_at = Some(
4136            Instant::now() + jittered_health_delay(&spec.module_id, self.probe_index, cadence),
4137        );
4138    }
4139}
4140
4141/// What a failed health probe actually OBSERVED, kept apart from how it reads.
4142///
4143/// This was a struct with a single `message: String`, and every one of the
4144/// fifteen construction sites collapsed into it. Each site knows exactly what it
4145/// saw -- the lane is gone, the module did not answer in time, the module
4146/// answered with the wrong thing -- and `handle_health_probe_failure` then
4147/// treated all of them identically: increment a counter, compare to a threshold,
4148/// restart the module. THE DISTINCTION EXISTED AT EVERY CALL SITE AND WAS
4149/// DESTROYED BEFORE THE DECISION THAT NEEDED IT.
4150///
4151/// The distinction that matters is not severity, it is EVIDENTIAL WEIGHT:
4152///
4153/// * `LaneDead` is PROOF. The module's control connection is gone; nothing will
4154///   answer on it again.
4155/// * `NoAnswer` is ABSENCE OF EVIDENCE. It is consistent with a wedged module
4156///   AND with a perfectly healthy one that lost a CPU race -- which is what
4157///   happens under machine load, and is how this supervisor killed a healthy
4158///   module three times in one day.
4159/// * `BadAnswer` proves the module is ALIVE. It replied; the reply was wrong.
4160///   Restarting on it is defensible, but it is not the silence case and should
4161///   never be counted as one.
4162/// * `Misconfigured` is a daemon-side fault. The module has not been asked
4163///   anything, so it cannot be evidence about the module at all.
4164///
4165/// The asymmetry is the whole point: under saturation the WEAKEST signal is the
4166/// one that fires most often, and while every variant collapsed into one string
4167/// it carried the same weight as the strongest.
4168///
4169/// LIVE BEHAVIOUR TODAY, stated here because this doc block describes the
4170/// DESIGN and a reader stopping at it gets the build backwards: the restart
4171/// decision does NOT yet consult this classification -- consecutive `NoAnswer`
4172/// probes still increment the failure streak and drive escalation at the
4173/// threshold (see `is_proof_of_death` below for why that is deliberate and
4174/// what gates the change). Absence of evidence restarts modules today.
4175#[derive(Debug)]
4176enum HealthProbeEvidence {
4177    /// The module's control lane is gone. Proof of death.
4178    LaneDead,
4179    /// No reply within the deadline. Proves nothing about the module's state.
4180    NoAnswer,
4181    /// The module replied, but not with a usable health report. Proves it is alive.
4182    BadAnswer,
4183    /// The daemon could not ask. Says nothing about the module.
4184    Misconfigured,
4185}
4186
4187#[derive(Debug)]
4188struct HealthProbeError {
4189    evidence: HealthProbeEvidence,
4190    message: String,
4191}
4192
4193impl HealthProbeError {
4194    fn lane_dead(message: impl Into<String>) -> Self {
4195        Self::with(HealthProbeEvidence::LaneDead, message)
4196    }
4197
4198    fn no_answer(message: impl Into<String>) -> Self {
4199        Self::with(HealthProbeEvidence::NoAnswer, message)
4200    }
4201
4202    fn bad_answer(message: impl Into<String>) -> Self {
4203        Self::with(HealthProbeEvidence::BadAnswer, message)
4204    }
4205
4206    fn misconfigured(message: impl Into<String>) -> Self {
4207        Self::with(HealthProbeEvidence::Misconfigured, message)
4208    }
4209
4210    fn with(evidence: HealthProbeEvidence, message: impl Into<String>) -> Self {
4211        Self {
4212            evidence,
4213            message: message.into(),
4214        }
4215    }
4216
4217    /// Whether this observation is proof the module cannot serve.
4218    ///
4219    /// Only `LaneDead` qualifies. `NoAnswer` is deliberately excluded: it is the
4220    /// variant that fires under CPU starvation, and treating it as proof is the
4221    /// defect this enum exists to make impossible to reintroduce silently.
4222    ///
4223    /// NOT YET CONSULTED BY THE RESTART DECISION, deliberately. Requiring proof
4224    /// to restart also needs a bound for the case it excludes -- a genuinely
4225    /// wedged module, alive but never answering -- and that bound must come from
4226    /// the distribution of real late-answer latencies, which nothing measures
4227    /// yet. Landing the classification first makes the later change a one-line
4228    /// decision against evidence that already exists, rather than two unproven
4229    /// changes at once.
4230    #[allow(dead_code)]
4231    fn is_proof_of_death(&self) -> bool {
4232        matches!(self.evidence, HealthProbeEvidence::LaneDead)
4233    }
4234
4235    /// Short stable label for logs and the health snapshot.
4236    ///
4237    /// An operator reading `ck health` currently cannot tell "the module is gone"
4238    /// from "the module did not answer in five seconds", because both render as
4239    /// prose in the same field. These labels are what make the two
4240    /// distinguishable at a glance, and they are what a later restart-policy
4241    /// change will be argued from.
4242    fn label(&self) -> &'static str {
4243        match self.evidence {
4244            HealthProbeEvidence::LaneDead => "lane-dead",
4245            HealthProbeEvidence::NoAnswer => "no-answer",
4246            HealthProbeEvidence::BadAnswer => "bad-answer",
4247            HealthProbeEvidence::Misconfigured => "daemon-misconfigured",
4248        }
4249    }
4250}
4251
4252impl fmt::Display for HealthProbeError {
4253    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
4254        f.write_str(&self.message)
4255    }
4256}
4257
4258async fn run_health_probe_cycle(
4259    spec: &ModuleSpec,
4260    runtime: &SupervisorRuntimeConfig,
4261    registry: &Registry,
4262    process_liveness: &SupervisorProcessLiveness,
4263    snapshot: &SharedSnapshot,
4264    child: &mut Option<SupervisedChild>,
4265) {
4266    let now_ms = unix_ms_now();
4267    let http = (running_protocol(spec, snapshot) == ModuleProtocol::None)
4268        .then_some(runtime.health.http.as_deref())
4269        .flatten();
4270    let result = match http {
4271        Some(url) => probe_http_health(url, runtime.health.deadline).await,
4272        None => probe_module_health(&spec.module_id, runtime, None).await,
4273    };
4274    match result {
4275        Ok(report) => {
4276            handle_health_report(
4277                spec,
4278                runtime,
4279                registry,
4280                process_liveness,
4281                snapshot,
4282                child,
4283                report,
4284                now_ms,
4285            )
4286            .await;
4287        }
4288        Err(err) => {
4289            if http.is_some() {
4290                let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4291                    state.health.status = SupervisorHealthStatus::Failing;
4292                });
4293            }
4294            handle_health_probe_failure(
4295                spec,
4296                runtime,
4297                registry,
4298                process_liveness,
4299                snapshot,
4300                child,
4301                err,
4302                now_ms,
4303            )
4304            .await;
4305        }
4306    }
4307}
4308
4309pub(crate) struct HttpProbeTarget<'a> {
4310    address: std::net::SocketAddr,
4311    localhost: bool,
4312    authority: &'a str,
4313    path: String,
4314}
4315
4316/// Resolve only literal loopback endpoints, without DNS, redirects, proxies,
4317/// or TLS. A URL cannot turn a local health check into an outbound connection.
4318pub(crate) fn parse_http_probe_url(url: &str) -> Result<HttpProbeTarget<'_>, String> {
4319    if url.bytes().any(|b| b <= b' ' || b == 127) || url.contains('#') {
4320        return Err("must not contain whitespace, controls, or a fragment".into());
4321    }
4322    let rest = url
4323        .strip_prefix("http://")
4324        .ok_or("must use plain http://")?;
4325    let split = rest.find(['/', '?']).unwrap_or(rest.len());
4326    let (authority, suffix) = rest.split_at(split);
4327    let (host, port) = if let Some(rest) = authority.strip_prefix("[::1]") {
4328        ("::1", rest)
4329    } else {
4330        let split = authority.find(':').unwrap_or(authority.len());
4331        authority.split_at(split)
4332    };
4333    let ip: std::net::IpAddr = match host {
4334        "127.0.0.1" | "localhost" => std::net::Ipv4Addr::LOCALHOST.into(),
4335        "::1" => std::net::Ipv6Addr::LOCALHOST.into(),
4336        _ => return Err("host must be 127.0.0.1, [::1], or localhost".into()),
4337    };
4338    let port = if port.is_empty() {
4339        80
4340    } else {
4341        port.strip_prefix(':')
4342            .and_then(|p| p.parse::<u16>().ok())
4343            .filter(|p| *p > 0)
4344            .ok_or("must have a valid nonzero TCP port")?
4345    };
4346    let path = if suffix.is_empty() {
4347        "/".into()
4348    } else if suffix.starts_with('?') {
4349        format!("/{suffix}")
4350    } else {
4351        suffix.into()
4352    };
4353    Ok(HttpProbeTarget {
4354        address: std::net::SocketAddr::new(ip, port),
4355        localhost: host == "localhost",
4356        authority,
4357        path,
4358    })
4359}
4360
4361async fn probe_http_health(
4362    url: &str,
4363    deadline: Duration,
4364) -> Result<HealthReport, HealthProbeError> {
4365    use tokio::io::{AsyncReadExt, AsyncWriteExt, BufReader};
4366    let target = parse_http_probe_url(url).map_err(HealthProbeError::misconfigured)?;
4367    // Keep partial diagnostics outside the timed future so cancellation does
4368    // not discard a status line or body bytes already received.
4369    let mut response_status = String::new();
4370    let mut body = Vec::new();
4371    let probe = async {
4372        // Resolve localhost ourselves so a hosts-file override cannot turn
4373        // this into an outbound request, while IPv6-only local servers work.
4374        let connection = match tokio::net::TcpStream::connect(target.address).await {
4375            Err(_) if target.localhost => {
4376                tokio::net::TcpStream::connect((
4377                    std::net::Ipv6Addr::LOCALHOST,
4378                    target.address.port(),
4379                ))
4380                .await
4381            }
4382            result => result,
4383        };
4384        let mut stream = connection.map_err(|error| {
4385            HealthProbeError::no_answer(format!("HTTP connect failed: {error}"))
4386        })?;
4387        stream.write_all(format!("GET {} HTTP/1.1\r\nHost: {}\r\nAccept-Encoding: identity\r\nConnection: close\r\n\r\n", target.path, target.authority).as_bytes()).await
4388            .map_err(|error| HealthProbeError::no_answer(format!("HTTP write failed: {error}")))?;
4389        let mut reader = BufReader::new(stream);
4390        let mut budget = 16 * 1024;
4391        let status = http_line(&mut reader, &mut budget).await?;
4392        let mut words = status.split_ascii_whitespace();
4393        let version = words.next();
4394        let code = words
4395            .next()
4396            .filter(|word| word.len() == 3)
4397            .and_then(|word| word.parse::<u16>().ok());
4398        if !matches!(version, Some("HTTP/1.1" | "HTTP/1.0"))
4399            || !code.is_some_and(|code| (100..600).contains(&code))
4400        {
4401            return Err(HealthProbeError::bad_answer(format!(
4402                "invalid HTTP status: {status}"
4403            )));
4404        }
4405        let code = code.expect("validated status code");
4406        response_status = status.clone();
4407        let mut length = None;
4408        let mut chunked = false;
4409        loop {
4410            let line = http_line(&mut reader, &mut budget).await?;
4411            if line.is_empty() {
4412                break;
4413            }
4414            if let Some((name, value)) = line.split_once(':') {
4415                if name.eq_ignore_ascii_case("content-length") {
4416                    length = Some(value.trim().parse::<u64>().map_err(|_| {
4417                        HealthProbeError::bad_answer("invalid HTTP Content-Length")
4418                    })?);
4419                } else if name.eq_ignore_ascii_case("transfer-encoding") {
4420                    chunked = value.trim().eq_ignore_ascii_case("chunked");
4421                }
4422            }
4423        }
4424        if chunked {
4425            while body.len() < 200 {
4426                let line = http_line(&mut reader, &mut budget).await?;
4427                let size = u64::from_str_radix(line.split(';').next().unwrap_or("").trim(), 16)
4428                    .map_err(|_| HealthProbeError::bad_answer("invalid HTTP chunk size"))?;
4429                if size == 0 {
4430                    break;
4431                }
4432                let count = size.min((200 - body.len()) as u64) as usize;
4433                let start = body.len();
4434                (&mut reader)
4435                    .take(count as u64)
4436                    .read_to_end(&mut body)
4437                    .await
4438                    .map_err(|error| {
4439                        HealthProbeError::no_answer(format!("HTTP body read failed: {error}"))
4440                    })?;
4441                if body.len() - start != count {
4442                    return Err(HealthProbeError::bad_answer("truncated HTTP chunk"));
4443                }
4444                if size > count as u64 || body.len() == 200 {
4445                    break;
4446                }
4447                if !http_line(&mut reader, &mut budget).await?.is_empty() {
4448                    return Err(HealthProbeError::bad_answer("invalid HTTP chunk delimiter"));
4449                }
4450            }
4451        } else {
4452            reader
4453                .take(length.unwrap_or(200).min(200))
4454                .read_to_end(&mut body)
4455                .await
4456                .map_err(|error| {
4457                    HealthProbeError::no_answer(format!("HTTP body read failed: {error}"))
4458                })?;
4459        }
4460        if (200..300).contains(&code) {
4461            Ok(HealthReport::ok())
4462        } else {
4463            Err(HealthProbeError::bad_answer(
4464                "HTTP health endpoint returned non-2xx",
4465            ))
4466        }
4467    };
4468    let mut result = timeout(deadline, probe).await.unwrap_or_else(|_| {
4469        Err(HealthProbeError::no_answer(format!(
4470            "HTTP probe timed out after {deadline:?}"
4471        )))
4472    });
4473    if let Err(error) = &mut result {
4474        if !response_status.is_empty() {
4475            error.message = format!(
4476                "{}; {response_status}: {}",
4477                error.message,
4478                String::from_utf8_lossy(&body)
4479            );
4480        }
4481    }
4482    result
4483}
4484
4485async fn http_line(
4486    reader: &mut tokio::io::BufReader<tokio::net::TcpStream>,
4487    remaining: &mut usize,
4488) -> Result<String, HealthProbeError> {
4489    use tokio::io::{AsyncBufReadExt, AsyncReadExt};
4490    let mut line = Vec::new();
4491    (&mut *reader)
4492        .take(*remaining as u64)
4493        .read_until(b'\n', &mut line)
4494        .await
4495        .map_err(|error| {
4496            HealthProbeError::no_answer(format!("HTTP header read failed: {error}"))
4497        })?;
4498    *remaining -= line.len();
4499    if !line.ends_with(b"\r\n") {
4500        return Err(HealthProbeError::bad_answer(
4501            "HTTP headers are incomplete or exceed 16 KiB",
4502        ));
4503    }
4504    line.truncate(line.len() - 2);
4505    String::from_utf8(line).map_err(|_| HealthProbeError::bad_answer("HTTP header is not UTF-8"))
4506}
4507
4508async fn probe_module_health(
4509    module_id: &str,
4510    runtime: &SupervisorRuntimeConfig,
4511    drain_deadline: Option<Instant>,
4512) -> Result<HealthReport, HealthProbeError> {
4513    let Some(forwarding) = runtime.forwarding.as_ref() else {
4514        return Err(HealthProbeError::misconfigured(
4515            "supervisor was not configured with a forwarding table",
4516        ));
4517    };
4518    let probe_started_at = Instant::now();
4519    let mut deadline = probe_started_at + runtime.health.deadline;
4520    if let Some(drain_deadline) = drain_deadline {
4521        deadline = deadline.min(drain_deadline);
4522    }
4523    let pending = if drain_deadline.is_some() {
4524        forwarding.begin_drain_health_probe_rpc_for(
4525            module_id,
4526            MODULE_CONTROL_OP_HEALTH_CHECK,
4527            probe_started_at,
4528            deadline,
4529        )
4530    } else {
4531        forwarding.begin_health_probe_rpc_for(
4532            module_id,
4533            MODULE_CONTROL_OP_HEALTH_CHECK,
4534            probe_started_at,
4535            deadline,
4536        )
4537    }
4538    .map_err(|err| {
4539        // The endpoint is not registered, so there is no live control lane to
4540        // ask. That is the module being absent, not slow.
4541        HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
4542    })?;
4543    await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
4544}
4545
4546/// [`probe_module_health`] for one endpoint rather than the id's active one.
4547///
4548/// A swap probes two processes that no by-id lookup reaches: its candidate
4549/// before cutover, and its superseded incumbent (for busy gauges) while the
4550/// incumbent drains. `deadline_cap` bounds the probe the way a drain deadline
4551/// bounds the by-id drain probe.
4552async fn probe_endpoint_health(
4553    endpoint: crate::ModuleEndpointId,
4554    runtime: &SupervisorRuntimeConfig,
4555    deadline_cap: Option<Instant>,
4556) -> Result<HealthReport, HealthProbeError> {
4557    let Some(forwarding) = runtime.forwarding.as_ref() else {
4558        return Err(HealthProbeError::misconfigured(
4559            "supervisor was not configured with a forwarding table",
4560        ));
4561    };
4562    let probe_started_at = Instant::now();
4563    let mut deadline = probe_started_at + runtime.health.deadline;
4564    if let Some(cap) = deadline_cap {
4565        deadline = deadline.min(cap);
4566    }
4567    let pending = forwarding
4568        .begin_endpoint_health_probe_rpc_for(
4569            endpoint,
4570            MODULE_CONTROL_OP_HEALTH_CHECK,
4571            probe_started_at,
4572            deadline,
4573        )
4574        .map_err(|err| {
4575            HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
4576        })?;
4577    await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
4578}
4579
4580/// Send a begun health probe and classify its answer.
4581async fn await_health_probe(
4582    forwarding: &ForwardingTable,
4583    pending: PendingModuleControlRpc,
4584    deadline: Instant,
4585    probe_budget: Duration,
4586) -> Result<HealthReport, HealthProbeError> {
4587    let PendingModuleControlRpc {
4588        endpoint,
4589        module_sink,
4590        negotiated_ver,
4591        corr,
4592        receiver,
4593    } = pending;
4594    let body = serde_json::to_vec(&ModuleControlRequest::HealthCheck {}).map_err(|err| {
4595        HealthProbeError::misconfigured(format!("failed to encode health.check: {err}"))
4596    })?;
4597    let frame = Frame::build_with_version(
4598        negotiated_ver,
4599        FrameType::Request,
4600        control_flags(),
4601        0,
4602        0,
4603        corr,
4604        body,
4605    )
4606    .map_err(|err| {
4607        HealthProbeError::misconfigured(format!("failed to build health.check frame: {err}"))
4608    })?;
4609
4610    // The enqueue itself must be bounded by the probe deadline: FrameSink.send
4611    // blocks waiting for capacity when the module's egress queue is full, and an
4612    // unbounded await here freezes the whole supervision actor (it stops polling
4613    // Child::wait and supervisor commands), making the module unrecoverable
4614    // in-band. On timeout the probe fails like any transport failure.
4615    match timeout_at(deadline, module_sink.send(frame)).await {
4616        Ok(Ok(())) => {}
4617        Ok(Err(err)) => {
4618            let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
4619            // A closed sink means the module's egress channel is gone -- the
4620            // receiving half is dropped when its connection tears down. Proof.
4621            return Err(HealthProbeError::lane_dead(format!(
4622                "failed to send health.check: {err}"
4623            )));
4624        }
4625        Err(_elapsed) => {
4626            let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
4627            // A full egress queue means the module is not draining its socket, which
4628            // is consistent with a wedged module AND with one whose reader is merely
4629            // starved. Silence, not proof.
4630            return Err(HealthProbeError::no_answer(
4631                "health.check send timed out before enqueue (module egress full)",
4632            ));
4633        }
4634    }
4635
4636    match timeout_at(deadline, receiver).await {
4637        // Each arm records WHAT WAS OBSERVED. Four of them are the module
4638        // demonstrably answering -- rejected, non-health, malformed, wrong op --
4639        // and those prove it is alive even though the probe failed.
4640        Ok(Ok(ModuleControlRpcOutcome::Response(response))) => {
4641            response.health_report().ok_or_else(|| {
4642                HealthProbeError::bad_answer("health.check RPC returned a non-health response")
4643            })
4644        }
4645        Ok(Ok(ModuleControlRpcOutcome::Rejected(body))) => Err(HealthProbeError::bad_answer(
4646            format!("health.check rejected: {}", body.message),
4647        )),
4648        Ok(Ok(ModuleControlRpcOutcome::ModuleGone(message))) => {
4649            Err(HealthProbeError::lane_dead(message))
4650        }
4651        Ok(Ok(ModuleControlRpcOutcome::MalformedResponse(message))) => {
4652            Err(HealthProbeError::bad_answer(message))
4653        }
4654        Ok(Ok(ModuleControlRpcOutcome::UnexpectedOp { expected, actual })) => {
4655            Err(HealthProbeError::bad_answer(format!(
4656                "expected module-control op '{expected}', got '{actual}'"
4657            )))
4658        }
4659        // A reply that crosses the deadline before this waiter observes it is
4660        // still proof of life. The forwarding path records its end-to-end latency
4661        // before delivering this classification.
4662        Ok(Ok(ModuleControlRpcOutcome::DeadlineElapsed)) => Err(HealthProbeError::bad_answer(
4663            "module answered health.check after its daemon deadline",
4664        )),
4665        Ok(Err(_)) => Err(HealthProbeError::misconfigured(
4666            "health.check waiter was canceled before the module responded",
4667        )),
4668        Err(_) => {
4669            let _ = forwarding.tombstone_health_probe_rpc(endpoint, corr);
4670            Err(HealthProbeError::no_answer(format!(
4671                "module did not answer health.check within {probe_budget:?}"
4672            )))
4673        }
4674    }
4675}
4676
4677#[allow(clippy::too_many_arguments)]
4678async fn handle_health_report(
4679    spec: &ModuleSpec,
4680    runtime: &SupervisorRuntimeConfig,
4681    registry: &Registry,
4682    process_liveness: &SupervisorProcessLiveness,
4683    snapshot: &SharedSnapshot,
4684    child: &mut Option<SupervisedChild>,
4685    report: HealthReport,
4686    now_ms: u64,
4687) {
4688    let status = supervisor_health_status(report.status);
4689    let detail = report.detail.clone();
4690    let metrics = truncate_health_metrics(report.metrics);
4691    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4692        state.health.status = status;
4693        state.health.last_probe_ms = Some(now_ms);
4694        state.health.detail = detail.clone();
4695        state.health.metrics = metrics.clone();
4696        state.health.consecutive_failures = 0;
4697    });
4698
4699    let action = match report.status {
4700        HealthStatus::Ok => return,
4701        HealthStatus::Degraded => runtime.health.on_degraded,
4702        HealthStatus::Failing => runtime.health.on_failing,
4703    };
4704    apply_l3_health_action(
4705        spec,
4706        runtime,
4707        registry,
4708        process_liveness,
4709        snapshot,
4710        child,
4711        status,
4712        detail.as_deref(),
4713        action,
4714        now_ms,
4715    )
4716    .await;
4717}
4718
4719#[allow(clippy::too_many_arguments)]
4720async fn handle_health_probe_failure(
4721    spec: &ModuleSpec,
4722    runtime: &SupervisorRuntimeConfig,
4723    registry: &Registry,
4724    process_liveness: &SupervisorProcessLiveness,
4725    snapshot: &SharedSnapshot,
4726    child: &mut Option<SupervisedChild>,
4727    err: HealthProbeError,
4728    now_ms: u64,
4729) {
4730    let threshold = runtime.health.failure_threshold.max(1);
4731    let mut failures = 0;
4732    // Carry the evidence class into the operator-visible detail. Without it,
4733    // "module did not answer within 5s" and "the control lane is gone" are two
4734    // prose strings in the same field, and the reader has to know the codebase to
4735    // tell which one is proof of anything.
4736    let detail = format!("[{}] {err}", err.label());
4737    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4738        // A failed wire probe invalidates the last report, even before the
4739        // restart threshold. HTTP probes already mark failures as Failing.
4740        if state.spawned_protocol.unwrap_or(spec.protocol) == ModuleProtocol::Subc {
4741            state.health.status = SupervisorHealthStatus::Unknown;
4742        }
4743        state.health.last_probe_ms = Some(now_ms);
4744        state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
4745        state.health.detail = Some(detail.clone());
4746        state.health.metrics = None;
4747        failures = state.health.consecutive_failures;
4748    });
4749
4750    if failures < threshold {
4751        warn!(
4752            module_id = %spec.module_id,
4753            consecutive_failures = failures,
4754            threshold,
4755            evidence = err.label(),
4756            detail = %detail,
4757            "health.check probe failed"
4758        );
4759        return;
4760    }
4761
4762    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4763        state.state = ModuleState::Unresponsive;
4764        state.health.status = SupervisorHealthStatus::Unresponsive;
4765    });
4766    // The evidence class is logged at the kill site because this is the line an
4767    // operator reads after an unexplained restart. A streak of `no-answer` under
4768    // machine load is the known false-positive shape; a `lane-dead` is not.
4769    if runtime.health.critical {
4770        error!(
4771            module_id = %spec.module_id,
4772            status = "unresponsive",
4773            evidence = err.label(),
4774            detail = %detail,
4775            "critical module health alert"
4776        );
4777    } else {
4778        warn!(
4779            module_id = %spec.module_id,
4780            status = "unresponsive",
4781            evidence = err.label(),
4782            detail = %detail,
4783            "module health threshold breached"
4784        );
4785    }
4786    if let Err(err) = health_restart_child(
4787        spec,
4788        runtime,
4789        registry,
4790        process_liveness,
4791        snapshot,
4792        child,
4793        SupervisorHealthStatus::Unresponsive,
4794        Some(&detail),
4795        now_ms,
4796    )
4797    .await
4798    {
4799        error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
4800    }
4801}
4802
4803#[allow(clippy::too_many_arguments)]
4804async fn apply_l3_health_action(
4805    spec: &ModuleSpec,
4806    runtime: &SupervisorRuntimeConfig,
4807    registry: &Registry,
4808    process_liveness: &SupervisorProcessLiveness,
4809    snapshot: &SharedSnapshot,
4810    child: &mut Option<SupervisedChild>,
4811    status: SupervisorHealthStatus,
4812    detail: Option<&str>,
4813    action: HealthAction,
4814    now_ms: u64,
4815) {
4816    record_health_action(snapshot, &spec.module_id, action.to_string(), now_ms);
4817    match action {
4818        HealthAction::Report => {
4819            info!(
4820                module_id = %spec.module_id,
4821                status = ?status,
4822                detail,
4823                "module reported non-ok health"
4824            );
4825        }
4826        HealthAction::Alert => {
4827            error!(
4828                module_id = %spec.module_id,
4829                status = ?status,
4830                detail,
4831                "module health alert"
4832            );
4833        }
4834        HealthAction::Restart => {
4835            if let Err(err) = health_restart_child(
4836                spec,
4837                runtime,
4838                registry,
4839                process_liveness,
4840                snapshot,
4841                child,
4842                status,
4843                detail,
4844                now_ms,
4845            )
4846            .await
4847            {
4848                error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
4849            }
4850        }
4851    }
4852}
4853
4854#[allow(clippy::too_many_arguments)]
4855async fn health_restart_child(
4856    spec: &ModuleSpec,
4857    runtime: &SupervisorRuntimeConfig,
4858    registry: &Registry,
4859    process_liveness: &SupervisorProcessLiveness,
4860    snapshot: &SharedSnapshot,
4861    child: &mut Option<SupervisedChild>,
4862    status: SupervisorHealthStatus,
4863    detail: Option<&str>,
4864    now_ms: u64,
4865) -> Result<(), SuperviseError> {
4866    let (enabled, schedule) = {
4867        let mut state = lock_snapshot(snapshot)?;
4868        let enabled = state.enabled;
4869        let schedule = if enabled {
4870            state.next_crash_restart(&runtime.restart_policy, Instant::now())
4871        } else {
4872            None
4873        };
4874        (enabled, schedule)
4875    };
4876
4877    if !enabled {
4878        return Err(SuperviseError::Disabled {
4879            module_id: spec.module_id.clone(),
4880        });
4881    }
4882
4883    if schedule.is_none() {
4884        record_health_action(snapshot, &spec.module_id, "failed".to_string(), now_ms);
4885        error!(
4886            module_id = %spec.module_id,
4887            status = ?status,
4888            detail,
4889            max_restarts = runtime.restart_policy.max_restarts,
4890            window_secs = runtime.restart_policy.window.as_secs(),
4891            reason = %runtime.restart_policy.budget_exhausted_detail(),
4892            "health restart budget exhausted; marking module failed"
4893        );
4894        let stop_notice = begin_forwarding_drain_if_configured(
4895            spec,
4896            runtime,
4897            registry,
4898            snapshot,
4899            Some(true),
4900            RouteCloseReason::Disable,
4901        )
4902        .await?;
4903        update_snapshot(snapshot, Some(&spec.module_id), |state| {
4904            state.drain_disposition_detail = Some(runtime.restart_policy.budget_exhausted_detail());
4905        })?;
4906        drain_optional_child(
4907            &spec.module_id,
4908            spec.protocol,
4909            stop_notice,
4910            registry,
4911            runtime.forwarding.as_deref(),
4912            snapshot,
4913            &runtime.terminal_ring,
4914            &runtime.spawn_events,
4915            child,
4916            runtime.drain_timeout,
4917            ModuleState::Failed,
4918            Some(true),
4919        )
4920        .await?;
4921        process_liveness.untrack_if_current(&spec.module_id, snapshot);
4922        return Ok(());
4923    }
4924
4925    let schedule = schedule.expect("a health restart must have a crash-restart schedule");
4926    let mut restart_count = 0;
4927    update_snapshot(snapshot, Some(&spec.module_id), |state| {
4928        restart_count = state.crash_restarts.len();
4929        state.state = ModuleState::Unresponsive;
4930        state.health.status = status;
4931        state.health.last_action = Some(HealthAction::Restart.to_string());
4932        state.health.last_action_ms = Some(now_ms);
4933    })?;
4934    warn!(
4935        module_id = %spec.module_id,
4936        status = ?status,
4937        detail,
4938        restart_count,
4939        restart_in_window = schedule.restart_in_window,
4940        delay_ms = schedule.delay.as_millis() as u64,
4941        "health-triggered module restart"
4942    );
4943
4944    let stop_notice = begin_forwarding_drain_if_configured(
4945        spec,
4946        runtime,
4947        registry,
4948        snapshot,
4949        Some(true),
4950        RouteCloseReason::Restart,
4951    )
4952    .await?;
4953    drain_optional_child(
4954        &spec.module_id,
4955        spec.protocol,
4956        stop_notice,
4957        registry,
4958        runtime.forwarding.as_deref(),
4959        snapshot,
4960        &runtime.terminal_ring,
4961        &runtime.spawn_events,
4962        child,
4963        runtime.drain_timeout,
4964        ModuleState::Restarting,
4965        Some(true),
4966    )
4967    .await?;
4968    schedule_respawn(
4969        runtime,
4970        snapshot,
4971        &spec.module_id,
4972        schedule.delay,
4973        RespawnKind::Spawn,
4974    )
4975}
4976
4977fn cancel_deferred_reload(runtime: &SupervisorRuntimeConfig, module_id: &str, reason: &str) {
4978    if let Some(reply) = runtime
4979        .deferred_reload_reply
4980        .lock()
4981        .unwrap_or_else(|p| p.into_inner())
4982        .take()
4983    {
4984        let _ = reply.send(Err(SuperviseError::ReloadFailed {
4985            module_id: module_id.to_string(),
4986            reason: reason.to_string(),
4987        }));
4988    }
4989}
4990
4991fn schedule_respawn(
4992    runtime: &SupervisorRuntimeConfig,
4993    snapshot: &SharedSnapshot,
4994    module_id: &str,
4995    delay: Duration,
4996    kind: RespawnKind,
4997) -> Result<(), SuperviseError> {
4998    cancel_deferred_reload(runtime, module_id, "respawn superseded by another restart");
4999    update_snapshot(snapshot, Some(module_id), |state| {
5000        state.respawn_pending = true
5001    })?;
5002    *runtime
5003        .scheduled_respawn
5004        .lock()
5005        .unwrap_or_else(|p| p.into_inner()) = Some(PendingRespawn {
5006        deadline: Instant::now() + delay,
5007        kind,
5008    });
5009    Ok(())
5010}
5011
5012fn record_health_action(snapshot: &SharedSnapshot, module_id: &str, action: String, now_ms: u64) {
5013    let _ = update_snapshot(snapshot, Some(module_id), |state| {
5014        state.health.last_action = Some(action);
5015        state.health.last_action_ms = Some(now_ms);
5016    });
5017}
5018
5019fn supervisor_health_status(status: HealthStatus) -> SupervisorHealthStatus {
5020    match status {
5021        HealthStatus::Ok => SupervisorHealthStatus::Ok,
5022        HealthStatus::Degraded => SupervisorHealthStatus::Degraded,
5023        HealthStatus::Failing => SupervisorHealthStatus::Failing,
5024    }
5025}
5026
5027/// Caps the metrics blob stored in the cached supervisor snapshot, which is
5028/// returned to every `supervisor.list` and `supervisor.health` caller.
5029///
5030/// This cap is deliberately NOT applied on the one-shot `supervisor.health_probe`
5031/// path: that request exists to return a module's complete metrics object, and
5032/// `ck health <module-id>` documents it as the way to see what the cached view
5033/// truncates. The asymmetry is the feature.
5034///
5035/// So a new caller must decide which side it is on rather than assume the cap is
5036/// universal. Reaching for it on a fresh-probe path would silently reintroduce
5037/// the truncation that path exists to avoid.
5038fn truncate_health_metrics(metrics: Option<Value>) -> Option<Value> {
5039    let metrics = metrics?;
5040    match serde_json::to_vec(&metrics) {
5041        Ok(encoded) if encoded.len() > MAX_HEALTH_METRICS_BYTES => Some(serde_json::json!({
5042            "truncated": true,
5043            "original_bytes": encoded.len(),
5044        })),
5045        Ok(_) | Err(_) => Some(metrics),
5046    }
5047}
5048
5049/// Spread health probes so a fleet-wide restart does not converge them.
5050///
5051/// The delay is derived from the module id and probe index rather than a random
5052/// source, so it is deterministic per module: a module keeps its own offset
5053/// across daemon restarts instead of re-rolling into a collision.
5054fn jittered_health_delay(module_id: &str, probe_index: u64, cadence: Duration) -> Duration {
5055    if cadence.is_zero() {
5056        return Duration::ZERO;
5057    }
5058    let cadence_ms = cadence.as_millis() as u64;
5059    // This early return is REDUNDANT, deliberately, and a mutation run will show
5060    // it surviving removal. Recording why here so the next person to notice does
5061    // not have to re-derive it:
5062    //
5063    // - It is unreachable in practice. `positive_millis` in daemon_config rejects
5064    //   a zero cadence and builds the Duration from whole milliseconds, so a
5065    //   sub-millisecond cadence cannot come from config.
5066    // - Even if reached it changes no answer. The `.max(1)` below makes the span
5067    //   1, and `hash % 1` is 0, so the fall-through returns `cadence` unchanged
5068    //   -- exactly what this returns.
5069    //
5070    // Kept as a guard against a future widening of the config parser (accepting
5071    // microseconds, say), which would make the sub-millisecond case reachable.
5072    // The `.max(1)` is the load-bearing half TODAY: remove it and the modulo
5073    // divides by zero. Remove this and nothing changes.
5074    if cadence_ms == 0 {
5075        return cadence;
5076    }
5077    // Note that this never returns less than one cadence, including for the FIRST
5078    // probe. So a freshly registered module reports health `unknown` for a full
5079    // cadence plus jitter -- 30-33s at the default -- no matter how quickly it is
5080    // ready to answer.
5081    //
5082    // That is a property of the supervisor's schedule, not of any module: an
5083    // operator watching a restart sees `unknown` and cannot tell it from a module
5084    // that is slow to warm. Measured on two unrelated modules, both flipping to
5085    // `ok` between 22s and 32s after restart.
5086    //
5087    // Left as-is because spreading the first probe is what keeps a fleet-wide
5088    // restart from firing fourteen simultaneous probes into a cold machine. The
5089    // alternative -- probe at t+0 and jitter only from the second onward -- trades
5090    // that thundering herd for a faster first reading.
5091    let jitter_span = (cadence_ms / 10).max(1);
5092    let hash = module_id.as_bytes().iter().fold(
5093        probe_index.wrapping_mul(0x9E37_79B9_7F4A_7C15),
5094        |acc, byte| {
5095            acc.wrapping_mul(1099511628211)
5096                .wrapping_add(u64::from(*byte))
5097        },
5098    );
5099    cadence + Duration::from_millis(hash % jitter_span)
5100}
5101
5102#[cfg(test)]
5103mod tests {
5104    use super::*;
5105
5106    #[test]
5107    fn readding_a_module_clears_its_rescan_removal_tombstone() {
5108        let handle = SupervisorHandle::new();
5109        let module_id = "readded-tombstone";
5110        handle.record_rescan_removal(module_id);
5111        assert!(handle.removal_tombstone_age_ms(module_id).is_some());
5112
5113        handle.apply_identity_configuration(&ModuleSpec {
5114            module_id: module_id.to_string(),
5115            program: PathBuf::from("/test/module"),
5116            args: Vec::new(),
5117            env: Vec::new(),
5118            reserved: false,
5119            reserved_prefixes: Vec::new(),
5120            protocol: ModuleProtocol::Subc,
5121            overlap: Default::default(),
5122        });
5123
5124        assert!(
5125            handle.removal_tombstone_age_ms(module_id).is_none(),
5126            "a re-added module must not retain a stale removal tombstone"
5127        );
5128    }
5129
5130    /// What one module's owner looked like from the control plane at the
5131    /// instant after its first process was spawned.
5132    #[derive(Debug, PartialEq, Eq)]
5133    struct OwnerInSpawnWindow {
5134        module_id: String,
5135        configured: bool,
5136        on_roster: bool,
5137        admission_refusal: Option<&'static str>,
5138    }
5139
5140    /// A supervised module's process can connect, register, sync its scopes
5141    /// and describe them as soon as it is spawned, which is BEFORE the
5142    /// supervisor puts the module on the roster. In that window the owner must
5143    /// already read as configured, so a scoped `route.open` against it is
5144    /// refused as retryable `scope_not_synced` and not as terminal
5145    /// `scope_not_live` ("will never sync").
5146    ///
5147    /// The hook runs in exactly that window on every path that takes on a new
5148    /// module, so no race with a real child is needed: `on_roster: false`
5149    /// proves each observation was taken before the roster insert.
5150    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5151    async fn a_new_module_is_configured_before_its_first_process_can_register() {
5152        use crate::scopes::ScopeTable;
5153        use subc_protocol::{error_codes, scope::ScopeSelector, Principal};
5154
5155        let dir = subc_test_support::TestTempDir::new("configured-before-spawn");
5156        let stub = |module_id: &str, program: PathBuf| ModuleSpec {
5157            module_id: module_id.to_string(),
5158            program,
5159            args: Vec::new(),
5160            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
5161                .into_iter()
5162                .map(|key| (key.to_string(), dir.path().display().to_string()))
5163                .chain([("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string())])
5164                .collect(),
5165            reserved: false,
5166            reserved_prefixes: Vec::new(),
5167            protocol: ModuleProtocol::Subc,
5168            overlap: Default::default(),
5169        };
5170        let live = super::terminal_history_tests::fake_aft_stub_path();
5171        let missing = dir.path().join("definitely-missing-module");
5172
5173        let handle = SupervisorHandle::new();
5174        let mut supervisor =
5175            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5176                .with_handle(handle.clone());
5177        let observed = Arc::new(Mutex::new(Vec::<OwnerInSpawnWindow>::new()));
5178        let hook_handle = handle.clone();
5179        let hook_observed = Arc::clone(&observed);
5180        supervisor.test_after_first_spawn = AfterFirstSpawnHook(Some(Arc::new(move |module_id| {
5181            // Exactly what the control plane computes for a scoped route.open
5182            // naming this module as the owner of a scope it has not synced.
5183            let configured = hook_handle.is_configured(module_id);
5184            let selector = ScopeSelector {
5185                owner: Principal::Reserved {
5186                    module_id: module_id.to_string(),
5187                },
5188                scope_ref: "s".to_string(),
5189                scope_epoch: Some(1),
5190            };
5191            let carrier = Principal::Reserved {
5192                module_id: "carrier".to_string(),
5193            };
5194            let admission_refusal = match ScopeTable::new(Vec::<String>::new())
5195                .admit(&carrier, module_id, &selector, configured)
5196            {
5197                Ok(_) => None,
5198                Err(refusal) => Some(refusal.code),
5199            };
5200            hook_observed.lock().unwrap().push(OwnerInSpawnWindow {
5201                module_id: module_id.to_string(),
5202                configured,
5203                on_roster: hook_handle.get(module_id).is_some(),
5204                admission_refusal,
5205            });
5206        })));
5207
5208        let plain = supervisor.spawn(stub("plain", live.clone())).unwrap();
5209        let configured = supervisor
5210            .supervise_configured(stub("configured", live.clone()), true)
5211            .unwrap();
5212        let with_health = supervisor
5213            .supervise_configured_with_health(
5214                stub("with-health", live.clone()),
5215                true,
5216                HealthConfig::default(),
5217                None,
5218                RestartPolicy::default(),
5219            )
5220            .unwrap();
5221        // The failed-spawn path still puts the module on the roster (as
5222        // failed), so it is configured throughout.
5223        let failed = supervisor
5224            .supervise_configured_with_health(
5225                stub("failed-spawn", missing.clone()),
5226                true,
5227                HealthConfig::default(),
5228                None,
5229                RestartPolicy::default(),
5230            )
5231            .unwrap();
5232        // A failed plain `spawn` puts nothing on the roster, so its mark is
5233        // taken back once the spawn has failed.
5234        assert!(supervisor.spawn(stub("spawn-error", missing)).is_err());
5235
5236        let expected = [
5237            "plain",
5238            "configured",
5239            "with-health",
5240            "failed-spawn",
5241            "spawn-error",
5242        ]
5243        .into_iter()
5244        .map(|module_id| OwnerInSpawnWindow {
5245            module_id: module_id.to_string(),
5246            configured: true,
5247            on_roster: false,
5248            admission_refusal: Some(error_codes::SCOPE_NOT_SYNCED),
5249        })
5250        .collect::<Vec<_>>();
5251        assert_eq!(*observed.lock().unwrap(), expected);
5252
5253        for module_id in ["plain", "configured", "with-health", "failed-spawn"] {
5254            assert!(
5255                handle.get(module_id).is_some(),
5256                "{module_id} is on the roster"
5257            );
5258            assert!(
5259                handle.is_configured(module_id),
5260                "{module_id} stays configured"
5261            );
5262        }
5263        assert!(handle.get("spawn-error").is_none());
5264        assert!(
5265            !handle.is_configured("spawn-error"),
5266            "a plain spawn that failed must not leave its module marked configured"
5267        );
5268
5269        // Leaving the roster clears the mark with it.
5270        handle.retire("failed-spawn");
5271        assert!(!handle.is_configured("failed-spawn"));
5272
5273        for module in [plain, configured, with_health] {
5274            module.stop().await.unwrap();
5275        }
5276        drop(failed);
5277    }
5278
5279    fn stale_process_snapshot(state: ModuleState, enabled: bool) -> SharedSnapshot {
5280        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(state, enabled)));
5281        update_snapshot(&snapshot, Some("stale-process-facts"), |snapshot| {
5282            snapshot.process_alive = true;
5283            snapshot.pid = Some(41);
5284            snapshot.spawned_at_ms = Some(42);
5285            snapshot.spawned_from = Some(PathBuf::from("/spawned/module"));
5286            snapshot.spawned_file_identity = Some(SpawnedFileIdentity {
5287                device: 43,
5288                inode: 44,
5289            });
5290        })
5291        .unwrap();
5292        snapshot
5293    }
5294
5295    fn assert_snapshot_process_facts_cleared(snapshot: &SharedSnapshot) {
5296        let snapshot = lock_snapshot(snapshot).unwrap();
5297        assert!(!snapshot.process_alive);
5298        assert_eq!(snapshot.pid, None);
5299        assert_eq!(snapshot.spawned_at_ms, None);
5300        assert_eq!(snapshot.spawned_from, None);
5301        assert_eq!(snapshot.spawned_file_identity, None);
5302    }
5303
5304    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5305    async fn failed_enable_spawn_clears_preexisting_current_process_facts() {
5306        let supervisor =
5307            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5308        let mut runtime = supervisor.runtime_config();
5309        runtime.test_seed_stale_facts_before_enable_spawn = true;
5310        let snapshot = stale_process_snapshot(ModuleState::Disabled, false);
5311        let mut child = None;
5312        let spec = ModuleSpec {
5313            module_id: "failed-enable-clears-facts".to_string(),
5314            program: PathBuf::from("/definitely/missing/failed-enable-module"),
5315            args: Vec::new(),
5316            env: Vec::new(),
5317            reserved: false,
5318            reserved_prefixes: Vec::new(),
5319            protocol: ModuleProtocol::Subc,
5320            overlap: Default::default(),
5321        };
5322
5323        let result = set_child_enabled(
5324            &spec,
5325            &runtime,
5326            &supervisor.registry,
5327            &supervisor.process_liveness,
5328            &snapshot,
5329            &mut child,
5330            true,
5331        )
5332        .await;
5333
5334        assert!(matches!(result, Err(SuperviseError::Spawn { .. })));
5335        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
5336        assert_snapshot_process_facts_cleared(&snapshot);
5337    }
5338
5339    #[tokio::test]
5340    async fn start_revives_stranded_restarting_but_not_pending_backoff() {
5341        let supervisor =
5342            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5343        let runtime = supervisor.runtime_config();
5344        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(
5345            ModuleState::Restarting,
5346            true,
5347        )));
5348        let spec = ModuleSpec {
5349            module_id: "start-stranded-restarting".to_string(),
5350            program: super::terminal_history_tests::fake_aft_stub_path(),
5351            args: Vec::new(),
5352            env: vec![("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string())],
5353            reserved: false,
5354            reserved_prefixes: Vec::new(),
5355            protocol: ModuleProtocol::None,
5356            overlap: Default::default(),
5357        };
5358        let mut child = None;
5359        lock_snapshot(&snapshot).unwrap().respawn_pending = true;
5360        assert!(!super::set_child_enabled(
5361            &spec,
5362            &runtime,
5363            &Registry::default(),
5364            &supervisor.process_liveness,
5365            &snapshot,
5366            &mut child,
5367            true
5368        )
5369        .await
5370        .unwrap());
5371        assert!(child.is_none());
5372        lock_snapshot(&snapshot).unwrap().respawn_pending = false;
5373        assert!(super::set_child_enabled(
5374            &spec,
5375            &runtime,
5376            &Registry::default(),
5377            &supervisor.process_liveness,
5378            &snapshot,
5379            &mut child,
5380            true
5381        )
5382        .await
5383        .unwrap());
5384        assert_eq!(
5385            lock_snapshot(&snapshot).unwrap().state,
5386            ModuleState::Running
5387        );
5388        let mut child = child.unwrap();
5389        child.start_kill().unwrap();
5390        child.wait().await.unwrap();
5391    }
5392
5393    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5394    async fn failed_reload_spawn_clears_current_process_facts() {
5395        let supervisor =
5396            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5397        let mut runtime = supervisor.runtime_config();
5398        runtime.restart_policy = RestartPolicy::new(0, Duration::ZERO);
5399        let snapshot = stale_process_snapshot(ModuleState::Running, true);
5400        let mut child = None;
5401        let spec = ModuleSpec {
5402            module_id: "failed-reload-clears-facts".to_string(),
5403            program: PathBuf::from("/unused/failed-reload-module"),
5404            args: Vec::new(),
5405            env: Vec::new(),
5406            reserved: false,
5407            reserved_prefixes: Vec::new(),
5408            protocol: ModuleProtocol::Subc,
5409            overlap: Default::default(),
5410        };
5411
5412        let result = handle_reload_spawn_failure(
5413            &spec,
5414            &runtime,
5415            &supervisor.process_liveness,
5416            &snapshot,
5417            &mut child,
5418            "forced reload spawn failure".to_string(),
5419        )
5420        .await;
5421
5422        assert!(matches!(result, Err(SuperviseError::ReloadFailed { .. })));
5423        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
5424        assert_snapshot_process_facts_cleared(&snapshot);
5425    }
5426
5427    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5428    async fn dropping_a_module_with_an_active_monitor_clears_current_process_facts() {
5429        let supervisor =
5430            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5431        let snapshot = stale_process_snapshot(ModuleState::Running, true);
5432        let module = supervisor.supervised_module(
5433            ModuleSpec {
5434                module_id: "drop-clears-facts".to_string(),
5435                program: PathBuf::from("/unused/drop-module"),
5436                args: Vec::new(),
5437                env: Vec::new(),
5438                reserved: false,
5439                reserved_prefixes: Vec::new(),
5440                protocol: ModuleProtocol::Subc,
5441                overlap: Default::default(),
5442            },
5443            supervisor.runtime_config(),
5444            Arc::clone(&snapshot),
5445            None,
5446        );
5447        assert!(!module
5448            .inner
5449            .monitor
5450            .lock()
5451            .unwrap()
5452            .as_ref()
5453            .unwrap()
5454            .is_finished());
5455
5456        drop(module);
5457
5458        assert_eq!(
5459            lock_snapshot(&snapshot).unwrap().state,
5460            ModuleState::Stopped
5461        );
5462        assert_snapshot_process_facts_cleared(&snapshot);
5463    }
5464
5465    #[cfg(unix)]
5466    #[tokio::test]
5467    async fn rescan_preserves_running_protocol_until_respawn() {
5468        let dir = subc_test_support::TestTempDir::new("rescan-protocol");
5469        let initial = ModuleSpec {
5470            module_id: "rescan-protocol".into(),
5471            program: PathBuf::from("/bin/sleep"),
5472            args: vec!["60".into()],
5473            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
5474                .into_iter()
5475                .map(|key| (key.into(), dir.path().to_string_lossy().into_owned()))
5476                .collect(),
5477            reserved: false,
5478            reserved_prefixes: vec![],
5479            protocol: ModuleProtocol::None,
5480            overlap: Default::default(),
5481        };
5482        let supervisor =
5483            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5484        let module = supervisor.spawn(initial.clone()).unwrap();
5485        assert!(module.status().unwrap().live);
5486        let mut next = initial;
5487        next.protocol = ModuleProtocol::Subc;
5488        module
5489            .update_configuration(next.clone(), HealthConfig::default(), None)
5490            .await
5491            .unwrap();
5492        assert!(
5493            module.status().unwrap().live,
5494            "rescan must not require HELLO from the old non-wire process"
5495        );
5496        let runtime = supervisor.runtime_config();
5497        let action = on_child_exit(
5498            &next,
5499            RestartPolicy::default(),
5500            &supervisor.registry,
5501            &module.inner.snapshot,
5502            &runtime.terminal_ring,
5503            &runtime.spawn_events,
5504            &runtime.child_roster,
5505            ExitReport {
5506                kind: ExitKind::Clean,
5507                code: Some(0),
5508                signal: None,
5509                at_ms: unix_ms_now(),
5510            },
5511        )
5512        .await;
5513        assert!(
5514            matches!(action, NextAction::Restart { .. }),
5515            "the old non-wire process's clean exit must restart"
5516        );
5517        module.drain().await.unwrap();
5518    }
5519
5520    #[cfg(unix)]
5521    #[tokio::test]
5522    async fn orphan_identity_matches_path_names_and_shebang_interpreters() {
5523        use std::os::unix::fs::PermissionsExt;
5524        let dir = subc_test_support::TestTempDir::new("spawn-image-identity");
5525        let script = dir.join("module.sh");
5526        std::fs::write(&script, "#!/bin/sh\nwhile :; do sleep 0.1; done\n").unwrap();
5527        std::fs::set_permissions(&script, std::fs::Permissions::from_mode(0o700)).unwrap();
5528        let record_path = dir.join("live-children.json");
5529        let supervisor =
5530            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5531                .with_live_children_record(&record_path);
5532        for (program, args) in [
5533            (PathBuf::from("sleep"), vec!["60".into()]),
5534            (script, vec![]),
5535        ] {
5536            let spec = ModuleSpec {
5537                module_id: "image-identity".into(),
5538                program: program.clone(),
5539                args,
5540                env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
5541                    .into_iter()
5542                    .map(|key| (key.into(), dir.to_string_lossy().into_owned()))
5543                    .collect(),
5544                reserved: false,
5545                reserved_prefixes: vec![],
5546                protocol: ModuleProtocol::None,
5547                overlap: Default::default(),
5548            };
5549            let module = supervisor.spawn(spec).unwrap();
5550            #[cfg(target_os = "macos")]
5551            {
5552                // SETEXEC confirmation is asynchronous; the orphan record must
5553                // identify the final image, never the intermediate trampoline.
5554                let deadline = Instant::now() + Duration::from_secs(5);
5555                while crate::live_children::read_record(&record_path)
5556                    .unwrap()
5557                    .iter()
5558                    .all(|entry| entry.executable.is_none())
5559                {
5560                    assert!(Instant::now() < deadline, "module image was not confirmed");
5561                    tokio::time::sleep(Duration::from_millis(5)).await;
5562                }
5563            }
5564            let entry = crate::live_children::read_record(&record_path)
5565                .unwrap()
5566                .pop()
5567                .unwrap();
5568            let observed = subc_os::Process::open(entry.pid)
5569                .unwrap()
5570                .unwrap()
5571                .observe()
5572                .unwrap();
5573            let verdict = crate::live_children::identity_verdict(&entry, entry.pid, &observed);
5574            module.drain().await.unwrap();
5575            assert_eq!(
5576                verdict,
5577                crate::live_children::IdentityVerdict::Matches,
5578                "program {program:?}: recorded {entry:?}, observed {observed:?}"
5579            );
5580        }
5581    }
5582
5583    #[cfg(unix)]
5584    fn http_fixture(
5585        dir: &std::path::Path,
5586        url: &str,
5587        threshold: u32,
5588    ) -> crate::daemon_config::ConfiguredModule {
5589        let path = dir.join("subc.jsonc");
5590        std::fs::write(&path, serde_json::json!({"version": 1, "modules": {"http-process": {
5591            "program": "/bin/sleep", "args": ["60"], "protocol": "none",
5592            "env": {"XDG_DATA_HOME":dir,"XDG_RUNTIME_DIR":dir,"XDG_CONFIG_HOME":dir},
5593            "health": {"http":url,"cadence_ms":20,"deadline_ms":200,"failure_threshold":threshold},
5594            "restart": {"backoff_ms":1,"max_backoff_ms":1}, "drain_timeout_ms":10
5595        }}}).to_string()).unwrap();
5596        crate::daemon_config::load(&path)
5597            .unwrap()
5598            .unwrap()
5599            .modules
5600            .pop()
5601            .unwrap()
5602    }
5603
5604    #[cfg(unix)]
5605    async fn wait_http_health(module: &SupervisedModule, status: SupervisorHealthStatus) {
5606        timeout(Duration::from_secs(5), async {
5607            loop {
5608                if module.status().unwrap().health.status == status {
5609                    break;
5610                }
5611                sleep(Duration::from_millis(5)).await;
5612            }
5613        })
5614        .await
5615        .unwrap_or_else(|_| {
5616            panic!(
5617                "expected {status:?}, got {:?}",
5618                module.status().unwrap().health
5619            )
5620        });
5621    }
5622
5623    #[cfg(unix)]
5624    #[tokio::test]
5625    async fn http_health_status_flips_ok_failing_ok() {
5626        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5627        let dir = subc_test_support::TestTempDir::new("http-status-flips");
5628        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5629        let status = Arc::new(std::sync::atomic::AtomicU16::new(200));
5630        let serving_status = status.clone();
5631        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5632        let server = tokio::spawn(async move {
5633            loop {
5634                let (mut stream, _) = listener.accept().await.unwrap();
5635                let mut request = [0u8; 2048];
5636                let count = stream.read(&mut request).await.unwrap();
5637                assert!(count > 0, "a probe must send an HTTP request");
5638                let code = serving_status.load(std::sync::atomic::Ordering::SeqCst);
5639                let body = if code == 200 {
5640                    "ready"
5641                } else {
5642                    "{\"status\":\"unavailable\",\"error\":\"scratch failure\"}"
5643                };
5644                let response = format!("HTTP/1.1 {code} scratch\r\nContent-Length: {}\r\nConnection: close\r\n\r\n{body}", body.len());
5645                let _ = stream.write_all(response.as_bytes()).await;
5646            }
5647        });
5648        let configured = http_fixture(&dir, &url, 1000);
5649        let module =
5650            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5651                .supervise_configured_with_health(
5652                    configured.module_spec(),
5653                    true,
5654                    configured.health,
5655                    configured.drain_timeout_ms,
5656                    configured.restart,
5657                )
5658                .unwrap();
5659        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5660        status.store(503, std::sync::atomic::Ordering::SeqCst);
5661        wait_http_health(&module, SupervisorHealthStatus::Failing).await;
5662        assert!(module
5663            .status()
5664            .unwrap()
5665            .health
5666            .detail
5667            .unwrap()
5668            .contains("scratch failure"));
5669        status.store(200, std::sync::atomic::Ordering::SeqCst);
5670        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5671        assert_eq!(module.status().unwrap().health.consecutive_failures, 0);
5672        let before = module.status().unwrap();
5673        let (spec, mut health) = module.configuration().unwrap();
5674        health.http = None;
5675        module
5676            .update_configuration(spec.clone(), health.clone(), Some(10))
5677            .await
5678            .unwrap();
5679        wait_http_health(&module, SupervisorHealthStatus::Unknown).await;
5680        health.http = Some(url);
5681        module
5682            .update_configuration(spec, health, Some(10))
5683            .await
5684            .unwrap();
5685        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5686        assert_eq!(
5687            module.status().unwrap().pid,
5688            before.pid,
5689            "changing a probe must apply live, not restart its process"
5690        );
5691        let (spec, mut health) = module.configuration().unwrap();
5692        health.failure_threshold = 2;
5693        module
5694            .update_configuration(spec, health, Some(10))
5695            .await
5696            .unwrap();
5697        status.store(503, std::sync::atomic::Ordering::SeqCst);
5698        timeout(Duration::from_secs(5), async {
5699            while module.status().unwrap().spawn_generation == before.spawn_generation {
5700                sleep(Duration::from_millis(5)).await;
5701            }
5702        })
5703        .await
5704        .expect("sustained HTTP 503 responses must trigger the health restart policy");
5705        module.drain().await.unwrap();
5706        server.abort();
5707    }
5708
5709    #[cfg(unix)]
5710    #[tokio::test]
5711    async fn http_health_no_listener_restarts_after_consecutive_failures() {
5712        let dir = subc_test_support::TestTempDir::new("http-refused");
5713        let unused = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5714        let url = format!("http://{}/healthz", unused.local_addr().unwrap());
5715        drop(unused);
5716        let configured = http_fixture(&dir, &url, 2);
5717        let module =
5718            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5719                .supervise_configured_with_health(
5720                    configured.module_spec(),
5721                    true,
5722                    configured.health,
5723                    configured.drain_timeout_ms,
5724                    configured.restart,
5725                )
5726                .unwrap();
5727        let before = module.status().unwrap().spawn_generation;
5728        timeout(Duration::from_secs(5), async {
5729            loop {
5730                let status = module.status().unwrap();
5731                if status.spawn_generation > before {
5732                    assert!(status.lifetime_restarts > 0);
5733                    break;
5734                }
5735                sleep(Duration::from_millis(5)).await;
5736            }
5737        })
5738        .await
5739        .expect("sustained HTTP refusal must trigger the health restart policy");
5740        module.drain().await.unwrap();
5741    }
5742
5743    #[cfg(unix)]
5744    #[tokio::test]
5745    async fn http_health_timeout_honours_deadline() {
5746        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5747        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5748        let server = tokio::spawn(async move {
5749            let _held = listener.accept().await.unwrap();
5750            std::future::pending::<()>().await;
5751        });
5752        let error = timeout(
5753            Duration::from_secs(1),
5754            probe_http_health(&url, Duration::from_millis(10)),
5755        )
5756        .await
5757        .expect("the probe must enforce its own deadline")
5758        .unwrap_err();
5759        server.abort();
5760        assert!(error.to_string().contains("timed out"));
5761    }
5762
5763    #[cfg(unix)]
5764    #[tokio::test]
5765    async fn http_health_localhost_reaches_an_ipv6_only_listener() {
5766        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5767        let listener = tokio::net::TcpListener::bind("[::1]:0").await.unwrap();
5768        let url = format!(
5769            "http://localhost:{}/healthz",
5770            listener.local_addr().unwrap().port()
5771        );
5772        let server = tokio::spawn(async move {
5773            let (mut stream, _) = listener.accept().await.unwrap();
5774            let mut request = [0u8; 2048];
5775            assert!(stream.read(&mut request).await.unwrap() > 0);
5776            stream
5777                .write_all(b"HTTP/1.1 200 OK\r\nContent-Length: 2\r\n\r\nok")
5778                .await
5779                .unwrap();
5780        });
5781        // The deadline only bounds a hang. A probe that never tried the IPv6
5782        // address would be refused on 127.0.0.1 and fail at once, so a longer
5783        // deadline does not weaken the assertion; one second timed out under a
5784        // loaded parallel test run.
5785        assert_eq!(
5786            probe_http_health(&url, Duration::from_secs(10))
5787                .await
5788                .unwrap()
5789                .status,
5790            HealthStatus::Ok
5791        );
5792        server.await.unwrap();
5793    }
5794
5795    #[cfg(unix)]
5796    #[tokio::test]
5797    async fn http_health_timeout_keeps_partial_status_and_body() {
5798        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5799        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5800        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5801        let server = tokio::spawn(async move {
5802            let (mut stream, _) = listener.accept().await.unwrap();
5803            let mut request = [0u8; 2048];
5804            assert!(stream.read(&mut request).await.unwrap() > 0);
5805            stream
5806                .write_all(
5807                    b"HTTP/1.1 503 Unavailable\r\nContent-Length: 100\r\n\r\npartial diagnostic",
5808                )
5809                .await
5810                .unwrap();
5811            std::future::pending::<()>().await;
5812        });
5813        let error = probe_http_health(&url, Duration::from_secs(1))
5814            .await
5815            .unwrap_err()
5816            .to_string();
5817        server.abort();
5818        assert!(
5819            error.contains("timed out")
5820                && error.contains("503 Unavailable")
5821                && error.contains("partial diagnostic"),
5822            "{error}"
5823        );
5824    }
5825
5826    #[cfg(unix)]
5827    #[tokio::test]
5828    async fn http_health_diagnostic_body_is_bounded_to_200_bytes() {
5829        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5830        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5831        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5832        let server = tokio::spawn(async move {
5833            let (mut stream, _) = listener.accept().await.unwrap();
5834            let mut request = [0u8; 2048];
5835            assert!(stream.read(&mut request).await.unwrap() > 0);
5836            let body = format!("{}not-in-diagnostic", "x".repeat(400));
5837            let response = format!(
5838                "HTTP/1.1 503 Unavailable\r\nContent-Length: {}\r\n\r\n{body}",
5839                body.len()
5840            );
5841            stream.write_all(response.as_bytes()).await.unwrap();
5842        });
5843        let error = probe_http_health(&url, Duration::from_secs(1))
5844            .await
5845            .unwrap_err()
5846            .to_string();
5847        server.await.unwrap();
5848        assert!(error.ends_with(&"x".repeat(200)), "{error}");
5849        assert_eq!(error.split(": ").last().unwrap().len(), 200);
5850        assert!(!error.contains("not-in-diagnostic"));
5851    }
5852
5853    #[cfg(unix)]
5854    #[tokio::test]
5855    async fn http_health_real_nats_server_monitoring() {
5856        if std::process::Command::new("nats-server")
5857            .arg("--version")
5858            .env("XDG_DATA_HOME", std::env::temp_dir())
5859            .env("XDG_RUNTIME_DIR", std::env::temp_dir())
5860            .env("XDG_CONFIG_HOME", std::env::temp_dir())
5861            .output()
5862            .is_err()
5863        {
5864            eprintln!(
5865                "skipped: http_health_real_nats_server_monitoring (nats-server is not on PATH)"
5866            );
5867            return;
5868        }
5869        let dir = subc_test_support::TestTempDir::new("nats-http-monitoring");
5870        let monitor = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5871        let port = monitor.local_addr().unwrap().port();
5872        let client = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5873        let client_port = client.local_addr().unwrap().port();
5874        let config = dir.join("server.conf");
5875        std::fs::write(&config, format!("listen: 127.0.0.1:{client_port}\nhttp: 127.0.0.1:{port}\njetstream {{ store_dir: \"{}\" }}\n", dir.join("jetstream").display())).unwrap();
5876        let mut configured = http_fixture(&dir, &format!("http://127.0.0.1:{port}/healthz"), 1000);
5877        configured.program = PathBuf::from("nats-server");
5878        configured.args = vec!["-c".into(), config.to_string_lossy().into_owned()];
5879        drop(monitor);
5880        drop(client);
5881        let module =
5882            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5883                .supervise_configured_with_health(
5884                    configured.module_spec(),
5885                    true,
5886                    configured.health,
5887                    configured.drain_timeout_ms,
5888                    configured.restart,
5889                )
5890                .unwrap();
5891        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5892        module.drain().await.unwrap();
5893    }
5894
5895    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5896    async fn configuration_update_does_not_replace_captured_running_process_facts() {
5897        let supervisor =
5898            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5899        let snapshot = stale_process_snapshot(ModuleState::Running, true);
5900        let initial = ModuleSpec {
5901            module_id: "rescan-preserves-spawn-facts".to_string(),
5902            program: PathBuf::from("/spawned/module"),
5903            args: Vec::new(),
5904            env: Vec::new(),
5905            reserved: false,
5906            reserved_prefixes: Vec::new(),
5907            protocol: ModuleProtocol::Subc,
5908            overlap: Default::default(),
5909        };
5910        let module = supervisor.supervised_module(
5911            initial.clone(),
5912            supervisor.runtime_config(),
5913            snapshot,
5914            None,
5915        );
5916        let before = module.status().unwrap();
5917        let mut replacement = initial;
5918        replacement.program = PathBuf::from("/rescanned/replacement-module");
5919
5920        module
5921            .update_configuration(replacement, HealthConfig::default(), None)
5922            .await
5923            .unwrap();
5924
5925        let after = module.status().unwrap();
5926        assert_eq!(after.pid, before.pid);
5927        assert_eq!(after.spawned_at_ms, before.spawned_at_ms);
5928        assert_eq!(after.spawned_from, before.spawned_from);
5929        drop(module);
5930    }
5931}
5932
5933fn unix_ms_now() -> u64 {
5934    SystemTime::now()
5935        .duration_since(UNIX_EPOCH)
5936        .map(|duration| duration.as_millis().min(u128::from(u64::MAX)) as u64)
5937        .unwrap_or(0)
5938}
5939
5940async fn supervise_loop(
5941    mut spec: ModuleSpec,
5942    mut runtime: SupervisorRuntimeConfig,
5943    registry: Arc<Registry>,
5944    process_liveness: Arc<SupervisorProcessLiveness>,
5945    snapshot: SharedSnapshot,
5946    mut child: Option<SupervisedChild>,
5947    mut commands: mpsc::Receiver<SupervisorCommand>,
5948) {
5949    let mut health_probe = HealthProbeRuntime::default();
5950    // All restart backoffs run here, including health and operator requests.
5951    // While one is pending the loop serves commands, so disable or drain can
5952    // cancel the replacement without spawning a process just to stop it.
5953    let mut pending_respawn: Option<PendingRespawn> = None;
5954    // Commands a swap handed back to run next (see `swap::SwapEnd`). Served
5955    // before anything else so a stop that interrupted a swap runs at once.
5956    let mut requeued: VecDeque<SupervisorCommand> = VecDeque::new();
5957    loop {
5958        #[cfg(target_os = "macos")]
5959        if let Some(active) = child.as_mut() {
5960            active.confirm_privacy_exec().await;
5961        }
5962        if let Some(scheduled) = runtime
5963            .scheduled_respawn
5964            .lock()
5965            .unwrap_or_else(|p| p.into_inner())
5966            .take()
5967        {
5968            pending_respawn = Some(scheduled);
5969        }
5970        if pending_respawn.is_some() && (child.is_some() || !respawn_still_pending(&snapshot)) {
5971            pending_respawn = None;
5972            cancel_deferred_reload(
5973                &runtime,
5974                &spec.module_id,
5975                "respawn cancelled by a supervisor command",
5976            );
5977        }
5978        if child.is_none() && pending_respawn.is_none() {
5979            cancel_deferred_reload(
5980                &runtime,
5981                &spec.module_id,
5982                "respawn cancelled before a replacement was spawned",
5983            );
5984            let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
5985                state.respawn_pending = false;
5986                state.coalesced_restart_pending = false;
5987                if matches!(
5988                    state.state,
5989                    ModuleState::Restarting
5990                        | ModuleState::Starting
5991                        | ModuleState::Draining
5992                        | ModuleState::Unresponsive
5993                ) {
5994                    error!(module_id = %spec.module_id, state = ?state.state,
5995                        "supervision operation ended without a child or pending respawn; marking failed so start can retry");
5996                    state.state = ModuleState::Failed;
5997                    clear_current_process_facts(state);
5998                }
5999            });
6000        }
6001        if let Some(command) = requeued.pop_front() {
6002            if !handle_supervisor_command(
6003                command,
6004                &mut spec,
6005                &mut runtime,
6006                &registry,
6007                &process_liveness,
6008                &snapshot,
6009                &mut child,
6010                &mut commands,
6011                &mut requeued,
6012            )
6013            .await
6014            {
6015                return;
6016            }
6017            if child.is_some() || !respawn_still_pending(&snapshot) {
6018                pending_respawn = None;
6019                let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
6020                    state.respawn_pending = false
6021                });
6022            }
6023            continue;
6024        }
6025        if child.is_some() {
6026            health_probe.refresh_registration(&spec, &runtime, &registry, &snapshot);
6027            let probe_sleep = sleep(health_probe.wake_after());
6028            tokio::pin!(probe_sleep);
6029            let active_child = child.as_mut().expect("child checked above");
6030            tokio::select! {
6031                wait_result = active_child.wait() => {
6032                    // Every arm below that gives up on the CHILD must keep the
6033                    // supervision task itself alive (child = None, loop
6034                    // continues into command-serving mode). Returning here
6035                    // closes the command channel, which makes the module
6036                    // permanently unrestartable in-band: a clean child exit
6037                    // of an enabled module once wedged the fleet this way
6038                    // ('supervisor command channel is closed') and required a
6039                    // full daemon restart to recover.
6040                    let exit_report = match wait_result {
6041                        Ok(status) => classify_reaped_child_exit(&snapshot, active_child, &status),
6042                        Err(err) => {
6043                            active_child.drain_stderr(&spec.module_id).await;
6044                            fail_snapshot(&snapshot, Some(&spec.module_id), None);
6045                            // Every other exit path (on_child_exit's Clean/Crash arms,
6046                            // the reload-registration-failure path) records a terminal
6047                            // before moving on. Without one here, a module whose wait()
6048                            // itself errored (e.g. already reaped) leaves no terminal
6049                            // record at all -- an empty ring reads as "nothing died".
6050                            record_wait_error_terminal(
6051                                &spec.module_id,
6052                                &runtime.terminal_ring,
6053                                &runtime.spawn_events,
6054                            );
6055                            untrack_if_registration_released(
6056                                &process_liveness,
6057                                &registry,
6058                                &spec.module_id,
6059                                &snapshot,
6060                            );
6061                            error!(module_id = %spec.module_id, error = %err, "failed to wait for supervised module");
6062                            child = None;
6063                            continue;
6064                        }
6065                    };
6066                    active_child.drain_stderr(&spec.module_id).await;
6067
6068                    let next = on_child_exit(
6069                        &spec,
6070                        runtime.restart_policy,
6071                        &registry,
6072                        &snapshot,
6073                        &runtime.terminal_ring,
6074                        &runtime.spawn_events,
6075                        &runtime.child_roster,
6076                        exit_report,
6077                    ).await;
6078                    // The exit is recorded, so a daemon shutdown may stop
6079                    // waiting for this child (see `SupervisedChild::wait`).
6080                    active_child.release_roster();
6081                    match next {
6082                        NextAction::Stop { registration_released } => {
6083                            if registration_released {
6084                                process_liveness.untrack_if_current(&spec.module_id, &snapshot);
6085                            }
6086                            child = None;
6087                        }
6088                        NextAction::Restart { schedule } => {
6089                            let delay = schedule.map_or(
6090                                runtime.restart_policy.delay_for_restart(0),
6091                                |schedule| schedule.delay,
6092                            );
6093                            if let Some(schedule) = schedule {
6094                                log_crash_respawn(&spec.module_id, schedule);
6095                            }
6096                            // The exited child is fully recorded at this point,
6097                            // so release it and count the backoff down in the
6098                            // command-serving branch below rather than sleeping
6099                            // here: commands cannot be received from inside this
6100                            // select arm, and an operator disable or drain that
6101                            // arrives during the backoff must cancel the pending
6102                            // respawn instead of waiting for it to spawn first.
6103                            child = None;
6104                            pending_respawn = Some(PendingRespawn { deadline: Instant::now() + delay, kind: RespawnKind::Spawn });
6105                             let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = true);
6106                        }
6107                    }
6108                }
6109                command = commands.recv() => {
6110                    let Some(command) = command else {
6111                        return;
6112                    };
6113                    if !handle_supervisor_command(
6114                        command,
6115                        &mut spec,
6116                        &mut runtime,
6117                        &registry,
6118                        &process_liveness,
6119                        &snapshot,
6120                        &mut child,
6121                        &mut commands,
6122                        &mut requeued,
6123                    ).await {
6124                        return;
6125                    }
6126                }
6127                _ = &mut probe_sleep => {
6128                    if health_probe.due() {
6129                        run_health_probe_cycle(
6130                            &spec,
6131                            &runtime,
6132                            &registry,
6133                            &process_liveness,
6134                            &snapshot,
6135                            &mut child,
6136                        ).await;
6137                        if child.is_some() {
6138                            health_probe.schedule_next(&spec, runtime.health.cadence);
6139                        }
6140                    }
6141                }
6142            }
6143        } else if let Some(pending) = pending_respawn {
6144            tokio::select! {
6145                _ = sleep_until(pending.deadline) => {
6146                    pending_respawn = None;
6147                let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = false);
6148                    // A command handled below while the backoff elapsed may
6149                    // have stopped the module; never respawn past an operator's
6150                    // disable or drain.
6151                    if !respawn_still_pending(&snapshot) {
6152                        continue;
6153                    }
6154                    // The daemon began shutting down during the backoff: the
6155                    // spawn would be refused anyway, and refusing it here
6156                    // leaves the module stopped instead of reporting a
6157                    // failed restart.
6158                    if runtime.child_roster.is_closed() {
6159                        let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
6160                            state.state = ModuleState::Stopped;
6161                        });
6162                        debug!(module_id = %spec.module_id, "crash respawn cancelled by daemon shutdown");
6163                        continue;
6164                    }
6165                    if let Err(err) = release_dead_registration(
6166                        &registry,
6167                        runtime.forwarding.as_deref(),
6168                        &snapshot,
6169                        &spec.module_id,
6170                    ).await {
6171                        fail_snapshot(&snapshot, Some(&spec.module_id), None);
6172                        error!(module_id = %spec.module_id, error = %err, "registration did not release before restart");
6173                        continue;
6174                    }
6175
6176                    if matches!(pending.kind, RespawnKind::Reload) {
6177                        let reply = runtime.deferred_reload_reply.lock().unwrap_or_else(|p| p.into_inner()).take();
6178                        let result = finish_reload_child(&spec, &runtime, &registry, &process_liveness, &snapshot, &mut child).await;
6179                        if let Some(reply) = reply { let _ = reply.send(result); }
6180                        continue;
6181                    }
6182                    process_liveness.track(spec.module_id.clone(), Arc::clone(&snapshot));
6183                    match spawn_and_mark_running(&spec, &runtime, &snapshot) {
6184                        Ok(next_child) => {
6185                            child = Some(next_child);
6186                            debug!(module_id = %spec.module_id, "supervised module restarted after crash");
6187                        }
6188                        Err(err) => {
6189                            fail_snapshot(&snapshot, Some(&spec.module_id), None);
6190                            process_liveness.untrack_if_current(&spec.module_id, &snapshot);
6191                            error!(module_id = %spec.module_id, error = %err, "failed to restart supervised module");
6192                        }
6193                    }
6194                }
6195                command = commands.recv() => {
6196                    let Some(command) = command else {
6197                        return;
6198                    };
6199                    if !handle_supervisor_command(
6200                        command,
6201                        &mut spec,
6202                        &mut runtime,
6203                        &registry,
6204                        &process_liveness,
6205                        &snapshot,
6206                        &mut child,
6207                        &mut commands,
6208                        &mut requeued,
6209                    ).await {
6210                        return;
6211                    }
6212                    // Reconcile the pending respawn with what the command did:
6213                    // a start may already have spawned a fresh child,
6214                    // while a disable or drain moved the snapshot out of the
6215                    // state the respawn was counting down from.
6216                    if child.is_some() || !respawn_still_pending(&snapshot) {
6217                        pending_respawn = None;
6218                let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = false);
6219                    }
6220                }
6221            }
6222        } else {
6223            let Some(command) = commands.recv().await else {
6224                return;
6225            };
6226            if !handle_supervisor_command(
6227                command,
6228                &mut spec,
6229                &mut runtime,
6230                &registry,
6231                &process_liveness,
6232                &snapshot,
6233                &mut child,
6234                &mut commands,
6235                &mut requeued,
6236            )
6237            .await
6238            {
6239                return;
6240            }
6241        }
6242    }
6243}
6244
6245fn log_crash_respawn(module_id: &str, schedule: CrashRestartSchedule) {
6246    info!(
6247        module_id,
6248        restart_in_window = schedule.restart_in_window,
6249        delay_ms = schedule.delay.as_millis() as u64,
6250        "respawning after crash"
6251    );
6252}
6253
6254/// Whether the respawn a backoff was counting down to is still wanted. A
6255/// disable or drain handled while the backoff elapsed moves the snapshot out
6256/// of `Restarting`, and the operator's stop must win over the pending respawn,
6257/// so every sleep-then-spawn path re-validates against the live snapshot
6258/// instead of assuming the state it left behind still holds.
6259fn respawn_still_pending(snapshot: &SharedSnapshot) -> bool {
6260    matches!(
6261        lock_snapshot(snapshot),
6262        Ok(state) if state.enabled && state.state == ModuleState::Restarting
6263    )
6264}
6265
6266enum NextAction {
6267    Stop {
6268        registration_released: bool,
6269    },
6270    Restart {
6271        schedule: Option<CrashRestartSchedule>,
6272    },
6273}
6274
6275#[allow(clippy::too_many_arguments)]
6276async fn handle_supervisor_command(
6277    command: SupervisorCommand,
6278    spec: &mut ModuleSpec,
6279    runtime: &mut SupervisorRuntimeConfig,
6280    registry: &Arc<Registry>,
6281    process_liveness: &SupervisorProcessLiveness,
6282    snapshot: &SharedSnapshot,
6283    child: &mut Option<SupervisedChild>,
6284    commands: &mut mpsc::Receiver<SupervisorCommand>,
6285    requeued: &mut VecDeque<SupervisorCommand>,
6286) -> bool {
6287    match command {
6288        SupervisorCommand::Drain { reply } => {
6289            // A plain stop runs no forwarding drain, so nothing reaches the
6290            // module over its connection before the wait: ask by signal.
6291            let result = drain_optional_child(
6292                &spec.module_id,
6293                spec.protocol,
6294                StopNotice::NotSent,
6295                registry,
6296                runtime.forwarding.as_deref(),
6297                snapshot,
6298                &runtime.terminal_ring,
6299                &runtime.spawn_events,
6300                child,
6301                runtime.drain_timeout,
6302                ModuleState::Stopped,
6303                None,
6304            )
6305            .await;
6306            let registration_released = result.is_ok();
6307            let _ = reply.send(result);
6308            if registration_released {
6309                process_liveness.untrack_if_current(&spec.module_id, snapshot);
6310            }
6311            false
6312        }
6313        SupervisorCommand::Retire { reply } => {
6314            let result = async {
6315                let stop_notice = begin_forwarding_drain_if_configured(
6316                    spec,
6317                    runtime,
6318                    registry,
6319                    snapshot,
6320                    None,
6321                    RouteCloseReason::Disable,
6322                )
6323                .await?;
6324                drain_optional_child(
6325                    &spec.module_id,
6326                    spec.protocol,
6327                    stop_notice,
6328                    registry,
6329                    runtime.forwarding.as_deref(),
6330                    snapshot,
6331                    &runtime.terminal_ring,
6332                    &runtime.spawn_events,
6333                    child,
6334                    runtime.drain_timeout,
6335                    ModuleState::Stopped,
6336                    None,
6337                )
6338                .await
6339            }
6340            .await;
6341            let registration_released = result.is_ok();
6342            let _ = reply.send(result);
6343            if registration_released {
6344                process_liveness.untrack_if_current(&spec.module_id, snapshot);
6345            }
6346            false
6347        }
6348        SupervisorCommand::Restart {
6349            drain_timeout_ms,
6350            received_at_generation,
6351            queued_at,
6352            reply,
6353        } => {
6354            // Without this line a restart that waited in the queue (behind a
6355            // health probe cycle or another command) was invisible: the log
6356            // showed only the drain timing out, minutes after the operator's call.
6357            info!(
6358                module_id = %spec.module_id,
6359                queued_ms = u64::try_from(queued_at.elapsed().as_millis()).unwrap_or(u64::MAX),
6360                "restart command dequeued"
6361            );
6362            // ACK AT INITIATION, not completion. The blocking form deadlocked any
6363            // caller whose own request lane rides the module being restarted: the
6364            // caller's in-flight request keeps the drain from quiescing, the drain
6365            // keeps the restart from completing, and the completion keeps the reply
6366            // from releasing the caller — so the drain always timed out and cut the
6367            // initiator with a GOODBYE, even on a healthy module. Replying once the
6368            // restart is validated lets a self-lane caller settle, which is exactly
6369            // what makes the drain succeed. Completion is observable via
6370            // supervisor.list / module status; a post-ack failure lands the module
6371            // in a visible terminal state below rather than in a reply nobody can
6372            // receive.
6373            let validation = match lock_snapshot(snapshot) {
6374                Ok(state) if !state.enabled => Err(SuperviseError::Disabled {
6375                    module_id: spec.module_id.clone(),
6376                }),
6377                Ok(_) => Ok(()),
6378                Err(err) => Err(err),
6379            };
6380            let initiated = validation.is_ok();
6381            let _ = reply.send(validation);
6382            // A restart asks for a fresh process. Commands run one at a time,
6383            // so a restart queued behind another restart (two operator calls
6384            // in quick succession) is dequeued the moment the first one has
6385            // spawned its replacement -- before that process has sent HELLO.
6386            // Running it would drain and kill the process the first restart
6387            // just produced, which is the opposite of what both callers asked
6388            // for. If a process spawned after this request was received is
6389            // still supervised, the request is already satisfied. Not when the
6390            // configuration changed since that spawn: then the newer process
6391            // predates the spec this restart may exist to apply.
6392            let satisfied_by_generation = if initiated && child.is_some() {
6393                lock_snapshot(snapshot).ok().and_then(|state| {
6394                    (state.spawn_generation > received_at_generation
6395                        && !state.configuration_updated_since_spawn)
6396                        .then_some(state.spawn_generation)
6397                })
6398            } else {
6399                None
6400            };
6401            let satisfied_by_pending = initiated
6402                && child.is_none()
6403                && lock_snapshot(snapshot).ok().is_some_and(|mut state| {
6404                    let pending = state.respawn_pending && !state.configuration_updated_since_spawn;
6405                    if pending {
6406                        state.coalesced_restart_pending = true;
6407                    }
6408                    pending
6409                });
6410            if satisfied_by_pending {
6411                debug!(module_id = %spec.module_id, "restart coalesced into the pending replacement");
6412            } else if let Some(generation) = satisfied_by_generation {
6413                info!(
6414                    module_id = %spec.module_id,
6415                    received_at_generation,
6416                    "restart already satisfied by generation {generation}; not restarting again"
6417                );
6418            } else if initiated {
6419                // Precedence: this restart's operator override, else the module's
6420                // configured budget (already resolved into the runtime).
6421                let drain_timeout = drain_timeout_ms
6422                    .map(Duration::from_millis)
6423                    .unwrap_or(runtime.drain_timeout);
6424                if let Err(err) = restart_child(
6425                    spec,
6426                    runtime,
6427                    registry,
6428                    process_liveness,
6429                    snapshot,
6430                    child,
6431                    drain_timeout,
6432                )
6433                .await
6434                {
6435                    warn!(
6436                        module_id = %spec.module_id,
6437                        error = %err,
6438                        "operator restart failed after initiation ack; module state carries the outcome"
6439                    );
6440                    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6441                        state.state = ModuleState::Failed;
6442                        clear_current_process_facts(state);
6443                    });
6444                }
6445            }
6446            true
6447        }
6448        SupervisorCommand::Reload { reply } => {
6449            let result =
6450                reload_child(spec, runtime, registry, process_liveness, snapshot, child).await;
6451            if result.is_ok()
6452                && runtime
6453                    .scheduled_respawn
6454                    .lock()
6455                    .unwrap_or_else(|p| p.into_inner())
6456                    .as_ref()
6457                    .is_some_and(|pending| matches!(pending.kind, RespawnKind::Reload))
6458            {
6459                *runtime
6460                    .deferred_reload_reply
6461                    .lock()
6462                    .unwrap_or_else(|p| p.into_inner()) = Some(reply);
6463            } else {
6464                let _ = reply.send(result);
6465            }
6466            true
6467        }
6468        SupervisorCommand::SetEnabled { enabled, reply } => {
6469            let result = set_child_enabled(
6470                spec,
6471                runtime,
6472                registry,
6473                process_liveness,
6474                snapshot,
6475                child,
6476                enabled,
6477            )
6478            .await;
6479            let _ = reply.send(result);
6480            true
6481        }
6482        SupervisorCommand::UpdateConfiguration {
6483            spec: next_spec,
6484            health,
6485            drain_timeout_ms,
6486            reply,
6487        } => {
6488            if let Some(handle) = &runtime.supervisor_handle {
6489                handle.apply_identity_configuration(&next_spec);
6490            }
6491            *spec = next_spec;
6492            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6493                state.configuration_updated_since_spawn = true;
6494            });
6495            let health_changed = runtime.health != health;
6496            runtime.health = health;
6497            // Reset the cadence and old endpoint's failure streak on a live
6498            // health-policy change rather than waiting for its old deadline.
6499            if health_changed {
6500                let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6501                    state.health = ModuleHealthStatus::default();
6502                });
6503            }
6504            runtime.drain_timeout = drain_timeout_ms
6505                .map(Duration::from_millis)
6506                .unwrap_or(runtime.default_drain_timeout);
6507            *runtime
6508                .effective_drain_timeout
6509                .lock()
6510                .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
6511            let _ = reply.send(());
6512            true
6513        }
6514        SupervisorCommand::Swap {
6515            ready_timeout,
6516            reply,
6517        } => {
6518            let end = swap::run_swap(
6519                spec,
6520                runtime,
6521                registry,
6522                process_liveness,
6523                snapshot,
6524                child,
6525                commands,
6526                ready_timeout.unwrap_or(DEFAULT_SWAP_READY_TIMEOUT),
6527                reply,
6528            )
6529            .await;
6530            requeued.extend(end.requeue);
6531            true
6532        }
6533    }
6534}
6535
6536async fn restart_child(
6537    spec: &ModuleSpec,
6538    runtime: &SupervisorRuntimeConfig,
6539    registry: &Registry,
6540    process_liveness: &SupervisorProcessLiveness,
6541    snapshot: &SharedSnapshot,
6542    child: &mut Option<SupervisedChild>,
6543    drain_timeout: Duration,
6544) -> Result<(), SuperviseError> {
6545    // Restart cycles a running module; it must not silently start a disabled one.
6546    if !lock_snapshot(snapshot)?.enabled {
6547        return Err(SuperviseError::Disabled {
6548            module_id: spec.module_id.clone(),
6549        });
6550    }
6551    let stop_notice = begin_forwarding_drain_with_timeout(
6552        spec,
6553        runtime,
6554        registry,
6555        snapshot,
6556        None,
6557        RouteCloseReason::Restart,
6558        drain_timeout,
6559    )
6560    .await?;
6561
6562    if child.is_some() {
6563        drain_optional_child(
6564            &spec.module_id,
6565            spec.protocol,
6566            stop_notice,
6567            registry,
6568            runtime.forwarding.as_deref(),
6569            snapshot,
6570            &runtime.terminal_ring,
6571            &runtime.spawn_events,
6572            child,
6573            drain_timeout,
6574            ModuleState::Restarting,
6575            Some(true),
6576        )
6577        .await?;
6578    } else {
6579        update_snapshot(snapshot, Some(&spec.module_id), |state| {
6580            state.enabled = true;
6581            state.state = ModuleState::Restarting;
6582            clear_current_process_facts(state);
6583        })?;
6584        release_dead_registration(
6585            registry,
6586            runtime.forwarding.as_deref(),
6587            snapshot,
6588            &spec.module_id,
6589        )
6590        .await?;
6591    }
6592
6593    reset_restart_count(snapshot, &spec.module_id)?;
6594    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6595    schedule_respawn(
6596        runtime,
6597        snapshot,
6598        &spec.module_id,
6599        runtime.restart_policy.backoff,
6600        RespawnKind::Spawn,
6601    )
6602}
6603
6604async fn reload_child(
6605    spec: &ModuleSpec,
6606    runtime: &SupervisorRuntimeConfig,
6607    registry: &Registry,
6608    process_liveness: &SupervisorProcessLiveness,
6609    snapshot: &SharedSnapshot,
6610    child: &mut Option<SupervisedChild>,
6611) -> Result<(), SuperviseError> {
6612    // Reload cycles a running module; it must not silently start a disabled one.
6613    if !lock_snapshot(snapshot)?.enabled {
6614        return Err(SuperviseError::Disabled {
6615            module_id: spec.module_id.clone(),
6616        });
6617    }
6618    let stop_notice = begin_forwarding_drain(
6619        spec,
6620        runtime,
6621        registry,
6622        snapshot,
6623        Some(true),
6624        RouteCloseReason::Reload,
6625    )
6626    .await?;
6627
6628    if child.is_some() {
6629        drain_optional_child(
6630            &spec.module_id,
6631            spec.protocol,
6632            stop_notice,
6633            registry,
6634            runtime.forwarding.as_deref(),
6635            snapshot,
6636            &runtime.terminal_ring,
6637            &runtime.spawn_events,
6638            child,
6639            runtime.drain_timeout,
6640            ModuleState::Restarting,
6641            Some(true),
6642        )
6643        .await?;
6644    } else {
6645        update_snapshot(snapshot, Some(&spec.module_id), |state| {
6646            state.enabled = true;
6647            state.state = ModuleState::Restarting;
6648            clear_current_process_facts(state);
6649        })?;
6650        release_dead_registration(
6651            registry,
6652            runtime.forwarding.as_deref(),
6653            snapshot,
6654            &spec.module_id,
6655        )
6656        .await?;
6657    }
6658
6659    reset_restart_count(snapshot, &spec.module_id)?;
6660    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6661    schedule_respawn(
6662        runtime,
6663        snapshot,
6664        &spec.module_id,
6665        runtime.restart_policy.backoff,
6666        RespawnKind::Reload,
6667    )
6668}
6669
6670async fn finish_reload_child(
6671    spec: &ModuleSpec,
6672    runtime: &SupervisorRuntimeConfig,
6673    registry: &Registry,
6674    process_liveness: &SupervisorProcessLiveness,
6675    snapshot: &SharedSnapshot,
6676    child: &mut Option<SupervisedChild>,
6677) -> Result<(), SuperviseError> {
6678    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6679    let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
6680        Ok(next_child) => next_child,
6681        Err(err) => {
6682            return handle_reload_spawn_failure(
6683                spec,
6684                runtime,
6685                process_liveness,
6686                snapshot,
6687                child,
6688                format!("new child failed to spawn: {err}"),
6689            )
6690            .await;
6691        }
6692    };
6693    *child = Some(next_child);
6694
6695    let wait_outcome = {
6696        let active_child = child.as_mut().expect("new reload child was just stored");
6697        wait_for_registration_after_reload(
6698            registry,
6699            &spec.module_id,
6700            snapshot,
6701            active_child,
6702            REGISTRY_RELEASE_TIMEOUT,
6703        )
6704        .await?
6705    };
6706
6707    match wait_outcome {
6708        RegistrationWaitOutcome::Registered => {
6709            debug!(module_id = %spec.module_id, "supervised module reloaded and registered");
6710            Ok(())
6711        }
6712        RegistrationWaitOutcome::Exited(exit_report) => {
6713            if let Some(active_child) = child.as_mut() {
6714                active_child.drain_stderr(&spec.module_id).await;
6715            }
6716            // Keep the reaped child's roster guard until its terminal is written.
6717            // Shutdown waits on that guard, not on the child Option used for respawn.
6718            let mut exited_child = child.take().expect("exited reload child is still stored");
6719            #[cfg(test)]
6720            if let Some(gate) = &runtime.test_reload_exit_record_gate {
6721                gate.reached.notify_one();
6722                gate.resume.notified().await;
6723            }
6724            let result = handle_reload_child_registration_failure(
6725                spec,
6726                runtime,
6727                registry,
6728                process_liveness,
6729                snapshot,
6730                child,
6731                ReloadRegistrationFailure {
6732                    exit_report: registration_failure_exit_report(exit_report),
6733                    reason: exited_child
6734                        .spawn_failure
6735                        .clone()
6736                        .unwrap_or_else(|| "new child exited before registering".to_string()),
6737                },
6738            )
6739            .await;
6740            exited_child.release_roster();
6741            result
6742        }
6743        RegistrationWaitOutcome::TimedOut => {
6744            let mut timed_out_child = child
6745                .take()
6746                .expect("timed-out reload child is still running");
6747            timed_out_child
6748                .start_kill()
6749                .map_err(|source| SuperviseError::Kill {
6750                    module_id: spec.module_id.clone(),
6751                    source,
6752                })?;
6753            let status = timed_out_child
6754                .wait()
6755                .await
6756                .map_err(|source| SuperviseError::Wait {
6757                    module_id: spec.module_id.clone(),
6758                    source,
6759                })?;
6760            timed_out_child.drain_stderr(&spec.module_id).await;
6761            handle_reload_child_registration_failure(
6762                spec,
6763                runtime,
6764                registry,
6765                process_liveness,
6766                snapshot,
6767                child,
6768                ReloadRegistrationFailure {
6769                    exit_report: registration_failure_exit_report(classify_reaped_child_exit(
6770                        snapshot,
6771                        &timed_out_child,
6772                        &status,
6773                    )),
6774                    reason: format!(
6775                        "new child did not register within {:?}",
6776                        REGISTRY_RELEASE_TIMEOUT
6777                    ),
6778                },
6779            )
6780            .await
6781        }
6782    }
6783}
6784
6785async fn set_child_enabled(
6786    spec: &ModuleSpec,
6787    runtime: &SupervisorRuntimeConfig,
6788    registry: &Registry,
6789    process_liveness: &SupervisorProcessLiveness,
6790    snapshot: &SharedSnapshot,
6791    child: &mut Option<SupervisedChild>,
6792    enabled: bool,
6793) -> Result<bool, SuperviseError> {
6794    let (current_enabled, current_state, respawn_pending) = {
6795        let state = lock_snapshot(snapshot)?;
6796        (state.enabled, state.state, state.respawn_pending)
6797    };
6798    // `start` (enable on an already-enabled module) heals TERMINAL states instead
6799    // of no-op'ing: a module whose restart budget exhausted (Failed) or that exited
6800    // clean (Stopped) has no live process and no other in-band recovery — the
6801    // operator's start is the explicit recovery act and resets the budget. Without
6802    // this arm the only revival was subc-probe --supervisor-restart in a terminal,
6803    // which the 2026-07-14 aft outage proved is a trap when the failed module is
6804    // the one providing every agent's shell.
6805    let revive_terminal = enabled
6806        && current_enabled
6807        && child.is_none()
6808        && (matches!(current_state, ModuleState::Failed | ModuleState::Stopped)
6809            || (current_state == ModuleState::Restarting && !respawn_pending));
6810    if current_enabled == enabled && !revive_terminal {
6811        return Ok(false);
6812    }
6813
6814    if enabled {
6815        update_snapshot(snapshot, Some(&spec.module_id), |state| {
6816            state.enabled = true;
6817            state.state = ModuleState::Starting;
6818            clear_current_process_facts(state);
6819        })?;
6820        #[cfg(test)]
6821        if runtime.test_seed_stale_facts_before_enable_spawn {
6822            update_snapshot(snapshot, Some(&spec.module_id), |state| {
6823                state.process_alive = true;
6824                state.pid = Some(41);
6825                state.spawned_at_ms = Some(42);
6826                state.spawned_from = Some(PathBuf::from("/spawned/module"));
6827                state.spawned_file_identity = Some(SpawnedFileIdentity {
6828                    device: 43,
6829                    inode: 44,
6830                });
6831            })?;
6832        }
6833        release_dead_registration(
6834            registry,
6835            runtime.forwarding.as_deref(),
6836            snapshot,
6837            &spec.module_id,
6838        )
6839        .await?;
6840        reset_restart_count(snapshot, &spec.module_id)?;
6841        process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6842        let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
6843            Ok(next_child) => next_child,
6844            Err(err) => {
6845                if let Err(state_err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6846                    state.state = ModuleState::Failed;
6847                    clear_current_process_facts(state);
6848                }) {
6849                    error!(module_id = %spec.module_id, error = %state_err, "failed to record enable spawn failure");
6850                }
6851                process_liveness.untrack_if_current(&spec.module_id, snapshot);
6852                return Err(err);
6853            }
6854        };
6855        *child = Some(next_child);
6856        debug!(module_id = %spec.module_id, "supervised module enabled");
6857        Ok(true)
6858    } else {
6859        let stop_notice = begin_forwarding_drain_if_configured(
6860            spec,
6861            runtime,
6862            registry,
6863            snapshot,
6864            Some(false),
6865            RouteCloseReason::Disable,
6866        )
6867        .await?;
6868        drain_optional_child(
6869            &spec.module_id,
6870            spec.protocol,
6871            stop_notice,
6872            registry,
6873            runtime.forwarding.as_deref(),
6874            snapshot,
6875            &runtime.terminal_ring,
6876            &runtime.spawn_events,
6877            child,
6878            runtime.drain_timeout,
6879            ModuleState::Disabled,
6880            Some(false),
6881        )
6882        .await?;
6883        debug!(module_id = %spec.module_id, "supervised module disabled");
6884        Ok(true)
6885    }
6886}
6887
6888#[allow(clippy::too_many_arguments)]
6889async fn on_child_exit(
6890    spec: &ModuleSpec,
6891    policy: RestartPolicy,
6892    registry: &Registry,
6893    snapshot: &SharedSnapshot,
6894    terminal_ring: &Arc<Mutex<TerminalRing>>,
6895    spawn_events: &SpawnEventFeed,
6896    roster: &ChildRoster,
6897    exit_report: ExitReport,
6898) -> NextAction {
6899    // Once the daemon has begun shutting down, no exit is a crash to recover
6900    // from: the module is exiting because the daemon is going away (EOF on its
6901    // connection, or a service manager signalling the whole cgroup). Record it
6902    // as such and never schedule a respawn, which would only start a process
6903    // for the shutdown to end again.
6904    if roster.is_closed() {
6905        return on_child_exit_during_daemon_shutdown(
6906            spec,
6907            registry,
6908            snapshot,
6909            terminal_ring,
6910            spawn_events,
6911            exit_report,
6912        )
6913        .await;
6914    }
6915    // Every stop the supervisor itself asks for (operator stop, disable,
6916    // restart, reload, swap, a health restart, a drain that runs out of budget)
6917    // takes the child out of the supervise loop and reaps it in
6918    // `drain_child_to_state`, and daemon shutdown is handled above. So an exit
6919    // that reaches this point was not requested by the daemon.
6920    //
6921    // For a subc-wire module a clean exit is still a stop: those modules are
6922    // written to re-raise SIGTERM, so a stray outside signal already reads as a
6923    // crash, and exiting 0 is a deliberate choice the module made. A
6924    // `protocol: "none"` module is a stock program we cannot change, and many
6925    // of them (nats-server among them) exit 0 on SIGTERM. Treating that as a
6926    // stop would leave the module down for good after any stray signal, so it
6927    // goes through the crash path instead: it spends restart budget, respawns
6928    // with the crash backoff, and ends `failed` when the budget runs out.
6929    let unrequested_clean_exit_of_protocol_none = exit_report.kind == ExitKind::Clean
6930        && running_protocol(spec, snapshot) == ModuleProtocol::None;
6931    match exit_report.kind {
6932        ExitKind::Clean if !unrequested_clean_exit_of_protocol_none => {
6933            info!(
6934                module_id = %spec.module_id,
6935                exit_code = ?exit_report.code,
6936                exit_signal = ?exit_report.signal,
6937                "supervised module exited cleanly"
6938            );
6939            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6940                state.state = ModuleState::Stopped;
6941                clear_current_process_facts(state);
6942                state.last_exit = Some(exit_report.clone());
6943            }) {
6944                error!(module_id = %spec.module_id, error = %err, "failed to record clean module exit");
6945            }
6946            record_terminal(
6947                &spec.module_id,
6948                terminal_ring,
6949                spawn_events,
6950                &exit_report,
6951                TerminalDisposition::Stopped,
6952            );
6953            let registration_released = match wait_for_registration_release(
6954                registry,
6955                &spec.module_id,
6956                REGISTRY_RELEASE_TIMEOUT,
6957            )
6958            .await
6959            {
6960                Ok(()) => true,
6961                Err(err) => {
6962                    warn!(module_id = %spec.module_id, error = %err, "registration still active after clean exit");
6963                    false
6964                }
6965            };
6966            NextAction::Stop {
6967                registration_released,
6968            }
6969        }
6970        ExitKind::Clean | ExitKind::Crash => {
6971            if unrequested_clean_exit_of_protocol_none {
6972                warn!(
6973                    module_id = %spec.module_id,
6974                    exit_code = ?exit_report.code,
6975                    exit_signal = ?exit_report.signal,
6976                    "protocol-none module exited cleanly without a stop request; handling it as a crash"
6977                );
6978            } else {
6979                warn!(
6980                    module_id = %spec.module_id,
6981                    exit_code = ?exit_report.code,
6982                    exit_signal = ?exit_report.signal,
6983                    "supervised module exited abnormally (crash)"
6984                );
6985            }
6986            let mut restart_schedule = None;
6987            let mut disposition = TerminalDisposition::Disabled;
6988            // Set only when the budget is what stopped the module, so the
6989            // terminal record says which limit was hit rather than leaving
6990            // `failed` to be read as "crashed once, badly".
6991            let mut disposition_detail = lock_snapshot(snapshot)
6992                .ok()
6993                .and_then(|mut state| state.spawn_failure.take());
6994            let now = Instant::now();
6995            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6996                clear_current_process_facts(state);
6997                state.last_exit = Some(exit_report.clone());
6998                if state.enabled {
6999                    if let Some(schedule) = state.next_crash_restart(&policy, now) {
7000                        state.state = ModuleState::Restarting;
7001                        restart_schedule = Some(schedule);
7002                        disposition = TerminalDisposition::Restarting;
7003                    } else {
7004                        disposition = TerminalDisposition::Failed;
7005                        let budget = policy.budget_exhausted_detail();
7006                        disposition_detail =
7007                            Some(disposition_detail.take().map_or_else(
7008                                || budget.clone(),
7009                                |cause| format!("{cause}; {budget}"),
7010                            ));
7011                    }
7012                } else {
7013                    state.state = ModuleState::Disabled;
7014                    disposition = TerminalDisposition::Disabled;
7015                }
7016            }) {
7017                error!(module_id = %spec.module_id, error = %err, "failed to record crashed module exit");
7018                return NextAction::Stop {
7019                    registration_released: false,
7020                };
7021            }
7022            if disposition == TerminalDisposition::Failed {
7023                // The window is in the message, not only in the fields: this line
7024                // is read in a scrollback where a bare `max_restarts=3` reads as a
7025                // lifetime cap and sends the operator looking for three crashes
7026                // that never happened together.
7027                error!(
7028                    module_id = %spec.module_id,
7029                    max_restarts = policy.max_restarts,
7030                    window_secs = policy.window.as_secs(),
7031                    "module stopped: {}",
7032                    policy.budget_exhausted_detail()
7033                );
7034            }
7035            let budget_exhausted = disposition == TerminalDisposition::Failed;
7036            let record_exit = || {
7037                record_terminal_with_detail(
7038                    &spec.module_id,
7039                    terminal_ring,
7040                    spawn_events,
7041                    &exit_report,
7042                    disposition,
7043                    disposition_detail,
7044                );
7045            };
7046            if budget_exhausted {
7047                // Publish Failed only after its terminal record is available.
7048                // Recording takes the event-feed lock, then ring -> journal
7049                // writer (with file I/O), all without the hot snapshot lock.
7050                // No lock is held when the final snapshot update runs, nor
7051                // across the registration-release await below. Commands and
7052                // health actions run on this same supervisor task, so none can
7053                // act on the old state during the write; process facts already
7054                // say the child is dead to concurrent liveness readers.
7055                // A journal error is retained in history, not returned. If
7056                // recording panics, still publish Failed before resuming the
7057                // original unwind rather than leaving a dead child Running.
7058                let recorded = std::panic::catch_unwind(std::panic::AssertUnwindSafe(record_exit));
7059                if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
7060                    state.state = ModuleState::Failed;
7061                }) {
7062                    error!(module_id = %spec.module_id, error = %err, "failed to publish exhausted restart budget");
7063                }
7064                if let Err(panic) = recorded {
7065                    std::panic::resume_unwind(panic);
7066                }
7067            } else {
7068                record_exit();
7069            }
7070
7071            if let Some(schedule) = restart_schedule {
7072                NextAction::Restart {
7073                    schedule: Some(schedule),
7074                }
7075            } else {
7076                let registration_released = match wait_for_registration_release(
7077                    registry,
7078                    &spec.module_id,
7079                    REGISTRY_RELEASE_TIMEOUT,
7080                )
7081                .await
7082                {
7083                    Ok(()) => true,
7084                    Err(err) => {
7085                        warn!(module_id = %spec.module_id, error = %err, "registration still active after failed module");
7086                        false
7087                    }
7088                };
7089                NextAction::Stop {
7090                    registration_released,
7091                }
7092            }
7093        }
7094        ExitKind::DeliberateSeverance => {
7095            warn!(
7096                module_id = %spec.module_id,
7097                exit_code = ?exit_report.code,
7098                exit_signal = ?exit_report.signal,
7099                "supervised module exited after deliberate connection severance"
7100            );
7101            let mut should_restart = false;
7102            let mut disposition = TerminalDisposition::Disabled;
7103            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
7104                clear_current_process_facts(state);
7105                state.last_exit = Some(exit_report.clone());
7106                state.lifetime_restarts += 1;
7107                if state.enabled {
7108                    state.state = ModuleState::Restarting;
7109                    should_restart = true;
7110                    disposition = TerminalDisposition::Restarting;
7111                } else {
7112                    state.state = ModuleState::Disabled;
7113                }
7114            }) {
7115                error!(module_id = %spec.module_id, error = %err, "failed to record deliberately severed module exit");
7116                return NextAction::Stop {
7117                    registration_released: false,
7118                };
7119            }
7120            record_terminal(
7121                &spec.module_id,
7122                terminal_ring,
7123                spawn_events,
7124                &exit_report,
7125                disposition,
7126            );
7127
7128            if should_restart {
7129                NextAction::Restart { schedule: None }
7130            } else {
7131                let registration_released = match wait_for_registration_release(
7132                    registry,
7133                    &spec.module_id,
7134                    REGISTRY_RELEASE_TIMEOUT,
7135                )
7136                .await
7137                {
7138                    Ok(()) => true,
7139                    Err(err) => {
7140                        warn!(module_id = %spec.module_id, error = %err, "registration still active after deliberately severed module exit");
7141                        false
7142                    }
7143                };
7144                NextAction::Stop {
7145                    registration_released,
7146                }
7147            }
7148        }
7149    }
7150}
7151
7152async fn on_child_exit_during_daemon_shutdown(
7153    spec: &ModuleSpec,
7154    registry: &Registry,
7155    snapshot: &SharedSnapshot,
7156    terminal_ring: &Arc<Mutex<TerminalRing>>,
7157    spawn_events: &SpawnEventFeed,
7158    exit_report: ExitReport,
7159) -> NextAction {
7160    info!(
7161        module_id = %spec.module_id,
7162        exit_code = ?exit_report.code,
7163        exit_signal = ?exit_report.signal,
7164        exit_kind = ?exit_report.kind,
7165        "supervised module exited during daemon shutdown; not restarting it"
7166    );
7167    if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
7168        state.state = ModuleState::Stopped;
7169        clear_current_process_facts(state);
7170        state.last_exit = Some(exit_report.clone());
7171    }) {
7172        error!(module_id = %spec.module_id, error = %err, "failed to record module exit during daemon shutdown");
7173    }
7174    record_terminal(
7175        &spec.module_id,
7176        terminal_ring,
7177        spawn_events,
7178        &exit_report,
7179        TerminalDisposition::DaemonShutdown,
7180    );
7181    let registration_released =
7182        wait_for_registration_release(registry, &spec.module_id, REGISTRY_RELEASE_TIMEOUT)
7183            .await
7184            .is_ok();
7185    NextAction::Stop {
7186        registration_released,
7187    }
7188}
7189
7190fn record_wait_error_terminal(
7191    module_id: &str,
7192    terminal_ring: &Arc<Mutex<TerminalRing>>,
7193    spawn_events: &SpawnEventFeed,
7194) {
7195    record_terminal(
7196        module_id,
7197        terminal_ring,
7198        spawn_events,
7199        &wait_error_exit_report(),
7200        TerminalDisposition::Failed,
7201    );
7202}
7203
7204fn record_terminal(
7205    module_id: &str,
7206    terminal_ring: &Arc<Mutex<TerminalRing>>,
7207    spawn_events: &SpawnEventFeed,
7208    exit_report: &ExitReport,
7209    disposition: TerminalDisposition,
7210) {
7211    record_terminal_with_detail(
7212        module_id,
7213        terminal_ring,
7214        spawn_events,
7215        exit_report,
7216        disposition,
7217        None,
7218    );
7219}
7220
7221/// The ring lock is held only to capture the read (see
7222/// `TerminalJournal::capture_read`), so this module's exits keep recording
7223/// while the journal files are read. Blocking: it reads files.
7224fn durable_terminal_history_of(
7225    terminal_ring: &Mutex<TerminalRing>,
7226    module_id: &str,
7227) -> subc_control::TerminalHistory {
7228    let read = terminal_ring
7229        .lock()
7230        .unwrap_or_else(|p| p.into_inner())
7231        .capture_durable_history();
7232    read.read(module_id)
7233}
7234
7235fn record_terminal_with_detail(
7236    module_id: &str,
7237    terminal_ring: &Arc<Mutex<TerminalRing>>,
7238    spawn_events: &SpawnEventFeed,
7239    exit_report: &ExitReport,
7240    disposition: TerminalDisposition,
7241    disposition_detail: Option<String>,
7242) {
7243    spawn_events.emit_exited(module_id, exit_report.code, exit_report.signal);
7244    let record = TerminalRecord {
7245        exit_code: exit_report.code,
7246        exit_signal: exit_report.signal,
7247        at_ms: exit_report.at_ms,
7248        disposition,
7249        exit_kind: exit_report.kind.into(),
7250        disposition_detail,
7251    };
7252    terminal_ring
7253        .lock()
7254        .unwrap_or_else(|poisoned| poisoned.into_inner())
7255        .record_exit(module_id, record);
7256}
7257
7258fn untrack_if_registration_released(
7259    process_liveness: &SupervisorProcessLiveness,
7260    registry: &Registry,
7261    module_id: &str,
7262    snapshot: &SharedSnapshot,
7263) {
7264    match registry.get_module(module_id) {
7265        Ok(None) => process_liveness.untrack_if_current(module_id, snapshot),
7266        Ok(Some(_)) => {}
7267        Err(err) => {
7268            warn!(module_id, error = %err, "could not determine whether supervisor liveness can be untracked");
7269        }
7270    }
7271}
7272
7273/// The child's environment plan: inherit the parent's, drop ambient `CK_LOG`,
7274/// then apply the module's configured entries minus daemon-private capture keys.
7275///
7276/// Separated from `spawn_child` only so it can be asserted without spawning a
7277/// process — a duplicate of this logic in a test would pass while the real one
7278/// drifted, which is the defect class this function exists to avoid.
7279/// The subc-wire half of a spawn: `--subc <connection file>` and the launch
7280/// nonce. A `protocol: "none"` module gets neither, because it cannot use
7281/// either and the argument would stop a stock binary from starting at all.
7282/// `SUBC_MODULE_ID` is set on every path since an unread variable is inert.
7283///
7284/// The plain-spawn form, kept for the tests that assert its plan; spawns go
7285/// through [`apply_wire_spawn_args_for_role`].
7286#[cfg(test)]
7287fn apply_wire_spawn_args(
7288    command: &mut Command,
7289    spec: &ModuleSpec,
7290    connection_file_path: Option<&std::path::Path>,
7291    handle: Option<&SupervisorHandle>,
7292) -> Result<Option<NonceHandoff>, SuperviseError> {
7293    apply_wire_spawn_args_for_role(
7294        command,
7295        spec,
7296        connection_file_path,
7297        handle,
7298        SpawnRole::Plain,
7299    )
7300}
7301
7302/// The read end of a spawn's launch-nonce pipe, prepared by
7303/// [`apply_wire_spawn_args_for_role`] and installed as the child's descriptor 3
7304/// by the last pre-exec step, just before `spawn()`. Windows has no descriptor
7305/// handoff and keeps only the environment copy.
7306#[cfg(unix)]
7307type NonceHandoff = subc_os::LaunchNonceHandoff;
7308#[cfg(not(unix))]
7309type NonceHandoff = std::convert::Infallible;
7310
7311/// Prepare wire identity for a plain spawn or a swap candidate.
7312///
7313/// A plain spawn replaces the module's recorded nonce. A swap candidate records
7314/// a separate candidate token so the still-serving incumbent and its consumers
7315/// keep their nonce. Both records are installed before the process exists, so
7316/// the child's initial HELLO registration cannot arrive ahead of its nonce.
7317///
7318/// On Unix the nonce is delivered only through a pipe. It is written into
7319/// a pipe whose read end the child gets as descriptor 3, named by
7320/// `SUBC_LAUNCH_NONCE_FD=3:<pipe inode>`: unlike the environment, another
7321/// process of the same user cannot read it with `ps eww`. That handoff is
7322/// returned rather than installed here, because installing it replaces
7323/// whatever the child has at descriptor 3 and so must be the last pre-exec
7324/// step, after the Linux cgroup placement that the caller registers later.
7325/// Windows retains the environment handoff until restricted handle inheritance
7326/// can be implemented outside std's process primitives.
7327fn apply_wire_spawn_args_for_role(
7328    command: &mut Command,
7329    spec: &ModuleSpec,
7330    connection_file_path: Option<&std::path::Path>,
7331    handle: Option<&SupervisorHandle>,
7332    role: SpawnRole,
7333) -> Result<Option<NonceHandoff>, SuperviseError> {
7334    command.env(SUBC_MODULE_ID_ENV, &spec.module_id);
7335    // SUBC_LAUNCH_NONCE_FD is removed for every child, `protocol: "none"`
7336    // included: a daemon started from a module's process tree inherits it,
7337    // and passing it on would point the child at a descriptor it does not
7338    // have.
7339    command.env_remove(subc_os::LAUNCH_NONCE_FD_ENV);
7340    // Remove inherited or configured copies too: withholding must mean absent.
7341    command.env_remove(SUBC_LAUNCH_NONCE_ENV);
7342    if spec.protocol == ModuleProtocol::None {
7343        return Ok(None);
7344    }
7345    if let Some(connection_file_path) = connection_file_path {
7346        command.arg(SUBC_ARG).arg(connection_file_path);
7347    }
7348
7349    // Every subc-wire spawn receives a fresh one-time launch nonce for consumer
7350    // route.open attestation. Reserved modules additionally use the same nonce
7351    // for HELLO id-squatting protection. A respawn rotates both records.
7352    let nonce = generate_launch_nonce()?;
7353    if let Some(handle) = handle {
7354        match role {
7355            SpawnRole::Plain => {
7356                handle.set_spawn_nonce(&spec.module_id, nonce.clone());
7357                if spec.reserved {
7358                    handle.set_reserved_nonce(&spec.module_id, nonce.clone());
7359                }
7360            }
7361            SpawnRole::SwapCandidate => handle.open_swap(&spec.module_id, nonce.clone()),
7362        }
7363    }
7364    #[cfg(unix)]
7365    let handoff = {
7366        let handoff =
7367            subc_os::LaunchNonceHandoff::new(&nonce).map_err(|source| SuperviseError::Spawn {
7368                program: spec.program.clone(),
7369                source,
7370                cgroup_path: None,
7371            })?;
7372        command.env(subc_os::LAUNCH_NONCE_FD_ENV, handoff.fd_env_value());
7373        Some(handoff)
7374    };
7375    #[cfg(not(unix))]
7376    let handoff = None;
7377    // Windows keeps the environment copy: std cannot restrict an inherited pipe
7378    // handle to this child without leaking it to concurrently spawned processes.
7379    #[cfg(not(unix))]
7380    command.env(SUBC_LAUNCH_NONCE_ENV, nonce);
7381    Ok(handoff)
7382}
7383
7384fn apply_child_env(command: &mut Command, spec: &ModuleSpec) {
7385    command.env_remove(CK_LOG_ENV);
7386    // The spawn role is the supervisor's to set, and only on a swap candidate
7387    // (see `apply_spawn_role`). Removing it here, rather than just not setting
7388    // it, is what makes it absent on a plain spawn: the daemon's own
7389    // environment could carry it, and so could a spec built outside daemon
7390    // config (config refuses it as an `env` key). A module reading it on a
7391    // plain restart would pick the long swap budget and leave callers waiting.
7392    command.env_remove(SUBC_SPAWN_ROLE_ENV);
7393    for (key, value) in &spec.env {
7394        // cortexkit-log currently exposes retention only as a Rust struct, not
7395        // environment names. These values are daemon-private sink metadata and
7396        // must never become a public child-process contract by being inherited.
7397        if matches!(
7398            key.as_str(),
7399            CAPTURE_MAX_FILE_MB_ENV | CAPTURE_KEEP_ENV | CAPTURE_MAX_AGE_DAYS_ENV
7400        ) || key == SUBC_SPAWN_ROLE_ENV
7401        {
7402            continue;
7403        }
7404        command.env(key, value);
7405    }
7406}
7407
7408/// Which slot a spawn fills: the module's ordinary one, or the candidate slot
7409/// of a blue/green swap.
7410#[derive(Debug, Clone, Copy, PartialEq, Eq)]
7411enum SpawnRole {
7412    Plain,
7413    SwapCandidate,
7414}
7415
7416/// Set the spawn role for a swap candidate. A plain spawn gets nothing here;
7417/// `apply_child_env` has already removed the variable for every spawn.
7418fn apply_spawn_role(command: &mut Command, role: SpawnRole) {
7419    if role == SpawnRole::SwapCandidate {
7420        command.env(SUBC_SPAWN_ROLE_ENV, SPAWN_ROLE_SWAP_CANDIDATE);
7421    }
7422}
7423
7424fn spawn_child(
7425    spec: &ModuleSpec,
7426    connection_file_path: Option<&std::path::Path>,
7427    handle: Option<&SupervisorHandle>,
7428    ring: &Arc<Mutex<StderrRing>>,
7429    capture_logs_dir: Option<&std::path::Path>,
7430    roster: &ChildRoster,
7431    #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
7432) -> Result<SupervisedChild, SuperviseError> {
7433    spawn_child_in_slot(
7434        spec,
7435        connection_file_path,
7436        handle,
7437        ring,
7438        capture_logs_dir,
7439        roster,
7440        #[cfg(target_os = "linux")]
7441        cgroup_placement,
7442        SpawnRole::Plain,
7443        false,
7444    )
7445}
7446
7447/// Spawn one process of `spec` into a slot.
7448///
7449/// `alternate_slot` picks the process's cgroup label on Linux (see `swap::cgroup_name`).
7450/// A swap candidate needs a different cgroup from the process it is replacing,
7451/// which is still alive: in the same cgroup the two would be one kill domain,
7452/// and killing a failed candidate could take the incumbent with it.
7453///
7454/// The stderr capture file is `<module_id>.stderr.log` for every process of
7455/// the module, whichever slot it is in, because that is the one file
7456/// `ck module logs` reads. During a swap's overlap both processes append to it;
7457/// the daemon writes whole lines, so the two interleave by line, which is also
7458/// the merged view an operator wants while a swap runs.
7459#[allow(clippy::too_many_arguments)]
7460fn spawn_child_in_slot(
7461    spec: &ModuleSpec,
7462    connection_file_path: Option<&std::path::Path>,
7463    handle: Option<&SupervisorHandle>,
7464    ring: &Arc<Mutex<StderrRing>>,
7465    capture_logs_dir: Option<&std::path::Path>,
7466    roster: &ChildRoster,
7467    #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
7468    role: SpawnRole,
7469    alternate_slot: bool,
7470) -> Result<SupervisedChild, SuperviseError> {
7471    if roster.is_closed() {
7472        return Err(SuperviseError::Spawn {
7473            program: spec.program.clone(),
7474            source: io::Error::other("the daemon is shutting down; not starting a new process"),
7475            cgroup_path: None,
7476        });
7477    }
7478    #[cfg(target_os = "linux")]
7479    let cgroup_name = {
7480        // Slot names alone are not kill domains: a retired incumbent may still
7481        // be draining when a later enable/restart spawns into the same slot.
7482        // Decimal entropy keeps the suffix unambiguous; Placement performs
7483        // the module-id escaping and constructs the filesystem path.
7484        if cgroup_placement.is_none() {
7485            swap::cgroup_name(&spec.module_id, alternate_slot)
7486        } else {
7487            let nonce = generate_launch_nonce()?;
7488            let suffix = u128::from_str_radix(&nonce[..32], 16).expect("hex launch nonce");
7489            // Leave room for byte escaping and the suffix under NAME_MAX. The
7490            // label is only for humans; the nonce identifies the kill domain.
7491            let mut end = spec.module_id.len().min(64);
7492            while !spec.module_id.is_char_boundary(end) {
7493                end -= 1;
7494            }
7495            format!(
7496                "{}_{suffix}",
7497                swap::cgroup_name(&spec.module_id[..end], alternate_slot)
7498            )
7499        }
7500    };
7501    #[cfg(not(target_os = "linux"))]
7502    let _ = alternate_slot;
7503    #[cfg(target_os = "macos")]
7504    let (mut command, privacy_exec, exec_ack) = privacy_command(spec, roster)?;
7505    #[cfg(not(target_os = "macos"))]
7506    let mut command = Command::new(&spec.program);
7507    command.args(&spec.args);
7508    // AMBIENT `CK_LOG` MUST NOT LEAK INTO AN OTHERWISE UNCONFIGURED MODULE — but
7509    // that is the whole of the intent, so remove that one key rather than the
7510    // environment.
7511    //
7512    // This was `env_clear()` from 0.17.41 until 0.18.3, which achieved the goal
7513    // and took the POSIX environment with it. Modules spawned that way had no
7514    // HOME, XDG_RUNTIME_DIR, TMPDIR or USER, and the consequences ran past
7515    // logging:
7516    //
7517    //   * `connection_file::discover` reads XDG_RUNTIME_DIR and HOME, so with
7518    //     both unset it fell back to the temp dir alone and `ck` could not find
7519    //     a daemon running on the same machine from inside any module's process
7520    //     tree — reporting a path the file has never lived at, which reads as
7521    //     "the daemon did not write its file".
7522    //   * `default_data_home()` with HOME and XDG_DATA_HOME both unset returns
7523    //     the RELATIVE `.local/share`, so a module deriving its own store path
7524    //     resolved it against its own CWD. That is the store-fragmentation
7525    //     defect the daemon already refuses in config (`parse_doc` rejects a
7526    //     relative `storage.data_home`) arriving by derivation instead.
7527    //   * anything a module spawns inherited it: git without ~/.gitconfig,
7528    //     cargo without CARGO_HOME, ssh, python user dirs — all degrading
7529    //     quietly rather than erroring.
7530    //
7531    // Reported by iceteaSA as #104 after deploying 0.18.2, where `ck daemon`
7532    // offered one candidate under /tmp while the file sat in /run/user/1000.
7533    //
7534    // A configured module is unaffected either way: `module_spec()` puts the
7535    // resolved CK_LOG into `spec.env`, which is applied below and therefore
7536    // wins over anything ambient.
7537    apply_child_env(&mut command, spec);
7538    apply_spawn_role(&mut command, role);
7539    let nonce_handoff =
7540        apply_wire_spawn_args_for_role(&mut command, spec, connection_file_path, handle, role)?;
7541
7542    #[cfg(target_os = "linux")]
7543    let cgroup_path = cgroup_placement
7544        .map(|placement| placement.module_path(&cgroup_name))
7545        .transpose()
7546        .map_err(|source| SuperviseError::Cgroup {
7547            module_id: spec.module_id.clone(),
7548            source,
7549        })?;
7550    #[cfg(not(target_os = "linux"))]
7551    let cgroup_path: Option<PathBuf> = None;
7552    #[cfg(target_os = "linux")]
7553    if let Some(path) = &cgroup_path {
7554        if let Err(error) = apply_cgroup_placement(&mut command, spec, path) {
7555            if let Some(placement) = cgroup_placement {
7556                remove_module_cgroup(placement, &cgroup_name);
7557            }
7558            return Err(error);
7559        }
7560    }
7561
7562    let output_sink = if let Some(logs_dir) = capture_logs_dir {
7563        let path = logs_dir.join(format!("{}.stderr.log", spec.module_id));
7564        match ChildOutputSink::open(&path, capture_retention(spec)) {
7565            Ok(sink) => sink,
7566            Err(error) => {
7567                warn!(
7568                    module_id = %spec.module_id,
7569                    path = %path.display(),
7570                    error = %error,
7571                    "could not open child output capture file; forwarding to stderr"
7572                );
7573                ChildOutputSink::Stderr
7574            }
7575        }
7576    } else {
7577        ChildOutputSink::Stderr
7578    };
7579
7580    command.stdout(Stdio::piped());
7581    command.stderr(Stdio::piped());
7582    command.kill_on_drop(true);
7583    // EACH MODULE LEADS ITS OWN PROCESS GROUP (the child calls setpgid(0, 0)
7584    // before exec). In the daemon's group, a service manager that kills the
7585    // job's process group when the daemon exits (launchd's default) killed
7586    // every module at the same moment its control connection closed, so no
7587    // module ever ran its EOF teardown on a daemon stop. Outside that group a
7588    // module is reached only by the daemon: the EOF it sees when its
7589    // connection closes, and the bounded stop in `child_roster` for anything
7590    // still running after that. On Linux this composes with the cgroup
7591    // placement above: that is a pre_exec write to cgroup.procs, std performs
7592    // setpgid in the child before running pre_exec callbacks, and the two
7593    // change independent process attributes.
7594    //
7595    // stdin is /dev/null because a process outside the terminal's foreground
7596    // group is stopped (SIGTTIN) if it reads the terminal, which a daemon run
7597    // by hand would otherwise hand down. Under a service manager stdin is
7598    // already /dev/null.
7599    #[cfg(unix)]
7600    command.process_group(0);
7601    command.stdin(Stdio::null());
7602    // The LAST pre-exec step, after the cgroup placement above: installing the
7603    // nonce at descriptor 3 replaces whatever the child had there, which could
7604    // be the descriptor an earlier step writes through.
7605    #[cfg(unix)]
7606    if let Some(handoff) = nonce_handoff {
7607        handoff.install_last(command.as_std_mut());
7608    }
7609    #[cfg(not(unix))]
7610    let _ = nonce_handoff;
7611
7612    // Containment, step 1 of 3 (issue #109): create the child suspended so it
7613    // cannot run a single instruction -- and therefore cannot spawn a
7614    // grandchild -- before it is in the job. See `contain_spawned_child` for the
7615    // other two steps and why the window matters.
7616    #[cfg(windows)]
7617    subc_jobobject::suspend_on_create_async(&mut command);
7618    let mut child = match command.spawn() {
7619        Ok(child) => child,
7620        Err(source) => {
7621            #[cfg(target_os = "linux")]
7622            if let Some(placement) = cgroup_placement {
7623                remove_module_cgroup(placement, &cgroup_name);
7624            }
7625            return Err(SuperviseError::Spawn {
7626                program: spec.program.clone(),
7627                source,
7628                cgroup_path,
7629            });
7630        }
7631    };
7632    // The parent must close its writer now: the acknowledgement pipe reports EOF
7633    // only when every writer is gone, and the child's copy closes when the
7634    // trampoline replaces itself with the module. Command holds only an integer
7635    // in its pre_exec callback, not another writer.
7636    #[cfg(target_os = "macos")]
7637    drop(exec_ack);
7638
7639    // Containment, steps 2 and 3: assign while suspended, then resume.
7640    #[cfg(windows)]
7641    let job = contain_spawned_child(&child, spec)?;
7642    let spawned_at_ms = unix_ms_now();
7643    let spawned_from = spec.program.clone();
7644    let spawned_file_identity = spawned_file_identity(&spawned_from);
7645    let pid = child.id().ok_or_else(|| SuperviseError::Spawn {
7646        program: spec.program.clone(),
7647        source: io::Error::other("spawned child exposed no live pid"),
7648        cgroup_path: cgroup_path.clone(),
7649    })?;
7650    let process_start_time = crate::provenance::process_start_time(pid);
7651    let process_identity = process_start_time.map(|start_time| ProcessIdentity { pid, start_time });
7652    #[cfg(all(test, target_os = "macos"))]
7653    privacy_exec_boundary_tests::before_image_sample(spec, pid);
7654    // Unix spawn returns after exec's error pipe closes. The kernel image is
7655    // therefore the executable to compare during a future orphan sweep: PATH
7656    // lookup and shebang interpretation may select a different file from the
7657    // configured program. Keep the literal program's identity for provenance,
7658    // but never use it as proof that a recorded pid may be signalled.
7659    let recorded_image = observe_spawned_image(pid);
7660    // spawn() confirms only the first exec, into the trampoline. Never persist
7661    // the trampoline image; the asynchronous acknowledgement publishes the
7662    // module image once the trampoline has replaced itself with the module.
7663    #[cfg(target_os = "macos")]
7664    let recorded_image = if privacy_exec.is_some() {
7665        None
7666    } else {
7667        recorded_image
7668    };
7669    #[cfg(target_os = "linux")]
7670    let recorded_cgroup_name = cgroup_path.as_ref().map(|_| cgroup_name.clone());
7671    #[cfg(not(target_os = "linux"))]
7672    let recorded_cgroup_name = None;
7673    let roster_guard = roster.admit(
7674        spec.module_id.clone(),
7675        pid,
7676        spec.protocol,
7677        process_start_time,
7678        crate::child_roster::RecordedIdentity {
7679            start_time: recorded_image.map(|image| image.start_time),
7680            executable: recorded_image
7681                .and_then(|image| image.executable)
7682                .map(crate::live_children::ExecutableIdentity::from),
7683            cgroup_name: recorded_cgroup_name,
7684            #[cfg(target_os = "linux")]
7685            cgroup_placement: cgroup_placement.cloned(),
7686        },
7687    );
7688    // The check at the top of this function can pass just before daemon
7689    // shutdown begins, and the process is only in the roster from here on.
7690    // The shutdown stop returns as soon as it finds the roster empty, so a
7691    // process admitted after that look would outlive the daemon. The roster
7692    // is closed before the stop first reads it and admission happens under
7693    // the roster's lock, so either the stop sees this process or this check
7694    // sees the roster closed: end the process now rather than start a module
7695    // the daemon is about to stop.
7696    if roster.is_closed() {
7697        // This child was never admitted, so there is no module protocol shutdown to wait for.
7698        #[cfg(target_os = "linux")]
7699        kill_module_cgroup(cgroup_placement, &cgroup_name);
7700        if let Err(error) = child.start_kill() {
7701            debug!(module_id = %spec.module_id, pid, %error, "kill of a process spawned during daemon shutdown failed; it may already have exited");
7702        }
7703        #[cfg(target_os = "linux")]
7704        if let Some(placement) = cgroup_placement {
7705            // This spawn was never admitted, so shutdown has no roster entry
7706            // to await. Do not detach its cleanup: the runtime could exit
7707            // before that task reaps the rejected child and removes its group.
7708            while matches!(child.try_wait(), Ok(None)) {
7709                std::thread::yield_now();
7710            }
7711            if matches!(
7712                subc_cgroup::kill_module(Some(placement), &cgroup_name),
7713                subc_cgroup::KillOutcome::Killed
7714            ) {
7715                if let Ok(path) = placement.module_path(&cgroup_name) {
7716                    while std::fs::read_to_string(path.join("cgroup.events"))
7717                        .ok()
7718                        .is_some_and(|events| events.lines().any(|line| line == "populated 1"))
7719                    {
7720                        std::thread::yield_now();
7721                    }
7722                }
7723            }
7724            remove_module_cgroup(placement, &cgroup_name);
7725        }
7726        drop(roster_guard);
7727        return Err(SuperviseError::Spawn {
7728            program: spec.program.clone(),
7729            source: io::Error::other(
7730                "the daemon began shutting down while this process was starting; ended it",
7731            ),
7732            cgroup_path,
7733        });
7734    }
7735
7736    let stdout_pump = match child.stdout.take() {
7737        Some(stdout) => Some(tokio::spawn(pump_stdout_to(stdout, output_sink.clone()))),
7738        None => {
7739            warn!(
7740                module_id = %spec.module_id,
7741                "spawned child exposed no stdout pipe; file capture will be incomplete"
7742            );
7743            None
7744        }
7745    };
7746    let stderr_pump = match child.stderr.take() {
7747        Some(stderr) => {
7748            let generation = ring
7749                .lock()
7750                .unwrap_or_else(|poisoned| poisoned.into_inner())
7751                .begin_process();
7752            Some(StderrPump {
7753                task: tokio::spawn(pump_stderr_to(
7754                    stderr,
7755                    Arc::clone(ring),
7756                    generation,
7757                    output_sink,
7758                )),
7759                generation,
7760            })
7761        }
7762        None => {
7763            // Spawning succeeded but the pipe did not materialise. Recording it as
7764            // uncaptured keeps the tail honest: the alternative is an empty tail
7765            // that reads as a module which printed nothing.
7766            ring.lock()
7767                .unwrap_or_else(|poisoned| poisoned.into_inner())
7768                .mark_not_captured("stderr pipe was not available on spawn");
7769            warn!(
7770                module_id = %spec.module_id,
7771                "spawned child exposed no stderr pipe; tail will be unavailable"
7772            );
7773            None
7774        }
7775    };
7776
7777    Ok(SupervisedChild {
7778        child,
7779        protocol: spec.protocol,
7780        #[cfg(target_os = "linux")]
7781        module_id: cgroup_name,
7782        #[cfg(target_os = "linux")]
7783        cgroup_placement: cgroup_placement.cloned(),
7784        #[cfg(windows)]
7785        job,
7786        stdout_pump,
7787        stderr_pump,
7788        stderr_ring: Arc::clone(ring),
7789        spawned_at_ms,
7790        spawned_from,
7791        spawned_file_identity,
7792        process_start_time,
7793        process_identity,
7794        pid,
7795        roster_guard: Some(roster_guard),
7796        #[cfg(target_os = "macos")]
7797        privacy_exec,
7798        #[cfg(target_os = "macos")]
7799        report_ready: Arc::new(OnceLock::new()),
7800        spawn_failure: None,
7801    })
7802}
7803
7804#[cfg(target_os = "linux")]
7805pub(crate) fn kill_module_cgroup(placement: Option<&subc_cgroup::Placement>, module_id: &str) {
7806    use subc_cgroup::KillOutcome;
7807    match subc_cgroup::kill_module(placement, module_id) {
7808        KillOutcome::Killed => {}
7809        KillOutcome::NotPlaced | KillOutcome::Unsupported => {
7810            debug!(
7811                module_id,
7812                "cgroup tree kill unavailable; using direct-child kill"
7813            );
7814        }
7815        KillOutcome::IoError { path, error } => {
7816            warn!(module_id, path = %path.display(), %error, "cgroup tree kill failed; using direct-child kill");
7817        }
7818    }
7819}
7820
7821/// Contain a freshly spawned Windows child and start it.
7822///
7823/// Steps 2 and 3 of the suspended-create contract: the job is created and the
7824/// child assigned **while it is still suspended** (step 1 is
7825/// `suspend_on_create_async` at the spawn site), then the child is resumed.
7826///
7827/// A child that is never resumed hangs forever holding a pid, so a resume
7828/// failure kills the child and fails the spawn rather than returning a
7829/// `SupervisedChild` that can never run.
7830///
7831/// An assignment failure is NOT fatal: an uncontained module behaves exactly as
7832/// it did before this existed, whereas refusing to start one would be a new
7833/// outage. It is logged at warn because it means a helper process could leak.
7834#[cfg(windows)]
7835fn contain_spawned_child(
7836    child: &Child,
7837    spec: &ModuleSpec,
7838) -> Result<Option<subc_jobobject::JobObject>, SuperviseError> {
7839    let module_id = spec.module_id.as_str();
7840    let Some(pid) = child.id() else {
7841        // The child exited between spawn and here. Its tree, if it made one,
7842        // needs no containment: nothing is left to contain.
7843        warn!(
7844            module_id,
7845            "spawned child had already exited before containment; no job object attached"
7846        );
7847        return Ok(None);
7848    };
7849
7850    let job = match subc_jobobject::JobObject::new() {
7851        Ok(job) => job,
7852        Err(source) => {
7853            warn!(
7854                module_id,
7855                error = %source,
7856                "could not create a job object; this module's helper processes will not be \
7857                 reaped on teardown"
7858            );
7859            // Resume regardless: leaving the child suspended would turn a
7860            // containment gap into a hung module.
7861            resume_suspended_child(pid, spec)?;
7862            return Ok(None);
7863        }
7864    };
7865
7866    if let Err(source) = job.assign(child) {
7867        warn!(
7868            module_id,
7869            error = %source,
7870            "could not assign the child to its job object; this module's helper processes \
7871             will not be reaped on teardown"
7872        );
7873        resume_suspended_child(pid, spec)?;
7874        return Ok(None);
7875    }
7876
7877    resume_suspended_child(pid, spec)?;
7878    Ok(Some(job))
7879}
7880
7881/// Resume a suspended child, killing it if it cannot be started.
7882///
7883/// A suspended process holds a pid and does nothing, so there is no useful
7884/// state to return: the caller gets an error and the spawn fails.
7885#[cfg(windows)]
7886fn resume_suspended_child(pid: u32, spec: &ModuleSpec) -> Result<(), SuperviseError> {
7887    if let Err(source) = subc_jobobject::resume_main_thread(pid) {
7888        // Kill it here rather than leaving a suspended process for the caller
7889        // to notice; `kill_on_drop` would eventually do this, but the module
7890        // would have been reported as running in between.
7891        let _ = std::process::Command::new("taskkill.exe")
7892            .args(["/PID", &pid.to_string(), "/T", "/F"])
7893            .stdin(Stdio::null())
7894            .stdout(Stdio::null())
7895            .stderr(Stdio::null())
7896            .status();
7897        return Err(SuperviseError::Spawn {
7898            program: spec.program.clone(),
7899            source,
7900            cgroup_path: None,
7901        });
7902    }
7903    Ok(())
7904}
7905
7906#[cfg(target_os = "linux")]
7907fn remove_module_cgroup(placement: &subc_cgroup::Placement, module_id: &str) {
7908    match placement.remove_module(module_id) {
7909        Ok(()) => debug!(module_id, "removed module cgroup after process exit"),
7910        Err(error) => warn!(
7911            module_id,
7912            error = %error,
7913            "could not remove module cgroup after process exit; continuing teardown"
7914        ),
7915    }
7916}
7917
7918#[cfg(target_os = "linux")]
7919async fn cleanup_reaped_cgroup(placement: &subc_cgroup::Placement, module_id: &str) {
7920    // Reaping the direct child is not proof its descendants exited. End the
7921    // residual tree and wait for the kernel's population fact before rmdir;
7922    // otherwise a successful parent wait leaks a directory on each restart.
7923    if matches!(
7924        subc_cgroup::kill_module(Some(placement), module_id),
7925        subc_cgroup::KillOutcome::Killed
7926    ) {
7927        if let Ok(path) = placement.module_path(module_id) {
7928            while std::fs::read_to_string(path.join("cgroup.events"))
7929                .ok()
7930                .is_some_and(|events| events.lines().any(|line| line == "populated 1"))
7931            {
7932                sleep(Duration::from_millis(1)).await;
7933            }
7934        }
7935    }
7936    remove_module_cgroup(placement, module_id);
7937}
7938
7939#[cfg(target_os = "linux")]
7940fn apply_cgroup_placement(
7941    command: &mut Command,
7942    spec: &ModuleSpec,
7943    path: &std::path::Path,
7944) -> Result<(), SuperviseError> {
7945    subc_cgroup::apply(command, path).map_err(|source| SuperviseError::Cgroup {
7946        module_id: spec.module_id.clone(),
7947        source,
7948    })
7949}
7950
7951fn capture_retention(spec: &ModuleSpec) -> Retention {
7952    let defaults = Retention::default();
7953    let value = |name: &str| {
7954        spec.env
7955            .iter()
7956            .rev()
7957            .find_map(|(key, value)| (key == name).then_some(value.as_str()))
7958    };
7959    Retention {
7960        max_file_mb: value(CAPTURE_MAX_FILE_MB_ENV)
7961            .and_then(|value| value.parse().ok())
7962            .unwrap_or(defaults.max_file_mb),
7963        keep: value(CAPTURE_KEEP_ENV)
7964            .and_then(|value| value.parse().ok())
7965            .unwrap_or(defaults.keep),
7966        max_age_days: value(CAPTURE_MAX_AGE_DAYS_ENV)
7967            .and_then(|value| value.parse().ok())
7968            .unwrap_or(defaults.max_age_days),
7969    }
7970}
7971
7972/// A fresh 256-bit CSPRNG launch nonce, lowercase hex. Used to bind a reserved
7973/// module's registration to the exact process the supervisor spawned.
7974fn generate_launch_nonce() -> Result<String, SuperviseError> {
7975    let mut bytes = [0u8; 32];
7976    getrandom::getrandom(&mut bytes).map_err(|source| SuperviseError::LaunchNonce {
7977        reason: source.to_string(),
7978    })?;
7979    let mut hex = String::with_capacity(64);
7980    for b in bytes {
7981        use std::fmt::Write;
7982        let _ = write!(hex, "{b:02x}");
7983    }
7984    Ok(hex)
7985}
7986
7987/// Constant-time byte comparison so a reserved-nonce mismatch leaks no timing
7988/// signal about how many leading bytes matched.
7989fn constant_time_eq(a: &[u8], b: &[u8]) -> bool {
7990    if a.len() != b.len() {
7991        return false;
7992    }
7993    let mut diff = 0u8;
7994    for (x, y) in a.iter().zip(b.iter()) {
7995        diff |= x ^ y;
7996    }
7997    diff == 0
7998}
7999
8000/// The kernel's image after an acknowledged exec, shared by ordinary launches
8001/// and privacy trampolines. PATH and shebang interpretation are kernel facts,
8002/// not identities inferred from a configured pathname.
8003fn observe_spawned_image(pid: u32) -> Option<subc_os::Observation> {
8004    subc_os::Process::open(pid)
8005        .ok()
8006        .flatten()
8007        .and_then(|process| process.observe())
8008}
8009
8010fn spawn_and_mark_running(
8011    spec: &ModuleSpec,
8012    runtime: &SupervisorRuntimeConfig,
8013    snapshot: &SharedSnapshot,
8014) -> Result<SupervisedChild, SuperviseError> {
8015    let child = spawn_child(
8016        spec,
8017        runtime.connection_file_path.as_deref(),
8018        runtime.supervisor_handle.as_ref(),
8019        &runtime.stderr_ring,
8020        runtime.capture_logs_dir.as_deref(),
8021        &runtime.child_roster,
8022        #[cfg(target_os = "linux")]
8023        runtime.cgroup_placement.as_ref(),
8024    )?;
8025    set_running(snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
8026    Ok(child)
8027}
8028
8029enum RegistrationWaitOutcome {
8030    Registered,
8031    Exited(ExitReport),
8032    TimedOut,
8033}
8034
8035struct ReloadRegistrationFailure {
8036    exit_report: ExitReport,
8037    reason: String,
8038}
8039
8040#[derive(Debug, Clone, Copy, PartialEq, Eq)]
8041enum BusyGaugeObservation {
8042    Quiescent,
8043    Busy,
8044    Omitted,
8045}
8046
8047fn busy_gauge_observation(metrics: Option<&Value>, gauges: &[String]) -> BusyGaugeObservation {
8048    let Some(metrics) = metrics.and_then(Value::as_object) else {
8049        return BusyGaugeObservation::Omitted;
8050    };
8051    let mut sum = 0u128;
8052    for gauge in gauges {
8053        let Some(value) = metrics.get(gauge) else {
8054            return BusyGaugeObservation::Omitted;
8055        };
8056        let Some(value) = value.as_u64() else {
8057            return BusyGaugeObservation::Busy;
8058        };
8059        sum = sum.saturating_add(u128::from(value));
8060    }
8061    if sum == 0 {
8062        BusyGaugeObservation::Quiescent
8063    } else {
8064        BusyGaugeObservation::Busy
8065    }
8066}
8067
8068fn declared_busy_gauges(
8069    registry: &Registry,
8070    module_id: &str,
8071) -> Result<Vec<String>, SuperviseError> {
8072    busy_gauges_of(
8073        registry
8074            .get_module(module_id)
8075            .map_err(SuperviseError::Registry)?,
8076    )
8077}
8078
8079/// [`declared_busy_gauges`] for the registration a connection holds, in any
8080/// slot: after cutover the incumbent is no longer the id's active
8081/// registration, and its own manifest is the one that names its gauges.
8082fn declared_busy_gauges_for_connection(
8083    registry: &Registry,
8084    connection_id: ConnectionId,
8085) -> Result<Vec<String>, SuperviseError> {
8086    busy_gauges_of(
8087        registry
8088            .get_module_by_connection(connection_id)
8089            .map_err(SuperviseError::Registry)?,
8090    )
8091}
8092
8093fn busy_gauges_of(
8094    registration: Option<crate::registry::ModuleRegistration>,
8095) -> Result<Vec<String>, SuperviseError> {
8096    let Some(registration) = registration else {
8097        return Ok(Vec::new());
8098    };
8099    let Some(self_signals) = registration.manifest.self_signals else {
8100        return Ok(Vec::new());
8101    };
8102
8103    let mut gauges = Vec::new();
8104    for declaration in self_signals {
8105        if declaration.kind != SelfSignalKind::Busy {
8106            continue;
8107        }
8108        match declaration.anchored_to {
8109            SignalAnchor::HealthGauges { gauges: declared } if !declared.is_empty() => {
8110                gauges.extend(declared)
8111            }
8112            _ => {
8113                // An invalid Busy anchor is fail-safe: the empty name cannot be
8114                // present in a conforming health report, so this drain stays busy.
8115                gauges.push(String::new());
8116            }
8117        }
8118    }
8119    Ok(gauges)
8120}
8121
8122/// Wait for `endpoint` to have nothing in flight and, when the module declares
8123/// busy gauges, for a health probe to report them quiet. The probe is addressed
8124/// by `scope`: a swap's superseded incumbent must be asked about its own
8125/// gauges, and by module id the probe would reach the promoted candidate.
8126async fn wait_for_forwarding_quiescence(
8127    forwarding: &ForwardingTable,
8128    module_id: &str,
8129    runtime: &SupervisorRuntimeConfig,
8130    endpoint: crate::ModuleEndpointId,
8131    deadline: Instant,
8132    busy_gauges: &[String],
8133    scope: DrainScope,
8134) -> Result<bool, SuperviseError> {
8135    let mut gauges_quiescent = busy_gauges.is_empty();
8136    let mut next_probe_at = Instant::now();
8137    let mut omission_counted = false;
8138
8139    loop {
8140        let now = Instant::now();
8141        if !busy_gauges.is_empty() && now >= next_probe_at && now < deadline {
8142            let report = match scope {
8143                DrainScope::Active => probe_module_health(module_id, runtime, Some(deadline)).await,
8144                DrainScope::Endpoint(endpoint) => {
8145                    probe_endpoint_health(endpoint, runtime, Some(deadline)).await
8146                }
8147            };
8148            gauges_quiescent = match report {
8149                Ok(report) => match busy_gauge_observation(report.metrics.as_ref(), busy_gauges) {
8150                    BusyGaugeObservation::Quiescent => true,
8151                    BusyGaugeObservation::Busy => false,
8152                    BusyGaugeObservation::Omitted => {
8153                        if !omission_counted {
8154                            forwarding
8155                                .counters()
8156                                .increment_drains_with_undeclared_gauge();
8157                            omission_counted = true;
8158                        }
8159                        false
8160                    }
8161                },
8162                Err(err) => {
8163                    warn!(
8164                        module_id,
8165                        error = %err,
8166                        "drain health.check did not produce declared busy gauges; treating module as busy"
8167                    );
8168                    false
8169                }
8170            };
8171            next_probe_at = Instant::now() + runtime.health.cadence.max(REGISTRY_RELEASE_POLL);
8172        }
8173
8174        let in_flight = forwarding
8175            .endpoint_in_flight_count(endpoint)
8176            .map_err(SuperviseError::Forwarding)?;
8177        if in_flight == 0 && gauges_quiescent {
8178            return Ok(true);
8179        }
8180
8181        let now = Instant::now();
8182        if now >= deadline {
8183            return Ok(false);
8184        }
8185        let mut wait = deadline
8186            .saturating_duration_since(now)
8187            .min(REGISTRY_RELEASE_POLL);
8188        if !busy_gauges.is_empty() {
8189            wait = wait.min(next_probe_at.saturating_duration_since(now));
8190        }
8191        sleep(wait).await;
8192    }
8193}
8194
8195/// The `route.closed` `drained` value implied by a quiescence-wait outcome.
8196///
8197/// `Ok` is always honest and passed straight through -- the wait actually measured
8198/// in-flight state. `Err` means the wait produced no measurement at all (the
8199/// forwarding table's lock was poisoned), so `false` is reported as the one honest
8200/// constant: the drain did not complete. Never recomputed from route state, never a
8201/// third "unknown" value -- the caller must still send a well-formed `route.closed`.
8202fn drained_after_quiescence_wait(wait_result: &Result<bool, SuperviseError>) -> bool {
8203    match wait_result {
8204        Ok(drained) => *drained,
8205        Err(_) => false,
8206    }
8207}
8208
8209fn send_route_goodbyes(forwarding: &ForwardingTable, released_routes: Vec<GoodbyeTarget>) {
8210    for released in released_routes {
8211        let frame = match Frame::build_with_version(
8212            released.negotiated_ver,
8213            FrameType::Goodbye,
8214            control_flags(),
8215            released.channel,
8216            released.epoch,
8217            0,
8218            Vec::new(),
8219        ) {
8220            Ok(frame) => frame,
8221            Err(err) => {
8222                warn!(
8223                    route_channel = released.channel,
8224                    error = %err,
8225                    "failed to build supervisor drain route GOODBYE frame"
8226                );
8227                continue;
8228            }
8229        };
8230        if !released.close_on_delivery_failure() {
8231            crate::forwarding::send_module_route_goodbye(
8232                &forwarding.counters(),
8233                &released.sink,
8234                frame,
8235                released.module_id.as_deref(),
8236                "supervisor drain",
8237            );
8238            continue;
8239        }
8240        if let Err(err) = released.sink.try_send(frame) {
8241            warn!(
8242                target_connection_id = released.connection_id.get(),
8243                route_channel = released.channel,
8244                error = %err,
8245                "supervisor drain route GOODBYE was not delivered to client; closing target connection"
8246            );
8247            let _ = forwarding.escalate_client_delivery_failure(
8248                released.connection_id,
8249                released.channel,
8250                released.epoch,
8251                CloseReason::new(
8252                    "route_goodbye_delivery_failed",
8253                    format!(
8254                        "failed to enqueue supervisor drain route GOODBYE for channel {}: {err}",
8255                        released.channel
8256                    ),
8257                ),
8258                crate::forwarding::UndeliveredFrame {
8259                    module_id: released.module_id.as_deref(),
8260                    sink: &released.sink,
8261                },
8262            );
8263        }
8264    }
8265}
8266
8267fn send_module_draining(
8268    module_id: &str,
8269    reason: RouteCloseReason,
8270    deadline_ms: u64,
8271    target: &ModuleDrainTarget,
8272) {
8273    let body = match serde_json::to_vec(&ModuleControlCommand::Draining {
8274        reason,
8275        deadline_ms,
8276    }) {
8277        Ok(body) => body,
8278        Err(err) => {
8279            warn!(
8280                module_id,
8281                error = %err,
8282                "failed to encode module draining command"
8283            );
8284            return;
8285        }
8286    };
8287    let frame = match Frame::build_with_version(
8288        target.negotiated_ver,
8289        FrameType::Push,
8290        control_flags(),
8291        0,
8292        0,
8293        0,
8294        body,
8295    ) {
8296        Ok(frame) => frame,
8297        Err(err) => {
8298            warn!(
8299                module_id,
8300                error = %err,
8301                "failed to build module draining command frame"
8302            );
8303            return;
8304        }
8305    };
8306    if let Err(err) = target.sink.try_send(frame) {
8307        warn!(
8308            module_id,
8309            target_connection_id = target.endpoint.connection_id.get(),
8310            error = %err,
8311            "module draining command was not delivered to peer"
8312        );
8313    }
8314}
8315
8316/// The channel-0 GOODBYE that tells a module its stop is planned.
8317fn module_goodbye_frame(module_id: &str, negotiated_ver: u8) -> Option<Frame> {
8318    match Frame::build_with_version(
8319        negotiated_ver,
8320        FrameType::Goodbye,
8321        control_flags(),
8322        0,
8323        0,
8324        0,
8325        Vec::new(),
8326    ) {
8327        Ok(frame) => Some(frame),
8328        Err(err) => {
8329            warn!(
8330                module_id,
8331                error = %err,
8332                "failed to build module GOODBYE frame"
8333            );
8334            None
8335        }
8336    }
8337}
8338
8339/// Send every registered module connection its module GOODBYE at daemon
8340/// shutdown, then request that connection's close.
8341///
8342/// A module tells a planned stop from a lost daemon by whether a GOODBYE came
8343/// before EOF, so the GOODBYE must reach the socket before the close. A close
8344/// request does not wait for the connection's queued frames: its writer gets a
8345/// bounded grace after the close, is aborted if it overruns it, and the daemon
8346/// process may exit before that grace ends. So with `wait_for_flush`, each
8347/// connection is closed only after its writer has acknowledged writing the
8348/// GOODBYE, or once a short shared budget runs out, so one module that is not
8349/// reading cannot hold up the others or the shutdown. Without it the GOODBYEs
8350/// are only queued, for a shutdown the operator has told to stop waiting.
8351/// A connection that is already gone is skipped.
8352#[cfg(unix)]
8353async fn send_module_goodbyes_for_daemon_shutdown(
8354    forwarding: &Arc<ForwardingTable>,
8355    reason: &CloseReason,
8356    wait_for_flush: bool,
8357) {
8358    const GOODBYE_BUDGET: Duration = Duration::from_millis(500);
8359    let targets = match forwarding.module_connections() {
8360        Ok(targets) => targets,
8361        Err(err) => {
8362            warn!(error = %err, "could not list module connections for shutdown GOODBYE");
8363            return;
8364        }
8365    };
8366    let deadline = Instant::now() + GOODBYE_BUDGET;
8367    let mut sends = tokio::task::JoinSet::new();
8368    for target in targets {
8369        let Some(frame) = module_goodbye_frame(&target.module_id, target.negotiated_ver) else {
8370            continue;
8371        };
8372        if !wait_for_flush {
8373            if let Err(err) = target.sink.try_send(frame) {
8374                debug!(
8375                    module_id = %target.module_id,
8376                    error = %err,
8377                    "shutdown module GOODBYE was not queued"
8378                );
8379            }
8380            continue;
8381        }
8382        let forwarding = Arc::clone(forwarding);
8383        let reason = reason.clone();
8384        sends.spawn(async move {
8385            match timeout_at(deadline, target.sink.send_flushed(frame)).await {
8386                Ok(Ok(())) => {}
8387                Ok(Err(err)) => debug!(
8388                    module_id = %target.module_id,
8389                    error = %err,
8390                    "module connection closed before its shutdown GOODBYE was written"
8391                ),
8392                Err(_) => warn!(
8393                    module_id = %target.module_id,
8394                    budget = ?GOODBYE_BUDGET,
8395                    "shutdown module GOODBYE was not written within its budget; closing anyway"
8396                ),
8397            }
8398            forwarding.request_connection_close(target.endpoint.connection_id, reason);
8399        });
8400    }
8401    // Every task ends by the shared deadline, so this wait is bounded too.
8402    while sends.join_next().await.is_some() {}
8403}
8404
8405fn send_module_goodbye(module_id: &str, forwarding: &ForwardingTable, target: &ModuleDrainTarget) {
8406    let Some(frame) = module_goodbye_frame(module_id, target.negotiated_ver) else {
8407        return;
8408    };
8409    if let Err(err) = target.sink.try_send(frame) {
8410        warn!(
8411            module_id,
8412            target_connection_id = target.endpoint.connection_id.get(),
8413            error = %err,
8414            "supervisor drain module GOODBYE was not delivered to peer; closing module connection"
8415        );
8416        forwarding.request_connection_close(
8417            target.endpoint.connection_id,
8418            CloseReason::new(
8419                "module_goodbye_delivery_failed",
8420                format!("failed to enqueue supervisor drain module GOODBYE for module '{module_id}': {err}"),
8421            ),
8422        );
8423    }
8424}
8425
8426#[derive(Clone, Copy)]
8427struct ForwardingDrainContext<'a> {
8428    spec: &'a ModuleSpec,
8429    runtime: &'a SupervisorRuntimeConfig,
8430    registry: &'a Registry,
8431    scope: DrainScope,
8432}
8433
8434/// Which process a forwarding drain addresses.
8435#[derive(Debug, Clone, Copy, PartialEq, Eq)]
8436enum DrainScope {
8437    /// Whatever endpoint is active for the module id: every plain stop,
8438    /// restart and reload. Also moves the module's state to `Draining`.
8439    Active,
8440    /// One specific endpoint: a swap's incumbent after cutover. Draining it by
8441    /// module id would resolve to the promoted candidate and leave neither
8442    /// process routable. The module's state is left alone, since the promoted
8443    /// candidate is what it describes and that process is running.
8444    Endpoint(crate::ModuleEndpointId),
8445}
8446
8447/// Whether a child being drained has already been asked to stop by the time
8448/// its drain wait starts.
8449///
8450/// The drain wait is the same budget whatever this says. What it decides is
8451/// whether the supervisor must ask by signal before that wait begins: a child
8452/// that nobody asked will sit out the whole budget and then be SIGKILLed,
8453/// healthy or not.
8454#[derive(Debug, Clone, Copy, PartialEq, Eq)]
8455enum StopNotice {
8456    /// The module was sent `module.draining` and a module GOODBYE over its own
8457    /// registered connection, and stops itself.
8458    SentOverConnection,
8459    /// The forwarding drain found no registered connection for the module: a
8460    /// subc child spawned moments ago that has not sent HELLO yet, or a
8461    /// `protocol: "none"` child, which never registers.
8462    NoConnection,
8463    /// This path sends nothing over the module's connection: the supervisor has
8464    /// no forwarding table, or the caller stops the child without a forwarding
8465    /// drain.
8466    NotSent,
8467}
8468
8469async fn begin_forwarding_drain(
8470    spec: &ModuleSpec,
8471    runtime: &SupervisorRuntimeConfig,
8472    registry: &Registry,
8473    snapshot: &SharedSnapshot,
8474    enabled: Option<bool>,
8475    reason: RouteCloseReason,
8476) -> Result<StopNotice, SuperviseError> {
8477    let Some(forwarding) = runtime.forwarding.as_ref() else {
8478        return Err(SuperviseError::ReloadUnavailable {
8479            module_id: spec.module_id.clone(),
8480            reason: "supervisor was not configured with a forwarding table".to_string(),
8481        });
8482    };
8483
8484    begin_forwarding_drain_with(
8485        forwarding,
8486        ForwardingDrainContext {
8487            spec,
8488            runtime,
8489            registry,
8490            scope: DrainScope::Active,
8491        },
8492        snapshot,
8493        enabled,
8494        reason,
8495        runtime.drain_timeout,
8496    )
8497    .await
8498}
8499
8500async fn begin_forwarding_drain_if_configured(
8501    spec: &ModuleSpec,
8502    runtime: &SupervisorRuntimeConfig,
8503    registry: &Registry,
8504    snapshot: &SharedSnapshot,
8505    enabled: Option<bool>,
8506    reason: RouteCloseReason,
8507) -> Result<StopNotice, SuperviseError> {
8508    begin_forwarding_drain_with_timeout(
8509        spec,
8510        runtime,
8511        registry,
8512        snapshot,
8513        enabled,
8514        reason,
8515        runtime.drain_timeout,
8516    )
8517    .await
8518}
8519
8520/// Like [`begin_forwarding_drain_if_configured`] but with an explicit drain
8521/// budget, for paths where the operator overrides the module's configured one
8522/// (`supervisor.restart{drain_timeout_ms}`).
8523async fn begin_forwarding_drain_with_timeout(
8524    spec: &ModuleSpec,
8525    runtime: &SupervisorRuntimeConfig,
8526    registry: &Registry,
8527    snapshot: &SharedSnapshot,
8528    enabled: Option<bool>,
8529    reason: RouteCloseReason,
8530    drain_timeout: Duration,
8531) -> Result<StopNotice, SuperviseError> {
8532    let Some(forwarding) = runtime.forwarding.as_ref() else {
8533        return Ok(StopNotice::NotSent);
8534    };
8535
8536    begin_forwarding_drain_with(
8537        forwarding,
8538        ForwardingDrainContext {
8539            spec,
8540            runtime,
8541            registry,
8542            scope: DrainScope::Active,
8543        },
8544        snapshot,
8545        enabled,
8546        reason,
8547        drain_timeout,
8548    )
8549    .await
8550}
8551
8552async fn begin_forwarding_drain_with(
8553    forwarding: &ForwardingTable,
8554    context: ForwardingDrainContext<'_>,
8555    snapshot: &SharedSnapshot,
8556    enabled: Option<bool>,
8557    reason: RouteCloseReason,
8558    drain_timeout: Duration,
8559) -> Result<StopNotice, SuperviseError> {
8560    let ForwardingDrainContext {
8561        spec,
8562        runtime,
8563        registry,
8564        scope,
8565    } = context;
8566    debug_assert_ne!(reason, RouteCloseReason::Crash);
8567    let terminal = matches!(reason, RouteCloseReason::Disable);
8568    let drain_started_at = Instant::now();
8569    let drain_deadline = drain_started_at + drain_timeout;
8570    let deadline_ms =
8571        unix_ms_now().saturating_add(u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX));
8572    let busy_gauges = match scope {
8573        DrainScope::Active => declared_busy_gauges(registry, &spec.module_id)?,
8574        DrainScope::Endpoint(endpoint) => {
8575            declared_busy_gauges_for_connection(registry, endpoint.connection_id)?
8576        }
8577    };
8578
8579    // Admission gate first: route.open/commit and route REQUEST admission are closed
8580    // before the first quiescence check, so the outstanding count can only fall.
8581    let gate_started = Instant::now();
8582    let drain_target = match scope {
8583        DrainScope::Active => forwarding.begin_module_drain(&spec.module_id, reason),
8584        DrainScope::Endpoint(endpoint) => forwarding.begin_endpoint_drain(endpoint, reason),
8585    }
8586    .map_err(SuperviseError::Forwarding)?;
8587    // The instant admission closed, and how long taking the forwarding write
8588    // lock to close it took. The timeout line reports only the quiescence
8589    // wait, so without this a drain that started late looked like one that
8590    // started on time.
8591    info!(
8592        module_id = %spec.module_id,
8593        ?reason,
8594        gate_ms = u64::try_from(gate_started.elapsed().as_millis()).unwrap_or(u64::MAX),
8595        connected = drain_target.is_some(),
8596        "module drain began; route admission closed"
8597    );
8598    if scope == DrainScope::Active {
8599        update_snapshot(snapshot, Some(&spec.module_id), |state| {
8600            state.state = ModuleState::Draining;
8601            state.draining_to_replace =
8602                matches!(reason, RouteCloseReason::Restart | RouteCloseReason::Reload);
8603            if let Some(enabled) = enabled {
8604                state.enabled = enabled;
8605            }
8606        })?;
8607    }
8608
8609    let Some(target) = drain_target.as_ref() else {
8610        // Nothing was sent: the module has no registered connection to carry
8611        // `module.draining` or a GOODBYE. The caller must not assume the child
8612        // was asked to stop.
8613        return Ok(StopNotice::NoConnection);
8614    };
8615    {
8616        send_module_draining(&spec.module_id, reason, deadline_ms, target);
8617        let routes = forwarding
8618            .endpoint_routes(target.endpoint)
8619            .map_err(SuperviseError::Forwarding)?;
8620        let routes_notified = routes.len();
8621        crate::control::send_route_control_pushes(
8622            forwarding,
8623            routes.clone(),
8624            ClientControlPush::RouteClosing {
8625                module_id: spec.module_id.clone(),
8626                channels: Vec::new(),
8627                reason,
8628            },
8629        );
8630        send_route_goodbyes(forwarding, target.abandoned_bindings.clone());
8631
8632        // `route.closing` was just sent above: from here on every return path,
8633        // including an early one, MUST send `route.closed` before propagating
8634        // anything else. A client holds `closing` as a promise that a verdict is
8635        // coming; leaving early without `closed` strands it waiting forever, since
8636        // `closing` carries no timeout of its own.
8637        let wait_result = wait_for_forwarding_quiescence(
8638            forwarding,
8639            &spec.module_id,
8640            runtime,
8641            target.endpoint,
8642            drain_deadline,
8643            &busy_gauges,
8644            scope,
8645        )
8646        .await;
8647        let drained = drained_after_quiescence_wait(&wait_result);
8648        if let Err(err) = &wait_result {
8649            error!(
8650                module_id = %spec.module_id,
8651                ?reason,
8652                error = %err,
8653                "forwarding quiescence wait failed after route.closing; forcing route.closed(drained: false) so the client is not left waiting on an unfulfilled promise"
8654            );
8655        } else if !drained {
8656            // Name what the drain waited on. Without it the line says only that
8657            // something did not settle, and "one wedged call" and "every
8658            // session's held stream" read the same; the first is a module bug,
8659            // the second is a module that should end its streams on
8660            // module.draining. Read before teardown releases the routes.
8661            let holdouts = forwarding
8662                .endpoint_drain_holdouts(target.endpoint)
8663                .unwrap_or_default();
8664            warn!(
8665                module_id = %spec.module_id,
8666                waited = ?drain_timeout,
8667                ?reason,
8668                held_requests = holdouts.requests,
8669                held_routes = holdouts.routes,
8670                total_routes = holdouts.total_routes,
8671                top_connections = ?holdouts.top_connections,
8672                // `module_channel:corr`, so the module can find each held request
8673                // in its own log; capped, so `held_requests` is the full count.
8674                held = %holdouts
8675                    .held
8676                    .iter()
8677                    .map(|(channel, corr)| format!("{channel}:{corr}"))
8678                    .collect::<Vec<_>>()
8679                    .join(","),
8680                "route drain timed out before request quiescence; forcing teardown"
8681            );
8682        }
8683        crate::control::send_route_control_pushes(
8684            forwarding,
8685            routes,
8686            ClientControlPush::RouteClosed {
8687                module_id: spec.module_id.clone(),
8688                channels: Vec::new(),
8689                reason,
8690                drained,
8691                abandoned: target.abandoned_bindings.len() as u32,
8692                excluded_subscriptions: target.excluded_subscriptions,
8693                terminal: Some(terminal),
8694            },
8695        );
8696        wait_result?;
8697
8698        // `route.closed` has now been sent unconditionally above. From here the
8699        // remaining steps are cleanup (route + module GOODBYE) rather than a
8700        // promise the client is waiting on, but a lock-poisoned
8701        // `release_module_endpoint_routes` would otherwise skip the module
8702        // GOODBYE silently too -- send it before propagating the error.
8703        let released_routes = match forwarding.release_module_endpoint_routes(target.endpoint) {
8704            Ok(routes) => routes,
8705            Err(err) => {
8706                warn!(
8707                    module_id = %spec.module_id,
8708                    ?reason,
8709                    error = %err,
8710                    "failed to release module endpoint routes after route.closed; module GOODBYE will still be sent"
8711                );
8712                send_module_goodbye(&spec.module_id, forwarding, target);
8713                return Err(SuperviseError::Forwarding(err));
8714            }
8715        };
8716        let route_goodbye_count = released_routes.len();
8717        send_route_goodbyes(forwarding, released_routes);
8718        send_module_goodbye(&spec.module_id, forwarding, target);
8719
8720        // The drain's happy path was previously silent: every emission above is
8721        // best-effort with only its failure arm logged, so "were consumers told"
8722        // was unprovable from the daemon log (surfaced by a 30-minute consumer
8723        // hang where the open question was exactly whether teardown notice went
8724        // out). One summary line makes that class decidable in one grep.
8725        info!(
8726            module_id = %spec.module_id,
8727            ?reason,
8728            routes_notified,
8729            route_goodbyes = route_goodbye_count,
8730            abandoned_reservations = target.abandoned_bindings.len(),
8731            excluded_subscriptions = target.excluded_subscriptions,
8732            drained,
8733            "module drain complete; consumers notified via route.closing/route.closed pushes and per-route GOODBYE frames"
8734        );
8735    }
8736
8737    Ok(StopNotice::SentOverConnection)
8738}
8739
8740/// Wait for the freshly spawned child to take the ACTIVE slot for `module_id`,
8741/// the only slot a plain (non-swap) spawn can register into.
8742async fn wait_for_registration_after_reload(
8743    registry: &Registry,
8744    module_id: &str,
8745    snapshot: &SharedSnapshot,
8746    child: &mut SupervisedChild,
8747    wait: Duration,
8748) -> Result<RegistrationWaitOutcome, SuperviseError> {
8749    wait_for_slot_registration(
8750        registry,
8751        crate::registry::RegistrationSlot::Active(module_id),
8752        module_id,
8753        snapshot,
8754        child,
8755        wait,
8756    )
8757    .await
8758}
8759
8760/// Wait for `child` to register into `slot`, or to exit, or for `wait` to pass.
8761///
8762/// Keyed on the slot rather than the bare module id because during a swap the
8763/// id's active slot is already held by the incumbent: an id-keyed wait would
8764/// report the incumbent's registration as the candidate's and a candidate that
8765/// never registers would look registered. A swap candidate waits on
8766/// `crate::registry::RegistrationSlot::Candidate`.
8767async fn wait_for_slot_registration(
8768    registry: &Registry,
8769    slot: crate::registry::RegistrationSlot<'_>,
8770    module_id: &str,
8771    snapshot: &SharedSnapshot,
8772    child: &mut SupervisedChild,
8773    wait: Duration,
8774) -> Result<RegistrationWaitOutcome, SuperviseError> {
8775    let deadline = Instant::now() + wait;
8776    loop {
8777        if registry
8778            .registration(slot)
8779            .map_err(SuperviseError::Registry)?
8780            .is_some()
8781        {
8782            return Ok(RegistrationWaitOutcome::Registered);
8783        }
8784
8785        let now = Instant::now();
8786        if now >= deadline {
8787            return Ok(RegistrationWaitOutcome::TimedOut);
8788        }
8789        let remaining = deadline.saturating_duration_since(now);
8790        let poll = remaining.min(REGISTRY_RELEASE_POLL);
8791
8792        tokio::select! {
8793            wait_result = child.wait() => {
8794                let status = wait_result.map_err(|source| SuperviseError::Wait {
8795                    module_id: module_id.to_string(),
8796                    source,
8797                })?;
8798                return Ok(RegistrationWaitOutcome::Exited(classify_reaped_child_exit(
8799                    snapshot,
8800                    child,
8801                    &status,
8802                )));
8803            }
8804            _ = sleep(poll) => {}
8805        }
8806    }
8807}
8808
8809fn registration_failure_exit_report(mut exit_report: ExitReport) -> ExitReport {
8810    // A replacement process that exits before HELLO did not provide service, even
8811    // if it used status 0. Count it against the restart cap as a new-binary failure.
8812    if exit_report.kind != ExitKind::DeliberateSeverance {
8813        exit_report.kind = ExitKind::Crash;
8814    }
8815    exit_report
8816}
8817
8818async fn handle_reload_child_registration_failure(
8819    spec: &ModuleSpec,
8820    runtime: &SupervisorRuntimeConfig,
8821    registry: &Registry,
8822    process_liveness: &SupervisorProcessLiveness,
8823    snapshot: &SharedSnapshot,
8824    _child: &mut Option<SupervisedChild>,
8825    failure: ReloadRegistrationFailure,
8826) -> Result<(), SuperviseError> {
8827    let ReloadRegistrationFailure {
8828        exit_report,
8829        reason,
8830    } = failure;
8831    match on_child_exit(
8832        spec,
8833        runtime.restart_policy,
8834        registry,
8835        snapshot,
8836        &runtime.terminal_ring,
8837        &runtime.spawn_events,
8838        &runtime.child_roster,
8839        exit_report,
8840    )
8841    .await
8842    {
8843        NextAction::Stop {
8844            registration_released,
8845        } => {
8846            if registration_released {
8847                process_liveness.untrack_if_current(&spec.module_id, snapshot);
8848            }
8849        }
8850        NextAction::Restart { schedule } => {
8851            let delay = schedule.map_or(runtime.restart_policy.delay_for_restart(0), |schedule| {
8852                schedule.delay
8853            });
8854            if let Some(schedule) = schedule {
8855                log_crash_respawn(&spec.module_id, schedule);
8856            }
8857            schedule_respawn(
8858                runtime,
8859                snapshot,
8860                &spec.module_id,
8861                delay,
8862                RespawnKind::Spawn,
8863            )?;
8864        }
8865    }
8866    Err(SuperviseError::ReloadFailed {
8867        module_id: spec.module_id.clone(),
8868        reason,
8869    })
8870}
8871
8872async fn handle_reload_spawn_failure(
8873    spec: &ModuleSpec,
8874    runtime: &SupervisorRuntimeConfig,
8875    process_liveness: &SupervisorProcessLiveness,
8876    snapshot: &SharedSnapshot,
8877    _child: &mut Option<SupervisedChild>,
8878    reason: String,
8879) -> Result<(), SuperviseError> {
8880    let now = Instant::now();
8881    let mut schedule = None;
8882    update_snapshot(snapshot, Some(&spec.module_id), |state| {
8883        clear_current_process_facts(state);
8884        if state.enabled {
8885            schedule = state.next_crash_restart(&runtime.restart_policy, now);
8886            state.state = if schedule.is_some() {
8887                ModuleState::Restarting
8888            } else {
8889                ModuleState::Failed
8890            };
8891        } else {
8892            state.state = ModuleState::Disabled;
8893        }
8894    })?;
8895    if let Some(schedule) = schedule {
8896        schedule_respawn(
8897            runtime,
8898            snapshot,
8899            &spec.module_id,
8900            schedule.delay,
8901            RespawnKind::Spawn,
8902        )?;
8903    } else {
8904        process_liveness.untrack_if_current(&spec.module_id, snapshot);
8905    }
8906    Err(SuperviseError::ReloadFailed {
8907        module_id: spec.module_id.clone(),
8908        reason,
8909    })
8910}
8911
8912fn control_flags() -> Flags {
8913    Flags::new(false, Priority::Passive, false)
8914}
8915
8916#[allow(clippy::too_many_arguments)]
8917async fn drain_optional_child(
8918    module_id: &str,
8919    protocol: ModuleProtocol,
8920    stop_notice: StopNotice,
8921    registry: &Registry,
8922    forwarding: Option<&ForwardingTable>,
8923    snapshot: &SharedSnapshot,
8924    terminal_ring: &Arc<Mutex<TerminalRing>>,
8925    spawn_events: &SpawnEventFeed,
8926    child: &mut Option<SupervisedChild>,
8927    drain_timeout: Duration,
8928    final_state: ModuleState,
8929    enabled: Option<bool>,
8930) -> Result<(), SuperviseError> {
8931    if let Some(child) = child.take() {
8932        drain_child_to_state(
8933            module_id,
8934            protocol,
8935            stop_notice,
8936            registry,
8937            forwarding,
8938            snapshot,
8939            terminal_ring,
8940            spawn_events,
8941            child,
8942            drain_timeout,
8943            final_state,
8944            enabled,
8945        )
8946        .await
8947    } else {
8948        update_snapshot(snapshot, Some(module_id), |state| {
8949            state.state = final_state;
8950            if let Some(enabled) = enabled {
8951                state.enabled = enabled;
8952            }
8953            clear_current_process_facts(state);
8954        })?;
8955        release_dead_registration(registry, forwarding, snapshot, module_id).await
8956    }
8957}
8958
8959#[allow(clippy::too_many_arguments)]
8960async fn drain_child_to_state(
8961    module_id: &str,
8962    _protocol: ModuleProtocol,
8963    stop_notice: StopNotice,
8964    registry: &Registry,
8965    forwarding: Option<&ForwardingTable>,
8966    snapshot: &SharedSnapshot,
8967    terminal_ring: &Arc<Mutex<TerminalRing>>,
8968    spawn_events: &SpawnEventFeed,
8969    mut child: SupervisedChild,
8970    drain_timeout: Duration,
8971    final_state: ModuleState,
8972    enabled: Option<bool>,
8973) -> Result<(), SuperviseError> {
8974    let protocol = child.protocol;
8975    update_snapshot(snapshot, Some(module_id), |state| {
8976        state.state = ModuleState::Draining;
8977        state.draining_to_replace = final_state == ModuleState::Restarting;
8978        if let Some(enabled) = enabled {
8979            state.enabled = enabled;
8980        }
8981    })?;
8982
8983    // The wait below is the same budget in every case; what differs is
8984    // whether anything has ASKED the child to stop before it starts. Only a
8985    // forwarding drain that reached the module's registered connection has
8986    // (`module.draining`, then a module GOODBYE). Every other child was told
8987    // nothing: a `protocol: "none"` module, which never registers; a subc
8988    // module spawned moments ago that has not sent HELLO yet; or a stop that
8989    // runs no forwarding drain. Without a signal the budget is only a delay
8990    // in front of SIGKILL -- and the not-yet-registered child is the worst
8991    // case, because it registers into a module that is already draining,
8992    // is never told, and is killed while healthy.
8993    if stop_notice != StopNotice::SentOverConnection {
8994        if protocol == ModuleProtocol::Subc && stop_notice == StopNotice::NoConnection {
8995            info!(
8996                module_id,
8997                pid = child.pid,
8998                budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
8999                "module has no connection yet; requesting stop by signal"
9000            );
9001        }
9002        request_graceful_stop(module_id, &child);
9003    }
9004
9005    let exit_report = match timeout(drain_timeout, child.wait()).await {
9006        Ok(Ok(status)) => classify_reaped_child_exit(snapshot, &child, &status),
9007        Ok(Err(source)) => {
9008            fail_snapshot(snapshot, Some(module_id), None);
9009            return Err(SuperviseError::Wait {
9010                module_id: module_id.to_string(),
9011                source,
9012            });
9013        }
9014        Err(_) => {
9015            // Mirror the sibling arm above: state is already `Draining`, and an
9016            // error propagated from here would strand it there -- a state
9017            // `set_enabled(true)` cannot heal (`revive_terminal` matches only
9018            // `Failed | Stopped`), leaving an operator Restart as the only exit.
9019            // `Failed` before `?` keeps the module operator-visible and
9020            // revivable. Trigger is an ESRCH race (process exits between the
9021            // drain timeout firing and the kill) or a post-kill wait failure
9022            // (issue #34).
9023            //
9024            // Logged because the kill is otherwise visible only as signal 9 in
9025            // the terminal ring, and the budget it follows can be long enough
9026            // that consumers see a stretch of refusals with no stated cause.
9027            warn!(
9028                module_id,
9029                pid = child.pid,
9030                budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
9031                reason = ?final_state,
9032                ?stop_notice,
9033                "drain budget expired before the module exited; killing it"
9034            );
9035            child.start_kill().map_err(|source| {
9036                fail_snapshot(snapshot, Some(module_id), None);
9037                SuperviseError::Kill {
9038                    module_id: module_id.to_string(),
9039                    source,
9040                }
9041            })?;
9042            let status = child.wait().await.map_err(|source| {
9043                fail_snapshot(snapshot, Some(module_id), None);
9044                SuperviseError::Wait {
9045                    module_id: module_id.to_string(),
9046                    source,
9047                }
9048            })?;
9049            classify_reaped_child_exit(snapshot, &child, &status)
9050        }
9051    };
9052
9053    update_snapshot(snapshot, Some(module_id), |state| {
9054        state.state = final_state;
9055        if let Some(enabled) = enabled {
9056            state.enabled = enabled;
9057        }
9058        clear_current_process_facts(state);
9059        state.last_exit = Some(exit_report.clone());
9060        if exit_report.kind == ExitKind::DeliberateSeverance {
9061            state.lifetime_restarts += 1;
9062        }
9063    })?;
9064    let detail = lock_snapshot(snapshot)?.drain_disposition_detail.take();
9065    record_terminal_with_detail(
9066        module_id,
9067        terminal_ring,
9068        spawn_events,
9069        &exit_report,
9070        terminal_disposition(final_state),
9071        detail,
9072    );
9073    child.drain_stderr(module_id).await;
9074
9075    release_dead_registration(registry, forwarding, snapshot, module_id).await
9076}
9077
9078/// Ask a child that nothing else has asked to stop, by signal.
9079///
9080/// A registered subc module is asked over its own connection: the drain sends
9081/// `route.closing`/`route.closed` to its consumers, a GOODBYE per route, then a
9082/// module GOODBYE, and the module stops itself. A module that speaks no subc
9083/// wire receives none of that, and neither does a subc module that has not
9084/// registered yet, so for them the drain budget would be pure delay in front of
9085/// a SIGKILL -- and for a process with a store to flush (JetStream is the
9086/// reason `protocol: "none"` exists) a SIGKILL turns every ordinary teardown
9087/// into a recovery on the next start.
9088///
9089/// NEVER CALLED FOR A MODULE THAT WAS TOLD OVER ITS CONNECTION, and that is a
9090/// rule rather than an optimisation: that module's graceful stop is already
9091/// running by the time its child is drained, and a signal would race it.
9092///
9093/// Best-effort by construction. A child that has already exited is the ordinary
9094/// case rather than an error (the kill lands on a reaped or exiting pid), so a
9095/// failure is logged at debug and the wait-then-kill below still decides the
9096/// outcome.
9097#[cfg(unix)]
9098fn request_graceful_stop(module_id: &str, child: &SupervisedChild) {
9099    let Some(pid) = child
9100        .id()
9101        .and_then(|pid| i32::try_from(pid).ok())
9102        .and_then(rustix::process::Pid::from_raw)
9103    else {
9104        debug!(
9105            module_id,
9106            "no pid to signal for teardown; falling through to the drain wait"
9107        );
9108        return;
9109    };
9110    match rustix::process::kill_process(pid, rustix::process::Signal::TERM) {
9111        Ok(()) => debug!(
9112            module_id,
9113            "sent SIGTERM to a module nothing else asked to stop"
9114        ),
9115        Err(err) => debug!(
9116            module_id,
9117            error = %err,
9118            "SIGTERM to module failed; the drain wait and kill still apply"
9119        ),
9120    }
9121}
9122
9123/// Windows has no SIGTERM and no portable stand-in for one. The graceful stops
9124/// Windows does offer need cooperation this supervisor cannot assume: a console
9125/// control event requires sharing a console with the child, and `WM_CLOSE`
9126/// requires the child to pump a message loop. A supervised server process does
9127/// neither, so there is nothing to send and teardown is the wait followed by the
9128/// kill. Emulating a signal here would mean inventing a stop protocol, which is
9129/// the thing `protocol: "none"` exists to avoid.
9130#[cfg(not(unix))]
9131fn request_graceful_stop(module_id: &str, _child: &SupervisedChild) {
9132    debug!(
9133        module_id,
9134        "no graceful stop signal exists on this platform; teardown of a module nothing asked to stop waits, then kills"
9135    );
9136}
9137
9138fn terminal_disposition(final_state: ModuleState) -> TerminalDisposition {
9139    match final_state {
9140        ModuleState::Stopped => TerminalDisposition::Stopped,
9141        ModuleState::Disabled => TerminalDisposition::Disabled,
9142        ModuleState::Restarting => TerminalDisposition::Restarting,
9143        ModuleState::Failed => TerminalDisposition::Failed,
9144        ModuleState::Starting
9145        | ModuleState::Running
9146        | ModuleState::Unresponsive
9147        | ModuleState::Draining => {
9148            unreachable!("terminal exits only finish in terminal or restarting states")
9149        }
9150    }
9151}
9152
9153/// Release a reaped child's registration before allowing another spawn.
9154///
9155/// EOF is not a process-lifetime signal: an inherited socket can stay open
9156/// indefinitely, and serial frame dispatch can be waiting on egress instead of
9157/// reading EOF. After the normal release grace, request connection close (which
9158/// cancels both reads and dispatch), then allow one more release grace for the
9159/// connection guard's forwarding cleanup. Never evict a different connection.
9160async fn release_dead_registration(
9161    registry: &Registry,
9162    forwarding: Option<&ForwardingTable>,
9163    snapshot: &SharedSnapshot,
9164    module_id: &str,
9165) -> Result<(), SuperviseError> {
9166    let result = async {
9167        let registration = registry
9168            .get_module(module_id)
9169            .map_err(SuperviseError::Registry)?;
9170        match wait_for_registration_release(registry, module_id, REGISTRY_RELEASE_TIMEOUT).await {
9171            Ok(()) => return Ok(()),
9172            Err(SuperviseError::RegistrationStillActive { .. }) => {}
9173            Err(err) => return Err(err),
9174        }
9175        let pid = lock_snapshot(snapshot)?.reaped_pid;
9176        if let (Some(registration), Some(forwarding), Some(pid)) = (registration, forwarding, pid) {
9177            warn!(
9178                module_id,
9179                pid,
9180                connection_id = registration.connection_id.get(),
9181                "reaped module registration outlived release grace; closing dead connection"
9182            );
9183            forwarding.request_connection_close(
9184                registration.connection_id,
9185                CloseReason::new(
9186                    "supervised_process_reaped",
9187                    format!("module '{module_id}' pid {pid} exited"),
9188                ),
9189            );
9190            wait_for_slot_registration_release(
9191                registry,
9192                crate::registry::RegistrationSlot::Connection(registration.connection_id),
9193                REGISTRY_RELEASE_TIMEOUT,
9194            )
9195            .await?;
9196        }
9197        wait_for_registration_release(registry, module_id, Duration::ZERO).await
9198    }
9199    .await;
9200    if let Err(err) = &result {
9201        fail_snapshot(snapshot, Some(module_id), None);
9202        error!(module_id, error = %err, "registration release failed after child exit; module is failed and start can retry");
9203    }
9204    result
9205}
9206
9207/// Wait for the ACTIVE registration of `module_id` to go away, which is what a
9208/// plain stop or restart waits for before it spawns a replacement.
9209async fn wait_for_registration_release(
9210    registry: &Registry,
9211    module_id: &str,
9212    wait: Duration,
9213) -> Result<(), SuperviseError> {
9214    wait_for_slot_registration_release(
9215        registry,
9216        crate::registry::RegistrationSlot::Active(module_id),
9217        wait,
9218    )
9219    .await
9220}
9221
9222/// Wait for the registration in `slot` to go away.
9223///
9224/// Keyed on the slot rather than the bare module id because a successful swap
9225/// never empties the id's active slot (the promoted candidate is in it), so an
9226/// id-keyed wait for the incumbent's release would always time out. Draining a
9227/// swap's incumbent waits on `crate::registry::RegistrationSlot::Connection` with the
9228/// incumbent's connection instead.
9229async fn wait_for_slot_registration_release(
9230    registry: &Registry,
9231    slot: crate::registry::RegistrationSlot<'_>,
9232    wait: Duration,
9233) -> Result<(), SuperviseError> {
9234    let deadline = Instant::now() + wait;
9235    let mut release_events = registration_release_events().subscribe();
9236    let still_active = |registration: &crate::registry::ModuleRegistration| {
9237        SuperviseError::RegistrationStillActive {
9238            module_id: registration.manifest.module_id.clone(),
9239            waited: wait,
9240        }
9241    };
9242    loop {
9243        let _observed_generation = *release_events.borrow_and_update();
9244        let Some(registration) = registry
9245            .registration(slot)
9246            .map_err(SuperviseError::Registry)?
9247        else {
9248            return Ok(());
9249        };
9250
9251        let now = Instant::now();
9252        if now >= deadline {
9253            return Err(still_active(&registration));
9254        }
9255
9256        let remaining = deadline.saturating_duration_since(now);
9257        match timeout(remaining, release_events.changed()).await {
9258            Ok(Ok(())) | Ok(Err(_)) => {}
9259            Err(_) => return Err(still_active(&registration)),
9260        }
9261    }
9262}
9263
9264#[cfg(test)]
9265mod slot_registration_wait_tests {
9266    use super::*;
9267    use crate::registry::{ConnectionId, RegistrationSlot};
9268    use subc_protocol::manifest::ModuleManifest;
9269
9270    #[tokio::test]
9271    async fn enable_release_failure_is_failed_and_a_second_enable_retries() {
9272        let registry = Arc::new(Registry::default());
9273        let supervisor = Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::default());
9274        let runtime = supervisor.runtime_config();
9275        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
9276        let spec = ModuleSpec {
9277            module_id: "enable-stale-registration".to_string(),
9278            program: PathBuf::from("/missing/enable-retry-test"),
9279            args: Vec::new(),
9280            env: Vec::new(),
9281            reserved: false,
9282            reserved_prefixes: Vec::new(),
9283            protocol: ModuleProtocol::Subc,
9284            overlap: Default::default(),
9285        };
9286        let connection = ConnectionId::new(90);
9287        registry
9288            .register_with_control_ops(
9289                ModuleManifest::builder(&spec.module_id, "0.1.0").build(),
9290                1,
9291                connection,
9292                Vec::new(),
9293            )
9294            .unwrap();
9295        let mut child = None;
9296        let err = set_child_enabled(
9297            &spec,
9298            &runtime,
9299            &registry,
9300            &supervisor.process_liveness,
9301            &snapshot,
9302            &mut child,
9303            true,
9304        )
9305        .await
9306        .unwrap_err();
9307        assert!(matches!(
9308            err,
9309            SuperviseError::RegistrationStillActive { .. }
9310        ));
9311        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
9312        assert!(child.is_none());
9313        registry.deregister_connection(connection).unwrap();
9314        let err = set_child_enabled(
9315            &spec,
9316            &runtime,
9317            &registry,
9318            &supervisor.process_liveness,
9319            &snapshot,
9320            &mut child,
9321            true,
9322        )
9323        .await
9324        .unwrap_err();
9325        assert!(
9326            matches!(err, SuperviseError::Spawn { .. }),
9327            "second enable must attempt a spawn: {err}"
9328        );
9329        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
9330    }
9331
9332    const INCUMBENT: u64 = 1;
9333    const CANDIDATE: u64 = 2;
9334
9335    fn swapped_registry() -> Arc<Registry> {
9336        let registry = Arc::new(Registry::default());
9337        let manifest = ModuleManifest::builder("m", "0.1.0").build();
9338        registry
9339            .register_with_control_ops(
9340                manifest.clone(),
9341                1,
9342                ConnectionId::new(INCUMBENT),
9343                Vec::new(),
9344            )
9345            .unwrap();
9346        registry
9347            .register_candidate_with_control_ops(
9348                manifest,
9349                1,
9350                ConnectionId::new(CANDIDATE),
9351                Vec::new(),
9352            )
9353            .unwrap();
9354        registry
9355    }
9356
9357    /// After a promotion the id's active slot is held by the new process, so an
9358    /// id-keyed wait for the incumbent's release can never succeed; the
9359    /// connection-keyed wait completes as soon as the incumbent deregisters.
9360    #[tokio::test]
9361    async fn incumbent_release_is_awaited_by_connection_not_by_module_id() {
9362        let registry = swapped_registry();
9363        registry.promote_candidate("m").unwrap().unwrap();
9364
9365        assert!(matches!(
9366            wait_for_registration_release(&registry, "m", Duration::from_millis(50)).await,
9367            Err(SuperviseError::RegistrationStillActive { .. })
9368        ));
9369
9370        // Still held while the incumbent's connection has not deregistered.
9371        assert!(matches!(
9372            wait_for_slot_registration_release(
9373                &registry,
9374                RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
9375                Duration::from_millis(50),
9376            )
9377            .await,
9378            Err(SuperviseError::RegistrationStillActive { .. })
9379        ));
9380
9381        let releaser = Arc::clone(&registry);
9382        let release = tokio::spawn(async move {
9383            sleep(Duration::from_millis(20)).await;
9384            releaser
9385                .deregister_connection(ConnectionId::new(INCUMBENT))
9386                .unwrap();
9387            notify_registration_release();
9388        });
9389        wait_for_slot_registration_release(
9390            &registry,
9391            RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
9392            Duration::from_secs(5),
9393        )
9394        .await
9395        .expect("the incumbent's own registration is released");
9396        release.await.unwrap();
9397        assert!(registry.get_module("m").unwrap().is_some());
9398    }
9399
9400    /// The candidate slot is waited on separately from the active slot: the
9401    /// incumbent's registration neither holds up nor stands in for it.
9402    #[tokio::test]
9403    async fn candidate_slot_wait_ignores_the_incumbents_registration() {
9404        let registry = swapped_registry();
9405        assert!(matches!(
9406            wait_for_slot_registration_release(
9407                &registry,
9408                RegistrationSlot::Candidate("m"),
9409                Duration::from_millis(50),
9410            )
9411            .await,
9412            Err(SuperviseError::RegistrationStillActive { .. })
9413        ));
9414        registry
9415            .deregister_connection(ConnectionId::new(CANDIDATE))
9416            .unwrap();
9417        wait_for_slot_registration_release(
9418            &registry,
9419            RegistrationSlot::Candidate("m"),
9420            Duration::from_millis(50),
9421        )
9422        .await
9423        .expect("a candidate slot with no candidate is released");
9424        assert!(registry
9425            .registration(RegistrationSlot::Active("m"))
9426            .unwrap()
9427            .is_some());
9428    }
9429}
9430
9431fn classify_exit(status: &ExitStatus) -> ExitReport {
9432    ExitReport {
9433        kind: if status.success() {
9434            ExitKind::Clean
9435        } else {
9436            ExitKind::Crash
9437        },
9438        code: status.code(),
9439        signal: exit_signal(status),
9440        at_ms: unix_ms_now(),
9441    }
9442}
9443
9444/// The terminal record for a module whose `wait()` call itself errored (e.g. the
9445/// child was already reaped out-of-band). There is no `ExitStatus` to read a code
9446/// or signal from -- `None`/`None` is the honest shape, not a guess -- but the
9447/// disposition still must be `Failed` so the terminal ring is not silently missing
9448/// an entry, matching what `fail_snapshot` records for this same arm.
9449fn wait_error_exit_report() -> ExitReport {
9450    ExitReport {
9451        kind: ExitKind::Crash,
9452        code: None,
9453        signal: None,
9454        at_ms: unix_ms_now(),
9455    }
9456}
9457
9458#[cfg(unix)]
9459fn exit_signal(status: &ExitStatus) -> Option<i32> {
9460    use std::os::unix::process::ExitStatusExt;
9461
9462    status.signal()
9463}
9464
9465#[cfg(not(unix))]
9466fn exit_signal(_status: &ExitStatus) -> Option<i32> {
9467    None
9468}
9469
9470/// Give an operator-touched module its full crash budget back.
9471///
9472/// Named for the counter it used to zero; it now empties the in-window ring,
9473/// which is the same act. `lifetime_restarts` is untouched on purpose -- the
9474/// ledger of what happened survives every operator action.
9475fn reset_restart_count(snapshot: &SharedSnapshot, module_id: &str) -> Result<(), SuperviseError> {
9476    update_snapshot(snapshot, Some(module_id), |state| {
9477        state.clear_crash_restarts();
9478    })
9479}
9480
9481fn set_running(
9482    snapshot: &SharedSnapshot,
9483    child: &SupervisedChild,
9484    module_id: &str,
9485    spawn_events: &SpawnEventFeed,
9486) -> Result<(), SuperviseError> {
9487    let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
9488        module_id: Some(module_id.to_string()),
9489    })?;
9490    state.spawn_generation = spawn_events.emit_spawned(module_id, child.pid, child.spawned_at_ms);
9491    if std::mem::take(&mut state.coalesced_restart_pending) {
9492        let generation = state.spawn_generation;
9493        info!(module_id, "restart already satisfied by generation {generation}; coalesced pending request completed");
9494    }
9495    state.drain_disposition_detail = None;
9496    state.spawn_failure = None;
9497    // Every caller of this is a plain spawn, which always uses the primary key;
9498    // a promoted swap candidate sets the flag itself after this returns.
9499    state.in_alternate_slot = false;
9500    state.configuration_updated_since_spawn = false;
9501    state.spawned_protocol = Some(child.protocol);
9502    state.state = ModuleState::Running;
9503    state.enabled = true;
9504    state.process_alive = true;
9505    state.pid = child.id();
9506    #[cfg(target_os = "macos")]
9507    {
9508        state.report_ready = Some(Arc::clone(&child.report_ready));
9509    }
9510    state.spawned_at_ms = Some(child.spawned_at_ms);
9511    state.spawned_from = Some(child.spawned_from.clone());
9512    state.spawned_file_identity = child.spawned_file_identity;
9513    state.process_start_time = child.process_start_time;
9514    Ok(())
9515}
9516
9517fn clear_current_process_facts(state: &mut SupervisorSnapshot) {
9518    state.process_alive = false;
9519    state.spawned_protocol = None;
9520    state.pid = None;
9521    #[cfg(target_os = "macos")]
9522    {
9523        state.report_ready = None;
9524    }
9525    state.spawned_at_ms = None;
9526    state.spawned_from = None;
9527    state.spawned_file_identity = None;
9528    state.process_start_time = None;
9529    state.deliberate_severance = None;
9530}
9531
9532#[cfg(test)]
9533fn record_deliberate_severance(
9534    snapshot: &SharedSnapshot,
9535    identity: ProcessIdentity,
9536) -> Result<(), SuperviseError> {
9537    update_snapshot(snapshot, None, |state| {
9538        state.deliberate_severance = Some(identity);
9539    })
9540}
9541
9542fn apply_deliberate_severance_marker(
9543    snapshot: &SharedSnapshot,
9544    exited_identity: Option<ProcessIdentity>,
9545    mut exit_report: ExitReport,
9546) -> ExitReport {
9547    let marker = lock_snapshot(snapshot)
9548        .ok()
9549        .and_then(|mut state| state.deliberate_severance.take());
9550    if marker.is_some() && marker == exited_identity {
9551        exit_report.kind = ExitKind::DeliberateSeverance;
9552    }
9553    exit_report
9554}
9555
9556fn classify_reaped_child_exit(
9557    snapshot: &SharedSnapshot,
9558    child: &SupervisedChild,
9559    status: &ExitStatus,
9560) -> ExitReport {
9561    let _ = update_snapshot(snapshot, None, |state| {
9562        state.reaped_pid = Some(child.pid);
9563        state.spawn_failure = child.spawn_failure.clone();
9564    });
9565    apply_deliberate_severance_marker(snapshot, child.process_identity(), classify_exit(status))
9566}
9567
9568fn fail_snapshot(
9569    snapshot: &SharedSnapshot,
9570    module_id: Option<&str>,
9571    last_exit: Option<ExitReport>,
9572) {
9573    if let Err(err) = update_snapshot(snapshot, module_id, |state| {
9574        state.state = ModuleState::Failed;
9575        clear_current_process_facts(state);
9576        if let Some(last_exit) = last_exit {
9577            state.last_exit = Some(last_exit);
9578        }
9579    }) {
9580        error!(error = %err, "failed to mark supervisor state failed");
9581    }
9582}
9583
9584fn update_snapshot(
9585    snapshot: &SharedSnapshot,
9586    module_id: Option<&str>,
9587    update: impl FnOnce(&mut SupervisorSnapshot),
9588) -> Result<(), SuperviseError> {
9589    let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
9590        module_id: module_id.map(ToOwned::to_owned),
9591    })?;
9592    update(&mut state);
9593    Ok(())
9594}
9595
9596const SLOW_SNAPSHOT_LOCK_THRESHOLD: Duration = Duration::from_millis(250);
9597
9598fn lock_snapshot_for_control<'a>(
9599    snapshot: &'a SharedSnapshot,
9600    module_id: &str,
9601    caller: &'static str,
9602) -> Result<std::sync::MutexGuard<'a, SupervisorSnapshot>, SuperviseError> {
9603    let started_at = Instant::now();
9604    let guard = lock_snapshot(snapshot)?;
9605    let waited = started_at.elapsed();
9606    if waited >= SLOW_SNAPSHOT_LOCK_THRESHOLD {
9607        warn!(
9608            module_id = %module_id,
9609            waited_ms = waited.as_millis() as u64,
9610            caller = %caller,
9611            "slow snapshot lock"
9612        );
9613    }
9614    Ok(guard)
9615}
9616
9617fn lock_snapshot(
9618    snapshot: &SharedSnapshot,
9619) -> Result<std::sync::MutexGuard<'_, SupervisorSnapshot>, SuperviseError> {
9620    snapshot
9621        .lock()
9622        .map_err(|_| SuperviseError::StatePoisoned { module_id: None })
9623}
9624
9625#[cfg(test)]
9626mod terminal_history_tests {
9627    use std::{
9628        path::PathBuf,
9629        sync::Arc,
9630        time::{Duration, Instant},
9631    };
9632
9633    use tokio::time::sleep;
9634
9635    use super::{
9636        apply_deliberate_severance_marker, daemon_will_restart, drain_child_to_state,
9637        drained_after_quiescence_wait, handle_reload_spawn_failure, health_restart_child,
9638        lock_snapshot, on_child_exit, record_deliberate_severance, record_wait_error_terminal,
9639        reset_restart_count, spawn_and_mark_running, update_snapshot, wait_error_exit_report,
9640        ExitKind, ExitReport, ModuleProtocol, ModuleSpec, ModuleState, NextAction, ProcessIdentity,
9641        RestartPolicy, SpawnEventKind, StopNotice, SuperviseError, SupervisedModule, Supervisor,
9642        SupervisorHandle, SupervisorHealthStatus, SupervisorSnapshot,
9643    };
9644    // The supervisor's clock, distinct from the `std::time::Instant` these tests
9645    // use for their own wall-clock deadlines: crash-restart instants must be on
9646    // the same clock the production code stamps them with, which is tokio's (and
9647    // is what `start_paused` tests can move).
9648    use super::Instant as ClockInstant;
9649    use crate::{
9650        registry::Registry,
9651        terminal_ring::{TerminalRing, TerminalRingConfig},
9652    };
9653    use std::sync::Mutex;
9654    use subc_control::TerminalDisposition;
9655
9656    /// See the twin in `control.rs` for why this derives the path from
9657    /// `current_exe()` and why the existence check is here: `--lib` alone does
9658    /// not build `[[bin]]` targets, and a bare spawn then fails with a raw
9659    /// `NotFound` that reads as a broken test rather than an unbuilt dependency.
9660    pub(super) fn fake_aft_stub_path() -> PathBuf {
9661        let mut path = std::env::current_exe().expect("current_exe available in tests");
9662        path.pop();
9663        path.pop();
9664        path.push(if cfg!(windows) {
9665            "fake-aft-stub.exe"
9666        } else {
9667            "fake-aft-stub"
9668        });
9669        assert!(
9670            path.exists(),
9671            "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
9672             [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
9673            path.display()
9674        );
9675        path
9676    }
9677
9678    #[test]
9679    fn reserved_never_spawned_refuses_every_hello() {
9680        // The canary hole: a reserved id whose module has never spawned had NO
9681        // gate entry and admitted anyone -- the reservation protected the nonce
9682        // holder, not the NAME. Now the entry is present with no legitimate
9683        // holder and refuses all comers.
9684        let supervisor = SupervisorHandle::default();
9685        supervisor.apply_identity_configuration(&ModuleSpec {
9686            module_id: "never-spawned".to_string(),
9687            program: PathBuf::from("/usr/bin/false"),
9688            args: Vec::new(),
9689            env: Vec::new(),
9690            reserved: true,
9691            reserved_prefixes: Vec::new(),
9692            protocol: ModuleProtocol::Subc,
9693            overlap: Default::default(),
9694        });
9695        assert!(
9696            supervisor
9697                .reserved_hello_rejection("never-spawned", Some("any-forged-nonce"))
9698                .is_some(),
9699            "forged nonce must refuse on a reserved never-spawned id"
9700        );
9701        assert!(
9702            supervisor
9703                .reserved_hello_rejection("never-spawned", None)
9704                .is_some(),
9705            "absent nonce must refuse on a reserved never-spawned id"
9706        );
9707        // And a real spawn nonce minted later admits exactly that nonce.
9708        supervisor.set_spawn_nonce("never-spawned", "minted".to_string());
9709        supervisor.apply_identity_configuration(&ModuleSpec {
9710            module_id: "never-spawned".to_string(),
9711            program: PathBuf::from("/usr/bin/false"),
9712            args: Vec::new(),
9713            env: Vec::new(),
9714            reserved: true,
9715            reserved_prefixes: Vec::new(),
9716            protocol: ModuleProtocol::Subc,
9717            overlap: Default::default(),
9718        });
9719        assert!(supervisor
9720            .reserved_hello_rejection("never-spawned", Some("minted"))
9721            .is_none());
9722        assert!(supervisor
9723            .reserved_hello_rejection("never-spawned", Some("forged"))
9724            .is_some());
9725    }
9726
9727    /// Put `count` crash restarts on a snapshot's ring as if they had all just
9728    /// happened, which is what "spent budget" looks like to every reader.
9729    fn seed_crash_restarts(state: &mut SupervisorSnapshot, count: u32) {
9730        let now = ClockInstant::now();
9731        for _ in 0..count {
9732            state.crash_restarts.push_back(now);
9733        }
9734    }
9735
9736    /// Age the oldest recorded restart out of `window`, standing in for the hours
9737    /// that would otherwise have to pass. Injecting the instant is the point: a
9738    /// test that slept a real window would take ten minutes and still prove less.
9739    fn age_oldest_crash_restart_out_of_window(state: &mut SupervisorSnapshot, window: Duration) {
9740        let aged = state
9741            .crash_restarts
9742            .front()
9743            .expect("a crash restart must be recorded before it can be aged")
9744            .checked_sub(window + Duration::from_secs(1))
9745            .expect("the test clock is far enough from its origin to age an instant");
9746        state.crash_restarts[0] = aged;
9747    }
9748
9749    fn snapshot_with_restarts(enabled: bool, count: u32) -> SupervisorSnapshot {
9750        let mut state = SupervisorSnapshot::new(ModuleState::Running, enabled);
9751        seed_crash_restarts(&mut state, count);
9752        state
9753    }
9754
9755    #[test]
9756    fn daemon_owned_recovery_predicate_uses_the_pre_increment_budget() {
9757        let policy = RestartPolicy::new(3, Duration::ZERO);
9758        let now = ClockInstant::now();
9759        assert!(daemon_will_restart(
9760            &mut snapshot_with_restarts(true, 2),
9761            &policy,
9762            now
9763        ));
9764        assert!(!daemon_will_restart(
9765            &mut snapshot_with_restarts(true, 3),
9766            &policy,
9767            now
9768        ));
9769        assert!(!daemon_will_restart(
9770            &mut snapshot_with_restarts(false, 0),
9771            &policy,
9772            now
9773        ));
9774    }
9775
9776    #[test]
9777    fn crash_restart_backoff_escalates_with_in_window_count() {
9778        let policy = RestartPolicy::new(4, Duration::from_millis(100))
9779            .with_max_backoff(Duration::from_secs(30));
9780        let now = ClockInstant::now();
9781        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
9782        let schedules = (0..4)
9783            .map(|_| {
9784                state
9785                    .next_crash_restart(&policy, now)
9786                    .expect("the test policy allows four crash restarts")
9787            })
9788            .collect::<Vec<_>>();
9789
9790        assert_eq!(
9791            schedules
9792                .iter()
9793                .map(|schedule| schedule.restart_in_window)
9794                .collect::<Vec<_>>(),
9795            vec![0, 1, 2, 3]
9796        );
9797        assert_eq!(
9798            schedules
9799                .iter()
9800                .map(|schedule| schedule.delay)
9801                .collect::<Vec<_>>(),
9802            vec![
9803                Duration::from_millis(100),
9804                Duration::from_secs(1),
9805                Duration::from_secs(10),
9806                Duration::from_secs(30),
9807            ]
9808        );
9809    }
9810
9811    #[test]
9812    fn crash_restart_backoff_resets_after_ring_clear() {
9813        let policy = RestartPolicy::new(3, Duration::from_millis(100));
9814        let now = ClockInstant::now();
9815        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
9816        assert_eq!(
9817            state.next_crash_restart(&policy, now).unwrap().delay,
9818            Duration::from_millis(100)
9819        );
9820        assert_eq!(
9821            state.next_crash_restart(&policy, now).unwrap().delay,
9822            Duration::from_secs(1)
9823        );
9824
9825        state.clear_crash_restarts();
9826        let schedule = state
9827            .next_crash_restart(&policy, now)
9828            .expect("a cleared ring must allow another restart");
9829        assert_eq!(schedule.restart_in_window, 0);
9830        assert_eq!(schedule.delay, Duration::from_millis(100));
9831    }
9832
9833    #[test]
9834    fn crash_restart_backoff_ignores_aged_restarts() {
9835        let policy = RestartPolicy::new(3, Duration::from_millis(100));
9836        let now = ClockInstant::now();
9837        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
9838        state
9839            .next_crash_restart(&policy, now)
9840            .expect("the first restart is allowed");
9841        state
9842            .next_crash_restart(&policy, now)
9843            .expect("the second restart is allowed");
9844        state.crash_restarts[0] = now
9845            .checked_sub(policy.window + Duration::from_secs(1))
9846            .expect("the fake clock can age a restart past the window");
9847
9848        let schedule = state
9849            .next_crash_restart(&policy, now)
9850            .expect("an aged restart must release its slot");
9851        assert_eq!(schedule.restart_in_window, 1);
9852        assert_eq!(schedule.delay, Duration::from_secs(1));
9853        assert_eq!(state.crash_restarts.len(), 2);
9854    }
9855
9856    /// The budget is a rate: the same three spent restarts refuse a respawn
9857    /// while they are recent and allow one once they have aged past the window.
9858    /// Nothing about the module changed in between, which is the whole point.
9859    #[test]
9860    fn a_budget_spent_before_the_window_no_longer_refuses() {
9861        let policy = RestartPolicy::new(3, Duration::ZERO);
9862        let mut state = snapshot_with_restarts(true, 3);
9863        let now = ClockInstant::now();
9864        assert!(!daemon_will_restart(&mut state, &policy, now));
9865
9866        assert!(daemon_will_restart(
9867            &mut state,
9868            &policy,
9869            now + policy.window + Duration::from_secs(1)
9870        ));
9871        assert!(
9872            state.crash_restarts.is_empty(),
9873            "reading the budget must drop the instants that left the window"
9874        );
9875    }
9876
9877    fn module_with_recovery_snapshot(
9878        state: ModuleState,
9879        enabled: bool,
9880        restart_count: u32,
9881    ) -> SupervisedModule {
9882        let registry = Arc::new(Registry::default());
9883        let supervisor =
9884            Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::new(3, Duration::ZERO));
9885        let module = supervisor
9886            .spawn(ModuleSpec {
9887                module_id: "recovery-snapshot".to_string(),
9888                program: fake_aft_stub_path(),
9889                args: Vec::new(),
9890                env: Vec::new(),
9891                reserved: false,
9892                reserved_prefixes: Vec::new(),
9893                protocol: ModuleProtocol::Subc,
9894                overlap: Default::default(),
9895            })
9896            .unwrap();
9897        update_snapshot(
9898            &module.inner.snapshot,
9899            Some("recovery-snapshot"),
9900            |snapshot| {
9901                snapshot.state = state;
9902                snapshot.enabled = enabled;
9903                seed_crash_restarts(snapshot, restart_count);
9904            },
9905        )
9906        .unwrap();
9907        module
9908    }
9909
9910    #[cfg(target_os = "linux")]
9911    #[tokio::test]
9912    async fn no_cgroup_placement_does_not_block_fake_aft_stub_spawn() {
9913        let supervisor =
9914            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
9915                .with_cgroup_placement(None);
9916        let result = supervisor.spawn(ModuleSpec {
9917            module_id: "no-cgroup-placement".to_string(),
9918            program: fake_aft_stub_path(),
9919            args: Vec::new(),
9920            env: Vec::new(),
9921            reserved: false,
9922            reserved_prefixes: Vec::new(),
9923            protocol: ModuleProtocol::Subc,
9924            overlap: Default::default(),
9925        });
9926
9927        assert!(
9928            result.is_ok(),
9929            "no delegation must not turn an otherwise valid spawn into a failure: {result:?}"
9930        );
9931    }
9932
9933    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9934    async fn undecided_snapshot_uses_shared_restart_predicate() {
9935        assert!(module_with_recovery_snapshot(ModuleState::Running, true, 2)
9936            .will_recover_after_connection_loss()
9937            .unwrap());
9938        assert!(
9939            !module_with_recovery_snapshot(ModuleState::Running, true, 3)
9940                .will_recover_after_connection_loss()
9941                .unwrap()
9942        );
9943    }
9944
9945    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9946    async fn restarting_snapshot_at_exhausted_budget_is_non_terminal() {
9947        assert!(
9948            module_with_recovery_snapshot(ModuleState::Restarting, true, 3)
9949                .will_recover_after_connection_loss()
9950                .unwrap()
9951        );
9952    }
9953
9954    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9955    async fn terminal_phase_snapshots_are_terminal_before_budget_exhaustion() {
9956        assert!(!module_with_recovery_snapshot(ModuleState::Failed, true, 0)
9957            .will_recover_after_connection_loss()
9958            .unwrap());
9959        assert!(
9960            !module_with_recovery_snapshot(ModuleState::Disabled, true, 0)
9961                .will_recover_after_connection_loss()
9962                .unwrap()
9963        );
9964    }
9965
9966    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9967    async fn warming_snapshot_is_limited_to_startup_phases() {
9968        for state in [
9969            ModuleState::Starting,
9970            ModuleState::Running,
9971            ModuleState::Restarting,
9972        ] {
9973            assert!(
9974                module_with_recovery_snapshot(state, true, 0)
9975                    .is_warming()
9976                    .unwrap(),
9977                "{state:?} should be warming"
9978            );
9979        }
9980        for state in [
9981            ModuleState::Unresponsive,
9982            ModuleState::Draining,
9983            ModuleState::Stopped,
9984            ModuleState::Failed,
9985            ModuleState::Disabled,
9986        ] {
9987            assert!(
9988                !module_with_recovery_snapshot(state, true, 0)
9989                    .is_warming()
9990                    .unwrap(),
9991                "{state:?} should not be warming"
9992            );
9993        }
9994    }
9995
9996    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9997    async fn terminal_history_survives_respawn_and_keeps_both_crashes_in_order() {
9998        let registry = Arc::new(Registry::default());
9999        let supervisor =
10000            Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::new(1, Duration::ZERO));
10001        let module = supervisor
10002            .spawn(ModuleSpec {
10003                module_id: "terminal-history".to_string(),
10004                program: fake_aft_stub_path(),
10005                args: Vec::new(),
10006                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
10007                reserved: false,
10008                reserved_prefixes: Vec::new(),
10009                protocol: ModuleProtocol::Subc,
10010                overlap: Default::default(),
10011            })
10012            .unwrap();
10013
10014        let deadline = Instant::now() + Duration::from_secs(5);
10015        loop {
10016            let history = module.terminal_history();
10017            if history.entries.len() == 2 {
10018                assert_eq!(module.status().unwrap().state, ModuleState::Failed);
10019                assert_eq!(history.dropped, 0);
10020                assert_eq!(
10021                    history
10022                        .entries
10023                        .iter()
10024                        .map(|entry| entry.exit_code)
10025                        .collect::<Vec<_>>(),
10026                    vec![Some(23), Some(23)]
10027                );
10028                assert!(history.entries[0].at_ms <= history.entries[1].at_ms);
10029                return;
10030            }
10031            assert!(
10032                Instant::now() < deadline,
10033                "module did not retain two terminal exits: {history:?}"
10034            );
10035            sleep(Duration::from_millis(10)).await;
10036        }
10037    }
10038
10039    /// A disable issued while a crash respawn is still backing off must preempt
10040    /// that respawn: the operator's stop wins, the disable must not queue behind
10041    /// the backoff, and the module must never come back up afterwards.
10042    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10043    async fn disable_during_crash_backoff_cancels_pending_respawn() {
10044        let backoff = Duration::from_secs(2);
10045        let supervisor = Supervisor::new_for_test(
10046            Arc::new(Registry::default()),
10047            RestartPolicy::new(10, backoff),
10048        );
10049        let module = supervisor
10050            .spawn(ModuleSpec {
10051                module_id: "disable-during-backoff".to_string(),
10052                program: fake_aft_stub_path(),
10053                args: Vec::new(),
10054                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
10055                reserved: false,
10056                reserved_prefixes: Vec::new(),
10057                protocol: ModuleProtocol::Subc,
10058                overlap: Default::default(),
10059            })
10060            .unwrap();
10061
10062        // Wait for the first crash to put the module into its backoff window.
10063        let deadline = Instant::now() + Duration::from_secs(5);
10064        loop {
10065            if module.status().unwrap().state == ModuleState::Restarting {
10066                break;
10067            }
10068            assert!(
10069                Instant::now() < deadline,
10070                "module never entered the crash backoff"
10071            );
10072            sleep(Duration::from_millis(10)).await;
10073        }
10074
10075        let started = Instant::now();
10076        module.set_enabled(false).await.unwrap();
10077        let waited = started.elapsed();
10078
10079        assert!(
10080            waited < backoff / 2,
10081            "disable waited {waited:?} behind the {backoff:?} crash backoff; the operator command must preempt the pending respawn"
10082        );
10083        assert_eq!(module.status().unwrap().state, ModuleState::Disabled);
10084
10085        // Outlast the backoff: the respawn it was counting down to must never run.
10086        sleep(backoff + Duration::from_millis(500)).await;
10087        let status = module.status().unwrap();
10088        assert_eq!(status.state, ModuleState::Disabled);
10089        assert_eq!(
10090            status.spawn_generation, 1,
10091            "module respawned after the operator disabled it"
10092        );
10093    }
10094
10095    /// A `protocol: "none"` child that stays parked and exits 0 on SIGTERM:
10096    /// the shape of nats-server, the program this rule exists for.
10097    #[cfg(unix)]
10098    fn protocol_none_sigterm_exits_clean_spec(
10099        module_id: &str,
10100        dir: &std::path::Path,
10101    ) -> (ModuleSpec, PathBuf, PathBuf) {
10102        let ready = dir.join("ready");
10103        let marker = dir.join("sigterm");
10104        let spec = ModuleSpec {
10105            module_id: module_id.to_string(),
10106            program: fake_aft_stub_path(),
10107            args: Vec::new(),
10108            env: vec![
10109                ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
10110                (
10111                    "FAKE_AFT_SIGTERM_MARKER_PATH".to_string(),
10112                    marker.display().to_string(),
10113                ),
10114                (
10115                    "FAKE_AFT_NEVER_CONNECT_READY_PATH".to_string(),
10116                    ready.display().to_string(),
10117                ),
10118            ],
10119            reserved: false,
10120            reserved_prefixes: Vec::new(),
10121            protocol: ModuleProtocol::None,
10122            overlap: Default::default(),
10123        };
10124        (spec, ready, marker)
10125    }
10126
10127    /// Wait for a file the child writes, so a signal is never sent before the
10128    /// child's SIGTERM handler is installed (the default disposition would
10129    /// kill it by signal and the exit would not be clean).
10130    #[cfg(unix)]
10131    async fn wait_for_file(path: &std::path::Path) {
10132        let deadline = Instant::now() + Duration::from_secs(10);
10133        while !path.exists() {
10134            assert!(
10135                Instant::now() < deadline,
10136                "{} never appeared",
10137                path.display()
10138            );
10139            sleep(Duration::from_millis(10)).await;
10140        }
10141    }
10142
10143    /// A protocol-none module that exits 0 because something OUTSIDE the
10144    /// supervisor sent it SIGTERM is respawned, and the terminal record carries
10145    /// the crash-path disposition rather than `stopped`.
10146    #[cfg(unix)]
10147    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10148    async fn protocol_none_unrequested_clean_exit_restarts_as_a_crash() {
10149        let dir = subc_test_support::TestTempDir::new("none-unrequested-clean-exit");
10150        let (spec, ready, marker) =
10151            protocol_none_sigterm_exits_clean_spec("none-unrequested-clean-exit", dir.path());
10152        let supervisor = Supervisor::new_for_test(
10153            Arc::new(Registry::default()),
10154            RestartPolicy::new(3, Duration::ZERO),
10155        );
10156        let module = supervisor.spawn(spec).unwrap();
10157        wait_for_file(&ready).await;
10158        let first_pid = module
10159            .status()
10160            .unwrap()
10161            .pid
10162            .expect("a running module reports its pid");
10163
10164        rustix::process::kill_process(
10165            rustix::process::Pid::from_raw(i32::try_from(first_pid).unwrap()).unwrap(),
10166            rustix::process::Signal::TERM,
10167        )
10168        .unwrap();
10169
10170        let deadline = Instant::now() + Duration::from_secs(10);
10171        let respawned = loop {
10172            let status = module.status().unwrap();
10173            if status.state == ModuleState::Running
10174                && status.pid.is_some_and(|pid| pid != first_pid)
10175            {
10176                break status;
10177            }
10178            assert!(
10179                Instant::now() < deadline,
10180                "protocol-none module was not respawned after an unrequested clean exit: {status:?}"
10181            );
10182            sleep(Duration::from_millis(10)).await;
10183        };
10184        assert_eq!(respawned.spawn_generation, 2);
10185        assert!(
10186            marker.exists(),
10187            "the child must have exited through its SIGTERM handler (exit 0), or this proves nothing about clean exits"
10188        );
10189
10190        let history = module.terminal_history();
10191        assert_eq!(history.entries.len(), 1, "{history:?}");
10192        let entry = &history.entries[0];
10193        assert_eq!(entry.exit_code, Some(0));
10194        assert_eq!(entry.exit_kind, subc_control::TerminalExitKind::Clean);
10195        assert_eq!(entry.disposition, TerminalDisposition::Restarting);
10196
10197        module.stop().await.unwrap();
10198    }
10199
10200    /// Repeated unrequested clean exits of a protocol-none module spend the
10201    /// restart budget exactly as crashes do, and the module ends `failed` with
10202    /// the budget named.
10203    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10204    async fn protocol_none_repeated_clean_exits_exhaust_the_restart_budget() {
10205        let supervisor = Supervisor::new_for_test(
10206            Arc::new(Registry::default()),
10207            RestartPolicy::new(1, Duration::ZERO),
10208        );
10209        let module = supervisor
10210            .spawn(ModuleSpec {
10211                module_id: "none-clean-exit-budget".to_string(),
10212                program: fake_aft_stub_path(),
10213                args: Vec::new(),
10214                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
10215                reserved: false,
10216                reserved_prefixes: Vec::new(),
10217                protocol: ModuleProtocol::None,
10218                overlap: Default::default(),
10219            })
10220            .unwrap();
10221
10222        // Failed follows the terminal write; this deadline only bounds a hang,
10223        // not an assumed duration for the two launches or their exit recording.
10224        let deadline = Instant::now() + Duration::from_secs(10);
10225        loop {
10226            let status = module.status().unwrap();
10227            if status.state == ModuleState::Failed {
10228                break;
10229            }
10230            assert!(
10231                Instant::now() < deadline,
10232                "module never exhausted its budget: {status:?} {:?}",
10233                module.terminal_history()
10234            );
10235            sleep(Duration::from_millis(10)).await;
10236        }
10237        let history = module.terminal_history();
10238        assert_eq!(
10239            history
10240                .entries
10241                .iter()
10242                .map(|entry| (entry.exit_code, entry.disposition.clone()))
10243                .collect::<Vec<_>>(),
10244            vec![
10245                (Some(0), TerminalDisposition::Restarting),
10246                (Some(0), TerminalDisposition::Failed),
10247            ]
10248        );
10249        let detail = history.entries[1]
10250            .disposition_detail
10251            .as_deref()
10252            .expect("a budget failure names the budget");
10253        assert!(detail.contains("max_restarts=1"), "{detail}");
10254        assert_eq!(module.status().unwrap().spawn_generation, 2);
10255    }
10256
10257    #[test]
10258    fn restart_budget_failure_is_published_after_its_terminal_record() {
10259        let supervisor = Supervisor::new_for_test(
10260            Arc::new(Registry::default()),
10261            RestartPolicy::new(0, Duration::ZERO),
10262        );
10263        let runtime = supervisor.runtime_config();
10264        let spec = ModuleSpec {
10265            protocol: ModuleProtocol::None,
10266            ..windowed_crash_spec("budget-publication-order")
10267        };
10268        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(
10269            ModuleState::Running,
10270            true,
10271        )));
10272        let reaped_snapshot = snapshot.clone();
10273        let events = runtime.spawn_events.clone();
10274        let history = runtime.terminal_ring.clone();
10275
10276        // Exit recording takes the event-feed lock before the history lock.
10277        // Holding it pauses the writer after choosing a disposition but before
10278        // recording history, without assuming anything about scheduler timing.
10279        let before_record = events.0.lock().unwrap();
10280        let reap = std::thread::spawn(move || {
10281            tokio::runtime::Builder::new_current_thread()
10282                .enable_all()
10283                .build()
10284                .unwrap()
10285                .block_on(on_child_exit(
10286                    &spec,
10287                    runtime.restart_policy,
10288                    &supervisor.registry,
10289                    &reaped_snapshot,
10290                    &runtime.terminal_ring,
10291                    &runtime.spawn_events,
10292                    &runtime.child_roster,
10293                    ExitReport {
10294                        kind: ExitKind::Clean,
10295                        code: Some(0),
10296                        signal: None,
10297                        at_ms: 1,
10298                    },
10299                ))
10300        });
10301        // The deadline bounds a hung writer only; last_exit is the handshake.
10302        let deadline = Instant::now() + Duration::from_secs(10);
10303        let before_state = loop {
10304            let state = lock_snapshot(&snapshot).unwrap();
10305            if state.last_exit.is_some() {
10306                break state.state;
10307            }
10308            drop(state);
10309            assert!(Instant::now() < deadline, "exit decision was not reached");
10310            std::thread::yield_now();
10311        };
10312        let before_history = history.lock().unwrap().snapshot();
10313        drop(before_record);
10314        assert!(matches!(reap.join().unwrap(), NextAction::Stop { .. }));
10315        assert!(before_history.entries.is_empty());
10316        assert_ne!(
10317            before_state,
10318            ModuleState::Failed,
10319            "Failed was visible before its terminal record could be written"
10320        );
10321        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
10322        let history = history.lock().unwrap().snapshot();
10323        assert_eq!(history.entries.len(), 1);
10324        assert_eq!(history.entries[0].exit_code, Some(0));
10325        assert_eq!(history.entries[0].disposition, TerminalDisposition::Failed);
10326    }
10327
10328    #[tokio::test]
10329    async fn restart_budget_failure_remains_failed_when_journal_append_fails() {
10330        let dir = subc_test_support::TestTempDir::new("budget-journal-failure");
10331        let path = dir.join("terminals.jsonl");
10332        std::fs::create_dir(&path).unwrap();
10333        let supervisor = Supervisor::new_for_test(
10334            Arc::new(Registry::default()),
10335            RestartPolicy::new(0, Duration::ZERO),
10336        )
10337        .with_terminal_journal(path, "budget-journal-failure".into());
10338        let runtime = supervisor.runtime_config();
10339        let spec = windowed_crash_spec("budget-journal-failure");
10340        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10341        assert!(matches!(
10342            on_child_exit(
10343                &spec,
10344                runtime.restart_policy,
10345                &supervisor.registry,
10346                &snapshot,
10347                &runtime.terminal_ring,
10348                &runtime.spawn_events,
10349                &runtime.child_roster,
10350                crash_exit_report(1),
10351            )
10352            .await,
10353            NextAction::Stop { .. }
10354        ));
10355        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
10356        let history = runtime
10357            .terminal_ring
10358            .lock()
10359            .unwrap()
10360            .durable_history(&spec.module_id);
10361        assert!(history.journal_write_failures > 0);
10362        assert_eq!(history.entries.len(), 1);
10363        assert_eq!(history.entries[0].disposition, TerminalDisposition::Failed);
10364    }
10365
10366    #[test]
10367    fn restart_budget_failure_remains_failed_when_exit_recording_panics() {
10368        let supervisor = Supervisor::new_for_test(
10369            Arc::new(Registry::default()),
10370            RestartPolicy::new(0, Duration::ZERO),
10371        );
10372        let runtime = supervisor.runtime_config();
10373        let spec = windowed_crash_spec("budget-recording-panic");
10374        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10375        runtime.spawn_events.emit_spawned(&spec.module_id, 1, 1);
10376        // Exhausting the event sequence makes emit_exited panic before the
10377        // terminal write, exercising publication on the recording unwind.
10378        runtime.spawn_events.0.lock().unwrap().seq = u64::MAX;
10379        let result = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {
10380            tokio::runtime::Builder::new_current_thread()
10381                .enable_all()
10382                .build()
10383                .unwrap()
10384                .block_on(on_child_exit(
10385                    &spec,
10386                    runtime.restart_policy,
10387                    &supervisor.registry,
10388                    &snapshot,
10389                    &runtime.terminal_ring,
10390                    &runtime.spawn_events,
10391                    &runtime.child_roster,
10392                    crash_exit_report(1),
10393                ))
10394        }));
10395        let panic = result.err().expect("recording must still unwind");
10396        assert_eq!(
10397            panic.downcast_ref::<String>().map(String::as_str),
10398            Some("spawn event sequence exhausted")
10399        );
10400        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
10401    }
10402
10403    /// A stop the supervisor itself requests still stops a protocol-none
10404    /// module, even though the child answers the SIGTERM with exit 0.
10405    #[cfg(unix)]
10406    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10407    async fn protocol_none_requested_stop_and_disable_do_not_respawn() {
10408        for disable in [false, true] {
10409            let label = if disable {
10410                "none-requested-disable"
10411            } else {
10412                "none-requested-stop"
10413            };
10414            let dir = subc_test_support::TestTempDir::new(label);
10415            let (spec, ready, marker) = protocol_none_sigterm_exits_clean_spec(label, dir.path());
10416            let supervisor = Supervisor::new_for_test(
10417                Arc::new(Registry::default()),
10418                RestartPolicy::new(3, Duration::ZERO),
10419            );
10420            let module = supervisor.spawn(spec).unwrap();
10421            wait_for_file(&ready).await;
10422
10423            if disable {
10424                module.set_enabled(false).await.unwrap();
10425            } else {
10426                module.stop().await.unwrap();
10427            }
10428            assert!(
10429                marker.exists(),
10430                "{label}: the child must have left through its SIGTERM handler with exit 0"
10431            );
10432
10433            // Long enough for a zero-backoff respawn to have happened if the
10434            // exit had been treated as a crash.
10435            sleep(Duration::from_millis(500)).await;
10436            let status = module.status().unwrap();
10437            let expected = if disable {
10438                ModuleState::Disabled
10439            } else {
10440                ModuleState::Stopped
10441            };
10442            assert_eq!(status.state, expected, "{label}");
10443            assert_eq!(
10444                status.spawn_generation, 1,
10445                "{label}: respawned after a requested stop"
10446            );
10447            let history = module.terminal_history();
10448            assert_eq!(history.entries.len(), 1, "{label}: {history:?}");
10449            assert_eq!(history.entries[0].exit_code, Some(0), "{label}");
10450            assert_ne!(
10451                history.entries[0].disposition,
10452                TerminalDisposition::Restarting,
10453                "{label}"
10454            );
10455        }
10456    }
10457
10458    /// A subc-wire module that exits 0 on its own is still a stop: the
10459    /// protocol-none rule must not reach it.
10460    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10461    async fn subc_wire_clean_exit_is_still_a_stop() {
10462        let supervisor = Supervisor::new_for_test(
10463            Arc::new(Registry::default()),
10464            RestartPolicy::new(3, Duration::ZERO),
10465        );
10466        let module = supervisor
10467            .spawn(ModuleSpec {
10468                module_id: "wire-clean-exit".to_string(),
10469                program: fake_aft_stub_path(),
10470                args: Vec::new(),
10471                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
10472                reserved: false,
10473                reserved_prefixes: Vec::new(),
10474                protocol: ModuleProtocol::Subc,
10475                overlap: Default::default(),
10476            })
10477            .unwrap();
10478
10479        let deadline = Instant::now() + Duration::from_secs(10);
10480        while module.terminal_history().entries.is_empty() {
10481            assert!(Instant::now() < deadline, "module never exited");
10482            sleep(Duration::from_millis(10)).await;
10483        }
10484        // Long enough for a zero-backoff respawn to have happened.
10485        sleep(Duration::from_millis(500)).await;
10486        let status = module.status().unwrap();
10487        assert_eq!(status.state, ModuleState::Stopped);
10488        assert_eq!(status.spawn_generation, 1);
10489        let history = module.terminal_history();
10490        assert_eq!(history.entries.len(), 1, "{history:?}");
10491        assert_eq!(history.entries[0].exit_code, Some(0));
10492        assert_eq!(history.entries[0].disposition, TerminalDisposition::Stopped);
10493    }
10494
10495    #[cfg(unix)]
10496    #[tokio::test]
10497    async fn reload_exit_keeps_roster_until_terminal_is_recorded_during_shutdown() {
10498        let dir = subc_test_support::TestTempDir::new("reload-roster-terminal-order");
10499        let record = dir.join("live-children.json");
10500        let supervisor = Supervisor::new_for_test(
10501            Arc::new(Registry::default()),
10502            RestartPolicy::new(0, Duration::ZERO),
10503        );
10504        let mut runtime = supervisor.runtime_config();
10505        runtime.child_roster.record_to(record.clone());
10506        let gate = Arc::new(super::ReloadExitRecordGate::default());
10507        runtime.test_reload_exit_record_gate = Some(Arc::clone(&gate));
10508        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10509        let spec = ModuleSpec {
10510            module_id: "reload-exit-roster".into(),
10511            program: fake_aft_stub_path(),
10512            args: Vec::new(),
10513            env: vec![("FAKE_AFT_EXIT_CODE".into(), "23".into())],
10514            reserved: false,
10515            reserved_prefixes: Vec::new(),
10516            protocol: ModuleProtocol::Subc,
10517            overlap: Default::default(),
10518        };
10519        let mut child = None;
10520        let reload = super::finish_reload_child(
10521            &spec,
10522            &runtime,
10523            &supervisor.registry,
10524            &supervisor.process_liveness,
10525            &snapshot,
10526            &mut child,
10527        );
10528        tokio::pin!(reload);
10529        tokio::select! {
10530            result = &mut reload => panic!("reload missed the before-record gate: {result:?}"),
10531            _ = gate.reached.notified() => {}
10532        }
10533        assert!(runtime
10534            .terminal_ring
10535            .lock()
10536            .unwrap()
10537            .snapshot()
10538            .entries
10539            .is_empty());
10540        assert_eq!(
10541            crate::live_children::read_record(&record).unwrap().len(),
10542            1,
10543            "shutdown must still wait for the reaped child until its terminal record exists"
10544        );
10545        runtime.child_roster.close();
10546        gate.resume.notify_one();
10547        assert!(reload.await.is_err());
10548        assert!(crate::live_children::read_record(&record)
10549            .unwrap()
10550            .is_empty());
10551        let history = runtime.terminal_ring.lock().unwrap().snapshot();
10552        assert_eq!(history.entries.len(), 1);
10553        assert_eq!(
10554            history.entries[0].disposition,
10555            TerminalDisposition::DaemonShutdown
10556        );
10557    }
10558
10559    /// Each restart-producing arm has its own state transition. Keeping their
10560    /// lifetime count assertions adjacent prevents a later new arm from silently
10561    /// spending budget without recording the historical restart.
10562    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10563    async fn every_restart_increment_path_advances_lifetime_count() {
10564        let supervisor = Supervisor::new_for_test(
10565            Arc::new(Registry::default()),
10566            RestartPolicy::new(1, Duration::ZERO),
10567        );
10568        let runtime = supervisor.runtime_config();
10569        let spec = ModuleSpec {
10570            module_id: "lifetime-increment-path".to_string(),
10571            program: PathBuf::from("/unused/lifetime-increment-path"),
10572            args: Vec::new(),
10573            env: Vec::new(),
10574            reserved: false,
10575            reserved_prefixes: Vec::new(),
10576            protocol: ModuleProtocol::Subc,
10577            overlap: Default::default(),
10578        };
10579
10580        let crash_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10581        assert!(matches!(
10582            on_child_exit(
10583                &spec,
10584                runtime.restart_policy,
10585                &supervisor.registry,
10586                &crash_snapshot,
10587                &runtime.terminal_ring,
10588                &runtime.spawn_events,
10589                &runtime.child_roster,
10590                ExitReport {
10591                    kind: ExitKind::Crash,
10592                    code: Some(1),
10593                    signal: None,
10594                    at_ms: 1,
10595                },
10596            )
10597            .await,
10598            NextAction::Restart { schedule: _ }
10599        ));
10600        let (crash_restarts, crash_lifetime) = {
10601            let state = lock_snapshot(&crash_snapshot).unwrap();
10602            (state.crash_restarts.len(), state.lifetime_restarts)
10603        };
10604        assert_eq!(crash_restarts, 1);
10605        assert_eq!(crash_lifetime, 1);
10606
10607        let health_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10608        let mut health_child = None;
10609        assert!(matches!(
10610            health_restart_child(
10611                &spec,
10612                &runtime,
10613                &supervisor.registry,
10614                &supervisor.process_liveness,
10615                &health_snapshot,
10616                &mut health_child,
10617                SupervisorHealthStatus::Failing,
10618                None,
10619                2,
10620            )
10621            .await,
10622            Ok(())
10623        ));
10624        assert!(health_child.is_none());
10625        assert!(lock_snapshot(&health_snapshot).unwrap().respawn_pending);
10626        assert!(runtime.scheduled_respawn.lock().unwrap().take().is_some());
10627        let (health_restarts, health_lifetime) = {
10628            let state = lock_snapshot(&health_snapshot).unwrap();
10629            (state.crash_restarts.len(), state.lifetime_restarts)
10630        };
10631        assert_eq!(health_restarts, 1);
10632        assert_eq!(health_lifetime, 1);
10633
10634        let reload_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10635        let mut reload_child = None;
10636        assert!(matches!(
10637            handle_reload_spawn_failure(
10638                &spec,
10639                &runtime,
10640                &supervisor.process_liveness,
10641                &reload_snapshot,
10642                &mut reload_child,
10643                "forced reload spawn failure".to_string(),
10644            )
10645            .await,
10646            Err(SuperviseError::ReloadFailed { .. })
10647        ));
10648        let (reload_restarts, reload_lifetime) = {
10649            let state = lock_snapshot(&reload_snapshot).unwrap();
10650            (state.crash_restarts.len(), state.lifetime_restarts)
10651        };
10652        assert_eq!(reload_restarts, 1);
10653        assert_eq!(reload_lifetime, 1);
10654    }
10655
10656    #[tokio::test]
10657    async fn deliberately_severed_live_child_records_lifetime_without_spending_restart_budget() {
10658        let supervisor = Supervisor::new_for_test(
10659            Arc::new(Registry::default()),
10660            RestartPolicy::new(3, Duration::ZERO),
10661        );
10662        let runtime = supervisor.runtime_config();
10663        let spec = ModuleSpec {
10664            module_id: "deliberately-severed".to_string(),
10665            program: PathBuf::from("/unused/deliberately-severed"),
10666            args: Vec::new(),
10667            env: Vec::new(),
10668            reserved: false,
10669            reserved_prefixes: Vec::new(),
10670            protocol: ModuleProtocol::Subc,
10671            overlap: Default::default(),
10672        };
10673        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10674        let process = ProcessIdentity {
10675            pid: 41,
10676            start_time: 101,
10677        };
10678        record_deliberate_severance(&snapshot, process).unwrap();
10679        let exit_report = apply_deliberate_severance_marker(
10680            &snapshot,
10681            Some(process),
10682            ExitReport {
10683                kind: ExitKind::Crash,
10684                code: Some(1),
10685                signal: None,
10686                at_ms: 1,
10687            },
10688        );
10689        assert_eq!(exit_report.kind, ExitKind::DeliberateSeverance);
10690
10691        assert!(matches!(
10692            on_child_exit(
10693                &spec,
10694                runtime.restart_policy,
10695                &supervisor.registry,
10696                &snapshot,
10697                &runtime.terminal_ring,
10698                &runtime.spawn_events,
10699                &runtime.child_roster,
10700                exit_report,
10701            )
10702            .await,
10703            NextAction::Restart { schedule: _ }
10704        ));
10705        let state = lock_snapshot(&snapshot).unwrap();
10706        assert_eq!(state.lifetime_restarts, 1);
10707        assert_eq!(state.crash_restarts.len(), 0);
10708    }
10709
10710    #[tokio::test]
10711    async fn genuine_crash_spends_restart_budget_and_records_lifetime() {
10712        let supervisor = Supervisor::new_for_test(
10713            Arc::new(Registry::default()),
10714            RestartPolicy::new(3, Duration::ZERO),
10715        );
10716        let runtime = supervisor.runtime_config();
10717        let spec = ModuleSpec {
10718            module_id: "genuine-crash".to_string(),
10719            program: PathBuf::from("/unused/genuine-crash"),
10720            args: Vec::new(),
10721            env: Vec::new(),
10722            reserved: false,
10723            reserved_prefixes: Vec::new(),
10724            protocol: ModuleProtocol::Subc,
10725            overlap: Default::default(),
10726        };
10727        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10728
10729        assert!(matches!(
10730            on_child_exit(
10731                &spec,
10732                runtime.restart_policy,
10733                &supervisor.registry,
10734                &snapshot,
10735                &runtime.terminal_ring,
10736                &runtime.spawn_events,
10737                &runtime.child_roster,
10738                ExitReport {
10739                    kind: ExitKind::Crash,
10740                    code: Some(1),
10741                    signal: None,
10742                    at_ms: 1,
10743                },
10744            )
10745            .await,
10746            NextAction::Restart { schedule: _ }
10747        ));
10748        let state = lock_snapshot(&snapshot).unwrap();
10749        assert_eq!(state.lifetime_restarts, 1);
10750        assert_eq!(state.crash_restarts.len(), 1);
10751    }
10752
10753    fn crash_exit_report(at_ms: u64) -> ExitReport {
10754        ExitReport {
10755            kind: ExitKind::Crash,
10756            code: Some(1),
10757            signal: None,
10758            at_ms,
10759        }
10760    }
10761
10762    fn windowed_crash_spec(module_id: &str) -> ModuleSpec {
10763        ModuleSpec {
10764            module_id: module_id.to_string(),
10765            program: PathBuf::from("/unused").join(module_id),
10766            args: Vec::new(),
10767            env: Vec::new(),
10768            reserved: false,
10769            reserved_prefixes: Vec::new(),
10770            protocol: ModuleProtocol::Subc,
10771            overlap: Default::default(),
10772        }
10773    }
10774
10775    /// A real crash loop still stops. Three crashes with nothing aging out spend
10776    /// a budget of two and the third respawn is refused, and both surfaces an
10777    /// operator has -- the log line and the retained terminal record -- name the
10778    /// window rather than only the cap, because `max_restarts=2` alone is what
10779    /// this budget used to mean.
10780    #[tokio::test]
10781    async fn three_crashes_inside_the_window_stop_the_module_and_name_the_window() {
10782        let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::ERROR);
10783        let supervisor = Supervisor::new_for_test(
10784            Arc::new(Registry::default()),
10785            RestartPolicy::new(2, Duration::ZERO),
10786        );
10787        let runtime = supervisor.runtime_config();
10788        let spec = windowed_crash_spec("crash-loop-in-window");
10789        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10790
10791        for attempt in 1..=2 {
10792            assert!(
10793                matches!(
10794                    on_child_exit(
10795                        &spec,
10796                        runtime.restart_policy,
10797                        &supervisor.registry,
10798                        &snapshot,
10799                        &runtime.terminal_ring,
10800                        &runtime.spawn_events,
10801                        &runtime.child_roster,
10802                        crash_exit_report(attempt),
10803                    )
10804                    .await,
10805                    NextAction::Restart { schedule: _ }
10806                ),
10807                "crash {attempt} is inside the budget and must respawn"
10808            );
10809        }
10810
10811        assert!(matches!(
10812            on_child_exit(
10813                &spec,
10814                runtime.restart_policy,
10815                &supervisor.registry,
10816                &snapshot,
10817                &runtime.terminal_ring,
10818                &runtime.spawn_events,
10819                &runtime.child_roster,
10820                crash_exit_report(3),
10821            )
10822            .await,
10823            NextAction::Stop { .. }
10824        ));
10825
10826        {
10827            let state = lock_snapshot(&snapshot).unwrap();
10828            assert_eq!(state.state, ModuleState::Failed);
10829            assert_eq!(state.crash_restarts.len(), 2);
10830            assert_eq!(state.lifetime_restarts, 2);
10831        }
10832
10833        let history = runtime
10834            .terminal_ring
10835            .lock()
10836            .expect("terminal ring is not poisoned")
10837            .snapshot();
10838        let last = history
10839            .entries
10840            .last()
10841            .expect("the refused crash is retained");
10842        assert_eq!(last.disposition, TerminalDisposition::Failed);
10843        assert_eq!(
10844            last.disposition_detail.as_deref(),
10845            Some("crash budget exhausted: max_restarts=2 within window_secs=600")
10846        );
10847
10848        let captured = crate::router::test_log::captured_logs(&logs);
10849        assert!(
10850            captured.contains("crash budget exhausted: max_restarts=2 within window_secs=600"),
10851            "the stop must be logged with its window: {captured}"
10852        );
10853    }
10854
10855    /// The rate, stated as a test: three crashes where the first has aged past
10856    /// the window are two crashes as far as the budget is concerned, so the
10857    /// third respawn is allowed and the ring holds only the two recent ones.
10858    ///
10859    /// This is the case a lifetime counter got wrong -- and the case the daemon
10860    /// now hits routinely, since a module exits non-zero every time its
10861    /// connection to the daemon drops.
10862    #[tokio::test]
10863    async fn a_crash_older_than_the_window_frees_its_slot_for_a_later_crash() {
10864        let supervisor = Supervisor::new_for_test(
10865            Arc::new(Registry::default()),
10866            RestartPolicy::new(2, Duration::ZERO),
10867        );
10868        let runtime = supervisor.runtime_config();
10869        let spec = windowed_crash_spec("crash-across-windows");
10870        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10871
10872        for attempt in 1..=2 {
10873            assert!(matches!(
10874                on_child_exit(
10875                    &spec,
10876                    runtime.restart_policy,
10877                    &supervisor.registry,
10878                    &snapshot,
10879                    &runtime.terminal_ring,
10880                    &runtime.spawn_events,
10881                    &runtime.child_roster,
10882                    crash_exit_report(attempt),
10883                )
10884                .await,
10885                NextAction::Restart { schedule: _ }
10886            ));
10887        }
10888
10889        // The oldest crash moves out of the window; nothing else about the
10890        // module changes.
10891        update_snapshot(&snapshot, Some(&spec.module_id), |state| {
10892            age_oldest_crash_restart_out_of_window(state, runtime.restart_policy.window);
10893        })
10894        .unwrap();
10895
10896        assert!(
10897            matches!(
10898                on_child_exit(
10899                    &spec,
10900                    runtime.restart_policy,
10901                    &supervisor.registry,
10902                    &snapshot,
10903                    &runtime.terminal_ring,
10904                    &runtime.spawn_events,
10905                    &runtime.child_roster,
10906                    crash_exit_report(3),
10907                )
10908                .await,
10909                NextAction::Restart { schedule: _ }
10910            ),
10911            "a crash older than the window must not hold a budget slot"
10912        );
10913
10914        let state = lock_snapshot(&snapshot).unwrap();
10915        assert_eq!(state.state, ModuleState::Restarting);
10916        assert_eq!(
10917            state.crash_restarts.len(),
10918            2,
10919            "the aged instant is dropped and the new one takes its place"
10920        );
10921        assert_eq!(
10922            state.lifetime_restarts, 3,
10923            "the ledger counts every restart, including the ones the window forgot"
10924        );
10925    }
10926
10927    /// An operator restart hands the budget back whole, and the ledger keeps
10928    /// counting. Those are different questions -- "how close is this module to
10929    /// being stopped" and "how many times has it been replaced" -- and the
10930    /// operator action answers only the first.
10931    #[tokio::test]
10932    async fn an_operator_restart_clears_the_ring_and_leaves_the_ledger_alone() {
10933        let supervisor = Supervisor::new_for_test(
10934            Arc::new(Registry::default()),
10935            RestartPolicy::new(2, Duration::ZERO),
10936        );
10937        let runtime = supervisor.runtime_config();
10938        let spec = windowed_crash_spec("operator-cleared-budget");
10939        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10940
10941        for attempt in 1..=2 {
10942            assert!(matches!(
10943                on_child_exit(
10944                    &spec,
10945                    runtime.restart_policy,
10946                    &supervisor.registry,
10947                    &snapshot,
10948                    &runtime.terminal_ring,
10949                    &runtime.spawn_events,
10950                    &runtime.child_roster,
10951                    crash_exit_report(attempt),
10952                )
10953                .await,
10954                NextAction::Restart { schedule: _ }
10955            ));
10956        }
10957
10958        reset_restart_count(&snapshot, &spec.module_id).unwrap();
10959        {
10960            let state = lock_snapshot(&snapshot).unwrap();
10961            assert!(
10962                state.crash_restarts.is_empty(),
10963                "an operator restart returns the full budget"
10964            );
10965            assert_eq!(
10966                state.lifetime_restarts, 2,
10967                "clearing the budget must not unmake the crashes"
10968            );
10969        }
10970
10971        assert!(
10972            matches!(
10973                on_child_exit(
10974                    &spec,
10975                    runtime.restart_policy,
10976                    &supervisor.registry,
10977                    &snapshot,
10978                    &runtime.terminal_ring,
10979                    &runtime.spawn_events,
10980                    &runtime.child_roster,
10981                    crash_exit_report(3),
10982                )
10983                .await,
10984                NextAction::Restart { schedule: _ }
10985            ),
10986            "the cleared budget must be spendable again"
10987        );
10988        let state = lock_snapshot(&snapshot).unwrap();
10989        assert_eq!(state.crash_restarts.len(), 1);
10990        assert_eq!(state.lifetime_restarts, 3);
10991    }
10992
10993    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10994    async fn severance_marker_for_a_dead_child_does_not_label_its_successor() {
10995        let severed = ProcessIdentity {
10996            pid: 41,
10997            start_time: 101,
10998        };
10999        let successor = ProcessIdentity {
11000            pid: 41,
11001            start_time: 202,
11002        };
11003        let module = module_with_recovery_snapshot(ModuleState::Running, true, 0);
11004        update_snapshot(&module.inner.snapshot, Some("recovery-snapshot"), |state| {
11005            state.pid = Some(successor.pid);
11006            state.process_start_time = Some(successor.start_time);
11007        })
11008        .unwrap();
11009        assert!(!module.record_deliberate_severance(severed).unwrap());
11010
11011        let exit_report = apply_deliberate_severance_marker(
11012            &module.inner.snapshot,
11013            Some(successor),
11014            ExitReport {
11015                kind: ExitKind::Crash,
11016                code: Some(1),
11017                signal: None,
11018                at_ms: 1,
11019            },
11020        );
11021
11022        assert_eq!(exit_report.kind, ExitKind::Crash);
11023    }
11024
11025    #[tokio::test]
11026    async fn drain_reap_marks_deliberate_severance_and_records_lifetime_without_budget() {
11027        let registry = Registry::default();
11028        let supervisor = Supervisor::new_for_test(
11029            Arc::new(Registry::default()),
11030            RestartPolicy::new(3, Duration::ZERO),
11031        );
11032        let runtime = supervisor.runtime_config();
11033        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
11034        let spec = ModuleSpec {
11035            module_id: "drain-deliberate-severance".to_string(),
11036            program: fake_aft_stub_path(),
11037            args: Vec::new(),
11038            env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
11039            reserved: false,
11040            reserved_prefixes: Vec::new(),
11041            protocol: ModuleProtocol::Subc,
11042            overlap: Default::default(),
11043        };
11044        let mut child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
11045        let process = ProcessIdentity {
11046            pid: 41,
11047            start_time: 101,
11048        };
11049        child.process_identity = Some(process);
11050        update_snapshot(&snapshot, Some(&spec.module_id), |state| {
11051            state.pid = Some(process.pid);
11052            state.process_start_time = Some(process.start_time);
11053        })
11054        .unwrap();
11055        record_deliberate_severance(&snapshot, process).unwrap();
11056
11057        drain_child_to_state(
11058            &spec.module_id,
11059            spec.protocol,
11060            // The child exits on its own; no signal may change the exit this
11061            // test classifies.
11062            StopNotice::SentOverConnection,
11063            &registry,
11064            None,
11065            &snapshot,
11066            &runtime.terminal_ring,
11067            &runtime.spawn_events,
11068            child,
11069            Duration::from_secs(1),
11070            ModuleState::Stopped,
11071            Some(false),
11072        )
11073        .await
11074        .unwrap();
11075
11076        let state = lock_snapshot(&snapshot).unwrap();
11077        assert_eq!(
11078            state.last_exit.as_ref().map(|exit| exit.kind),
11079            Some(ExitKind::DeliberateSeverance)
11080        );
11081        assert_eq!(state.lifetime_restarts, 1);
11082        assert_eq!(state.crash_restarts.len(), 0);
11083        drop(state);
11084        let history = runtime.terminal_ring.lock().unwrap().snapshot();
11085        assert_eq!(
11086            history.entries[0].exit_kind,
11087            subc_control::TerminalExitKind::DeliberateSeverance
11088        );
11089    }
11090
11091    #[tokio::test]
11092    async fn ordinary_drain_reap_does_not_record_a_lifetime_restart() {
11093        let registry = Registry::default();
11094        let supervisor = Supervisor::new_for_test(
11095            Arc::new(Registry::default()),
11096            RestartPolicy::new(3, Duration::ZERO),
11097        );
11098        let runtime = supervisor.runtime_config();
11099        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
11100        let spec = ModuleSpec {
11101            module_id: "ordinary-drain".to_string(),
11102            program: fake_aft_stub_path(),
11103            args: Vec::new(),
11104            env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
11105            reserved: false,
11106            reserved_prefixes: Vec::new(),
11107            protocol: ModuleProtocol::Subc,
11108            overlap: Default::default(),
11109        };
11110        let child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
11111
11112        drain_child_to_state(
11113            &spec.module_id,
11114            spec.protocol,
11115            // The child exits on its own; no signal may change the exit this
11116            // test classifies.
11117            StopNotice::SentOverConnection,
11118            &registry,
11119            None,
11120            &snapshot,
11121            &runtime.terminal_ring,
11122            &runtime.spawn_events,
11123            child,
11124            Duration::from_secs(1),
11125            ModuleState::Stopped,
11126            Some(false),
11127        )
11128        .await
11129        .unwrap();
11130
11131        let state = lock_snapshot(&snapshot).unwrap();
11132        assert_eq!(
11133            state.last_exit.as_ref().map(|exit| exit.kind),
11134            Some(ExitKind::Crash)
11135        );
11136        assert_eq!(state.lifetime_restarts, 0);
11137        assert_eq!(state.crash_restarts.len(), 0);
11138    }
11139
11140    #[test]
11141    fn fatal_connection_teardown_cannot_arm_a_marker_for_a_surviving_process() {
11142        // The server's generic fatal-routing branch only knows that the
11143        // connection failed; it does not know that the daemon deliberately
11144        // initiated a process-killing severance. Keep this seam explicit so a
11145        // future connection error path cannot silently reintroduce the stale
11146        // exemption that mislabels a later genuine crash.
11147        assert!(!include_str!("server.rs")
11148            .contains("router.record_deliberate_connection_severance(ctx.connection_id)"));
11149    }
11150
11151    /// The `route.closed` `drained` value must be the quiescence wait's own
11152    /// measurement (`Ok`), never invented -- except on `Err`, where there is no
11153    /// measurement at all and `false` is the one honest constant. This is the exact
11154    /// logic `begin_forwarding_drain_with` now applies before sending `route.closed`
11155    /// on every return path, including the one that used to return early via `?`
11156    /// with `route.closing` already sent and no `route.closed` ever following.
11157    #[test]
11158    fn drained_after_quiescence_wait_passes_ok_through_and_forces_false_on_err() {
11159        assert!(drained_after_quiescence_wait(&Ok(true)));
11160        assert!(!drained_after_quiescence_wait(&Ok(false)));
11161        assert!(!drained_after_quiescence_wait(&Err(
11162            SuperviseError::StatePoisoned { module_id: None }
11163        )));
11164    }
11165
11166    /// `supervise_loop`'s `wait()`-error arm now calls `record_terminal` like every
11167    /// other exit path does, so a module whose child `wait()` itself errored (e.g.
11168    /// already reaped out-of-band) still leaves a terminal record rather than none
11169    /// at all. Triggering the real `wait()` I/O error from an integration test would
11170    /// need a genuine already-reaped-child race, which is OS-specific and not
11171    /// something this suite attempts elsewhere; this test instead verifies the
11172    /// record produced for that arm end-to-end through the real `TerminalRing`, and
11173    /// the call site itself is verified by inspection to sit in that exact arm.
11174    #[test]
11175    fn wait_error_exit_report_records_a_failed_terminal_with_no_code_or_signal() {
11176        let ring = Arc::new(Mutex::new(TerminalRing::new(
11177            TerminalRingConfig::default(),
11178            0,
11179        )));
11180        record_wait_error_terminal("wait-error", &ring, &super::SpawnEventFeed::default());
11181
11182        let snapshot = ring.lock().unwrap().snapshot();
11183        assert_eq!(snapshot.entries.len(), 1);
11184        let entry = &snapshot.entries[0];
11185        assert_eq!(entry.exit_code, None);
11186        assert_eq!(entry.exit_signal, None);
11187        assert_eq!(entry.disposition, TerminalDisposition::Failed);
11188    }
11189
11190    #[test]
11191    fn wait_error_exit_path_preserves_spawn_event_density() {
11192        let feed = super::SpawnEventFeed::default();
11193        feed.configure_incarnation("wait-error-density".to_string());
11194        feed.emit_spawned("wait-error", 41, 1);
11195        let ring = Arc::new(Mutex::new(TerminalRing::new(
11196            TerminalRingConfig::default(),
11197            0,
11198        )));
11199
11200        record_wait_error_terminal("wait-error", &ring, &feed);
11201        feed.emit_spawned("after-wait-error", 42, 2);
11202
11203        let state = feed.0.lock().unwrap();
11204        let sequences = state
11205            .events
11206            .iter()
11207            .map(|event| event.cursor.seq)
11208            .collect::<Vec<_>>();
11209        assert_eq!(sequences, vec![1, 2, 3]);
11210        assert_eq!(state.events[1].kind, SpawnEventKind::Exited);
11211        assert_eq!(state.events[1].exit_code, None);
11212        assert_eq!(state.events[1].exit_signal, None);
11213    }
11214
11215    /// Pins the report's `kind` too: the wait-error arm treats an unwaitable child
11216    /// as a crash (matching `fail_snapshot`'s `Failed` disposition for this arm),
11217    /// not a clean exit it never actually observed.
11218    #[test]
11219    fn wait_error_exit_report_is_classified_as_a_crash() {
11220        assert_eq!(wait_error_exit_report().kind, ExitKind::Crash);
11221    }
11222}
11223
11224#[cfg(test)]
11225mod health_evidence_tests {
11226    use super::{HealthProbeError, HealthProbeEvidence};
11227    use std::collections::HashSet;
11228
11229    /// The evidential asymmetry, asserted rather than described.
11230    ///
11231    /// Exactly ONE observation is proof a module cannot serve, and the one that
11232    /// fires under CPU starvation is not it. Before the split, all fifteen
11233    /// construction sites collapsed into a single String, so a timeout carried the
11234    /// same weight as a dead lane -- which is how a healthy module was restarted
11235    /// three times in one day.
11236    #[test]
11237    fn only_a_dead_lane_is_proof_of_death() {
11238        assert!(HealthProbeError::lane_dead("gone").is_proof_of_death());
11239        // Three non-proof classes, each for a different reason: silence is
11240        // consistent with health, a bad answer proves the module ALIVE, and a
11241        // daemon-side fault never reached the module at all.
11242        assert!(!HealthProbeError::no_answer("timed out").is_proof_of_death());
11243        assert!(!HealthProbeError::bad_answer("garbage").is_proof_of_death());
11244        assert!(!HealthProbeError::misconfigured("no table").is_proof_of_death());
11245    }
11246
11247    /// Labels must be distinct, or the operator-facing distinction is cosmetic.
11248    ///
11249    /// A shared label renders two different observations identically in the line an
11250    /// operator reads after an unexplained restart -- the exact confusion this
11251    /// change removes.
11252    #[test]
11253    fn every_evidence_class_has_a_distinct_label() {
11254        let labels = [
11255            HealthProbeError::lane_dead("").label(),
11256            HealthProbeError::no_answer("").label(),
11257            HealthProbeError::bad_answer("").label(),
11258            HealthProbeError::misconfigured("").label(),
11259        ];
11260        let unique: HashSet<_> = labels.iter().collect();
11261        assert_eq!(unique.len(), labels.len(), "labels collided: {labels:?}");
11262    }
11263
11264    /// The class is additional information, not a replacement.
11265    ///
11266    /// An operator needs both "this was silence" and the specific text saying how
11267    /// long we waited; a classification that swallowed the message would trade one
11268    /// missing distinction for another.
11269    #[test]
11270    fn classification_preserves_the_original_message() {
11271        let err = HealthProbeError::no_answer("module did not answer within 5s");
11272        assert_eq!(err.to_string(), "module did not answer within 5s");
11273        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
11274    }
11275}
11276
11277#[cfg(test)]
11278mod health_tombstone_tests {
11279    use std::{path::PathBuf, sync::Arc, time::Duration};
11280
11281    use subc_protocol::{
11282        manifest::Concurrency,
11283        session::{HealthStatus, ModuleControlResponse},
11284    };
11285    use tokio::sync::mpsc;
11286
11287    use super::{
11288        probe_module_health, HealthAction, HealthConfig, HealthProbeEvidence, ModuleProtocol,
11289        ModuleSpec, RestartPolicy, Supervisor, SupervisorRuntimeConfig,
11290    };
11291    use crate::{
11292        control::ControlHandler,
11293        forwarding::{ForwardingTable, ModuleControlRpcCompletion, ModuleControlRpcOutcome},
11294        registry::{ConnectionId, Registry},
11295        router::FrameSink,
11296    };
11297
11298    struct ProbeHarness {
11299        spec: ModuleSpec,
11300        runtime: SupervisorRuntimeConfig,
11301        forwarding: Arc<ForwardingTable>,
11302        module_connection: ConnectionId,
11303        module_rx: mpsc::Receiver<crate::router::OutboundFrame>,
11304        handler: ControlHandler,
11305        module: super::SupervisedModule,
11306    }
11307
11308    fn probe_harness() -> ProbeHarness {
11309        let registry = Arc::new(Registry::default());
11310        let forwarding = Arc::new(ForwardingTable::default());
11311        let supervisor_handle = super::SupervisorHandle::new();
11312        let health = HealthConfig {
11313            http: None,
11314            cadence: Duration::from_secs(30),
11315            deadline: Duration::from_secs(5),
11316            failure_threshold: 3,
11317            on_degraded: HealthAction::Report,
11318            on_failing: HealthAction::Report,
11319            critical: false,
11320        };
11321        let supervisor = Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::default())
11322            .with_forwarding(Arc::clone(&forwarding))
11323            .with_handle(supervisor_handle.clone())
11324            .with_health_config(health);
11325        let spec = ModuleSpec {
11326            module_id: "late-health-module".to_string(),
11327            program: PathBuf::from("disabled-module"),
11328            args: Vec::new(),
11329            env: Vec::new(),
11330            reserved: false,
11331            reserved_prefixes: Vec::new(),
11332            protocol: ModuleProtocol::Subc,
11333            overlap: Default::default(),
11334        };
11335        let module = supervisor
11336            .supervise_configured(spec.clone(), false)
11337            .unwrap();
11338        let runtime = supervisor.runtime_config();
11339        let handler = ControlHandler::with_forwarding(registry, Arc::clone(&forwarding))
11340            .with_supervisor(supervisor_handle);
11341        let module_connection = ConnectionId::new(700);
11342        let (module_tx, module_rx) = mpsc::channel(8);
11343        forwarding
11344            .register_module_connection(
11345                module_connection,
11346                spec.module_id.clone(),
11347                subc_protocol::PROTOCOL_VERSION,
11348                Concurrency::ModuleManaged,
11349                FrameSink::new(module_tx),
11350            )
11351            .unwrap();
11352
11353        ProbeHarness {
11354            spec,
11355            runtime,
11356            forwarding,
11357            module_connection,
11358            module_rx,
11359            handler,
11360            module,
11361        }
11362    }
11363
11364    async fn finish_after(
11365        harness: &mut ProbeHarness,
11366        stall: Duration,
11367    ) -> ModuleControlRpcCompletion {
11368        assert!(stall > harness.runtime.health.deadline);
11369        let deadline = harness.runtime.health.deadline;
11370        let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
11371        let answer = async {
11372            let frame = harness.module_rx.recv().await.expect("health.check frame");
11373            tokio::time::advance(deadline).await;
11374            tokio::task::yield_now().await;
11375            tokio::time::advance(stall - deadline).await;
11376            harness
11377                .forwarding
11378                .complete_module_control_rpc(
11379                    harness.module_connection,
11380                    frame.header.corr,
11381                    Some("health.check"),
11382                    ModuleControlRpcOutcome::Response(ModuleControlResponse::HealthCheck {
11383                        status: HealthStatus::Ok,
11384                        detail: None,
11385                        metrics: None,
11386                    }),
11387                )
11388                .unwrap()
11389        };
11390        let (probe_result, completion) = tokio::join!(probe, answer);
11391        let err = probe_result.expect_err("probe must miss its deadline");
11392        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
11393        completion
11394    }
11395
11396    async fn time_out_without_answer(harness: &mut ProbeHarness) {
11397        let deadline = harness.runtime.health.deadline;
11398        let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
11399        let exhaust_deadline = async {
11400            let _frame = harness.module_rx.recv().await.expect("health.check frame");
11401            tokio::time::advance(deadline).await;
11402            tokio::task::yield_now().await;
11403        };
11404        let (probe_result, ()) = tokio::join!(probe, exhaust_deadline);
11405        let err = probe_result.expect_err("probe must miss its deadline");
11406        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
11407    }
11408
11409    async fn run_probe_cycle(harness: &mut ProbeHarness, answer: bool) {
11410        let registry = Arc::clone(&harness.module.inner.registry);
11411        let snapshot = Arc::clone(&harness.module.inner.snapshot);
11412        let process_liveness = super::SupervisorProcessLiveness::default();
11413        let mut child = None;
11414        let cycle = super::run_health_probe_cycle(
11415            &harness.spec,
11416            &harness.runtime,
11417            &registry,
11418            &process_liveness,
11419            &snapshot,
11420            &mut child,
11421        );
11422        let peer = async {
11423            let frame = harness.module_rx.recv().await.expect("health.check frame");
11424            if answer {
11425                harness
11426                    .forwarding
11427                    .complete_module_control_rpc(
11428                        harness.module_connection,
11429                        frame.header.corr,
11430                        Some("health.check"),
11431                        ModuleControlRpcOutcome::Response(ModuleControlResponse::HealthCheck {
11432                            status: HealthStatus::Ok,
11433                            detail: None,
11434                            metrics: Some(serde_json::json!({"ready": true})),
11435                        }),
11436                    )
11437                    .unwrap();
11438            } else {
11439                tokio::time::advance(harness.runtime.health.deadline).await;
11440                tokio::task::yield_now().await;
11441            }
11442        };
11443        tokio::join!(cycle, peer);
11444    }
11445
11446    #[tokio::test(start_paused = true)]
11447    async fn unanswered_probe_is_unknown_until_threshold_and_ok_report_recovers() {
11448        let mut harness = probe_harness();
11449        // Drive the probe cycle directly with an in-memory wire peer. Stop the
11450        // disabled module's monitor so only this test owns lifecycle transitions;
11451        // no OS process is launched, and a restart is observed at scheduling.
11452        let monitor = harness.module.inner.monitor.lock().unwrap().take().unwrap();
11453        monitor.abort();
11454        let _ = monitor.await;
11455        super::update_snapshot(&harness.module.inner.snapshot, None, |state| {
11456            state.enabled = true;
11457            state.state = super::ModuleState::Running;
11458            state.process_alive = true;
11459        })
11460        .unwrap();
11461
11462        run_probe_cycle(&mut harness, true).await;
11463        assert_eq!(
11464            harness.module.status().unwrap().health.status,
11465            super::SupervisorHealthStatus::Ok
11466        );
11467
11468        for failures in 1..harness.runtime.health.failure_threshold {
11469            run_probe_cycle(&mut harness, false).await;
11470            let status = harness.module.status().unwrap();
11471            assert_eq!(status.health.status, super::SupervisorHealthStatus::Unknown);
11472            assert_eq!(status.health.consecutive_failures, failures);
11473            assert!(status.health.last_probe_ms.is_some());
11474            assert!(status.health.detail.unwrap().starts_with("[no-answer]"));
11475            assert!(status.health.metrics.is_none());
11476            assert_eq!(status.state, super::ModuleState::Running);
11477            assert!(status.process_alive);
11478            assert_eq!(status.restart_count, 0);
11479            assert_eq!(status.lifetime_restarts, 0);
11480            assert!(status.health.last_action.is_none());
11481            assert!(harness.runtime.scheduled_respawn.lock().unwrap().is_none());
11482        }
11483
11484        run_probe_cycle(&mut harness, true).await;
11485        let recovered = harness.module.status().unwrap();
11486        assert_eq!(recovered.health.status, super::SupervisorHealthStatus::Ok);
11487        assert_eq!(recovered.health.consecutive_failures, 0);
11488        assert!(recovered.health.detail.is_none());
11489        assert_eq!(
11490            recovered.health.metrics,
11491            Some(serde_json::json!({"ready": true}))
11492        );
11493        assert_eq!(recovered.lifetime_restarts, 0);
11494
11495        for failures in 1..=harness.runtime.health.failure_threshold {
11496            run_probe_cycle(&mut harness, false).await;
11497            let status = harness.module.status().unwrap();
11498            assert_eq!(status.health.consecutive_failures, failures);
11499            if failures < harness.runtime.health.failure_threshold {
11500                assert_eq!(status.health.status, super::SupervisorHealthStatus::Unknown);
11501                assert_eq!(status.state, super::ModuleState::Running);
11502                assert_eq!(status.lifetime_restarts, 0);
11503                assert!(harness.runtime.scheduled_respawn.lock().unwrap().is_none());
11504            } else {
11505                assert_eq!(
11506                    status.health.status,
11507                    super::SupervisorHealthStatus::Unresponsive
11508                );
11509                assert_eq!(status.state, super::ModuleState::Restarting);
11510                assert_eq!(status.restart_count, 1);
11511                assert_eq!(status.lifetime_restarts, 1);
11512                assert_eq!(status.health.last_action.as_deref(), Some("restart"));
11513                assert!(harness.runtime.scheduled_respawn.lock().unwrap().is_some());
11514            }
11515        }
11516    }
11517
11518    #[tokio::test(start_paused = true)]
11519    async fn late_health_answers_record_start_anchored_latency_for_two_stalls() {
11520        let mut harness = probe_harness();
11521
11522        let first = finish_after(&mut harness, Duration::from_secs(8)).await;
11523        let first_latency = match &first {
11524            ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
11525            other => panic!("late answer was not retained: {other:?}"),
11526        };
11527        assert!(harness.handler.observe_module_control_completion(first));
11528
11529        let second = finish_after(&mut harness, Duration::from_secs(11)).await;
11530        let second_latency = match &second {
11531            ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
11532            other => panic!("late answer was not retained: {other:?}"),
11533        };
11534        assert!(harness.handler.observe_module_control_completion(second));
11535
11536        assert_eq!(first_latency, Duration::from_secs(8));
11537        assert_eq!(
11538            second_latency - first_latency,
11539            Duration::from_secs(3),
11540            "latency must grow linearly with the additional stall"
11541        );
11542        let health = harness.module.status().unwrap().health;
11543        assert_eq!(health.late_answer_count, 2);
11544        assert_eq!(health.last_late_answer_latency_ms, Some(11_000));
11545    }
11546
11547    /// A module that answers every probe late must never march to the kill
11548    /// threshold: the late answer proves it is alive, so it must clear the miss
11549    /// streak the timeout recorded. Without the reset, a CPU-starved module
11550    /// that serves every probe seconds past the deadline accumulates
11551    /// `consecutive_failures` to the threshold and is killed — the exact
11552    /// sequence from the 2026-08-14 aft disable, where the daemon logged
11553    /// "proves the module is alive" five times while counting five misses.
11554    #[tokio::test(start_paused = true)]
11555    async fn late_answer_clears_the_consecutive_failure_streak() {
11556        let mut harness = probe_harness();
11557
11558        // Timeout recorded first: the probe path saw no answer in time.
11559        time_out_without_answer(&mut harness).await;
11560        harness
11561            .module
11562            .record_health_probe_failure_for_test("[no-answer] test miss")
11563            .unwrap();
11564        assert_eq!(
11565            harness.module.status().unwrap().health.consecutive_failures,
11566            1,
11567            "precondition: the miss must be on the streak before the late answer"
11568        );
11569
11570        // The stalled reply then lands: proof of life.
11571        let late = finish_after(&mut harness, Duration::from_secs(9)).await;
11572        assert!(matches!(
11573            late,
11574            ModuleControlRpcCompletion::LateHealthAnswer { .. }
11575        ));
11576        assert!(harness.handler.observe_module_control_completion(late));
11577
11578        let health = harness.module.status().unwrap().health;
11579        assert_eq!(
11580            health.consecutive_failures, 0,
11581            "a late answer is an answer: the streak must reset"
11582        );
11583        assert_eq!(health.late_answer_count, 1);
11584    }
11585
11586    #[tokio::test(start_paused = true)]
11587    async fn repeated_serial_probe_cycles_keep_one_tombstone_per_endpoint() {
11588        let mut harness = probe_harness();
11589
11590        for _ in 0..20 {
11591            time_out_without_answer(&mut harness).await;
11592            assert_eq!(
11593                harness.forwarding.health_probe_tombstone_count().unwrap(),
11594                1
11595            );
11596        }
11597    }
11598}
11599
11600#[cfg(test)]
11601mod child_env_tests {
11602    use super::{
11603        apply_child_env, apply_spawn_role, apply_wire_spawn_args, ModuleProtocol, ModuleSpec,
11604        SpawnRole, SupervisorHandle, SPAWN_ROLE_SWAP_CANDIDATE, SUBC_ARG, SUBC_LAUNCH_NONCE_ENV,
11605        SUBC_MODULE_ID_ENV, SUBC_SPAWN_ROLE_ENV,
11606    };
11607    use std::{ffi::OsStr, path::PathBuf};
11608    use tokio::process::Command;
11609
11610    fn spec(env: Vec<(String, String)>) -> ModuleSpec {
11611        ModuleSpec {
11612            module_id: "env-plan".to_string(),
11613            program: PathBuf::from("/nonexistent"),
11614            args: Vec::new(),
11615            env,
11616            reserved: false,
11617            reserved_prefixes: Vec::new(),
11618            protocol: ModuleProtocol::Subc,
11619            overlap: Default::default(),
11620        }
11621    }
11622
11623    /// Ambient `CK_LOG` is REMOVED for an unconfigured module, and a configured
11624    /// one still gets its own.
11625    ///
11626    /// This is the narrow goal `env_clear()` was reached for, and the reason the
11627    /// fix is `env_remove` rather than deleting the line: an operator's ambient
11628    /// filter silently becoming an unconfigured module's log level is a real
11629    /// defect, just a much smaller one than clearing the environment.
11630    ///
11631    /// Asserted on the command plan rather than a spawned child because proving
11632    /// the ABSENCE of an inherited variable needs the parent's environment
11633    /// mutated, and `forbid(unsafe_code)` refuses that. `get_envs()` reports a
11634    /// removal as `(key, None)`, which is exactly the distinction wanted: not
11635    /// "absent because nobody set it" but "explicitly unset for the child".
11636    #[test]
11637    fn ambient_ck_log_is_removed_and_a_configured_one_survives() {
11638        let mut command = Command::new("/nonexistent");
11639        apply_child_env(&mut command, &spec(Vec::new()));
11640        let removed = command
11641            .as_std()
11642            .get_envs()
11643            .any(|(key, value)| key == OsStr::new("CK_LOG") && value.is_none());
11644        assert!(
11645            removed,
11646            "ambient CK_LOG must be explicitly removed for an unconfigured module"
11647        );
11648
11649        let mut configured = Command::new("/nonexistent");
11650        apply_child_env(
11651            &mut configured,
11652            &spec(vec![("CK_LOG".to_string(), "debug".to_string())]),
11653        );
11654        let effective = configured
11655            .as_std()
11656            .get_envs()
11657            .filter(|(key, _)| *key == OsStr::new("CK_LOG"))
11658            .last()
11659            .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
11660        assert_eq!(
11661            effective,
11662            Some(Some("debug".to_string())),
11663            "a module's configured CK_LOG must survive the ambient removal"
11664        );
11665    }
11666
11667    /// A `protocol: "none"` spawn carries NO `--subc` argument and NO launch
11668    /// nonce; a subc-wire spawn carries both. Asserted on the command plan for
11669    /// the same reason as the CK_LOG test above.
11670    ///
11671    /// The argument is the load-bearing half: a stock binary exits on an
11672    /// unknown flag before it listens, so with `--subc` appended the mode
11673    /// could not supervise the one process it exists for. Found by the first
11674    /// conformance run (nats-server: `flag provided but not defined: -subc`).
11675    #[test]
11676    fn protocol_none_spawn_carries_no_subc_argument_and_no_nonce() {
11677        let connection_file = std::path::Path::new("/run/subc-connection.json");
11678        let handle = SupervisorHandle::new();
11679
11680        let mut none_spec = spec(Vec::new());
11681        none_spec.protocol = ModuleProtocol::None;
11682        let mut none = Command::new("/nonexistent");
11683        let none_handoff =
11684            apply_wire_spawn_args(&mut none, &none_spec, Some(connection_file), Some(&handle))
11685                .expect("protocol-none spawn args apply");
11686        assert!(
11687            none_handoff.is_none(),
11688            "protocol:none spawn must not receive a nonce descriptor"
11689        );
11690        assert!(
11691            !none.as_std().get_envs().any(|(key, value)| key
11692                == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV)
11693                && value.is_some()),
11694            "protocol:none spawn must not name a nonce descriptor"
11695        );
11696        let none_args: Vec<String> = none
11697            .as_std()
11698            .get_args()
11699            .map(|a| a.to_string_lossy().into_owned())
11700            .collect();
11701        assert!(
11702            !none_args.iter().any(|a| a == SUBC_ARG),
11703            "protocol:none argv must not carry --subc; got {none_args:?}"
11704        );
11705        let none_has_nonce = none
11706            .as_std()
11707            .get_envs()
11708            .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some());
11709        assert!(
11710            !none_has_nonce,
11711            "protocol:none spawn must not receive a launch nonce"
11712        );
11713        let none_has_module_id = none
11714            .as_std()
11715            .get_envs()
11716            .any(|(key, value)| key == OsStr::new(SUBC_MODULE_ID_ENV) && value.is_some());
11717        assert!(
11718            none_has_module_id,
11719            "SUBC_MODULE_ID is inert and stays on every path"
11720        );
11721        assert!(
11722            handle.spawn_nonce(&none_spec.module_id).is_none(),
11723            "no nonce record for a process that will never present one"
11724        );
11725
11726        // Control: the subc-wire path is unchanged by the branch above.
11727        let wire_spec = spec(Vec::new());
11728        let mut wire = Command::new("/nonexistent");
11729        let wire_handoff =
11730            apply_wire_spawn_args(&mut wire, &wire_spec, Some(connection_file), Some(&handle))
11731                .expect("subc-wire spawn args apply");
11732        let wire_fd_env = wire
11733            .as_std()
11734            .get_envs()
11735            .find(|(key, _)| *key == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV))
11736            .and_then(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
11737        #[cfg(unix)]
11738        assert_eq!(
11739            wire_fd_env,
11740            Some(wire_handoff.expect("a descriptor handoff").fd_env_value()),
11741            "a subc-wire spawn names the pipe it will receive at descriptor 3"
11742        );
11743        #[cfg(not(unix))]
11744        assert!(wire_handoff.is_none() && wire_fd_env.is_none());
11745        let wire_args: Vec<String> = wire
11746            .as_std()
11747            .get_args()
11748            .map(|a| a.to_string_lossy().into_owned())
11749            .collect();
11750        assert_eq!(
11751            wire_args,
11752            vec![
11753                SUBC_ARG.to_string(),
11754                connection_file.to_string_lossy().into_owned()
11755            ],
11756            "a subc-wire spawn still carries --subc <path>"
11757        );
11758        assert_eq!(
11759            wire.as_std()
11760                .get_envs()
11761                .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some()),
11762            !cfg!(unix),
11763            "only Windows supplies the environment nonce"
11764        );
11765        assert!(handle.spawn_nonce(&wire_spec.module_id).is_some());
11766    }
11767
11768    /// A plain spawn EXPLICITLY REMOVES the spawn role, even when the module's
11769    /// spec tries to set it; only a swap candidate carries it.
11770    ///
11771    /// "Set it only on candidates" is not enough, because spawn applies the
11772    /// spec's env verbatim and the daemon's own environment is inherited: either
11773    /// could hand a plain restart the swap role, and a module reading it would
11774    /// warm on its long swap budget while callers wait. Asserted as an explicit
11775    /// removal (`(key, None)`), not mere absence, for the reason the `CK_LOG`
11776    /// test above gives.
11777    #[test]
11778    fn plain_spawn_removes_the_spawn_role_even_when_the_spec_sets_it() {
11779        let role = |command: &Command| {
11780            command
11781                .as_std()
11782                .get_envs()
11783                .filter(|(key, _)| *key == OsStr::new(SUBC_SPAWN_ROLE_ENV))
11784                .last()
11785                .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()))
11786        };
11787        let forged = spec(vec![(
11788            SUBC_SPAWN_ROLE_ENV.to_string(),
11789            SPAWN_ROLE_SWAP_CANDIDATE.to_string(),
11790        )]);
11791
11792        let mut plain = Command::new("/nonexistent");
11793        apply_child_env(&mut plain, &forged);
11794        apply_spawn_role(&mut plain, SpawnRole::Plain);
11795        assert_eq!(
11796            role(&plain),
11797            Some(None),
11798            "a plain spawn must remove SUBC_SPAWN_ROLE, whatever the spec says"
11799        );
11800
11801        let mut candidate = Command::new("/nonexistent");
11802        apply_child_env(&mut candidate, &spec(Vec::new()));
11803        apply_spawn_role(&mut candidate, SpawnRole::SwapCandidate);
11804        assert_eq!(
11805            role(&candidate),
11806            Some(Some(SPAWN_ROLE_SWAP_CANDIDATE.to_string()))
11807        );
11808    }
11809
11810    /// Daemon-private capture retention keys never reach the child.
11811    ///
11812    /// cortexkit-log exposes retention as a Rust struct with no environment
11813    /// names, so these entries are supervisor metadata. Passing them through
11814    /// would invent a public child-process contract by accident.
11815    #[test]
11816    fn daemon_private_capture_keys_are_not_passed_to_the_child() {
11817        let mut command = Command::new("/nonexistent");
11818        apply_child_env(
11819            &mut command,
11820            &spec(vec![
11821                (super::CAPTURE_KEEP_ENV.to_string(), "5".to_string()),
11822                ("KEPT".to_string(), "yes".to_string()),
11823            ]),
11824        );
11825        let keys: Vec<String> = command
11826            .as_std()
11827            .get_envs()
11828            .filter(|(_, value)| value.is_some())
11829            .map(|(key, _)| key.to_string_lossy().into_owned())
11830            .collect();
11831        assert!(keys.contains(&"KEPT".to_string()), "got {keys:?}");
11832        assert!(
11833            !keys.contains(&super::CAPTURE_KEEP_ENV.to_string()),
11834            "daemon-private capture key leaked to the child: {keys:?}"
11835        );
11836    }
11837}
11838
11839#[cfg(test)]
11840mod jitter_tests {
11841    use super::jittered_health_delay;
11842    use std::{collections::HashSet, time::Duration};
11843
11844    /// Module ids drawn from a real fleet, so the dispersal claim is about names
11845    /// that actually occur rather than invented ones.
11846    ///
11847    /// This is a SAMPLE, not a registry: the property under test is that distinct
11848    /// ids disperse, which holds for any set of distinct strings. Several entries
11849    /// are already historical (modules get renamed), and that costs nothing here --
11850    /// but it means a reader must not mistake this for the live module set, and a
11851    /// rename sweep will match it without there being anything to change.
11852    const FLEET: [&str; 14] = [
11853        "aft",
11854        "alfonso-core",
11855        "magic-context",
11856        "broca",
11857        "thalamus",
11858        "quota",
11859        "engram",
11860        "plexus",
11861        "cerebellum",
11862        "astrocyte",
11863        "synapse",
11864        "subc-mcp",
11865        "cortexkit-credentials",
11866        "subc-federation",
11867    ];
11868
11869    /// Probes must not converge after a fleet-wide restart.
11870    ///
11871    /// This is the property the jitter exists for: every module reconnects at
11872    /// once, and without dispersal all fourteen would then probe on the same
11873    /// tick forever. Nothing failed visibly when this went untested -- a
11874    /// convergent fleet still probes correctly, just in a burst, so the symptom
11875    /// is a periodic load spike that looks like whatever else is running.
11876    #[test]
11877    fn probe_delays_disperse_across_the_fleet() {
11878        let cadence = Duration::from_secs(30);
11879        let delays: HashSet<Duration> = FLEET
11880            .iter()
11881            .map(|id| jittered_health_delay(id, 0, cadence))
11882            .collect();
11883        assert_eq!(
11884            delays.len(),
11885            FLEET.len(),
11886            "every supervised module must land on its own probe offset"
11887        );
11888    }
11889
11890    /// The offset may only ever DELAY a probe, never bring it forward.
11891    ///
11892    /// A delay below the cadence would probe a module more often than
11893    /// configured, which is the opposite of what an operator asked for and
11894    /// would tighten the failure budget without anyone changing it.
11895    #[test]
11896    fn jitter_only_delays_and_stays_within_one_tenth_of_cadence() {
11897        let cadence = Duration::from_secs(30);
11898        let span = cadence / 10;
11899        for id in FLEET {
11900            for probe_index in 0..8 {
11901                let delay = jittered_health_delay(id, probe_index, cadence);
11902                assert!(
11903                    delay >= cadence,
11904                    "{id}#{probe_index}: jitter must not shorten the cadence"
11905                );
11906                assert!(
11907                    delay < cadence + span,
11908                    "{id}#{probe_index}: jitter must stay inside one tenth of the cadence"
11909                );
11910            }
11911        }
11912    }
11913
11914    /// A module keeps its offset across daemon restarts.
11915    ///
11916    /// The delay is derived rather than randomised precisely so a restart does
11917    /// not re-roll every module into a fresh chance of collision. A random
11918    /// source would satisfy the dispersal test above and quietly lose this.
11919    #[test]
11920    fn a_module_offset_is_stable_across_restarts() {
11921        let cadence = Duration::from_secs(30);
11922        for id in FLEET {
11923            assert_eq!(
11924                jittered_health_delay(id, 0, cadence),
11925                jittered_health_delay(id, 0, cadence),
11926                "{id}: the same module and probe index must produce the same offset"
11927            );
11928        }
11929    }
11930
11931    /// A zero cadence disables probing rather than producing a busy loop.
11932    #[test]
11933    fn zero_cadence_yields_zero_delay() {
11934        assert_eq!(
11935            jittered_health_delay("aft", 0, Duration::ZERO),
11936            Duration::ZERO
11937        );
11938    }
11939}
11940
11941#[cfg(all(test, target_os = "linux"))]
11942mod cgroup_placement_tests {
11943    use super::{
11944        apply_cgroup_placement, remove_module_cgroup, ModuleProtocol, ModuleSpec, SuperviseError,
11945        SupervisedChild,
11946    };
11947    use crate::stderr_tail::{StderrRing, StderrTailConfig};
11948    use std::{
11949        fs, io,
11950        path::{Path, PathBuf},
11951        sync::{Arc, Mutex},
11952    };
11953    use subc_test_support::TestTempDir;
11954    use tokio::process::Command;
11955
11956    #[tokio::test]
11957    async fn unique_spawn_cgroups_clean_up_across_restarts_and_shutdown() {
11958        use super::*;
11959        let dir = TestTempDir::new("unique-spawn-cgroups");
11960        let root = PathBuf::from(format!(
11961            "/sys/fs/cgroup/subc-unique-test-{}-{}",
11962            std::process::id(),
11963            unix_ms_now()
11964        ));
11965        if let Err(error) = fs::create_dir(&root) {
11966            assert!(
11967                std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
11968                "required cgroup test cannot execute: {error}"
11969            );
11970            eprintln!("SKIP unique_spawn_cgroups_clean_up_across_restarts_and_shutdown: {error}");
11971            return;
11972        }
11973        let placement = subc_cgroup::prepare_at(&root).unwrap().unwrap();
11974        let group_count = || {
11975            fs::read_dir(root.join("subc-modules"))
11976                .unwrap()
11977                .map(|entry| entry.unwrap().file_type().unwrap())
11978                .filter(|kind| kind.is_dir())
11979                .count()
11980        };
11981        let supervisor =
11982            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
11983                .with_cgroup_placement(Some(placement.clone()));
11984        let runtime = supervisor.runtime_config();
11985        let mut spec = ModuleSpec {
11986            module_id: "unique-spawn".into(),
11987            program: PathBuf::from("/bin/sleep"),
11988            args: vec!["60".into()],
11989            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
11990                .into_iter()
11991                .map(|key| (key.into(), dir.to_string_lossy().into_owned()))
11992                .collect(),
11993            reserved: false,
11994            reserved_prefixes: vec![],
11995            protocol: ModuleProtocol::None,
11996            overlap: Default::default(),
11997        };
11998        let spawn = |spec: &ModuleSpec| {
11999            spawn_child(
12000                spec,
12001                None,
12002                None,
12003                &runtime.stderr_ring,
12004                None,
12005                &runtime.child_roster,
12006                Some(&placement),
12007            )
12008            .unwrap()
12009        };
12010        let mut live = spawn(&spec);
12011        for _ in 0..3 {
12012            // A new process can enter the old slot while retirement is pending.
12013            let next = spawn(&spec);
12014            assert_ne!(live.module_id, next.module_id);
12015            live.start_kill().unwrap();
12016            live.wait().await.unwrap();
12017            live = next;
12018            assert!(
12019                live.child.try_wait().unwrap().is_none(),
12020                "retiring the old slot must not kill the replacement"
12021            );
12022            assert_eq!(
12023                group_count(),
12024                1,
12025                "only the live spawn's cgroup should remain"
12026            );
12027        }
12028        supervisor.begin_daemon_shutdown();
12029        let reap = tokio::spawn(async move {
12030            live.wait().await.unwrap();
12031        });
12032        supervisor
12033            .end_children_for_daemon_shutdown(false, std::future::pending())
12034            .await;
12035        reap.await.unwrap();
12036        assert_eq!(group_count(), 0);
12037        // A normal exit uses the same tree-cleanup path as a killed spawn.
12038        spec.program = PathBuf::from("/bin/true");
12039        spec.args.clear();
12040        let fresh_roster = ChildRoster::default();
12041        let mut short = spawn_child(
12042            &spec,
12043            None,
12044            None,
12045            &runtime.stderr_ring,
12046            None,
12047            &fresh_roster,
12048            Some(&placement),
12049        )
12050        .unwrap();
12051        short.wait().await.unwrap();
12052        assert_eq!(group_count(), 0);
12053        spec.module_id = "_".repeat(255);
12054        let mut long_id = spawn_child(
12055            &spec,
12056            None,
12057            None,
12058            &runtime.stderr_ring,
12059            None,
12060            &fresh_roster,
12061            Some(&placement),
12062        )
12063        .unwrap();
12064        long_id.wait().await.unwrap();
12065        assert_eq!(
12066            group_count(),
12067            0,
12068            "valid long module IDs must not exceed cgroup NAME_MAX"
12069        );
12070        fs::remove_dir(root.join("subc-modules")).unwrap();
12071        fs::remove_dir(root).unwrap();
12072        eprintln!("EXECUTED unique_spawn_cgroups_clean_up_across_restarts_and_shutdown");
12073    }
12074
12075    #[test]
12076    fn failed_parent_cgroup_open_is_a_cgroup_supervision_error() {
12077        let path = Path::new("/definitely-missing-subc-cgroup");
12078        let mut command = Command::new("true");
12079        let error = apply_cgroup_placement(
12080            &mut command,
12081            &ModuleSpec {
12082                module_id: "broken-cgroup".to_string(),
12083                program: PathBuf::from("true"),
12084                args: Vec::new(),
12085                env: Vec::new(),
12086                reserved: false,
12087                reserved_prefixes: Vec::new(),
12088                protocol: ModuleProtocol::Subc,
12089                overlap: Default::default(),
12090            },
12091            path,
12092        )
12093        .expect_err("a parent cgroup open failure must reject the supervised spawn");
12094        let reason = error.to_string();
12095
12096        assert!(
12097            matches!(error, SuperviseError::Cgroup { .. }),
12098            "parent cgroup open must be reported as a cgroup supervision error: {reason}"
12099        );
12100        assert!(
12101            reason.contains("/definitely-missing-subc-cgroup/cgroup.procs"),
12102            "parent cgroup open failure must name cgroup.procs: {reason}"
12103        );
12104    }
12105
12106    #[tokio::test]
12107    async fn reaping_a_child_removes_its_empty_module_cgroup() {
12108        let root = TestTempDir::new("supervisor-reap-cgroup");
12109        fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
12110        let placement = subc_cgroup::prepare_at(&root)
12111            .expect("prepare scratch cgroup root")
12112            .expect("scratch root has a cgroup.procs marker");
12113        let module_id = "reaped-module";
12114        let module = placement
12115            .module_path(module_id)
12116            .expect("create scratch module cgroup");
12117        let child = Command::new("true")
12118            .env("XDG_DATA_HOME", root.path())
12119            .env("XDG_RUNTIME_DIR", root.path())
12120            .env("XDG_CONFIG_HOME", root.path())
12121            .spawn()
12122            .expect("spawn short-lived child");
12123        let pid = child.id().expect("spawned child has pid");
12124        let mut child = SupervisedChild {
12125            child,
12126            protocol: ModuleProtocol::Subc,
12127            module_id: module_id.to_string(),
12128            cgroup_placement: Some(placement),
12129            stdout_pump: None,
12130            stderr_pump: None,
12131            stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
12132            spawned_at_ms: 0,
12133            spawned_from: PathBuf::from("true"),
12134            spawned_file_identity: None,
12135            process_start_time: None,
12136            process_identity: None,
12137            pid,
12138            roster_guard: None,
12139            #[cfg(target_os = "macos")]
12140            privacy_exec: None,
12141            spawn_failure: None,
12142        };
12143
12144        child.wait().await.expect("reap short-lived child");
12145
12146        assert!(
12147            !module.exists(),
12148            "reaping the supervised child must remove its empty cgroup"
12149        );
12150    }
12151
12152    #[test]
12153    fn non_empty_cgroup_removal_is_reported_without_blocking_teardown() {
12154        let root = TestTempDir::new("supervisor-non-empty-cgroup");
12155        fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
12156        let placement = subc_cgroup::prepare_at(&root)
12157            .expect("prepare scratch cgroup root")
12158            .expect("scratch root has a cgroup.procs marker");
12159        let module = placement
12160            .module_path("surviving-module")
12161            .expect("create scratch module cgroup");
12162        fs::write(module.join("surviving-process"), b"still present")
12163            .expect("make scratch cgroup non-empty");
12164        let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::WARN);
12165
12166        remove_module_cgroup(&placement, "surviving-module");
12167
12168        let logs = crate::router::test_log::captured_logs(&logs);
12169        assert!(
12170            module.exists(),
12171            "failed removal must leave the cgroup intact"
12172        );
12173        assert!(
12174            logs.contains("could not remove module cgroup after process exit; continuing teardown")
12175                && logs.contains("surviving-module"),
12176            "best-effort removal must report the failure without returning it: {logs}"
12177        );
12178    }
12179
12180    #[test]
12181    fn cgroup_pre_exec_spawn_failure_names_the_cgroup_path() {
12182        let cgroup_path = PathBuf::from("/sys/fs/cgroup/subc-modules/broken-module");
12183        let reason = SuperviseError::Spawn {
12184            program: PathBuf::from("/bin/true"),
12185            source: io::Error::from_raw_os_error(13),
12186            cgroup_path: Some(cgroup_path.clone()),
12187        }
12188        .to_string();
12189
12190        assert!(
12191            reason.contains(&cgroup_path.display().to_string()),
12192            "a pre_exec spawn failure must name the cgroup path: {reason}"
12193        );
12194    }
12195}
12196
12197#[cfg(test)]
12198mod spawn_subscriber_lag_tests {
12199    use super::*;
12200
12201    /// A subscriber whose connection stops draining is dropped once its frame
12202    /// channel fills. The client must learn that from a terminal Error frame
12203    /// after the frames already queued for it, not from a stream that simply
12204    /// goes quiet.
12205    #[tokio::test]
12206    async fn lagged_spawn_subscriber_receives_a_terminal_lagged_error_after_its_queued_frames() {
12207        let feed = SpawnEventFeed::default();
12208        feed.configure_incarnation("lag-incarnation".to_string());
12209        // A one-slot connection queue that nobody reads until the emits are
12210        // done: the forwarder parks on it and the subscriber channel fills.
12211        let (tx, mut rx) = mpsc::channel(1);
12212        feed.subscribe(ConnectionId::new(1), 7, 1, None, FrameSink::new(tx))
12213            .expect("subscribe");
12214        let emitted = SPAWN_SUBSCRIBER_BUFFER + 16;
12215        for index in 0..emitted {
12216            feed.emit_spawned(&format!("lag-module-{index}"), 1000, 0);
12217            // Let the forwarder take what it can so the fill point is the
12218            // subscriber channel, not a scheduling accident.
12219            tokio::task::yield_now().await;
12220        }
12221        assert_eq!(
12222            feed.subscriber_count(),
12223            0,
12224            "the lagged subscriber must be removed"
12225        );
12226
12227        let mut data = Vec::new();
12228        let mut last = None;
12229        loop {
12230            let next = tokio::time::timeout(Duration::from_secs(5), rx.recv())
12231                .await
12232                .expect("the forwarder must finish once the subscriber is dropped");
12233            let Some(outbound) = next else { break };
12234            let frame = outbound.frame;
12235            if frame.header.ty == FrameType::StreamData {
12236                assert!(last.is_none(), "no data may follow the terminal frame");
12237                let event: SpawnEvent = serde_json::from_slice(&frame.body).unwrap();
12238                data.push(event.cursor.seq);
12239            } else {
12240                assert!(last.is_none(), "exactly one terminal frame");
12241                last = Some(frame);
12242            }
12243        }
12244        assert!(!data.is_empty(), "queued frames drain before the terminal");
12245        for pair in data.windows(2) {
12246            assert_eq!(
12247                pair[1],
12248                pair[0] + 1,
12249                "queued frames arrive dense and in order"
12250            );
12251        }
12252        let terminal = last.expect("a lagged subscriber must receive a terminal frame");
12253        assert_eq!(terminal.header.ty, FrameType::Error);
12254        assert_eq!(terminal.header.corr, 7);
12255        let body: subc_protocol::ErrorBody = serde_json::from_slice(&terminal.body).unwrap();
12256        assert_eq!(body.code, SPAWN_SUBSCRIBER_LAGGED_CODE);
12257        let detail = body.detail.expect("lagged error carries detail");
12258        assert_eq!(
12259            detail["first_undelivered_cursor"]["seq"],
12260            data.last().unwrap() + 1,
12261            "the named cursor is the first event the subscriber did not receive"
12262        );
12263        assert_eq!(
12264            detail["first_undelivered_cursor"]["daemon_incarnation"],
12265            "lag-incarnation"
12266        );
12267    }
12268}
12269
12270#[cfg(test)]
12271mod terminal_history_read_concurrency_tests {
12272    use super::*;
12273    use crate::terminal_journal::read_pause;
12274    use std::sync::mpsc as std_mpsc;
12275    use subc_test_support::TestTempDir;
12276
12277    fn journaled_ring(
12278        journal: &Arc<crate::terminal_journal::TerminalJournal>,
12279    ) -> Arc<Mutex<TerminalRing>> {
12280        Arc::new(Mutex::new(
12281            TerminalRing::new(TerminalRingConfig::default(), 1)
12282                .with_journal(Some(Arc::clone(journal))),
12283        ))
12284    }
12285
12286    fn crash(at_ms: u64) -> ExitReport {
12287        ExitReport {
12288            kind: ExitKind::Crash,
12289            code: Some(1),
12290            signal: None,
12291            at_ms,
12292        }
12293    }
12294
12295    /// Record an exit on another thread and report whether it finished within
12296    /// `bound`. The recorder thread is left running if it did not.
12297    fn record_within(
12298        module_id: &'static str,
12299        ring: &Arc<Mutex<TerminalRing>>,
12300        at_ms: u64,
12301        bound: Duration,
12302    ) -> bool {
12303        let ring = Arc::clone(ring);
12304        let (done, done_rx) = std_mpsc::channel();
12305        std::thread::spawn(move || {
12306            record_terminal(
12307                module_id,
12308                &ring,
12309                &SpawnEventFeed::default(),
12310                &crash(at_ms),
12311                TerminalDisposition::Restarting,
12312            );
12313            let _ = done.send(());
12314        });
12315        done_rx.recv_timeout(bound).is_ok()
12316    }
12317
12318    /// A history read in progress must not hold the journal writer (which every
12319    /// module's exit recording needs) or the module's own ring. Exits recorded
12320    /// while the read is paused complete promptly; the paused read answers as of
12321    /// the moment it started, and the next read has each exit exactly once.
12322    #[test]
12323    fn exits_recorded_during_a_paused_history_read_are_not_blocked_or_half_merged() {
12324        let dir = TestTempDir::new("terminal-history-concurrent-read");
12325        let path = dir.join("terminals.jsonl");
12326        let journal = Arc::new(crate::terminal_journal::TerminalJournal::open(
12327            path.clone(),
12328            "daemon".into(),
12329        ));
12330        let reader_ring = journaled_ring(&journal);
12331        let other_ring = journaled_ring(&journal);
12332        assert!(record_within(
12333            "reader-module",
12334            &reader_ring,
12335            10,
12336            Duration::from_secs(5)
12337        ));
12338
12339        let (started, release) = read_pause::install(&path);
12340        let reading = {
12341            let ring = Arc::clone(&reader_ring);
12342            std::thread::spawn(move || durable_terminal_history_of(&ring, "reader-module"))
12343        };
12344        started
12345            .recv_timeout(Duration::from_secs(5))
12346            .expect("the history read reached its pause");
12347
12348        let bound = Duration::from_secs(1);
12349        assert!(
12350            record_within("other-module", &other_ring, 20, bound),
12351            "another module's exit waited on a history read (journal writer held)"
12352        );
12353        assert!(
12354            record_within("reader-module", &reader_ring, 30, bound),
12355            "the read module's own exit waited on its history read (ring held)"
12356        );
12357
12358        drop(release);
12359        let paused = reading.join().unwrap();
12360        assert_eq!(
12361            paused.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
12362            vec![10],
12363            "an exit recorded after the read began lands in neither half of it"
12364        );
12365        assert_eq!(paused.journal_skipped_lines, 0);
12366        assert_eq!(paused.journal_read_errors, 0);
12367
12368        let after = durable_terminal_history_of(&reader_ring, "reader-module");
12369        assert_eq!(
12370            after.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
12371            vec![10, 30],
12372            "the next read merges ring and journal with no duplicate"
12373        );
12374        assert_eq!(after.journal_skipped_lines, 0);
12375    }
12376}
12377
12378/// What a restart does with the exited process's stderr reader. These drive
12379/// the same `settle_stderr_pump` the supervisor calls, with a reader the test
12380/// holds, so a reader that has not been scheduled by the bound is a controlled
12381/// input rather than something only a loaded machine produces.
12382#[cfg(test)]
12383mod stderr_settle_tests {
12384    use std::{
12385        future::Future,
12386        io,
12387        pin::Pin,
12388        sync::{Arc, Mutex},
12389        task::{Context, Poll},
12390        time::Duration,
12391    };
12392
12393    use tokio::{
12394        io::{AsyncRead, ReadBuf},
12395        sync::oneshot,
12396        time::Instant,
12397    };
12398
12399    use super::{settle_stderr_pump, StderrPump};
12400    use crate::stderr_tail::{
12401        pump_stderr_to, untimed, CaptureState, OutputSink, StderrRing, StderrTailConfig, TailEntry,
12402    };
12403
12404    const BOUND: Duration = Duration::from_millis(250);
12405
12406    /// Yields `before`, then stays pending until the gate is released, then
12407    /// yields `after` and reaches EOF. The bytes after the gate were written
12408    /// by a process that has already exited; only the reader is behind.
12409    struct HeldReader {
12410        before: Option<Vec<u8>>,
12411        gate: Option<oneshot::Receiver<()>>,
12412        after: io::Cursor<Vec<u8>>,
12413    }
12414
12415    impl AsyncRead for HeldReader {
12416        fn poll_read(
12417            mut self: Pin<&mut Self>,
12418            cx: &mut Context<'_>,
12419            buf: &mut ReadBuf<'_>,
12420        ) -> Poll<io::Result<()>> {
12421            if let Some(bytes) = self.before.take() {
12422                buf.put_slice(&bytes);
12423                return Poll::Ready(Ok(()));
12424            }
12425            if let Some(gate) = self.gate.as_mut() {
12426                match Pin::new(gate).poll(cx) {
12427                    Poll::Pending => return Poll::Pending,
12428                    Poll::Ready(_) => self.gate = None,
12429                }
12430            }
12431            Pin::new(&mut self.after).poll_read(cx, buf)
12432        }
12433    }
12434
12435    struct DiscardSink;
12436
12437    impl OutputSink for DiscardSink {
12438        fn write_line(&mut self, _line: &[u8]) {}
12439    }
12440
12441    fn line(text: &str) -> TailEntry {
12442        TailEntry::Line {
12443            text: text.to_string(),
12444            truncated: false,
12445            at_ms: None,
12446        }
12447    }
12448
12449    fn lock(ring: &Arc<Mutex<StderrRing>>) -> std::sync::MutexGuard<'_, StderrRing> {
12450        ring.lock().unwrap()
12451    }
12452
12453    /// Start a reader for a new process generation that delivers `before`
12454    /// immediately and `after` only once the returned sender fires (or is
12455    /// dropped).
12456    fn held_pump(
12457        ring: &Arc<Mutex<StderrRing>>,
12458        before: &str,
12459        after: &str,
12460    ) -> (StderrPump, oneshot::Sender<()>) {
12461        let generation = lock(ring).begin_process();
12462        let (release, gate) = oneshot::channel();
12463        let reader = HeldReader {
12464            before: Some(before.as_bytes().to_vec()),
12465            gate: Some(gate),
12466            after: io::Cursor::new(after.as_bytes().to_vec()),
12467        };
12468        let task = tokio::spawn(pump_stderr_to(
12469            reader,
12470            Arc::clone(ring),
12471            generation,
12472            DiscardSink,
12473        ));
12474        (StderrPump { task, generation }, release)
12475    }
12476
12477    async fn wait_until(ring: &Arc<Mutex<StderrRing>>, done: impl Fn(&StderrRing) -> bool) {
12478        for _ in 0..1000 {
12479            if done(&lock(ring)) {
12480                return;
12481            }
12482            tokio::time::sleep(Duration::from_millis(1)).await;
12483        }
12484        panic!(
12485            "ring never reached the expected state: {:?}",
12486            lock(ring).snapshot(None, None)
12487        );
12488    }
12489
12490    #[tokio::test(start_paused = true)]
12491    async fn a_crash_line_the_reader_had_not_reached_by_the_bound_is_kept_before_the_restart() {
12492        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
12493        let (pump, release) = held_pump(&ring, "booting\n", "config error: missing storage\n");
12494
12495        settle_stderr_pump("crasher", &ring, pump, BOUND).await;
12496        let before_release = lock(&ring).snapshot(None, None);
12497        assert!(
12498            matches!(before_release.capture, CaptureState::Incomplete { .. }),
12499            "a reader that has not reached EOF cannot claim a whole tail: {before_release:?}"
12500        );
12501
12502        // The restart: the next process starts and writes before the old
12503        // reader catches up.
12504        let next = lock(&ring).begin_process();
12505        lock(&ring).push_line_from(next, "next process booting");
12506        release.send(()).unwrap();
12507        wait_until(&ring, |ring| {
12508            ring.snapshot(None, None).capture == CaptureState::Captured
12509        })
12510        .await;
12511
12512        assert_eq!(
12513            untimed(lock(&ring).snapshot(None, None).entries),
12514            vec![
12515                line("booting"),
12516                line("config error: missing storage"),
12517                TailEntry::ProcessStart,
12518                line("next process booting"),
12519            ],
12520            "the crash's last line must survive a slow reader and stay in the crashed process's section"
12521        );
12522    }
12523
12524    #[tokio::test(start_paused = true)]
12525    async fn a_pipe_held_open_by_a_descendant_reads_incomplete_without_delaying_the_restart_past_the_bound(
12526    ) {
12527        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
12528        // `_held` is never fired: a descendant keeps the pipe open for the
12529        // whole test.
12530        let (pump, _held) = held_pump(&ring, "parent exiting\n", "");
12531
12532        let started = Instant::now();
12533        settle_stderr_pump("orphaning", &ring, pump, BOUND).await;
12534        assert_eq!(
12535            started.elapsed(),
12536            BOUND,
12537            "the restart must wait exactly the bound for a pipe that stays open, no longer"
12538        );
12539
12540        let next = lock(&ring).begin_process();
12541        lock(&ring).push_line_from(next, "next process booting");
12542        tokio::time::sleep(Duration::from_secs(60)).await;
12543
12544        let snapshot = lock(&ring).snapshot(None, None);
12545        match &snapshot.capture {
12546            CaptureState::Incomplete { reason } => assert!(
12547                reason.contains("had not reached EOF") && reason.contains("250ms"),
12548                "the reason must say what is missing and after how long: {reason}"
12549            ),
12550            other => panic!("expected Incomplete while the pipe is held open, got {other:?}"),
12551        }
12552        assert_eq!(
12553            untimed(snapshot.entries),
12554            vec![
12555                line("parent exiting"),
12556                TailEntry::ProcessStart,
12557                line("next process booting"),
12558            ]
12559        );
12560    }
12561
12562    #[tokio::test(start_paused = true)]
12563    async fn a_reader_that_reaches_eof_within_the_bound_leaves_the_tail_captured() {
12564        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
12565        let (pump, release) = held_pump(&ring, "one\n", "two\n");
12566        release.send(()).unwrap();
12567
12568        settle_stderr_pump("clean", &ring, pump, BOUND).await;
12569
12570        let snapshot = lock(&ring).snapshot(None, None);
12571        assert_eq!(snapshot.capture, CaptureState::Captured);
12572        assert_eq!(untimed(snapshot.entries), vec![line("one"), line("two")]);
12573    }
12574}
12575
12576/// Containment of a module's process tree (issue #109).
12577///
12578/// The behaviour these defend against is a module helper surviving its module:
12579/// on a real machine the Synapse embedding module's CUDA worker holds ~2.2 GB of
12580/// VRAM, so a leaked grandchild is a leaked GPU allocation, and a day of restarts
12581/// compounds it.
12582///
12583/// They run against the SUPERVISOR rather than the job-object crate because the
12584/// claim is about teardown: a crate-level test proves a job can reap a tree, not
12585/// that the daemon's drain path reaches it.
12586///
12587/// Windows-only, like the mechanism. On Unix this arm compiles out; the cgroup
12588/// lane there is a separate containment path with its own tests.
12589#[cfg(all(test, windows))]
12590mod job_containment_tests {
12591    use super::*;
12592    use std::{
12593        path::{Path, PathBuf},
12594        sync::{Arc, Mutex},
12595        time::{Duration, Instant},
12596    };
12597    use subc_test_support::TestTempDir;
12598
12599    /// The stub, expected beside this test executable.
12600    ///
12601    /// The existence check is here for the reason its twin at `fake_aft_stub_path`
12602    /// documents: `--lib` does not build `[[bin]]` targets, and a bare spawn
12603    /// failure then reads as a broken test rather than an unbuilt dependency.
12604    fn stub_path() -> PathBuf {
12605        let mut path = std::env::current_exe().expect("current_exe available in tests");
12606        path.pop();
12607        path.pop();
12608        path.push("fake-aft-stub.exe");
12609        assert!(
12610            path.exists(),
12611            "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
12612             [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
12613            path.display()
12614        );
12615        path
12616    }
12617
12618    /// Poll for the grandchild pid the stub records, and parse it.
12619    fn read_grandchild_pid(path: &Path) -> u32 {
12620        let deadline = Instant::now() + Duration::from_secs(10);
12621        loop {
12622            if let Ok(contents) = std::fs::read_to_string(path) {
12623                if let Ok(pid) = contents.trim().parse() {
12624                    return pid;
12625                }
12626            }
12627            assert!(
12628                Instant::now() < deadline,
12629                "the stub never recorded a grandchild pid at {}",
12630                path.display()
12631            );
12632            std::thread::sleep(Duration::from_millis(10));
12633        }
12634    }
12635
12636    /// Everything one fixture run needs, so the two tests below differ in exactly
12637    /// one place: whether the child is contained.
12638    struct Fixture {
12639        _dir: TestTempDir,
12640        module_id: String,
12641        grandchild: u32,
12642        child: Option<SupervisedChild>,
12643        registry: Arc<Registry>,
12644        snapshot: Arc<Mutex<SupervisorSnapshot>>,
12645        terminal_ring: Arc<Mutex<TerminalRing>>,
12646        spawn_events: SpawnEventFeed,
12647    }
12648
12649    fn fixture(label: &str, module_id: &str) -> Fixture {
12650        let dir = TestTempDir::new(label);
12651        let pid_file = dir.join("grandchild.pid");
12652        let supervisor = Supervisor::new_for_test(
12653            Arc::new(Registry::default()),
12654            RestartPolicy::new(3, Duration::ZERO),
12655        );
12656        let runtime = supervisor.runtime_config();
12657        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
12658        let spec = ModuleSpec {
12659            module_id: module_id.to_string(),
12660            program: stub_path(),
12661            // Zero args deliberately: a `--subc` argument would make the stub dial
12662            // a daemon that is not there, and the failure would land in the same
12663            // stderr ring this fixture exists to keep quiet.
12664            args: Vec::new(),
12665            env: vec![
12666                ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
12667                (
12668                    "FAKE_AFT_GRANDCHILD_PID_FILE".to_string(),
12669                    pid_file.display().to_string(),
12670                ),
12671            ],
12672            reserved: false,
12673            reserved_prefixes: Vec::new(),
12674            protocol: ModuleProtocol::Subc,
12675            overlap: Default::default(),
12676        };
12677        let child = spawn_and_mark_running(&spec, &runtime, &snapshot)
12678            .expect("spawn the supervised fixture");
12679        let grandchild = read_grandchild_pid(&pid_file);
12680        Fixture {
12681            _dir: dir,
12682            module_id: module_id.to_string(),
12683            grandchild,
12684            child: Some(child),
12685            registry: Arc::new(Registry::default()),
12686            snapshot,
12687            terminal_ring: Arc::clone(&runtime.terminal_ring),
12688            spawn_events: SpawnEventFeed::default(),
12689        }
12690    }
12691
12692    impl Fixture {
12693        /// Drain through the supervisor's own teardown path.
12694        async fn drain(&mut self) {
12695            let child = self
12696                .child
12697                .take()
12698                .expect("the fixture child is still present");
12699            drain_child_to_state(
12700                &self.module_id,
12701                ModuleProtocol::Subc,
12702                // No forwarding table in this fixture, so nothing reaches the
12703                // child over a connection.
12704                StopNotice::NotSent,
12705                &self.registry,
12706                None,
12707                &self.snapshot,
12708                &self.terminal_ring,
12709                &self.spawn_events,
12710                child,
12711                Duration::from_millis(500),
12712                ModuleState::Stopped,
12713                Some(false),
12714            )
12715            .await
12716            .expect("drain the supervised fixture");
12717        }
12718    }
12719
12720    /// Teardown reaps the grandchild, not merely the direct child.
12721    ///
12722    /// This is the assertion the change exists for. Before containment the
12723    /// grandchild survived: it is a separate process, and `start_kill` is
12724    /// `TerminateProcess` scoped to one pid.
12725    #[tokio::test]
12726    async fn teardown_reaps_the_grandchild() {
12727        let mut fixture = fixture("teardown-grandchild", "tree-teardown");
12728        let grandchild = fixture.grandchild;
12729
12730        assert!(
12731            subc_jobobject::process_exists(grandchild),
12732            "grandchild {grandchild} must be alive before teardown, or this proves nothing"
12733        );
12734
12735        fixture.drain().await;
12736
12737        assert!(
12738            subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
12739            "grandchild {grandchild} outlived module teardown: the tree was not contained"
12740        );
12741    }
12742
12743    /// The mutation control: with containment withheld, the grandchild survives
12744    /// the same kill.
12745    ///
12746    /// This is the defect reproduction from #109 — a direct-child kill reaches
12747    /// one pid, and the grandchild is a different process. It spawns OUTSIDE the
12748    /// supervisor because `spawn_and_mark_running` now always contains on
12749    /// Windows, which is the point: there is no longer a path that spawns
12750    /// uncontained, so the control has to construct one.
12751    ///
12752    /// Its job is to keep `teardown_reaps_the_grandchild` honest. If the
12753    /// grandchild ever dies here, that test is passing for a reason unrelated to
12754    /// the job object and the containment claim is unproven.
12755    #[test]
12756    fn an_uncontained_grandchild_survives_a_direct_child_kill() {
12757        let dir = TestTempDir::new("teardown-uncontained");
12758        let pid_file = dir.join("grandchild.pid");
12759        let mut child = std::process::Command::new(stub_path())
12760            .env("FAKE_AFT_NEVER_CONNECT", "1")
12761            .env(
12762                "FAKE_AFT_GRANDCHILD_PID_FILE",
12763                pid_file.display().to_string(),
12764            )
12765            .stdin(std::process::Stdio::null())
12766            .stdout(std::process::Stdio::null())
12767            .stderr(std::process::Stdio::null())
12768            .spawn()
12769            .expect("spawn the uncontained fixture");
12770        let grandchild = read_grandchild_pid(&pid_file);
12771
12772        // Exactly what the pre-fix teardown did: kill the direct child.
12773        child.kill().expect("kill the direct child");
12774        let _ = child.wait();
12775
12776        assert!(
12777            subc_jobobject::process_exists(grandchild),
12778            "grandchild {grandchild} died with the direct child, so this control no longer \
12779             distinguishes contained from uncontained teardown and the regression test is \
12780             passing vacuously"
12781        );
12782
12783        // The orphan this control demonstrates is the leak the fix prevents, so
12784        // the control must not leave one behind.
12785        kill_tree(grandchild);
12786    }
12787
12788    /// Crash durability: closing the containment handle reaps the tree with no
12789    /// teardown code running at all.
12790    ///
12791    /// This is the case `taskkill /T` cannot cover — a daemon that dies cannot
12792    /// call anything — and it is why containment is a kernel property of the
12793    /// handle rather than a step in the drain. Discovered by getting the
12794    /// mutation control wrong: clearing `job` to "disable" containment instead
12795    /// killed the tree, which is the guarantee, not a mistake.
12796    #[tokio::test]
12797    async fn dropping_containment_reaps_the_grandchild() {
12798        let mut fixture = fixture("drop-containment", "tree-drop");
12799        let grandchild = fixture.grandchild;
12800
12801        assert!(subc_jobobject::process_exists(grandchild));
12802
12803        // No `drain` call, no kill: dropping the handle is the entire mechanism.
12804        fixture.child.as_mut().expect("child present").job = None;
12805
12806        assert!(
12807            subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
12808            "grandchild {grandchild} survived the containment handle closing, so a daemon \
12809             crash would leave the tree behind"
12810        );
12811    }
12812
12813    /// Kill a pid and its tree, then confirm it is gone.
12814    fn kill_tree(pid: u32) {
12815        let _ = std::process::Command::new("taskkill.exe")
12816            .args(["/PID", &pid.to_string(), "/T", "/F"])
12817            .stdin(std::process::Stdio::null())
12818            .stdout(std::process::Stdio::null())
12819            .stderr(std::process::Stdio::null())
12820            .status();
12821        assert!(
12822            subc_jobobject::wait_for_process_exit(pid, Duration::from_secs(10)),
12823            "could not clean up grandchild {pid}"
12824        );
12825    }
12826}
12827
12828#[cfg(test)]
12829mod privacy_trampoline_configuration_tests {
12830    #[tokio::test]
12831    #[cfg_attr(not(target_os = "macos"), ignore = "requires macOS privacy trampoline")]
12832    async fn macos_spawn_without_a_configured_trampoline_refuses_by_name() {
12833        #[cfg(target_os = "macos")]
12834        {
12835            let supervisor = super::Supervisor::new(
12836                std::sync::Arc::new(crate::Registry::default()),
12837                super::RestartPolicy::default(),
12838            );
12839            let error = supervisor.spawn(spec()).unwrap_err();
12840            assert!(
12841                error
12842                    .to_string()
12843                    .contains("no privacy trampoline configured"),
12844                "{error}"
12845            );
12846        }
12847    }
12848
12849    #[tokio::test]
12850    #[cfg_attr(not(target_os = "macos"), ignore = "requires macOS privacy trampoline")]
12851    async fn macos_wrong_trampoline_is_rejected_by_the_startup_probe() {
12852        #[cfg(target_os = "macos")]
12853        {
12854            let supervisor = super::Supervisor::new(
12855                std::sync::Arc::new(crate::Registry::default()),
12856                super::RestartPolicy::default(),
12857            )
12858            .with_privacy_trampoline(std::env::current_exe().unwrap());
12859            let error = supervisor.spawn(spec()).unwrap_err();
12860            assert!(
12861                error
12862                    .to_string()
12863                    .contains("binary does not implement the privacy trampoline protocol"),
12864                "{error}"
12865            );
12866        }
12867    }
12868
12869    #[cfg(target_os = "macos")]
12870    fn spec() -> super::ModuleSpec {
12871        super::ModuleSpec {
12872            module_id: "privacy-configuration".into(),
12873            program: "/bin/sleep".into(),
12874            args: vec!["30".into()],
12875            env: vec![],
12876            reserved: false,
12877            reserved_prefixes: vec![],
12878            protocol: subc_control::ModuleProtocol::None,
12879            overlap: super::ModuleOverlap::Exclusive,
12880        }
12881    }
12882}
12883
12884#[cfg(test)]
12885mod privacy_exec_boundary_tests {
12886    #[cfg(target_os = "macos")]
12887    use super::*;
12888    #[cfg(target_os = "macos")]
12889    use std::{
12890        io::{Read, Write},
12891        net::{TcpListener, TcpStream},
12892    };
12893
12894    /// Unit-test-only pause at the actual early image read, not at a later
12895    /// status read. Production supervisors never inspect this environment key.
12896    #[cfg(target_os = "macos")]
12897    pub(super) fn before_image_sample(spec: &ModuleSpec, pid: u32) {
12898        if let Some((_, path)) = spec
12899            .env
12900            .iter()
12901            .find(|(key, _)| key == "SUBC_TEST_PRIVACY_SAMPLE_BARRIER")
12902        {
12903            let mut barrier = TcpStream::connect(path).unwrap();
12904            barrier
12905                .set_read_timeout(Some(Duration::from_secs(30)))
12906                .unwrap();
12907            barrier.write_all(&pid.to_ne_bytes()).unwrap();
12908            let mut release = [0];
12909            barrier.read_exact(&mut release).unwrap();
12910            assert_eq!(&release, b"X");
12911        }
12912    }
12913
12914    #[cfg(target_os = "macos")]
12915    fn spec(program: &str, args: &[&str]) -> ModuleSpec {
12916        ModuleSpec {
12917            module_id: "privacy-boundary".into(),
12918            program: program.into(),
12919            args: args.iter().map(|arg| (*arg).into()).collect(),
12920            env: vec![],
12921            reserved: false,
12922            reserved_prefixes: vec![],
12923            protocol: ModuleProtocol::None,
12924            overlap: ModuleOverlap::Exclusive,
12925        }
12926    }
12927
12928    #[cfg(target_os = "macos")]
12929    async fn accept(listener: TcpListener) -> TcpStream {
12930        // Socket readiness, not elapsed time, establishes both pause points.
12931        let listener = tokio::net::TcpListener::from_std({
12932            listener.set_nonblocking(true).unwrap();
12933            listener
12934        })
12935        .unwrap();
12936        let (stream, _) = tokio::time::timeout(Duration::from_secs(30), listener.accept())
12937            .await
12938            .unwrap()
12939            .unwrap();
12940        let stream = stream.into_std().unwrap();
12941        stream.set_nonblocking(false).unwrap();
12942        stream
12943            .set_read_timeout(Some(Duration::from_secs(30)))
12944            .unwrap();
12945        stream
12946    }
12947
12948    #[tokio::test]
12949    #[cfg_attr(not(target_os = "macos"), ignore = "requires macOS privacy trampoline")]
12950    async fn macos_roster_withholds_a_nonnull_trampoline_image_until_exec_confirmation() {
12951        #[cfg(target_os = "macos")]
12952        {
12953            let root = subc_test_support::TestTempDir::new("privacy-roster-barrier");
12954            // Loopback sockets also work when the replay adapter's TMPDIR is
12955            // longer than Darwin's Unix-domain socket path limit.
12956            let exec_listener = TcpListener::bind("127.0.0.1:0").unwrap();
12957            let sample_listener = TcpListener::bind("127.0.0.1:0").unwrap();
12958            let record = root.join("live-children.json");
12959            let supervisor =
12960                Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
12961                    .with_live_children_record(&record);
12962            let runtime = supervisor.runtime_config();
12963            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
12964            let mut spec = spec("/bin/sleep", &["30"]);
12965            spec.env = vec![
12966                (
12967                    "SUBC_TEST_PRIVACY_EXEC_BARRIER".into(),
12968                    exec_listener.local_addr().unwrap().to_string(),
12969                ),
12970                (
12971                    "SUBC_TEST_PRIVACY_SAMPLE_BARRIER".into(),
12972                    sample_listener.local_addr().unwrap().to_string(),
12973                ),
12974            ];
12975            let spawn = tokio::task::spawn_blocking(move || {
12976                spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap()
12977            });
12978            let mut sample = accept(sample_listener).await;
12979            let mut pid = [0; 4];
12980            sample.read_exact(&mut pid).unwrap();
12981            let pid = u32::from_ne_bytes(pid);
12982            let mut exec = accept(exec_listener).await;
12983            let mut ready = [0];
12984            exec.read_exact(&mut ready).unwrap();
12985            assert_eq!(&ready, b"R");
12986            // The early read is guaranteed to see a real, non-null trampoline
12987            // image: the fixture has reached its barrier and cannot exec yet.
12988            let trampoline = subc_os::file_identity(&test_privacy_trampoline()).unwrap();
12989            assert_eq!(
12990                observe_spawned_image(pid).unwrap().executable,
12991                Some(trampoline)
12992            );
12993            sample.write_all(b"X").unwrap();
12994            let mut child = spawn.await.unwrap();
12995            let early = crate::live_children::read_record(&record).unwrap();
12996            assert_eq!(early.len(), 1);
12997            assert_eq!(early[0].pid, pid);
12998            assert_eq!(
12999                early[0].executable, None,
13000                "unconfirmed trampoline image entered the roster"
13001            );
13002            assert!(child.report_ready.get().is_none());
13003            // The barrier's duration is unrelated to the production five-second
13004            // exec budget. Start the test's confirmation budget upon release.
13005            child.privacy_exec.as_mut().unwrap().deadline =
13006                tokio::time::Instant::now() + Duration::from_secs(30);
13007            exec.write_all(b"X").unwrap();
13008            child.confirm_privacy_exec().await;
13009            assert_eq!(child.spawn_failure, None);
13010            assert!(child.report_ready.get().is_some());
13011            let confirmed = crate::live_children::read_record(&record).unwrap();
13012            let module = subc_os::file_identity(std::path::Path::new("/bin/sleep")).unwrap();
13013            assert_ne!(module, trampoline);
13014            assert_eq!(confirmed[0].executable, Some(module.into()));
13015            child.start_kill().unwrap();
13016            child.wait().await.unwrap();
13017            child.release_roster();
13018        }
13019    }
13020
13021    #[tokio::test]
13022    #[cfg_attr(not(target_os = "macos"), ignore = "requires macOS privacy trampoline")]
13023    async fn macos_already_exited_121_in_try_wait_is_a_module_exit_not_a_trampoline_refusal() {
13024        #[cfg(target_os = "macos")]
13025        {
13026            let registry = Arc::new(Registry::default());
13027            let policy = RestartPolicy::new(0, Duration::ZERO);
13028            let supervisor = Supervisor::new_for_test(Arc::clone(&registry), policy);
13029            let runtime = supervisor.runtime_config();
13030            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
13031            let spec = spec("/bin/sh", &["-c", "exit 121"]);
13032            // Drive spawn and confirmation separately instead of starting the
13033            // monitor. WNOWAIT observes a real exit without consuming its status,
13034            // so confirmation's first try_wait must take the already-exited arm.
13035            let mut child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
13036            let pid = child.pid;
13037            tokio::task::spawn_blocking(move || {
13038                subc_os::fork_exec_test::wait_for_child_exit_without_reaping(pid)
13039            })
13040            .await
13041            .unwrap()
13042            .unwrap();
13043            child.privacy_exec.as_mut().unwrap().deadline =
13044                tokio::time::Instant::now() + Duration::from_secs(30);
13045            let status = child.wait().await.unwrap();
13046            assert_eq!(status.code(), Some(121));
13047            assert!(child.privacy_exec.is_none());
13048            assert!(
13049                child.report_ready.get().is_none(),
13050                "an exited module must not publish a live pid"
13051            );
13052            let report = classify_reaped_child_exit(&snapshot, &child, &status);
13053            on_child_exit(
13054                &spec,
13055                policy,
13056                &registry,
13057                &snapshot,
13058                &runtime.terminal_ring,
13059                &runtime.spawn_events,
13060                &runtime.child_roster,
13061                report,
13062            )
13063            .await;
13064            let state = lock_snapshot(&snapshot).unwrap();
13065            assert_eq!(state.state, ModuleState::Failed);
13066            assert_eq!(state.last_exit.as_ref().unwrap().code, Some(121));
13067            assert_eq!(state.reported_pid(), None);
13068            drop(state);
13069            let history = runtime.terminal_ring.lock().unwrap().snapshot();
13070            assert_eq!(history.entries.len(), 1);
13071            let terminal = &history.entries[0];
13072            assert_eq!(terminal.exit_code, Some(121));
13073            assert_eq!(terminal.exit_kind, subc_control::TerminalExitKind::Crash);
13074            assert_eq!(terminal.disposition, TerminalDisposition::Failed);
13075            assert_eq!(
13076                terminal.disposition_detail.as_deref(),
13077                Some(policy.budget_exhausted_detail().as_str()),
13078                "module exit 121 was classified as a trampoline refusal: {terminal:?}"
13079            );
13080            assert_eq!(child.spawn_failure, None);
13081            child.release_roster();
13082        }
13083    }
13084}
13085
13086/// The daemon's real spawn path hands a subc-wire child its launch nonce on
13087/// descriptor 3, without an environment copy. The shell records the nonce
13088/// and its environment after exec so these tests observe the real handover.
13089#[cfg(all(test, unix))]
13090mod launch_nonce_descriptor_tests {
13091    use super::{ChildRoster, ModuleProtocol, ModuleSpec, SupervisorHandle};
13092    use crate::stderr_tail::{StderrRing, StderrTailConfig};
13093    use std::{
13094        path::PathBuf,
13095        sync::{Arc, Mutex},
13096        time::{Duration, Instant},
13097    };
13098    use subc_test_support::TestTempDir;
13099
13100    async fn probe(role: super::SpawnRole) {
13101        let scratch = TestTempDir::new("launch-nonce-descriptor");
13102        let fd_copy = scratch.join("from-descriptor");
13103        let env_copy = scratch.join("environment");
13104        let script = format!(
13105            "env > '{env}.tmp' && mv '{env}.tmp' '{env}'; cat <&3 > '{fd}.tmp' && mv '{fd}.tmp' '{fd}'; sleep 30",
13106            fd = fd_copy.display(), env = env_copy.display(),
13107        );
13108        let xdg = |name: &str| (name.to_string(), scratch.join(name).display().to_string());
13109        let spec = ModuleSpec {
13110            module_id: "nonce-descriptor-probe".to_string(),
13111            program: PathBuf::from("/bin/sh"),
13112            args: vec!["-c".to_string(), script],
13113            env: vec![
13114                xdg("XDG_DATA_HOME"),
13115                xdg("XDG_RUNTIME_DIR"),
13116                xdg("XDG_CONFIG_HOME"),
13117                ("SUBC_LAUNCH_NONCE".to_string(), "must-not-leak".to_string()),
13118            ],
13119            reserved: true,
13120            reserved_prefixes: Vec::new(),
13121            protocol: ModuleProtocol::Subc,
13122            overlap: Default::default(),
13123        };
13124        let handle = SupervisorHandle::new();
13125        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
13126        let roster = ChildRoster::default();
13127        #[cfg(target_os = "macos")]
13128        {
13129            let path = super::test_privacy_trampoline();
13130            roster.set_privacy_trampoline(super::probe_privacy_trampoline(&path).map(|()| path));
13131        }
13132        let child = super::spawn_child_in_slot(
13133            &spec,
13134            None,
13135            Some(&handle),
13136            &ring,
13137            None,
13138            &roster,
13139            #[cfg(target_os = "linux")]
13140            None,
13141            role,
13142            matches!(role, super::SpawnRole::SwapCandidate),
13143        )
13144        .expect("spawn probe");
13145        let deadline = Instant::now() + Duration::from_secs(10);
13146        while !(fd_copy.exists() && env_copy.exists()) {
13147            assert!(Instant::now() < deadline, "probe never wrote its copies");
13148            tokio::time::sleep(Duration::from_millis(20)).await;
13149        }
13150        let nonce = std::fs::read_to_string(fd_copy).unwrap();
13151        assert!(!nonce.is_empty());
13152        let environment = std::fs::read_to_string(env_copy).unwrap();
13153        assert!(environment
13154            .lines()
13155            .any(|line| line.starts_with("SUBC_LAUNCH_NONCE_FD=3:")));
13156        let copy = environment
13157            .lines()
13158            .find_map(|line| line.strip_prefix("SUBC_LAUNCH_NONCE="));
13159        assert_eq!(
13160            copy, None,
13161            "Unix children must never receive the environment nonce"
13162        );
13163        if matches!(role, super::SpawnRole::Plain) {
13164            assert_eq!(
13165                handle.spawn_nonce(&spec.module_id).as_deref(),
13166                Some(nonce.as_str())
13167            );
13168        }
13169        drop(child);
13170    }
13171
13172    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
13173    async fn a_spawned_module_receives_its_nonce_only_on_descriptor_3() {
13174        probe(super::SpawnRole::Plain).await;
13175    }
13176
13177    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
13178    async fn swap_candidate_receives_its_nonce_only_on_descriptor_3() {
13179        probe(super::SpawnRole::SwapCandidate).await;
13180    }
13181}
13182
13183#[cfg(all(test, target_os = "linux"))]
13184mod cgroup_containment_tests {
13185    use super::*;
13186    use subc_test_support::TestTempDir;
13187
13188    fn running(pid: u32) -> bool {
13189        // An orphan can remain a zombie until the container init reaps it.
13190        std::fs::read_to_string(format!("/proc/{pid}/stat"))
13191            .ok()
13192            .and_then(|stat| {
13193                stat.rsplit_once(") ")
13194                    .map(|(_, rest)| rest.starts_with('Z'))
13195            })
13196            .is_some_and(|zombie| !zombie)
13197    }
13198
13199    #[tokio::test]
13200    async fn linux_teardown_reaps_the_grandchild() {
13201        teardown_tree("linux_teardown_reaps_the_grandchild", false).await;
13202    }
13203
13204    #[tokio::test]
13205    async fn linux_shutdown_straggler_reaps_the_grandchild() {
13206        teardown_tree("linux_shutdown_straggler_reaps_the_grandchild", true).await;
13207    }
13208
13209    async fn teardown_tree(test_name: &str, shutdown: bool) {
13210        let dir = TestTempDir::new(test_name);
13211        let root = PathBuf::from(format!(
13212            "/sys/fs/cgroup/subc-tree-test-{test_name}-{}-{}",
13213            std::process::id(),
13214            unix_ms_now()
13215        ));
13216        if let Err(error) = std::fs::create_dir(&root) {
13217            assert!(
13218                std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
13219                "required cgroup test cannot execute: {error}"
13220            );
13221            eprintln!(
13222                "SKIP {test_name}: no writable delegated cgroup at {}: {error}",
13223                root.display()
13224            );
13225            return;
13226        }
13227        let placement = subc_cgroup::prepare_at(&root)
13228            .expect("prepare isolated kernel cgroup")
13229            .expect("isolated cgroup is delegated");
13230        let module_id = "tree-teardown";
13231        let module = placement
13232            .module_path(module_id)
13233            .expect("create isolated module cgroup");
13234        if !module.join("cgroup.kill").exists() {
13235            std::fs::remove_dir(&module).unwrap();
13236            std::fs::remove_dir(root.join("subc-modules")).unwrap();
13237            std::fs::remove_dir(&root).unwrap();
13238            assert!(
13239                std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
13240                "required cgroup.kill interface unavailable"
13241            );
13242            eprintln!("SKIP {test_name}: cgroup.kill unavailable (kernel < 5.14)");
13243            return;
13244        }
13245        let supervisor = Supervisor::new_for_test(
13246            Arc::new(Registry::default()),
13247            RestartPolicy::new(3, Duration::ZERO),
13248        )
13249        .with_cgroup_placement(Some(placement));
13250        let mut runtime = supervisor.runtime_config();
13251        runtime.child_roster = runtime
13252            .child_roster
13253            .for_module(Arc::new(Mutex::new(Duration::from_millis(100))));
13254        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
13255        let pid_file = dir.join("grandchild.pid");
13256        let spec = ModuleSpec {
13257            module_id: module_id.to_string(),
13258            program: PathBuf::from("/bin/sh"),
13259            args: vec![
13260                "-c".into(),
13261                "trap '' TERM; sleep 600 & echo $! > \"$1\"; wait".into(),
13262                "fixture".into(),
13263                pid_file.display().to_string(),
13264            ],
13265            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
13266                .into_iter()
13267                .map(|key| (key.to_string(), dir.display().to_string()))
13268                .collect(),
13269            reserved: false,
13270            reserved_prefixes: Vec::new(),
13271            protocol: ModuleProtocol::None,
13272            overlap: Default::default(),
13273        };
13274        let child =
13275            spawn_and_mark_running(&spec, &runtime, &snapshot).expect("spawn supervised tree");
13276        let deadline = tokio::time::Instant::now() + Duration::from_secs(10);
13277        let grandchild: u32 = loop {
13278            if let Ok(contents) = std::fs::read_to_string(&pid_file) {
13279                if let Ok(pid) = contents.trim().parse() {
13280                    break pid;
13281                }
13282            }
13283            assert!(
13284                tokio::time::Instant::now() < deadline,
13285                "grandchild pid was not recorded"
13286            );
13287            tokio::time::sleep(Duration::from_millis(10)).await;
13288        };
13289        assert!(
13290            running(grandchild),
13291            "grandchild must be alive before teardown"
13292        );
13293        if shutdown {
13294            let mut child = child;
13295            crate::child_roster::end_children_for_daemon_shutdown(
13296                &runtime.child_roster,
13297                false,
13298                std::future::pending(),
13299            )
13300            .await;
13301            child.wait().await.expect("reap shutdown straggler");
13302        } else {
13303            drain_child_to_state(
13304                module_id,
13305                ModuleProtocol::None,
13306                StopNotice::NotSent,
13307                &Registry::default(),
13308                None,
13309                &snapshot,
13310                &runtime.terminal_ring,
13311                &SpawnEventFeed::default(),
13312                child,
13313                Duration::from_millis(100),
13314                ModuleState::Stopped,
13315                Some(false),
13316            )
13317            .await
13318            .expect("real supervisor teardown");
13319        }
13320        let deadline = tokio::time::Instant::now() + Duration::from_secs(5);
13321        while running(grandchild) && tokio::time::Instant::now() < deadline {
13322            tokio::time::sleep(Duration::from_millis(10)).await;
13323        }
13324        let survived = running(grandchild);
13325        // Kill a surviving grandchild so a failed test does not leave it behind.
13326        if survived {
13327            let pid = rustix::process::Pid::from_raw(grandchild as i32).unwrap();
13328            let _ = rustix::process::kill_process(pid, rustix::process::Signal::KILL);
13329            tokio::time::sleep(Duration::from_millis(100)).await;
13330        }
13331        if module.exists() {
13332            std::fs::remove_dir(&module).expect("remove empty module cgroup");
13333        }
13334        std::fs::remove_dir(root.join("subc-modules")).unwrap();
13335        std::fs::remove_dir(&root).unwrap();
13336        assert!(
13337            !survived,
13338            "grandchild {grandchild} outlived module teardown"
13339        );
13340        eprintln!("EXECUTED {test_name}: grandchild {grandchild} killed in isolated cgroup");
13341    }
13342}