Skip to main content

subc_daemon/
supervise.rs

1use std::{
2    collections::{HashMap, HashSet, VecDeque},
3    error::Error,
4    fmt, io,
5    path::PathBuf,
6    process::{ExitStatus, Stdio},
7    sync::{Arc, Mutex, OnceLock},
8    time::{Duration, SystemTime, UNIX_EPOCH},
9};
10
11use cortexkit_log::Retention;
12use serde_json::Value;
13use subc_control::{
14    ClientControlPush, LiveSpawn, ModuleProtocol, RouteCloseReason, SpawnCursor, SpawnEvent,
15    SpawnEventKind, SpawnSnapshot, SupervisorHealthStatus, TerminalDisposition, TerminalExitKind,
16};
17use subc_protocol::{
18    manifest::{SelfSignalKind, SignalAnchor},
19    session::{
20        HealthReport, HealthStatus, ModuleControlCommand, ModuleControlRequest,
21        MODULE_CONTROL_OP_HEALTH_CHECK,
22    },
23    Flags, FrameType, Priority, SUBC_LAUNCH_NONCE_ENV, SUBC_MODULE_ID_ENV,
24};
25use tokio::{
26    process::{Child, Command},
27    sync::{mpsc, oneshot, watch, Mutex as AsyncMutex},
28    task::JoinHandle,
29    time::{sleep, sleep_until, timeout, timeout_at, Instant},
30};
31use tracing::{debug, error, info, warn};
32
33use crate::{
34    child_roster::ChildRoster,
35    daemon_config::{
36        CAPTURE_KEEP_ENV, CAPTURE_MAX_AGE_DAYS_ENV, CAPTURE_MAX_FILE_MB_ENV, CK_LOG_ENV,
37    },
38    forwarding::{
39        CloseReason, ForwardingError, ForwardingTable, GoodbyeTarget, ModuleControlRpcOutcome,
40        ModuleDrainTarget, PendingModuleControlRpc,
41    },
42    provenance::{spawned_file_identity, ExecutableIdentityProbe, SpawnedFileIdentity},
43    registry::{ConnectionId, RegistrationEndReason, RegistryError},
44    stderr_tail::{
45        pump_stderr_to, pump_stdout_to, ChildOutputSink, StderrRing, StderrTailConfig,
46        StderrTailSnapshot,
47    },
48    terminal_ring::{TerminalHistorySnapshot, TerminalRecord, TerminalRing, TerminalRingConfig},
49    Frame, FrameSink, Registry,
50};
51
52#[path = "supervise_swap.rs"]
53mod swap;
54
55/// Command-line flag used by supervised modules to find subc.
56///
57/// subc launches module-mode children as `<module> --subc <connection-file-path>`.
58/// The path points at the TCP+key connection file; it is not an ambient signal and
59/// is never inherited by standalone children.
60pub const SUBC_ARG: &str = "--subc";
61
62const DEFAULT_MAX_RESTARTS: u32 = 3;
63const DEFAULT_BACKOFF: Duration = Duration::from_millis(100);
64const DEFAULT_MAX_BACKOFF: Duration = Duration::from_secs(30);
65/// The span `DEFAULT_MAX_RESTARTS` is counted over. Ten minutes is long enough
66/// to contain a real crash loop (which respawns in seconds) and short enough
67/// that unrelated crashes hours apart never accumulate into a permanent stop.
68const DEFAULT_RESTART_WINDOW: Duration = Duration::from_secs(600);
69/// How long a drain waits for already-dispatched requests to finalize before
70/// the child is torn down. Sized for TOOL-SCALE work (bash, inspect, builds),
71/// not RPC-scale: the original 2s value silently cut nearly every real tool
72/// call at the fence, making the wait-for-finalize design decorative for the
73/// workloads it existed for. Quiescence short-circuits, so an idle module
74/// restarts immediately regardless of this value; the budget is spent only
75/// when a genuine in-flight request is worth finishing. Per-module override:
76/// `drain_timeout_ms` in subc.jsonc; per-restart override: the operator's
77/// `supervisor.restart{drain_timeout_ms}` (0 = cut now, for wedge bounces
78/// where a stuck request will never settle).
79pub const DEFAULT_DRAIN_TIMEOUT: Duration = Duration::from_secs(30);
80const REGISTRY_RELEASE_TIMEOUT: Duration = Duration::from_secs(1);
81const REGISTRY_RELEASE_POLL: Duration = Duration::from_millis(10);
82/// How long a restart waits for the exited process's output readers.
83///
84/// The restart does not depend on the stderr reader finishing. A reader still
85/// running at this bound is left running, and whatever it delivers later goes
86/// into the exited process's own section of the stderr ring (see
87/// `StderrRing::push_line_from`), ending naturally at EOF on its pipe. So the
88/// bound no longer decides whether a crash's last lines are kept: under load
89/// the reader may simply not have been scheduled yet, and cutting it there
90/// discarded exactly the lines that explained the crash.
91///
92/// What the bound still decides is when the tail starts reporting
93/// `Incomplete`: a pipe open past it usually means a descendant of the exited
94/// process still holds it, and the tail cannot claim to be whole until that
95/// pipe closes. That is also the only case in which the wait costs the restart
96/// anything, because a pipe with no other holder reaches EOF when the process
97/// exits. Under load a slow reader can show `Incomplete` briefly; it returns to
98/// `Captured` at EOF with nothing lost.
99///
100/// The stdout reader carries no ring, only the capture file, and is still
101/// stopped at this bound so an old process's stdout cannot trail into the file
102/// after its successor starts.
103const STDERR_PUMP_DRAIN_TIMEOUT: Duration = Duration::from_millis(250);
104/// Maximum number of supervised process spawn/exit facts retained per daemon incarnation.
105pub const SPAWN_EVENT_RING_CAPACITY: usize = 4096;
106const SPAWN_SUBSCRIBER_BUFFER: usize = SPAWN_EVENT_RING_CAPACITY + 1;
107/// Terminal Error code for a `supervisor.spawn_subscribe` stream the daemon
108/// dropped because the subscriber stopped draining its frames. The detail's
109/// `first_undelivered_cursor` names the first event it did not receive; the
110/// client resubscribes from the last cursor it did receive.
111pub(crate) const SPAWN_SUBSCRIBER_LAGGED_CODE: &str = "spawn_subscriber_lagged";
112
113#[cfg(target_os = "macos")]
114struct PrivacyExec {
115    reader: tokio::io::unix::AsyncFd<std::io::PipeReader>,
116    deadline: tokio::time::Instant,
117    expected: Option<subc_os::FileIdentity>,
118    trampoline: Option<subc_os::FileIdentity>,
119    script: bool,
120    module_id: String,
121}
122
123#[cfg(all(test, target_os = "macos"))]
124pub(crate) fn test_privacy_trampoline() -> PathBuf {
125    // Cargo's unit-test executable lives in <profile>/deps; its fixture bin
126    // lives beside that directory. This honors custom CARGO_TARGET_DIR too.
127    let path = std::env::current_exe()
128        .unwrap()
129        .parent()
130        .unwrap()
131        .parent()
132        .unwrap()
133        .join("privacy-trampoline-fixture");
134    // Without the fixture every macOS spawn is refused, and the tests that
135    // spawn fail later as a module in state Failed, which names the wrong
136    // cause. `cargo test -p subc-daemon --lib` alone does not build it.
137    assert!(
138        path.exists(),
139        "privacy-trampoline-fixture not built at {}: run `cargo build -p subc-daemon \
140         --bins --features test-support` or `cargo test -p subc-daemon` first",
141        path.display()
142    );
143    path
144}
145
146#[cfg(target_os = "macos")]
147fn probe_privacy_trampoline(path: &std::path::Path) -> Result<(), String> {
148    use std::io::Read;
149    let mut probe = std::process::Command::new(path)
150        .args(["__disclaim-exec", "--probe"])
151        .stdin(Stdio::null())
152        .stdout(Stdio::piped())
153        .stderr(Stdio::piped())
154        .spawn()
155        .map_err(|error| {
156            format!(
157                "privacy trampoline probe failed for {}: {error}",
158                path.display()
159            )
160        })?;
161    let deadline = std::time::Instant::now() + Duration::from_secs(5);
162    let status = loop {
163        match probe.try_wait() {
164            Ok(Some(status)) => break status,
165            Ok(None) if std::time::Instant::now() < deadline => {
166                std::thread::sleep(Duration::from_millis(5))
167            }
168            result => {
169                let _ = probe.kill();
170                let _ = probe.wait();
171                return Err(format!(
172                    "privacy trampoline probe failed or timed out for {}: {result:?}",
173                    path.display()
174                ));
175            }
176        }
177    };
178    let mut answer = String::new();
179    if let Some(stdout) = probe.stdout.take() {
180        let _ = stdout.take(256).read_to_string(&mut answer);
181    }
182    if status.success() && answer.trim() == subc_os::privacy_identity::TRAMPOLINE_PROBE {
183        return Ok(());
184    }
185    let mut diagnostic = String::new();
186    if let Some(stderr) = probe.stderr.take() {
187        let _ = stderr.take(1024).read_to_string(&mut diagnostic);
188    }
189    let cause = diagnostic
190        .trim()
191        .strip_prefix("ck-subc: own privacy identity refused: ")
192        .unwrap_or("binary does not implement the privacy trampoline protocol");
193    Err(format!(
194        "{cause}: probe of {} exited {status}",
195        path.display()
196    ))
197}
198
199#[cfg(target_os = "macos")]
200fn privacy_command(
201    spec: &ModuleSpec,
202    roster: &ChildRoster,
203) -> Result<
204    (
205        Command,
206        Option<PrivacyExec>,
207        subc_os::privacy_identity::ExecAcknowledgement,
208    ),
209    SuperviseError,
210> {
211    let failure = |cause: String| {
212        warn!(module_id = %spec.module_id, %cause, "privacy identity trampoline refused module spawn");
213        SuperviseError::Spawn {
214            program: spec.program.clone(),
215            source: io::Error::other(cause),
216            cgroup_path: None,
217        }
218    };
219    let trampoline = roster.privacy_trampoline().map_err(failure)?;
220    // Resolve PATH with the same environment the Command will receive. For
221    // scripts retain the existing orphan-identity rule: the kernel chooses
222    // the interpreter, and its observed image is the one recorded. Do not
223    // duplicate the kernel's shebang/PATH interpreter resolution in Rust.
224    let program = if spec.program.components().count() == 1 && !spec.program.is_absolute() {
225        let path = spec
226            .env
227            .iter()
228            .find(|(key, _)| key == "PATH")
229            .map(|(_, value)| std::ffi::OsString::from(value))
230            .or_else(|| std::env::var_os("PATH"))
231            .unwrap_or_else(|| "/usr/bin:/bin".into());
232        std::env::split_paths(&path)
233            .map(|dir| dir.join(&spec.program))
234            .find(|path| path.is_file())
235            .unwrap_or_else(|| spec.program.clone())
236    } else {
237        spec.program.clone()
238    };
239    let expected = subc_os::file_identity(&program);
240    let trampoline_image = subc_os::file_identity(&trampoline);
241    let script = {
242        use std::io::Read;
243        let mut prefix = [0u8; 2];
244        std::fs::File::open(&program)
245            .is_ok_and(|mut file| file.read_exact(&mut prefix).is_ok() && &prefix == b"#!")
246    };
247    if expected.is_none() || expected == subc_os::file_identity(&trampoline) {
248        return Err(failure(
249            "privacy identity module executable is missing or is the trampoline itself".to_string(),
250        ));
251    }
252    let (reader, ack) = subc_os::privacy_identity::ExecAcknowledgement::pipe()
253        .map_err(|error| failure(error.to_string()))?;
254    let reader =
255        tokio::io::unix::AsyncFd::new(reader).map_err(|error| failure(error.to_string()))?;
256    let mut command = Command::new(&trampoline);
257    command
258        .arg("__disclaim-exec")
259        .arg(ack.fd().to_string())
260        .arg(&program);
261    ack.install(command.as_std_mut());
262    Ok((
263        command,
264        Some(PrivacyExec {
265            reader,
266            deadline: tokio::time::Instant::now() + Duration::from_secs(5),
267            expected,
268            trampoline: trampoline_image,
269            script,
270            module_id: spec.module_id.clone(),
271        }),
272        ack,
273    ))
274}
275
276struct SupervisedChild {
277    child: Child,
278    #[cfg(target_os = "macos")]
279    privacy_exec: Option<PrivacyExec>,
280    /// Set once this launch's exec acknowledgement confirms the module image.
281    /// On macOS the pid first runs the `ck-subc` launch trampoline (see
282    /// `subc_os::privacy_identity`), which then replaces itself with the
283    /// module. The supervisor owns and can kill that pid from spawn, but
284    /// status readers report it only after this latch is set, so nothing
285    /// reports the trampoline's image as the module's.
286    #[cfg(target_os = "macos")]
287    report_ready: Arc<OnceLock<()>>,
288    /// Refusal before the module image was accepted, retained for terminal records.
289    spawn_failure: Option<String>,
290    /// The protocol this process was launched with. A reload can store a new
291    /// launch spec with a different protocol, but that takes effect only at the
292    /// next spawn, so this process keeps being handled by the protocol it
293    /// actually speaks.
294    protocol: ModuleProtocol,
295    /// This process's cgroup name: a bounded module/slot label followed by a
296    /// spawn suffix unique to this process (when cgroup placement is on). A
297    /// retired process in a slot may still be draining when a later one is
298    /// spawned into that slot, so the suffix keeps the later process out of
299    /// the retired one's cgroup, which is the domain a kill applies to.
300    #[cfg(target_os = "linux")]
301    module_id: String,
302    #[cfg(target_os = "linux")]
303    cgroup_placement: Option<subc_cgroup::Placement>,
304    /// The job that contains this child and every process it spawns (issue #109).
305    ///
306    /// Dropping this handle is what reaps a surviving tree when no supervisor
307    /// code runs — a daemon crash — because the job carries
308    /// `JOB_OBJECT_LIMIT_KILL_ON_JOB_CLOSE`.
309    ///
310    /// That limit is not crash-only, and the difference is worth knowing: a
311    /// Windows daemon *stop* is `taskkill` or the scheduler's `/End` — the
312    /// SIGTERM handler is `#[cfg(unix)]` — so the daemon dies with no stop
313    /// notice and the kernel closes the job handle, `TerminateProcess`ing every
314    /// module at once. Before this change they survived that, saw EOF on the
315    /// control socket, and ran their own teardown; Unix keeps that path
316    /// deliberately, so a module can seal a WAL or close a capture rather than
317    /// be killed mid-write. So this trades graceful teardown on every Windows
318    /// daemon stop for containment on a crash, which is the right way round
319    /// today: orphaned GPU workers are a reported, recurring problem, and the
320    /// modules that write most heavily do not run on Windows.
321    ///
322    /// The fix is a real Windows stop path — the daemon draining before it
323    /// exits, the twin of the Unix SIGTERM handler. Once it exists, this limit
324    /// reaches only what the drain left behind, which is what it should reach.
325    #[cfg(windows)]
326    job: Option<subc_jobobject::JobObject>,
327    stdout_pump: Option<JoinHandle<()>>,
328    stderr_pump: Option<StderrPump>,
329    stderr_ring: Arc<Mutex<StderrRing>>,
330    spawned_at_ms: u64,
331    spawned_from: PathBuf,
332    spawned_file_identity: Option<SpawnedFileIdentity>,
333    process_start_time: Option<u64>,
334    process_identity: Option<ProcessIdentity>,
335    pid: u32,
336    /// This process's entry in the daemon's child roster, released when the
337    /// process is reaped or this handle is dropped.
338    roster_guard: Option<crate::child_roster::RosterGuard>,
339}
340
341impl SupervisedChild {
342    fn id(&self) -> Option<u32> {
343        Some(self.pid)
344    }
345
346    fn process_identity(&self) -> Option<ProcessIdentity> {
347        self.process_identity
348    }
349
350    async fn wait(&mut self) -> io::Result<ExitStatus> {
351        #[cfg(target_os = "macos")]
352        self.confirm_privacy_exec().await;
353        // The roster entry is NOT released here. A daemon shutdown waits for the
354        // roster to empty and then exits the process, so releasing at the reap
355        // let it exit before the exit handler wrote this child's terminal record
356        // (the stderr drain and snapshot update sit in between), and the
357        // shutdown's own `daemon_shutdown` record was intermittently lost. The
358        // caller releases it after recording the exit (`release_roster`), and
359        // dropping the handle releases it too.
360        let result = self.child.wait().await;
361        #[cfg(target_os = "linux")]
362        if result.is_ok() {
363            if let Some(placement) = self.cgroup_placement.as_ref() {
364                cleanup_reaped_cgroup(placement, &self.module_id).await;
365                // Keep ownership while awaiting kernel population changes: a
366                // drain timeout may cancel this wait and then escalate/reap.
367                self.cgroup_placement = None;
368            }
369        }
370        result
371    }
372
373    #[cfg(target_os = "macos")]
374    async fn confirm_privacy_exec(&mut self) {
375        let Some(pending) = &mut self.privacy_exec else {
376            return;
377        };
378        let result = tokio::time::timeout_at(pending.deadline, async {
379            let mut record = Vec::new();
380            loop {
381                let mut ready = pending.reader.readable().await?;
382                let read = ready.try_io(|reader| {
383                    use std::io::Read;
384                    let mut reader = reader.get_ref();
385                    let mut buffer = [0u8; 256];
386                    reader.read(&mut buffer).map(|count| (count, buffer))
387                });
388                match read {
389                    Ok(Ok((0, _))) => return Ok::<_, io::Error>(record),
390                    Ok(Ok((count, buffer))) => {
391                        if record.len() + count > 1024 {
392                            return Err(io::Error::other(
393                                "privacy exec refusal record is too long",
394                            ));
395                        }
396                        record.extend_from_slice(&buffer[..count]);
397                    }
398                    Ok(Err(error)) => return Err(error),
399                    Err(_) => continue,
400                }
401            }
402        })
403        .await;
404        // Keep the reader in self across await: select cancellation must not
405        // discard the handshake or reset its original five-second deadline.
406        let pending = self.privacy_exec.as_ref().expect("pending exec");
407        let cause = match result {
408            Err(_) => Some("privacy identity trampoline did not exec within 5s".to_string()),
409            Ok(Err(error)) => Some(format!(
410                "privacy identity exec acknowledgement failed: {error}"
411            )),
412            Ok(Ok(record)) if !record.is_empty() => Some(
413                std::str::from_utf8(&record)
414                    .ok()
415                    .and_then(|record| {
416                        record
417                            .trim()
418                            .strip_prefix(subc_os::privacy_identity::EXEC_REFUSAL_TAG)
419                    })
420                    .filter(|cause| !cause.is_empty())
421                    .unwrap_or("invalid privacy exec refusal record")
422                    .to_string(),
423            ),
424            Ok(Ok(_)) => match self.child.try_wait() {
425                // Empty EOF is the exec acknowledgement. A real module may exit
426                // immediately, including with a reserved trampoline status; no
427                // image is admitted, and its ordinary exit contract stays intact.
428                Ok(Some(_status)) => None,
429                Err(error) => Some(format!("privacy identity trampoline wait failed: {error}")),
430                Ok(None) => {
431                    let image = observe_spawned_image(self.pid);
432                    if let Some(image) = image.filter(|image| {
433                        image.executable.is_some()
434                            && image.executable != pending.trampoline
435                            && (image.executable == pending.expected || pending.script)
436                    }) {
437                        if let Some(guard) = &self.roster_guard {
438                            guard.confirm_executable(image);
439                        }
440                        let _ = self.report_ready.set(());
441                        info!(module_id = %pending.module_id, pid = self.pid,
442                            "module spawned with own privacy identity (responsibility disclaimed)");
443                        None
444                    } else if image.is_none()
445                        || image.is_some_and(|image| image.executable.is_none())
446                    {
447                        // A process can exit between try_wait and the kernel
448                        // image read. Empty EOF already acknowledged exec, so
449                        // preserve that module's ordinary exit rather than
450                        // mislabel a disappearing image as trampoline refusal.
451                        // Keep pending in self across await so cancellation does
452                        // not discard validation or reset its original deadline.
453                        match tokio::time::timeout_at(pending.deadline, self.child.wait()).await {
454                            Ok(Ok(_status)) => None,
455                            Ok(Err(error)) => Some(format!("privacy identity trampoline wait failed: {error}")),
456                            Err(_) => Some("privacy identity module executable remained unreadable until the 5s exec deadline".to_string()),
457                        }
458                    } else {
459                        Some("privacy identity trampoline executable mismatch: expected the module program, not ck-subc".to_string())
460                    }
461                }
462            },
463        };
464        let pending = self.privacy_exec.take().expect("pending exec");
465        if let Some(cause) = cause {
466            warn!(module_id = %pending.module_id, pid = self.pid, %cause, "privacy identity trampoline refused module spawn");
467            self.spawn_failure = Some(cause);
468            // No image is admitted on failure. Reach the entire fresh process
469            // group, including a module which spawned a helper before refusal.
470            if let Some(pid) = rustix::process::Pid::from_raw(self.pid as i32) {
471                let _ = rustix::process::kill_process_group(pid, rustix::process::Signal::KILL);
472            }
473            let _ = self.child.start_kill();
474        }
475    }
476
477    /// Releases this child's daemon-shutdown roster entry once its exit has
478    /// been recorded. The pid is already reaped and free for reuse, so the
479    /// entry must not outlive the record any longer than that.
480    fn release_roster(&mut self) {
481        self.roster_guard = None;
482    }
483
484    /// Kill the child and, where containment is available, its process tree.
485    ///
486    /// `Child::kill` is `TerminateProcess` scoped to one pid, so a module with a
487    /// helper process leaked the helper — the Synapse embedding module's CUDA
488    /// worker holds the GPU allocation, so the leak cost VRAM until the next
489    /// restart of something else. Terminating the job reaches grandchildren that
490    /// a tree walk cannot, including one whose parent has already exited and
491    /// been reparented away.
492    ///
493    /// Best-effort like `request_graceful_stop`: a job failure is logged and the
494    /// direct-child kill still decides the outcome, so containment can never
495    /// change whether a module is reported as stopped.
496    fn start_kill(&mut self) -> io::Result<()> {
497        #[cfg(windows)]
498        if let Some(job) = &self.job {
499            if let Err(error) = job.terminate() {
500                debug!(
501                    error = %error,
502                    "job termination failed; the direct-child kill still owns the outcome"
503                );
504            }
505        }
506        #[cfg(target_os = "linux")]
507        kill_module_cgroup(self.cgroup_placement.as_ref(), &self.module_id);
508        self.child.start_kill()
509    }
510
511    async fn drain_stderr(&mut self, module_id: &str) {
512        if let Some(mut pump) = self.stdout_pump.take() {
513            match timeout(STDERR_PUMP_DRAIN_TIMEOUT, &mut pump).await {
514                Ok(Ok(())) => {}
515                Ok(Err(error)) => {
516                    warn!(module_id, error = %error, "stdout pump ended unexpectedly");
517                }
518                Err(_) => {
519                    pump.abort();
520                    warn!(
521                        module_id,
522                        waited = ?STDERR_PUMP_DRAIN_TIMEOUT,
523                        "stdout pump did not drain before restart; stopped it before the next process"
524                    );
525                }
526            }
527        }
528
529        let Some(pump) = self.stderr_pump.take() else {
530            return;
531        };
532        settle_stderr_pump(
533            module_id,
534            &self.stderr_ring,
535            pump,
536            STDERR_PUMP_DRAIN_TIMEOUT,
537        )
538        .await;
539    }
540}
541
542/// The reader task for one process's stderr, with the ring generation its
543/// lines are attributed to.
544struct StderrPump {
545    task: JoinHandle<()>,
546    generation: u64,
547}
548
549/// Retire an exited process's stderr reader and wait up to `bound` for it to
550/// reach EOF. A reader still running at the bound is detached, not stopped: it
551/// keeps filling the exited process's section of the ring until its pipe
552/// closes, and the tail reads `Incomplete` until then. See
553/// [`STDERR_PUMP_DRAIN_TIMEOUT`] for why.
554async fn settle_stderr_pump(
555    module_id: &str,
556    ring: &Arc<Mutex<StderrRing>>,
557    pump: StderrPump,
558    bound: Duration,
559) {
560    let lock = || ring.lock().unwrap_or_else(|poisoned| poisoned.into_inner());
561    let StderrPump {
562        mut task,
563        generation,
564    } = pump;
565    lock().retire_pump(generation);
566    match timeout(bound, &mut task).await {
567        Ok(Ok(())) => {}
568        Ok(Err(err)) => {
569            let mut ring = lock();
570            ring.mark_incomplete_from(generation, format!("stderr pump ended unexpectedly: {err}"));
571            ring.finish_pump(generation);
572            warn!(module_id, error = %err, "stderr pump ended before clean EOF");
573        }
574        Err(_) => {
575            // Dropping the handle detaches the task; it ends at EOF on its pipe.
576            drop(task);
577            lock().mark_pump_late(
578                generation,
579                format!(
580                    "stderr of the exited process had not reached EOF {bound:?} after it was \
581                     retired (a descendant may still hold the pipe open); lines it still \
582                     writes are kept in that process's section"
583                ),
584            );
585            warn!(
586                module_id,
587                waited = ?bound,
588                "stderr pipe of the exited process is still open; its reader keeps running without delaying the restart"
589            );
590        }
591    }
592}
593
594fn registration_release_events() -> &'static watch::Sender<u64> {
595    static EVENTS: OnceLock<watch::Sender<u64>> = OnceLock::new();
596    EVENTS.get_or_init(|| {
597        let (sender, _receiver) = watch::channel(0);
598        sender
599    })
600}
601
602pub(crate) fn notify_registration_release() {
603    let events = registration_release_events();
604    let next_generation = (*events.borrow()).wrapping_add(1);
605    events.send_replace(next_generation);
606}
607
608/// How to launch one singleton module process.
609#[derive(Debug, Clone, PartialEq, Eq)]
610pub struct ModuleSpec {
611    pub module_id: String,
612    pub program: PathBuf,
613    pub args: Vec<String>,
614    pub env: Vec<(String, String)>,
615    /// When true this is a reserved module: each spawn gets a fresh one-time launch
616    /// nonce that the child must echo in its HELLO, so only the daemon-spawned
617    /// process can register this module_id (a security-boundary module like the
618    /// credential vault must not be impersonable while it is down/restarting).
619    pub reserved: bool,
620    /// Module-id prefixes this supervised module owns for reserved HELLO checks.
621    /// Prefixes come from daemon config and must end in `:` before they reach the
622    /// supervisor; the owner module's current spawn nonce authorizes claims under
623    /// each prefix.
624    pub reserved_prefixes: Vec<String>,
625    /// The wire protocol this module speaks, as DECLARED in daemon config.
626    ///
627    /// [`ModuleProtocol::None`] changes five things and nothing else: wire health
628    /// probing is suppressed (an optional HTTP probe can replace it), teardown sends SIGTERM before waiting,
629    /// `route.open` is refused, the spawn passes NO `--subc <path>` argument
630    /// and NO launch nonce, and a clean exit the daemon did not request is
631    /// restarted as a crash rather than recorded as a stop (see `on_child_exit`:
632    /// a stock program exits 0 on a stray SIGTERM, and a stop would leave it
633    /// down for good). `SUBC_MODULE_ID` still goes into the environment,
634    /// because a process ignores an environment variable it does not read.
635    ///
636    /// The argument is the part that cannot be "harmless to a process that
637    /// ignores it": a stock binary exits on an unknown flag before it listens
638    /// (`nats-server`: "flag provided but not defined: -subc"), which is how the
639    /// first conformance run against this mode found it. The nonce is withheld
640    /// because a process that will never present it gains nothing from holding
641    /// it, and a secret in the environment of a process that does not need it is
642    /// a leak surface for no benefit.
643    pub protocol: ModuleProtocol,
644    /// Whether two processes of this module may run at once, which is what a
645    /// blue/green swap does for the length of its overlap. Declared in daemon
646    /// config because the daemon must be able to answer it while the module is
647    /// down, and so a module cannot talk itself into it after registering.
648    pub overlap: ModuleOverlap,
649}
650
651/// Whether a module tolerates a second process of itself running alongside.
652///
653/// Most modules are single-writer on their store (a WAL, a capture log, a
654/// resident index behind a writer barrier), and two processes on one store
655/// corrupt it. So a swap, which overlaps the old and new process by design,
656/// is refused unless the module's config opts in with `overlap: "safe"`.
657#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
658pub enum ModuleOverlap {
659    /// Never run two processes of this module at once. The default.
660    #[default]
661    Exclusive,
662    /// The module has said a second process of itself is harmless for the
663    /// length of a swap.
664    ///
665    /// Declare it only if a second instance can run for a few seconds without
666    /// touching ANY single-writer store: every database, WAL, index, projector
667    /// and scheduled job the module owns. A lease on part of that state is not
668    /// enough. broca's session lease guards WAL appends while its run index, its
669    /// store projector and its archive fold timer (which unlinks live WAL files)
670    /// stay single-writer, so broca is exclusive despite holding a lease. The
671    /// refusal only fires after this has been decided, so the decision is the
672    /// check.
673    Safe,
674}
675
676impl ModuleOverlap {
677    pub fn as_str(self) -> &'static str {
678        match self {
679            Self::Exclusive => "exclusive",
680            Self::Safe => "safe",
681        }
682    }
683}
684
685/// Environment variable telling a spawned module which case it was started
686/// for, before it sends HELLO. Only a swap candidate carries it, as
687/// [`SPAWN_ROLE_SWAP_CANDIDATE`]; every other spawn has it removed.
688///
689/// It chooses a warm-up budget, nothing else: a swap candidate can warm for
690/// longer because nobody waits on it, while a plain restart must flip ready
691/// quickly because callers see `module_warming` until it does. Absence means
692/// plain restart, the safe reading. The daemon trusts nothing about it; the
693/// candidate is proven by its launch nonce at HELLO.
694pub const SUBC_SPAWN_ROLE_ENV: &str = "SUBC_SPAWN_ROLE";
695/// The one value of [`SUBC_SPAWN_ROLE_ENV`] the daemon sets.
696pub const SPAWN_ROLE_SWAP_CANDIDATE: &str = "swap_candidate";
697/// How long a swap waits for its candidate to register and declare itself
698/// ready when the operator does not say. A module warming as a swap candidate
699/// may take up to 90 s (aft's ceiling, the largest in the fleet), so the
700/// daemon allows that plus time to start the process and send HELLO.
701pub const DEFAULT_SWAP_READY_TIMEOUT: Duration = Duration::from_secs(100);
702
703/// Bounded restart policy for crash exits.
704///
705/// `max_restarts` is the number of replacement processes allowed after the
706/// initial spawn WITHIN `window`. After that many crash restarts inside one
707/// window the module enters [`ModuleState::Failed`] and the supervisor stops
708/// the crash loop.
709///
710/// The budget is a RATE, not a lifetime total. It used to be a lifetime total,
711/// and that only survived because crashes were rare: a module that crashed
712/// three times across a week was disabled forever by crashes that had nothing
713/// to do with each other. That stopped being survivable once modules began
714/// exiting non-zero whenever the daemon's connection to them drops, because
715/// then every daemon-side connection drop spends a unit of the same budget and
716/// one flappy hour permanently stops a healthy module. Restarts older than
717/// `window` release their slot, so a module that crashed twice yesterday has a
718/// full budget today, while a genuine crash loop -- which is fast by
719/// definition -- still reaches the cap and stops.
720#[derive(Debug, Clone, Copy, PartialEq, Eq)]
721pub struct RestartPolicy {
722    pub max_restarts: u32,
723    /// Base delay before a crash replacement. The actual delay escalates with
724    /// the number of recent crash replacements and is capped by `max_backoff`.
725    pub backoff: Duration,
726    /// Maximum delay before a crash replacement.
727    pub max_backoff: Duration,
728    /// The span `max_restarts` is counted over. `Duration::ZERO` makes the
729    /// budget effectively infinite (nothing is ever in-window), which is why
730    /// daemon config refuses `window_secs: 0` rather than quietly accepting it.
731    pub window: Duration,
732}
733
734impl RestartPolicy {
735    /// A policy with the default crash window. Callers that care about the
736    /// window say so with [`Self::with_window`]; the ones that do not are
737    /// asking for the standard rate limit, not for no limit.
738    pub fn new(max_restarts: u32, backoff: Duration) -> Self {
739        Self {
740            max_restarts,
741            backoff,
742            max_backoff: DEFAULT_MAX_BACKOFF,
743            window: DEFAULT_RESTART_WINDOW,
744        }
745    }
746
747    pub fn with_max_backoff(mut self, max_backoff: Duration) -> Self {
748        self.max_backoff = max_backoff;
749        self
750    }
751
752    pub fn with_window(mut self, window: Duration) -> Self {
753        self.window = window;
754        self
755    }
756
757    /// Calculate the capped exponential delay for the next crash replacement.
758    /// `restart_in_window` is zero for the first replacement after an operator
759    /// action (restart, reload, re-enable) cleared the crash ring, or after all
760    /// older crash replacements have aged out of the window.
761    fn delay_for_restart(&self, restart_in_window: u32) -> Duration {
762        if self.backoff.is_zero() || self.max_backoff.is_zero() {
763            return Duration::ZERO;
764        }
765
766        let mut delay = self.backoff;
767        for _ in 0..restart_in_window {
768            if delay >= self.max_backoff {
769                return self.max_backoff;
770            }
771            delay = delay
772                .checked_mul(10)
773                .unwrap_or(self.max_backoff)
774                .min(self.max_backoff);
775        }
776        delay.min(self.max_backoff)
777    }
778
779    /// The one sentence that explains a budget-exhausted stop, used for both the
780    /// log line and the terminal record so the two cannot drift. It names the
781    /// window because `max_restarts=3` alone reads as a lifetime cap, which is
782    /// exactly what this budget is not.
783    fn budget_exhausted_detail(&self) -> String {
784        format!(
785            "crash budget exhausted: max_restarts={} within window_secs={}",
786            self.max_restarts,
787            self.window.as_secs()
788        )
789    }
790}
791
792impl Default for RestartPolicy {
793    fn default() -> Self {
794        Self {
795            max_restarts: DEFAULT_MAX_RESTARTS,
796            backoff: DEFAULT_BACKOFF,
797            max_backoff: DEFAULT_MAX_BACKOFF,
798            window: DEFAULT_RESTART_WINDOW,
799        }
800    }
801}
802
803#[derive(Debug, Clone, Copy, PartialEq, Eq)]
804struct CrashRestartSchedule {
805    restart_in_window: u32,
806    delay: Duration,
807}
808
809/// Whether the daemon itself will bring this module back after the exit being
810/// handled: it is enabled AND its in-window crash restarts are below the cap.
811///
812/// Takes `&mut` because reading the budget prunes it. Instants that fell out of
813/// the window are dropped here rather than by a timer, so the count is right
814/// the moment somebody asks and no bookkeeping runs for idle modules.
815fn daemon_will_restart(
816    state: &mut SupervisorSnapshot,
817    policy: &RestartPolicy,
818    now: Instant,
819) -> bool {
820    state.enabled && state.crash_restarts_in_window(policy.window, now) < policy.max_restarts
821}
822
823const DEFAULT_HEALTH_CADENCE: Duration = Duration::from_secs(30);
824const DEFAULT_HEALTH_DEADLINE: Duration = Duration::from_secs(5);
825const DEFAULT_HEALTH_FAILURE_THRESHOLD: u32 = 3;
826const MAX_HEALTH_METRICS_BYTES: usize = 16 * 1024;
827
828#[derive(Debug, Clone, Copy, PartialEq, Eq)]
829pub enum HealthAction {
830    Report,
831    Restart,
832    Alert,
833}
834
835impl fmt::Display for HealthAction {
836    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
837        f.write_str(match self {
838            Self::Report => "report",
839            Self::Restart => "restart",
840            Self::Alert => "alert",
841        })
842    }
843}
844
845#[derive(Debug, Clone, PartialEq, Eq)]
846pub struct HealthConfig {
847    /// Optional loopback HTTP endpoint for a managed non-wire process.
848    /// Changing it applies live on rescan; the process protocol changes only
849    /// at its next spawn.
850    pub http: Option<String>,
851    pub cadence: Duration,
852    pub deadline: Duration,
853    pub failure_threshold: u32,
854    pub on_degraded: HealthAction,
855    pub on_failing: HealthAction,
856    pub critical: bool,
857}
858
859impl Default for HealthConfig {
860    fn default() -> Self {
861        Self {
862            http: None,
863            cadence: DEFAULT_HEALTH_CADENCE,
864            deadline: DEFAULT_HEALTH_DEADLINE,
865            failure_threshold: DEFAULT_HEALTH_FAILURE_THRESHOLD,
866            on_degraded: HealthAction::Report,
867            on_failing: HealthAction::Report,
868            critical: false,
869        }
870    }
871}
872
873/// The supervisor's view of one module's health, relayed to clients over
874/// channel-0 and rendered by `ck health`.
875///
876/// THIS TYPE IS WHERE THE ABSENCE MEANINGS ARE CREATED, which is why they are
877/// stated here rather than only at the wire type a consumer reads. A reader can
878/// look up what `None` means; only a writer can silently change it, and the
879/// writer has no reason to go looking at a downstream contract before editing.
880///
881/// `last_probe_ms: None` MEANS NEVER PROBED, not probed-long-ago. It is cleared
882/// back to `None` on re-registration precisely so a respawned module does not
883/// carry its predecessor's timestamp — so an old value and an absent one call for
884/// opposite readings, and anything that defaulted this to a number would make a
885/// never-probed module indistinguishable from one probed at the epoch.
886///
887/// `detail` and `metrics` are `None` when the module published none on this
888/// probe, which does not mean it reported nothing wrong — it is also the shape
889/// when the probe never reached it. `last_probe_ms` is what separates those.
890#[derive(Debug, Clone, PartialEq)]
891pub struct ModuleHealthStatus {
892    pub status: SupervisorHealthStatus,
893    pub last_probe_ms: Option<u64>,
894    pub detail: Option<String>,
895    pub metrics: Option<Value>,
896    pub consecutive_failures: u32,
897    /// Number of replies received after a recurring health probe's deadline.
898    /// Unlike a timeout, every increment proves the module was alive.
899    pub late_answer_count: u64,
900    /// End-to-end latency of the newest late reply, measured from probe start.
901    pub last_late_answer_latency_ms: Option<u64>,
902    pub last_action: Option<String>,
903    /// Set together with `last_action`; the pair moves as one, and both being
904    /// absent means no escalation has ever been taken rather than that the last
905    /// one succeeded.
906    pub last_action_ms: Option<u64>,
907}
908
909impl Default for ModuleHealthStatus {
910    fn default() -> Self {
911        Self {
912            status: SupervisorHealthStatus::Unknown,
913            last_probe_ms: None,
914            detail: None,
915            metrics: None,
916            consecutive_failures: 0,
917            late_answer_count: 0,
918            last_late_answer_latency_ms: None,
919            last_action: None,
920            last_action_ms: None,
921        }
922    }
923}
924
925/// Typed lifecycle state for a supervised module.
926#[derive(Debug, Clone, Copy, PartialEq, Eq)]
927pub enum ModuleState {
928    Starting,
929    Running,
930    Unresponsive,
931    Restarting,
932    Draining,
933    Stopped,
934    Failed,
935    Disabled,
936}
937
938impl fmt::Display for ModuleState {
939    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
940        f.write_str(match self {
941            Self::Starting => "starting",
942            Self::Running => "running",
943            Self::Unresponsive => "unresponsive",
944            Self::Restarting => "restarting",
945            Self::Draining => "draining",
946            Self::Stopped => "stopped",
947            Self::Failed => "failed",
948            Self::Disabled => "disabled",
949        })
950    }
951}
952
953/// Supervisor classification of a child-process exit.
954#[derive(Debug, Clone, Copy, PartialEq, Eq)]
955pub enum ExitKind {
956    Clean,
957    Crash,
958    DeliberateSeverance,
959}
960
961impl From<ExitKind> for TerminalExitKind {
962    fn from(kind: ExitKind) -> Self {
963        match kind {
964            ExitKind::Clean => Self::Clean,
965            ExitKind::Crash => Self::Crash,
966            ExitKind::DeliberateSeverance => Self::DeliberateSeverance,
967        }
968    }
969}
970
971/// Exact process identity retained when a supervised module registers its
972/// connection. PID reuse makes a PID alone insufficient evidence of ownership.
973#[derive(Debug, Clone, Copy, PartialEq, Eq)]
974pub(crate) struct ProcessIdentity {
975    pub(crate) pid: u32,
976    pub(crate) start_time: u64,
977}
978
979/// Last observed child exit, if any.
980#[derive(Debug, Clone, PartialEq, Eq)]
981pub struct ExitReport {
982    pub kind: ExitKind,
983    pub code: Option<i32>,
984    pub signal: Option<i32>,
985    pub at_ms: u64,
986}
987
988/// Point-in-time module status answerable by subc without forwarding to the
989/// module process.
990#[derive(Debug, Clone, PartialEq)]
991pub struct ModuleStatus {
992    pub module_id: String,
993    pub state: ModuleState,
994    pub enabled: bool,
995    pub process_alive: bool,
996    pub registration_active: bool,
997    /// The module's declared wire protocol, carried beside `live` because it is
998    /// what makes `live` readable: the two fields answer one question together.
999    /// While a process is alive this is its launch declaration, not a later
1000    /// pending-reload edit. When down it is the configured next launch protocol.
1001    pub protocol: ModuleProtocol,
1002    /// Whether the module is serving, under the strongest definition the daemon
1003    /// can assert for its protocol.
1004    ///
1005    /// A subc module must also be REGISTERED: its process being alive says
1006    /// nothing about whether it can take a request. A `protocol: "none"` module
1007    /// never registers, so that term is dropped and this falls back to "enabled,
1008    /// running, and the process the daemon launched is alive" -- which is all
1009    /// the daemon observes about a process that speaks no subc wire. It stays a
1010    /// `bool` on the wire for compatibility; renderers pair it with `protocol`
1011    /// rather than printing it bare.
1012    pub live: bool,
1013    /// Crash restarts spent INSIDE `restart_window` as of this read. Older
1014    /// restarts have already released their slot, so this count can go down
1015    /// without anybody touching the module.
1016    pub restart_count: u32,
1017    /// Replacement processes spawned over this module's entire supervisor lifetime;
1018    /// unlike `restart_count`, this value is never reset by an operator action
1019    /// and never falls out of a window.
1020    pub lifetime_restarts: u32,
1021    pub spawn_generation: u64,
1022    /// The budget `restart_count` is spent against. Carried alongside the count
1023    /// because the count alone does not say how close the module is to being
1024    /// disabled, and reporting one without the other is what makes an
1025    /// about-to-be-retired module look ordinary.
1026    pub max_restarts: u32,
1027    /// The span `restart_count` is counted over. Carried with the pair above for
1028    /// the same reason they are carried together: "2 of 3" means one thing for a
1029    /// ten-minute window and something else entirely for a lifetime.
1030    pub restart_window: Duration,
1031    /// Effective drain and restart timing policy used by this running module.
1032    /// These values are carried together with the restart budget so status
1033    /// readers can compare configured intent with what the supervisor applied.
1034    pub drain_timeout: Duration,
1035    pub restart_backoff: Duration,
1036    pub restart_max_backoff: Duration,
1037    /// The module's process. On macOS this stays absent while the `ck-subc`
1038    /// launch trampoline is still running in that pid, and appears once the
1039    /// exec acknowledgement confirms the module image has replaced it. Launch
1040    /// time and the supervisor's own hold on the process are unaffected.
1041    pub pid: Option<u32>,
1042    pub spawned_at_ms: Option<u64>,
1043    pub spawned_from: Option<PathBuf>,
1044    pub process_start_time: Option<u64>,
1045    pub last_exit: Option<ExitReport>,
1046    pub health: ModuleHealthStatus,
1047}
1048
1049#[cfg(test)]
1050#[derive(Debug, Clone, Copy, PartialEq)]
1051struct ActorSelectCheckpoint {
1052    generation: u64,
1053    turn: u64,
1054    registered_connection: Option<ConnectionId>,
1055    next_probe_at: Option<Instant>,
1056    wake_after: Option<Duration>,
1057}
1058
1059#[derive(Debug, Clone, PartialEq)]
1060struct SupervisorSnapshot {
1061    /// Counts running-child loop turns, not executor polls of a parked wait.
1062    #[cfg(test)]
1063    actor_turns: u64,
1064    /// Running select after exec confirmation and with no unconsumed registry
1065    /// event. Recording a spawn or starting a loop turn is not this barrier.
1066    #[cfg(test)]
1067    actor_select: Option<ActorSelectCheckpoint>,
1068    state: ModuleState,
1069    enabled: bool,
1070    process_alive: bool,
1071    spawned_protocol: Option<ModuleProtocol>,
1072    spawn_failure: Option<String>,
1073    /// When each crash restart was spent, oldest first. This IS the crash
1074    /// budget: its in-window length is the count an operator sees and the count
1075    /// the restart decision is made against, so there is no second counter that
1076    /// can disagree with it. Bounded by `max_restarts`, and cleared by the same
1077    /// operator actions that used to zero the old lifetime counter.
1078    crash_restarts: VecDeque<Instant>,
1079    lifetime_restarts: u32,
1080    /// Successful child spawns in this daemon incarnation.
1081    ///
1082    /// `lifetime_restarts` was considered and rejected: it starts at zero
1083    /// (line 640), successful initial/operator spawns in `set_running` do not
1084    /// increment it (lines 5264-5274), and crash/deliberate retry bookkeeping
1085    /// increments before a successful replacement exists (lines 604, 3846,
1086    /// and 3921), so a failed spawn can consume it. This counter moves only
1087    /// when a live PID is accepted below.
1088    spawn_generation: u64,
1089    pid: Option<u32>,
1090    #[cfg(target_os = "macos")]
1091    report_ready: Option<Arc<OnceLock<()>>>,
1092    /// Last reaped child, retained after current process facts are cleared.
1093    reaped_pid: Option<u32>,
1094    /// Whether the command-serving supervision loop has a scheduled respawn.
1095    respawn_pending: bool,
1096    /// A second restart is waiting for the replacement already scheduled.
1097    coalesced_restart_pending: bool,
1098    spawned_at_ms: Option<u64>,
1099    spawned_from: Option<PathBuf>,
1100    spawned_file_identity: Option<SpawnedFileIdentity>,
1101    process_start_time: Option<u64>,
1102    deliberate_severance: Option<ProcessIdentity>,
1103    last_exit: Option<ExitReport>,
1104    /// Diagnostic attached to the next drain's terminal record, if any.
1105    drain_disposition_detail: Option<String>,
1106    health: ModuleHealthStatus,
1107    /// Whether the current process was started as a swap candidate and so
1108    /// lives in the module's alternate cgroup. The next swap's candidate takes
1109    /// the other one, so the two processes of a swap never share a cgroup. A
1110    /// plain spawn always uses the primary cgroup.
1111    in_alternate_slot: bool,
1112    /// Whether the current `Draining` state ends in a replacement process
1113    /// (restart, reload, health restart) rather than a stop. Only meaningful
1114    /// while `state` is `Draining`; every entry into that state rewrites it.
1115    /// It is what lets route.open answer the retryable `module_reloading` to a
1116    /// consumer that reaches a still-registered process mid-restart, instead of
1117    /// the `supervisor_not_live` a stop or disable deserves.
1118    draining_to_replace: bool,
1119    /// Whether a configuration update has been applied since the current
1120    /// process was spawned, so that process runs an older spec than the one
1121    /// the supervisor now holds. A queued restart is only coalesced into a
1122    /// fresher process when this is false: a restart requested to pick up a
1123    /// new configuration must not be satisfied by a process that predates it.
1124    configuration_updated_since_spawn: bool,
1125}
1126
1127impl SupervisorSnapshot {
1128    /// The pid that status, provenance and resource readings may report. While
1129    /// the launch trampoline still runs in the pid, reading its executable or
1130    /// resource use would describe `ck-subc`, not the module, so none is
1131    /// reported until the exec acknowledgement confirms the module image.
1132    fn reported_pid(&self) -> Option<u32> {
1133        #[cfg(target_os = "macos")]
1134        if self
1135            .report_ready
1136            .as_ref()
1137            .is_some_and(|ready| ready.get().is_none())
1138        {
1139            return None;
1140        }
1141        self.pid
1142    }
1143
1144    fn starting() -> Self {
1145        Self::new(ModuleState::Starting, true)
1146    }
1147
1148    fn disabled() -> Self {
1149        Self::new(ModuleState::Disabled, false)
1150    }
1151
1152    fn failed() -> Self {
1153        Self::new(ModuleState::Failed, true)
1154    }
1155
1156    /// Crash restarts still inside `window`, having dropped the ones that are
1157    /// not. Pruning on read is what makes the budget a rate: an instant older
1158    /// than the window stops holding a slot the moment anybody counts.
1159    fn crash_restarts_in_window(&mut self, window: Duration, now: Instant) -> u32 {
1160        while let Some(oldest) = self.crash_restarts.front() {
1161            if now.duration_since(*oldest) > window {
1162                self.crash_restarts.pop_front();
1163            } else {
1164                break;
1165            }
1166        }
1167        u32::try_from(self.crash_restarts.len()).unwrap_or(u32::MAX)
1168    }
1169
1170    /// Spend one unit of the crash budget and record the restart in the ledger.
1171    ///
1172    /// The ring is bounded by the cap because more than `max_restarts` in-window
1173    /// instants can never be reached (the caller refuses the restart first), so
1174    /// anything beyond that is an unbounded queue waiting to happen.
1175    fn record_crash_restart(&mut self, policy: &RestartPolicy, now: Instant) {
1176        self.crash_restarts.push_back(now);
1177        while self.crash_restarts.len() > policy.max_restarts as usize {
1178            self.crash_restarts.pop_front();
1179        }
1180        self.lifetime_restarts += 1;
1181    }
1182
1183    /// Reserve one crash-restart slot and calculate the delay before respawning.
1184    /// The count is captured before recording this restart, so the first retry
1185    /// uses the base delay and each later in-window retry escalates once.
1186    fn next_crash_restart(
1187        &mut self,
1188        policy: &RestartPolicy,
1189        now: Instant,
1190    ) -> Option<CrashRestartSchedule> {
1191        let restart_in_window = self.crash_restarts_in_window(policy.window, now);
1192        if restart_in_window >= policy.max_restarts {
1193            return None;
1194        }
1195        self.record_crash_restart(policy, now);
1196        Some(CrashRestartSchedule {
1197            restart_in_window,
1198            delay: policy.delay_for_restart(restart_in_window),
1199        })
1200    }
1201
1202    /// Give the module its full budget back, as an operator restart, reload, or
1203    /// re-enable does. `lifetime_restarts` deliberately does not move: it is the
1204    /// ledger of what actually happened, and an operator action does not unmake
1205    /// the crashes.
1206    fn clear_crash_restarts(&mut self) {
1207        self.crash_restarts.clear();
1208    }
1209
1210    fn new(state: ModuleState, enabled: bool) -> Self {
1211        Self {
1212            #[cfg(test)]
1213            actor_turns: 0,
1214            #[cfg(test)]
1215            actor_select: None,
1216            state,
1217            enabled,
1218            process_alive: false,
1219            spawned_protocol: None,
1220            spawn_failure: None,
1221            crash_restarts: VecDeque::new(),
1222            lifetime_restarts: 0,
1223            spawn_generation: 0,
1224            pid: None,
1225            #[cfg(target_os = "macos")]
1226            report_ready: None,
1227            reaped_pid: None,
1228            respawn_pending: false,
1229            coalesced_restart_pending: false,
1230            spawned_at_ms: None,
1231            spawned_from: None,
1232            spawned_file_identity: None,
1233            process_start_time: None,
1234            deliberate_severance: None,
1235            last_exit: None,
1236            drain_disposition_detail: None,
1237            health: ModuleHealthStatus::default(),
1238            in_alternate_slot: false,
1239            draining_to_replace: false,
1240            configuration_updated_since_spawn: false,
1241        }
1242    }
1243}
1244
1245type SharedSnapshot = Arc<Mutex<SupervisorSnapshot>>;
1246
1247type SpawnSubscriberKey = (ConnectionId, u64);
1248
1249#[derive(Debug)]
1250struct SpawnSubscriber {
1251    version: u8,
1252    frames: mpsc::Sender<Frame>,
1253    /// Tells this subscriber's forwarder that it was dropped for lagging, and
1254    /// from which event. The full frame channel cannot carry that news, so it
1255    /// travels beside it; see `SpawnEventFeed::subscribe`.
1256    lagged: Option<oneshot::Sender<SpawnCursor>>,
1257}
1258
1259#[derive(Debug)]
1260struct SpawnEventState {
1261    daemon_incarnation: String,
1262    seq: u64,
1263    capacity: usize,
1264    live: HashMap<String, LiveSpawn>,
1265    generations: HashMap<String, u64>,
1266    events: VecDeque<SpawnEvent>,
1267    subscribers: HashMap<SpawnSubscriberKey, SpawnSubscriber>,
1268}
1269
1270impl Default for SpawnEventState {
1271    fn default() -> Self {
1272        Self {
1273            daemon_incarnation: "unconfigured".to_string(),
1274            seq: 0,
1275            capacity: SPAWN_EVENT_RING_CAPACITY,
1276            live: HashMap::new(),
1277            generations: HashMap::new(),
1278            events: VecDeque::new(),
1279            subscribers: HashMap::new(),
1280        }
1281    }
1282}
1283
1284#[derive(Debug, Clone, Default)]
1285struct SpawnEventFeed(Arc<Mutex<SpawnEventState>>);
1286
1287#[derive(Debug, Clone, PartialEq, Eq)]
1288pub(crate) enum SpawnSubscribeRefusal {
1289    ForeignIncarnation { current: String },
1290    TooOld { oldest: SpawnCursor },
1291    Frame(String),
1292}
1293
1294impl SpawnEventFeed {
1295    fn configure_incarnation(&self, daemon_incarnation: String) {
1296        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1297        state.daemon_incarnation = daemon_incarnation;
1298        state.seq = 0;
1299        state.live.clear();
1300        state.generations.clear();
1301        state.events.clear();
1302        state.subscribers.clear();
1303    }
1304
1305    fn cursor(state: &SpawnEventState) -> SpawnCursor {
1306        SpawnCursor {
1307            daemon_incarnation: state.daemon_incarnation.clone(),
1308            seq: state.seq,
1309        }
1310    }
1311
1312    fn snapshot(&self) -> SpawnSnapshot {
1313        let state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1314        let mut live = state.live.values().cloned().collect::<Vec<_>>();
1315        live.sort_by(|left, right| left.module_id.cmp(&right.module_id));
1316        SpawnSnapshot {
1317            cursor: Self::cursor(&state),
1318            ring_bound: state.capacity as u64,
1319            live,
1320        }
1321    }
1322
1323    fn emit_spawned(&self, module_id: &str, pid: u32, spawned_at_ms: u64) -> u64 {
1324        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1325        let generation = state
1326            .generations
1327            .get(module_id)
1328            .copied()
1329            .unwrap_or(0)
1330            .checked_add(1)
1331            .expect("spawn generation exhausted");
1332        state.generations.insert(module_id.to_string(), generation);
1333        let live = LiveSpawn {
1334            module_id: module_id.to_string(),
1335            spawn_generation: generation,
1336            pid,
1337            spawned_at_ms,
1338        };
1339        state.live.insert(module_id.to_string(), live);
1340        Self::emit_locked(
1341            &mut state,
1342            SpawnEventKind::Spawned,
1343            module_id.to_string(),
1344            generation,
1345            pid,
1346            None,
1347            None,
1348        );
1349        generation
1350    }
1351
1352    fn emit_exited(&self, module_id: &str, exit_code: Option<i32>, exit_signal: Option<i32>) {
1353        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1354        let Some(live) = state.live.remove(module_id) else {
1355            warn!(
1356                module_id,
1357                "terminal record had no live spawn event identity"
1358            );
1359            return;
1360        };
1361        Self::emit_locked(
1362            &mut state,
1363            SpawnEventKind::Exited,
1364            module_id.to_string(),
1365            live.spawn_generation,
1366            live.pid,
1367            exit_code,
1368            exit_signal,
1369        );
1370    }
1371
1372    /// Report the exit of a process that a swap has already replaced.
1373    ///
1374    /// `emit_exited` removes the module's live entry, which after a swap's
1375    /// cutover describes the promoted candidate, not the old process now
1376    /// exiting. This emits the old generation's exit and leaves the live entry
1377    /// alone unless it still names that generation.
1378    fn emit_superseded_exited(
1379        &self,
1380        module_id: &str,
1381        spawn_generation: u64,
1382        pid: u32,
1383        exit_code: Option<i32>,
1384        exit_signal: Option<i32>,
1385    ) {
1386        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1387        if state
1388            .live
1389            .get(module_id)
1390            .is_some_and(|live| live.spawn_generation == spawn_generation)
1391        {
1392            state.live.remove(module_id);
1393        }
1394        Self::emit_locked(
1395            &mut state,
1396            SpawnEventKind::Exited,
1397            module_id.to_string(),
1398            spawn_generation,
1399            pid,
1400            exit_code,
1401            exit_signal,
1402        );
1403    }
1404
1405    #[allow(clippy::too_many_arguments)]
1406    fn emit_locked(
1407        state: &mut SpawnEventState,
1408        kind: SpawnEventKind,
1409        module_id: String,
1410        spawn_generation: u64,
1411        pid: u32,
1412        exit_code: Option<i32>,
1413        exit_signal: Option<i32>,
1414    ) {
1415        state.seq = state
1416            .seq
1417            .checked_add(1)
1418            .expect("spawn event sequence exhausted");
1419        let event = SpawnEvent {
1420            cursor: Self::cursor(state),
1421            kind,
1422            module_id,
1423            spawn_generation,
1424            pid,
1425            exit_code,
1426            exit_signal,
1427        };
1428        state.events.push_back(event.clone());
1429        while state.events.len() > state.capacity {
1430            state.events.pop_front();
1431        }
1432        let body = match serde_json::to_vec(&event) {
1433            Ok(body) => body,
1434            Err(error) => {
1435                error!(%error, "failed to serialize supervisor spawn event");
1436                return;
1437            }
1438        };
1439        state.subscribers.retain(|(connection_id, corr), subscriber| {
1440            let frame = Frame::build_with_version(
1441                subscriber.version,
1442                FrameType::StreamData,
1443                control_flags(),
1444                0,
1445                0,
1446                *corr,
1447                body.clone(),
1448            );
1449            match frame {
1450                Ok(frame) => {
1451                    if subscriber.frames.try_send(frame).is_ok() {
1452                        true
1453                    } else {
1454                        warn!(connection_id = connection_id.get(), corr, "dropping lagged supervisor spawn subscriber");
1455                        if let Some(lagged) = subscriber.lagged.take() {
1456                            let _ = lagged.send(event.cursor.clone());
1457                        }
1458                        false
1459                    }
1460                }
1461                Err(error) => {
1462                    warn!(connection_id = connection_id.get(), corr, %error, "dropping supervisor spawn subscriber after frame build failure");
1463                    false
1464                }
1465            }
1466        });
1467    }
1468
1469    fn subscribe(
1470        &self,
1471        connection_id: ConnectionId,
1472        corr: u64,
1473        version: u8,
1474        since: Option<SpawnCursor>,
1475        sink: FrameSink,
1476    ) -> Result<(), SpawnSubscribeRefusal> {
1477        let (frames, mut receiver) = mpsc::channel(SPAWN_SUBSCRIBER_BUFFER);
1478        let (lagged, mut lagged_rx) = oneshot::channel::<SpawnCursor>();
1479        {
1480            let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1481            let replay = if let Some(since) = since {
1482                if since.daemon_incarnation != state.daemon_incarnation {
1483                    return Err(SpawnSubscribeRefusal::ForeignIncarnation {
1484                        current: state.daemon_incarnation.clone(),
1485                    });
1486                }
1487                if let Some(oldest) = state.events.front().map(|event| event.cursor.clone()) {
1488                    if since.seq < oldest.seq.saturating_sub(1) {
1489                        return Err(SpawnSubscribeRefusal::TooOld { oldest });
1490                    }
1491                }
1492                state
1493                    .events
1494                    .iter()
1495                    .filter(|event| event.cursor.seq > since.seq)
1496                    .cloned()
1497                    .collect::<Vec<_>>()
1498            } else {
1499                Vec::new()
1500            };
1501            for event in replay {
1502                let body = serde_json::to_vec(&event)
1503                    .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1504                let frame = Frame::build_with_version(
1505                    version,
1506                    FrameType::StreamData,
1507                    control_flags(),
1508                    0,
1509                    0,
1510                    corr,
1511                    body,
1512                )
1513                .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1514                frames
1515                    .try_send(frame)
1516                    .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1517            }
1518            state.subscribers.insert(
1519                (connection_id, corr),
1520                SpawnSubscriber {
1521                    version,
1522                    frames,
1523                    lagged: Some(lagged),
1524                },
1525            );
1526        }
1527        // The lagged terminal is sent here, by the forwarder, rather than by
1528        // the emitter: at the moment of the drop the subscriber's own channel
1529        // is full, and writing to the connection sink directly from the emitter
1530        // would put the Error AHEAD of the events still queued in that channel
1531        // (and the emitter holds the feed lock, so it cannot await the sink).
1532        // Dropping the subscriber drops the only sender, so `recv` drains every
1533        // queued event and then returns `None`; only then is the Error sent, so
1534        // the client sees each event it can keep, then the reason it was cut.
1535        // Cancel and connection removal drop the oneshot unsent, so they end
1536        // the stream with no Error.
1537        tokio::spawn(async move {
1538            while let Some(frame) = receiver.recv().await {
1539                if sink.send(frame).await.is_err() {
1540                    return;
1541                }
1542            }
1543            let Ok(first_undelivered) = lagged_rx.try_recv() else {
1544                return;
1545            };
1546            match spawn_subscriber_lagged_frame(version, corr, first_undelivered) {
1547                Ok(frame) => {
1548                    let _ = sink.send(frame).await;
1549                }
1550                Err(error) => {
1551                    error!(%error, corr, "failed to build lagged spawn subscriber terminal frame");
1552                }
1553            }
1554        });
1555        Ok(())
1556    }
1557
1558    fn cancel(&self, connection_id: ConnectionId, corr: u64) -> bool {
1559        let Some(subscriber) = self
1560            .0
1561            .lock()
1562            .unwrap_or_else(|p| p.into_inner())
1563            .subscribers
1564            .remove(&(connection_id, corr))
1565        else {
1566            return false;
1567        };
1568        if let Ok(frame) = Frame::build_with_version(
1569            subscriber.version,
1570            FrameType::StreamEnd,
1571            control_flags(),
1572            0,
1573            0,
1574            corr,
1575            Vec::new(),
1576        ) {
1577            tokio::spawn(async move {
1578                let _ = subscriber.frames.send(frame).await;
1579            });
1580        }
1581        true
1582    }
1583
1584    fn remove_connection(&self, connection_id: ConnectionId) {
1585        self.0
1586            .lock()
1587            .unwrap_or_else(|p| p.into_inner())
1588            .subscribers
1589            .retain(|(subscriber_connection, _), _| *subscriber_connection != connection_id);
1590    }
1591
1592    #[cfg(any(test, feature = "test-support"))]
1593    fn set_capacity(&self, capacity: usize) {
1594        self.0.lock().unwrap_or_else(|p| p.into_inner()).capacity = capacity;
1595    }
1596
1597    #[cfg(any(test, feature = "test-support"))]
1598    fn subscriber_count(&self) -> usize {
1599        self.0
1600            .lock()
1601            .unwrap_or_else(|p| p.into_inner())
1602            .subscribers
1603            .len()
1604    }
1605}
1606
1607/// Narrow process-liveness signal published by supervisors and consumed by passive liveness polls.
1608/// The terminal Error a lagged spawn subscriber receives after its queued events.
1609fn spawn_subscriber_lagged_frame(
1610    version: u8,
1611    corr: u64,
1612    first_undelivered: SpawnCursor,
1613) -> Result<Frame, String> {
1614    let body = serde_json::to_vec(&subc_protocol::ErrorBody {
1615        code: SPAWN_SUBSCRIBER_LAGGED_CODE.to_string(),
1616        message: "spawn subscriber fell behind and was dropped; resubscribe from the last cursor received"
1617            .to_string(),
1618        detail: Some(serde_json::json!({
1619            "first_undelivered_cursor": first_undelivered
1620        })),
1621    })
1622    .map_err(|error| error.to_string())?;
1623    Frame::build_with_version(version, FrameType::Error, control_flags(), 0, 0, corr, body)
1624        .map_err(|error| error.to_string())
1625}
1626
1627pub trait ModuleProcessLiveness: Send + Sync {
1628    fn process_live(&self, module_id: &str) -> Option<bool>;
1629
1630    /// Whether the supervisor is replacing this module's process right now: an
1631    /// operator restart or reload, a health restart, or a crash respawn whose
1632    /// backoff is running. A module in that state is not live, but a consumer
1633    /// refused now should retry shortly rather than treat the target as gone.
1634    /// Stopped, failed, and disabled modules are not replacing.
1635    fn process_replacing(&self, _module_id: &str) -> bool {
1636        false
1637    }
1638}
1639
1640/// Shared process-liveness registry keyed by supervised `module_id`.
1641#[derive(Debug, Clone, Default)]
1642pub struct SupervisorProcessLiveness {
1643    snapshots: Arc<Mutex<HashMap<String, SharedSnapshot>>>,
1644}
1645
1646impl SupervisorProcessLiveness {
1647    pub fn new() -> Self {
1648        Self::default()
1649    }
1650
1651    fn track(&self, module_id: String, snapshot: SharedSnapshot) {
1652        let mut snapshots = self
1653            .snapshots
1654            .lock()
1655            .unwrap_or_else(|poisoned| poisoned.into_inner());
1656        snapshots.insert(module_id, snapshot);
1657    }
1658
1659    fn untrack_if_current(&self, module_id: &str, snapshot: &SharedSnapshot) {
1660        let mut snapshots = self
1661            .snapshots
1662            .lock()
1663            .unwrap_or_else(|poisoned| poisoned.into_inner());
1664        let is_current = snapshots
1665            .get(module_id)
1666            .map(|tracked| Arc::ptr_eq(tracked, snapshot))
1667            .unwrap_or(false);
1668        if is_current {
1669            snapshots.remove(module_id);
1670        }
1671    }
1672}
1673
1674impl ModuleProcessLiveness for SupervisorProcessLiveness {
1675    fn process_live(&self, module_id: &str) -> Option<bool> {
1676        let snapshot = {
1677            let snapshots = self
1678                .snapshots
1679                .lock()
1680                .unwrap_or_else(|poisoned| poisoned.into_inner());
1681            snapshots.get(module_id).cloned()
1682        }?;
1683        let snapshot = snapshot
1684            .lock()
1685            .unwrap_or_else(|poisoned| poisoned.into_inner());
1686        Some(snapshot.state == ModuleState::Running && snapshot.process_alive)
1687    }
1688
1689    fn process_replacing(&self, module_id: &str) -> bool {
1690        let Some(snapshot) = self
1691            .snapshots
1692            .lock()
1693            .unwrap_or_else(|poisoned| poisoned.into_inner())
1694            .get(module_id)
1695            .cloned()
1696        else {
1697            return false;
1698        };
1699        let snapshot = snapshot
1700            .lock()
1701            .unwrap_or_else(|poisoned| poisoned.into_inner());
1702        snapshot.enabled
1703            && match snapshot.state {
1704                ModuleState::Restarting => true,
1705                ModuleState::Draining => snapshot.draining_to_replace,
1706                ModuleState::Starting
1707                | ModuleState::Running
1708                | ModuleState::Unresponsive
1709                | ModuleState::Stopped
1710                | ModuleState::Failed
1711                | ModuleState::Disabled => false,
1712            }
1713    }
1714}
1715
1716#[cfg(test)]
1717#[derive(Debug, Default)]
1718struct ReloadExitRecordGate {
1719    reached: tokio::sync::Notify,
1720    resume: tokio::sync::Notify,
1721}
1722
1723#[derive(Debug, Clone, Copy)]
1724enum RespawnKind {
1725    Spawn,
1726    Reload,
1727}
1728
1729#[derive(Debug, Clone, Copy)]
1730struct PendingRespawn {
1731    deadline: Instant,
1732    kind: RespawnKind,
1733}
1734
1735type ReloadReply = oneshot::Sender<Result<(), SuperviseError>>;
1736
1737#[derive(Debug, Clone)]
1738struct SupervisorRuntimeConfig {
1739    /// Restart operations hand their backoff to the loop, which keeps serving commands.
1740    scheduled_respawn: Arc<Mutex<Option<PendingRespawn>>>,
1741    /// A reload acknowledges completion only after its replacement registers.
1742    deferred_reload_reply: Arc<Mutex<Option<ReloadReply>>>,
1743    restart_policy: RestartPolicy,
1744    /// This module's RESOLVED drain budget: per-module config when present,
1745    /// else `default_drain_timeout`.
1746    drain_timeout: Duration,
1747    /// Shared with the status handle so the attested value changes atomically
1748    /// when a rescan updates the running drain policy.
1749    effective_drain_timeout: Arc<Mutex<Duration>>,
1750    /// The supervisor-wide fallback, kept so a configuration update that
1751    /// REMOVES the per-module override can re-resolve to it.
1752    default_drain_timeout: Duration,
1753    health: HealthConfig,
1754    connection_file_path: Option<PathBuf>,
1755    capture_logs_dir: Option<PathBuf>,
1756    forwarding: Option<Arc<ForwardingTable>>,
1757    /// The shared handle, so every spawn path (initial, restart, reload) records the
1758    /// reserved-module launch nonce the HELLO verifier checks against.
1759    supervisor_handle: Option<SupervisorHandle>,
1760    /// This module's stderr tail, shared with the [`SupervisedModule`] that answers
1761    /// status queries.
1762    ///
1763    /// One ring per module, held across every respawn. The lines explaining an exit
1764    /// are written BEFORE that exit, so a ring recreated per process would be empty
1765    /// exactly when it is asked for.
1766    stderr_ring: Arc<Mutex<StderrRing>>,
1767    terminal_ring: Arc<Mutex<TerminalRing>>,
1768    spawn_events: SpawnEventFeed,
1769    child_roster: ChildRoster,
1770    #[cfg(target_os = "linux")]
1771    cgroup_placement: Option<subc_cgroup::Placement>,
1772    #[cfg(test)]
1773    test_seed_stale_facts_before_enable_spawn: bool,
1774    #[cfg(test)]
1775    test_reload_exit_record_gate: Option<Arc<ReloadExitRecordGate>>,
1776}
1777
1778#[derive(Debug, Clone, PartialEq, Eq)]
1779struct SupervisedConfiguration {
1780    spec: ModuleSpec,
1781    health: HealthConfig,
1782}
1783
1784/// Shared daemon lookup table for supervised module handles.
1785///
1786/// Shared by clone between the [`Supervisor`] (which spawns processes) and the
1787/// channel-0 control handler (which verifies HELLOs and consumer route opens), so
1788/// launch nonces recorded at spawn are checked by the same daemon instance.
1789#[derive(Debug, Clone, Default)]
1790pub struct SupervisorHandle {
1791    modules: Arc<Mutex<HashMap<String, SupervisedModule>>>,
1792    /// Module ids the supervisor has taken on. An id is added BEFORE the
1793    /// module's first process is spawned and removed only when the module
1794    /// leaves the roster (see [`Self::retire`]), so it is always a superset of
1795    /// the keys of `modules`.
1796    ///
1797    /// `modules` cannot answer "is this module configured?" on its own: a
1798    /// [`SupervisedModule`] only exists once its process has been spawned, and
1799    /// a fast child can connect, register, sync its scopes and ask about them
1800    /// before the supervisor has inserted it. Answering "not configured" in that
1801    /// gap makes scope admission refuse with the terminal "will never sync"
1802    /// instead of the retryable "has not synced yet".
1803    configured_ids: Arc<Mutex<HashSet<String>>>,
1804    spawn_events: SpawnEventFeed,
1805    /// The current expected launch nonce for each reserved module_id. Set when the
1806    /// supervisor spawns the reserved module; checked when a HELLO claims that id. A
1807    /// non-reserved module never has an entry here and is never nonce-checked.
1808    /// Reserved module ids and the nonce that authorizes their next HELLO.
1809    /// `None` means RESERVED WITH NO LEGITIMATE HOLDER — a reserved module that
1810    /// has never been spawned (e.g. configured `enabled: false`) — and refuses
1811    /// every HELLO. Before this was expressible, a reserved-but-never-spawned id
1812    /// had NO entry and admitted anyone: the reservation protected the nonce
1813    /// holder, not the NAME (found live by CKCRED's canary probe registering
1814    /// against a reserved scratch id).
1815    reserved_nonces: Arc<Mutex<HashMap<String, Option<String>>>>,
1816    /// Module ids removed by an executed rescan and the unix-millisecond removal time.
1817    ///
1818    /// This is deliberately in-memory only: subc is state-free across daemon
1819    /// restarts, and the tombstone only explains the hours-after-removal window
1820    /// while this executing daemon is still alive. Do not persist it in a store.
1821    removal_tombstones: Arc<Mutex<HashMap<String, u64>>>,
1822    /// The current launch nonce for every supervised spawn. This is separate from
1823    /// reserved_nonces because consumer route.open attestation applies to all spawned
1824    /// modules, while HELLO id-squatting protection remains opt-in via `reserved`.
1825    spawn_nonces: Arc<Mutex<HashMap<String, String>>>,
1826    /// Reserved namespace prefixes mapped to the supervised owner module whose
1827    /// current spawn nonce authorizes HELLO claims below the prefix.
1828    ///
1829    /// Per §2.6 this is not a same-user security barrier: a same-user process can
1830    /// read the key file and launch nonce env. Like exact reserved ids, it prevents
1831    /// accidental collisions and lower-trust processes from squatting protected
1832    /// namespaces.
1833    reserved_prefix_owners: Arc<Mutex<HashMap<String, String>>>,
1834    /// Blue/green swaps in progress, by module id. An entry exists from just
1835    /// before the candidate process is spawned until the swap has failed, or
1836    /// has cut over and the old process is gone. While it exists, HELLO for the
1837    /// id is gated on the swap token (see [`Self::swap_hello_admission`]) and
1838    /// consumer attestation accepts both processes' nonces.
1839    swaps: Arc<Mutex<HashMap<String, OpenSwap>>>,
1840    /// Told when a swap promotes its candidate; see [`SwapPromotionObserver`].
1841    promotion_observer: PromotionObserverSlot,
1842    /// Serializes module-set reconciliation with operator lifecycle commands. Without
1843    /// this daemon-wide ordering, a rescan could retire or update a module while a
1844    /// concurrent reload still held its old handle and launch specification.
1845    operation_lock: Arc<AsyncMutex<()>>,
1846}
1847
1848/// Told when a swap has promoted its candidate to be the module's active
1849/// registration.
1850///
1851/// An ordinary HELLO runs the control plane's registration side effects (the
1852/// capability cache, the deny census, the requirement recompute) as it
1853/// registers. A swap candidate's HELLO does not, because it is not routable;
1854/// promotion is when those must run instead, and promotion happens in the
1855/// supervisor, which has no other way into the control handler.
1856pub(crate) trait SwapPromotionObserver: Send + Sync {
1857    fn swap_promoted(&self, registration: &crate::registry::ModuleRegistration);
1858}
1859
1860/// The installed [`SwapPromotionObserver`], held weakly: the observer (the
1861/// control handler) owns this handle, so a strong reference back would be a
1862/// cycle that keeps both alive.
1863#[derive(Clone, Default)]
1864struct PromotionObserverSlot(Arc<Mutex<Option<std::sync::Weak<dyn SwapPromotionObserver>>>>);
1865
1866impl fmt::Debug for PromotionObserverSlot {
1867    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1868        f.write_str("PromotionObserverSlot")
1869    }
1870}
1871
1872/// The nonces of one open swap.
1873#[derive(Debug, Clone)]
1874struct OpenSwap {
1875    /// The launch nonce minted for the candidate process. It is the swap
1876    /// token: the only thing that admits a HELLO into the candidate slot.
1877    candidate_nonce: String,
1878    /// The incumbent's launch nonce, captured when the swap opened. It is kept
1879    /// here because cutover moves the module's recorded spawn nonce to the
1880    /// candidate while the incumbent is still draining and its consumers are
1881    /// still attesting with this one.
1882    incumbent_nonce: Option<String>,
1883    /// Set once a HELLO has been admitted with the swap token, so the token
1884    /// admits one registration and cannot be replayed after cutover empties
1885    /// the candidate slot.
1886    candidate_admitted: bool,
1887}
1888
1889/// What the swap gate says about a HELLO. See
1890/// [`SupervisorHandle::swap_hello_admission`].
1891#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1892pub(crate) enum SwapHelloAdmission {
1893    /// No swap is open for the id (or the HELLO carries the incumbent's own
1894    /// nonce); the ordinary gates decide.
1895    NotSwapping,
1896    /// The HELLO carries the swap token: register it into the candidate slot.
1897    Candidate,
1898    /// A swap is open and the HELLO carries a nonce the supervisor did not
1899    /// mint for this id, no nonce, or a token already used.
1900    Refused,
1901}
1902
1903#[derive(Debug, Clone, PartialEq, Eq)]
1904pub(crate) enum ReservedHelloRejection {
1905    Exact {
1906        module_id: String,
1907    },
1908    Prefix {
1909        prefix: String,
1910        owner_module_id: String,
1911    },
1912}
1913
1914impl SupervisorHandle {
1915    pub fn new() -> Self {
1916        Self::default()
1917    }
1918
1919    pub(crate) fn spawn_snapshot(&self) -> SpawnSnapshot {
1920        self.spawn_events.snapshot()
1921    }
1922
1923    pub(crate) fn subscribe_spawns(
1924        &self,
1925        connection_id: ConnectionId,
1926        corr: u64,
1927        version: u8,
1928        since: Option<SpawnCursor>,
1929        sink: FrameSink,
1930    ) -> Result<(), SpawnSubscribeRefusal> {
1931        self.spawn_events
1932            .subscribe(connection_id, corr, version, since, sink)
1933    }
1934
1935    pub(crate) fn cancel_spawn_subscription(&self, connection_id: ConnectionId, corr: u64) -> bool {
1936        self.spawn_events.cancel(connection_id, corr)
1937    }
1938
1939    pub(crate) fn remove_spawn_subscribers(&self, connection_id: ConnectionId) {
1940        self.spawn_events.remove_connection(connection_id);
1941    }
1942
1943    #[cfg(any(test, feature = "test-support"))]
1944    pub fn set_spawn_event_capacity_for_test(&self, capacity: usize) {
1945        assert!(capacity > 0, "spawn event capacity must be non-zero");
1946        self.spawn_events.set_capacity(capacity);
1947    }
1948
1949    #[cfg(any(test, feature = "test-support"))]
1950    pub fn spawn_subscriber_count_for_test(&self) -> usize {
1951        self.spawn_events.subscriber_count()
1952    }
1953
1954    /// Record the launch nonce from a supervised spawn, replacing any prior nonce so
1955    /// a respawn invalidates stale consumer identities.
1956    pub fn set_spawn_nonce(&self, module_id: &str, nonce: String) {
1957        self.spawn_nonces
1958            .lock()
1959            .unwrap_or_else(|poisoned| poisoned.into_inner())
1960            .insert(module_id.to_string(), nonce);
1961    }
1962
1963    /// Record the launch nonce expected from the next HELLO for a reserved module,
1964    /// replacing any prior nonce (a respawn invalidates the previous one).
1965    pub fn set_reserved_nonce(&self, module_id: &str, nonce: String) {
1966        self.reserved_nonces
1967            .lock()
1968            .unwrap_or_else(|poisoned| poisoned.into_inner())
1969            .insert(module_id.to_string(), Some(nonce));
1970    }
1971
1972    /// Record namespace prefixes owned by a supervised module.
1973    pub fn set_reserved_prefixes(&self, owner_module_id: &str, prefixes: &[String]) {
1974        let mut owners = self
1975            .reserved_prefix_owners
1976            .lock()
1977            .unwrap_or_else(|poisoned| poisoned.into_inner());
1978        owners.retain(|_, owner| owner != owner_module_id);
1979        for prefix in prefixes {
1980            owners.insert(prefix.clone(), owner_module_id.to_string());
1981        }
1982    }
1983
1984    /// The launch nonce most recently minted for a module's spawn, if any.
1985    #[cfg(test)]
1986    pub(crate) fn spawn_nonce(&self, module_id: &str) -> Option<String> {
1987        self.spawn_nonces
1988            .lock()
1989            .unwrap_or_else(|poisoned| poisoned.into_inner())
1990            .get(module_id)
1991            .cloned()
1992    }
1993
1994    fn apply_identity_configuration(&self, spec: &ModuleSpec) {
1995        self.set_reserved_prefixes(&spec.module_id, &spec.reserved_prefixes);
1996        let spawn_nonce = self
1997            .spawn_nonces
1998            .lock()
1999            .unwrap_or_else(|poisoned| poisoned.into_inner())
2000            .get(&spec.module_id)
2001            .cloned();
2002        let mut reserved_nonces = self
2003            .reserved_nonces
2004            .lock()
2005            .unwrap_or_else(|poisoned| poisoned.into_inner());
2006        if spec.reserved {
2007            // `None` (no spawn nonce minted) is INSERTED, not skipped: a
2008            // reserved name whose module has never spawned has no legitimate
2009            // holder, and the entry's absence is what used to leave the name
2010            // open to the first claimant.
2011            reserved_nonces.insert(spec.module_id.clone(), spawn_nonce);
2012        }
2013        drop(reserved_nonces);
2014        // A later unreserved declaration must not silently unreserve an id that
2015        // was retained after its reserved configuration was removed. The explicit
2016        // release ceremony is the only operation that retires that gate.
2017        self.removal_tombstones
2018            .lock()
2019            .unwrap_or_else(|poisoned| poisoned.into_inner())
2020            .remove(&spec.module_id);
2021    }
2022
2023    /// Whether a HELLO claiming `module_id` is authorized. An exact reserved id is
2024    /// authorized only by its expected nonce; otherwise a matching reserved prefix
2025    /// is authorized by the owner module's current spawn nonce. Non-reserved ids
2026    /// with no matching prefix are always authorized.
2027    pub fn reserved_hello_authorized(&self, module_id: &str, presented: Option<&str>) -> bool {
2028        self.reserved_hello_rejection(module_id, presented)
2029            .is_none()
2030    }
2031
2032    pub(crate) fn reserved_hello_rejection(
2033        &self,
2034        module_id: &str,
2035        presented: Option<&str>,
2036    ) -> Option<ReservedHelloRejection> {
2037        let nonces = self
2038            .reserved_nonces
2039            .lock()
2040            .unwrap_or_else(|poisoned| poisoned.into_inner());
2041        if let Some(expected) = nonces.get(module_id) {
2042            // `None` = reserved with no legitimate holder: refuse every
2043            // presentation, because no process can hold a nonce that was never
2044            // minted. Only a real minted nonce admits, in constant time.
2045            let authorized = match expected {
2046                Some(expected) => {
2047                    presented.is_some_and(|p| constant_time_eq(expected.as_bytes(), p.as_bytes()))
2048                }
2049                None => false,
2050            };
2051            if authorized {
2052                return None;
2053            }
2054            return Some(ReservedHelloRejection::Exact {
2055                module_id: module_id.to_string(),
2056            });
2057        }
2058        drop(nonces);
2059
2060        let matched_prefix = self
2061            .reserved_prefix_owners
2062            .lock()
2063            .unwrap_or_else(|poisoned| poisoned.into_inner())
2064            .iter()
2065            .filter(|(prefix, _)| module_id.starts_with(prefix.as_str()))
2066            .max_by_key(|(prefix, _)| prefix.len())
2067            .map(|(prefix, owner)| (prefix.clone(), owner.clone()));
2068        let (prefix, owner_module_id) = matched_prefix?;
2069
2070        let authorized = presented.is_some_and(|presented| {
2071            self.spawn_nonces
2072                .lock()
2073                .unwrap_or_else(|poisoned| poisoned.into_inner())
2074                .get(&owner_module_id)
2075                .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()))
2076                // While the owner is being swapped, children started by
2077                // either of its two processes hold that process's nonce.
2078                || self.swap_nonce_matches(&owner_module_id, presented)
2079        });
2080        if authorized {
2081            None
2082        } else {
2083            Some(ReservedHelloRejection::Prefix {
2084                prefix,
2085                owner_module_id,
2086            })
2087        }
2088    }
2089
2090    /// Whether a consumer connection proved it came from a daemon-spawned module.
2091    ///
2092    /// Absence of an expected spawn nonce is a hard failure: consumer_identity is
2093    /// accepted only for module ids the supervisor has spawned.
2094    pub fn spawned_consumer_authorized(&self, module_id: &str, presented: &str) -> bool {
2095        if presented.is_empty() {
2096            return false;
2097        }
2098        let nonces = self
2099            .spawn_nonces
2100            .lock()
2101            .unwrap_or_else(|poisoned| poisoned.into_inner());
2102        let current = nonces
2103            .get(module_id)
2104            .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()));
2105        drop(nonces);
2106        // During a swap two processes of the module are alive, and a consumer
2107        // started by either one presents that process's nonce. Accepting only
2108        // the recorded one would fail the incumbent's consumers for the whole
2109        // overlap once cutover moves the record to the candidate.
2110        current || self.swap_nonce_matches(module_id, presented)
2111    }
2112
2113    /// Whether `presented` is either nonce of an open swap for `module_id`.
2114    fn swap_nonce_matches(&self, module_id: &str, presented: &str) -> bool {
2115        let swaps = self
2116            .swaps
2117            .lock()
2118            .unwrap_or_else(|poisoned| poisoned.into_inner());
2119        swaps.get(module_id).is_some_and(|swap| {
2120            constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes())
2121                || swap.incumbent_nonce.as_deref().is_some_and(|incumbent| {
2122                    constant_time_eq(incumbent.as_bytes(), presented.as_bytes())
2123                })
2124        })
2125    }
2126
2127    /// Open a swap for `module_id` with the candidate's freshly minted nonce.
2128    /// Called before the candidate process exists.
2129    pub(crate) fn open_swap(&self, module_id: &str, candidate_nonce: String) {
2130        let incumbent_nonce = self
2131            .spawn_nonces
2132            .lock()
2133            .unwrap_or_else(|poisoned| poisoned.into_inner())
2134            .get(module_id)
2135            .cloned();
2136        self.swaps
2137            .lock()
2138            .unwrap_or_else(|poisoned| poisoned.into_inner())
2139            .insert(
2140                module_id.to_string(),
2141                OpenSwap {
2142                    candidate_nonce,
2143                    incumbent_nonce,
2144                    candidate_admitted: false,
2145                },
2146            );
2147    }
2148
2149    /// Close the swap for `module_id`, releasing whichever nonce is no longer
2150    /// the module's recorded one.
2151    pub(crate) fn close_swap(&self, module_id: &str) {
2152        self.swaps
2153            .lock()
2154            .unwrap_or_else(|poisoned| poisoned.into_inner())
2155            .remove(module_id);
2156    }
2157
2158    /// Install the observer told about swap promotions, replacing any earlier
2159    /// one.
2160    pub(crate) fn set_swap_promotion_observer(
2161        &self,
2162        observer: std::sync::Weak<dyn SwapPromotionObserver>,
2163    ) {
2164        *self
2165            .promotion_observer
2166            .0
2167            .lock()
2168            .unwrap_or_else(|poisoned| poisoned.into_inner()) = Some(observer);
2169    }
2170
2171    /// Tell the installed observer, if it is still alive, that a swap promoted
2172    /// `registration`.
2173    fn notify_swap_promoted(&self, registration: &crate::registry::ModuleRegistration) {
2174        let observer = self
2175            .promotion_observer
2176            .0
2177            .lock()
2178            .unwrap_or_else(|poisoned| poisoned.into_inner())
2179            .as_ref()
2180            .and_then(std::sync::Weak::upgrade);
2181        if let Some(observer) = observer {
2182            observer.swap_promoted(registration);
2183        }
2184    }
2185
2186    /// Whether a swap is open for `module_id`.
2187    pub(crate) fn swap_open(&self, module_id: &str) -> bool {
2188        self.swaps
2189            .lock()
2190            .unwrap_or_else(|poisoned| poisoned.into_inner())
2191            .contains_key(module_id)
2192    }
2193
2194    /// Make the candidate's nonce the module's recorded spawn nonce, as a plain
2195    /// respawn would, once cutover has made the candidate the module's process.
2196    /// The swap stays open so the incumbent's nonce keeps attesting until the
2197    /// incumbent has drained and exited.
2198    fn promote_swap_nonce(&self, module_id: &str, reserved: bool) {
2199        let candidate_nonce = self
2200            .swaps
2201            .lock()
2202            .unwrap_or_else(|poisoned| poisoned.into_inner())
2203            .get(module_id)
2204            .map(|swap| swap.candidate_nonce.clone());
2205        let Some(nonce) = candidate_nonce else {
2206            return;
2207        };
2208        self.set_spawn_nonce(module_id, nonce.clone());
2209        if reserved {
2210            self.set_reserved_nonce(module_id, nonce);
2211        }
2212    }
2213
2214    /// The swap gate for a HELLO claiming `module_id`.
2215    ///
2216    /// This runs BEFORE the reserved-module gate. A reserved module's candidate
2217    /// presents the candidate nonce, which the reserved gate (holding the
2218    /// incumbent's nonce) would refuse as `reserved_module` before swap
2219    /// admission was ever reached. And it applies to unreserved ids too: for an
2220    /// unreserved id the only thing that ever stopped a second process claiming
2221    /// a live id was the `duplicate_module_id` refusal, which is exactly the
2222    /// refusal a swap lifts for its candidate.
2223    ///
2224    /// The incumbent's own nonce falls through to the ordinary gates, which
2225    /// treat it as they always have (a live incumbent is refused as a
2226    /// duplicate). Anything else while a swap is open is refused, including an
2227    /// absent nonce.
2228    pub(crate) fn swap_hello_admission(
2229        &self,
2230        module_id: &str,
2231        presented: Option<&str>,
2232    ) -> SwapHelloAdmission {
2233        let swaps = self
2234            .swaps
2235            .lock()
2236            .unwrap_or_else(|poisoned| poisoned.into_inner());
2237        let Some(swap) = swaps.get(module_id) else {
2238            return SwapHelloAdmission::NotSwapping;
2239        };
2240        let Some(presented) = presented else {
2241            return SwapHelloAdmission::Refused;
2242        };
2243        if constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes()) {
2244            return if swap.candidate_admitted {
2245                SwapHelloAdmission::Refused
2246            } else {
2247                SwapHelloAdmission::Candidate
2248            };
2249        }
2250        if swap
2251            .incumbent_nonce
2252            .as_deref()
2253            .is_some_and(|incumbent| constant_time_eq(incumbent.as_bytes(), presented.as_bytes()))
2254        {
2255            return SwapHelloAdmission::NotSwapping;
2256        }
2257        SwapHelloAdmission::Refused
2258    }
2259
2260    /// Record that the swap token has registered a candidate, so it admits no
2261    /// second HELLO.
2262    pub(crate) fn mark_swap_candidate_admitted(&self, module_id: &str) {
2263        if let Some(swap) = self
2264            .swaps
2265            .lock()
2266            .unwrap_or_else(|poisoned| poisoned.into_inner())
2267            .get_mut(module_id)
2268        {
2269            swap.candidate_admitted = true;
2270        }
2271    }
2272
2273    /// Test/support lookup for the current launch nonce of a supervised spawn.
2274    pub fn spawn_launch_nonce_for(&self, module_id: &str) -> Option<String> {
2275        self.spawn_nonces
2276            .lock()
2277            .unwrap_or_else(|poisoned| poisoned.into_inner())
2278            .get(module_id)
2279            .cloned()
2280    }
2281
2282    /// Test/support lookup for the HELLO-gating nonce of a reserved module.
2283    pub fn reserved_launch_nonce_for(&self, module_id: &str) -> Option<String> {
2284        self.reserved_nonces
2285            .lock()
2286            .unwrap_or_else(|poisoned| poisoned.into_inner())
2287            .get(module_id)
2288            .cloned()
2289            .flatten()
2290    }
2291
2292    pub fn insert(&self, module: SupervisedModule) -> Option<SupervisedModule> {
2293        // Normally already marked before the process was spawned; marking here
2294        // too keeps `configured_ids` a superset of the roster for any caller
2295        // that inserts a module directly.
2296        self.mark_configured(module.module_id());
2297        let mut modules = self
2298            .modules
2299            .lock()
2300            .unwrap_or_else(|poisoned| poisoned.into_inner());
2301        modules.insert(module.module_id().to_string(), module)
2302    }
2303
2304    /// Record that the supervisor has taken on `module_id`. Called before the
2305    /// module's first process is spawned, so that by the time that process can
2306    /// register, [`Self::is_configured`] already answers true.
2307    fn mark_configured(&self, module_id: &str) {
2308        self.configured_ids
2309            .lock()
2310            .unwrap_or_else(|poisoned| poisoned.into_inner())
2311            .insert(module_id.to_string());
2312    }
2313
2314    /// Undo [`Self::mark_configured`] for a module whose first spawn failed
2315    /// before it was ever put on the roster. A module already on the roster
2316    /// keeps its mark: only [`Self::retire`] takes a rostered module off.
2317    fn unmark_configured_unless_rostered(&self, module_id: &str) {
2318        let modules = self
2319            .modules
2320            .lock()
2321            .unwrap_or_else(|poisoned| poisoned.into_inner());
2322        if !modules.contains_key(module_id) {
2323            self.configured_ids
2324                .lock()
2325                .unwrap_or_else(|poisoned| poisoned.into_inner())
2326                .remove(module_id);
2327        }
2328    }
2329
2330    /// Whether `module_id` is a module this daemon supervises: on the roster,
2331    /// or about to be (its process is being spawned right now).
2332    ///
2333    /// This, not `get(..).is_some()`, is what "the owner is configured" means
2334    /// for scopes: a supervised module's process can register and sync before
2335    /// [`Self::get`] can return it, and in that window it is still a module
2336    /// that will sync, not one that never will.
2337    pub(crate) fn is_configured(&self, module_id: &str) -> bool {
2338        self.configured_ids
2339            .lock()
2340            .unwrap_or_else(|poisoned| poisoned.into_inner())
2341            .contains(module_id)
2342    }
2343
2344    pub fn get(&self, module_id: &str) -> Option<SupervisedModule> {
2345        let modules = self
2346            .modules
2347            .lock()
2348            .unwrap_or_else(|poisoned| poisoned.into_inner());
2349        modules.get(module_id).cloned()
2350    }
2351
2352    pub(crate) fn record_late_health_answer(
2353        &self,
2354        module_id: &str,
2355        latency_ms: u64,
2356    ) -> Result<bool, SuperviseError> {
2357        let Some(module) = self.get(module_id) else {
2358            return Ok(false);
2359        };
2360        update_snapshot(&module.inner.snapshot, Some(module_id), |state| {
2361            state.health.late_answer_count = state.health.late_answer_count.saturating_add(1);
2362            state.health.last_late_answer_latency_ms = Some(latency_ms);
2363            // A late answer is an answer: the module served the probe, just past
2364            // the deadline. Leaving the miss streak in place while logging
2365            // "proves the module is alive" is how a CPU-starved module that
2366            // answers every probe a few seconds late still marches to the
2367            // threshold and gets killed — the exact kill class `NoAnswer` is
2368            // excluded from `is_proof_of_death` to prevent. Slow-but-answering
2369            // is degradation, and degradation reports; it does not restart.
2370            state.health.consecutive_failures = 0;
2371        })?;
2372        Ok(true)
2373    }
2374
2375    /// Arm the one-shot marker for the module process that this caller
2376    /// deliberately initiated severance against. Generic connection teardown
2377    /// must not call this:
2378    /// a surviving process would otherwise retain an exemption for a later
2379    /// genuine crash.
2380    pub fn record_deliberate_severance(&self, module_id: &str) -> Result<bool, SuperviseError> {
2381        let Some(module) = self.get(module_id) else {
2382            return Ok(false);
2383        };
2384        let snapshot = lock_snapshot(&module.inner.snapshot)?;
2385        let Some((pid, start_time)) = snapshot.pid.zip(snapshot.process_start_time) else {
2386            return Ok(false);
2387        };
2388        drop(snapshot);
2389        module.record_deliberate_severance(ProcessIdentity { pid, start_time })
2390    }
2391
2392    pub fn list(&self) -> Vec<SupervisedModule> {
2393        let modules = self
2394            .modules
2395            .lock()
2396            .unwrap_or_else(|poisoned| poisoned.into_inner());
2397        let mut modules = modules.values().cloned().collect::<Vec<_>>();
2398        modules.sort_by(|left, right| left.module_id().cmp(right.module_id()));
2399        modules
2400    }
2401
2402    pub(crate) fn retire(&self, module_id: &str) -> Option<SupervisedModule> {
2403        self.spawn_nonces
2404            .lock()
2405            .unwrap_or_else(|poisoned| poisoned.into_inner())
2406            .remove(module_id);
2407        self.close_swap(module_id);
2408        let mut reserved_nonces = self
2409            .reserved_nonces
2410            .lock()
2411            .unwrap_or_else(|poisoned| poisoned.into_inner());
2412        if reserved_nonces.contains_key(module_id) {
2413            // The old nonce must die with the removed process, but the exact-id
2414            // gate remains until an operator explicitly releases it.
2415            reserved_nonces.insert(module_id.to_string(), None);
2416        }
2417        drop(reserved_nonces);
2418        self.reserved_prefix_owners
2419            .lock()
2420            .unwrap_or_else(|poisoned| poisoned.into_inner())
2421            .retain(|_, owner| owner != module_id);
2422        let removed = self
2423            .modules
2424            .lock()
2425            .unwrap_or_else(|poisoned| poisoned.into_inner())
2426            .remove(module_id);
2427        self.configured_ids
2428            .lock()
2429            .unwrap_or_else(|poisoned| poisoned.into_inner())
2430            .remove(module_id);
2431        removed
2432    }
2433
2434    /// Remember a module removed by a non-preview rescan so route.open can
2435    /// distinguish that intentional removal from an unknown id.
2436    pub(crate) fn record_rescan_removal(&self, module_id: &str) {
2437        self.removal_tombstones
2438            .lock()
2439            .unwrap_or_else(|poisoned| poisoned.into_inner())
2440            .insert(module_id.to_string(), unix_ms_now());
2441    }
2442
2443    /// Return how long ago a rescan removed this module in milliseconds.
2444    pub(crate) fn removal_tombstone_age_ms(&self, module_id: &str) -> Option<u64> {
2445        self.removal_tombstones
2446            .lock()
2447            .unwrap_or_else(|poisoned| poisoned.into_inner())
2448            .get(module_id)
2449            .copied()
2450            .map(|removed_at_ms| unix_ms_now().saturating_sub(removed_at_ms))
2451    }
2452
2453    /// Retire a reserved-id gate only after its module has left supervision.
2454    ///
2455    /// A retained gate has no live nonce (`None`), so releasing any other entry
2456    /// would weaken a currently configured or otherwise active reservation.
2457    pub(crate) fn release_retained_reserved_gate(&self, module_id: &str) -> bool {
2458        if self.get(module_id).is_some() {
2459            return false;
2460        }
2461        let mut reserved_nonces = self
2462            .reserved_nonces
2463            .lock()
2464            .unwrap_or_else(|poisoned| poisoned.into_inner());
2465        if !matches!(reserved_nonces.get(module_id), Some(None)) {
2466            return false;
2467        }
2468        reserved_nonces.remove(module_id);
2469        true
2470    }
2471
2472    pub(crate) fn operation_lock(&self) -> Arc<AsyncMutex<()>> {
2473        Arc::clone(&self.operation_lock)
2474    }
2475}
2476
2477/// Process supervisor for subc-owned singleton modules.
2478#[derive(Debug, Clone)]
2479pub struct Supervisor {
2480    registry: Arc<Registry>,
2481    restart_policy: RestartPolicy,
2482    drain_timeout: Duration,
2483    connection_file_path: Option<PathBuf>,
2484    capture_logs_dir: Option<PathBuf>,
2485    forwarding: Option<Arc<ForwardingTable>>,
2486    process_liveness: Arc<SupervisorProcessLiveness>,
2487    supervisor_handle: Option<SupervisorHandle>,
2488    health: HealthConfig,
2489    daemon_start_clock: crate::clock::StartClock,
2490    terminal_journal: Option<Arc<crate::terminal_journal::TerminalJournal>>,
2491    spawn_events: SpawnEventFeed,
2492    provenance_probe: ExecutableIdentityProbe,
2493    /// Every process spawned through this supervisor (and its clones) and not
2494    /// yet reaped, so daemon shutdown can end them.
2495    child_roster: ChildRoster,
2496    #[cfg(target_os = "linux")]
2497    cgroup_placement: Option<subc_cgroup::Placement>,
2498    #[cfg(test)]
2499    test_after_first_spawn: AfterFirstSpawnHook,
2500}
2501
2502/// Test-only hook run on the path that takes on a new module, right after its
2503/// first `spawn_child` returns (the process exists and could already be
2504/// registering) and before that process is handed to the module's supervise
2505/// loop and put on the roster. Lets a test observe what a fast child would see
2506/// in that window without racing a real one.
2507#[cfg(test)]
2508#[derive(Clone, Default)]
2509struct AfterFirstSpawnHook(Option<AfterFirstSpawnFn>);
2510
2511#[cfg(test)]
2512type AfterFirstSpawnFn = Arc<dyn Fn(&str) + Send + Sync>;
2513
2514#[cfg(test)]
2515impl fmt::Debug for AfterFirstSpawnHook {
2516    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
2517        f.write_str("AfterFirstSpawnHook")
2518    }
2519}
2520
2521#[cfg(test)]
2522impl AfterFirstSpawnHook {
2523    fn run(&self, module_id: &str) {
2524        if let Some(hook) = &self.0 {
2525            hook(module_id);
2526        }
2527    }
2528}
2529
2530impl Supervisor {
2531    #[cfg(test)]
2532    pub(crate) fn new_for_test(registry: Arc<Registry>, policy: RestartPolicy) -> Self {
2533        let supervisor = Self::new(registry, policy);
2534        #[cfg(target_os = "macos")]
2535        let supervisor = supervisor.with_privacy_trampoline(test_privacy_trampoline());
2536        supervisor
2537    }
2538    /// Verify the trampoline once when configured. Missing private OS support
2539    /// refuses every macOS launch by name but does not stop the daemon's control
2540    /// server. Embedders must explicitly provide a binary with the subc-os hidden
2541    /// entry point; the library must not exec an arbitrary hosting program.
2542    pub fn with_privacy_trampoline(self, path: impl Into<PathBuf>) -> Self {
2543        let path = path.into();
2544        #[cfg(target_os = "macos")]
2545        {
2546            let result = probe_privacy_trampoline(&path).map(|()| path);
2547            if let Err(cause) = &result {
2548                error!(%cause, "macOS privacy identity unavailable; supervised launches will refuse");
2549            }
2550            self.child_roster.set_privacy_trampoline(result);
2551        }
2552        #[cfg(not(target_os = "macos"))]
2553        let _ = path;
2554        self
2555    }
2556    /// The first step of an announced daemon shutdown, before the notice and
2557    /// before any connection is closed.
2558    ///
2559    /// Sets the daemon-shutdown flag first: from here on no module is
2560    /// respawned (crash restart, operator restart, or swap), and every child
2561    /// exit is recorded as `daemon_shutdown` rather than as a crash, whether
2562    /// the module exits on the EOF this shutdown gives it or is signalled by a
2563    /// service manager that kills the whole cgroup. Then writes the journal's
2564    /// shutdown marker, which records the instant and closes this daemon
2565    /// incarnation's stretch of the journal.
2566    #[cfg(unix)]
2567    pub(crate) fn begin_daemon_shutdown(&self) {
2568        self.child_roster.close();
2569        if let Some(journal) = &self.terminal_journal {
2570            journal.stamp_shutdown();
2571        }
2572    }
2573
2574    /// Announce a cut while established connections can still carry replies.
2575    /// These budgets promise notice and a bounded wait, not child completion;
2576    /// they are local policy, not an estimate of launchd's unknown kill ceiling.
2577    #[cfg(unix)]
2578    pub(crate) async fn drain_for_daemon_shutdown(&self) -> Result<(), SuperviseError> {
2579        const NOTICE_BUDGET: Duration = Duration::from_millis(500);
2580        const DRAIN_BUDGET: Duration = Duration::from_secs(2);
2581        let Some(forwarding) = &self.forwarding else {
2582            return Ok(());
2583        };
2584        let module_ids = forwarding
2585            .begin_daemon_drain()
2586            .map_err(SuperviseError::Forwarding)?;
2587        let deadline_ms =
2588            unix_ms_now().saturating_add((NOTICE_BUDGET + DRAIN_BUDGET).as_millis() as u64);
2589        let mut notices = tokio::task::JoinSet::new();
2590        let mut drains = Vec::new();
2591        for module_id in module_ids {
2592            let Some(target) = forwarding
2593                .begin_module_drain(&module_id, RouteCloseReason::Restart)
2594                .map_err(SuperviseError::Forwarding)?
2595            else {
2596                continue;
2597            };
2598            let routes = forwarding
2599                .endpoint_routes(target.endpoint)
2600                .map_err(SuperviseError::Forwarding)?;
2601            // Restart allows deployed consumers to reopen after the new daemon
2602            // appears. The wire reason stays `restart`; what tells a daemon cut
2603            // apart from a module restart afterwards is the terminal record
2604            // itself, whose disposition is `daemon_shutdown` for every exit
2605            // observed once `begin_daemon_shutdown` has run.
2606            let command = serde_json::to_vec(&ModuleControlCommand::Draining {
2607                reason: RouteCloseReason::Restart,
2608                deadline_ms,
2609            })
2610            .expect("module draining serializes");
2611            let mut recipients = vec![(target.sink.clone(), target.negotiated_ver, command)];
2612            let mut clients: Vec<(crate::forwarding::GoodbyeTarget, Vec<u16>)> = Vec::new();
2613            for route in routes {
2614                let client = route.goodbye_target;
2615                if let Some((_, channels)) = clients
2616                    .iter_mut()
2617                    .find(|(existing, _)| existing.connection_id == client.connection_id)
2618                {
2619                    channels.push(client.channel);
2620                } else {
2621                    let channel = client.channel;
2622                    clients.push((client, vec![channel]));
2623                }
2624            }
2625            for (client, mut channels) in clients {
2626                channels.sort_unstable();
2627                channels.dedup();
2628                let closing = serde_json::to_vec(&ClientControlPush::RouteClosing {
2629                    module_id: module_id.clone(),
2630                    channels,
2631                    reason: RouteCloseReason::Restart,
2632                })
2633                .expect("route closing serializes");
2634                recipients.push((client.sink, client.negotiated_ver, closing));
2635            }
2636            for (sink, version, body) in recipients {
2637                notices.spawn(async move {
2638                    let frame = Frame::build_with_version(
2639                        version,
2640                        FrameType::Push,
2641                        control_flags(),
2642                        0,
2643                        0,
2644                        0,
2645                        body,
2646                    )
2647                    .expect("bounded lifecycle notice frame builds");
2648                    sink.send_flushed(frame).await
2649                });
2650            }
2651            let gauges = declared_busy_gauges(&self.registry, &module_id)?;
2652            drains.push((module_id, target.endpoint, gauges));
2653        }
2654        // A quiet forwarding table is not proof that queued notices reached the
2655        // socket. Wait for writer flush acknowledgements before testing quiescence.
2656        let notice_deadline = Instant::now() + NOTICE_BUDGET;
2657        while let Ok(Some(result)) = timeout_at(notice_deadline, notices.join_next()).await {
2658            if !matches!(result, Ok(Ok(()))) {
2659                warn!(?result, "daemon shutdown notice delivery failed");
2660            }
2661        }
2662        notices.abort_all();
2663        let deadline = Instant::now() + DRAIN_BUDGET;
2664        let mut waits = tokio::task::JoinSet::new();
2665        for (module_id, endpoint, gauges) in drains {
2666            let forwarding = Arc::clone(forwarding);
2667            let mut runtime = self.runtime_config();
2668            runtime.health.cadence = Duration::from_millis(100);
2669            waits.spawn(async move {
2670                wait_for_forwarding_quiescence(
2671                    &forwarding,
2672                    &module_id,
2673                    &runtime,
2674                    endpoint,
2675                    deadline,
2676                    &gauges,
2677                    DrainScope::Active,
2678                )
2679                .await
2680            });
2681        }
2682        while let Ok(Some(result)) = timeout_at(deadline, waits.join_next()).await {
2683            if !matches!(result, Ok(Ok(true))) {
2684                warn!(?result, "daemon shutdown drain did not reach quiescence");
2685            }
2686        }
2687        Ok(())
2688    }
2689
2690    /// The last step of an announced daemon shutdown, after the notice and the
2691    /// drain: send every registered module a module GOODBYE, the same planned
2692    /// stop signal `ck module stop` gives, then close every connection so each
2693    /// subc module sees EOF and starts its own teardown, then end every
2694    /// supervised child that has not exited
2695    /// by its own deadline (its drain budget, capped). Modules lead their own
2696    /// process groups, so a
2697    /// service manager's group kill no longer reaches them; without this a
2698    /// child that does not stop on EOF (every `protocol: "none"` child, which
2699    /// has no connection) would outlive the daemon. Every wait is bounded (see
2700    /// `child_roster`), and `escalate` resolving (a second SIGTERM) cuts them.
2701    #[cfg(unix)]
2702    pub(crate) async fn end_children_for_daemon_shutdown(
2703        &self,
2704        already_escalated: bool,
2705        escalate: impl std::future::Future<Output = ()>,
2706    ) {
2707        tokio::pin!(escalate);
2708        let mut escalated = already_escalated;
2709        if let Some(forwarding) = &self.forwarding {
2710            let reason = CloseReason::new(
2711                "daemon_shutdown",
2712                "the daemon is exiting after its shutdown notice and drain",
2713            );
2714            if escalated {
2715                // The operator asked to stop waiting: queue the GOODBYEs but
2716                // do not wait for them to be written.
2717                send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, false).await;
2718            } else {
2719                tokio::select! {
2720                    biased;
2721                    _ = escalate.as_mut() => {
2722                        info!("second SIGTERM: abandoning module GOODBYE delivery");
2723                        escalated = true;
2724                    }
2725                    _ = send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, true) => {}
2726                }
2727            }
2728            let closed = forwarding.close_all_connections(&reason);
2729            debug!(closed, "closed established connections for daemon shutdown");
2730        }
2731        // If the second SIGTERM arrived during GOODBYE delivery, `escalate` has
2732        // already completed and must not be polled again; the child shutdown
2733        // wait is told it is escalated and gets a future that never fires.
2734        let escalated_here = escalated && !already_escalated;
2735        let remaining_escalate = async move {
2736            if escalated_here {
2737                std::future::pending::<()>().await;
2738            } else {
2739                escalate.await;
2740            }
2741        };
2742        crate::child_roster::end_children_for_daemon_shutdown(
2743            &self.child_roster,
2744            escalated,
2745            remaining_escalate,
2746        )
2747        .await;
2748    }
2749
2750    pub fn new(registry: Arc<Registry>, restart_policy: RestartPolicy) -> Self {
2751        Self {
2752            registry,
2753            restart_policy,
2754            drain_timeout: DEFAULT_DRAIN_TIMEOUT,
2755            connection_file_path: None,
2756            capture_logs_dir: None,
2757            forwarding: None,
2758            process_liveness: Arc::new(SupervisorProcessLiveness::default()),
2759            supervisor_handle: None,
2760            health: HealthConfig::default(),
2761            daemon_start_clock: crate::clock::StartClock::capture(),
2762            terminal_journal: None,
2763            spawn_events: SpawnEventFeed::default(),
2764            provenance_probe: ExecutableIdentityProbe::default(),
2765            child_roster: ChildRoster::default(),
2766            #[cfg(target_os = "linux")]
2767            cgroup_placement: None,
2768            #[cfg(test)]
2769            test_after_first_spawn: AfterFirstSpawnHook::default(),
2770        }
2771    }
2772
2773    pub fn with_drain_timeout(mut self, drain_timeout: Duration) -> Self {
2774        self.drain_timeout = drain_timeout;
2775        self
2776    }
2777
2778    pub fn with_process_liveness(
2779        mut self,
2780        process_liveness: Arc<SupervisorProcessLiveness>,
2781    ) -> Self {
2782        self.process_liveness = process_liveness;
2783        self
2784    }
2785
2786    pub fn with_connection_file_path(mut self, connection_file_path: impl Into<PathBuf>) -> Self {
2787        self.connection_file_path = Some(connection_file_path.into());
2788        self
2789    }
2790
2791    /// Enables daemon-owned capture files for supervised stdout and stderr.
2792    pub fn with_capture_logs_dir(mut self, logs_dir: impl Into<PathBuf>) -> Self {
2793        self.capture_logs_dir = Some(logs_dir.into());
2794        self
2795    }
2796
2797    /// Names this daemon lifetime in spawn events, independently of whether a
2798    /// terminal journal is configured.
2799    pub fn with_daemon_incarnation(self, daemon_incarnation: String) -> Self {
2800        // A millisecond start stamp can repeat after clock rollback or a rapid
2801        // restart. Use the connection file's random daemon_id instead: it already
2802        // identifies this daemon lifetime independently of the wall clock.
2803        self.spawn_events.configure_incarnation(daemon_incarnation);
2804        self
2805    }
2806
2807    /// Enables best-effort history shared by every supervised module. Without
2808    /// it, terminal history is kept only in each module's in-memory ring.
2809    pub fn with_terminal_journal(self, path: PathBuf, daemon_incarnation: String) -> Self {
2810        let mut this = self.with_daemon_incarnation(daemon_incarnation.clone());
2811        this.terminal_journal = Some(Arc::new(crate::terminal_journal::TerminalJournal::open(
2812            path,
2813            daemon_incarnation,
2814        )));
2815        this
2816    }
2817
2818    pub fn with_forwarding(mut self, forwarding: Arc<ForwardingTable>) -> Self {
2819        self.forwarding = Some(forwarding);
2820        self
2821    }
2822
2823    pub fn with_handle(mut self, supervisor_handle: SupervisorHandle) -> Self {
2824        self.spawn_events = supervisor_handle.spawn_events.clone();
2825        self.supervisor_handle = Some(supervisor_handle);
2826        self
2827    }
2828
2829    pub fn with_health_config(mut self, health: HealthConfig) -> Self {
2830        self.health = health;
2831        self
2832    }
2833
2834    /// Keeps a record of every live child at `path`, rewritten on each spawn and
2835    /// reap, for the orphan sweep a later daemon runs at boot. Without it no
2836    /// record is kept.
2837    pub fn with_live_children_record(self, path: impl Into<PathBuf>) -> Self {
2838        self.child_roster.record_to(path.into());
2839        self
2840    }
2841
2842    #[cfg(target_os = "linux")]
2843    pub fn with_cgroup_placement(
2844        mut self,
2845        cgroup_placement: Option<subc_cgroup::Placement>,
2846    ) -> Self {
2847        self.cgroup_placement = cgroup_placement;
2848        self
2849    }
2850
2851    /// Spawn `spec.program` and start monitoring it.
2852    ///
2853    /// The child is expected to parse `--subc <connection-file-path>`, read the
2854    /// TCP+key connection file, authenticate to the already-running listener, and
2855    /// register with channel-0 `HELLO` using `spec.module_id` as its manifest id.
2856    pub fn spawn(&self, spec: ModuleSpec) -> Result<SupervisedModule, SuperviseError> {
2857        validate_spec(&spec)?;
2858        self.establish_identity(&spec);
2859
2860        let runtime = self.runtime_config();
2861        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2862        let spawned = spawn_child(
2863            &spec,
2864            runtime.connection_file_path.as_deref(),
2865            self.supervisor_handle.as_ref(),
2866            &runtime.stderr_ring,
2867            runtime.capture_logs_dir.as_deref(),
2868            &runtime.child_roster,
2869            #[cfg(target_os = "linux")]
2870            runtime.cgroup_placement.as_ref(),
2871        );
2872        #[cfg(test)]
2873        self.test_after_first_spawn.run(&spec.module_id);
2874        let child = match spawned {
2875            Ok(child) => child,
2876            Err(err) => {
2877                // Unlike the configured paths, a failed `spawn` leaves nothing
2878                // on the roster, so the module must not stay marked configured.
2879                self.abandon_unrostered(&spec.module_id);
2880                return Err(err);
2881            }
2882        };
2883        self.mark_first_process_running(&spec, &runtime, &snapshot, &child)?;
2884
2885        Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2886    }
2887
2888    /// Make `spec`'s module count as configured, with its identity gates
2889    /// (reserved id, reserved prefixes) in place, BEFORE any process of it
2890    /// exists.
2891    ///
2892    /// Every path that takes on a new module calls this before `spawn_child`.
2893    /// The order is the point: the child can connect, register, sync its
2894    /// scopes and ask about them as soon as it is spawned, and the module is
2895    /// only put on the roster after `spawn_child` returns. Were the mark set
2896    /// with the roster entry, a fast child would see its own owner reported
2897    /// as not configured, and a scoped `route.open` in that window would be
2898    /// refused as terminal `scope_not_live` ("will never sync") instead of
2899    /// retryable `scope_not_synced`.
2900    fn establish_identity(&self, spec: &ModuleSpec) {
2901        if let Some(supervisor_handle) = &self.supervisor_handle {
2902            supervisor_handle.apply_identity_configuration(spec);
2903            supervisor_handle.mark_configured(&spec.module_id);
2904        }
2905    }
2906
2907    /// Take back [`Self::establish_identity`]'s configured mark when the
2908    /// module will not be put on the roster after all.
2909    fn abandon_unrostered(&self, module_id: &str) {
2910        if let Some(supervisor_handle) = &self.supervisor_handle {
2911            supervisor_handle.unmark_configured_unless_rostered(module_id);
2912        }
2913    }
2914
2915    /// Record a freshly spawned first process as running. On failure the
2916    /// module never reaches the roster, so its configured mark is taken back.
2917    fn mark_first_process_running(
2918        &self,
2919        spec: &ModuleSpec,
2920        runtime: &SupervisorRuntimeConfig,
2921        snapshot: &SharedSnapshot,
2922        child: &SupervisedChild,
2923    ) -> Result<(), SuperviseError> {
2924        if let Err(err) = set_running(snapshot, child, &spec.module_id, &runtime.spawn_events) {
2925            self.abandon_unrostered(&spec.module_id);
2926            return Err(err);
2927        }
2928        self.process_liveness
2929            .track(spec.module_id.clone(), Arc::clone(snapshot));
2930        Ok(())
2931    }
2932
2933    /// Start supervising a module declared in daemon configuration.
2934    ///
2935    /// Unlike [`Self::spawn`], this records disabled modules and immediate spawn
2936    /// failures in the supervisor handle so operator-facing `supervisor.list`
2937    /// reflects every configured module while daemon startup continues.
2938    pub fn supervise_configured(
2939        &self,
2940        spec: ModuleSpec,
2941        enabled: bool,
2942    ) -> Result<SupervisedModule, SuperviseError> {
2943        validate_spec(&spec)?;
2944        self.establish_identity(&spec);
2945
2946        let runtime = self.runtime_config();
2947        if !enabled {
2948            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
2949            return Ok(self.supervised_module(spec, runtime, snapshot, None));
2950        }
2951
2952        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2953        let spawned = spawn_child(
2954            &spec,
2955            runtime.connection_file_path.as_deref(),
2956            self.supervisor_handle.as_ref(),
2957            &runtime.stderr_ring,
2958            runtime.capture_logs_dir.as_deref(),
2959            &runtime.child_roster,
2960            #[cfg(target_os = "linux")]
2961            runtime.cgroup_placement.as_ref(),
2962        );
2963        #[cfg(test)]
2964        self.test_after_first_spawn.run(&spec.module_id);
2965        match spawned {
2966            Ok(child) => {
2967                self.mark_first_process_running(&spec, &runtime, &snapshot, &child)?;
2968                Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2969            }
2970            Err(err) => {
2971                error!(
2972                    module_id = %spec.module_id,
2973                    program = %spec.program.display(),
2974                    error = %err,
2975                    "configured module failed to spawn; marking failed and continuing"
2976                );
2977                let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
2978                Ok(self.supervised_module(spec, runtime, snapshot, None))
2979            }
2980        }
2981    }
2982
2983    /// Supervise a configured module with its own health, drain, and crash
2984    /// budget. The restart policy is per-module because the config file is:
2985    /// `modules.<id>.restart` resolves to a full policy at parse time, and a
2986    /// module that is expensive to restart should not be forced onto the same
2987    /// budget as one that is cheap.
2988    pub fn supervise_configured_with_health(
2989        &self,
2990        spec: ModuleSpec,
2991        enabled: bool,
2992        health: HealthConfig,
2993        drain_timeout_ms: Option<u64>,
2994        restart_policy: RestartPolicy,
2995    ) -> Result<SupervisedModule, SuperviseError> {
2996        validate_spec(&spec)?;
2997        self.establish_identity(&spec);
2998
2999        let mut runtime = self.runtime_config();
3000        runtime.health = health.clone();
3001        runtime.restart_policy = restart_policy;
3002        if let Some(ms) = drain_timeout_ms {
3003            runtime.drain_timeout = Duration::from_millis(ms);
3004            *runtime
3005                .effective_drain_timeout
3006                .lock()
3007                .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
3008        }
3009        if !enabled {
3010            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
3011            return Ok(self.supervised_module(spec, runtime, snapshot, None));
3012        }
3013
3014        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
3015        let spawned = spawn_child(
3016            &spec,
3017            runtime.connection_file_path.as_deref(),
3018            self.supervisor_handle.as_ref(),
3019            &runtime.stderr_ring,
3020            runtime.capture_logs_dir.as_deref(),
3021            &runtime.child_roster,
3022            #[cfg(target_os = "linux")]
3023            runtime.cgroup_placement.as_ref(),
3024        );
3025        #[cfg(test)]
3026        self.test_after_first_spawn.run(&spec.module_id);
3027        match spawned {
3028            Ok(child) => {
3029                self.mark_first_process_running(&spec, &runtime, &snapshot, &child)?;
3030                Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
3031            }
3032            Err(err) => {
3033                if health.critical {
3034                    error!(
3035                        module_id = %spec.module_id,
3036                        program = %spec.program.display(),
3037                        error = %err,
3038                        "critical configured module failed to spawn; marking failed and alerting"
3039                    );
3040                } else {
3041                    error!(
3042                        module_id = %spec.module_id,
3043                        program = %spec.program.display(),
3044                        error = %err,
3045                        "configured module failed to spawn; marking failed and continuing"
3046                    );
3047                }
3048                let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
3049                Ok(self.supervised_module(spec, runtime, snapshot, None))
3050            }
3051        }
3052    }
3053
3054    fn runtime_config(&self) -> SupervisorRuntimeConfig {
3055        let effective_drain_timeout = Arc::new(Mutex::new(self.drain_timeout));
3056        SupervisorRuntimeConfig {
3057            scheduled_respawn: Arc::default(),
3058            deferred_reload_reply: Arc::default(),
3059            restart_policy: self.restart_policy,
3060            drain_timeout: self.drain_timeout,
3061            // Shared with this module's roster copy: daemon shutdown waits on
3062            // each child for the module's own drain budget, as resolved now.
3063            child_roster: self
3064                .child_roster
3065                .for_module(Arc::clone(&effective_drain_timeout)),
3066            effective_drain_timeout,
3067            default_drain_timeout: self.drain_timeout,
3068            health: self.health.clone(),
3069            connection_file_path: self.connection_file_path.clone(),
3070            capture_logs_dir: self.capture_logs_dir.clone(),
3071            forwarding: self.forwarding.clone(),
3072            supervisor_handle: self.supervisor_handle.clone(),
3073            stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
3074            terminal_ring: Arc::new(Mutex::new(
3075                TerminalRing::new(
3076                    TerminalRingConfig::default(),
3077                    self.daemon_start_clock.started_at_ms(),
3078                )
3079                .with_start_clock(self.daemon_start_clock)
3080                .with_journal(self.terminal_journal.clone())
3081                .with_daemon_shutdown(self.child_roster.shutdown_flag()),
3082            )),
3083            spawn_events: self.spawn_events.clone(),
3084            #[cfg(target_os = "linux")]
3085            cgroup_placement: self.cgroup_placement.clone(),
3086            #[cfg(test)]
3087            test_seed_stale_facts_before_enable_spawn: false,
3088            #[cfg(test)]
3089            test_reload_exit_record_gate: None,
3090        }
3091    }
3092
3093    fn supervised_module(
3094        &self,
3095        spec: ModuleSpec,
3096        runtime: SupervisorRuntimeConfig,
3097        snapshot: SharedSnapshot,
3098        child: Option<SupervisedChild>,
3099    ) -> SupervisedModule {
3100        let configuration = Arc::new(Mutex::new(SupervisedConfiguration {
3101            spec: spec.clone(),
3102            health: runtime.health.clone(),
3103        }));
3104        let stderr_ring = Arc::clone(&runtime.stderr_ring);
3105        let terminal_ring = Arc::clone(&runtime.terminal_ring);
3106        // The module's OWN policy, which may be its per-module config rather than
3107        // the supervisor-wide one; status must report the budget the supervise
3108        // loop actually enforces.
3109        let restart_policy = runtime.restart_policy;
3110        let effective_drain_timeout = Arc::clone(&runtime.effective_drain_timeout);
3111        let (tx, rx) = mpsc::channel(4);
3112        let monitor = tokio::spawn(supervise_loop(
3113            spec.clone(),
3114            runtime,
3115            Arc::clone(&self.registry),
3116            Arc::clone(&self.process_liveness),
3117            Arc::clone(&snapshot),
3118            child,
3119            rx,
3120        ));
3121
3122        let module_id = spec.module_id.clone();
3123        let module = SupervisedModule {
3124            inner: Arc::new(SupervisedModuleInner {
3125                module_id: module_id.clone(),
3126                registry: Arc::clone(&self.registry),
3127                snapshot,
3128                configuration,
3129                stderr_ring,
3130                terminal_ring,
3131                commands: tx,
3132                monitor: Mutex::new(Some(monitor)),
3133                restart_policy,
3134                effective_drain_timeout,
3135                provenance_probe: self.provenance_probe.clone(),
3136            }),
3137        };
3138        // The identity gates and the configured mark were set by
3139        // `establish_identity` before any process was spawned; only the roster
3140        // entry waits for the module handle, which needs the spawned child.
3141        if let Some(supervisor_handle) = &self.supervisor_handle {
3142            supervisor_handle.insert(module.clone());
3143        }
3144        module
3145    }
3146}
3147
3148impl Default for Supervisor {
3149    fn default() -> Self {
3150        Self::new(Arc::new(Registry::default()), RestartPolicy::default())
3151    }
3152}
3153
3154/// Handle to one supervised child process.
3155#[derive(Clone)]
3156pub struct SupervisedModule {
3157    inner: Arc<SupervisedModuleInner>,
3158}
3159
3160struct SupervisedModuleInner {
3161    module_id: String,
3162    registry: Arc<Registry>,
3163    snapshot: SharedSnapshot,
3164    configuration: Arc<Mutex<SupervisedConfiguration>>,
3165    stderr_ring: Arc<Mutex<StderrRing>>,
3166    terminal_ring: Arc<Mutex<TerminalRing>>,
3167    commands: mpsc::Sender<SupervisorCommand>,
3168    monitor: Mutex<Option<JoinHandle<()>>>,
3169    /// Copied from the supervisor's runtime config at spawn so `status()` can
3170    /// report the restart budget without reaching back into the supervisor. The
3171    /// policy is fixed for the process's lifetime, so a copy cannot drift.
3172    restart_policy: RestartPolicy,
3173    effective_drain_timeout: Arc<Mutex<Duration>>,
3174    provenance_probe: ExecutableIdentityProbe,
3175}
3176
3177impl fmt::Debug for SupervisedModule {
3178    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3179        f.debug_struct("SupervisedModule")
3180            .field("module_id", &self.inner.module_id)
3181            .field("status", &self.status())
3182            .finish_non_exhaustive()
3183    }
3184}
3185
3186impl SupervisedModule {
3187    pub fn module_id(&self) -> &str {
3188        &self.inner.module_id
3189    }
3190
3191    /// Test-only: put one probe miss on the streak, the way
3192    /// `handle_health_probe_failure` does, so tests can assert what a later
3193    /// event does to the streak without driving the whole probe loop.
3194    #[cfg(test)]
3195    pub(crate) fn record_health_probe_failure_for_test(
3196        &self,
3197        detail: &str,
3198    ) -> Result<(), SuperviseError> {
3199        update_snapshot(&self.inner.snapshot, Some(&self.inner.module_id), |state| {
3200            state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
3201            state.health.detail = Some(detail.to_string());
3202        })
3203    }
3204
3205    pub fn state(&self) -> Result<ModuleState, SuperviseError> {
3206        Ok(lock_snapshot(&self.inner.snapshot)?.state)
3207    }
3208
3209    pub(crate) fn registration_end_reason(
3210        &self,
3211    ) -> Result<Option<RegistrationEndReason>, SuperviseError> {
3212        let snapshot = lock_snapshot(&self.inner.snapshot)?;
3213        Ok(match snapshot.state {
3214            ModuleState::Draining if snapshot.draining_to_replace => {
3215                Some(RegistrationEndReason::SupervisorRestart)
3216            }
3217            ModuleState::Draining => Some(RegistrationEndReason::SupervisorStop),
3218            ModuleState::Restarting if snapshot.draining_to_replace => {
3219                Some(RegistrationEndReason::SupervisorRestart)
3220            }
3221            _ => None,
3222        })
3223    }
3224
3225    /// The module's retained stderr, newest lines last.
3226    ///
3227    /// Deliberately NOT on [`Self::status`]: a bounded tail is kilobytes per
3228    /// module, `supervisor.list` renders every module, and putting it in the
3229    /// shared snapshot would make each status read carry a payload almost nobody
3230    /// asked for. Callers that want the text ask for it.
3231    pub fn stderr_tail(
3232        &self,
3233        max_lines: Option<usize>,
3234        max_bytes: Option<usize>,
3235    ) -> StderrTailSnapshot {
3236        self.inner
3237            .stderr_ring
3238            .lock()
3239            .unwrap_or_else(|poisoned| poisoned.into_inner())
3240            .snapshot(max_lines, max_bytes)
3241    }
3242
3243    /// The module's bounded terminal history, oldest retained exit first.
3244    ///
3245    /// The daemon-start stamp distinguishes a quiet supervisor from a replacement
3246    /// daemon whose in-memory history was necessarily reset.
3247    pub fn terminal_history(&self) -> TerminalHistorySnapshot {
3248        self.inner
3249            .terminal_ring
3250            .lock()
3251            .unwrap_or_else(|poisoned| poisoned.into_inner())
3252            .snapshot()
3253    }
3254
3255    /// Retained observations from the current ring and all journal generations.
3256    ///
3257    /// Blocking: this reads the journal files. Async callers use
3258    /// [`Self::read_durable_terminal_history`].
3259    pub fn durable_terminal_history(&self) -> subc_control::TerminalHistory {
3260        durable_terminal_history_of(&self.inner.terminal_ring, &self.inner.module_id)
3261    }
3262
3263    /// [`Self::durable_terminal_history`] on a blocking thread, so the journal
3264    /// read (up to every retained generation) never occupies a runtime worker.
3265    /// Fails only if the blocking task could not finish (runtime shutdown or a
3266    /// panic in the read).
3267    pub(crate) async fn read_durable_terminal_history(
3268        &self,
3269    ) -> Result<subc_control::TerminalHistory, tokio::task::JoinError> {
3270        let terminal_ring = Arc::clone(&self.inner.terminal_ring);
3271        let module_id = self.inner.module_id.clone();
3272        tokio::task::spawn_blocking(move || durable_terminal_history_of(&terminal_ring, &module_id))
3273            .await
3274    }
3275
3276    pub fn status(&self) -> Result<ModuleStatus, SuperviseError> {
3277        self.status_with_snapshot_lock(&self.inner.snapshot, None)
3278            .map(|(status, _)| status)
3279    }
3280
3281    pub(crate) fn record_deliberate_severance(
3282        &self,
3283        identity: ProcessIdentity,
3284    ) -> Result<bool, SuperviseError> {
3285        let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
3286        if snapshot.pid != Some(identity.pid)
3287            || snapshot.process_start_time != Some(identity.start_time)
3288        {
3289            return Ok(false);
3290        }
3291        snapshot.deliberate_severance = Some(identity);
3292        Ok(true)
3293    }
3294
3295    /// Read status for a channel-0 renderer and report a contended snapshot lock.
3296    ///
3297    /// Internal supervision callers use [`Self::status`] so writer-side machinery
3298    /// does not produce reader-observability logs.
3299    pub(crate) fn status_for_control(
3300        &self,
3301        caller: &'static str,
3302    ) -> Result<ModuleStatus, SuperviseError> {
3303        self.status_with_snapshot_lock(&self.inner.snapshot, Some(caller))
3304            .map(|(status, _)| status)
3305    }
3306
3307    fn status_with_snapshot_lock(
3308        &self,
3309        snapshot: &SharedSnapshot,
3310        caller: Option<&'static str>,
3311    ) -> Result<(ModuleStatus, Option<SpawnedFileIdentity>), SuperviseError> {
3312        let mut guard = match caller {
3313            Some(caller) => lock_snapshot_for_control(snapshot, &self.inner.module_id, caller)?,
3314            None => lock_snapshot(snapshot)?,
3315        };
3316        // Read the budget through the pruning path so a reader sees the same
3317        // in-window count the restart decision would use, not a stale total.
3318        let restart_count =
3319            guard.crash_restarts_in_window(self.inner.restart_policy.window, Instant::now());
3320        let snapshot = guard.clone();
3321        drop(guard);
3322        let drain_timeout = *self.inner.effective_drain_timeout.lock().map_err(|_| {
3323            SuperviseError::StatePoisoned {
3324                module_id: Some(self.inner.module_id.clone()),
3325            }
3326        })?;
3327        let registration_active = self
3328            .inner
3329            .registry
3330            .get_module(&self.inner.module_id)
3331            .map_err(SuperviseError::Registry)?
3332            .is_some();
3333        let protocol = snapshot
3334            .spawned_protocol
3335            .unwrap_or(self.declared_protocol()?);
3336        let running_process =
3337            snapshot.enabled && snapshot.state == ModuleState::Running && snapshot.process_alive;
3338        // Registration is the difference between the two protocols and the only
3339        // one: a subc module that has not registered cannot serve a request even
3340        // though its process is up, and a `none` module never registers at all,
3341        // so requiring it there would pin `live` to false for the whole life of
3342        // a perfectly healthy process.
3343        let live = match protocol {
3344            ModuleProtocol::Subc => running_process && registration_active,
3345            ModuleProtocol::None => running_process,
3346        };
3347
3348        Ok((
3349            ModuleStatus {
3350                module_id: self.inner.module_id.clone(),
3351                state: snapshot.state,
3352                enabled: snapshot.enabled,
3353                process_alive: snapshot.process_alive,
3354                registration_active,
3355                protocol,
3356                live,
3357                restart_count,
3358                lifetime_restarts: snapshot.lifetime_restarts,
3359                spawn_generation: snapshot.spawn_generation,
3360                max_restarts: self.inner.restart_policy.max_restarts,
3361                restart_window: self.inner.restart_policy.window,
3362                drain_timeout,
3363                restart_backoff: self.inner.restart_policy.backoff,
3364                restart_max_backoff: self.inner.restart_policy.max_backoff,
3365                pid: snapshot.reported_pid(),
3366                spawned_at_ms: snapshot.spawned_at_ms,
3367                spawned_from: snapshot.spawned_from,
3368                process_start_time: snapshot.process_start_time,
3369                last_exit: snapshot.last_exit,
3370                health: snapshot.health,
3371            },
3372            snapshot.spawned_file_identity,
3373        ))
3374    }
3375
3376    #[cfg(test)]
3377    pub(crate) fn hold_snapshot_for_test(
3378        &self,
3379        acquired: std::sync::mpsc::Sender<()>,
3380        hold: Duration,
3381    ) -> std::thread::JoinHandle<()> {
3382        let snapshot = Arc::clone(&self.inner.snapshot);
3383        std::thread::spawn(move || {
3384            let _guard = snapshot.lock().expect("test snapshot lock is not poisoned");
3385            acquired
3386                .send(())
3387                .expect("test receiver waits for snapshot lock");
3388            std::thread::sleep(hold);
3389        })
3390    }
3391
3392    /// The status and the running-image check for `supervisor.provenance`,
3393    /// taken from one status read. The exec acknowledgement can land between
3394    /// two separate reads, and the reply would then pair "no pid yet" with an
3395    /// image observed after the module started, which describes no single
3396    /// moment.
3397    pub(crate) async fn status_and_running_image_agreement(
3398        &self,
3399    ) -> Result<(ModuleStatus, subc_control::RunningImageAgreement), SuperviseError> {
3400        let (status, identity) = self.status_with_snapshot_lock(&self.inner.snapshot, None)?;
3401        let image = self
3402            .inner
3403            .provenance_probe
3404            .observe(
3405                status.pid,
3406                status.spawned_from.as_deref(),
3407                identity,
3408                status.process_start_time,
3409            )
3410            .await;
3411        Ok((status, image))
3412    }
3413
3414    pub(crate) async fn running_image_agreement(&self) -> subc_control::RunningImageAgreement {
3415        let snapshot = match lock_snapshot(&self.inner.snapshot) {
3416            Ok(snapshot) => snapshot.clone(),
3417            Err(_) => {
3418                return subc_control::RunningImageAgreement::Unavailable {
3419                    reason: subc_control::RunningImageUnavailableReason::NotRunning,
3420                };
3421            }
3422        };
3423        self.inner
3424            .provenance_probe
3425            .observe(
3426                snapshot.reported_pid(),
3427                snapshot.spawned_from.as_deref(),
3428                snapshot.spawned_file_identity,
3429                snapshot.process_start_time,
3430            )
3431            .await
3432    }
3433
3434    /// Memory and CPU time of the module's current process, read now. Only the
3435    /// process the supervisor spawned is read, not processes it has started.
3436    pub(crate) fn child_resource_usage(&self) -> subc_control::ChildResourceUsage {
3437        let (pid, start_time) = match lock_snapshot(&self.inner.snapshot) {
3438            Ok(snapshot) => (snapshot.reported_pid(), snapshot.process_start_time),
3439            Err(_) => {
3440                return subc_control::ChildResourceUsage::Unavailable {
3441                    reason: subc_control::ChildResourceUnavailableReason::Unreadable,
3442                }
3443            }
3444        };
3445        crate::child_resources::read(pid, start_time)
3446    }
3447
3448    pub(crate) fn will_recover_after_connection_loss(&self) -> Result<bool, SuperviseError> {
3449        let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
3450        Ok(match snapshot.state {
3451            ModuleState::Restarting => true,
3452            ModuleState::Failed | ModuleState::Disabled => false,
3453            _ => daemon_will_restart(&mut snapshot, &self.inner.restart_policy, Instant::now()),
3454        })
3455    }
3456
3457    #[cfg(test)]
3458    pub(crate) fn is_warming(&self) -> Result<bool, SuperviseError> {
3459        self.is_warming_with_snapshot_lock(None)
3460    }
3461
3462    pub(crate) fn is_warming_for_control(
3463        &self,
3464        caller: &'static str,
3465    ) -> Result<bool, SuperviseError> {
3466        self.is_warming_with_snapshot_lock(Some(caller))
3467    }
3468
3469    fn is_warming_with_snapshot_lock(
3470        &self,
3471        caller: Option<&'static str>,
3472    ) -> Result<bool, SuperviseError> {
3473        let snapshot = match caller {
3474            Some(caller) => {
3475                lock_snapshot_for_control(&self.inner.snapshot, &self.inner.module_id, caller)?
3476            }
3477            None => lock_snapshot(&self.inner.snapshot)?,
3478        }
3479        .clone();
3480        Ok(matches!(
3481            snapshot.state,
3482            ModuleState::Starting | ModuleState::Running | ModuleState::Restarting
3483        ))
3484    }
3485
3486    /// Drain the module and stop monitoring it.
3487    pub async fn drain(&self) -> Result<(), SuperviseError> {
3488        self.stop().await
3489    }
3490
3491    pub(crate) async fn retire(&self) -> Result<(), SuperviseError> {
3492        match self.state()? {
3493            ModuleState::Stopped | ModuleState::Failed => return Ok(()),
3494            ModuleState::Starting
3495            | ModuleState::Running
3496            | ModuleState::Unresponsive
3497            | ModuleState::Restarting
3498            | ModuleState::Draining
3499            | ModuleState::Disabled => {}
3500        }
3501
3502        let (reply_tx, reply_rx) = oneshot::channel();
3503        self.inner
3504            .commands
3505            .send(SupervisorCommand::Retire { reply: reply_tx })
3506            .await
3507            .map_err(|_| SuperviseError::CommandClosed {
3508                module_id: self.inner.module_id.clone(),
3509            })?;
3510        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3511            module_id: self.inner.module_id.clone(),
3512        })?
3513    }
3514
3515    pub async fn stop(&self) -> Result<(), SuperviseError> {
3516        match self.state()? {
3517            ModuleState::Stopped | ModuleState::Failed => return Ok(()),
3518            ModuleState::Starting
3519            | ModuleState::Running
3520            | ModuleState::Unresponsive
3521            | ModuleState::Restarting
3522            | ModuleState::Draining
3523            | ModuleState::Disabled => {}
3524        }
3525
3526        let (reply_tx, reply_rx) = oneshot::channel();
3527        self.inner
3528            .commands
3529            .send(SupervisorCommand::Drain { reply: reply_tx })
3530            .await
3531            .map_err(|_| SuperviseError::CommandClosed {
3532                module_id: self.inner.module_id.clone(),
3533            })?;
3534        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3535            module_id: self.inner.module_id.clone(),
3536        })?
3537    }
3538
3539    pub async fn restart(&self, drain_timeout_ms: Option<u64>) -> Result<(), SuperviseError> {
3540        let received_at_generation = lock_snapshot(&self.inner.snapshot)?.spawn_generation;
3541        let (reply_tx, reply_rx) = oneshot::channel();
3542        self.inner
3543            .commands
3544            .send(SupervisorCommand::Restart {
3545                drain_timeout_ms,
3546                received_at_generation,
3547                queued_at: Instant::now(),
3548                reply: reply_tx,
3549            })
3550            .await
3551            .map_err(|_| SuperviseError::CommandClosed {
3552                module_id: self.inner.module_id.clone(),
3553            })?;
3554        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3555            module_id: self.inner.module_id.clone(),
3556        })?
3557    }
3558
3559    /// Blue/green restart: see [`SupervisorCommand::Swap`] and the
3560    /// `supervisor_swap` module. Returns once the swap has cut over (the old
3561    /// process then drains in the background of the supervise loop) or has
3562    /// failed, leaving the old process serving.
3563    pub async fn swap(&self, ready_timeout: Option<Duration>) -> Result<(), SuperviseError> {
3564        let (reply_tx, reply_rx) = oneshot::channel();
3565        self.inner
3566            .commands
3567            .send(SupervisorCommand::Swap {
3568                ready_timeout,
3569                reply: reply_tx,
3570            })
3571            .await
3572            .map_err(|_| SuperviseError::CommandClosed {
3573                module_id: self.inner.module_id.clone(),
3574            })?;
3575        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3576            module_id: self.inner.module_id.clone(),
3577        })?
3578    }
3579
3580    pub async fn reload(&self) -> Result<(), SuperviseError> {
3581        let (reply_tx, reply_rx) = oneshot::channel();
3582        self.inner
3583            .commands
3584            .send(SupervisorCommand::Reload { reply: reply_tx })
3585            .await
3586            .map_err(|_| SuperviseError::CommandClosed {
3587                module_id: self.inner.module_id.clone(),
3588            })?;
3589        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3590            module_id: self.inner.module_id.clone(),
3591        })?
3592    }
3593
3594    pub async fn set_enabled(&self, enabled: bool) -> Result<bool, SuperviseError> {
3595        let (reply_tx, reply_rx) = oneshot::channel();
3596        self.inner
3597            .commands
3598            .send(SupervisorCommand::SetEnabled {
3599                enabled,
3600                reply: reply_tx,
3601            })
3602            .await
3603            .map_err(|_| SuperviseError::CommandClosed {
3604                module_id: self.inner.module_id.clone(),
3605            })?;
3606        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3607            module_id: self.inner.module_id.clone(),
3608        })?
3609    }
3610
3611    /// The current process's protocol, or the configured protocol when down.
3612    /// A rescan stores the next launch spec without changing how an existing
3613    /// process registers, serves routes, is probed, or exits.
3614    pub(crate) fn declared_protocol(&self) -> Result<ModuleProtocol, SuperviseError> {
3615        let configured = self
3616            .inner
3617            .configuration
3618            .lock()
3619            .map_err(|_| SuperviseError::StatePoisoned {
3620                module_id: Some(self.inner.module_id.clone()),
3621            })?
3622            .spec
3623            .protocol;
3624        let state = lock_snapshot(&self.inner.snapshot)?;
3625        Ok(state.spawned_protocol.unwrap_or(configured))
3626    }
3627
3628    pub(crate) fn configuration(&self) -> Result<(ModuleSpec, HealthConfig), SuperviseError> {
3629        let configuration =
3630            self.inner
3631                .configuration
3632                .lock()
3633                .map_err(|_| SuperviseError::StatePoisoned {
3634                    module_id: Some(self.inner.module_id.clone()),
3635                })?;
3636        Ok((configuration.spec.clone(), configuration.health.clone()))
3637    }
3638
3639    /// Replace this module's launch spec, keeping its health and drain policy,
3640    /// the way a rescan does for a changed config entry. The running process is
3641    /// untouched; the next spawn (a restart, or a swap's candidate) uses it.
3642    #[cfg(any(test, feature = "test-support"))]
3643    pub async fn update_spec_for_test(&self, spec: ModuleSpec) -> Result<(), SuperviseError> {
3644        let (_, health) = self.configuration()?;
3645        let drain_timeout_ms = u64::try_from(
3646            self.inner
3647                .effective_drain_timeout
3648                .lock()
3649                .unwrap_or_else(|poisoned| poisoned.into_inner())
3650                .as_millis(),
3651        )
3652        .ok();
3653        self.update_configuration(spec, health, drain_timeout_ms)
3654            .await
3655    }
3656
3657    pub(crate) async fn update_configuration(
3658        &self,
3659        spec: ModuleSpec,
3660        health: HealthConfig,
3661        drain_timeout_ms: Option<u64>,
3662    ) -> Result<(), SuperviseError> {
3663        if spec.module_id != self.inner.module_id {
3664            return Err(SuperviseError::InvalidSpec {
3665                reason: "a supervised module's module_id cannot be changed".to_string(),
3666            });
3667        }
3668        validate_spec(&spec)?;
3669        let (reply_tx, reply_rx) = oneshot::channel();
3670        self.inner
3671            .commands
3672            .send(SupervisorCommand::UpdateConfiguration {
3673                spec: spec.clone(),
3674                health: health.clone(),
3675                drain_timeout_ms,
3676                reply: reply_tx,
3677            })
3678            .await
3679            .map_err(|_| SuperviseError::CommandClosed {
3680                module_id: self.inner.module_id.clone(),
3681            })?;
3682        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3683            module_id: self.inner.module_id.clone(),
3684        })?;
3685        let mut configuration =
3686            self.inner
3687                .configuration
3688                .lock()
3689                .map_err(|_| SuperviseError::StatePoisoned {
3690                    module_id: Some(self.inner.module_id.clone()),
3691                })?;
3692        configuration.spec = spec;
3693        configuration.health = health;
3694        Ok(())
3695    }
3696}
3697
3698impl Drop for SupervisedModuleInner {
3699    fn drop(&mut self) {
3700        let Ok(mut monitor) = self.monitor.lock() else {
3701            return;
3702        };
3703        if let Some(monitor) = monitor.as_ref().filter(|monitor| !monitor.is_finished()) {
3704            let _ = update_snapshot(&self.snapshot, Some(&self.module_id), |state| {
3705                state.state = ModuleState::Stopped;
3706                clear_current_process_facts(state);
3707            });
3708            monitor.abort();
3709        }
3710        let _ = monitor.take();
3711    }
3712}
3713
3714#[derive(Debug)]
3715enum SupervisorCommand {
3716    Drain {
3717        reply: oneshot::Sender<Result<(), SuperviseError>>,
3718    },
3719    Retire {
3720        reply: oneshot::Sender<Result<(), SuperviseError>>,
3721    },
3722    Restart {
3723        /// Operator override for this one restart's drain budget, in ms. `None`
3724        /// uses the module's configured/default budget; `Some(0)` cuts
3725        /// immediately (wedge bounce: a stuck request never settles, so
3726        /// waiting only delays recovery).
3727        drain_timeout_ms: Option<u64>,
3728        /// The module's `spawn_generation` when the request was received, before
3729        /// it waited in the command queue. A queued restart whose module has
3730        /// since spawned a newer process is already satisfied (see the handler).
3731        received_at_generation: u64,
3732        /// When the request entered the command queue, so the handler can log
3733        /// how long it waited behind the loop's other work.
3734        queued_at: Instant,
3735        reply: oneshot::Sender<Result<(), SuperviseError>>,
3736    },
3737    Reload {
3738        reply: oneshot::Sender<Result<(), SuperviseError>>,
3739    },
3740    SetEnabled {
3741        enabled: bool,
3742        reply: oneshot::Sender<Result<bool, SuperviseError>>,
3743    },
3744    UpdateConfiguration {
3745        spec: ModuleSpec,
3746        health: HealthConfig,
3747        /// Per-module drain override from the new config; `None` re-resolves to
3748        /// the supervisor-wide default.
3749        drain_timeout_ms: Option<u64>,
3750        reply: oneshot::Sender<()>,
3751    },
3752    Swap {
3753        /// How long the candidate may take to register and declare itself
3754        /// ready. `None` uses [`DEFAULT_SWAP_READY_TIMEOUT`].
3755        ready_timeout: Option<Duration>,
3756        /// Answered at cutover or failure; the incumbent's drain follows.
3757        reply: oneshot::Sender<Result<(), SuperviseError>>,
3758    },
3759}
3760
3761#[derive(Debug)]
3762pub enum SuperviseError {
3763    InvalidSpec {
3764        reason: String,
3765    },
3766    Spawn {
3767        program: PathBuf,
3768        source: io::Error,
3769        cgroup_path: Option<PathBuf>,
3770    },
3771    Cgroup {
3772        module_id: String,
3773        source: io::Error,
3774    },
3775    /// CSPRNG failure generating a reserved module's launch nonce. Fail loud rather
3776    /// than spawn a reserved module without its identity binding.
3777    LaunchNonce {
3778        reason: String,
3779    },
3780    Wait {
3781        module_id: String,
3782        source: io::Error,
3783    },
3784    Kill {
3785        module_id: String,
3786        source: io::Error,
3787    },
3788    Forwarding(ForwardingError),
3789    Registry(RegistryError),
3790    ReloadUnavailable {
3791        module_id: String,
3792        reason: String,
3793    },
3794    /// An operator restart/reload was requested for a module that is currently
3795    /// disabled. Restart/reload cycle a *running* module; a disabled module must
3796    /// be explicitly re-enabled (set_enabled(true)) rather than silently started
3797    /// by a restart, so these commands are rejected instead of re-enabling it.
3798    Disabled {
3799        module_id: String,
3800    },
3801    ReloadFailed {
3802        module_id: String,
3803        reason: String,
3804    },
3805    RegistrationStillActive {
3806        module_id: String,
3807        waited: Duration,
3808    },
3809    StatePoisoned {
3810        module_id: Option<String>,
3811    },
3812    CommandClosed {
3813        module_id: String,
3814    },
3815    /// A restart or reload arrived while a swap's candidate was warming. The
3816    /// swap owns the module until it cuts over or fails; a stop or disable
3817    /// would have aborted it instead.
3818    SwapInProgress {
3819        module_id: String,
3820    },
3821    /// A swap was refused before anything was spawned.
3822    SwapRefused {
3823        module_id: String,
3824        reason: SwapRefusal,
3825    },
3826    /// A swap spawned a candidate and gave up on it. The candidate has been
3827    /// killed and its slot freed; the incumbent was left serving and was never
3828    /// drained, except in the one `CutoverLost` case described on that arm.
3829    SwapFailed {
3830        module_id: String,
3831        arm: SwapFailureArm,
3832        detail: String,
3833        /// How the candidate exited, when it exited on its own before the
3834        /// supervisor gave up on it.
3835        candidate_exit: Option<ExitReport>,
3836    },
3837}
3838
3839/// Why a swap was refused before a candidate was spawned.
3840#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3841pub enum SwapRefusal {
3842    /// The module's config does not declare `overlap: "safe"`.
3843    OverlapExclusive,
3844    /// The module is not registered, so there is no incumbent to keep serving
3845    /// and nothing a swap would improve on; a plain restart is the tool.
3846    NotRegistered,
3847    /// The module does not speak the subc wire, so a candidate could never
3848    /// register or declare itself ready.
3849    ProtocolNone,
3850    /// The supervisor lacks the forwarding table (to cut routes over) or the
3851    /// shared handle (to admit the candidate's HELLO) that a swap needs.
3852    NotConfigured,
3853    /// A swap is already open for this module.
3854    AlreadySwapping,
3855}
3856
3857impl SwapRefusal {
3858    pub fn as_str(self) -> &'static str {
3859        match self {
3860            Self::OverlapExclusive => "overlap_exclusive",
3861            Self::NotRegistered => "not_registered",
3862            Self::ProtocolNone => "protocol_none",
3863            Self::NotConfigured => "not_configured",
3864            Self::AlreadySwapping => "already_swapping",
3865        }
3866    }
3867}
3868
3869/// Which failure arm ended a swap. Every arm but one leaves the incumbent
3870/// serving and undrained; see `CutoverLost`.
3871#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3872pub enum SwapFailureArm {
3873    /// The candidate process could not be started.
3874    SpawnFailed,
3875    /// The candidate did not register within the readiness budget.
3876    NeverRegistered,
3877    /// The candidate registered but did not declare itself ready in time.
3878    NeverReady,
3879    /// The candidate exited before cutover.
3880    CandidateExited,
3881    /// The candidate declared itself ready but failed its health probe.
3882    CandidateUnhealthy,
3883    /// An operator stop, disable or retire arrived while the candidate warmed.
3884    /// The candidate was killed and the operator's command then carried out on
3885    /// the incumbent.
3886    Interrupted,
3887    /// The candidate's connection closed at the moment of cutover. If it
3888    /// closed before forwarding moved, the incumbent is untouched. If it closed
3889    /// between the forwarding and registry halves of cutover, forwarding can no
3890    /// longer route to the incumbent, so the module is restarted plainly.
3891    CutoverLost,
3892}
3893
3894impl SwapFailureArm {
3895    pub fn as_str(self) -> &'static str {
3896        match self {
3897            Self::SpawnFailed => "spawn_failed",
3898            Self::NeverRegistered => "never_registered",
3899            Self::NeverReady => "never_ready",
3900            Self::CandidateExited => "candidate_exited",
3901            Self::CandidateUnhealthy => "candidate_unhealthy",
3902            Self::Interrupted => "interrupted",
3903            Self::CutoverLost => "cutover_lost",
3904        }
3905    }
3906}
3907
3908impl fmt::Display for SuperviseError {
3909    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3910        match self {
3911            Self::InvalidSpec { reason } => write!(f, "invalid module spec: {reason}"),
3912            Self::Spawn {
3913                program,
3914                source,
3915                cgroup_path: Some(cgroup_path),
3916            } => write!(
3917                f,
3918                "failed to place module in cgroup '{}' while spawning '{}': {source}",
3919                cgroup_path.display(),
3920                program.display()
3921            ),
3922            Self::Spawn {
3923                program,
3924                source,
3925                cgroup_path: None,
3926            } => write!(
3927                f,
3928                "failed to spawn module '{}': {source}",
3929                program.display()
3930            ),
3931            Self::Cgroup { module_id, source } => {
3932                write!(
3933                    f,
3934                    "failed to prepare cgroup for module '{module_id}': {source}"
3935                )
3936            }
3937            Self::LaunchNonce { reason } => {
3938                write!(
3939                    f,
3940                    "failed to generate reserved-module launch nonce: {reason}"
3941                )
3942            }
3943            Self::Wait { module_id, source } => {
3944                write!(f, "failed to wait for module '{module_id}': {source}")
3945            }
3946            Self::Kill { module_id, source } => {
3947                write!(f, "failed to kill module '{module_id}': {source}")
3948            }
3949            Self::Forwarding(err) => write!(f, "forwarding error: {err}"),
3950            Self::Registry(err) => write!(f, "registry error: {err}"),
3951            Self::ReloadUnavailable { module_id, reason } => {
3952                write!(f, "reload unavailable for module '{module_id}': {reason}")
3953            }
3954            Self::Disabled { module_id } => {
3955                write!(
3956                    f,
3957                    "module '{module_id}' is disabled; enable it before restart or reload"
3958                )
3959            }
3960            Self::ReloadFailed { module_id, reason } => {
3961                write!(f, "reload failed for module '{module_id}': {reason}")
3962            }
3963            Self::RegistrationStillActive { module_id, waited } => write!(
3964                f,
3965                "module '{module_id}' registration remained active after waiting {waited:?}"
3966            ),
3967            Self::StatePoisoned { module_id } => match module_id {
3968                Some(module_id) => {
3969                    write!(f, "supervisor state for module '{module_id}' was poisoned")
3970                }
3971                None => write!(f, "supervisor state was poisoned"),
3972            },
3973            Self::CommandClosed { module_id } => {
3974                write!(
3975                    f,
3976                    "supervisor command channel for module '{module_id}' is closed"
3977                )
3978            }
3979            Self::SwapInProgress { module_id } => write!(
3980                f,
3981                "module '{module_id}' is being swapped; retry once the swap has cut over or failed, or stop the module to abort the swap"
3982            ),
3983            Self::SwapRefused { module_id, reason } => match reason {
3984                SwapRefusal::OverlapExclusive => write!(
3985                    f,
3986                    "module '{module_id}' is declared overlap: \"exclusive\" (the default): two processes of it must not run at once, so it cannot be swapped; use a plain restart, or declare overlap: \"safe\" in its config if it really tolerates a second process"
3987                ),
3988                SwapRefusal::NotRegistered => write!(
3989                    f,
3990                    "module '{module_id}' is not registered, so there is no serving process to keep while a replacement warms; use a plain restart"
3991                ),
3992                SwapRefusal::ProtocolNone => write!(
3993                    f,
3994                    "module '{module_id}' is protocol: \"none\" and never registers, so a swap could never see its replacement become ready; use a plain restart"
3995                ),
3996                SwapRefusal::NotConfigured => write!(
3997                    f,
3998                    "module '{module_id}' cannot be swapped: the supervisor was built without the forwarding table or shared handle a swap needs"
3999                ),
4000                SwapRefusal::AlreadySwapping => {
4001                    write!(f, "module '{module_id}' is already being swapped")
4002                }
4003            },
4004            Self::SwapFailed {
4005                module_id,
4006                arm,
4007                detail,
4008                ..
4009            } => write!(
4010                f,
4011                "swap of module '{module_id}' failed ({}): {detail}; the running process was left serving",
4012                arm.as_str()
4013            ),
4014        }
4015    }
4016}
4017
4018impl Error for SuperviseError {
4019    fn source(&self) -> Option<&(dyn Error + 'static)> {
4020        match self {
4021            Self::Spawn { source, .. }
4022            | Self::Cgroup { source, .. }
4023            | Self::Wait { source, .. }
4024            | Self::Kill { source, .. } => Some(source),
4025            Self::Forwarding(err) => Some(err),
4026            Self::Registry(err) => Some(err),
4027            Self::LaunchNonce { .. }
4028            | Self::InvalidSpec { .. }
4029            | Self::ReloadUnavailable { .. }
4030            | Self::Disabled { .. }
4031            | Self::ReloadFailed { .. }
4032            | Self::RegistrationStillActive { .. }
4033            | Self::StatePoisoned { .. }
4034            | Self::CommandClosed { .. }
4035            | Self::SwapInProgress { .. }
4036            | Self::SwapRefused { .. }
4037            | Self::SwapFailed { .. } => None,
4038        }
4039    }
4040}
4041
4042pub(crate) fn validate_spec(spec: &ModuleSpec) -> Result<(), SuperviseError> {
4043    if spec.module_id.trim().is_empty() {
4044        return Err(SuperviseError::InvalidSpec {
4045            reason: "module_id must not be empty".to_string(),
4046        });
4047    }
4048
4049    Ok(())
4050}
4051
4052#[derive(Debug, Default)]
4053struct HealthProbeRuntime {
4054    configured_health: Option<HealthConfig>,
4055    registered_connection: Option<crate::ConnectionId>,
4056    advertised: bool,
4057    next_probe_at: Option<Instant>,
4058    probe_index: u64,
4059}
4060
4061fn running_protocol(spec: &ModuleSpec, snapshot: &SharedSnapshot) -> ModuleProtocol {
4062    lock_snapshot(snapshot)
4063        .ok()
4064        .and_then(|state| state.spawned_protocol)
4065        .unwrap_or(spec.protocol)
4066}
4067
4068impl HealthProbeRuntime {
4069    fn refresh_registration(
4070        &mut self,
4071        spec: &ModuleSpec,
4072        runtime: &SupervisorRuntimeConfig,
4073        registry: &Registry,
4074        snapshot: &SharedSnapshot,
4075    ) {
4076        if self.configured_health.as_ref() != Some(&runtime.health) {
4077            self.configured_health = Some(runtime.health.clone());
4078            self.next_probe_at = None;
4079            self.registered_connection = None;
4080            self.probe_index = 0;
4081        }
4082        // A non-wire process never registers. Only an explicitly configured
4083        // HTTP endpoint can arm its health probe; an absent HELLO is not a
4084        // health failure for that kind of process.
4085        if running_protocol(spec, snapshot) == ModuleProtocol::None {
4086            self.registered_connection = None;
4087            self.advertised = runtime.health.http.is_some();
4088            if !self.advertised {
4089                self.next_probe_at = None;
4090                let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4091                    let unknown = ModuleHealthStatus::default();
4092                    if state.health != unknown {
4093                        state.health = unknown;
4094                    }
4095                });
4096            } else if self.next_probe_at.is_none() {
4097                self.next_probe_at = Some(
4098                    Instant::now()
4099                        + jittered_health_delay(&spec.module_id, 0, runtime.health.cadence),
4100                );
4101            }
4102            return;
4103        }
4104
4105        let registration = match registry.get_module(&spec.module_id) {
4106            Ok(registration) => registration,
4107            Err(err) => {
4108                warn!(module_id = %spec.module_id, error = %err, "health prober could not read registry");
4109                self.advertised = false;
4110                self.next_probe_at = None;
4111                return;
4112            }
4113        };
4114
4115        let Some(registration) = registration else {
4116            self.registered_connection = None;
4117            self.advertised = false;
4118            self.next_probe_at = None;
4119            return;
4120        };
4121
4122        let advertised = registration
4123            .control_ops
4124            .iter()
4125            .any(|op| op == MODULE_CONTROL_OP_HEALTH_CHECK);
4126        if !advertised {
4127            self.registered_connection = Some(registration.connection_id);
4128            self.advertised = false;
4129            self.next_probe_at = None;
4130            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4131                if state.health.status != SupervisorHealthStatus::Unknown
4132                    || state.health.consecutive_failures != 0
4133                    || state.health.last_probe_ms.is_some()
4134                    || state.health.detail.is_some()
4135                    || state.health.metrics.is_some()
4136                {
4137                    state.health.status = SupervisorHealthStatus::Unknown;
4138                    state.health.consecutive_failures = 0;
4139                    state.health.last_probe_ms = None;
4140                    state.health.detail = None;
4141                    state.health.metrics = None;
4142                }
4143            });
4144            return;
4145        }
4146
4147        let reregistered = self.registered_connection != Some(registration.connection_id);
4148        self.registered_connection = Some(registration.connection_id);
4149        self.advertised = true;
4150        if reregistered || self.next_probe_at.is_none() {
4151            self.probe_index = 0;
4152            self.next_probe_at = Some(
4153                Instant::now() + jittered_health_delay(&spec.module_id, 0, runtime.health.cadence),
4154            );
4155            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4156                state.health.status = SupervisorHealthStatus::Unknown;
4157                state.health.consecutive_failures = 0;
4158                state.health.detail = None;
4159                state.health.metrics = None;
4160            });
4161        }
4162    }
4163
4164    fn wake_after(&self) -> Option<Duration> {
4165        if !self.advertised {
4166            return None;
4167        }
4168        self.next_probe_at
4169            .map(|next| next.saturating_duration_since(Instant::now()))
4170    }
4171
4172    fn due(&self) -> bool {
4173        self.advertised
4174            && self
4175                .next_probe_at
4176                .is_some_and(|next| Instant::now() >= next)
4177    }
4178
4179    fn schedule_next(&mut self, spec: &ModuleSpec, cadence: Duration) {
4180        self.probe_index = self.probe_index.wrapping_add(1);
4181        self.next_probe_at = Some(
4182            Instant::now() + jittered_health_delay(&spec.module_id, self.probe_index, cadence),
4183        );
4184    }
4185}
4186
4187/// What a failed health probe actually OBSERVED, kept apart from how it reads.
4188///
4189/// This was a struct with a single `message: String`, and every one of the
4190/// fifteen construction sites collapsed into it. Each site knows exactly what it
4191/// saw -- the lane is gone, the module did not answer in time, the module
4192/// answered with the wrong thing -- and `handle_health_probe_failure` then
4193/// treated all of them identically: increment a counter, compare to a threshold,
4194/// restart the module. THE DISTINCTION EXISTED AT EVERY CALL SITE AND WAS
4195/// DESTROYED BEFORE THE DECISION THAT NEEDED IT.
4196///
4197/// The distinction that matters is not severity, it is EVIDENTIAL WEIGHT:
4198///
4199/// * `LaneDead` is PROOF. The module's control connection is gone; nothing will
4200///   answer on it again.
4201/// * `NoAnswer` is ABSENCE OF EVIDENCE. It is consistent with a wedged module
4202///   AND with a perfectly healthy one that lost a CPU race -- which is what
4203///   happens under machine load, and is how this supervisor killed a healthy
4204///   module three times in one day.
4205/// * `BadAnswer` proves the module is ALIVE. It replied; the reply was wrong.
4206///   Restarting on it is defensible, but it is not the silence case and should
4207///   never be counted as one.
4208/// * `Misconfigured` is a daemon-side fault. The module has not been asked
4209///   anything, so it cannot be evidence about the module at all.
4210///
4211/// The asymmetry is the whole point: under saturation the WEAKEST signal is the
4212/// one that fires most often, and while every variant collapsed into one string
4213/// it carried the same weight as the strongest.
4214///
4215/// LIVE BEHAVIOUR TODAY, stated here because this doc block describes the
4216/// DESIGN and a reader stopping at it gets the build backwards: the restart
4217/// decision does NOT yet consult this classification -- consecutive `NoAnswer`
4218/// probes still increment the failure streak and drive escalation at the
4219/// threshold (see `is_proof_of_death` below for why that is deliberate and
4220/// what gates the change). Absence of evidence restarts modules today.
4221#[derive(Debug)]
4222enum HealthProbeEvidence {
4223    /// The module's control lane is gone. Proof of death.
4224    LaneDead,
4225    /// No reply within the deadline. Proves nothing about the module's state.
4226    NoAnswer,
4227    /// The module replied, but not with a usable health report. Proves it is alive.
4228    BadAnswer,
4229    /// The daemon could not ask. Says nothing about the module.
4230    Misconfigured,
4231}
4232
4233#[derive(Debug)]
4234struct HealthProbeError {
4235    evidence: HealthProbeEvidence,
4236    message: String,
4237}
4238
4239impl HealthProbeError {
4240    fn lane_dead(message: impl Into<String>) -> Self {
4241        Self::with(HealthProbeEvidence::LaneDead, message)
4242    }
4243
4244    fn no_answer(message: impl Into<String>) -> Self {
4245        Self::with(HealthProbeEvidence::NoAnswer, message)
4246    }
4247
4248    fn bad_answer(message: impl Into<String>) -> Self {
4249        Self::with(HealthProbeEvidence::BadAnswer, message)
4250    }
4251
4252    fn misconfigured(message: impl Into<String>) -> Self {
4253        Self::with(HealthProbeEvidence::Misconfigured, message)
4254    }
4255
4256    fn with(evidence: HealthProbeEvidence, message: impl Into<String>) -> Self {
4257        Self {
4258            evidence,
4259            message: message.into(),
4260        }
4261    }
4262
4263    /// Whether this observation is proof the module cannot serve.
4264    ///
4265    /// Only `LaneDead` qualifies. `NoAnswer` is deliberately excluded: it is the
4266    /// variant that fires under CPU starvation, and treating it as proof is the
4267    /// defect this enum exists to make impossible to reintroduce silently.
4268    ///
4269    /// NOT YET CONSULTED BY THE RESTART DECISION, deliberately. Requiring proof
4270    /// to restart also needs a bound for the case it excludes -- a genuinely
4271    /// wedged module, alive but never answering -- and that bound must come from
4272    /// the distribution of real late-answer latencies, which nothing measures
4273    /// yet. Landing the classification first makes the later change a one-line
4274    /// decision against evidence that already exists, rather than two unproven
4275    /// changes at once.
4276    #[allow(dead_code)]
4277    fn is_proof_of_death(&self) -> bool {
4278        matches!(self.evidence, HealthProbeEvidence::LaneDead)
4279    }
4280
4281    /// Short stable label for logs and the health snapshot.
4282    ///
4283    /// An operator reading `ck health` currently cannot tell "the module is gone"
4284    /// from "the module did not answer in five seconds", because both render as
4285    /// prose in the same field. These labels are what make the two
4286    /// distinguishable at a glance, and they are what a later restart-policy
4287    /// change will be argued from.
4288    fn label(&self) -> &'static str {
4289        match self.evidence {
4290            HealthProbeEvidence::LaneDead => "lane-dead",
4291            HealthProbeEvidence::NoAnswer => "no-answer",
4292            HealthProbeEvidence::BadAnswer => "bad-answer",
4293            HealthProbeEvidence::Misconfigured => "daemon-misconfigured",
4294        }
4295    }
4296}
4297
4298impl fmt::Display for HealthProbeError {
4299    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
4300        f.write_str(&self.message)
4301    }
4302}
4303
4304async fn run_health_probe_cycle(
4305    spec: &ModuleSpec,
4306    runtime: &SupervisorRuntimeConfig,
4307    registry: &Registry,
4308    process_liveness: &SupervisorProcessLiveness,
4309    snapshot: &SharedSnapshot,
4310    child: &mut Option<SupervisedChild>,
4311) {
4312    let now_ms = unix_ms_now();
4313    let http = (running_protocol(spec, snapshot) == ModuleProtocol::None)
4314        .then_some(runtime.health.http.as_deref())
4315        .flatten();
4316    let result = match http {
4317        Some(url) => probe_http_health(url, runtime.health.deadline).await,
4318        None => probe_module_health(&spec.module_id, runtime, None).await,
4319    };
4320    match result {
4321        Ok(report) => {
4322            handle_health_report(
4323                spec,
4324                runtime,
4325                registry,
4326                process_liveness,
4327                snapshot,
4328                child,
4329                report,
4330                now_ms,
4331            )
4332            .await;
4333        }
4334        Err(err) => {
4335            if http.is_some() {
4336                let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4337                    state.health.status = SupervisorHealthStatus::Failing;
4338                });
4339            }
4340            handle_health_probe_failure(
4341                spec,
4342                runtime,
4343                registry,
4344                process_liveness,
4345                snapshot,
4346                child,
4347                err,
4348                now_ms,
4349            )
4350            .await;
4351        }
4352    }
4353}
4354
4355pub(crate) struct HttpProbeTarget<'a> {
4356    address: std::net::SocketAddr,
4357    localhost: bool,
4358    authority: &'a str,
4359    path: String,
4360}
4361
4362/// Resolve only literal loopback endpoints, without DNS, redirects, proxies,
4363/// or TLS. A URL cannot turn a local health check into an outbound connection.
4364pub(crate) fn parse_http_probe_url(url: &str) -> Result<HttpProbeTarget<'_>, String> {
4365    if url.bytes().any(|b| b <= b' ' || b == 127) || url.contains('#') {
4366        return Err("must not contain whitespace, controls, or a fragment".into());
4367    }
4368    let rest = url
4369        .strip_prefix("http://")
4370        .ok_or("must use plain http://")?;
4371    let split = rest.find(['/', '?']).unwrap_or(rest.len());
4372    let (authority, suffix) = rest.split_at(split);
4373    let (host, port) = if let Some(rest) = authority.strip_prefix("[::1]") {
4374        ("::1", rest)
4375    } else {
4376        let split = authority.find(':').unwrap_or(authority.len());
4377        authority.split_at(split)
4378    };
4379    let ip: std::net::IpAddr = match host {
4380        "127.0.0.1" | "localhost" => std::net::Ipv4Addr::LOCALHOST.into(),
4381        "::1" => std::net::Ipv6Addr::LOCALHOST.into(),
4382        _ => return Err("host must be 127.0.0.1, [::1], or localhost".into()),
4383    };
4384    let port = if port.is_empty() {
4385        80
4386    } else {
4387        port.strip_prefix(':')
4388            .and_then(|p| p.parse::<u16>().ok())
4389            .filter(|p| *p > 0)
4390            .ok_or("must have a valid nonzero TCP port")?
4391    };
4392    let path = if suffix.is_empty() {
4393        "/".into()
4394    } else if suffix.starts_with('?') {
4395        format!("/{suffix}")
4396    } else {
4397        suffix.into()
4398    };
4399    Ok(HttpProbeTarget {
4400        address: std::net::SocketAddr::new(ip, port),
4401        localhost: host == "localhost",
4402        authority,
4403        path,
4404    })
4405}
4406
4407async fn probe_http_health(
4408    url: &str,
4409    deadline: Duration,
4410) -> Result<HealthReport, HealthProbeError> {
4411    use tokio::io::{AsyncReadExt, AsyncWriteExt, BufReader};
4412    let target = parse_http_probe_url(url).map_err(HealthProbeError::misconfigured)?;
4413    // Keep partial diagnostics outside the timed future so cancellation does
4414    // not discard a status line or body bytes already received.
4415    let mut response_status = String::new();
4416    let mut body = Vec::new();
4417    let probe = async {
4418        // Resolve localhost ourselves so a hosts-file override cannot turn
4419        // this into an outbound request, while IPv6-only local servers work.
4420        let connection = match tokio::net::TcpStream::connect(target.address).await {
4421            Err(_) if target.localhost => {
4422                tokio::net::TcpStream::connect((
4423                    std::net::Ipv6Addr::LOCALHOST,
4424                    target.address.port(),
4425                ))
4426                .await
4427            }
4428            result => result,
4429        };
4430        let mut stream = connection.map_err(|error| {
4431            HealthProbeError::no_answer(format!("HTTP connect failed: {error}"))
4432        })?;
4433        stream.write_all(format!("GET {} HTTP/1.1\r\nHost: {}\r\nAccept-Encoding: identity\r\nConnection: close\r\n\r\n", target.path, target.authority).as_bytes()).await
4434            .map_err(|error| HealthProbeError::no_answer(format!("HTTP write failed: {error}")))?;
4435        let mut reader = BufReader::new(stream);
4436        let mut budget = 16 * 1024;
4437        let status = http_line(&mut reader, &mut budget).await?;
4438        let mut words = status.split_ascii_whitespace();
4439        let version = words.next();
4440        let code = words
4441            .next()
4442            .filter(|word| word.len() == 3)
4443            .and_then(|word| word.parse::<u16>().ok());
4444        if !matches!(version, Some("HTTP/1.1" | "HTTP/1.0"))
4445            || !code.is_some_and(|code| (100..600).contains(&code))
4446        {
4447            return Err(HealthProbeError::bad_answer(format!(
4448                "invalid HTTP status: {status}"
4449            )));
4450        }
4451        let code = code.expect("validated status code");
4452        response_status = status.clone();
4453        let mut length = None;
4454        let mut chunked = false;
4455        loop {
4456            let line = http_line(&mut reader, &mut budget).await?;
4457            if line.is_empty() {
4458                break;
4459            }
4460            if let Some((name, value)) = line.split_once(':') {
4461                if name.eq_ignore_ascii_case("content-length") {
4462                    length = Some(value.trim().parse::<u64>().map_err(|_| {
4463                        HealthProbeError::bad_answer("invalid HTTP Content-Length")
4464                    })?);
4465                } else if name.eq_ignore_ascii_case("transfer-encoding") {
4466                    chunked = value.trim().eq_ignore_ascii_case("chunked");
4467                }
4468            }
4469        }
4470        if chunked {
4471            while body.len() < 200 {
4472                let line = http_line(&mut reader, &mut budget).await?;
4473                let size = u64::from_str_radix(line.split(';').next().unwrap_or("").trim(), 16)
4474                    .map_err(|_| HealthProbeError::bad_answer("invalid HTTP chunk size"))?;
4475                if size == 0 {
4476                    break;
4477                }
4478                let count = size.min((200 - body.len()) as u64) as usize;
4479                let start = body.len();
4480                (&mut reader)
4481                    .take(count as u64)
4482                    .read_to_end(&mut body)
4483                    .await
4484                    .map_err(|error| {
4485                        HealthProbeError::no_answer(format!("HTTP body read failed: {error}"))
4486                    })?;
4487                if body.len() - start != count {
4488                    return Err(HealthProbeError::bad_answer("truncated HTTP chunk"));
4489                }
4490                if size > count as u64 || body.len() == 200 {
4491                    break;
4492                }
4493                if !http_line(&mut reader, &mut budget).await?.is_empty() {
4494                    return Err(HealthProbeError::bad_answer("invalid HTTP chunk delimiter"));
4495                }
4496            }
4497        } else {
4498            reader
4499                .take(length.unwrap_or(200).min(200))
4500                .read_to_end(&mut body)
4501                .await
4502                .map_err(|error| {
4503                    HealthProbeError::no_answer(format!("HTTP body read failed: {error}"))
4504                })?;
4505        }
4506        if (200..300).contains(&code) {
4507            Ok(HealthReport::ok())
4508        } else {
4509            Err(HealthProbeError::bad_answer(
4510                "HTTP health endpoint returned non-2xx",
4511            ))
4512        }
4513    };
4514    let mut result = timeout(deadline, probe).await.unwrap_or_else(|_| {
4515        Err(HealthProbeError::no_answer(format!(
4516            "HTTP probe timed out after {deadline:?}"
4517        )))
4518    });
4519    if let Err(error) = &mut result {
4520        if !response_status.is_empty() {
4521            error.message = format!(
4522                "{}; {response_status}: {}",
4523                error.message,
4524                String::from_utf8_lossy(&body)
4525            );
4526        }
4527    }
4528    result
4529}
4530
4531async fn http_line(
4532    reader: &mut tokio::io::BufReader<tokio::net::TcpStream>,
4533    remaining: &mut usize,
4534) -> Result<String, HealthProbeError> {
4535    use tokio::io::{AsyncBufReadExt, AsyncReadExt};
4536    let mut line = Vec::new();
4537    (&mut *reader)
4538        .take(*remaining as u64)
4539        .read_until(b'\n', &mut line)
4540        .await
4541        .map_err(|error| {
4542            HealthProbeError::no_answer(format!("HTTP header read failed: {error}"))
4543        })?;
4544    *remaining -= line.len();
4545    if !line.ends_with(b"\r\n") {
4546        return Err(HealthProbeError::bad_answer(
4547            "HTTP headers are incomplete or exceed 16 KiB",
4548        ));
4549    }
4550    line.truncate(line.len() - 2);
4551    String::from_utf8(line).map_err(|_| HealthProbeError::bad_answer("HTTP header is not UTF-8"))
4552}
4553
4554async fn probe_module_health(
4555    module_id: &str,
4556    runtime: &SupervisorRuntimeConfig,
4557    drain_deadline: Option<Instant>,
4558) -> Result<HealthReport, HealthProbeError> {
4559    let Some(forwarding) = runtime.forwarding.as_ref() else {
4560        return Err(HealthProbeError::misconfigured(
4561            "supervisor was not configured with a forwarding table",
4562        ));
4563    };
4564    let probe_started_at = Instant::now();
4565    let mut deadline = probe_started_at + runtime.health.deadline;
4566    if let Some(drain_deadline) = drain_deadline {
4567        deadline = deadline.min(drain_deadline);
4568    }
4569    let pending = if drain_deadline.is_some() {
4570        forwarding.begin_drain_health_probe_rpc_for(
4571            module_id,
4572            MODULE_CONTROL_OP_HEALTH_CHECK,
4573            probe_started_at,
4574            deadline,
4575        )
4576    } else {
4577        forwarding.begin_health_probe_rpc_for(
4578            module_id,
4579            MODULE_CONTROL_OP_HEALTH_CHECK,
4580            probe_started_at,
4581            deadline,
4582        )
4583    }
4584    .map_err(|err| {
4585        // The endpoint is not registered, so there is no live control lane to
4586        // ask. That is the module being absent, not slow.
4587        HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
4588    })?;
4589    await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
4590}
4591
4592/// [`probe_module_health`] for one endpoint rather than the id's active one.
4593///
4594/// A swap probes two processes that no by-id lookup reaches: its candidate
4595/// before cutover, and its superseded incumbent (for busy gauges) while the
4596/// incumbent drains. `deadline_cap` bounds the probe the way a drain deadline
4597/// bounds the by-id drain probe.
4598async fn probe_endpoint_health(
4599    endpoint: crate::ModuleEndpointId,
4600    runtime: &SupervisorRuntimeConfig,
4601    deadline_cap: Option<Instant>,
4602) -> Result<HealthReport, HealthProbeError> {
4603    let Some(forwarding) = runtime.forwarding.as_ref() else {
4604        return Err(HealthProbeError::misconfigured(
4605            "supervisor was not configured with a forwarding table",
4606        ));
4607    };
4608    let probe_started_at = Instant::now();
4609    let mut deadline = probe_started_at + runtime.health.deadline;
4610    if let Some(cap) = deadline_cap {
4611        deadline = deadline.min(cap);
4612    }
4613    let pending = forwarding
4614        .begin_endpoint_health_probe_rpc_for(
4615            endpoint,
4616            MODULE_CONTROL_OP_HEALTH_CHECK,
4617            probe_started_at,
4618            deadline,
4619        )
4620        .map_err(|err| {
4621            HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
4622        })?;
4623    await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
4624}
4625
4626/// Send a begun health probe and classify its answer.
4627async fn await_health_probe(
4628    forwarding: &ForwardingTable,
4629    pending: PendingModuleControlRpc,
4630    deadline: Instant,
4631    probe_budget: Duration,
4632) -> Result<HealthReport, HealthProbeError> {
4633    let PendingModuleControlRpc {
4634        endpoint,
4635        module_sink,
4636        negotiated_ver,
4637        corr,
4638        receiver,
4639    } = pending;
4640    let body = serde_json::to_vec(&ModuleControlRequest::HealthCheck {}).map_err(|err| {
4641        HealthProbeError::misconfigured(format!("failed to encode health.check: {err}"))
4642    })?;
4643    let frame = Frame::build_with_version(
4644        negotiated_ver,
4645        FrameType::Request,
4646        control_flags(),
4647        0,
4648        0,
4649        corr,
4650        body,
4651    )
4652    .map_err(|err| {
4653        HealthProbeError::misconfigured(format!("failed to build health.check frame: {err}"))
4654    })?;
4655
4656    // The enqueue itself must be bounded by the probe deadline: FrameSink.send
4657    // blocks waiting for capacity when the module's egress queue is full, and an
4658    // unbounded await here freezes the whole supervision actor (it stops polling
4659    // Child::wait and supervisor commands), making the module unrecoverable
4660    // in-band. On timeout the probe fails like any transport failure.
4661    match timeout_at(deadline, module_sink.send(frame)).await {
4662        Ok(Ok(())) => {}
4663        Ok(Err(err)) => {
4664            let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
4665            // A closed sink means the module's egress channel is gone -- the
4666            // receiving half is dropped when its connection tears down. Proof.
4667            return Err(HealthProbeError::lane_dead(format!(
4668                "failed to send health.check: {err}"
4669            )));
4670        }
4671        Err(_elapsed) => {
4672            let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
4673            // A full egress queue means the module is not draining its socket, which
4674            // is consistent with a wedged module AND with one whose reader is merely
4675            // starved. Silence, not proof.
4676            return Err(HealthProbeError::no_answer(
4677                "health.check send timed out before enqueue (module egress full)",
4678            ));
4679        }
4680    }
4681
4682    match timeout_at(deadline, receiver).await {
4683        // Each arm records WHAT WAS OBSERVED. Four of them are the module
4684        // demonstrably answering -- rejected, non-health, malformed, wrong op --
4685        // and those prove it is alive even though the probe failed.
4686        Ok(Ok(ModuleControlRpcOutcome::Response(response))) => {
4687            response.health_report().ok_or_else(|| {
4688                HealthProbeError::bad_answer("health.check RPC returned a non-health response")
4689            })
4690        }
4691        Ok(Ok(ModuleControlRpcOutcome::Rejected(body))) => Err(HealthProbeError::bad_answer(
4692            format!("health.check rejected: {}", body.message),
4693        )),
4694        Ok(Ok(ModuleControlRpcOutcome::ModuleGone(message))) => {
4695            Err(HealthProbeError::lane_dead(message))
4696        }
4697        Ok(Ok(ModuleControlRpcOutcome::MalformedResponse(message))) => {
4698            Err(HealthProbeError::bad_answer(message))
4699        }
4700        Ok(Ok(ModuleControlRpcOutcome::UnexpectedOp { expected, actual })) => {
4701            Err(HealthProbeError::bad_answer(format!(
4702                "expected module-control op '{expected}', got '{actual}'"
4703            )))
4704        }
4705        // A reply that crosses the deadline before this waiter observes it is
4706        // still proof of life. The forwarding path records its end-to-end latency
4707        // before delivering this classification.
4708        Ok(Ok(ModuleControlRpcOutcome::DeadlineElapsed)) => Err(HealthProbeError::bad_answer(
4709            "module answered health.check after its daemon deadline",
4710        )),
4711        Ok(Err(_)) => Err(HealthProbeError::misconfigured(
4712            "health.check waiter was canceled before the module responded",
4713        )),
4714        Err(_) => {
4715            let _ = forwarding.tombstone_health_probe_rpc(endpoint, corr);
4716            Err(HealthProbeError::no_answer(format!(
4717                "module did not answer health.check within {probe_budget:?}"
4718            )))
4719        }
4720    }
4721}
4722
4723#[allow(clippy::too_many_arguments)]
4724async fn handle_health_report(
4725    spec: &ModuleSpec,
4726    runtime: &SupervisorRuntimeConfig,
4727    registry: &Registry,
4728    process_liveness: &SupervisorProcessLiveness,
4729    snapshot: &SharedSnapshot,
4730    child: &mut Option<SupervisedChild>,
4731    report: HealthReport,
4732    now_ms: u64,
4733) {
4734    let status = supervisor_health_status(report.status);
4735    let detail = report.detail.clone();
4736    let metrics = truncate_health_metrics(report.metrics);
4737    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4738        state.health.status = status;
4739        state.health.last_probe_ms = Some(now_ms);
4740        state.health.detail = detail.clone();
4741        state.health.metrics = metrics.clone();
4742        state.health.consecutive_failures = 0;
4743    });
4744
4745    let action = match report.status {
4746        HealthStatus::Ok => return,
4747        HealthStatus::Degraded => runtime.health.on_degraded,
4748        HealthStatus::Failing => runtime.health.on_failing,
4749    };
4750    apply_l3_health_action(
4751        spec,
4752        runtime,
4753        registry,
4754        process_liveness,
4755        snapshot,
4756        child,
4757        status,
4758        detail.as_deref(),
4759        action,
4760        now_ms,
4761    )
4762    .await;
4763}
4764
4765#[allow(clippy::too_many_arguments)]
4766async fn handle_health_probe_failure(
4767    spec: &ModuleSpec,
4768    runtime: &SupervisorRuntimeConfig,
4769    registry: &Registry,
4770    process_liveness: &SupervisorProcessLiveness,
4771    snapshot: &SharedSnapshot,
4772    child: &mut Option<SupervisedChild>,
4773    err: HealthProbeError,
4774    now_ms: u64,
4775) {
4776    let threshold = runtime.health.failure_threshold.max(1);
4777    let mut failures = 0;
4778    // Carry the evidence class into the operator-visible detail. Without it,
4779    // "module did not answer within 5s" and "the control lane is gone" are two
4780    // prose strings in the same field, and the reader has to know the codebase to
4781    // tell which one is proof of anything.
4782    let detail = format!("[{}] {err}", err.label());
4783    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4784        // A failed wire probe invalidates the last report, even before the
4785        // restart threshold. HTTP probes already mark failures as Failing.
4786        if state.spawned_protocol.unwrap_or(spec.protocol) == ModuleProtocol::Subc {
4787            state.health.status = SupervisorHealthStatus::Unknown;
4788        }
4789        state.health.last_probe_ms = Some(now_ms);
4790        state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
4791        state.health.detail = Some(detail.clone());
4792        state.health.metrics = None;
4793        failures = state.health.consecutive_failures;
4794    });
4795
4796    if failures < threshold {
4797        warn!(
4798            module_id = %spec.module_id,
4799            consecutive_failures = failures,
4800            threshold,
4801            evidence = err.label(),
4802            detail = %detail,
4803            "health.check probe failed"
4804        );
4805        return;
4806    }
4807
4808    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4809        state.state = ModuleState::Unresponsive;
4810        state.health.status = SupervisorHealthStatus::Unresponsive;
4811    });
4812    // The evidence class is logged at the kill site because this is the line an
4813    // operator reads after an unexplained restart. A streak of `no-answer` under
4814    // machine load is the known false-positive shape; a `lane-dead` is not.
4815    if runtime.health.critical {
4816        error!(
4817            module_id = %spec.module_id,
4818            status = "unresponsive",
4819            evidence = err.label(),
4820            detail = %detail,
4821            "critical module health alert"
4822        );
4823    } else {
4824        warn!(
4825            module_id = %spec.module_id,
4826            status = "unresponsive",
4827            evidence = err.label(),
4828            detail = %detail,
4829            "module health threshold breached"
4830        );
4831    }
4832    if let Err(err) = health_restart_child(
4833        spec,
4834        runtime,
4835        registry,
4836        process_liveness,
4837        snapshot,
4838        child,
4839        SupervisorHealthStatus::Unresponsive,
4840        Some(&detail),
4841        now_ms,
4842    )
4843    .await
4844    {
4845        error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
4846    }
4847}
4848
4849#[allow(clippy::too_many_arguments)]
4850async fn apply_l3_health_action(
4851    spec: &ModuleSpec,
4852    runtime: &SupervisorRuntimeConfig,
4853    registry: &Registry,
4854    process_liveness: &SupervisorProcessLiveness,
4855    snapshot: &SharedSnapshot,
4856    child: &mut Option<SupervisedChild>,
4857    status: SupervisorHealthStatus,
4858    detail: Option<&str>,
4859    action: HealthAction,
4860    now_ms: u64,
4861) {
4862    record_health_action(snapshot, &spec.module_id, action.to_string(), now_ms);
4863    match action {
4864        HealthAction::Report => {
4865            info!(
4866                module_id = %spec.module_id,
4867                status = ?status,
4868                detail,
4869                "module reported non-ok health"
4870            );
4871        }
4872        HealthAction::Alert => {
4873            error!(
4874                module_id = %spec.module_id,
4875                status = ?status,
4876                detail,
4877                "module health alert"
4878            );
4879        }
4880        HealthAction::Restart => {
4881            if let Err(err) = health_restart_child(
4882                spec,
4883                runtime,
4884                registry,
4885                process_liveness,
4886                snapshot,
4887                child,
4888                status,
4889                detail,
4890                now_ms,
4891            )
4892            .await
4893            {
4894                error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
4895            }
4896        }
4897    }
4898}
4899
4900#[allow(clippy::too_many_arguments)]
4901async fn health_restart_child(
4902    spec: &ModuleSpec,
4903    runtime: &SupervisorRuntimeConfig,
4904    registry: &Registry,
4905    process_liveness: &SupervisorProcessLiveness,
4906    snapshot: &SharedSnapshot,
4907    child: &mut Option<SupervisedChild>,
4908    status: SupervisorHealthStatus,
4909    detail: Option<&str>,
4910    now_ms: u64,
4911) -> Result<(), SuperviseError> {
4912    let (enabled, schedule) = {
4913        let mut state = lock_snapshot(snapshot)?;
4914        let enabled = state.enabled;
4915        let schedule = if enabled {
4916            state.next_crash_restart(&runtime.restart_policy, Instant::now())
4917        } else {
4918            None
4919        };
4920        (enabled, schedule)
4921    };
4922
4923    if !enabled {
4924        return Err(SuperviseError::Disabled {
4925            module_id: spec.module_id.clone(),
4926        });
4927    }
4928
4929    if schedule.is_none() {
4930        record_health_action(snapshot, &spec.module_id, "failed".to_string(), now_ms);
4931        error!(
4932            module_id = %spec.module_id,
4933            status = ?status,
4934            detail,
4935            max_restarts = runtime.restart_policy.max_restarts,
4936            window_secs = runtime.restart_policy.window.as_secs(),
4937            reason = %runtime.restart_policy.budget_exhausted_detail(),
4938            "health restart budget exhausted; marking module failed"
4939        );
4940        let stop_notice = begin_forwarding_drain_if_configured(
4941            spec,
4942            runtime,
4943            registry,
4944            snapshot,
4945            Some(true),
4946            RouteCloseReason::Disable,
4947        )
4948        .await?;
4949        update_snapshot(snapshot, Some(&spec.module_id), |state| {
4950            state.drain_disposition_detail = Some(runtime.restart_policy.budget_exhausted_detail());
4951        })?;
4952        drain_optional_child(
4953            &spec.module_id,
4954            spec.protocol,
4955            stop_notice,
4956            registry,
4957            runtime.forwarding.as_deref(),
4958            snapshot,
4959            &runtime.terminal_ring,
4960            &runtime.spawn_events,
4961            child,
4962            runtime.drain_timeout,
4963            ModuleState::Failed,
4964            Some(true),
4965        )
4966        .await?;
4967        process_liveness.untrack_if_current(&spec.module_id, snapshot);
4968        return Ok(());
4969    }
4970
4971    let schedule = schedule.expect("a health restart must have a crash-restart schedule");
4972    let mut restart_count = 0;
4973    update_snapshot(snapshot, Some(&spec.module_id), |state| {
4974        restart_count = state.crash_restarts.len();
4975        state.state = ModuleState::Unresponsive;
4976        state.health.status = status;
4977        state.health.last_action = Some(HealthAction::Restart.to_string());
4978        state.health.last_action_ms = Some(now_ms);
4979    })?;
4980    warn!(
4981        module_id = %spec.module_id,
4982        status = ?status,
4983        detail,
4984        restart_count,
4985        restart_in_window = schedule.restart_in_window,
4986        delay_ms = schedule.delay.as_millis() as u64,
4987        "health-triggered module restart"
4988    );
4989
4990    let stop_notice = begin_forwarding_drain_if_configured(
4991        spec,
4992        runtime,
4993        registry,
4994        snapshot,
4995        Some(true),
4996        RouteCloseReason::Restart,
4997    )
4998    .await?;
4999    drain_optional_child(
5000        &spec.module_id,
5001        spec.protocol,
5002        stop_notice,
5003        registry,
5004        runtime.forwarding.as_deref(),
5005        snapshot,
5006        &runtime.terminal_ring,
5007        &runtime.spawn_events,
5008        child,
5009        runtime.drain_timeout,
5010        ModuleState::Restarting,
5011        Some(true),
5012    )
5013    .await?;
5014    schedule_respawn(
5015        runtime,
5016        snapshot,
5017        &spec.module_id,
5018        schedule.delay,
5019        RespawnKind::Spawn,
5020    )
5021}
5022
5023fn cancel_deferred_reload(runtime: &SupervisorRuntimeConfig, module_id: &str, reason: &str) {
5024    if let Some(reply) = runtime
5025        .deferred_reload_reply
5026        .lock()
5027        .unwrap_or_else(|p| p.into_inner())
5028        .take()
5029    {
5030        let _ = reply.send(Err(SuperviseError::ReloadFailed {
5031            module_id: module_id.to_string(),
5032            reason: reason.to_string(),
5033        }));
5034    }
5035}
5036
5037fn schedule_respawn(
5038    runtime: &SupervisorRuntimeConfig,
5039    snapshot: &SharedSnapshot,
5040    module_id: &str,
5041    delay: Duration,
5042    kind: RespawnKind,
5043) -> Result<(), SuperviseError> {
5044    cancel_deferred_reload(runtime, module_id, "respawn superseded by another restart");
5045    update_snapshot(snapshot, Some(module_id), |state| {
5046        state.respawn_pending = true
5047    })?;
5048    *runtime
5049        .scheduled_respawn
5050        .lock()
5051        .unwrap_or_else(|p| p.into_inner()) = Some(PendingRespawn {
5052        deadline: Instant::now() + delay,
5053        kind,
5054    });
5055    Ok(())
5056}
5057
5058fn record_health_action(snapshot: &SharedSnapshot, module_id: &str, action: String, now_ms: u64) {
5059    let _ = update_snapshot(snapshot, Some(module_id), |state| {
5060        state.health.last_action = Some(action);
5061        state.health.last_action_ms = Some(now_ms);
5062    });
5063}
5064
5065fn supervisor_health_status(status: HealthStatus) -> SupervisorHealthStatus {
5066    match status {
5067        HealthStatus::Ok => SupervisorHealthStatus::Ok,
5068        HealthStatus::Degraded => SupervisorHealthStatus::Degraded,
5069        HealthStatus::Failing => SupervisorHealthStatus::Failing,
5070    }
5071}
5072
5073/// Caps the metrics blob stored in the cached supervisor snapshot, which is
5074/// returned to every `supervisor.list` and `supervisor.health` caller.
5075///
5076/// This cap is deliberately NOT applied on the one-shot `supervisor.health_probe`
5077/// path: that request exists to return a module's complete metrics object, and
5078/// `ck health <module-id>` documents it as the way to see what the cached view
5079/// truncates. The asymmetry is the feature.
5080///
5081/// So a new caller must decide which side it is on rather than assume the cap is
5082/// universal. Reaching for it on a fresh-probe path would silently reintroduce
5083/// the truncation that path exists to avoid.
5084fn truncate_health_metrics(metrics: Option<Value>) -> Option<Value> {
5085    let metrics = metrics?;
5086    match serde_json::to_vec(&metrics) {
5087        Ok(encoded) if encoded.len() > MAX_HEALTH_METRICS_BYTES => Some(serde_json::json!({
5088            "truncated": true,
5089            "original_bytes": encoded.len(),
5090        })),
5091        Ok(_) | Err(_) => Some(metrics),
5092    }
5093}
5094
5095/// Spread health probes so a fleet-wide restart does not converge them.
5096///
5097/// The delay is derived from the module id and probe index rather than a random
5098/// source, so it is deterministic per module: a module keeps its own offset
5099/// across daemon restarts instead of re-rolling into a collision.
5100fn jittered_health_delay(module_id: &str, probe_index: u64, cadence: Duration) -> Duration {
5101    if cadence.is_zero() {
5102        return Duration::ZERO;
5103    }
5104    let cadence_ms = cadence.as_millis() as u64;
5105    // This early return is REDUNDANT, deliberately, and a mutation run will show
5106    // it surviving removal. Recording why here so the next person to notice does
5107    // not have to re-derive it:
5108    //
5109    // - It is unreachable in practice. `positive_millis` in daemon_config rejects
5110    //   a zero cadence and builds the Duration from whole milliseconds, so a
5111    //   sub-millisecond cadence cannot come from config.
5112    // - Even if reached it changes no answer. The `.max(1)` below makes the span
5113    //   1, and `hash % 1` is 0, so the fall-through returns `cadence` unchanged
5114    //   -- exactly what this returns.
5115    //
5116    // Kept as a guard against a future widening of the config parser (accepting
5117    // microseconds, say), which would make the sub-millisecond case reachable.
5118    // The `.max(1)` is the load-bearing half TODAY: remove it and the modulo
5119    // divides by zero. Remove this and nothing changes.
5120    if cadence_ms == 0 {
5121        return cadence;
5122    }
5123    // Note that this never returns less than one cadence, including for the FIRST
5124    // probe. So a freshly registered module reports health `unknown` for a full
5125    // cadence plus jitter -- 30-33s at the default -- no matter how quickly it is
5126    // ready to answer.
5127    //
5128    // That is a property of the supervisor's schedule, not of any module: an
5129    // operator watching a restart sees `unknown` and cannot tell it from a module
5130    // that is slow to warm. Measured on two unrelated modules, both flipping to
5131    // `ok` between 22s and 32s after restart.
5132    //
5133    // Left as-is because spreading the first probe is what keeps a fleet-wide
5134    // restart from firing fourteen simultaneous probes into a cold machine. The
5135    // alternative -- probe at t+0 and jitter only from the second onward -- trades
5136    // that thundering herd for a faster first reading.
5137    let jitter_span = (cadence_ms / 10).max(1);
5138    let hash = module_id.as_bytes().iter().fold(
5139        probe_index.wrapping_mul(0x9E37_79B9_7F4A_7C15),
5140        |acc, byte| {
5141            acc.wrapping_mul(1099511628211)
5142                .wrapping_add(u64::from(*byte))
5143        },
5144    );
5145    cadence + Duration::from_millis(hash % jitter_span)
5146}
5147
5148#[cfg(test)]
5149mod tests {
5150    use super::*;
5151
5152    #[test]
5153    fn readding_a_module_clears_its_rescan_removal_tombstone() {
5154        let handle = SupervisorHandle::new();
5155        let module_id = "readded-tombstone";
5156        handle.record_rescan_removal(module_id);
5157        assert!(handle.removal_tombstone_age_ms(module_id).is_some());
5158
5159        handle.apply_identity_configuration(&ModuleSpec {
5160            module_id: module_id.to_string(),
5161            program: PathBuf::from("/test/module"),
5162            args: Vec::new(),
5163            env: Vec::new(),
5164            reserved: false,
5165            reserved_prefixes: Vec::new(),
5166            protocol: ModuleProtocol::Subc,
5167            overlap: Default::default(),
5168        });
5169
5170        assert!(
5171            handle.removal_tombstone_age_ms(module_id).is_none(),
5172            "a re-added module must not retain a stale removal tombstone"
5173        );
5174    }
5175
5176    /// What one module's owner looked like from the control plane at the
5177    /// instant after its first process was spawned.
5178    #[derive(Debug, PartialEq, Eq)]
5179    struct OwnerInSpawnWindow {
5180        module_id: String,
5181        configured: bool,
5182        on_roster: bool,
5183        admission_refusal: Option<&'static str>,
5184    }
5185
5186    /// A supervised module's process can connect, register, sync its scopes
5187    /// and describe them as soon as it is spawned, which is BEFORE the
5188    /// supervisor puts the module on the roster. In that window the owner must
5189    /// already read as configured, so a scoped `route.open` against it is
5190    /// refused as retryable `scope_not_synced` and not as terminal
5191    /// `scope_not_live` ("will never sync").
5192    ///
5193    /// The hook runs in exactly that window on every path that takes on a new
5194    /// module, so no race with a real child is needed: `on_roster: false`
5195    /// proves each observation was taken before the roster insert.
5196    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5197    async fn a_new_module_is_configured_before_its_first_process_can_register() {
5198        use crate::scopes::ScopeTable;
5199        use subc_protocol::{error_codes, scope::ScopeSelector, Principal};
5200
5201        let dir = subc_test_support::TestTempDir::new("configured-before-spawn");
5202        let stub = |module_id: &str, program: PathBuf| ModuleSpec {
5203            module_id: module_id.to_string(),
5204            program,
5205            args: Vec::new(),
5206            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
5207                .into_iter()
5208                .map(|key| (key.to_string(), dir.path().display().to_string()))
5209                .chain([("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string())])
5210                .collect(),
5211            reserved: false,
5212            reserved_prefixes: Vec::new(),
5213            protocol: ModuleProtocol::Subc,
5214            overlap: Default::default(),
5215        };
5216        let live = super::terminal_history_tests::fake_aft_stub_path();
5217        let missing = dir.path().join("definitely-missing-module");
5218
5219        let handle = SupervisorHandle::new();
5220        let mut supervisor =
5221            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5222                .with_handle(handle.clone());
5223        let observed = Arc::new(Mutex::new(Vec::<OwnerInSpawnWindow>::new()));
5224        let hook_handle = handle.clone();
5225        let hook_observed = Arc::clone(&observed);
5226        supervisor.test_after_first_spawn = AfterFirstSpawnHook(Some(Arc::new(move |module_id| {
5227            // Exactly what the control plane computes for a scoped route.open
5228            // naming this module as the owner of a scope it has not synced.
5229            let configured = hook_handle.is_configured(module_id);
5230            let selector = ScopeSelector {
5231                owner: Principal::Reserved {
5232                    module_id: module_id.to_string(),
5233                },
5234                scope_ref: "s".to_string(),
5235                scope_epoch: Some(1),
5236            };
5237            let carrier = Principal::Reserved {
5238                module_id: "carrier".to_string(),
5239            };
5240            let admission_refusal = match ScopeTable::new(Vec::<String>::new())
5241                .admit(&carrier, module_id, &selector, configured)
5242            {
5243                Ok(_) => None,
5244                Err(refusal) => Some(refusal.code),
5245            };
5246            hook_observed.lock().unwrap().push(OwnerInSpawnWindow {
5247                module_id: module_id.to_string(),
5248                configured,
5249                on_roster: hook_handle.get(module_id).is_some(),
5250                admission_refusal,
5251            });
5252        })));
5253
5254        let plain = supervisor.spawn(stub("plain", live.clone())).unwrap();
5255        let configured = supervisor
5256            .supervise_configured(stub("configured", live.clone()), true)
5257            .unwrap();
5258        let with_health = supervisor
5259            .supervise_configured_with_health(
5260                stub("with-health", live.clone()),
5261                true,
5262                HealthConfig::default(),
5263                None,
5264                RestartPolicy::default(),
5265            )
5266            .unwrap();
5267        // The failed-spawn path still puts the module on the roster (as
5268        // failed), so it is configured throughout.
5269        let failed = supervisor
5270            .supervise_configured_with_health(
5271                stub("failed-spawn", missing.clone()),
5272                true,
5273                HealthConfig::default(),
5274                None,
5275                RestartPolicy::default(),
5276            )
5277            .unwrap();
5278        // A failed plain `spawn` puts nothing on the roster, so its mark is
5279        // taken back once the spawn has failed.
5280        assert!(supervisor.spawn(stub("spawn-error", missing)).is_err());
5281
5282        let expected = [
5283            "plain",
5284            "configured",
5285            "with-health",
5286            "failed-spawn",
5287            "spawn-error",
5288        ]
5289        .into_iter()
5290        .map(|module_id| OwnerInSpawnWindow {
5291            module_id: module_id.to_string(),
5292            configured: true,
5293            on_roster: false,
5294            admission_refusal: Some(error_codes::SCOPE_NOT_SYNCED),
5295        })
5296        .collect::<Vec<_>>();
5297        assert_eq!(*observed.lock().unwrap(), expected);
5298
5299        for module_id in ["plain", "configured", "with-health", "failed-spawn"] {
5300            assert!(
5301                handle.get(module_id).is_some(),
5302                "{module_id} is on the roster"
5303            );
5304            assert!(
5305                handle.is_configured(module_id),
5306                "{module_id} stays configured"
5307            );
5308        }
5309        assert!(handle.get("spawn-error").is_none());
5310        assert!(
5311            !handle.is_configured("spawn-error"),
5312            "a plain spawn that failed must not leave its module marked configured"
5313        );
5314
5315        // Leaving the roster clears the mark with it.
5316        handle.retire("failed-spawn");
5317        assert!(!handle.is_configured("failed-spawn"));
5318
5319        for module in [plain, configured, with_health] {
5320            module.stop().await.unwrap();
5321        }
5322        drop(failed);
5323    }
5324
5325    fn stale_process_snapshot(state: ModuleState, enabled: bool) -> SharedSnapshot {
5326        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(state, enabled)));
5327        update_snapshot(&snapshot, Some("stale-process-facts"), |snapshot| {
5328            snapshot.process_alive = true;
5329            snapshot.pid = Some(41);
5330            snapshot.spawned_at_ms = Some(42);
5331            snapshot.spawned_from = Some(PathBuf::from("/spawned/module"));
5332            snapshot.spawned_file_identity = Some(SpawnedFileIdentity {
5333                device: 43,
5334                inode: 44,
5335            });
5336        })
5337        .unwrap();
5338        snapshot
5339    }
5340
5341    fn assert_snapshot_process_facts_cleared(snapshot: &SharedSnapshot) {
5342        let snapshot = lock_snapshot(snapshot).unwrap();
5343        assert!(!snapshot.process_alive);
5344        assert_eq!(snapshot.pid, None);
5345        assert_eq!(snapshot.spawned_at_ms, None);
5346        assert_eq!(snapshot.spawned_from, None);
5347        assert_eq!(snapshot.spawned_file_identity, None);
5348    }
5349
5350    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5351    async fn failed_enable_spawn_clears_preexisting_current_process_facts() {
5352        let supervisor =
5353            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5354        let mut runtime = supervisor.runtime_config();
5355        runtime.test_seed_stale_facts_before_enable_spawn = true;
5356        let snapshot = stale_process_snapshot(ModuleState::Disabled, false);
5357        let mut child = None;
5358        let spec = ModuleSpec {
5359            module_id: "failed-enable-clears-facts".to_string(),
5360            program: PathBuf::from("/definitely/missing/failed-enable-module"),
5361            args: Vec::new(),
5362            env: Vec::new(),
5363            reserved: false,
5364            reserved_prefixes: Vec::new(),
5365            protocol: ModuleProtocol::Subc,
5366            overlap: Default::default(),
5367        };
5368
5369        let result = set_child_enabled(
5370            &spec,
5371            &runtime,
5372            &supervisor.registry,
5373            &supervisor.process_liveness,
5374            &snapshot,
5375            &mut child,
5376            true,
5377        )
5378        .await;
5379
5380        assert!(matches!(result, Err(SuperviseError::Spawn { .. })));
5381        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
5382        assert_snapshot_process_facts_cleared(&snapshot);
5383    }
5384
5385    #[tokio::test]
5386    async fn start_revives_stranded_restarting_but_not_pending_backoff() {
5387        let supervisor =
5388            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5389        let runtime = supervisor.runtime_config();
5390        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(
5391            ModuleState::Restarting,
5392            true,
5393        )));
5394        let spec = ModuleSpec {
5395            module_id: "start-stranded-restarting".to_string(),
5396            program: super::terminal_history_tests::fake_aft_stub_path(),
5397            args: Vec::new(),
5398            env: vec![("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string())],
5399            reserved: false,
5400            reserved_prefixes: Vec::new(),
5401            protocol: ModuleProtocol::None,
5402            overlap: Default::default(),
5403        };
5404        let mut child = None;
5405        lock_snapshot(&snapshot).unwrap().respawn_pending = true;
5406        assert!(!super::set_child_enabled(
5407            &spec,
5408            &runtime,
5409            &Registry::default(),
5410            &supervisor.process_liveness,
5411            &snapshot,
5412            &mut child,
5413            true
5414        )
5415        .await
5416        .unwrap());
5417        assert!(child.is_none());
5418        lock_snapshot(&snapshot).unwrap().respawn_pending = false;
5419        assert!(super::set_child_enabled(
5420            &spec,
5421            &runtime,
5422            &Registry::default(),
5423            &supervisor.process_liveness,
5424            &snapshot,
5425            &mut child,
5426            true
5427        )
5428        .await
5429        .unwrap());
5430        assert_eq!(
5431            lock_snapshot(&snapshot).unwrap().state,
5432            ModuleState::Running
5433        );
5434        let mut child = child.unwrap();
5435        child.start_kill().unwrap();
5436        child.wait().await.unwrap();
5437    }
5438
5439    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5440    async fn failed_reload_spawn_clears_current_process_facts() {
5441        let supervisor =
5442            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5443        let mut runtime = supervisor.runtime_config();
5444        runtime.restart_policy = RestartPolicy::new(0, Duration::ZERO);
5445        let snapshot = stale_process_snapshot(ModuleState::Running, true);
5446        let mut child = None;
5447        let spec = ModuleSpec {
5448            module_id: "failed-reload-clears-facts".to_string(),
5449            program: PathBuf::from("/unused/failed-reload-module"),
5450            args: Vec::new(),
5451            env: Vec::new(),
5452            reserved: false,
5453            reserved_prefixes: Vec::new(),
5454            protocol: ModuleProtocol::Subc,
5455            overlap: Default::default(),
5456        };
5457
5458        let result = handle_reload_spawn_failure(
5459            &spec,
5460            &runtime,
5461            &supervisor.process_liveness,
5462            &snapshot,
5463            &mut child,
5464            "forced reload spawn failure".to_string(),
5465        )
5466        .await;
5467
5468        assert!(matches!(result, Err(SuperviseError::ReloadFailed { .. })));
5469        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
5470        assert_snapshot_process_facts_cleared(&snapshot);
5471    }
5472
5473    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5474    async fn dropping_a_module_with_an_active_monitor_clears_current_process_facts() {
5475        let supervisor =
5476            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5477        let snapshot = stale_process_snapshot(ModuleState::Running, true);
5478        let module = supervisor.supervised_module(
5479            ModuleSpec {
5480                module_id: "drop-clears-facts".to_string(),
5481                program: PathBuf::from("/unused/drop-module"),
5482                args: Vec::new(),
5483                env: Vec::new(),
5484                reserved: false,
5485                reserved_prefixes: Vec::new(),
5486                protocol: ModuleProtocol::Subc,
5487                overlap: Default::default(),
5488            },
5489            supervisor.runtime_config(),
5490            Arc::clone(&snapshot),
5491            None,
5492        );
5493        assert!(!module
5494            .inner
5495            .monitor
5496            .lock()
5497            .unwrap()
5498            .as_ref()
5499            .unwrap()
5500            .is_finished());
5501
5502        drop(module);
5503
5504        assert_eq!(
5505            lock_snapshot(&snapshot).unwrap().state,
5506            ModuleState::Stopped
5507        );
5508        assert_snapshot_process_facts_cleared(&snapshot);
5509    }
5510
5511    #[cfg(unix)]
5512    #[tokio::test]
5513    async fn rescan_preserves_running_protocol_until_respawn() {
5514        let dir = subc_test_support::TestTempDir::new("rescan-protocol");
5515        let initial = ModuleSpec {
5516            module_id: "rescan-protocol".into(),
5517            program: PathBuf::from("/bin/sleep"),
5518            args: vec!["60".into()],
5519            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
5520                .into_iter()
5521                .map(|key| (key.into(), dir.path().to_string_lossy().into_owned()))
5522                .collect(),
5523            reserved: false,
5524            reserved_prefixes: vec![],
5525            protocol: ModuleProtocol::None,
5526            overlap: Default::default(),
5527        };
5528        let supervisor =
5529            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5530        let module = supervisor.spawn(initial.clone()).unwrap();
5531        assert!(module.status().unwrap().live);
5532        let mut next = initial;
5533        next.protocol = ModuleProtocol::Subc;
5534        module
5535            .update_configuration(next.clone(), HealthConfig::default(), None)
5536            .await
5537            .unwrap();
5538        assert!(
5539            module.status().unwrap().live,
5540            "rescan must not require HELLO from the old non-wire process"
5541        );
5542        let runtime = supervisor.runtime_config();
5543        let action = on_child_exit(
5544            &next,
5545            RestartPolicy::default(),
5546            &supervisor.registry,
5547            &module.inner.snapshot,
5548            &runtime.terminal_ring,
5549            &runtime.spawn_events,
5550            &runtime.child_roster,
5551            ExitReport {
5552                kind: ExitKind::Clean,
5553                code: Some(0),
5554                signal: None,
5555                at_ms: unix_ms_now(),
5556            },
5557        )
5558        .await;
5559        assert!(
5560            matches!(action, NextAction::Restart { .. }),
5561            "the old non-wire process's clean exit must restart"
5562        );
5563        module.drain().await.unwrap();
5564    }
5565
5566    #[cfg(unix)]
5567    #[tokio::test]
5568    async fn orphan_identity_matches_path_names_and_shebang_interpreters() {
5569        use std::os::unix::fs::PermissionsExt;
5570        let dir = subc_test_support::TestTempDir::new("spawn-image-identity");
5571        let script = dir.join("module.sh");
5572        std::fs::write(&script, "#!/bin/sh\nwhile :; do sleep 0.1; done\n").unwrap();
5573        std::fs::set_permissions(&script, std::fs::Permissions::from_mode(0o700)).unwrap();
5574        let record_path = dir.join("live-children.json");
5575        let supervisor =
5576            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5577                .with_live_children_record(&record_path);
5578        for (program, args) in [
5579            (PathBuf::from("sleep"), vec!["60".into()]),
5580            (script, vec![]),
5581        ] {
5582            let spec = ModuleSpec {
5583                module_id: "image-identity".into(),
5584                program: program.clone(),
5585                args,
5586                env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
5587                    .into_iter()
5588                    .map(|key| (key.into(), dir.to_string_lossy().into_owned()))
5589                    .collect(),
5590                reserved: false,
5591                reserved_prefixes: vec![],
5592                protocol: ModuleProtocol::None,
5593                overlap: Default::default(),
5594            };
5595            let module = supervisor.spawn(spec).unwrap();
5596            #[cfg(target_os = "macos")]
5597            {
5598                // SETEXEC confirmation is asynchronous; the orphan record must
5599                // identify the final image, never the intermediate trampoline.
5600                let deadline = Instant::now() + Duration::from_secs(5);
5601                while crate::live_children::read_record(&record_path)
5602                    .unwrap()
5603                    .iter()
5604                    .all(|entry| entry.executable.is_none())
5605                {
5606                    assert!(Instant::now() < deadline, "module image was not confirmed");
5607                    tokio::time::sleep(Duration::from_millis(5)).await;
5608                }
5609            }
5610            let entry = crate::live_children::read_record(&record_path)
5611                .unwrap()
5612                .pop()
5613                .unwrap();
5614            let observed = subc_os::Process::open(entry.pid)
5615                .unwrap()
5616                .unwrap()
5617                .observe()
5618                .unwrap();
5619            let verdict = crate::live_children::identity_verdict(&entry, entry.pid, &observed);
5620            module.drain().await.unwrap();
5621            assert_eq!(
5622                verdict,
5623                crate::live_children::IdentityVerdict::Matches,
5624                "program {program:?}: recorded {entry:?}, observed {observed:?}"
5625            );
5626        }
5627    }
5628
5629    #[cfg(unix)]
5630    fn http_fixture(
5631        dir: &std::path::Path,
5632        url: &str,
5633        threshold: u32,
5634    ) -> crate::daemon_config::ConfiguredModule {
5635        let path = dir.join("subc.jsonc");
5636        std::fs::write(&path, serde_json::json!({"version": 1, "modules": {"http-process": {
5637            "program": "/bin/sleep", "args": ["60"], "protocol": "none",
5638            "env": {"XDG_DATA_HOME":dir,"XDG_RUNTIME_DIR":dir,"XDG_CONFIG_HOME":dir},
5639            "health": {"http":url,"cadence_ms":20,"deadline_ms":200,"failure_threshold":threshold},
5640            "restart": {"backoff_ms":1,"max_backoff_ms":1}, "drain_timeout_ms":10
5641        }}}).to_string()).unwrap();
5642        crate::daemon_config::load(&path)
5643            .unwrap()
5644            .unwrap()
5645            .modules
5646            .pop()
5647            .unwrap()
5648    }
5649
5650    #[cfg(unix)]
5651    async fn wait_http_health(module: &SupervisedModule, status: SupervisorHealthStatus) {
5652        timeout(Duration::from_secs(5), async {
5653            loop {
5654                if module.status().unwrap().health.status == status {
5655                    break;
5656                }
5657                sleep(Duration::from_millis(5)).await;
5658            }
5659        })
5660        .await
5661        .unwrap_or_else(|_| {
5662            panic!(
5663                "expected {status:?}, got {:?}",
5664                module.status().unwrap().health
5665            )
5666        });
5667    }
5668
5669    #[cfg(unix)]
5670    #[tokio::test]
5671    async fn http_health_status_flips_ok_failing_ok() {
5672        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5673        let dir = subc_test_support::TestTempDir::new("http-status-flips");
5674        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5675        let status = Arc::new(std::sync::atomic::AtomicU16::new(200));
5676        let serving_status = status.clone();
5677        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5678        let server = tokio::spawn(async move {
5679            loop {
5680                let (mut stream, _) = listener.accept().await.unwrap();
5681                let mut request = [0u8; 2048];
5682                let count = stream.read(&mut request).await.unwrap();
5683                assert!(count > 0, "a probe must send an HTTP request");
5684                let code = serving_status.load(std::sync::atomic::Ordering::SeqCst);
5685                let body = if code == 200 {
5686                    "ready"
5687                } else {
5688                    "{\"status\":\"unavailable\",\"error\":\"scratch failure\"}"
5689                };
5690                let response = format!("HTTP/1.1 {code} scratch\r\nContent-Length: {}\r\nConnection: close\r\n\r\n{body}", body.len());
5691                let _ = stream.write_all(response.as_bytes()).await;
5692            }
5693        });
5694        let configured = http_fixture(&dir, &url, 1000);
5695        let module =
5696            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5697                .supervise_configured_with_health(
5698                    configured.module_spec(),
5699                    true,
5700                    configured.health,
5701                    configured.drain_timeout_ms,
5702                    configured.restart,
5703                )
5704                .unwrap();
5705        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5706        status.store(503, std::sync::atomic::Ordering::SeqCst);
5707        wait_http_health(&module, SupervisorHealthStatus::Failing).await;
5708        assert!(module
5709            .status()
5710            .unwrap()
5711            .health
5712            .detail
5713            .unwrap()
5714            .contains("scratch failure"));
5715        status.store(200, std::sync::atomic::Ordering::SeqCst);
5716        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5717        assert_eq!(module.status().unwrap().health.consecutive_failures, 0);
5718        let before = module.status().unwrap();
5719        let (spec, mut health) = module.configuration().unwrap();
5720        health.http = None;
5721        module
5722            .update_configuration(spec.clone(), health.clone(), Some(10))
5723            .await
5724            .unwrap();
5725        wait_http_health(&module, SupervisorHealthStatus::Unknown).await;
5726        health.http = Some(url);
5727        module
5728            .update_configuration(spec, health, Some(10))
5729            .await
5730            .unwrap();
5731        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5732        assert_eq!(
5733            module.status().unwrap().pid,
5734            before.pid,
5735            "changing a probe must apply live, not restart its process"
5736        );
5737        let (spec, mut health) = module.configuration().unwrap();
5738        health.failure_threshold = 2;
5739        module
5740            .update_configuration(spec, health, Some(10))
5741            .await
5742            .unwrap();
5743        status.store(503, std::sync::atomic::Ordering::SeqCst);
5744        timeout(Duration::from_secs(5), async {
5745            while module.status().unwrap().spawn_generation == before.spawn_generation {
5746                sleep(Duration::from_millis(5)).await;
5747            }
5748        })
5749        .await
5750        .expect("sustained HTTP 503 responses must trigger the health restart policy");
5751        module.drain().await.unwrap();
5752        server.abort();
5753    }
5754
5755    #[cfg(unix)]
5756    #[tokio::test]
5757    async fn http_health_no_listener_restarts_after_consecutive_failures() {
5758        let dir = subc_test_support::TestTempDir::new("http-refused");
5759        let unused = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5760        let url = format!("http://{}/healthz", unused.local_addr().unwrap());
5761        drop(unused);
5762        let configured = http_fixture(&dir, &url, 2);
5763        let module =
5764            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5765                .supervise_configured_with_health(
5766                    configured.module_spec(),
5767                    true,
5768                    configured.health,
5769                    configured.drain_timeout_ms,
5770                    configured.restart,
5771                )
5772                .unwrap();
5773        let before = module.status().unwrap().spawn_generation;
5774        timeout(Duration::from_secs(5), async {
5775            loop {
5776                let status = module.status().unwrap();
5777                if status.spawn_generation > before {
5778                    assert!(status.lifetime_restarts > 0);
5779                    break;
5780                }
5781                sleep(Duration::from_millis(5)).await;
5782            }
5783        })
5784        .await
5785        .expect("sustained HTTP refusal must trigger the health restart policy");
5786        module.drain().await.unwrap();
5787    }
5788
5789    #[cfg(unix)]
5790    #[tokio::test]
5791    async fn http_health_timeout_honours_deadline() {
5792        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5793        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5794        let server = tokio::spawn(async move {
5795            let _held = listener.accept().await.unwrap();
5796            std::future::pending::<()>().await;
5797        });
5798        let error = timeout(
5799            Duration::from_secs(1),
5800            probe_http_health(&url, Duration::from_millis(10)),
5801        )
5802        .await
5803        .expect("the probe must enforce its own deadline")
5804        .unwrap_err();
5805        server.abort();
5806        assert!(error.to_string().contains("timed out"));
5807    }
5808
5809    #[cfg(unix)]
5810    #[tokio::test]
5811    async fn http_health_localhost_reaches_an_ipv6_only_listener() {
5812        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5813        let listener = tokio::net::TcpListener::bind("[::1]:0").await.unwrap();
5814        let url = format!(
5815            "http://localhost:{}/healthz",
5816            listener.local_addr().unwrap().port()
5817        );
5818        let server = tokio::spawn(async move {
5819            let (mut stream, _) = listener.accept().await.unwrap();
5820            let mut request = [0u8; 2048];
5821            assert!(stream.read(&mut request).await.unwrap() > 0);
5822            stream
5823                .write_all(b"HTTP/1.1 200 OK\r\nContent-Length: 2\r\n\r\nok")
5824                .await
5825                .unwrap();
5826        });
5827        // The deadline only bounds a hang. A probe that never tried the IPv6
5828        // address would be refused on 127.0.0.1 and fail at once, so a longer
5829        // deadline does not weaken the assertion; one second timed out under a
5830        // loaded parallel test run.
5831        assert_eq!(
5832            probe_http_health(&url, Duration::from_secs(10))
5833                .await
5834                .unwrap()
5835                .status,
5836            HealthStatus::Ok
5837        );
5838        server.await.unwrap();
5839    }
5840
5841    #[cfg(unix)]
5842    #[tokio::test]
5843    async fn http_health_timeout_keeps_partial_status_and_body() {
5844        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5845        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5846        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5847        let server = tokio::spawn(async move {
5848            let (mut stream, _) = listener.accept().await.unwrap();
5849            let mut request = [0u8; 2048];
5850            assert!(stream.read(&mut request).await.unwrap() > 0);
5851            stream
5852                .write_all(
5853                    b"HTTP/1.1 503 Unavailable\r\nContent-Length: 100\r\n\r\npartial diagnostic",
5854                )
5855                .await
5856                .unwrap();
5857            std::future::pending::<()>().await;
5858        });
5859        let error = probe_http_health(&url, Duration::from_secs(1))
5860            .await
5861            .unwrap_err()
5862            .to_string();
5863        server.abort();
5864        assert!(
5865            error.contains("timed out")
5866                && error.contains("503 Unavailable")
5867                && error.contains("partial diagnostic"),
5868            "{error}"
5869        );
5870    }
5871
5872    #[cfg(unix)]
5873    #[tokio::test]
5874    async fn http_health_diagnostic_body_is_bounded_to_200_bytes() {
5875        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5876        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5877        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5878        let server = tokio::spawn(async move {
5879            let (mut stream, _) = listener.accept().await.unwrap();
5880            let mut request = [0u8; 2048];
5881            assert!(stream.read(&mut request).await.unwrap() > 0);
5882            let body = format!("{}not-in-diagnostic", "x".repeat(400));
5883            let response = format!(
5884                "HTTP/1.1 503 Unavailable\r\nContent-Length: {}\r\n\r\n{body}",
5885                body.len()
5886            );
5887            stream.write_all(response.as_bytes()).await.unwrap();
5888        });
5889        let error = probe_http_health(&url, Duration::from_secs(1))
5890            .await
5891            .unwrap_err()
5892            .to_string();
5893        server.await.unwrap();
5894        assert!(error.ends_with(&"x".repeat(200)), "{error}");
5895        assert_eq!(error.split(": ").last().unwrap().len(), 200);
5896        assert!(!error.contains("not-in-diagnostic"));
5897    }
5898
5899    #[cfg(unix)]
5900    #[tokio::test]
5901    async fn http_health_real_nats_server_monitoring() {
5902        if std::process::Command::new("nats-server")
5903            .arg("--version")
5904            .env("XDG_DATA_HOME", std::env::temp_dir())
5905            .env("XDG_RUNTIME_DIR", std::env::temp_dir())
5906            .env("XDG_CONFIG_HOME", std::env::temp_dir())
5907            .output()
5908            .is_err()
5909        {
5910            eprintln!(
5911                "skipped: http_health_real_nats_server_monitoring (nats-server is not on PATH)"
5912            );
5913            return;
5914        }
5915        let dir = subc_test_support::TestTempDir::new("nats-http-monitoring");
5916        let monitor = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5917        let port = monitor.local_addr().unwrap().port();
5918        let client = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5919        let client_port = client.local_addr().unwrap().port();
5920        let config = dir.join("server.conf");
5921        std::fs::write(&config, format!("listen: 127.0.0.1:{client_port}\nhttp: 127.0.0.1:{port}\njetstream {{ store_dir: \"{}\" }}\n", dir.join("jetstream").display())).unwrap();
5922        let mut configured = http_fixture(&dir, &format!("http://127.0.0.1:{port}/healthz"), 1000);
5923        configured.program = PathBuf::from("nats-server");
5924        configured.args = vec!["-c".into(), config.to_string_lossy().into_owned()];
5925        drop(monitor);
5926        drop(client);
5927        let module =
5928            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5929                .supervise_configured_with_health(
5930                    configured.module_spec(),
5931                    true,
5932                    configured.health,
5933                    configured.drain_timeout_ms,
5934                    configured.restart,
5935                )
5936                .unwrap();
5937        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5938        module.drain().await.unwrap();
5939    }
5940
5941    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5942    async fn configuration_update_does_not_replace_captured_running_process_facts() {
5943        let supervisor =
5944            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5945        let snapshot = stale_process_snapshot(ModuleState::Running, true);
5946        let initial = ModuleSpec {
5947            module_id: "rescan-preserves-spawn-facts".to_string(),
5948            program: PathBuf::from("/spawned/module"),
5949            args: Vec::new(),
5950            env: Vec::new(),
5951            reserved: false,
5952            reserved_prefixes: Vec::new(),
5953            protocol: ModuleProtocol::Subc,
5954            overlap: Default::default(),
5955        };
5956        let module = supervisor.supervised_module(
5957            initial.clone(),
5958            supervisor.runtime_config(),
5959            snapshot,
5960            None,
5961        );
5962        let before = module.status().unwrap();
5963        let mut replacement = initial;
5964        replacement.program = PathBuf::from("/rescanned/replacement-module");
5965
5966        module
5967            .update_configuration(replacement, HealthConfig::default(), None)
5968            .await
5969            .unwrap();
5970
5971        let after = module.status().unwrap();
5972        assert_eq!(after.pid, before.pid);
5973        assert_eq!(after.spawned_at_ms, before.spawned_at_ms);
5974        assert_eq!(after.spawned_from, before.spawned_from);
5975        drop(module);
5976    }
5977}
5978
5979fn unix_ms_now() -> u64 {
5980    SystemTime::now()
5981        .duration_since(UNIX_EPOCH)
5982        .map(|duration| duration.as_millis().min(u128::from(u64::MAX)) as u64)
5983        .unwrap_or(0)
5984}
5985
5986async fn supervise_loop(
5987    mut spec: ModuleSpec,
5988    mut runtime: SupervisorRuntimeConfig,
5989    registry: Arc<Registry>,
5990    process_liveness: Arc<SupervisorProcessLiveness>,
5991    snapshot: SharedSnapshot,
5992    mut child: Option<SupervisedChild>,
5993    mut commands: mpsc::Receiver<SupervisorCommand>,
5994) {
5995    let mut health_probe = HealthProbeRuntime::default();
5996    // Registry writes (including embedded callers) notify this module only.
5997    // Subscribe before the first refresh; watch retains changes that arrive
5998    // while commands or probes are running. No polling fallback is needed.
5999    let mut registration_changes = match registry.subscribe_module_changes(&spec.module_id) {
6000        Ok(changes) => Some(changes),
6001        Err(err) => {
6002            // A poisoned registry cannot accept further writes, so it cannot
6003            // recover via a timer. Keep exit and command handling alive.
6004            warn!(module_id = %spec.module_id, error = %err, "health prober could not subscribe to registry");
6005            None
6006        }
6007    };
6008    // All restart backoffs run here, including health and operator requests.
6009    // While one is pending the loop serves commands, so disable or drain can
6010    // cancel the replacement without spawning a process just to stop it.
6011    let mut pending_respawn: Option<PendingRespawn> = None;
6012    // Commands a swap handed back to run next (see `swap::SwapEnd`). Served
6013    // before anything else so a stop that interrupted a swap runs at once.
6014    let mut requeued: VecDeque<SupervisorCommand> = VecDeque::new();
6015    loop {
6016        #[cfg(test)]
6017        {
6018            lock_snapshot(&snapshot).unwrap().actor_select = None;
6019        }
6020        #[cfg(target_os = "macos")]
6021        if let Some(active) = child.as_mut() {
6022            active.confirm_privacy_exec().await;
6023        }
6024        if let Some(scheduled) = runtime
6025            .scheduled_respawn
6026            .lock()
6027            .unwrap_or_else(|p| p.into_inner())
6028            .take()
6029        {
6030            pending_respawn = Some(scheduled);
6031        }
6032        if pending_respawn.is_some() && (child.is_some() || !respawn_still_pending(&snapshot)) {
6033            pending_respawn = None;
6034            cancel_deferred_reload(
6035                &runtime,
6036                &spec.module_id,
6037                "respawn cancelled by a supervisor command",
6038            );
6039        }
6040        if child.is_none() && pending_respawn.is_none() {
6041            cancel_deferred_reload(
6042                &runtime,
6043                &spec.module_id,
6044                "respawn cancelled before a replacement was spawned",
6045            );
6046            let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
6047                state.respawn_pending = false;
6048                state.coalesced_restart_pending = false;
6049                if matches!(
6050                    state.state,
6051                    ModuleState::Restarting
6052                        | ModuleState::Starting
6053                        | ModuleState::Draining
6054                        | ModuleState::Unresponsive
6055                ) {
6056                    error!(module_id = %spec.module_id, state = ?state.state,
6057                        "supervision operation ended without a child or pending respawn; marking failed so start can retry");
6058                    state.state = ModuleState::Failed;
6059                    clear_current_process_facts(state);
6060                }
6061            });
6062        }
6063        if let Some(command) = requeued.pop_front() {
6064            if !handle_supervisor_command(
6065                command,
6066                &mut spec,
6067                &mut runtime,
6068                &registry,
6069                &process_liveness,
6070                &snapshot,
6071                &mut child,
6072                &mut commands,
6073                &mut requeued,
6074            )
6075            .await
6076            {
6077                return;
6078            }
6079            if child.is_some() || !respawn_still_pending(&snapshot) {
6080                pending_respawn = None;
6081                let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
6082                    state.respawn_pending = false
6083                });
6084            }
6085            continue;
6086        }
6087        if child.is_some() {
6088            #[cfg(test)]
6089            {
6090                lock_snapshot(&snapshot).unwrap().actor_turns += 1;
6091            }
6092            health_probe.refresh_registration(&spec, &runtime, &registry, &snapshot);
6093            let wake_after = health_probe.wake_after();
6094            let probe_sleep = async move {
6095                match wake_after {
6096                    Some(delay) => sleep(delay).await,
6097                    None => std::future::pending().await,
6098                }
6099            };
6100            tokio::pin!(probe_sleep);
6101            // A `protocol: "none"` child (a plain process that never registers
6102            // over the subc wire, such as nats-server) with no HTTP health
6103            // check has nothing to probe, so it waits only for its exit or a
6104            // supervisor command. A rescan that adds an HTTP check is a command.
6105            let wire_child = running_protocol(&spec, &snapshot) == ModuleProtocol::Subc;
6106            #[cfg(test)]
6107            {
6108                let registration_pending = wire_child
6109                    && registration_changes
6110                        .as_ref()
6111                        .is_some_and(|changes| changes.has_changed().unwrap());
6112                let mut state = lock_snapshot(&snapshot).unwrap();
6113                state.actor_select = (!registration_pending).then_some(ActorSelectCheckpoint {
6114                    generation: state.spawn_generation,
6115                    turn: state.actor_turns,
6116                    registered_connection: health_probe.registered_connection,
6117                    next_probe_at: health_probe.next_probe_at,
6118                    wake_after,
6119                });
6120            }
6121            let active_child = child.as_mut().expect("child checked above");
6122            tokio::select! {
6123                wait_result = active_child.wait() => {
6124                    #[cfg(test)]
6125                    { lock_snapshot(&snapshot).unwrap().actor_select = None; }
6126                    // Every arm below that gives up on the CHILD must keep the
6127                    // supervision task itself alive (child = None, loop
6128                    // continues into command-serving mode). Returning here
6129                    // closes the command channel, which makes the module
6130                    // permanently unrestartable in-band: a clean child exit
6131                    // of an enabled module once wedged the fleet this way
6132                    // ('supervisor command channel is closed') and required a
6133                    // full daemon restart to recover.
6134                    let exit_report = match wait_result {
6135                        Ok(status) => classify_reaped_child_exit(&snapshot, active_child, &status),
6136                        Err(err) => {
6137                            active_child.drain_stderr(&spec.module_id).await;
6138                            fail_snapshot(&snapshot, Some(&spec.module_id), None);
6139                            // Every other exit path (on_child_exit's Clean/Crash arms,
6140                            // the reload-registration-failure path) records a terminal
6141                            // before moving on. Without one here, a module whose wait()
6142                            // itself errored (e.g. already reaped) leaves no terminal
6143                            // record at all -- an empty ring reads as "nothing died".
6144                            record_wait_error_terminal(
6145                                &spec.module_id,
6146                                &runtime.terminal_ring,
6147                                &runtime.spawn_events,
6148                            );
6149                            untrack_if_registration_released(
6150                                &process_liveness,
6151                                &registry,
6152                                &spec.module_id,
6153                                &snapshot,
6154                            );
6155                            error!(module_id = %spec.module_id, error = %err, "failed to wait for supervised module");
6156                            child = None;
6157                            continue;
6158                        }
6159                    };
6160                    active_child.drain_stderr(&spec.module_id).await;
6161
6162                    let next = on_child_exit(
6163                        &spec,
6164                        runtime.restart_policy,
6165                        &registry,
6166                        &snapshot,
6167                        &runtime.terminal_ring,
6168                        &runtime.spawn_events,
6169                        &runtime.child_roster,
6170                        exit_report,
6171                    ).await;
6172                    // The exit is recorded, so a daemon shutdown may stop
6173                    // waiting for this child (see `SupervisedChild::wait`).
6174                    active_child.release_roster();
6175                    match next {
6176                        NextAction::Stop { registration_released } => {
6177                            if registration_released {
6178                                process_liveness.untrack_if_current(&spec.module_id, &snapshot);
6179                            }
6180                            child = None;
6181                        }
6182                        NextAction::Restart { schedule } => {
6183                            let delay = schedule.map_or(
6184                                runtime.restart_policy.delay_for_restart(0),
6185                                |schedule| schedule.delay,
6186                            );
6187                            if let Some(schedule) = schedule {
6188                                log_crash_respawn(&spec.module_id, schedule);
6189                            }
6190                            // The exited child is fully recorded at this point,
6191                            // so release it and count the backoff down in the
6192                            // command-serving branch below rather than sleeping
6193                            // here: commands cannot be received from inside this
6194                            // select arm, and an operator disable or drain that
6195                            // arrives during the backoff must cancel the pending
6196                            // respawn instead of waiting for it to spawn first.
6197                            child = None;
6198                            pending_respawn = Some(PendingRespawn { deadline: Instant::now() + delay, kind: RespawnKind::Spawn });
6199                             let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = true);
6200                        }
6201                    }
6202                }
6203                command = commands.recv() => {
6204                    #[cfg(test)]
6205                    { lock_snapshot(&snapshot).unwrap().actor_select = None; }
6206                    let Some(command) = command else {
6207                        return;
6208                    };
6209                    if !handle_supervisor_command(
6210                        command,
6211                        &mut spec,
6212                        &mut runtime,
6213                        &registry,
6214                        &process_liveness,
6215                        &snapshot,
6216                        &mut child,
6217                        &mut commands,
6218                        &mut requeued,
6219                    ).await {
6220                        return;
6221                    }
6222                }
6223                _ = async {
6224                    match registration_changes.as_mut() {
6225                        Some(changes) => { let _ = changes.changed().await; }
6226                        None => std::future::pending().await,
6227                    }
6228                }, if wire_child => {
6229                    #[cfg(test)]
6230                    { lock_snapshot(&snapshot).unwrap().actor_select = None; }
6231                }
6232                _ = &mut probe_sleep => {
6233                    #[cfg(test)]
6234                    { lock_snapshot(&snapshot).unwrap().actor_select = None; }
6235                    if health_probe.due() {
6236                        run_health_probe_cycle(
6237                            &spec,
6238                            &runtime,
6239                            &registry,
6240                            &process_liveness,
6241                            &snapshot,
6242                            &mut child,
6243                        ).await;
6244                        if child.is_some() {
6245                            health_probe.schedule_next(&spec, runtime.health.cadence);
6246                        }
6247                    }
6248                }
6249            }
6250        } else if let Some(pending) = pending_respawn {
6251            tokio::select! {
6252                _ = sleep_until(pending.deadline) => {
6253                    pending_respawn = None;
6254                let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = false);
6255                    // A command handled below while the backoff elapsed may
6256                    // have stopped the module; never respawn past an operator's
6257                    // disable or drain.
6258                    if !respawn_still_pending(&snapshot) {
6259                        continue;
6260                    }
6261                    // The daemon began shutting down during the backoff: the
6262                    // spawn would be refused anyway, and refusing it here
6263                    // leaves the module stopped instead of reporting a
6264                    // failed restart.
6265                    if runtime.child_roster.is_closed() {
6266                        let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
6267                            state.state = ModuleState::Stopped;
6268                        });
6269                        debug!(module_id = %spec.module_id, "crash respawn cancelled by daemon shutdown");
6270                        continue;
6271                    }
6272                    if let Err(err) = release_dead_registration(
6273                        &registry,
6274                        runtime.forwarding.as_deref(),
6275                        &snapshot,
6276                        &spec.module_id,
6277                    ).await {
6278                        fail_snapshot(&snapshot, Some(&spec.module_id), None);
6279                        error!(module_id = %spec.module_id, error = %err, "registration did not release before restart");
6280                        continue;
6281                    }
6282
6283                    if matches!(pending.kind, RespawnKind::Reload) {
6284                        let reply = runtime.deferred_reload_reply.lock().unwrap_or_else(|p| p.into_inner()).take();
6285                        let result = finish_reload_child(&spec, &runtime, &registry, &process_liveness, &snapshot, &mut child).await;
6286                        if let Some(reply) = reply { let _ = reply.send(result); }
6287                        continue;
6288                    }
6289                    process_liveness.track(spec.module_id.clone(), Arc::clone(&snapshot));
6290                    match spawn_and_mark_running(&spec, &runtime, &snapshot) {
6291                        Ok(next_child) => {
6292                            child = Some(next_child);
6293                            debug!(module_id = %spec.module_id, "supervised module restarted after crash");
6294                        }
6295                        Err(err) => {
6296                            fail_snapshot(&snapshot, Some(&spec.module_id), None);
6297                            process_liveness.untrack_if_current(&spec.module_id, &snapshot);
6298                            error!(module_id = %spec.module_id, error = %err, "failed to restart supervised module");
6299                        }
6300                    }
6301                }
6302                command = commands.recv() => {
6303                    let Some(command) = command else {
6304                        return;
6305                    };
6306                    if !handle_supervisor_command(
6307                        command,
6308                        &mut spec,
6309                        &mut runtime,
6310                        &registry,
6311                        &process_liveness,
6312                        &snapshot,
6313                        &mut child,
6314                        &mut commands,
6315                        &mut requeued,
6316                    ).await {
6317                        return;
6318                    }
6319                    // Reconcile the pending respawn with what the command did:
6320                    // a start may already have spawned a fresh child,
6321                    // while a disable or drain moved the snapshot out of the
6322                    // state the respawn was counting down from.
6323                    if child.is_some() || !respawn_still_pending(&snapshot) {
6324                        pending_respawn = None;
6325                let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = false);
6326                    }
6327                }
6328            }
6329        } else {
6330            let Some(command) = commands.recv().await else {
6331                return;
6332            };
6333            if !handle_supervisor_command(
6334                command,
6335                &mut spec,
6336                &mut runtime,
6337                &registry,
6338                &process_liveness,
6339                &snapshot,
6340                &mut child,
6341                &mut commands,
6342                &mut requeued,
6343            )
6344            .await
6345            {
6346                return;
6347            }
6348        }
6349    }
6350}
6351
6352fn log_crash_respawn(module_id: &str, schedule: CrashRestartSchedule) {
6353    info!(
6354        module_id,
6355        restart_in_window = schedule.restart_in_window,
6356        delay_ms = schedule.delay.as_millis() as u64,
6357        "respawning after crash"
6358    );
6359}
6360
6361/// Whether the respawn a backoff was counting down to is still wanted. A
6362/// disable or drain handled while the backoff elapsed moves the snapshot out
6363/// of `Restarting`, and the operator's stop must win over the pending respawn,
6364/// so every sleep-then-spawn path re-validates against the live snapshot
6365/// instead of assuming the state it left behind still holds.
6366fn respawn_still_pending(snapshot: &SharedSnapshot) -> bool {
6367    matches!(
6368        lock_snapshot(snapshot),
6369        Ok(state) if state.enabled && state.state == ModuleState::Restarting
6370    )
6371}
6372
6373enum NextAction {
6374    Stop {
6375        registration_released: bool,
6376    },
6377    Restart {
6378        schedule: Option<CrashRestartSchedule>,
6379    },
6380}
6381
6382#[allow(clippy::too_many_arguments)]
6383async fn handle_supervisor_command(
6384    command: SupervisorCommand,
6385    spec: &mut ModuleSpec,
6386    runtime: &mut SupervisorRuntimeConfig,
6387    registry: &Arc<Registry>,
6388    process_liveness: &SupervisorProcessLiveness,
6389    snapshot: &SharedSnapshot,
6390    child: &mut Option<SupervisedChild>,
6391    commands: &mut mpsc::Receiver<SupervisorCommand>,
6392    requeued: &mut VecDeque<SupervisorCommand>,
6393) -> bool {
6394    match command {
6395        SupervisorCommand::Drain { reply } => {
6396            // A plain stop runs no forwarding drain, so nothing reaches the
6397            // module over its connection before the wait: ask by signal.
6398            let result = drain_optional_child(
6399                &spec.module_id,
6400                spec.protocol,
6401                StopNotice::NotSent,
6402                registry,
6403                runtime.forwarding.as_deref(),
6404                snapshot,
6405                &runtime.terminal_ring,
6406                &runtime.spawn_events,
6407                child,
6408                runtime.drain_timeout,
6409                ModuleState::Stopped,
6410                None,
6411            )
6412            .await;
6413            let registration_released = result.is_ok();
6414            let _ = reply.send(result);
6415            if registration_released {
6416                process_liveness.untrack_if_current(&spec.module_id, snapshot);
6417            }
6418            false
6419        }
6420        SupervisorCommand::Retire { reply } => {
6421            let result = async {
6422                let stop_notice = begin_forwarding_drain_if_configured(
6423                    spec,
6424                    runtime,
6425                    registry,
6426                    snapshot,
6427                    None,
6428                    RouteCloseReason::Disable,
6429                )
6430                .await?;
6431                drain_optional_child(
6432                    &spec.module_id,
6433                    spec.protocol,
6434                    stop_notice,
6435                    registry,
6436                    runtime.forwarding.as_deref(),
6437                    snapshot,
6438                    &runtime.terminal_ring,
6439                    &runtime.spawn_events,
6440                    child,
6441                    runtime.drain_timeout,
6442                    ModuleState::Stopped,
6443                    None,
6444                )
6445                .await
6446            }
6447            .await;
6448            let registration_released = result.is_ok();
6449            let _ = reply.send(result);
6450            if registration_released {
6451                process_liveness.untrack_if_current(&spec.module_id, snapshot);
6452            }
6453            false
6454        }
6455        SupervisorCommand::Restart {
6456            drain_timeout_ms,
6457            received_at_generation,
6458            queued_at,
6459            reply,
6460        } => {
6461            // Without this line a restart that waited in the queue (behind a
6462            // health probe cycle or another command) was invisible: the log
6463            // showed only the drain timing out, minutes after the operator's call.
6464            info!(
6465                module_id = %spec.module_id,
6466                queued_ms = u64::try_from(queued_at.elapsed().as_millis()).unwrap_or(u64::MAX),
6467                "restart command dequeued"
6468            );
6469            // ACK AT INITIATION, not completion. The blocking form deadlocked any
6470            // caller whose own request lane rides the module being restarted: the
6471            // caller's in-flight request keeps the drain from quiescing, the drain
6472            // keeps the restart from completing, and the completion keeps the reply
6473            // from releasing the caller — so the drain always timed out and cut the
6474            // initiator with a GOODBYE, even on a healthy module. Replying once the
6475            // restart is validated lets a self-lane caller settle, which is exactly
6476            // what makes the drain succeed. Completion is observable via
6477            // supervisor.list / module status; a post-ack failure lands the module
6478            // in a visible terminal state below rather than in a reply nobody can
6479            // receive.
6480            let validation = match lock_snapshot(snapshot) {
6481                Ok(state) if !state.enabled => Err(SuperviseError::Disabled {
6482                    module_id: spec.module_id.clone(),
6483                }),
6484                Ok(_) => Ok(()),
6485                Err(err) => Err(err),
6486            };
6487            let initiated = validation.is_ok();
6488            let _ = reply.send(validation);
6489            // A restart asks for a fresh process. Commands run one at a time,
6490            // so a restart queued behind another restart (two operator calls
6491            // in quick succession) is dequeued the moment the first one has
6492            // spawned its replacement -- before that process has sent HELLO.
6493            // Running it would drain and kill the process the first restart
6494            // just produced, which is the opposite of what both callers asked
6495            // for. If a process spawned after this request was received is
6496            // still supervised, the request is already satisfied. Not when the
6497            // configuration changed since that spawn: then the newer process
6498            // predates the spec this restart may exist to apply.
6499            let satisfied_by_generation = if initiated && child.is_some() {
6500                lock_snapshot(snapshot).ok().and_then(|state| {
6501                    (state.spawn_generation > received_at_generation
6502                        && !state.configuration_updated_since_spawn)
6503                        .then_some(state.spawn_generation)
6504                })
6505            } else {
6506                None
6507            };
6508            let satisfied_by_pending = initiated
6509                && child.is_none()
6510                && lock_snapshot(snapshot).ok().is_some_and(|mut state| {
6511                    let pending = state.respawn_pending && !state.configuration_updated_since_spawn;
6512                    if pending {
6513                        state.coalesced_restart_pending = true;
6514                    }
6515                    pending
6516                });
6517            if satisfied_by_pending {
6518                debug!(module_id = %spec.module_id, "restart coalesced into the pending replacement");
6519            } else if let Some(generation) = satisfied_by_generation {
6520                info!(
6521                    module_id = %spec.module_id,
6522                    received_at_generation,
6523                    "restart already satisfied by generation {generation}; not restarting again"
6524                );
6525            } else if initiated {
6526                // Precedence: this restart's operator override, else the module's
6527                // configured budget (already resolved into the runtime).
6528                let drain_timeout = drain_timeout_ms
6529                    .map(Duration::from_millis)
6530                    .unwrap_or(runtime.drain_timeout);
6531                if let Err(err) = restart_child(
6532                    spec,
6533                    runtime,
6534                    registry,
6535                    process_liveness,
6536                    snapshot,
6537                    child,
6538                    drain_timeout,
6539                )
6540                .await
6541                {
6542                    warn!(
6543                        module_id = %spec.module_id,
6544                        error = %err,
6545                        "operator restart failed after initiation ack; module state carries the outcome"
6546                    );
6547                    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6548                        state.state = ModuleState::Failed;
6549                        clear_current_process_facts(state);
6550                    });
6551                }
6552            }
6553            true
6554        }
6555        SupervisorCommand::Reload { reply } => {
6556            let result =
6557                reload_child(spec, runtime, registry, process_liveness, snapshot, child).await;
6558            if result.is_ok()
6559                && runtime
6560                    .scheduled_respawn
6561                    .lock()
6562                    .unwrap_or_else(|p| p.into_inner())
6563                    .as_ref()
6564                    .is_some_and(|pending| matches!(pending.kind, RespawnKind::Reload))
6565            {
6566                *runtime
6567                    .deferred_reload_reply
6568                    .lock()
6569                    .unwrap_or_else(|p| p.into_inner()) = Some(reply);
6570            } else {
6571                let _ = reply.send(result);
6572            }
6573            true
6574        }
6575        SupervisorCommand::SetEnabled { enabled, reply } => {
6576            let result = set_child_enabled(
6577                spec,
6578                runtime,
6579                registry,
6580                process_liveness,
6581                snapshot,
6582                child,
6583                enabled,
6584            )
6585            .await;
6586            let _ = reply.send(result);
6587            true
6588        }
6589        SupervisorCommand::UpdateConfiguration {
6590            spec: next_spec,
6591            health,
6592            drain_timeout_ms,
6593            reply,
6594        } => {
6595            if let Some(handle) = &runtime.supervisor_handle {
6596                handle.apply_identity_configuration(&next_spec);
6597            }
6598            *spec = next_spec;
6599            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6600                state.configuration_updated_since_spawn = true;
6601            });
6602            let health_changed = runtime.health != health;
6603            runtime.health = health;
6604            // Reset the cadence and old endpoint's failure streak on a live
6605            // health-policy change rather than waiting for its old deadline.
6606            if health_changed {
6607                let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6608                    state.health = ModuleHealthStatus::default();
6609                });
6610            }
6611            runtime.drain_timeout = drain_timeout_ms
6612                .map(Duration::from_millis)
6613                .unwrap_or(runtime.default_drain_timeout);
6614            *runtime
6615                .effective_drain_timeout
6616                .lock()
6617                .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
6618            let _ = reply.send(());
6619            true
6620        }
6621        SupervisorCommand::Swap {
6622            ready_timeout,
6623            reply,
6624        } => {
6625            let end = swap::run_swap(
6626                spec,
6627                runtime,
6628                registry,
6629                process_liveness,
6630                snapshot,
6631                child,
6632                commands,
6633                ready_timeout.unwrap_or(DEFAULT_SWAP_READY_TIMEOUT),
6634                reply,
6635            )
6636            .await;
6637            requeued.extend(end.requeue);
6638            true
6639        }
6640    }
6641}
6642
6643async fn restart_child(
6644    spec: &ModuleSpec,
6645    runtime: &SupervisorRuntimeConfig,
6646    registry: &Registry,
6647    process_liveness: &SupervisorProcessLiveness,
6648    snapshot: &SharedSnapshot,
6649    child: &mut Option<SupervisedChild>,
6650    drain_timeout: Duration,
6651) -> Result<(), SuperviseError> {
6652    // Restart cycles a running module; it must not silently start a disabled one.
6653    if !lock_snapshot(snapshot)?.enabled {
6654        return Err(SuperviseError::Disabled {
6655            module_id: spec.module_id.clone(),
6656        });
6657    }
6658    let stop_notice = begin_forwarding_drain_with_timeout(
6659        spec,
6660        runtime,
6661        registry,
6662        snapshot,
6663        None,
6664        RouteCloseReason::Restart,
6665        drain_timeout,
6666    )
6667    .await?;
6668
6669    if child.is_some() {
6670        drain_optional_child(
6671            &spec.module_id,
6672            spec.protocol,
6673            stop_notice,
6674            registry,
6675            runtime.forwarding.as_deref(),
6676            snapshot,
6677            &runtime.terminal_ring,
6678            &runtime.spawn_events,
6679            child,
6680            drain_timeout,
6681            ModuleState::Restarting,
6682            Some(true),
6683        )
6684        .await?;
6685    } else {
6686        update_snapshot(snapshot, Some(&spec.module_id), |state| {
6687            state.enabled = true;
6688            state.state = ModuleState::Restarting;
6689            clear_current_process_facts(state);
6690        })?;
6691        release_dead_registration(
6692            registry,
6693            runtime.forwarding.as_deref(),
6694            snapshot,
6695            &spec.module_id,
6696        )
6697        .await?;
6698    }
6699
6700    reset_restart_count(snapshot, &spec.module_id)?;
6701    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6702    schedule_respawn(
6703        runtime,
6704        snapshot,
6705        &spec.module_id,
6706        runtime.restart_policy.backoff,
6707        RespawnKind::Spawn,
6708    )
6709}
6710
6711async fn reload_child(
6712    spec: &ModuleSpec,
6713    runtime: &SupervisorRuntimeConfig,
6714    registry: &Registry,
6715    process_liveness: &SupervisorProcessLiveness,
6716    snapshot: &SharedSnapshot,
6717    child: &mut Option<SupervisedChild>,
6718) -> Result<(), SuperviseError> {
6719    // Reload cycles a running module; it must not silently start a disabled one.
6720    if !lock_snapshot(snapshot)?.enabled {
6721        return Err(SuperviseError::Disabled {
6722            module_id: spec.module_id.clone(),
6723        });
6724    }
6725    let stop_notice = begin_forwarding_drain(
6726        spec,
6727        runtime,
6728        registry,
6729        snapshot,
6730        Some(true),
6731        RouteCloseReason::Reload,
6732    )
6733    .await?;
6734
6735    if child.is_some() {
6736        drain_optional_child(
6737            &spec.module_id,
6738            spec.protocol,
6739            stop_notice,
6740            registry,
6741            runtime.forwarding.as_deref(),
6742            snapshot,
6743            &runtime.terminal_ring,
6744            &runtime.spawn_events,
6745            child,
6746            runtime.drain_timeout,
6747            ModuleState::Restarting,
6748            Some(true),
6749        )
6750        .await?;
6751    } else {
6752        update_snapshot(snapshot, Some(&spec.module_id), |state| {
6753            state.enabled = true;
6754            state.state = ModuleState::Restarting;
6755            clear_current_process_facts(state);
6756        })?;
6757        release_dead_registration(
6758            registry,
6759            runtime.forwarding.as_deref(),
6760            snapshot,
6761            &spec.module_id,
6762        )
6763        .await?;
6764    }
6765
6766    reset_restart_count(snapshot, &spec.module_id)?;
6767    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6768    schedule_respawn(
6769        runtime,
6770        snapshot,
6771        &spec.module_id,
6772        runtime.restart_policy.backoff,
6773        RespawnKind::Reload,
6774    )
6775}
6776
6777async fn finish_reload_child(
6778    spec: &ModuleSpec,
6779    runtime: &SupervisorRuntimeConfig,
6780    registry: &Registry,
6781    process_liveness: &SupervisorProcessLiveness,
6782    snapshot: &SharedSnapshot,
6783    child: &mut Option<SupervisedChild>,
6784) -> Result<(), SuperviseError> {
6785    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6786    let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
6787        Ok(next_child) => next_child,
6788        Err(err) => {
6789            return handle_reload_spawn_failure(
6790                spec,
6791                runtime,
6792                process_liveness,
6793                snapshot,
6794                child,
6795                format!("new child failed to spawn: {err}"),
6796            )
6797            .await;
6798        }
6799    };
6800    *child = Some(next_child);
6801
6802    let wait_outcome = {
6803        let active_child = child.as_mut().expect("new reload child was just stored");
6804        wait_for_registration_after_reload(
6805            registry,
6806            &spec.module_id,
6807            snapshot,
6808            active_child,
6809            REGISTRY_RELEASE_TIMEOUT,
6810        )
6811        .await?
6812    };
6813
6814    match wait_outcome {
6815        RegistrationWaitOutcome::Registered => {
6816            debug!(module_id = %spec.module_id, "supervised module reloaded and registered");
6817            Ok(())
6818        }
6819        RegistrationWaitOutcome::Exited(exit_report) => {
6820            if let Some(active_child) = child.as_mut() {
6821                active_child.drain_stderr(&spec.module_id).await;
6822            }
6823            // Keep the reaped child's roster guard until its terminal is written.
6824            // Shutdown waits on that guard, not on the child Option used for respawn.
6825            let mut exited_child = child.take().expect("exited reload child is still stored");
6826            #[cfg(test)]
6827            if let Some(gate) = &runtime.test_reload_exit_record_gate {
6828                gate.reached.notify_one();
6829                gate.resume.notified().await;
6830            }
6831            let result = handle_reload_child_registration_failure(
6832                spec,
6833                runtime,
6834                registry,
6835                process_liveness,
6836                snapshot,
6837                child,
6838                ReloadRegistrationFailure {
6839                    exit_report: registration_failure_exit_report(exit_report),
6840                    reason: exited_child
6841                        .spawn_failure
6842                        .clone()
6843                        .unwrap_or_else(|| "new child exited before registering".to_string()),
6844                },
6845            )
6846            .await;
6847            exited_child.release_roster();
6848            result
6849        }
6850        RegistrationWaitOutcome::TimedOut => {
6851            let mut timed_out_child = child
6852                .take()
6853                .expect("timed-out reload child is still running");
6854            timed_out_child
6855                .start_kill()
6856                .map_err(|source| SuperviseError::Kill {
6857                    module_id: spec.module_id.clone(),
6858                    source,
6859                })?;
6860            let status = timed_out_child
6861                .wait()
6862                .await
6863                .map_err(|source| SuperviseError::Wait {
6864                    module_id: spec.module_id.clone(),
6865                    source,
6866                })?;
6867            timed_out_child.drain_stderr(&spec.module_id).await;
6868            handle_reload_child_registration_failure(
6869                spec,
6870                runtime,
6871                registry,
6872                process_liveness,
6873                snapshot,
6874                child,
6875                ReloadRegistrationFailure {
6876                    exit_report: registration_failure_exit_report(classify_reaped_child_exit(
6877                        snapshot,
6878                        &timed_out_child,
6879                        &status,
6880                    )),
6881                    reason: format!(
6882                        "new child did not register within {:?}",
6883                        REGISTRY_RELEASE_TIMEOUT
6884                    ),
6885                },
6886            )
6887            .await
6888        }
6889    }
6890}
6891
6892async fn set_child_enabled(
6893    spec: &ModuleSpec,
6894    runtime: &SupervisorRuntimeConfig,
6895    registry: &Registry,
6896    process_liveness: &SupervisorProcessLiveness,
6897    snapshot: &SharedSnapshot,
6898    child: &mut Option<SupervisedChild>,
6899    enabled: bool,
6900) -> Result<bool, SuperviseError> {
6901    let (current_enabled, current_state, respawn_pending) = {
6902        let state = lock_snapshot(snapshot)?;
6903        (state.enabled, state.state, state.respawn_pending)
6904    };
6905    // `start` (enable on an already-enabled module) heals TERMINAL states instead
6906    // of no-op'ing: a module whose restart budget exhausted (Failed) or that exited
6907    // clean (Stopped) has no live process and no other in-band recovery — the
6908    // operator's start is the explicit recovery act and resets the budget. Without
6909    // this arm the only revival was subc-probe --supervisor-restart in a terminal,
6910    // which the 2026-07-14 aft outage proved is a trap when the failed module is
6911    // the one providing every agent's shell.
6912    let revive_terminal = enabled
6913        && current_enabled
6914        && child.is_none()
6915        && (matches!(current_state, ModuleState::Failed | ModuleState::Stopped)
6916            || (current_state == ModuleState::Restarting && !respawn_pending));
6917    if current_enabled == enabled && !revive_terminal {
6918        return Ok(false);
6919    }
6920
6921    if enabled {
6922        update_snapshot(snapshot, Some(&spec.module_id), |state| {
6923            state.enabled = true;
6924            state.state = ModuleState::Starting;
6925            clear_current_process_facts(state);
6926        })?;
6927        #[cfg(test)]
6928        if runtime.test_seed_stale_facts_before_enable_spawn {
6929            update_snapshot(snapshot, Some(&spec.module_id), |state| {
6930                state.process_alive = true;
6931                state.pid = Some(41);
6932                state.spawned_at_ms = Some(42);
6933                state.spawned_from = Some(PathBuf::from("/spawned/module"));
6934                state.spawned_file_identity = Some(SpawnedFileIdentity {
6935                    device: 43,
6936                    inode: 44,
6937                });
6938            })?;
6939        }
6940        release_dead_registration(
6941            registry,
6942            runtime.forwarding.as_deref(),
6943            snapshot,
6944            &spec.module_id,
6945        )
6946        .await?;
6947        reset_restart_count(snapshot, &spec.module_id)?;
6948        process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6949        let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
6950            Ok(next_child) => next_child,
6951            Err(err) => {
6952                if let Err(state_err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6953                    state.state = ModuleState::Failed;
6954                    clear_current_process_facts(state);
6955                }) {
6956                    error!(module_id = %spec.module_id, error = %state_err, "failed to record enable spawn failure");
6957                }
6958                process_liveness.untrack_if_current(&spec.module_id, snapshot);
6959                return Err(err);
6960            }
6961        };
6962        *child = Some(next_child);
6963        debug!(module_id = %spec.module_id, "supervised module enabled");
6964        Ok(true)
6965    } else {
6966        let stop_notice = begin_forwarding_drain_if_configured(
6967            spec,
6968            runtime,
6969            registry,
6970            snapshot,
6971            Some(false),
6972            RouteCloseReason::Disable,
6973        )
6974        .await?;
6975        drain_optional_child(
6976            &spec.module_id,
6977            spec.protocol,
6978            stop_notice,
6979            registry,
6980            runtime.forwarding.as_deref(),
6981            snapshot,
6982            &runtime.terminal_ring,
6983            &runtime.spawn_events,
6984            child,
6985            runtime.drain_timeout,
6986            ModuleState::Disabled,
6987            Some(false),
6988        )
6989        .await?;
6990        debug!(module_id = %spec.module_id, "supervised module disabled");
6991        Ok(true)
6992    }
6993}
6994
6995#[allow(clippy::too_many_arguments)]
6996async fn on_child_exit(
6997    spec: &ModuleSpec,
6998    policy: RestartPolicy,
6999    registry: &Registry,
7000    snapshot: &SharedSnapshot,
7001    terminal_ring: &Arc<Mutex<TerminalRing>>,
7002    spawn_events: &SpawnEventFeed,
7003    roster: &ChildRoster,
7004    exit_report: ExitReport,
7005) -> NextAction {
7006    // Once the daemon has begun shutting down, no exit is a crash to recover
7007    // from: the module is exiting because the daemon is going away (EOF on its
7008    // connection, or a service manager signalling the whole cgroup). Record it
7009    // as such and never schedule a respawn, which would only start a process
7010    // for the shutdown to end again.
7011    if roster.is_closed() {
7012        return on_child_exit_during_daemon_shutdown(
7013            spec,
7014            registry,
7015            snapshot,
7016            terminal_ring,
7017            spawn_events,
7018            exit_report,
7019        )
7020        .await;
7021    }
7022    // Every stop the supervisor itself asks for (operator stop, disable,
7023    // restart, reload, swap, a health restart, a drain that runs out of budget)
7024    // takes the child out of the supervise loop and reaps it in
7025    // `drain_child_to_state`, and daemon shutdown is handled above. So an exit
7026    // that reaches this point was not requested by the daemon.
7027    //
7028    // For a subc-wire module a clean exit is still a stop: those modules are
7029    // written to re-raise SIGTERM, so a stray outside signal already reads as a
7030    // crash, and exiting 0 is a deliberate choice the module made. A
7031    // `protocol: "none"` module is a stock program we cannot change, and many
7032    // of them (nats-server among them) exit 0 on SIGTERM. Treating that as a
7033    // stop would leave the module down for good after any stray signal, so it
7034    // goes through the crash path instead: it spends restart budget, respawns
7035    // with the crash backoff, and ends `failed` when the budget runs out.
7036    let unrequested_clean_exit_of_protocol_none = exit_report.kind == ExitKind::Clean
7037        && running_protocol(spec, snapshot) == ModuleProtocol::None;
7038    match exit_report.kind {
7039        ExitKind::Clean if !unrequested_clean_exit_of_protocol_none => {
7040            info!(
7041                module_id = %spec.module_id,
7042                exit_code = ?exit_report.code,
7043                exit_signal = ?exit_report.signal,
7044                "supervised module exited cleanly"
7045            );
7046            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
7047                state.state = ModuleState::Stopped;
7048                clear_current_process_facts(state);
7049                state.last_exit = Some(exit_report.clone());
7050            }) {
7051                error!(module_id = %spec.module_id, error = %err, "failed to record clean module exit");
7052            }
7053            record_terminal(
7054                &spec.module_id,
7055                terminal_ring,
7056                spawn_events,
7057                &exit_report,
7058                TerminalDisposition::Stopped,
7059            );
7060            let registration_released = match wait_for_registration_release(
7061                registry,
7062                &spec.module_id,
7063                REGISTRY_RELEASE_TIMEOUT,
7064            )
7065            .await
7066            {
7067                Ok(()) => true,
7068                Err(err) => {
7069                    warn!(module_id = %spec.module_id, error = %err, "registration still active after clean exit");
7070                    false
7071                }
7072            };
7073            NextAction::Stop {
7074                registration_released,
7075            }
7076        }
7077        ExitKind::Clean | ExitKind::Crash => {
7078            if unrequested_clean_exit_of_protocol_none {
7079                warn!(
7080                    module_id = %spec.module_id,
7081                    exit_code = ?exit_report.code,
7082                    exit_signal = ?exit_report.signal,
7083                    "protocol-none module exited cleanly without a stop request; handling it as a crash"
7084                );
7085            } else {
7086                warn!(
7087                    module_id = %spec.module_id,
7088                    exit_code = ?exit_report.code,
7089                    exit_signal = ?exit_report.signal,
7090                    "supervised module exited abnormally (crash)"
7091                );
7092            }
7093            let mut restart_schedule = None;
7094            let mut disposition = TerminalDisposition::Disabled;
7095            // Set only when the budget is what stopped the module, so the
7096            // terminal record says which limit was hit rather than leaving
7097            // `failed` to be read as "crashed once, badly".
7098            let mut disposition_detail = lock_snapshot(snapshot)
7099                .ok()
7100                .and_then(|mut state| state.spawn_failure.take());
7101            let now = Instant::now();
7102            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
7103                clear_current_process_facts(state);
7104                state.last_exit = Some(exit_report.clone());
7105                if state.enabled {
7106                    if let Some(schedule) = state.next_crash_restart(&policy, now) {
7107                        state.state = ModuleState::Restarting;
7108                        restart_schedule = Some(schedule);
7109                        disposition = TerminalDisposition::Restarting;
7110                    } else {
7111                        disposition = TerminalDisposition::Failed;
7112                        let budget = policy.budget_exhausted_detail();
7113                        disposition_detail =
7114                            Some(disposition_detail.take().map_or_else(
7115                                || budget.clone(),
7116                                |cause| format!("{cause}; {budget}"),
7117                            ));
7118                    }
7119                } else {
7120                    state.state = ModuleState::Disabled;
7121                    disposition = TerminalDisposition::Disabled;
7122                }
7123            }) {
7124                error!(module_id = %spec.module_id, error = %err, "failed to record crashed module exit");
7125                return NextAction::Stop {
7126                    registration_released: false,
7127                };
7128            }
7129            if disposition == TerminalDisposition::Failed {
7130                // The window is in the message, not only in the fields: this line
7131                // is read in a scrollback where a bare `max_restarts=3` reads as a
7132                // lifetime cap and sends the operator looking for three crashes
7133                // that never happened together.
7134                error!(
7135                    module_id = %spec.module_id,
7136                    max_restarts = policy.max_restarts,
7137                    window_secs = policy.window.as_secs(),
7138                    "module stopped: {}",
7139                    policy.budget_exhausted_detail()
7140                );
7141            }
7142            let budget_exhausted = disposition == TerminalDisposition::Failed;
7143            let record_exit = || {
7144                record_terminal_with_detail(
7145                    &spec.module_id,
7146                    terminal_ring,
7147                    spawn_events,
7148                    &exit_report,
7149                    disposition,
7150                    disposition_detail,
7151                );
7152            };
7153            if budget_exhausted {
7154                // Publish Failed only after its terminal record is available.
7155                // Recording takes the event-feed lock, then ring -> journal
7156                // writer (with file I/O), all without the hot snapshot lock.
7157                // No lock is held when the final snapshot update runs, nor
7158                // across the registration-release await below. Commands and
7159                // health actions run on this same supervisor task, so none can
7160                // act on the old state during the write; process facts already
7161                // say the child is dead to concurrent liveness readers.
7162                // A journal error is retained in history, not returned. If
7163                // recording panics, still publish Failed before resuming the
7164                // original unwind rather than leaving a dead child Running.
7165                let recorded = std::panic::catch_unwind(std::panic::AssertUnwindSafe(record_exit));
7166                if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
7167                    state.state = ModuleState::Failed;
7168                }) {
7169                    error!(module_id = %spec.module_id, error = %err, "failed to publish exhausted restart budget");
7170                }
7171                if let Err(panic) = recorded {
7172                    std::panic::resume_unwind(panic);
7173                }
7174            } else {
7175                record_exit();
7176            }
7177
7178            if let Some(schedule) = restart_schedule {
7179                NextAction::Restart {
7180                    schedule: Some(schedule),
7181                }
7182            } else {
7183                let registration_released = match wait_for_registration_release(
7184                    registry,
7185                    &spec.module_id,
7186                    REGISTRY_RELEASE_TIMEOUT,
7187                )
7188                .await
7189                {
7190                    Ok(()) => true,
7191                    Err(err) => {
7192                        warn!(module_id = %spec.module_id, error = %err, "registration still active after failed module");
7193                        false
7194                    }
7195                };
7196                NextAction::Stop {
7197                    registration_released,
7198                }
7199            }
7200        }
7201        ExitKind::DeliberateSeverance => {
7202            warn!(
7203                module_id = %spec.module_id,
7204                exit_code = ?exit_report.code,
7205                exit_signal = ?exit_report.signal,
7206                "supervised module exited after deliberate connection severance"
7207            );
7208            let mut should_restart = false;
7209            let mut disposition = TerminalDisposition::Disabled;
7210            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
7211                clear_current_process_facts(state);
7212                state.last_exit = Some(exit_report.clone());
7213                state.lifetime_restarts += 1;
7214                if state.enabled {
7215                    state.state = ModuleState::Restarting;
7216                    should_restart = true;
7217                    disposition = TerminalDisposition::Restarting;
7218                } else {
7219                    state.state = ModuleState::Disabled;
7220                }
7221            }) {
7222                error!(module_id = %spec.module_id, error = %err, "failed to record deliberately severed module exit");
7223                return NextAction::Stop {
7224                    registration_released: false,
7225                };
7226            }
7227            record_terminal(
7228                &spec.module_id,
7229                terminal_ring,
7230                spawn_events,
7231                &exit_report,
7232                disposition,
7233            );
7234
7235            if should_restart {
7236                NextAction::Restart { schedule: None }
7237            } else {
7238                let registration_released = match wait_for_registration_release(
7239                    registry,
7240                    &spec.module_id,
7241                    REGISTRY_RELEASE_TIMEOUT,
7242                )
7243                .await
7244                {
7245                    Ok(()) => true,
7246                    Err(err) => {
7247                        warn!(module_id = %spec.module_id, error = %err, "registration still active after deliberately severed module exit");
7248                        false
7249                    }
7250                };
7251                NextAction::Stop {
7252                    registration_released,
7253                }
7254            }
7255        }
7256    }
7257}
7258
7259async fn on_child_exit_during_daemon_shutdown(
7260    spec: &ModuleSpec,
7261    registry: &Registry,
7262    snapshot: &SharedSnapshot,
7263    terminal_ring: &Arc<Mutex<TerminalRing>>,
7264    spawn_events: &SpawnEventFeed,
7265    exit_report: ExitReport,
7266) -> NextAction {
7267    info!(
7268        module_id = %spec.module_id,
7269        exit_code = ?exit_report.code,
7270        exit_signal = ?exit_report.signal,
7271        exit_kind = ?exit_report.kind,
7272        "supervised module exited during daemon shutdown; not restarting it"
7273    );
7274    if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
7275        state.state = ModuleState::Stopped;
7276        clear_current_process_facts(state);
7277        state.last_exit = Some(exit_report.clone());
7278    }) {
7279        error!(module_id = %spec.module_id, error = %err, "failed to record module exit during daemon shutdown");
7280    }
7281    record_terminal(
7282        &spec.module_id,
7283        terminal_ring,
7284        spawn_events,
7285        &exit_report,
7286        TerminalDisposition::DaemonShutdown,
7287    );
7288    let registration_released =
7289        wait_for_registration_release(registry, &spec.module_id, REGISTRY_RELEASE_TIMEOUT)
7290            .await
7291            .is_ok();
7292    NextAction::Stop {
7293        registration_released,
7294    }
7295}
7296
7297fn record_wait_error_terminal(
7298    module_id: &str,
7299    terminal_ring: &Arc<Mutex<TerminalRing>>,
7300    spawn_events: &SpawnEventFeed,
7301) {
7302    record_terminal(
7303        module_id,
7304        terminal_ring,
7305        spawn_events,
7306        &wait_error_exit_report(),
7307        TerminalDisposition::Failed,
7308    );
7309}
7310
7311fn record_terminal(
7312    module_id: &str,
7313    terminal_ring: &Arc<Mutex<TerminalRing>>,
7314    spawn_events: &SpawnEventFeed,
7315    exit_report: &ExitReport,
7316    disposition: TerminalDisposition,
7317) {
7318    record_terminal_with_detail(
7319        module_id,
7320        terminal_ring,
7321        spawn_events,
7322        exit_report,
7323        disposition,
7324        None,
7325    );
7326}
7327
7328/// The ring lock is held only to capture the read (see
7329/// `TerminalJournal::capture_read`), so this module's exits keep recording
7330/// while the journal files are read. Blocking: it reads files.
7331fn durable_terminal_history_of(
7332    terminal_ring: &Mutex<TerminalRing>,
7333    module_id: &str,
7334) -> subc_control::TerminalHistory {
7335    let read = terminal_ring
7336        .lock()
7337        .unwrap_or_else(|p| p.into_inner())
7338        .capture_durable_history();
7339    read.read(module_id)
7340}
7341
7342fn record_terminal_with_detail(
7343    module_id: &str,
7344    terminal_ring: &Arc<Mutex<TerminalRing>>,
7345    spawn_events: &SpawnEventFeed,
7346    exit_report: &ExitReport,
7347    disposition: TerminalDisposition,
7348    disposition_detail: Option<String>,
7349) {
7350    spawn_events.emit_exited(module_id, exit_report.code, exit_report.signal);
7351    let record = TerminalRecord {
7352        exit_code: exit_report.code,
7353        exit_signal: exit_report.signal,
7354        at_ms: exit_report.at_ms,
7355        disposition,
7356        exit_kind: exit_report.kind.into(),
7357        disposition_detail,
7358    };
7359    terminal_ring
7360        .lock()
7361        .unwrap_or_else(|poisoned| poisoned.into_inner())
7362        .record_exit(module_id, record);
7363}
7364
7365fn untrack_if_registration_released(
7366    process_liveness: &SupervisorProcessLiveness,
7367    registry: &Registry,
7368    module_id: &str,
7369    snapshot: &SharedSnapshot,
7370) {
7371    match registry.get_module(module_id) {
7372        Ok(None) => process_liveness.untrack_if_current(module_id, snapshot),
7373        Ok(Some(_)) => {}
7374        Err(err) => {
7375            warn!(module_id, error = %err, "could not determine whether supervisor liveness can be untracked");
7376        }
7377    }
7378}
7379
7380/// The child's environment plan: inherit the parent's, drop ambient `CK_LOG`,
7381/// then apply the module's configured entries minus daemon-private capture keys.
7382///
7383/// Separated from `spawn_child` only so it can be asserted without spawning a
7384/// process — a duplicate of this logic in a test would pass while the real one
7385/// drifted, which is the defect class this function exists to avoid.
7386/// The subc-wire half of a spawn: `--subc <connection file>` and the launch
7387/// nonce. A `protocol: "none"` module gets neither, because it cannot use
7388/// either and the argument would stop a stock binary from starting at all.
7389/// `SUBC_MODULE_ID` is set on every path since an unread variable is inert.
7390///
7391/// The plain-spawn form, kept for the tests that assert its plan; spawns go
7392/// through [`apply_wire_spawn_args_for_role`].
7393#[cfg(test)]
7394fn apply_wire_spawn_args(
7395    command: &mut Command,
7396    spec: &ModuleSpec,
7397    connection_file_path: Option<&std::path::Path>,
7398    handle: Option<&SupervisorHandle>,
7399) -> Result<Option<NonceHandoff>, SuperviseError> {
7400    apply_wire_spawn_args_for_role(
7401        command,
7402        spec,
7403        connection_file_path,
7404        handle,
7405        SpawnRole::Plain,
7406    )
7407}
7408
7409/// The read end of a spawn's launch-nonce pipe, prepared by
7410/// [`apply_wire_spawn_args_for_role`] and installed as the child's descriptor 3
7411/// by the last pre-exec step, just before `spawn()`. Windows has no descriptor
7412/// handoff and keeps only the environment copy.
7413#[cfg(unix)]
7414type NonceHandoff = subc_os::LaunchNonceHandoff;
7415#[cfg(not(unix))]
7416type NonceHandoff = std::convert::Infallible;
7417
7418/// Prepare wire identity for a plain spawn or a swap candidate.
7419///
7420/// A plain spawn replaces the module's recorded nonce. A swap candidate records
7421/// a separate candidate token so the still-serving incumbent and its consumers
7422/// keep their nonce. Both records are installed before the process exists, so
7423/// the child's initial HELLO registration cannot arrive ahead of its nonce.
7424///
7425/// On Unix the nonce is delivered only through a pipe. It is written into
7426/// a pipe whose read end the child gets as descriptor 3, named by
7427/// `SUBC_LAUNCH_NONCE_FD=3:<pipe inode>`: unlike the environment, another
7428/// process of the same user cannot read it with `ps eww`. That handoff is
7429/// returned rather than installed here, because installing it replaces
7430/// whatever the child has at descriptor 3 and so must be the last pre-exec
7431/// step, after the Linux cgroup placement that the caller registers later.
7432/// Windows retains the environment handoff until restricted handle inheritance
7433/// can be implemented outside std's process primitives.
7434fn apply_wire_spawn_args_for_role(
7435    command: &mut Command,
7436    spec: &ModuleSpec,
7437    connection_file_path: Option<&std::path::Path>,
7438    handle: Option<&SupervisorHandle>,
7439    role: SpawnRole,
7440) -> Result<Option<NonceHandoff>, SuperviseError> {
7441    command.env(SUBC_MODULE_ID_ENV, &spec.module_id);
7442    // SUBC_LAUNCH_NONCE_FD is removed for every child, `protocol: "none"`
7443    // included: a daemon started from a module's process tree inherits it,
7444    // and passing it on would point the child at a descriptor it does not
7445    // have.
7446    command.env_remove(subc_os::LAUNCH_NONCE_FD_ENV);
7447    // Remove inherited or configured copies too: withholding must mean absent.
7448    command.env_remove(SUBC_LAUNCH_NONCE_ENV);
7449    if spec.protocol == ModuleProtocol::None {
7450        return Ok(None);
7451    }
7452    if let Some(connection_file_path) = connection_file_path {
7453        command.arg(SUBC_ARG).arg(connection_file_path);
7454    }
7455
7456    // Every subc-wire spawn receives a fresh one-time launch nonce for consumer
7457    // route.open attestation. Reserved modules additionally use the same nonce
7458    // for HELLO id-squatting protection. A respawn rotates both records.
7459    let nonce = generate_launch_nonce()?;
7460    if let Some(handle) = handle {
7461        match role {
7462            SpawnRole::Plain => {
7463                handle.set_spawn_nonce(&spec.module_id, nonce.clone());
7464                if spec.reserved {
7465                    handle.set_reserved_nonce(&spec.module_id, nonce.clone());
7466                }
7467            }
7468            SpawnRole::SwapCandidate => handle.open_swap(&spec.module_id, nonce.clone()),
7469        }
7470    }
7471    #[cfg(unix)]
7472    let handoff = {
7473        let handoff =
7474            subc_os::LaunchNonceHandoff::new(&nonce).map_err(|source| SuperviseError::Spawn {
7475                program: spec.program.clone(),
7476                source,
7477                cgroup_path: None,
7478            })?;
7479        command.env(subc_os::LAUNCH_NONCE_FD_ENV, handoff.fd_env_value());
7480        Some(handoff)
7481    };
7482    #[cfg(not(unix))]
7483    let handoff = None;
7484    // Windows keeps the environment copy: std cannot restrict an inherited pipe
7485    // handle to this child without leaking it to concurrently spawned processes.
7486    #[cfg(not(unix))]
7487    command.env(SUBC_LAUNCH_NONCE_ENV, nonce);
7488    Ok(handoff)
7489}
7490
7491fn apply_child_env(command: &mut Command, spec: &ModuleSpec) {
7492    command.env_remove(CK_LOG_ENV);
7493    // The spawn role is the supervisor's to set, and only on a swap candidate
7494    // (see `apply_spawn_role`). Removing it here, rather than just not setting
7495    // it, is what makes it absent on a plain spawn: the daemon's own
7496    // environment could carry it, and so could a spec built outside daemon
7497    // config (config refuses it as an `env` key). A module reading it on a
7498    // plain restart would pick the long swap budget and leave callers waiting.
7499    command.env_remove(SUBC_SPAWN_ROLE_ENV);
7500    for (key, value) in &spec.env {
7501        // cortexkit-log currently exposes retention only as a Rust struct, not
7502        // environment names. These values are daemon-private sink metadata and
7503        // must never become a public child-process contract by being inherited.
7504        if matches!(
7505            key.as_str(),
7506            CAPTURE_MAX_FILE_MB_ENV | CAPTURE_KEEP_ENV | CAPTURE_MAX_AGE_DAYS_ENV
7507        ) || key == SUBC_SPAWN_ROLE_ENV
7508        {
7509            continue;
7510        }
7511        command.env(key, value);
7512    }
7513}
7514
7515/// Which slot a spawn fills: the module's ordinary one, or the candidate slot
7516/// of a blue/green swap.
7517#[derive(Debug, Clone, Copy, PartialEq, Eq)]
7518enum SpawnRole {
7519    Plain,
7520    SwapCandidate,
7521}
7522
7523/// Set the spawn role for a swap candidate. A plain spawn gets nothing here;
7524/// `apply_child_env` has already removed the variable for every spawn.
7525fn apply_spawn_role(command: &mut Command, role: SpawnRole) {
7526    if role == SpawnRole::SwapCandidate {
7527        command.env(SUBC_SPAWN_ROLE_ENV, SPAWN_ROLE_SWAP_CANDIDATE);
7528    }
7529}
7530
7531fn spawn_child(
7532    spec: &ModuleSpec,
7533    connection_file_path: Option<&std::path::Path>,
7534    handle: Option<&SupervisorHandle>,
7535    ring: &Arc<Mutex<StderrRing>>,
7536    capture_logs_dir: Option<&std::path::Path>,
7537    roster: &ChildRoster,
7538    #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
7539) -> Result<SupervisedChild, SuperviseError> {
7540    spawn_child_in_slot(
7541        spec,
7542        connection_file_path,
7543        handle,
7544        ring,
7545        capture_logs_dir,
7546        roster,
7547        #[cfg(target_os = "linux")]
7548        cgroup_placement,
7549        SpawnRole::Plain,
7550        false,
7551    )
7552}
7553
7554/// Spawn one process of `spec` into a slot.
7555///
7556/// `alternate_slot` picks the process's cgroup label on Linux (see `swap::cgroup_name`).
7557/// A swap candidate needs a different cgroup from the process it is replacing,
7558/// which is still alive: in the same cgroup the two would be one kill domain,
7559/// and killing a failed candidate could take the incumbent with it.
7560///
7561/// The stderr capture file is `<module_id>.stderr.log` for every process of
7562/// the module, whichever slot it is in, because that is the one file
7563/// `ck module logs` reads. During a swap's overlap both processes append to it;
7564/// the daemon writes whole lines, so the two interleave by line, which is also
7565/// the merged view an operator wants while a swap runs.
7566#[allow(clippy::too_many_arguments)]
7567fn spawn_child_in_slot(
7568    spec: &ModuleSpec,
7569    connection_file_path: Option<&std::path::Path>,
7570    handle: Option<&SupervisorHandle>,
7571    ring: &Arc<Mutex<StderrRing>>,
7572    capture_logs_dir: Option<&std::path::Path>,
7573    roster: &ChildRoster,
7574    #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
7575    role: SpawnRole,
7576    alternate_slot: bool,
7577) -> Result<SupervisedChild, SuperviseError> {
7578    if roster.is_closed() {
7579        return Err(SuperviseError::Spawn {
7580            program: spec.program.clone(),
7581            source: io::Error::other("the daemon is shutting down; not starting a new process"),
7582            cgroup_path: None,
7583        });
7584    }
7585    #[cfg(target_os = "linux")]
7586    let cgroup_name = {
7587        // Slot names alone are not kill domains: a retired incumbent may still
7588        // be draining when a later enable/restart spawns into the same slot.
7589        // Decimal entropy keeps the suffix unambiguous; Placement performs
7590        // the module-id escaping and constructs the filesystem path.
7591        if cgroup_placement.is_none() {
7592            swap::cgroup_name(&spec.module_id, alternate_slot)
7593        } else {
7594            let nonce = generate_launch_nonce()?;
7595            let suffix = u128::from_str_radix(&nonce[..32], 16).expect("hex launch nonce");
7596            // Leave room for byte escaping and the suffix under NAME_MAX. The
7597            // label is only for humans; the nonce identifies the kill domain.
7598            let mut end = spec.module_id.len().min(64);
7599            while !spec.module_id.is_char_boundary(end) {
7600                end -= 1;
7601            }
7602            format!(
7603                "{}_{suffix}",
7604                swap::cgroup_name(&spec.module_id[..end], alternate_slot)
7605            )
7606        }
7607    };
7608    #[cfg(not(target_os = "linux"))]
7609    let _ = alternate_slot;
7610    #[cfg(target_os = "macos")]
7611    let (mut command, privacy_exec, exec_ack) = privacy_command(spec, roster)?;
7612    #[cfg(not(target_os = "macos"))]
7613    let mut command = Command::new(&spec.program);
7614    command.args(&spec.args);
7615    // AMBIENT `CK_LOG` MUST NOT LEAK INTO AN OTHERWISE UNCONFIGURED MODULE — but
7616    // that is the whole of the intent, so remove that one key rather than the
7617    // environment.
7618    //
7619    // This was `env_clear()` from 0.17.41 until 0.18.3, which achieved the goal
7620    // and took the POSIX environment with it. Modules spawned that way had no
7621    // HOME, XDG_RUNTIME_DIR, TMPDIR or USER, and the consequences ran past
7622    // logging:
7623    //
7624    //   * `connection_file::discover` reads XDG_RUNTIME_DIR and HOME, so with
7625    //     both unset it fell back to the temp dir alone and `ck` could not find
7626    //     a daemon running on the same machine from inside any module's process
7627    //     tree — reporting a path the file has never lived at, which reads as
7628    //     "the daemon did not write its file".
7629    //   * `default_data_home()` with HOME and XDG_DATA_HOME both unset returns
7630    //     the RELATIVE `.local/share`, so a module deriving its own store path
7631    //     resolved it against its own CWD. That is the store-fragmentation
7632    //     defect the daemon already refuses in config (`parse_doc` rejects a
7633    //     relative `storage.data_home`) arriving by derivation instead.
7634    //   * anything a module spawns inherited it: git without ~/.gitconfig,
7635    //     cargo without CARGO_HOME, ssh, python user dirs — all degrading
7636    //     quietly rather than erroring.
7637    //
7638    // Reported by iceteaSA as #104 after deploying 0.18.2, where `ck daemon`
7639    // offered one candidate under /tmp while the file sat in /run/user/1000.
7640    //
7641    // A configured module is unaffected either way: `module_spec()` puts the
7642    // resolved CK_LOG into `spec.env`, which is applied below and therefore
7643    // wins over anything ambient.
7644    apply_child_env(&mut command, spec);
7645    apply_spawn_role(&mut command, role);
7646    let nonce_handoff =
7647        apply_wire_spawn_args_for_role(&mut command, spec, connection_file_path, handle, role)?;
7648
7649    #[cfg(target_os = "linux")]
7650    let cgroup_path = cgroup_placement
7651        .map(|placement| placement.module_path(&cgroup_name))
7652        .transpose()
7653        .map_err(|source| SuperviseError::Cgroup {
7654            module_id: spec.module_id.clone(),
7655            source,
7656        })?;
7657    #[cfg(not(target_os = "linux"))]
7658    let cgroup_path: Option<PathBuf> = None;
7659    #[cfg(target_os = "linux")]
7660    if let Some(path) = &cgroup_path {
7661        if let Err(error) = apply_cgroup_placement(&mut command, spec, path) {
7662            if let Some(placement) = cgroup_placement {
7663                remove_module_cgroup(placement, &cgroup_name);
7664            }
7665            return Err(error);
7666        }
7667    }
7668
7669    let output_sink = if let Some(logs_dir) = capture_logs_dir {
7670        let path = logs_dir.join(format!("{}.stderr.log", spec.module_id));
7671        match ChildOutputSink::open(&path, capture_retention(spec)) {
7672            Ok(sink) => sink,
7673            Err(error) => {
7674                warn!(
7675                    module_id = %spec.module_id,
7676                    path = %path.display(),
7677                    error = %error,
7678                    "could not open child output capture file; forwarding to stderr"
7679                );
7680                ChildOutputSink::Stderr
7681            }
7682        }
7683    } else {
7684        ChildOutputSink::Stderr
7685    };
7686
7687    command.stdout(Stdio::piped());
7688    command.stderr(Stdio::piped());
7689    command.kill_on_drop(true);
7690    // EACH MODULE LEADS ITS OWN PROCESS GROUP (the child calls setpgid(0, 0)
7691    // before exec). In the daemon's group, a service manager that kills the
7692    // job's process group when the daemon exits (launchd's default) killed
7693    // every module at the same moment its control connection closed, so no
7694    // module ever ran its EOF teardown on a daemon stop. Outside that group a
7695    // module is reached only by the daemon: the EOF it sees when its
7696    // connection closes, and the bounded stop in `child_roster` for anything
7697    // still running after that. On Linux this composes with the cgroup
7698    // placement above: that is a pre_exec write to cgroup.procs, std performs
7699    // setpgid in the child before running pre_exec callbacks, and the two
7700    // change independent process attributes.
7701    //
7702    // stdin is /dev/null because a process outside the terminal's foreground
7703    // group is stopped (SIGTTIN) if it reads the terminal, which a daemon run
7704    // by hand would otherwise hand down. Under a service manager stdin is
7705    // already /dev/null.
7706    #[cfg(unix)]
7707    command.process_group(0);
7708    command.stdin(Stdio::null());
7709    // The LAST pre-exec step, after the cgroup placement above: installing the
7710    // nonce at descriptor 3 replaces whatever the child had there, which could
7711    // be the descriptor an earlier step writes through.
7712    #[cfg(unix)]
7713    if let Some(handoff) = nonce_handoff {
7714        handoff.install_last(command.as_std_mut());
7715    }
7716    #[cfg(not(unix))]
7717    let _ = nonce_handoff;
7718
7719    // Containment, step 1 of 3 (issue #109): create the child suspended so it
7720    // cannot run a single instruction -- and therefore cannot spawn a
7721    // grandchild -- before it is in the job. See `contain_spawned_child` for the
7722    // other two steps and why the window matters.
7723    #[cfg(windows)]
7724    subc_jobobject::suspend_on_create_async(&mut command);
7725    let mut child = match command.spawn() {
7726        Ok(child) => child,
7727        Err(source) => {
7728            #[cfg(target_os = "linux")]
7729            if let Some(placement) = cgroup_placement {
7730                remove_module_cgroup(placement, &cgroup_name);
7731            }
7732            return Err(SuperviseError::Spawn {
7733                program: spec.program.clone(),
7734                source,
7735                cgroup_path,
7736            });
7737        }
7738    };
7739    // The parent must close its writer now: the acknowledgement pipe reports EOF
7740    // only when every writer is gone, and the child's copy closes when the
7741    // trampoline replaces itself with the module. Command holds only an integer
7742    // in its pre_exec callback, not another writer.
7743    #[cfg(target_os = "macos")]
7744    drop(exec_ack);
7745
7746    // Containment, steps 2 and 3: assign while suspended, then resume.
7747    #[cfg(windows)]
7748    let job = contain_spawned_child(&child, spec)?;
7749    let spawned_at_ms = unix_ms_now();
7750    let spawned_from = spec.program.clone();
7751    let spawned_file_identity = spawned_file_identity(&spawned_from);
7752    let pid = child.id().ok_or_else(|| SuperviseError::Spawn {
7753        program: spec.program.clone(),
7754        source: io::Error::other("spawned child exposed no live pid"),
7755        cgroup_path: cgroup_path.clone(),
7756    })?;
7757    let process_start_time = crate::provenance::process_start_time(pid);
7758    let process_identity = process_start_time.map(|start_time| ProcessIdentity { pid, start_time });
7759    #[cfg(all(test, target_os = "macos"))]
7760    privacy_exec_boundary_tests::before_image_sample(spec, pid);
7761    // Unix spawn returns after exec's error pipe closes. The kernel image is
7762    // therefore the executable to compare during a future orphan sweep: PATH
7763    // lookup and shebang interpretation may select a different file from the
7764    // configured program. Keep the literal program's identity for provenance,
7765    // but never use it as proof that a recorded pid may be signalled.
7766    let recorded_image = observe_spawned_image(pid);
7767    // spawn() confirms only the first exec, into the trampoline. Never persist
7768    // the trampoline image; the asynchronous acknowledgement publishes the
7769    // module image once the trampoline has replaced itself with the module.
7770    #[cfg(target_os = "macos")]
7771    let recorded_image = if privacy_exec.is_some() {
7772        None
7773    } else {
7774        recorded_image
7775    };
7776    #[cfg(target_os = "linux")]
7777    let recorded_cgroup_name = cgroup_path.as_ref().map(|_| cgroup_name.clone());
7778    #[cfg(not(target_os = "linux"))]
7779    let recorded_cgroup_name = None;
7780    let roster_guard = roster.admit(
7781        spec.module_id.clone(),
7782        pid,
7783        spec.protocol,
7784        process_start_time,
7785        crate::child_roster::RecordedIdentity {
7786            start_time: recorded_image.map(|image| image.start_time),
7787            executable: recorded_image
7788                .and_then(|image| image.executable)
7789                .map(crate::live_children::ExecutableIdentity::from),
7790            cgroup_name: recorded_cgroup_name,
7791            #[cfg(target_os = "linux")]
7792            cgroup_placement: cgroup_placement.cloned(),
7793        },
7794    );
7795    // The check at the top of this function can pass just before daemon
7796    // shutdown begins, and the process is only in the roster from here on.
7797    // The shutdown stop returns as soon as it finds the roster empty, so a
7798    // process admitted after that look would outlive the daemon. The roster
7799    // is closed before the stop first reads it and admission happens under
7800    // the roster's lock, so either the stop sees this process or this check
7801    // sees the roster closed: end the process now rather than start a module
7802    // the daemon is about to stop.
7803    if roster.is_closed() {
7804        // This child was never admitted, so there is no module protocol shutdown to wait for.
7805        #[cfg(target_os = "linux")]
7806        kill_module_cgroup(cgroup_placement, &cgroup_name);
7807        if let Err(error) = child.start_kill() {
7808            debug!(module_id = %spec.module_id, pid, %error, "kill of a process spawned during daemon shutdown failed; it may already have exited");
7809        }
7810        #[cfg(target_os = "linux")]
7811        if let Some(placement) = cgroup_placement {
7812            // This spawn was never admitted, so shutdown has no roster entry
7813            // to await. Do not detach its cleanup: the runtime could exit
7814            // before that task reaps the rejected child and removes its group.
7815            while matches!(child.try_wait(), Ok(None)) {
7816                std::thread::yield_now();
7817            }
7818            if matches!(
7819                subc_cgroup::kill_module(Some(placement), &cgroup_name),
7820                subc_cgroup::KillOutcome::Killed
7821            ) {
7822                if let Ok(path) = placement.module_path(&cgroup_name) {
7823                    while std::fs::read_to_string(path.join("cgroup.events"))
7824                        .ok()
7825                        .is_some_and(|events| events.lines().any(|line| line == "populated 1"))
7826                    {
7827                        std::thread::yield_now();
7828                    }
7829                }
7830            }
7831            remove_module_cgroup(placement, &cgroup_name);
7832        }
7833        drop(roster_guard);
7834        return Err(SuperviseError::Spawn {
7835            program: spec.program.clone(),
7836            source: io::Error::other(
7837                "the daemon began shutting down while this process was starting; ended it",
7838            ),
7839            cgroup_path,
7840        });
7841    }
7842
7843    let stdout_pump = match child.stdout.take() {
7844        Some(stdout) => Some(tokio::spawn(pump_stdout_to(stdout, output_sink.clone()))),
7845        None => {
7846            warn!(
7847                module_id = %spec.module_id,
7848                "spawned child exposed no stdout pipe; file capture will be incomplete"
7849            );
7850            None
7851        }
7852    };
7853    let stderr_pump = match child.stderr.take() {
7854        Some(stderr) => {
7855            let generation = ring
7856                .lock()
7857                .unwrap_or_else(|poisoned| poisoned.into_inner())
7858                .begin_process();
7859            Some(StderrPump {
7860                task: tokio::spawn(pump_stderr_to(
7861                    stderr,
7862                    Arc::clone(ring),
7863                    generation,
7864                    output_sink,
7865                )),
7866                generation,
7867            })
7868        }
7869        None => {
7870            // Spawning succeeded but the pipe did not materialise. Recording it as
7871            // uncaptured keeps the tail honest: the alternative is an empty tail
7872            // that reads as a module which printed nothing.
7873            ring.lock()
7874                .unwrap_or_else(|poisoned| poisoned.into_inner())
7875                .mark_not_captured("stderr pipe was not available on spawn");
7876            warn!(
7877                module_id = %spec.module_id,
7878                "spawned child exposed no stderr pipe; tail will be unavailable"
7879            );
7880            None
7881        }
7882    };
7883
7884    Ok(SupervisedChild {
7885        child,
7886        protocol: spec.protocol,
7887        #[cfg(target_os = "linux")]
7888        module_id: cgroup_name,
7889        #[cfg(target_os = "linux")]
7890        cgroup_placement: cgroup_placement.cloned(),
7891        #[cfg(windows)]
7892        job,
7893        stdout_pump,
7894        stderr_pump,
7895        stderr_ring: Arc::clone(ring),
7896        spawned_at_ms,
7897        spawned_from,
7898        spawned_file_identity,
7899        process_start_time,
7900        process_identity,
7901        pid,
7902        roster_guard: Some(roster_guard),
7903        #[cfg(target_os = "macos")]
7904        privacy_exec,
7905        #[cfg(target_os = "macos")]
7906        report_ready: Arc::new(OnceLock::new()),
7907        spawn_failure: None,
7908    })
7909}
7910
7911#[cfg(target_os = "linux")]
7912pub(crate) fn kill_module_cgroup(placement: Option<&subc_cgroup::Placement>, module_id: &str) {
7913    use subc_cgroup::KillOutcome;
7914    match subc_cgroup::kill_module(placement, module_id) {
7915        KillOutcome::Killed => {}
7916        KillOutcome::NotPlaced | KillOutcome::Unsupported => {
7917            debug!(
7918                module_id,
7919                "cgroup tree kill unavailable; using direct-child kill"
7920            );
7921        }
7922        KillOutcome::IoError { path, error } => {
7923            warn!(module_id, path = %path.display(), %error, "cgroup tree kill failed; using direct-child kill");
7924        }
7925    }
7926}
7927
7928/// Contain a freshly spawned Windows child and start it.
7929///
7930/// Steps 2 and 3 of the suspended-create contract: the job is created and the
7931/// child assigned **while it is still suspended** (step 1 is
7932/// `suspend_on_create_async` at the spawn site), then the child is resumed.
7933///
7934/// A child that is never resumed hangs forever holding a pid, so a resume
7935/// failure kills the child and fails the spawn rather than returning a
7936/// `SupervisedChild` that can never run.
7937///
7938/// An assignment failure is NOT fatal: an uncontained module behaves exactly as
7939/// it did before this existed, whereas refusing to start one would be a new
7940/// outage. It is logged at warn because it means a helper process could leak.
7941#[cfg(windows)]
7942fn contain_spawned_child(
7943    child: &Child,
7944    spec: &ModuleSpec,
7945) -> Result<Option<subc_jobobject::JobObject>, SuperviseError> {
7946    let module_id = spec.module_id.as_str();
7947    let Some(pid) = child.id() else {
7948        // The child exited between spawn and here. Its tree, if it made one,
7949        // needs no containment: nothing is left to contain.
7950        warn!(
7951            module_id,
7952            "spawned child had already exited before containment; no job object attached"
7953        );
7954        return Ok(None);
7955    };
7956
7957    let job = match subc_jobobject::JobObject::new() {
7958        Ok(job) => job,
7959        Err(source) => {
7960            warn!(
7961                module_id,
7962                error = %source,
7963                "could not create a job object; this module's helper processes will not be \
7964                 reaped on teardown"
7965            );
7966            // Resume regardless: leaving the child suspended would turn a
7967            // containment gap into a hung module.
7968            resume_suspended_child(pid, spec)?;
7969            return Ok(None);
7970        }
7971    };
7972
7973    if let Err(source) = job.assign(child) {
7974        warn!(
7975            module_id,
7976            error = %source,
7977            "could not assign the child to its job object; this module's helper processes \
7978             will not be reaped on teardown"
7979        );
7980        resume_suspended_child(pid, spec)?;
7981        return Ok(None);
7982    }
7983
7984    resume_suspended_child(pid, spec)?;
7985    Ok(Some(job))
7986}
7987
7988/// Resume a suspended child, killing it if it cannot be started.
7989///
7990/// A suspended process holds a pid and does nothing, so there is no useful
7991/// state to return: the caller gets an error and the spawn fails.
7992#[cfg(windows)]
7993fn resume_suspended_child(pid: u32, spec: &ModuleSpec) -> Result<(), SuperviseError> {
7994    if let Err(source) = subc_jobobject::resume_main_thread(pid) {
7995        // Kill it here rather than leaving a suspended process for the caller
7996        // to notice; `kill_on_drop` would eventually do this, but the module
7997        // would have been reported as running in between.
7998        let _ = std::process::Command::new("taskkill.exe")
7999            .args(["/PID", &pid.to_string(), "/T", "/F"])
8000            .stdin(Stdio::null())
8001            .stdout(Stdio::null())
8002            .stderr(Stdio::null())
8003            .status();
8004        return Err(SuperviseError::Spawn {
8005            program: spec.program.clone(),
8006            source,
8007            cgroup_path: None,
8008        });
8009    }
8010    Ok(())
8011}
8012
8013#[cfg(target_os = "linux")]
8014fn remove_module_cgroup(placement: &subc_cgroup::Placement, module_id: &str) {
8015    match placement.remove_module(module_id) {
8016        Ok(()) => debug!(module_id, "removed module cgroup after process exit"),
8017        Err(error) => warn!(
8018            module_id,
8019            error = %error,
8020            "could not remove module cgroup after process exit; continuing teardown"
8021        ),
8022    }
8023}
8024
8025#[cfg(target_os = "linux")]
8026async fn cleanup_reaped_cgroup(placement: &subc_cgroup::Placement, module_id: &str) {
8027    // Reaping the direct child is not proof its descendants exited. End the
8028    // residual tree and wait for the kernel's population fact before rmdir;
8029    // otherwise a successful parent wait leaks a directory on each restart.
8030    if matches!(
8031        subc_cgroup::kill_module(Some(placement), module_id),
8032        subc_cgroup::KillOutcome::Killed
8033    ) {
8034        if let Ok(path) = placement.module_path(module_id) {
8035            while std::fs::read_to_string(path.join("cgroup.events"))
8036                .ok()
8037                .is_some_and(|events| events.lines().any(|line| line == "populated 1"))
8038            {
8039                sleep(Duration::from_millis(1)).await;
8040            }
8041        }
8042    }
8043    remove_module_cgroup(placement, module_id);
8044}
8045
8046#[cfg(target_os = "linux")]
8047fn apply_cgroup_placement(
8048    command: &mut Command,
8049    spec: &ModuleSpec,
8050    path: &std::path::Path,
8051) -> Result<(), SuperviseError> {
8052    subc_cgroup::apply(command, path).map_err(|source| SuperviseError::Cgroup {
8053        module_id: spec.module_id.clone(),
8054        source,
8055    })
8056}
8057
8058fn capture_retention(spec: &ModuleSpec) -> Retention {
8059    let defaults = Retention::default();
8060    let value = |name: &str| {
8061        spec.env
8062            .iter()
8063            .rev()
8064            .find_map(|(key, value)| (key == name).then_some(value.as_str()))
8065    };
8066    Retention {
8067        max_file_mb: value(CAPTURE_MAX_FILE_MB_ENV)
8068            .and_then(|value| value.parse().ok())
8069            .unwrap_or(defaults.max_file_mb),
8070        keep: value(CAPTURE_KEEP_ENV)
8071            .and_then(|value| value.parse().ok())
8072            .unwrap_or(defaults.keep),
8073        max_age_days: value(CAPTURE_MAX_AGE_DAYS_ENV)
8074            .and_then(|value| value.parse().ok())
8075            .unwrap_or(defaults.max_age_days),
8076    }
8077}
8078
8079/// A fresh 256-bit CSPRNG launch nonce, lowercase hex. Used to bind a reserved
8080/// module's registration to the exact process the supervisor spawned.
8081fn generate_launch_nonce() -> Result<String, SuperviseError> {
8082    let mut bytes = [0u8; 32];
8083    getrandom::getrandom(&mut bytes).map_err(|source| SuperviseError::LaunchNonce {
8084        reason: source.to_string(),
8085    })?;
8086    let mut hex = String::with_capacity(64);
8087    for b in bytes {
8088        use std::fmt::Write;
8089        let _ = write!(hex, "{b:02x}");
8090    }
8091    Ok(hex)
8092}
8093
8094/// Constant-time byte comparison so a reserved-nonce mismatch leaks no timing
8095/// signal about how many leading bytes matched.
8096fn constant_time_eq(a: &[u8], b: &[u8]) -> bool {
8097    if a.len() != b.len() {
8098        return false;
8099    }
8100    let mut diff = 0u8;
8101    for (x, y) in a.iter().zip(b.iter()) {
8102        diff |= x ^ y;
8103    }
8104    diff == 0
8105}
8106
8107/// The kernel's image after an acknowledged exec, shared by ordinary launches
8108/// and privacy trampolines. PATH and shebang interpretation are kernel facts,
8109/// not identities inferred from a configured pathname.
8110fn observe_spawned_image(pid: u32) -> Option<subc_os::Observation> {
8111    subc_os::Process::open(pid)
8112        .ok()
8113        .flatten()
8114        .and_then(|process| process.observe())
8115}
8116
8117fn spawn_and_mark_running(
8118    spec: &ModuleSpec,
8119    runtime: &SupervisorRuntimeConfig,
8120    snapshot: &SharedSnapshot,
8121) -> Result<SupervisedChild, SuperviseError> {
8122    let child = spawn_child(
8123        spec,
8124        runtime.connection_file_path.as_deref(),
8125        runtime.supervisor_handle.as_ref(),
8126        &runtime.stderr_ring,
8127        runtime.capture_logs_dir.as_deref(),
8128        &runtime.child_roster,
8129        #[cfg(target_os = "linux")]
8130        runtime.cgroup_placement.as_ref(),
8131    )?;
8132    set_running(snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
8133    Ok(child)
8134}
8135
8136enum RegistrationWaitOutcome {
8137    Registered,
8138    Exited(ExitReport),
8139    TimedOut,
8140}
8141
8142struct ReloadRegistrationFailure {
8143    exit_report: ExitReport,
8144    reason: String,
8145}
8146
8147#[derive(Debug, Clone, Copy, PartialEq, Eq)]
8148enum BusyGaugeObservation {
8149    Quiescent,
8150    Busy,
8151    Omitted,
8152}
8153
8154fn busy_gauge_observation(metrics: Option<&Value>, gauges: &[String]) -> BusyGaugeObservation {
8155    let Some(metrics) = metrics.and_then(Value::as_object) else {
8156        return BusyGaugeObservation::Omitted;
8157    };
8158    let mut sum = 0u128;
8159    for gauge in gauges {
8160        let Some(value) = metrics.get(gauge) else {
8161            return BusyGaugeObservation::Omitted;
8162        };
8163        let Some(value) = value.as_u64() else {
8164            return BusyGaugeObservation::Busy;
8165        };
8166        sum = sum.saturating_add(u128::from(value));
8167    }
8168    if sum == 0 {
8169        BusyGaugeObservation::Quiescent
8170    } else {
8171        BusyGaugeObservation::Busy
8172    }
8173}
8174
8175fn declared_busy_gauges(
8176    registry: &Registry,
8177    module_id: &str,
8178) -> Result<Vec<String>, SuperviseError> {
8179    busy_gauges_of(
8180        registry
8181            .get_module(module_id)
8182            .map_err(SuperviseError::Registry)?,
8183    )
8184}
8185
8186/// [`declared_busy_gauges`] for the registration a connection holds, in any
8187/// slot: after cutover the incumbent is no longer the id's active
8188/// registration, and its own manifest is the one that names its gauges.
8189fn declared_busy_gauges_for_connection(
8190    registry: &Registry,
8191    connection_id: ConnectionId,
8192) -> Result<Vec<String>, SuperviseError> {
8193    busy_gauges_of(
8194        registry
8195            .get_module_by_connection(connection_id)
8196            .map_err(SuperviseError::Registry)?,
8197    )
8198}
8199
8200fn busy_gauges_of(
8201    registration: Option<crate::registry::ModuleRegistration>,
8202) -> Result<Vec<String>, SuperviseError> {
8203    let Some(registration) = registration else {
8204        return Ok(Vec::new());
8205    };
8206    let Some(self_signals) = registration.manifest.self_signals else {
8207        return Ok(Vec::new());
8208    };
8209
8210    let mut gauges = Vec::new();
8211    for declaration in self_signals {
8212        if declaration.kind != SelfSignalKind::Busy {
8213            continue;
8214        }
8215        match declaration.anchored_to {
8216            SignalAnchor::HealthGauges { gauges: declared } if !declared.is_empty() => {
8217                gauges.extend(declared)
8218            }
8219            _ => {
8220                // An invalid Busy anchor is fail-safe: the empty name cannot be
8221                // present in a conforming health report, so this drain stays busy.
8222                gauges.push(String::new());
8223            }
8224        }
8225    }
8226    Ok(gauges)
8227}
8228
8229/// Wait for `endpoint` to have nothing in flight and, when the module declares
8230/// busy gauges, for a health probe to report them quiet. The probe is addressed
8231/// by `scope`: a swap's superseded incumbent must be asked about its own
8232/// gauges, and by module id the probe would reach the promoted candidate.
8233async fn wait_for_forwarding_quiescence(
8234    forwarding: &ForwardingTable,
8235    module_id: &str,
8236    runtime: &SupervisorRuntimeConfig,
8237    endpoint: crate::ModuleEndpointId,
8238    deadline: Instant,
8239    busy_gauges: &[String],
8240    scope: DrainScope,
8241) -> Result<bool, SuperviseError> {
8242    let mut gauges_quiescent = busy_gauges.is_empty();
8243    let mut next_probe_at = Instant::now();
8244    let mut omission_counted = false;
8245
8246    loop {
8247        let now = Instant::now();
8248        if !busy_gauges.is_empty() && now >= next_probe_at && now < deadline {
8249            let report = match scope {
8250                DrainScope::Active => probe_module_health(module_id, runtime, Some(deadline)).await,
8251                DrainScope::Endpoint(endpoint) => {
8252                    probe_endpoint_health(endpoint, runtime, Some(deadline)).await
8253                }
8254            };
8255            gauges_quiescent = match report {
8256                Ok(report) => match busy_gauge_observation(report.metrics.as_ref(), busy_gauges) {
8257                    BusyGaugeObservation::Quiescent => true,
8258                    BusyGaugeObservation::Busy => false,
8259                    BusyGaugeObservation::Omitted => {
8260                        if !omission_counted {
8261                            forwarding
8262                                .counters()
8263                                .increment_drains_with_undeclared_gauge();
8264                            omission_counted = true;
8265                        }
8266                        false
8267                    }
8268                },
8269                Err(err) => {
8270                    warn!(
8271                        module_id,
8272                        error = %err,
8273                        "drain health.check did not produce declared busy gauges; treating module as busy"
8274                    );
8275                    false
8276                }
8277            };
8278            next_probe_at = Instant::now() + runtime.health.cadence.max(REGISTRY_RELEASE_POLL);
8279        }
8280
8281        let in_flight = forwarding
8282            .endpoint_in_flight_count(endpoint)
8283            .map_err(SuperviseError::Forwarding)?;
8284        if in_flight == 0 && gauges_quiescent {
8285            return Ok(true);
8286        }
8287
8288        let now = Instant::now();
8289        if now >= deadline {
8290            return Ok(false);
8291        }
8292        let mut wait = deadline
8293            .saturating_duration_since(now)
8294            .min(REGISTRY_RELEASE_POLL);
8295        if !busy_gauges.is_empty() {
8296            wait = wait.min(next_probe_at.saturating_duration_since(now));
8297        }
8298        sleep(wait).await;
8299    }
8300}
8301
8302/// The `route.closed` `drained` value implied by a quiescence-wait outcome.
8303///
8304/// `Ok` is always honest and passed straight through -- the wait actually measured
8305/// in-flight state. `Err` means the wait produced no measurement at all (the
8306/// forwarding table's lock was poisoned), so `false` is reported as the one honest
8307/// constant: the drain did not complete. Never recomputed from route state, never a
8308/// third "unknown" value -- the caller must still send a well-formed `route.closed`.
8309fn drained_after_quiescence_wait(wait_result: &Result<bool, SuperviseError>) -> bool {
8310    match wait_result {
8311        Ok(drained) => *drained,
8312        Err(_) => false,
8313    }
8314}
8315
8316fn send_route_goodbyes(forwarding: &ForwardingTable, released_routes: Vec<GoodbyeTarget>) {
8317    for released in released_routes {
8318        let frame = match Frame::build_with_version(
8319            released.negotiated_ver,
8320            FrameType::Goodbye,
8321            control_flags(),
8322            released.channel,
8323            released.epoch,
8324            0,
8325            Vec::new(),
8326        ) {
8327            Ok(frame) => frame,
8328            Err(err) => {
8329                warn!(
8330                    route_channel = released.channel,
8331                    error = %err,
8332                    "failed to build supervisor drain route GOODBYE frame"
8333                );
8334                continue;
8335            }
8336        };
8337        if !released.close_on_delivery_failure() {
8338            crate::forwarding::send_module_route_goodbye(
8339                &forwarding.counters(),
8340                &released.sink,
8341                frame,
8342                released.module_id.as_deref(),
8343                "supervisor drain",
8344            );
8345            continue;
8346        }
8347        if let Err(err) = released.sink.try_send(frame) {
8348            warn!(
8349                target_connection_id = released.connection_id.get(),
8350                route_channel = released.channel,
8351                error = %err,
8352                "supervisor drain route GOODBYE was not delivered to client; closing target connection"
8353            );
8354            let _ = forwarding.escalate_client_delivery_failure(
8355                released.connection_id,
8356                released.channel,
8357                released.epoch,
8358                CloseReason::new(
8359                    "route_goodbye_delivery_failed",
8360                    format!(
8361                        "failed to enqueue supervisor drain route GOODBYE for channel {}: {err}",
8362                        released.channel
8363                    ),
8364                ),
8365                crate::forwarding::UndeliveredFrame {
8366                    module_id: released.module_id.as_deref(),
8367                    sink: &released.sink,
8368                },
8369            );
8370        }
8371    }
8372}
8373
8374fn send_module_draining(
8375    module_id: &str,
8376    reason: RouteCloseReason,
8377    deadline_ms: u64,
8378    target: &ModuleDrainTarget,
8379) {
8380    let body = match serde_json::to_vec(&ModuleControlCommand::Draining {
8381        reason,
8382        deadline_ms,
8383    }) {
8384        Ok(body) => body,
8385        Err(err) => {
8386            warn!(
8387                module_id,
8388                error = %err,
8389                "failed to encode module draining command"
8390            );
8391            return;
8392        }
8393    };
8394    let frame = match Frame::build_with_version(
8395        target.negotiated_ver,
8396        FrameType::Push,
8397        control_flags(),
8398        0,
8399        0,
8400        0,
8401        body,
8402    ) {
8403        Ok(frame) => frame,
8404        Err(err) => {
8405            warn!(
8406                module_id,
8407                error = %err,
8408                "failed to build module draining command frame"
8409            );
8410            return;
8411        }
8412    };
8413    if let Err(err) = target.sink.try_send(frame) {
8414        warn!(
8415            module_id,
8416            target_connection_id = target.endpoint.connection_id.get(),
8417            error = %err,
8418            "module draining command was not delivered to peer"
8419        );
8420    }
8421}
8422
8423/// The channel-0 GOODBYE that tells a module its stop is planned.
8424fn module_goodbye_frame(module_id: &str, negotiated_ver: u8) -> Option<Frame> {
8425    match Frame::build_with_version(
8426        negotiated_ver,
8427        FrameType::Goodbye,
8428        control_flags(),
8429        0,
8430        0,
8431        0,
8432        Vec::new(),
8433    ) {
8434        Ok(frame) => Some(frame),
8435        Err(err) => {
8436            warn!(
8437                module_id,
8438                error = %err,
8439                "failed to build module GOODBYE frame"
8440            );
8441            None
8442        }
8443    }
8444}
8445
8446/// Send every registered module connection its module GOODBYE at daemon
8447/// shutdown, then request that connection's close.
8448///
8449/// A module tells a planned stop from a lost daemon by whether a GOODBYE came
8450/// before EOF, so the GOODBYE must reach the socket before the close. A close
8451/// request does not wait for the connection's queued frames: its writer gets a
8452/// bounded grace after the close, is aborted if it overruns it, and the daemon
8453/// process may exit before that grace ends. So with `wait_for_flush`, each
8454/// connection is closed only after its writer has acknowledged writing the
8455/// GOODBYE, or once a short shared budget runs out, so one module that is not
8456/// reading cannot hold up the others or the shutdown. Without it the GOODBYEs
8457/// are only queued, for a shutdown the operator has told to stop waiting.
8458/// A connection that is already gone is skipped.
8459#[cfg(unix)]
8460async fn send_module_goodbyes_for_daemon_shutdown(
8461    forwarding: &Arc<ForwardingTable>,
8462    reason: &CloseReason,
8463    wait_for_flush: bool,
8464) {
8465    const GOODBYE_BUDGET: Duration = Duration::from_millis(500);
8466    let targets = match forwarding.module_connections() {
8467        Ok(targets) => targets,
8468        Err(err) => {
8469            warn!(error = %err, "could not list module connections for shutdown GOODBYE");
8470            return;
8471        }
8472    };
8473    let deadline = Instant::now() + GOODBYE_BUDGET;
8474    let mut sends = tokio::task::JoinSet::new();
8475    for target in targets {
8476        let Some(frame) = module_goodbye_frame(&target.module_id, target.negotiated_ver) else {
8477            continue;
8478        };
8479        if !wait_for_flush {
8480            if let Err(err) = target.sink.try_send(frame) {
8481                debug!(
8482                    module_id = %target.module_id,
8483                    error = %err,
8484                    "shutdown module GOODBYE was not queued"
8485                );
8486            }
8487            continue;
8488        }
8489        let forwarding = Arc::clone(forwarding);
8490        let reason = reason.clone();
8491        sends.spawn(async move {
8492            match timeout_at(deadline, target.sink.send_flushed(frame)).await {
8493                Ok(Ok(())) => {}
8494                Ok(Err(err)) => debug!(
8495                    module_id = %target.module_id,
8496                    error = %err,
8497                    "module connection closed before its shutdown GOODBYE was written"
8498                ),
8499                Err(_) => warn!(
8500                    module_id = %target.module_id,
8501                    budget = ?GOODBYE_BUDGET,
8502                    "shutdown module GOODBYE was not written within its budget; closing anyway"
8503                ),
8504            }
8505            forwarding.request_connection_close(target.endpoint.connection_id, reason);
8506        });
8507    }
8508    // Every task ends by the shared deadline, so this wait is bounded too.
8509    while sends.join_next().await.is_some() {}
8510}
8511
8512fn send_module_goodbye(module_id: &str, forwarding: &ForwardingTable, target: &ModuleDrainTarget) {
8513    let Some(frame) = module_goodbye_frame(module_id, target.negotiated_ver) else {
8514        return;
8515    };
8516    if let Err(err) = target.sink.try_send(frame) {
8517        warn!(
8518            module_id,
8519            target_connection_id = target.endpoint.connection_id.get(),
8520            error = %err,
8521            "supervisor drain module GOODBYE was not delivered to peer; closing module connection"
8522        );
8523        forwarding.request_connection_close(
8524            target.endpoint.connection_id,
8525            CloseReason::new(
8526                "module_goodbye_delivery_failed",
8527                format!("failed to enqueue supervisor drain module GOODBYE for module '{module_id}': {err}"),
8528            ),
8529        );
8530    }
8531}
8532
8533#[derive(Clone, Copy)]
8534struct ForwardingDrainContext<'a> {
8535    spec: &'a ModuleSpec,
8536    runtime: &'a SupervisorRuntimeConfig,
8537    registry: &'a Registry,
8538    scope: DrainScope,
8539}
8540
8541/// Which process a forwarding drain addresses.
8542#[derive(Debug, Clone, Copy, PartialEq, Eq)]
8543enum DrainScope {
8544    /// Whatever endpoint is active for the module id: every plain stop,
8545    /// restart and reload. Also moves the module's state to `Draining`.
8546    Active,
8547    /// One specific endpoint: a swap's incumbent after cutover. Draining it by
8548    /// module id would resolve to the promoted candidate and leave neither
8549    /// process routable. The module's state is left alone, since the promoted
8550    /// candidate is what it describes and that process is running.
8551    Endpoint(crate::ModuleEndpointId),
8552}
8553
8554/// Whether a child being drained has already been asked to stop by the time
8555/// its drain wait starts.
8556///
8557/// The drain wait is the same budget whatever this says. What it decides is
8558/// whether the supervisor must ask by signal before that wait begins: a child
8559/// that nobody asked will sit out the whole budget and then be SIGKILLed,
8560/// healthy or not.
8561#[derive(Debug, Clone, Copy, PartialEq, Eq)]
8562enum StopNotice {
8563    /// The module was sent `module.draining` and a module GOODBYE over its own
8564    /// registered connection, and stops itself.
8565    SentOverConnection,
8566    /// The forwarding drain found no registered connection for the module: a
8567    /// subc child spawned moments ago that has not sent HELLO yet, or a
8568    /// `protocol: "none"` child, which never registers.
8569    NoConnection,
8570    /// This path sends nothing over the module's connection: the supervisor has
8571    /// no forwarding table, or the caller stops the child without a forwarding
8572    /// drain.
8573    NotSent,
8574}
8575
8576async fn begin_forwarding_drain(
8577    spec: &ModuleSpec,
8578    runtime: &SupervisorRuntimeConfig,
8579    registry: &Registry,
8580    snapshot: &SharedSnapshot,
8581    enabled: Option<bool>,
8582    reason: RouteCloseReason,
8583) -> Result<StopNotice, SuperviseError> {
8584    let Some(forwarding) = runtime.forwarding.as_ref() else {
8585        return Err(SuperviseError::ReloadUnavailable {
8586            module_id: spec.module_id.clone(),
8587            reason: "supervisor was not configured with a forwarding table".to_string(),
8588        });
8589    };
8590
8591    begin_forwarding_drain_with(
8592        forwarding,
8593        ForwardingDrainContext {
8594            spec,
8595            runtime,
8596            registry,
8597            scope: DrainScope::Active,
8598        },
8599        snapshot,
8600        enabled,
8601        reason,
8602        runtime.drain_timeout,
8603    )
8604    .await
8605}
8606
8607async fn begin_forwarding_drain_if_configured(
8608    spec: &ModuleSpec,
8609    runtime: &SupervisorRuntimeConfig,
8610    registry: &Registry,
8611    snapshot: &SharedSnapshot,
8612    enabled: Option<bool>,
8613    reason: RouteCloseReason,
8614) -> Result<StopNotice, SuperviseError> {
8615    begin_forwarding_drain_with_timeout(
8616        spec,
8617        runtime,
8618        registry,
8619        snapshot,
8620        enabled,
8621        reason,
8622        runtime.drain_timeout,
8623    )
8624    .await
8625}
8626
8627/// Like [`begin_forwarding_drain_if_configured`] but with an explicit drain
8628/// budget, for paths where the operator overrides the module's configured one
8629/// (`supervisor.restart{drain_timeout_ms}`).
8630async fn begin_forwarding_drain_with_timeout(
8631    spec: &ModuleSpec,
8632    runtime: &SupervisorRuntimeConfig,
8633    registry: &Registry,
8634    snapshot: &SharedSnapshot,
8635    enabled: Option<bool>,
8636    reason: RouteCloseReason,
8637    drain_timeout: Duration,
8638) -> Result<StopNotice, SuperviseError> {
8639    let Some(forwarding) = runtime.forwarding.as_ref() else {
8640        return Ok(StopNotice::NotSent);
8641    };
8642
8643    begin_forwarding_drain_with(
8644        forwarding,
8645        ForwardingDrainContext {
8646            spec,
8647            runtime,
8648            registry,
8649            scope: DrainScope::Active,
8650        },
8651        snapshot,
8652        enabled,
8653        reason,
8654        drain_timeout,
8655    )
8656    .await
8657}
8658
8659async fn begin_forwarding_drain_with(
8660    forwarding: &ForwardingTable,
8661    context: ForwardingDrainContext<'_>,
8662    snapshot: &SharedSnapshot,
8663    enabled: Option<bool>,
8664    reason: RouteCloseReason,
8665    drain_timeout: Duration,
8666) -> Result<StopNotice, SuperviseError> {
8667    let ForwardingDrainContext {
8668        spec,
8669        runtime,
8670        registry,
8671        scope,
8672    } = context;
8673    debug_assert_ne!(reason, RouteCloseReason::Crash);
8674    let terminal = matches!(reason, RouteCloseReason::Disable);
8675    let drain_started_at = Instant::now();
8676    let drain_deadline = drain_started_at + drain_timeout;
8677    let deadline_ms =
8678        unix_ms_now().saturating_add(u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX));
8679    let busy_gauges = match scope {
8680        DrainScope::Active => declared_busy_gauges(registry, &spec.module_id)?,
8681        DrainScope::Endpoint(endpoint) => {
8682            declared_busy_gauges_for_connection(registry, endpoint.connection_id)?
8683        }
8684    };
8685
8686    // Admission gate first: route.open/commit and route REQUEST admission are closed
8687    // before the first quiescence check, so the outstanding count can only fall.
8688    let gate_started = Instant::now();
8689    let drain_target = match scope {
8690        DrainScope::Active => forwarding.begin_module_drain(&spec.module_id, reason),
8691        DrainScope::Endpoint(endpoint) => forwarding.begin_endpoint_drain(endpoint, reason),
8692    }
8693    .map_err(SuperviseError::Forwarding)?;
8694    // The instant admission closed, and how long taking the forwarding write
8695    // lock to close it took. The timeout line reports only the quiescence
8696    // wait, so without this a drain that started late looked like one that
8697    // started on time.
8698    info!(
8699        module_id = %spec.module_id,
8700        ?reason,
8701        gate_ms = u64::try_from(gate_started.elapsed().as_millis()).unwrap_or(u64::MAX),
8702        connected = drain_target.is_some(),
8703        "module drain began; route admission closed"
8704    );
8705    if scope == DrainScope::Active {
8706        update_snapshot(snapshot, Some(&spec.module_id), |state| {
8707            state.state = ModuleState::Draining;
8708            state.draining_to_replace =
8709                matches!(reason, RouteCloseReason::Restart | RouteCloseReason::Reload);
8710            if let Some(enabled) = enabled {
8711                state.enabled = enabled;
8712            }
8713        })?;
8714    }
8715
8716    let Some(target) = drain_target.as_ref() else {
8717        // Nothing was sent: the module has no registered connection to carry
8718        // `module.draining` or a GOODBYE. The caller must not assume the child
8719        // was asked to stop.
8720        return Ok(StopNotice::NoConnection);
8721    };
8722    {
8723        send_module_draining(&spec.module_id, reason, deadline_ms, target);
8724        let routes = forwarding
8725            .endpoint_routes(target.endpoint)
8726            .map_err(SuperviseError::Forwarding)?;
8727        let routes_notified = routes.len();
8728        crate::control::send_route_control_pushes(
8729            forwarding,
8730            routes.clone(),
8731            ClientControlPush::RouteClosing {
8732                module_id: spec.module_id.clone(),
8733                channels: Vec::new(),
8734                reason,
8735            },
8736        );
8737        send_route_goodbyes(forwarding, target.abandoned_bindings.clone());
8738
8739        // `route.closing` was just sent above: from here on every return path,
8740        // including an early one, MUST send `route.closed` before propagating
8741        // anything else. A client holds `closing` as a promise that a verdict is
8742        // coming; leaving early without `closed` strands it waiting forever, since
8743        // `closing` carries no timeout of its own.
8744        let wait_result = wait_for_forwarding_quiescence(
8745            forwarding,
8746            &spec.module_id,
8747            runtime,
8748            target.endpoint,
8749            drain_deadline,
8750            &busy_gauges,
8751            scope,
8752        )
8753        .await;
8754        let drained = drained_after_quiescence_wait(&wait_result);
8755        if let Err(err) = &wait_result {
8756            error!(
8757                module_id = %spec.module_id,
8758                ?reason,
8759                error = %err,
8760                "forwarding quiescence wait failed after route.closing; forcing route.closed(drained: false) so the client is not left waiting on an unfulfilled promise"
8761            );
8762        } else if !drained {
8763            // Name what the drain waited on. Without it the line says only that
8764            // something did not settle, and "one wedged call" and "every
8765            // session's held stream" read the same; the first is a module bug,
8766            // the second is a module that should end its streams on
8767            // module.draining. Read before teardown releases the routes.
8768            let holdouts = forwarding
8769                .endpoint_drain_holdouts(target.endpoint)
8770                .unwrap_or_default();
8771            warn!(
8772                module_id = %spec.module_id,
8773                waited = ?drain_timeout,
8774                ?reason,
8775                held_requests = holdouts.requests,
8776                held_routes = holdouts.routes,
8777                total_routes = holdouts.total_routes,
8778                top_connections = ?holdouts.top_connections,
8779                // `module_channel:corr`, so the module can find each held request
8780                // in its own log; capped, so `held_requests` is the full count.
8781                held = %holdouts
8782                    .held
8783                    .iter()
8784                    .map(|(channel, corr)| format!("{channel}:{corr}"))
8785                    .collect::<Vec<_>>()
8786                    .join(","),
8787                "route drain timed out before request quiescence; forcing teardown"
8788            );
8789        }
8790        crate::control::send_route_control_pushes(
8791            forwarding,
8792            routes,
8793            ClientControlPush::RouteClosed {
8794                module_id: spec.module_id.clone(),
8795                channels: Vec::new(),
8796                reason,
8797                drained,
8798                abandoned: target.abandoned_bindings.len() as u32,
8799                excluded_subscriptions: target.excluded_subscriptions,
8800                terminal: Some(terminal),
8801            },
8802        );
8803        wait_result?;
8804
8805        // `route.closed` has now been sent unconditionally above. From here the
8806        // remaining steps are cleanup (route + module GOODBYE) rather than a
8807        // promise the client is waiting on, but a lock-poisoned
8808        // `release_module_endpoint_routes` would otherwise skip the module
8809        // GOODBYE silently too -- send it before propagating the error.
8810        let released_routes = match forwarding.release_module_endpoint_routes(target.endpoint) {
8811            Ok(routes) => routes,
8812            Err(err) => {
8813                warn!(
8814                    module_id = %spec.module_id,
8815                    ?reason,
8816                    error = %err,
8817                    "failed to release module endpoint routes after route.closed; module GOODBYE will still be sent"
8818                );
8819                send_module_goodbye(&spec.module_id, forwarding, target);
8820                return Err(SuperviseError::Forwarding(err));
8821            }
8822        };
8823        let route_goodbye_count = released_routes.len();
8824        send_route_goodbyes(forwarding, released_routes);
8825        send_module_goodbye(&spec.module_id, forwarding, target);
8826
8827        // The drain's happy path was previously silent: every emission above is
8828        // best-effort with only its failure arm logged, so "were consumers told"
8829        // was unprovable from the daemon log (surfaced by a 30-minute consumer
8830        // hang where the open question was exactly whether teardown notice went
8831        // out). One summary line makes that class decidable in one grep.
8832        info!(
8833            module_id = %spec.module_id,
8834            ?reason,
8835            routes_notified,
8836            route_goodbyes = route_goodbye_count,
8837            abandoned_reservations = target.abandoned_bindings.len(),
8838            excluded_subscriptions = target.excluded_subscriptions,
8839            drained,
8840            "module drain complete; consumers notified via route.closing/route.closed pushes and per-route GOODBYE frames"
8841        );
8842    }
8843
8844    Ok(StopNotice::SentOverConnection)
8845}
8846
8847/// Wait for the freshly spawned child to take the ACTIVE slot for `module_id`,
8848/// the only slot a plain (non-swap) spawn can register into.
8849async fn wait_for_registration_after_reload(
8850    registry: &Registry,
8851    module_id: &str,
8852    snapshot: &SharedSnapshot,
8853    child: &mut SupervisedChild,
8854    wait: Duration,
8855) -> Result<RegistrationWaitOutcome, SuperviseError> {
8856    wait_for_slot_registration(
8857        registry,
8858        crate::registry::RegistrationSlot::Active(module_id),
8859        module_id,
8860        snapshot,
8861        child,
8862        wait,
8863    )
8864    .await
8865}
8866
8867/// Wait for `child` to register into `slot`, or to exit, or for `wait` to pass.
8868///
8869/// Keyed on the slot rather than the bare module id because during a swap the
8870/// id's active slot is already held by the incumbent: an id-keyed wait would
8871/// report the incumbent's registration as the candidate's and a candidate that
8872/// never registers would look registered. A swap candidate waits on
8873/// `crate::registry::RegistrationSlot::Candidate`.
8874async fn wait_for_slot_registration(
8875    registry: &Registry,
8876    slot: crate::registry::RegistrationSlot<'_>,
8877    module_id: &str,
8878    snapshot: &SharedSnapshot,
8879    child: &mut SupervisedChild,
8880    wait: Duration,
8881) -> Result<RegistrationWaitOutcome, SuperviseError> {
8882    let deadline = Instant::now() + wait;
8883    loop {
8884        if registry
8885            .registration(slot)
8886            .map_err(SuperviseError::Registry)?
8887            .is_some()
8888        {
8889            return Ok(RegistrationWaitOutcome::Registered);
8890        }
8891
8892        let now = Instant::now();
8893        if now >= deadline {
8894            return Ok(RegistrationWaitOutcome::TimedOut);
8895        }
8896        let remaining = deadline.saturating_duration_since(now);
8897        let poll = remaining.min(REGISTRY_RELEASE_POLL);
8898
8899        tokio::select! {
8900            wait_result = child.wait() => {
8901                let status = wait_result.map_err(|source| SuperviseError::Wait {
8902                    module_id: module_id.to_string(),
8903                    source,
8904                })?;
8905                return Ok(RegistrationWaitOutcome::Exited(classify_reaped_child_exit(
8906                    snapshot,
8907                    child,
8908                    &status,
8909                )));
8910            }
8911            _ = sleep(poll) => {}
8912        }
8913    }
8914}
8915
8916fn registration_failure_exit_report(mut exit_report: ExitReport) -> ExitReport {
8917    // A replacement process that exits before HELLO did not provide service, even
8918    // if it used status 0. Count it against the restart cap as a new-binary failure.
8919    if exit_report.kind != ExitKind::DeliberateSeverance {
8920        exit_report.kind = ExitKind::Crash;
8921    }
8922    exit_report
8923}
8924
8925async fn handle_reload_child_registration_failure(
8926    spec: &ModuleSpec,
8927    runtime: &SupervisorRuntimeConfig,
8928    registry: &Registry,
8929    process_liveness: &SupervisorProcessLiveness,
8930    snapshot: &SharedSnapshot,
8931    _child: &mut Option<SupervisedChild>,
8932    failure: ReloadRegistrationFailure,
8933) -> Result<(), SuperviseError> {
8934    let ReloadRegistrationFailure {
8935        exit_report,
8936        reason,
8937    } = failure;
8938    match on_child_exit(
8939        spec,
8940        runtime.restart_policy,
8941        registry,
8942        snapshot,
8943        &runtime.terminal_ring,
8944        &runtime.spawn_events,
8945        &runtime.child_roster,
8946        exit_report,
8947    )
8948    .await
8949    {
8950        NextAction::Stop {
8951            registration_released,
8952        } => {
8953            if registration_released {
8954                process_liveness.untrack_if_current(&spec.module_id, snapshot);
8955            }
8956        }
8957        NextAction::Restart { schedule } => {
8958            let delay = schedule.map_or(runtime.restart_policy.delay_for_restart(0), |schedule| {
8959                schedule.delay
8960            });
8961            if let Some(schedule) = schedule {
8962                log_crash_respawn(&spec.module_id, schedule);
8963            }
8964            schedule_respawn(
8965                runtime,
8966                snapshot,
8967                &spec.module_id,
8968                delay,
8969                RespawnKind::Spawn,
8970            )?;
8971        }
8972    }
8973    Err(SuperviseError::ReloadFailed {
8974        module_id: spec.module_id.clone(),
8975        reason,
8976    })
8977}
8978
8979async fn handle_reload_spawn_failure(
8980    spec: &ModuleSpec,
8981    runtime: &SupervisorRuntimeConfig,
8982    process_liveness: &SupervisorProcessLiveness,
8983    snapshot: &SharedSnapshot,
8984    _child: &mut Option<SupervisedChild>,
8985    reason: String,
8986) -> Result<(), SuperviseError> {
8987    let now = Instant::now();
8988    let mut schedule = None;
8989    update_snapshot(snapshot, Some(&spec.module_id), |state| {
8990        clear_current_process_facts(state);
8991        if state.enabled {
8992            schedule = state.next_crash_restart(&runtime.restart_policy, now);
8993            state.state = if schedule.is_some() {
8994                ModuleState::Restarting
8995            } else {
8996                ModuleState::Failed
8997            };
8998        } else {
8999            state.state = ModuleState::Disabled;
9000        }
9001    })?;
9002    if let Some(schedule) = schedule {
9003        schedule_respawn(
9004            runtime,
9005            snapshot,
9006            &spec.module_id,
9007            schedule.delay,
9008            RespawnKind::Spawn,
9009        )?;
9010    } else {
9011        process_liveness.untrack_if_current(&spec.module_id, snapshot);
9012    }
9013    Err(SuperviseError::ReloadFailed {
9014        module_id: spec.module_id.clone(),
9015        reason,
9016    })
9017}
9018
9019fn control_flags() -> Flags {
9020    Flags::new(false, Priority::Passive, false)
9021}
9022
9023#[allow(clippy::too_many_arguments)]
9024async fn drain_optional_child(
9025    module_id: &str,
9026    protocol: ModuleProtocol,
9027    stop_notice: StopNotice,
9028    registry: &Registry,
9029    forwarding: Option<&ForwardingTable>,
9030    snapshot: &SharedSnapshot,
9031    terminal_ring: &Arc<Mutex<TerminalRing>>,
9032    spawn_events: &SpawnEventFeed,
9033    child: &mut Option<SupervisedChild>,
9034    drain_timeout: Duration,
9035    final_state: ModuleState,
9036    enabled: Option<bool>,
9037) -> Result<(), SuperviseError> {
9038    if let Some(child) = child.take() {
9039        drain_child_to_state(
9040            module_id,
9041            protocol,
9042            stop_notice,
9043            registry,
9044            forwarding,
9045            snapshot,
9046            terminal_ring,
9047            spawn_events,
9048            child,
9049            drain_timeout,
9050            final_state,
9051            enabled,
9052        )
9053        .await
9054    } else {
9055        update_snapshot(snapshot, Some(module_id), |state| {
9056            state.state = final_state;
9057            if let Some(enabled) = enabled {
9058                state.enabled = enabled;
9059            }
9060            clear_current_process_facts(state);
9061        })?;
9062        release_dead_registration(registry, forwarding, snapshot, module_id).await
9063    }
9064}
9065
9066#[allow(clippy::too_many_arguments)]
9067async fn drain_child_to_state(
9068    module_id: &str,
9069    _protocol: ModuleProtocol,
9070    stop_notice: StopNotice,
9071    registry: &Registry,
9072    forwarding: Option<&ForwardingTable>,
9073    snapshot: &SharedSnapshot,
9074    terminal_ring: &Arc<Mutex<TerminalRing>>,
9075    spawn_events: &SpawnEventFeed,
9076    mut child: SupervisedChild,
9077    drain_timeout: Duration,
9078    final_state: ModuleState,
9079    enabled: Option<bool>,
9080) -> Result<(), SuperviseError> {
9081    let protocol = child.protocol;
9082    update_snapshot(snapshot, Some(module_id), |state| {
9083        state.state = ModuleState::Draining;
9084        state.draining_to_replace = final_state == ModuleState::Restarting;
9085        if let Some(enabled) = enabled {
9086            state.enabled = enabled;
9087        }
9088    })?;
9089
9090    // The wait below is the same budget in every case; what differs is
9091    // whether anything has ASKED the child to stop before it starts. Only a
9092    // forwarding drain that reached the module's registered connection has
9093    // (`module.draining`, then a module GOODBYE). Every other child was told
9094    // nothing: a `protocol: "none"` module, which never registers; a subc
9095    // module spawned moments ago that has not sent HELLO yet; or a stop that
9096    // runs no forwarding drain. Without a signal the budget is only a delay
9097    // in front of SIGKILL -- and the not-yet-registered child is the worst
9098    // case, because it registers into a module that is already draining,
9099    // is never told, and is killed while healthy.
9100    if stop_notice != StopNotice::SentOverConnection {
9101        if protocol == ModuleProtocol::Subc && stop_notice == StopNotice::NoConnection {
9102            info!(
9103                module_id,
9104                pid = child.pid,
9105                budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
9106                "module has no connection yet; requesting stop by signal"
9107            );
9108        }
9109        request_graceful_stop(module_id, &child);
9110    }
9111
9112    let exit_report = match timeout(drain_timeout, child.wait()).await {
9113        Ok(Ok(status)) => classify_reaped_child_exit(snapshot, &child, &status),
9114        Ok(Err(source)) => {
9115            fail_snapshot(snapshot, Some(module_id), None);
9116            return Err(SuperviseError::Wait {
9117                module_id: module_id.to_string(),
9118                source,
9119            });
9120        }
9121        Err(_) => {
9122            // Mirror the sibling arm above: state is already `Draining`, and an
9123            // error propagated from here would strand it there -- a state
9124            // `set_enabled(true)` cannot heal (`revive_terminal` matches only
9125            // `Failed | Stopped`), leaving an operator Restart as the only exit.
9126            // `Failed` before `?` keeps the module operator-visible and
9127            // revivable. Trigger is an ESRCH race (process exits between the
9128            // drain timeout firing and the kill) or a post-kill wait failure
9129            // (issue #34).
9130            //
9131            // Logged because the kill is otherwise visible only as signal 9 in
9132            // the terminal ring, and the budget it follows can be long enough
9133            // that consumers see a stretch of refusals with no stated cause.
9134            warn!(
9135                module_id,
9136                pid = child.pid,
9137                budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
9138                reason = ?final_state,
9139                ?stop_notice,
9140                "drain budget expired before the module exited; killing it"
9141            );
9142            child.start_kill().map_err(|source| {
9143                fail_snapshot(snapshot, Some(module_id), None);
9144                SuperviseError::Kill {
9145                    module_id: module_id.to_string(),
9146                    source,
9147                }
9148            })?;
9149            let status = child.wait().await.map_err(|source| {
9150                fail_snapshot(snapshot, Some(module_id), None);
9151                SuperviseError::Wait {
9152                    module_id: module_id.to_string(),
9153                    source,
9154                }
9155            })?;
9156            classify_reaped_child_exit(snapshot, &child, &status)
9157        }
9158    };
9159
9160    update_snapshot(snapshot, Some(module_id), |state| {
9161        state.state = final_state;
9162        if let Some(enabled) = enabled {
9163            state.enabled = enabled;
9164        }
9165        clear_current_process_facts(state);
9166        state.last_exit = Some(exit_report.clone());
9167        if exit_report.kind == ExitKind::DeliberateSeverance {
9168            state.lifetime_restarts += 1;
9169        }
9170    })?;
9171    let detail = lock_snapshot(snapshot)?.drain_disposition_detail.take();
9172    record_terminal_with_detail(
9173        module_id,
9174        terminal_ring,
9175        spawn_events,
9176        &exit_report,
9177        terminal_disposition(final_state),
9178        detail,
9179    );
9180    child.drain_stderr(module_id).await;
9181
9182    release_dead_registration(registry, forwarding, snapshot, module_id).await
9183}
9184
9185/// Ask a child that nothing else has asked to stop, by signal.
9186///
9187/// A registered subc module is asked over its own connection: the drain sends
9188/// `route.closing`/`route.closed` to its consumers, a GOODBYE per route, then a
9189/// module GOODBYE, and the module stops itself. A module that speaks no subc
9190/// wire receives none of that, and neither does a subc module that has not
9191/// registered yet, so for them the drain budget would be pure delay in front of
9192/// a SIGKILL -- and for a process with a store to flush (JetStream is the
9193/// reason `protocol: "none"` exists) a SIGKILL turns every ordinary teardown
9194/// into a recovery on the next start.
9195///
9196/// NEVER CALLED FOR A MODULE THAT WAS TOLD OVER ITS CONNECTION, and that is a
9197/// rule rather than an optimisation: that module's graceful stop is already
9198/// running by the time its child is drained, and a signal would race it.
9199///
9200/// Best-effort by construction. A child that has already exited is the ordinary
9201/// case rather than an error (the kill lands on a reaped or exiting pid), so a
9202/// failure is logged at debug and the wait-then-kill below still decides the
9203/// outcome.
9204#[cfg(unix)]
9205fn request_graceful_stop(module_id: &str, child: &SupervisedChild) {
9206    let Some(pid) = child
9207        .id()
9208        .and_then(|pid| i32::try_from(pid).ok())
9209        .and_then(rustix::process::Pid::from_raw)
9210    else {
9211        debug!(
9212            module_id,
9213            "no pid to signal for teardown; falling through to the drain wait"
9214        );
9215        return;
9216    };
9217    match rustix::process::kill_process(pid, rustix::process::Signal::TERM) {
9218        Ok(()) => debug!(
9219            module_id,
9220            "sent SIGTERM to a module nothing else asked to stop"
9221        ),
9222        Err(err) => debug!(
9223            module_id,
9224            error = %err,
9225            "SIGTERM to module failed; the drain wait and kill still apply"
9226        ),
9227    }
9228}
9229
9230/// Windows has no SIGTERM and no portable stand-in for one. The graceful stops
9231/// Windows does offer need cooperation this supervisor cannot assume: a console
9232/// control event requires sharing a console with the child, and `WM_CLOSE`
9233/// requires the child to pump a message loop. A supervised server process does
9234/// neither, so there is nothing to send and teardown is the wait followed by the
9235/// kill. Emulating a signal here would mean inventing a stop protocol, which is
9236/// the thing `protocol: "none"` exists to avoid.
9237#[cfg(not(unix))]
9238fn request_graceful_stop(module_id: &str, _child: &SupervisedChild) {
9239    debug!(
9240        module_id,
9241        "no graceful stop signal exists on this platform; teardown of a module nothing asked to stop waits, then kills"
9242    );
9243}
9244
9245fn terminal_disposition(final_state: ModuleState) -> TerminalDisposition {
9246    match final_state {
9247        ModuleState::Stopped => TerminalDisposition::Stopped,
9248        ModuleState::Disabled => TerminalDisposition::Disabled,
9249        ModuleState::Restarting => TerminalDisposition::Restarting,
9250        ModuleState::Failed => TerminalDisposition::Failed,
9251        ModuleState::Starting
9252        | ModuleState::Running
9253        | ModuleState::Unresponsive
9254        | ModuleState::Draining => {
9255            unreachable!("terminal exits only finish in terminal or restarting states")
9256        }
9257    }
9258}
9259
9260/// Release a reaped child's registration before allowing another spawn.
9261///
9262/// EOF is not a process-lifetime signal: an inherited socket can stay open
9263/// indefinitely, and serial frame dispatch can be waiting on egress instead of
9264/// reading EOF. After the normal release grace, request connection close (which
9265/// cancels both reads and dispatch), then allow one more release grace for the
9266/// connection guard's forwarding cleanup. Never evict a different connection.
9267async fn release_dead_registration(
9268    registry: &Registry,
9269    forwarding: Option<&ForwardingTable>,
9270    snapshot: &SharedSnapshot,
9271    module_id: &str,
9272) -> Result<(), SuperviseError> {
9273    let result = async {
9274        let registration = registry
9275            .get_module(module_id)
9276            .map_err(SuperviseError::Registry)?;
9277        match wait_for_registration_release(registry, module_id, REGISTRY_RELEASE_TIMEOUT).await {
9278            Ok(()) => return Ok(()),
9279            Err(SuperviseError::RegistrationStillActive { .. }) => {}
9280            Err(err) => return Err(err),
9281        }
9282        let pid = lock_snapshot(snapshot)?.reaped_pid;
9283        if let (Some(registration), Some(forwarding), Some(pid)) = (registration, forwarding, pid) {
9284            warn!(
9285                module_id,
9286                pid,
9287                connection_id = registration.connection_id.get(),
9288                "reaped module registration outlived release grace; closing dead connection"
9289            );
9290            forwarding.request_connection_close(
9291                registration.connection_id,
9292                CloseReason::new(
9293                    "supervised_process_reaped",
9294                    format!("module '{module_id}' pid {pid} exited"),
9295                ),
9296            );
9297            wait_for_slot_registration_release(
9298                registry,
9299                crate::registry::RegistrationSlot::Connection(registration.connection_id),
9300                REGISTRY_RELEASE_TIMEOUT,
9301            )
9302            .await?;
9303        }
9304        wait_for_registration_release(registry, module_id, Duration::ZERO).await
9305    }
9306    .await;
9307    if let Err(err) = &result {
9308        fail_snapshot(snapshot, Some(module_id), None);
9309        error!(module_id, error = %err, "registration release failed after child exit; module is failed and start can retry");
9310    }
9311    result
9312}
9313
9314/// Wait for the ACTIVE registration of `module_id` to go away, which is what a
9315/// plain stop or restart waits for before it spawns a replacement.
9316async fn wait_for_registration_release(
9317    registry: &Registry,
9318    module_id: &str,
9319    wait: Duration,
9320) -> Result<(), SuperviseError> {
9321    wait_for_slot_registration_release(
9322        registry,
9323        crate::registry::RegistrationSlot::Active(module_id),
9324        wait,
9325    )
9326    .await
9327}
9328
9329/// Wait for the registration in `slot` to go away.
9330///
9331/// Keyed on the slot rather than the bare module id because a successful swap
9332/// never empties the id's active slot (the promoted candidate is in it), so an
9333/// id-keyed wait for the incumbent's release would always time out. Draining a
9334/// swap's incumbent waits on `crate::registry::RegistrationSlot::Connection` with the
9335/// incumbent's connection instead.
9336async fn wait_for_slot_registration_release(
9337    registry: &Registry,
9338    slot: crate::registry::RegistrationSlot<'_>,
9339    wait: Duration,
9340) -> Result<(), SuperviseError> {
9341    let deadline = Instant::now() + wait;
9342    let mut release_events = registration_release_events().subscribe();
9343    let still_active = |registration: &crate::registry::ModuleRegistration| {
9344        SuperviseError::RegistrationStillActive {
9345            module_id: registration.manifest.module_id.clone(),
9346            waited: wait,
9347        }
9348    };
9349    loop {
9350        let _observed_generation = *release_events.borrow_and_update();
9351        let Some(registration) = registry
9352            .registration(slot)
9353            .map_err(SuperviseError::Registry)?
9354        else {
9355            return Ok(());
9356        };
9357
9358        let now = Instant::now();
9359        if now >= deadline {
9360            return Err(still_active(&registration));
9361        }
9362
9363        let remaining = deadline.saturating_duration_since(now);
9364        match timeout(remaining, release_events.changed()).await {
9365            Ok(Ok(())) | Ok(Err(_)) => {}
9366            Err(_) => return Err(still_active(&registration)),
9367        }
9368    }
9369}
9370
9371#[cfg(test)]
9372mod slot_registration_wait_tests {
9373    use super::*;
9374    use crate::registry::{ConnectionId, RegistrationSlot};
9375    use subc_protocol::manifest::ModuleManifest;
9376
9377    #[tokio::test]
9378    async fn enable_release_failure_is_failed_and_a_second_enable_retries() {
9379        let registry = Arc::new(Registry::default());
9380        let supervisor = Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::default());
9381        let runtime = supervisor.runtime_config();
9382        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
9383        let spec = ModuleSpec {
9384            module_id: "enable-stale-registration".to_string(),
9385            program: PathBuf::from("/missing/enable-retry-test"),
9386            args: Vec::new(),
9387            env: Vec::new(),
9388            reserved: false,
9389            reserved_prefixes: Vec::new(),
9390            protocol: ModuleProtocol::Subc,
9391            overlap: Default::default(),
9392        };
9393        let connection = ConnectionId::new(90);
9394        registry
9395            .register_with_control_ops(
9396                ModuleManifest::builder(&spec.module_id, "0.1.0").build(),
9397                1,
9398                connection,
9399                Vec::new(),
9400            )
9401            .unwrap();
9402        let mut child = None;
9403        let err = set_child_enabled(
9404            &spec,
9405            &runtime,
9406            &registry,
9407            &supervisor.process_liveness,
9408            &snapshot,
9409            &mut child,
9410            true,
9411        )
9412        .await
9413        .unwrap_err();
9414        assert!(matches!(
9415            err,
9416            SuperviseError::RegistrationStillActive { .. }
9417        ));
9418        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
9419        assert!(child.is_none());
9420        registry.deregister_connection(connection).unwrap();
9421        let err = set_child_enabled(
9422            &spec,
9423            &runtime,
9424            &registry,
9425            &supervisor.process_liveness,
9426            &snapshot,
9427            &mut child,
9428            true,
9429        )
9430        .await
9431        .unwrap_err();
9432        assert!(
9433            matches!(err, SuperviseError::Spawn { .. }),
9434            "second enable must attempt a spawn: {err}"
9435        );
9436        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
9437    }
9438
9439    const INCUMBENT: u64 = 1;
9440    const CANDIDATE: u64 = 2;
9441
9442    fn swapped_registry() -> Arc<Registry> {
9443        let registry = Arc::new(Registry::default());
9444        let manifest = ModuleManifest::builder("m", "0.1.0").build();
9445        registry
9446            .register_with_control_ops(
9447                manifest.clone(),
9448                1,
9449                ConnectionId::new(INCUMBENT),
9450                Vec::new(),
9451            )
9452            .unwrap();
9453        registry
9454            .register_candidate_with_control_ops(
9455                manifest,
9456                1,
9457                ConnectionId::new(CANDIDATE),
9458                Vec::new(),
9459            )
9460            .unwrap();
9461        registry
9462    }
9463
9464    /// After a promotion the id's active slot is held by the new process, so an
9465    /// id-keyed wait for the incumbent's release can never succeed; the
9466    /// connection-keyed wait completes as soon as the incumbent deregisters.
9467    #[tokio::test]
9468    async fn incumbent_release_is_awaited_by_connection_not_by_module_id() {
9469        let registry = swapped_registry();
9470        registry.promote_candidate("m").unwrap().unwrap();
9471
9472        assert!(matches!(
9473            wait_for_registration_release(&registry, "m", Duration::from_millis(50)).await,
9474            Err(SuperviseError::RegistrationStillActive { .. })
9475        ));
9476
9477        // Still held while the incumbent's connection has not deregistered.
9478        assert!(matches!(
9479            wait_for_slot_registration_release(
9480                &registry,
9481                RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
9482                Duration::from_millis(50),
9483            )
9484            .await,
9485            Err(SuperviseError::RegistrationStillActive { .. })
9486        ));
9487
9488        let releaser = Arc::clone(&registry);
9489        let release = tokio::spawn(async move {
9490            sleep(Duration::from_millis(20)).await;
9491            releaser
9492                .deregister_connection(ConnectionId::new(INCUMBENT))
9493                .unwrap();
9494            notify_registration_release();
9495        });
9496        wait_for_slot_registration_release(
9497            &registry,
9498            RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
9499            Duration::from_secs(5),
9500        )
9501        .await
9502        .expect("the incumbent's own registration is released");
9503        release.await.unwrap();
9504        assert!(registry.get_module("m").unwrap().is_some());
9505    }
9506
9507    /// The candidate slot is waited on separately from the active slot: the
9508    /// incumbent's registration neither holds up nor stands in for it.
9509    #[tokio::test]
9510    async fn candidate_slot_wait_ignores_the_incumbents_registration() {
9511        let registry = swapped_registry();
9512        assert!(matches!(
9513            wait_for_slot_registration_release(
9514                &registry,
9515                RegistrationSlot::Candidate("m"),
9516                Duration::from_millis(50),
9517            )
9518            .await,
9519            Err(SuperviseError::RegistrationStillActive { .. })
9520        ));
9521        registry
9522            .deregister_connection(ConnectionId::new(CANDIDATE))
9523            .unwrap();
9524        wait_for_slot_registration_release(
9525            &registry,
9526            RegistrationSlot::Candidate("m"),
9527            Duration::from_millis(50),
9528        )
9529        .await
9530        .expect("a candidate slot with no candidate is released");
9531        assert!(registry
9532            .registration(RegistrationSlot::Active("m"))
9533            .unwrap()
9534            .is_some());
9535    }
9536}
9537
9538fn classify_exit(status: &ExitStatus) -> ExitReport {
9539    ExitReport {
9540        kind: if status.success() {
9541            ExitKind::Clean
9542        } else {
9543            ExitKind::Crash
9544        },
9545        code: status.code(),
9546        signal: exit_signal(status),
9547        at_ms: unix_ms_now(),
9548    }
9549}
9550
9551/// The terminal record for a module whose `wait()` call itself errored (e.g. the
9552/// child was already reaped out-of-band). There is no `ExitStatus` to read a code
9553/// or signal from -- `None`/`None` is the honest shape, not a guess -- but the
9554/// disposition still must be `Failed` so the terminal ring is not silently missing
9555/// an entry, matching what `fail_snapshot` records for this same arm.
9556fn wait_error_exit_report() -> ExitReport {
9557    ExitReport {
9558        kind: ExitKind::Crash,
9559        code: None,
9560        signal: None,
9561        at_ms: unix_ms_now(),
9562    }
9563}
9564
9565#[cfg(unix)]
9566fn exit_signal(status: &ExitStatus) -> Option<i32> {
9567    use std::os::unix::process::ExitStatusExt;
9568
9569    status.signal()
9570}
9571
9572#[cfg(not(unix))]
9573fn exit_signal(_status: &ExitStatus) -> Option<i32> {
9574    None
9575}
9576
9577/// Give an operator-touched module its full crash budget back.
9578///
9579/// Named for the counter it used to zero; it now empties the in-window ring,
9580/// which is the same act. `lifetime_restarts` is untouched on purpose -- the
9581/// ledger of what happened survives every operator action.
9582fn reset_restart_count(snapshot: &SharedSnapshot, module_id: &str) -> Result<(), SuperviseError> {
9583    update_snapshot(snapshot, Some(module_id), |state| {
9584        state.clear_crash_restarts();
9585    })
9586}
9587
9588fn set_running(
9589    snapshot: &SharedSnapshot,
9590    child: &SupervisedChild,
9591    module_id: &str,
9592    spawn_events: &SpawnEventFeed,
9593) -> Result<(), SuperviseError> {
9594    let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
9595        module_id: Some(module_id.to_string()),
9596    })?;
9597    state.spawn_generation = spawn_events.emit_spawned(module_id, child.pid, child.spawned_at_ms);
9598    if std::mem::take(&mut state.coalesced_restart_pending) {
9599        let generation = state.spawn_generation;
9600        info!(module_id, "restart already satisfied by generation {generation}; coalesced pending request completed");
9601    }
9602    state.drain_disposition_detail = None;
9603    state.spawn_failure = None;
9604    // Every caller of this is a plain spawn, which always uses the primary key;
9605    // a promoted swap candidate sets the flag itself after this returns.
9606    state.in_alternate_slot = false;
9607    state.configuration_updated_since_spawn = false;
9608    state.spawned_protocol = Some(child.protocol);
9609    state.state = ModuleState::Running;
9610    state.enabled = true;
9611    state.process_alive = true;
9612    state.pid = child.id();
9613    #[cfg(target_os = "macos")]
9614    {
9615        state.report_ready = Some(Arc::clone(&child.report_ready));
9616    }
9617    state.spawned_at_ms = Some(child.spawned_at_ms);
9618    state.spawned_from = Some(child.spawned_from.clone());
9619    state.spawned_file_identity = child.spawned_file_identity;
9620    state.process_start_time = child.process_start_time;
9621    Ok(())
9622}
9623
9624fn clear_current_process_facts(state: &mut SupervisorSnapshot) {
9625    state.process_alive = false;
9626    state.spawned_protocol = None;
9627    state.pid = None;
9628    #[cfg(target_os = "macos")]
9629    {
9630        state.report_ready = None;
9631    }
9632    state.spawned_at_ms = None;
9633    state.spawned_from = None;
9634    state.spawned_file_identity = None;
9635    state.process_start_time = None;
9636    state.deliberate_severance = None;
9637}
9638
9639#[cfg(test)]
9640fn record_deliberate_severance(
9641    snapshot: &SharedSnapshot,
9642    identity: ProcessIdentity,
9643) -> Result<(), SuperviseError> {
9644    update_snapshot(snapshot, None, |state| {
9645        state.deliberate_severance = Some(identity);
9646    })
9647}
9648
9649fn apply_deliberate_severance_marker(
9650    snapshot: &SharedSnapshot,
9651    exited_identity: Option<ProcessIdentity>,
9652    mut exit_report: ExitReport,
9653) -> ExitReport {
9654    let marker = lock_snapshot(snapshot)
9655        .ok()
9656        .and_then(|mut state| state.deliberate_severance.take());
9657    if marker.is_some() && marker == exited_identity {
9658        exit_report.kind = ExitKind::DeliberateSeverance;
9659    }
9660    exit_report
9661}
9662
9663fn classify_reaped_child_exit(
9664    snapshot: &SharedSnapshot,
9665    child: &SupervisedChild,
9666    status: &ExitStatus,
9667) -> ExitReport {
9668    let _ = update_snapshot(snapshot, None, |state| {
9669        state.reaped_pid = Some(child.pid);
9670        state.spawn_failure = child.spawn_failure.clone();
9671    });
9672    apply_deliberate_severance_marker(snapshot, child.process_identity(), classify_exit(status))
9673}
9674
9675fn fail_snapshot(
9676    snapshot: &SharedSnapshot,
9677    module_id: Option<&str>,
9678    last_exit: Option<ExitReport>,
9679) {
9680    if let Err(err) = update_snapshot(snapshot, module_id, |state| {
9681        state.state = ModuleState::Failed;
9682        clear_current_process_facts(state);
9683        if let Some(last_exit) = last_exit {
9684            state.last_exit = Some(last_exit);
9685        }
9686    }) {
9687        error!(error = %err, "failed to mark supervisor state failed");
9688    }
9689}
9690
9691fn update_snapshot(
9692    snapshot: &SharedSnapshot,
9693    module_id: Option<&str>,
9694    update: impl FnOnce(&mut SupervisorSnapshot),
9695) -> Result<(), SuperviseError> {
9696    let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
9697        module_id: module_id.map(ToOwned::to_owned),
9698    })?;
9699    update(&mut state);
9700    Ok(())
9701}
9702
9703const SLOW_SNAPSHOT_LOCK_THRESHOLD: Duration = Duration::from_millis(250);
9704
9705fn lock_snapshot_for_control<'a>(
9706    snapshot: &'a SharedSnapshot,
9707    module_id: &str,
9708    caller: &'static str,
9709) -> Result<std::sync::MutexGuard<'a, SupervisorSnapshot>, SuperviseError> {
9710    let started_at = Instant::now();
9711    let guard = lock_snapshot(snapshot)?;
9712    let waited = started_at.elapsed();
9713    if waited >= SLOW_SNAPSHOT_LOCK_THRESHOLD {
9714        warn!(
9715            module_id = %module_id,
9716            waited_ms = waited.as_millis() as u64,
9717            caller = %caller,
9718            "slow snapshot lock"
9719        );
9720    }
9721    Ok(guard)
9722}
9723
9724fn lock_snapshot(
9725    snapshot: &SharedSnapshot,
9726) -> Result<std::sync::MutexGuard<'_, SupervisorSnapshot>, SuperviseError> {
9727    snapshot
9728        .lock()
9729        .map_err(|_| SuperviseError::StatePoisoned { module_id: None })
9730}
9731
9732#[cfg(test)]
9733mod terminal_history_tests {
9734    use std::{
9735        path::PathBuf,
9736        sync::Arc,
9737        time::{Duration, Instant},
9738    };
9739
9740    use tokio::{sync::mpsc, time::sleep};
9741
9742    use super::{
9743        apply_deliberate_severance_marker, daemon_will_restart, drain_child_to_state,
9744        drained_after_quiescence_wait, handle_reload_spawn_failure, health_restart_child,
9745        lock_snapshot, on_child_exit, record_deliberate_severance, record_wait_error_terminal,
9746        reset_restart_count, spawn_and_mark_running, update_snapshot, wait_error_exit_report,
9747        ExitKind, ExitReport, ModuleProtocol, ModuleSpec, ModuleState, NextAction, ProcessIdentity,
9748        RestartPolicy, SpawnEventKind, StopNotice, SuperviseError, SupervisedConfiguration,
9749        SupervisedModule, SupervisedModuleInner, Supervisor, SupervisorHandle,
9750        SupervisorHealthStatus, SupervisorSnapshot,
9751    };
9752    // The supervisor's clock, distinct from the `std::time::Instant` these tests
9753    // use for their own wall-clock deadlines: crash-restart instants must be on
9754    // the same clock the production code stamps them with, which is tokio's (and
9755    // is what `start_paused` tests can move).
9756    use super::Instant as ClockInstant;
9757    use crate::{
9758        registry::Registry,
9759        terminal_ring::{TerminalRing, TerminalRingConfig},
9760    };
9761    use std::sync::Mutex;
9762    use subc_control::TerminalDisposition;
9763
9764    /// See the twin in `control.rs` for why this derives the path from
9765    /// `current_exe()` and why the existence check is here: `--lib` alone does
9766    /// not build `[[bin]]` targets, and a bare spawn then fails with a raw
9767    /// `NotFound` that reads as a broken test rather than an unbuilt dependency.
9768    pub(super) fn fake_aft_stub_path() -> PathBuf {
9769        let mut path = std::env::current_exe().expect("current_exe available in tests");
9770        path.pop();
9771        path.pop();
9772        path.push(if cfg!(windows) {
9773            "fake-aft-stub.exe"
9774        } else {
9775            "fake-aft-stub"
9776        });
9777        assert!(
9778            path.exists(),
9779            "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
9780             [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
9781            path.display()
9782        );
9783        path
9784    }
9785
9786    #[test]
9787    fn reserved_never_spawned_refuses_every_hello() {
9788        // The canary hole: a reserved id whose module has never spawned had NO
9789        // gate entry and admitted anyone -- the reservation protected the nonce
9790        // holder, not the NAME. Now the entry is present with no legitimate
9791        // holder and refuses all comers.
9792        let supervisor = SupervisorHandle::default();
9793        supervisor.apply_identity_configuration(&ModuleSpec {
9794            module_id: "never-spawned".to_string(),
9795            program: PathBuf::from("/usr/bin/false"),
9796            args: Vec::new(),
9797            env: Vec::new(),
9798            reserved: true,
9799            reserved_prefixes: Vec::new(),
9800            protocol: ModuleProtocol::Subc,
9801            overlap: Default::default(),
9802        });
9803        assert!(
9804            supervisor
9805                .reserved_hello_rejection("never-spawned", Some("any-forged-nonce"))
9806                .is_some(),
9807            "forged nonce must refuse on a reserved never-spawned id"
9808        );
9809        assert!(
9810            supervisor
9811                .reserved_hello_rejection("never-spawned", None)
9812                .is_some(),
9813            "absent nonce must refuse on a reserved never-spawned id"
9814        );
9815        // And a real spawn nonce minted later admits exactly that nonce.
9816        supervisor.set_spawn_nonce("never-spawned", "minted".to_string());
9817        supervisor.apply_identity_configuration(&ModuleSpec {
9818            module_id: "never-spawned".to_string(),
9819            program: PathBuf::from("/usr/bin/false"),
9820            args: Vec::new(),
9821            env: Vec::new(),
9822            reserved: true,
9823            reserved_prefixes: Vec::new(),
9824            protocol: ModuleProtocol::Subc,
9825            overlap: Default::default(),
9826        });
9827        assert!(supervisor
9828            .reserved_hello_rejection("never-spawned", Some("minted"))
9829            .is_none());
9830        assert!(supervisor
9831            .reserved_hello_rejection("never-spawned", Some("forged"))
9832            .is_some());
9833    }
9834
9835    /// Put `count` crash restarts on a snapshot's ring as if they had all just
9836    /// happened, which is what "spent budget" looks like to every reader.
9837    fn seed_crash_restarts(state: &mut SupervisorSnapshot, count: u32) {
9838        let now = ClockInstant::now();
9839        for _ in 0..count {
9840            state.crash_restarts.push_back(now);
9841        }
9842    }
9843
9844    /// Age the oldest recorded restart out of `window`, standing in for the hours
9845    /// that would otherwise have to pass. Injecting the instant is the point: a
9846    /// test that slept a real window would take ten minutes and still prove less.
9847    fn age_oldest_crash_restart_out_of_window(state: &mut SupervisorSnapshot, window: Duration) {
9848        let aged = state
9849            .crash_restarts
9850            .front()
9851            .expect("a crash restart must be recorded before it can be aged")
9852            .checked_sub(window + Duration::from_secs(1))
9853            .expect("the test clock is far enough from its origin to age an instant");
9854        state.crash_restarts[0] = aged;
9855    }
9856
9857    fn snapshot_with_restarts(enabled: bool, count: u32) -> SupervisorSnapshot {
9858        let mut state = SupervisorSnapshot::new(ModuleState::Running, enabled);
9859        seed_crash_restarts(&mut state, count);
9860        state
9861    }
9862
9863    #[test]
9864    fn daemon_owned_recovery_predicate_uses_the_pre_increment_budget() {
9865        let policy = RestartPolicy::new(3, Duration::ZERO);
9866        let now = ClockInstant::now();
9867        assert!(daemon_will_restart(
9868            &mut snapshot_with_restarts(true, 2),
9869            &policy,
9870            now
9871        ));
9872        assert!(!daemon_will_restart(
9873            &mut snapshot_with_restarts(true, 3),
9874            &policy,
9875            now
9876        ));
9877        assert!(!daemon_will_restart(
9878            &mut snapshot_with_restarts(false, 0),
9879            &policy,
9880            now
9881        ));
9882    }
9883
9884    #[test]
9885    fn crash_restart_backoff_escalates_with_in_window_count() {
9886        let policy = RestartPolicy::new(4, Duration::from_millis(100))
9887            .with_max_backoff(Duration::from_secs(30));
9888        let now = ClockInstant::now();
9889        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
9890        let schedules = (0..4)
9891            .map(|_| {
9892                state
9893                    .next_crash_restart(&policy, now)
9894                    .expect("the test policy allows four crash restarts")
9895            })
9896            .collect::<Vec<_>>();
9897
9898        assert_eq!(
9899            schedules
9900                .iter()
9901                .map(|schedule| schedule.restart_in_window)
9902                .collect::<Vec<_>>(),
9903            vec![0, 1, 2, 3]
9904        );
9905        assert_eq!(
9906            schedules
9907                .iter()
9908                .map(|schedule| schedule.delay)
9909                .collect::<Vec<_>>(),
9910            vec![
9911                Duration::from_millis(100),
9912                Duration::from_secs(1),
9913                Duration::from_secs(10),
9914                Duration::from_secs(30),
9915            ]
9916        );
9917    }
9918
9919    #[test]
9920    fn crash_restart_backoff_resets_after_ring_clear() {
9921        let policy = RestartPolicy::new(3, Duration::from_millis(100));
9922        let now = ClockInstant::now();
9923        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
9924        assert_eq!(
9925            state.next_crash_restart(&policy, now).unwrap().delay,
9926            Duration::from_millis(100)
9927        );
9928        assert_eq!(
9929            state.next_crash_restart(&policy, now).unwrap().delay,
9930            Duration::from_secs(1)
9931        );
9932
9933        state.clear_crash_restarts();
9934        let schedule = state
9935            .next_crash_restart(&policy, now)
9936            .expect("a cleared ring must allow another restart");
9937        assert_eq!(schedule.restart_in_window, 0);
9938        assert_eq!(schedule.delay, Duration::from_millis(100));
9939    }
9940
9941    #[test]
9942    fn crash_restart_backoff_ignores_aged_restarts() {
9943        let policy = RestartPolicy::new(3, Duration::from_millis(100));
9944        let now = ClockInstant::now();
9945        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
9946        state
9947            .next_crash_restart(&policy, now)
9948            .expect("the first restart is allowed");
9949        state
9950            .next_crash_restart(&policy, now)
9951            .expect("the second restart is allowed");
9952        state.crash_restarts[0] = now
9953            .checked_sub(policy.window + Duration::from_secs(1))
9954            .expect("the fake clock can age a restart past the window");
9955
9956        let schedule = state
9957            .next_crash_restart(&policy, now)
9958            .expect("an aged restart must release its slot");
9959        assert_eq!(schedule.restart_in_window, 1);
9960        assert_eq!(schedule.delay, Duration::from_secs(1));
9961        assert_eq!(state.crash_restarts.len(), 2);
9962    }
9963
9964    /// The budget is a rate: the same three spent restarts refuse a respawn
9965    /// while they are recent and allow one once they have aged past the window.
9966    /// Nothing about the module changed in between, which is the whole point.
9967    #[test]
9968    fn a_budget_spent_before_the_window_no_longer_refuses() {
9969        let policy = RestartPolicy::new(3, Duration::ZERO);
9970        let mut state = snapshot_with_restarts(true, 3);
9971        let now = ClockInstant::now();
9972        assert!(!daemon_will_restart(&mut state, &policy, now));
9973
9974        assert!(daemon_will_restart(
9975            &mut state,
9976            &policy,
9977            now + policy.window + Duration::from_secs(1)
9978        ));
9979        assert!(
9980            state.crash_restarts.is_empty(),
9981            "reading the budget must drop the instants that left the window"
9982        );
9983    }
9984
9985    fn module_with_recovery_snapshot(
9986        state: ModuleState,
9987        enabled: bool,
9988        restart_count: u32,
9989    ) -> SupervisedModule {
9990        let registry = Arc::new(Registry::default());
9991        let supervisor =
9992            Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::new(3, Duration::ZERO));
9993        let runtime = supervisor.runtime_config();
9994        let spec = ModuleSpec {
9995            module_id: "recovery-snapshot".to_string(),
9996            program: fake_aft_stub_path(),
9997            args: Vec::new(),
9998            env: Vec::new(),
9999            reserved: false,
10000            reserved_prefixes: Vec::new(),
10001            protocol: ModuleProtocol::Subc,
10002            overlap: Default::default(),
10003        };
10004        let mut snapshot = SupervisorSnapshot::new(state, enabled);
10005        seed_crash_restarts(&mut snapshot, restart_count);
10006        // These tests read synthetic snapshots. A real child and monitor would
10007        // race those reads by replacing the requested state during startup.
10008        let (commands, _rx) = mpsc::channel(4);
10009        SupervisedModule {
10010            inner: Arc::new(SupervisedModuleInner {
10011                module_id: spec.module_id.clone(),
10012                registry,
10013                snapshot: Arc::new(Mutex::new(snapshot)),
10014                configuration: Arc::new(Mutex::new(SupervisedConfiguration {
10015                    spec,
10016                    health: runtime.health,
10017                })),
10018                stderr_ring: runtime.stderr_ring,
10019                terminal_ring: runtime.terminal_ring,
10020                commands,
10021                monitor: Mutex::new(None),
10022                restart_policy: runtime.restart_policy,
10023                effective_drain_timeout: runtime.effective_drain_timeout,
10024                provenance_probe: supervisor.provenance_probe.clone(),
10025            }),
10026        }
10027    }
10028
10029    #[cfg(target_os = "linux")]
10030    #[tokio::test]
10031    async fn no_cgroup_placement_does_not_block_fake_aft_stub_spawn() {
10032        let supervisor =
10033            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
10034                .with_cgroup_placement(None);
10035        let result = supervisor.spawn(ModuleSpec {
10036            module_id: "no-cgroup-placement".to_string(),
10037            program: fake_aft_stub_path(),
10038            args: Vec::new(),
10039            env: Vec::new(),
10040            reserved: false,
10041            reserved_prefixes: Vec::new(),
10042            protocol: ModuleProtocol::Subc,
10043            overlap: Default::default(),
10044        });
10045
10046        assert!(
10047            result.is_ok(),
10048            "no delegation must not turn an otherwise valid spawn into a failure: {result:?}"
10049        );
10050    }
10051
10052    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10053    async fn undecided_snapshot_uses_shared_restart_predicate() {
10054        assert!(module_with_recovery_snapshot(ModuleState::Running, true, 2)
10055            .will_recover_after_connection_loss()
10056            .unwrap());
10057        assert!(
10058            !module_with_recovery_snapshot(ModuleState::Running, true, 3)
10059                .will_recover_after_connection_loss()
10060                .unwrap()
10061        );
10062    }
10063
10064    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10065    async fn restarting_snapshot_at_exhausted_budget_is_non_terminal() {
10066        assert!(
10067            module_with_recovery_snapshot(ModuleState::Restarting, true, 3)
10068                .will_recover_after_connection_loss()
10069                .unwrap()
10070        );
10071    }
10072
10073    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10074    async fn terminal_phase_snapshots_are_terminal_before_budget_exhaustion() {
10075        assert!(!module_with_recovery_snapshot(ModuleState::Failed, true, 0)
10076            .will_recover_after_connection_loss()
10077            .unwrap());
10078        assert!(
10079            !module_with_recovery_snapshot(ModuleState::Disabled, true, 0)
10080                .will_recover_after_connection_loss()
10081                .unwrap()
10082        );
10083    }
10084
10085    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10086    async fn warming_snapshot_is_limited_to_startup_phases() {
10087        for state in [
10088            ModuleState::Starting,
10089            ModuleState::Running,
10090            ModuleState::Restarting,
10091        ] {
10092            assert!(
10093                module_with_recovery_snapshot(state, true, 0)
10094                    .is_warming()
10095                    .unwrap(),
10096                "{state:?} should be warming"
10097            );
10098        }
10099        for state in [
10100            ModuleState::Unresponsive,
10101            ModuleState::Draining,
10102            ModuleState::Stopped,
10103            ModuleState::Failed,
10104            ModuleState::Disabled,
10105        ] {
10106            assert!(
10107                !module_with_recovery_snapshot(state, true, 0)
10108                    .is_warming()
10109                    .unwrap(),
10110                "{state:?} should not be warming"
10111            );
10112        }
10113    }
10114
10115    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10116    async fn terminal_history_survives_respawn_and_keeps_both_crashes_in_order() {
10117        let registry = Arc::new(Registry::default());
10118        let supervisor =
10119            Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::new(1, Duration::ZERO));
10120        let module = supervisor
10121            .spawn(ModuleSpec {
10122                module_id: "terminal-history".to_string(),
10123                program: fake_aft_stub_path(),
10124                args: Vec::new(),
10125                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
10126                reserved: false,
10127                reserved_prefixes: Vec::new(),
10128                protocol: ModuleProtocol::Subc,
10129                overlap: Default::default(),
10130            })
10131            .unwrap();
10132
10133        let deadline = Instant::now() + Duration::from_secs(5);
10134        loop {
10135            let history = module.terminal_history();
10136            if history.entries.len() == 2 {
10137                assert_eq!(module.status().unwrap().state, ModuleState::Failed);
10138                assert_eq!(history.dropped, 0);
10139                assert_eq!(
10140                    history
10141                        .entries
10142                        .iter()
10143                        .map(|entry| entry.exit_code)
10144                        .collect::<Vec<_>>(),
10145                    vec![Some(23), Some(23)]
10146                );
10147                assert!(history.entries[0].at_ms <= history.entries[1].at_ms);
10148                return;
10149            }
10150            assert!(
10151                Instant::now() < deadline,
10152                "module did not retain two terminal exits: {history:?}"
10153            );
10154            sleep(Duration::from_millis(10)).await;
10155        }
10156    }
10157
10158    /// A disable issued while a crash respawn is still backing off must preempt
10159    /// that respawn: the operator's stop wins, the disable must not queue behind
10160    /// the backoff, and the module must never come back up afterwards.
10161    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10162    async fn disable_during_crash_backoff_cancels_pending_respawn() {
10163        let backoff = Duration::from_secs(2);
10164        let supervisor = Supervisor::new_for_test(
10165            Arc::new(Registry::default()),
10166            RestartPolicy::new(10, backoff),
10167        );
10168        let module = supervisor
10169            .spawn(ModuleSpec {
10170                module_id: "disable-during-backoff".to_string(),
10171                program: fake_aft_stub_path(),
10172                args: Vec::new(),
10173                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
10174                reserved: false,
10175                reserved_prefixes: Vec::new(),
10176                protocol: ModuleProtocol::Subc,
10177                overlap: Default::default(),
10178            })
10179            .unwrap();
10180
10181        // Wait for the first crash to put the module into its backoff window.
10182        let deadline = Instant::now() + Duration::from_secs(5);
10183        loop {
10184            if module.status().unwrap().state == ModuleState::Restarting {
10185                break;
10186            }
10187            assert!(
10188                Instant::now() < deadline,
10189                "module never entered the crash backoff"
10190            );
10191            sleep(Duration::from_millis(10)).await;
10192        }
10193
10194        let started = Instant::now();
10195        module.set_enabled(false).await.unwrap();
10196        let waited = started.elapsed();
10197
10198        assert!(
10199            waited < backoff / 2,
10200            "disable waited {waited:?} behind the {backoff:?} crash backoff; the operator command must preempt the pending respawn"
10201        );
10202        assert_eq!(module.status().unwrap().state, ModuleState::Disabled);
10203
10204        // Outlast the backoff: the respawn it was counting down to must never run.
10205        sleep(backoff + Duration::from_millis(500)).await;
10206        let status = module.status().unwrap();
10207        assert_eq!(status.state, ModuleState::Disabled);
10208        assert_eq!(
10209            status.spawn_generation, 1,
10210            "module respawned after the operator disabled it"
10211        );
10212    }
10213
10214    /// A `protocol: "none"` child that stays parked and exits 0 on SIGTERM:
10215    /// the shape of nats-server, the program this rule exists for.
10216    #[cfg(unix)]
10217    fn protocol_none_sigterm_exits_clean_spec(
10218        module_id: &str,
10219        dir: &std::path::Path,
10220    ) -> (ModuleSpec, PathBuf, PathBuf) {
10221        let ready = dir.join("ready");
10222        let marker = dir.join("sigterm");
10223        let spec = ModuleSpec {
10224            module_id: module_id.to_string(),
10225            program: fake_aft_stub_path(),
10226            args: Vec::new(),
10227            env: vec![
10228                ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
10229                (
10230                    "FAKE_AFT_SIGTERM_MARKER_PATH".to_string(),
10231                    marker.display().to_string(),
10232                ),
10233                (
10234                    "FAKE_AFT_NEVER_CONNECT_READY_PATH".to_string(),
10235                    ready.display().to_string(),
10236                ),
10237            ],
10238            reserved: false,
10239            reserved_prefixes: Vec::new(),
10240            protocol: ModuleProtocol::None,
10241            overlap: Default::default(),
10242        };
10243        (spec, ready, marker)
10244    }
10245
10246    /// Wait for a file the child writes, so a signal is never sent before the
10247    /// child's SIGTERM handler is installed (the default disposition would
10248    /// kill it by signal and the exit would not be clean).
10249    #[cfg(unix)]
10250    async fn wait_for_file(path: &std::path::Path) {
10251        let deadline = Instant::now() + Duration::from_secs(10);
10252        while !path.exists() {
10253            assert!(
10254                Instant::now() < deadline,
10255                "{} never appeared",
10256                path.display()
10257            );
10258            sleep(Duration::from_millis(10)).await;
10259        }
10260    }
10261
10262    /// A protocol-none module that exits 0 because something OUTSIDE the
10263    /// supervisor sent it SIGTERM is respawned, and the terminal record carries
10264    /// the crash-path disposition rather than `stopped`.
10265    #[cfg(unix)]
10266    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10267    async fn protocol_none_unrequested_clean_exit_restarts_as_a_crash() {
10268        let dir = subc_test_support::TestTempDir::new("none-unrequested-clean-exit");
10269        let (spec, ready, marker) =
10270            protocol_none_sigterm_exits_clean_spec("none-unrequested-clean-exit", dir.path());
10271        let supervisor = Supervisor::new_for_test(
10272            Arc::new(Registry::default()),
10273            RestartPolicy::new(3, Duration::ZERO),
10274        );
10275        let module = supervisor.spawn(spec).unwrap();
10276        wait_for_file(&ready).await;
10277        // The ready file proves the child installed its SIGTERM handler, not
10278        // that the supervisor has processed the privacy trampoline's exec
10279        // acknowledgement. On macOS status withholds the pid until then.
10280        let deadline = Instant::now() + Duration::from_secs(10);
10281        let first_pid = loop {
10282            let status = module.status().unwrap();
10283            if status.state == ModuleState::Running {
10284                if let Some(pid) = status.pid {
10285                    break pid;
10286                }
10287            }
10288            assert!(
10289                Instant::now() < deadline,
10290                "a running module must report its pid after exec confirmation: {status:?}"
10291            );
10292            sleep(Duration::from_millis(10)).await;
10293        };
10294
10295        rustix::process::kill_process(
10296            rustix::process::Pid::from_raw(i32::try_from(first_pid).unwrap()).unwrap(),
10297            rustix::process::Signal::TERM,
10298        )
10299        .unwrap();
10300
10301        let deadline = Instant::now() + Duration::from_secs(10);
10302        let respawned = loop {
10303            let status = module.status().unwrap();
10304            if status.state == ModuleState::Running
10305                && status.pid.is_some_and(|pid| pid != first_pid)
10306            {
10307                break status;
10308            }
10309            assert!(
10310                Instant::now() < deadline,
10311                "protocol-none module was not respawned after an unrequested clean exit: {status:?}"
10312            );
10313            sleep(Duration::from_millis(10)).await;
10314        };
10315        assert_eq!(respawned.spawn_generation, 2);
10316        assert!(
10317            marker.exists(),
10318            "the child must have exited through its SIGTERM handler (exit 0), or this proves nothing about clean exits"
10319        );
10320
10321        let history = module.terminal_history();
10322        assert_eq!(history.entries.len(), 1, "{history:?}");
10323        let entry = &history.entries[0];
10324        assert_eq!(entry.exit_code, Some(0));
10325        assert_eq!(entry.exit_kind, subc_control::TerminalExitKind::Clean);
10326        assert_eq!(entry.disposition, TerminalDisposition::Restarting);
10327
10328        module.stop().await.unwrap();
10329    }
10330
10331    /// Repeated unrequested clean exits of a protocol-none module spend the
10332    /// restart budget exactly as crashes do, and the module ends `failed` with
10333    /// the budget named.
10334    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10335    async fn protocol_none_repeated_clean_exits_exhaust_the_restart_budget() {
10336        let supervisor = Supervisor::new_for_test(
10337            Arc::new(Registry::default()),
10338            RestartPolicy::new(1, Duration::ZERO),
10339        );
10340        let module = supervisor
10341            .spawn(ModuleSpec {
10342                module_id: "none-clean-exit-budget".to_string(),
10343                program: fake_aft_stub_path(),
10344                args: Vec::new(),
10345                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
10346                reserved: false,
10347                reserved_prefixes: Vec::new(),
10348                protocol: ModuleProtocol::None,
10349                overlap: Default::default(),
10350            })
10351            .unwrap();
10352
10353        // Failed follows the terminal write; this deadline only bounds a hang,
10354        // not an assumed duration for the two launches or their exit recording.
10355        let deadline = Instant::now() + Duration::from_secs(10);
10356        loop {
10357            let status = module.status().unwrap();
10358            if status.state == ModuleState::Failed {
10359                break;
10360            }
10361            assert!(
10362                Instant::now() < deadline,
10363                "module never exhausted its budget: {status:?} {:?}",
10364                module.terminal_history()
10365            );
10366            sleep(Duration::from_millis(10)).await;
10367        }
10368        let history = module.terminal_history();
10369        assert_eq!(
10370            history
10371                .entries
10372                .iter()
10373                .map(|entry| (entry.exit_code, entry.disposition.clone()))
10374                .collect::<Vec<_>>(),
10375            vec![
10376                (Some(0), TerminalDisposition::Restarting),
10377                (Some(0), TerminalDisposition::Failed),
10378            ]
10379        );
10380        let detail = history.entries[1]
10381            .disposition_detail
10382            .as_deref()
10383            .expect("a budget failure names the budget");
10384        assert!(detail.contains("max_restarts=1"), "{detail}");
10385        assert_eq!(module.status().unwrap().spawn_generation, 2);
10386    }
10387
10388    #[test]
10389    fn restart_budget_failure_is_published_after_its_terminal_record() {
10390        let supervisor = Supervisor::new_for_test(
10391            Arc::new(Registry::default()),
10392            RestartPolicy::new(0, Duration::ZERO),
10393        );
10394        let runtime = supervisor.runtime_config();
10395        let spec = ModuleSpec {
10396            protocol: ModuleProtocol::None,
10397            ..windowed_crash_spec("budget-publication-order")
10398        };
10399        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(
10400            ModuleState::Running,
10401            true,
10402        )));
10403        let reaped_snapshot = snapshot.clone();
10404        let events = runtime.spawn_events.clone();
10405        let history = runtime.terminal_ring.clone();
10406
10407        // Exit recording takes the event-feed lock before the history lock.
10408        // Holding it pauses the writer after choosing a disposition but before
10409        // recording history, without assuming anything about scheduler timing.
10410        let before_record = events.0.lock().unwrap();
10411        let reap = std::thread::spawn(move || {
10412            tokio::runtime::Builder::new_current_thread()
10413                .enable_all()
10414                .build()
10415                .unwrap()
10416                .block_on(on_child_exit(
10417                    &spec,
10418                    runtime.restart_policy,
10419                    &supervisor.registry,
10420                    &reaped_snapshot,
10421                    &runtime.terminal_ring,
10422                    &runtime.spawn_events,
10423                    &runtime.child_roster,
10424                    ExitReport {
10425                        kind: ExitKind::Clean,
10426                        code: Some(0),
10427                        signal: None,
10428                        at_ms: 1,
10429                    },
10430                ))
10431        });
10432        // The deadline bounds a hung writer only; last_exit is the handshake.
10433        let deadline = Instant::now() + Duration::from_secs(10);
10434        let before_state = loop {
10435            let state = lock_snapshot(&snapshot).unwrap();
10436            if state.last_exit.is_some() {
10437                break state.state;
10438            }
10439            drop(state);
10440            assert!(Instant::now() < deadline, "exit decision was not reached");
10441            std::thread::yield_now();
10442        };
10443        let before_history = history.lock().unwrap().snapshot();
10444        drop(before_record);
10445        assert!(matches!(reap.join().unwrap(), NextAction::Stop { .. }));
10446        assert!(before_history.entries.is_empty());
10447        assert_ne!(
10448            before_state,
10449            ModuleState::Failed,
10450            "Failed was visible before its terminal record could be written"
10451        );
10452        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
10453        let history = history.lock().unwrap().snapshot();
10454        assert_eq!(history.entries.len(), 1);
10455        assert_eq!(history.entries[0].exit_code, Some(0));
10456        assert_eq!(history.entries[0].disposition, TerminalDisposition::Failed);
10457    }
10458
10459    #[tokio::test]
10460    async fn restart_budget_failure_remains_failed_when_journal_append_fails() {
10461        let dir = subc_test_support::TestTempDir::new("budget-journal-failure");
10462        let path = dir.join("terminals.jsonl");
10463        std::fs::create_dir(&path).unwrap();
10464        let supervisor = Supervisor::new_for_test(
10465            Arc::new(Registry::default()),
10466            RestartPolicy::new(0, Duration::ZERO),
10467        )
10468        .with_terminal_journal(path, "budget-journal-failure".into());
10469        let runtime = supervisor.runtime_config();
10470        let spec = windowed_crash_spec("budget-journal-failure");
10471        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10472        assert!(matches!(
10473            on_child_exit(
10474                &spec,
10475                runtime.restart_policy,
10476                &supervisor.registry,
10477                &snapshot,
10478                &runtime.terminal_ring,
10479                &runtime.spawn_events,
10480                &runtime.child_roster,
10481                crash_exit_report(1),
10482            )
10483            .await,
10484            NextAction::Stop { .. }
10485        ));
10486        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
10487        let history = runtime
10488            .terminal_ring
10489            .lock()
10490            .unwrap()
10491            .durable_history(&spec.module_id);
10492        assert!(history.journal_write_failures > 0);
10493        assert_eq!(history.entries.len(), 1);
10494        assert_eq!(history.entries[0].disposition, TerminalDisposition::Failed);
10495    }
10496
10497    #[test]
10498    fn restart_budget_failure_remains_failed_when_exit_recording_panics() {
10499        let supervisor = Supervisor::new_for_test(
10500            Arc::new(Registry::default()),
10501            RestartPolicy::new(0, Duration::ZERO),
10502        );
10503        let runtime = supervisor.runtime_config();
10504        let spec = windowed_crash_spec("budget-recording-panic");
10505        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10506        runtime.spawn_events.emit_spawned(&spec.module_id, 1, 1);
10507        // Exhausting the event sequence makes emit_exited panic before the
10508        // terminal write, exercising publication on the recording unwind.
10509        runtime.spawn_events.0.lock().unwrap().seq = u64::MAX;
10510        let result = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {
10511            tokio::runtime::Builder::new_current_thread()
10512                .enable_all()
10513                .build()
10514                .unwrap()
10515                .block_on(on_child_exit(
10516                    &spec,
10517                    runtime.restart_policy,
10518                    &supervisor.registry,
10519                    &snapshot,
10520                    &runtime.terminal_ring,
10521                    &runtime.spawn_events,
10522                    &runtime.child_roster,
10523                    crash_exit_report(1),
10524                ))
10525        }));
10526        let panic = result.err().expect("recording must still unwind");
10527        assert_eq!(
10528            panic.downcast_ref::<String>().map(String::as_str),
10529            Some("spawn event sequence exhausted")
10530        );
10531        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
10532    }
10533
10534    /// A stop the supervisor itself requests still stops a protocol-none
10535    /// module, even though the child answers the SIGTERM with exit 0.
10536    #[cfg(unix)]
10537    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10538    async fn protocol_none_requested_stop_and_disable_do_not_respawn() {
10539        for disable in [false, true] {
10540            let label = if disable {
10541                "none-requested-disable"
10542            } else {
10543                "none-requested-stop"
10544            };
10545            let dir = subc_test_support::TestTempDir::new(label);
10546            let (spec, ready, marker) = protocol_none_sigterm_exits_clean_spec(label, dir.path());
10547            let supervisor = Supervisor::new_for_test(
10548                Arc::new(Registry::default()),
10549                RestartPolicy::new(3, Duration::ZERO),
10550            );
10551            let module = supervisor.spawn(spec).unwrap();
10552            wait_for_file(&ready).await;
10553
10554            if disable {
10555                module.set_enabled(false).await.unwrap();
10556            } else {
10557                module.stop().await.unwrap();
10558            }
10559            assert!(
10560                marker.exists(),
10561                "{label}: the child must have left through its SIGTERM handler with exit 0"
10562            );
10563
10564            // Long enough for a zero-backoff respawn to have happened if the
10565            // exit had been treated as a crash.
10566            sleep(Duration::from_millis(500)).await;
10567            let status = module.status().unwrap();
10568            let expected = if disable {
10569                ModuleState::Disabled
10570            } else {
10571                ModuleState::Stopped
10572            };
10573            assert_eq!(status.state, expected, "{label}");
10574            assert_eq!(
10575                status.spawn_generation, 1,
10576                "{label}: respawned after a requested stop"
10577            );
10578            let history = module.terminal_history();
10579            assert_eq!(history.entries.len(), 1, "{label}: {history:?}");
10580            assert_eq!(history.entries[0].exit_code, Some(0), "{label}");
10581            assert_ne!(
10582                history.entries[0].disposition,
10583                TerminalDisposition::Restarting,
10584                "{label}"
10585            );
10586        }
10587    }
10588
10589    /// A subc-wire module that exits 0 on its own is still a stop: the
10590    /// protocol-none rule must not reach it.
10591    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10592    async fn subc_wire_clean_exit_is_still_a_stop() {
10593        let supervisor = Supervisor::new_for_test(
10594            Arc::new(Registry::default()),
10595            RestartPolicy::new(3, Duration::ZERO),
10596        );
10597        let module = supervisor
10598            .spawn(ModuleSpec {
10599                module_id: "wire-clean-exit".to_string(),
10600                program: fake_aft_stub_path(),
10601                args: Vec::new(),
10602                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
10603                reserved: false,
10604                reserved_prefixes: Vec::new(),
10605                protocol: ModuleProtocol::Subc,
10606                overlap: Default::default(),
10607            })
10608            .unwrap();
10609
10610        let deadline = Instant::now() + Duration::from_secs(10);
10611        while module.terminal_history().entries.is_empty() {
10612            assert!(Instant::now() < deadline, "module never exited");
10613            sleep(Duration::from_millis(10)).await;
10614        }
10615        // Long enough for a zero-backoff respawn to have happened.
10616        sleep(Duration::from_millis(500)).await;
10617        let status = module.status().unwrap();
10618        assert_eq!(status.state, ModuleState::Stopped);
10619        assert_eq!(status.spawn_generation, 1);
10620        let history = module.terminal_history();
10621        assert_eq!(history.entries.len(), 1, "{history:?}");
10622        assert_eq!(history.entries[0].exit_code, Some(0));
10623        assert_eq!(history.entries[0].disposition, TerminalDisposition::Stopped);
10624    }
10625
10626    #[cfg(unix)]
10627    #[tokio::test]
10628    async fn reload_exit_keeps_roster_until_terminal_is_recorded_during_shutdown() {
10629        let dir = subc_test_support::TestTempDir::new("reload-roster-terminal-order");
10630        let record = dir.join("live-children.json");
10631        let supervisor = Supervisor::new_for_test(
10632            Arc::new(Registry::default()),
10633            RestartPolicy::new(0, Duration::ZERO),
10634        );
10635        let mut runtime = supervisor.runtime_config();
10636        runtime.child_roster.record_to(record.clone());
10637        let gate = Arc::new(super::ReloadExitRecordGate::default());
10638        runtime.test_reload_exit_record_gate = Some(Arc::clone(&gate));
10639        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10640        let spec = ModuleSpec {
10641            module_id: "reload-exit-roster".into(),
10642            program: fake_aft_stub_path(),
10643            args: Vec::new(),
10644            env: vec![("FAKE_AFT_EXIT_CODE".into(), "23".into())],
10645            reserved: false,
10646            reserved_prefixes: Vec::new(),
10647            protocol: ModuleProtocol::Subc,
10648            overlap: Default::default(),
10649        };
10650        let mut child = None;
10651        let reload = super::finish_reload_child(
10652            &spec,
10653            &runtime,
10654            &supervisor.registry,
10655            &supervisor.process_liveness,
10656            &snapshot,
10657            &mut child,
10658        );
10659        tokio::pin!(reload);
10660        tokio::select! {
10661            result = &mut reload => panic!("reload missed the before-record gate: {result:?}"),
10662            _ = gate.reached.notified() => {}
10663        }
10664        assert!(runtime
10665            .terminal_ring
10666            .lock()
10667            .unwrap()
10668            .snapshot()
10669            .entries
10670            .is_empty());
10671        assert_eq!(
10672            crate::live_children::read_record(&record).unwrap().len(),
10673            1,
10674            "shutdown must still wait for the reaped child until its terminal record exists"
10675        );
10676        runtime.child_roster.close();
10677        gate.resume.notify_one();
10678        assert!(reload.await.is_err());
10679        assert!(crate::live_children::read_record(&record)
10680            .unwrap()
10681            .is_empty());
10682        let history = runtime.terminal_ring.lock().unwrap().snapshot();
10683        assert_eq!(history.entries.len(), 1);
10684        assert_eq!(
10685            history.entries[0].disposition,
10686            TerminalDisposition::DaemonShutdown
10687        );
10688    }
10689
10690    /// Each restart-producing arm has its own state transition. Keeping their
10691    /// lifetime count assertions adjacent prevents a later new arm from silently
10692    /// spending budget without recording the historical restart.
10693    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10694    async fn every_restart_increment_path_advances_lifetime_count() {
10695        let supervisor = Supervisor::new_for_test(
10696            Arc::new(Registry::default()),
10697            RestartPolicy::new(1, Duration::ZERO),
10698        );
10699        let runtime = supervisor.runtime_config();
10700        let spec = ModuleSpec {
10701            module_id: "lifetime-increment-path".to_string(),
10702            program: PathBuf::from("/unused/lifetime-increment-path"),
10703            args: Vec::new(),
10704            env: Vec::new(),
10705            reserved: false,
10706            reserved_prefixes: Vec::new(),
10707            protocol: ModuleProtocol::Subc,
10708            overlap: Default::default(),
10709        };
10710
10711        let crash_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10712        assert!(matches!(
10713            on_child_exit(
10714                &spec,
10715                runtime.restart_policy,
10716                &supervisor.registry,
10717                &crash_snapshot,
10718                &runtime.terminal_ring,
10719                &runtime.spawn_events,
10720                &runtime.child_roster,
10721                ExitReport {
10722                    kind: ExitKind::Crash,
10723                    code: Some(1),
10724                    signal: None,
10725                    at_ms: 1,
10726                },
10727            )
10728            .await,
10729            NextAction::Restart { schedule: _ }
10730        ));
10731        let (crash_restarts, crash_lifetime) = {
10732            let state = lock_snapshot(&crash_snapshot).unwrap();
10733            (state.crash_restarts.len(), state.lifetime_restarts)
10734        };
10735        assert_eq!(crash_restarts, 1);
10736        assert_eq!(crash_lifetime, 1);
10737
10738        let health_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10739        let mut health_child = None;
10740        assert!(matches!(
10741            health_restart_child(
10742                &spec,
10743                &runtime,
10744                &supervisor.registry,
10745                &supervisor.process_liveness,
10746                &health_snapshot,
10747                &mut health_child,
10748                SupervisorHealthStatus::Failing,
10749                None,
10750                2,
10751            )
10752            .await,
10753            Ok(())
10754        ));
10755        assert!(health_child.is_none());
10756        assert!(lock_snapshot(&health_snapshot).unwrap().respawn_pending);
10757        assert!(runtime.scheduled_respawn.lock().unwrap().take().is_some());
10758        let (health_restarts, health_lifetime) = {
10759            let state = lock_snapshot(&health_snapshot).unwrap();
10760            (state.crash_restarts.len(), state.lifetime_restarts)
10761        };
10762        assert_eq!(health_restarts, 1);
10763        assert_eq!(health_lifetime, 1);
10764
10765        let reload_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10766        let mut reload_child = None;
10767        assert!(matches!(
10768            handle_reload_spawn_failure(
10769                &spec,
10770                &runtime,
10771                &supervisor.process_liveness,
10772                &reload_snapshot,
10773                &mut reload_child,
10774                "forced reload spawn failure".to_string(),
10775            )
10776            .await,
10777            Err(SuperviseError::ReloadFailed { .. })
10778        ));
10779        let (reload_restarts, reload_lifetime) = {
10780            let state = lock_snapshot(&reload_snapshot).unwrap();
10781            (state.crash_restarts.len(), state.lifetime_restarts)
10782        };
10783        assert_eq!(reload_restarts, 1);
10784        assert_eq!(reload_lifetime, 1);
10785    }
10786
10787    #[tokio::test]
10788    async fn deliberately_severed_live_child_records_lifetime_without_spending_restart_budget() {
10789        let supervisor = Supervisor::new_for_test(
10790            Arc::new(Registry::default()),
10791            RestartPolicy::new(3, Duration::ZERO),
10792        );
10793        let runtime = supervisor.runtime_config();
10794        let spec = ModuleSpec {
10795            module_id: "deliberately-severed".to_string(),
10796            program: PathBuf::from("/unused/deliberately-severed"),
10797            args: Vec::new(),
10798            env: Vec::new(),
10799            reserved: false,
10800            reserved_prefixes: Vec::new(),
10801            protocol: ModuleProtocol::Subc,
10802            overlap: Default::default(),
10803        };
10804        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10805        let process = ProcessIdentity {
10806            pid: 41,
10807            start_time: 101,
10808        };
10809        record_deliberate_severance(&snapshot, process).unwrap();
10810        let exit_report = apply_deliberate_severance_marker(
10811            &snapshot,
10812            Some(process),
10813            ExitReport {
10814                kind: ExitKind::Crash,
10815                code: Some(1),
10816                signal: None,
10817                at_ms: 1,
10818            },
10819        );
10820        assert_eq!(exit_report.kind, ExitKind::DeliberateSeverance);
10821
10822        assert!(matches!(
10823            on_child_exit(
10824                &spec,
10825                runtime.restart_policy,
10826                &supervisor.registry,
10827                &snapshot,
10828                &runtime.terminal_ring,
10829                &runtime.spawn_events,
10830                &runtime.child_roster,
10831                exit_report,
10832            )
10833            .await,
10834            NextAction::Restart { schedule: _ }
10835        ));
10836        let state = lock_snapshot(&snapshot).unwrap();
10837        assert_eq!(state.lifetime_restarts, 1);
10838        assert_eq!(state.crash_restarts.len(), 0);
10839    }
10840
10841    #[tokio::test]
10842    async fn genuine_crash_spends_restart_budget_and_records_lifetime() {
10843        let supervisor = Supervisor::new_for_test(
10844            Arc::new(Registry::default()),
10845            RestartPolicy::new(3, Duration::ZERO),
10846        );
10847        let runtime = supervisor.runtime_config();
10848        let spec = ModuleSpec {
10849            module_id: "genuine-crash".to_string(),
10850            program: PathBuf::from("/unused/genuine-crash"),
10851            args: Vec::new(),
10852            env: Vec::new(),
10853            reserved: false,
10854            reserved_prefixes: Vec::new(),
10855            protocol: ModuleProtocol::Subc,
10856            overlap: Default::default(),
10857        };
10858        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10859
10860        assert!(matches!(
10861            on_child_exit(
10862                &spec,
10863                runtime.restart_policy,
10864                &supervisor.registry,
10865                &snapshot,
10866                &runtime.terminal_ring,
10867                &runtime.spawn_events,
10868                &runtime.child_roster,
10869                ExitReport {
10870                    kind: ExitKind::Crash,
10871                    code: Some(1),
10872                    signal: None,
10873                    at_ms: 1,
10874                },
10875            )
10876            .await,
10877            NextAction::Restart { schedule: _ }
10878        ));
10879        let state = lock_snapshot(&snapshot).unwrap();
10880        assert_eq!(state.lifetime_restarts, 1);
10881        assert_eq!(state.crash_restarts.len(), 1);
10882    }
10883
10884    fn crash_exit_report(at_ms: u64) -> ExitReport {
10885        ExitReport {
10886            kind: ExitKind::Crash,
10887            code: Some(1),
10888            signal: None,
10889            at_ms,
10890        }
10891    }
10892
10893    fn windowed_crash_spec(module_id: &str) -> ModuleSpec {
10894        ModuleSpec {
10895            module_id: module_id.to_string(),
10896            program: PathBuf::from("/unused").join(module_id),
10897            args: Vec::new(),
10898            env: Vec::new(),
10899            reserved: false,
10900            reserved_prefixes: Vec::new(),
10901            protocol: ModuleProtocol::Subc,
10902            overlap: Default::default(),
10903        }
10904    }
10905
10906    /// A real crash loop still stops. Three crashes with nothing aging out spend
10907    /// a budget of two and the third respawn is refused, and both surfaces an
10908    /// operator has -- the log line and the retained terminal record -- name the
10909    /// window rather than only the cap, because `max_restarts=2` alone is what
10910    /// this budget used to mean.
10911    #[tokio::test]
10912    async fn three_crashes_inside_the_window_stop_the_module_and_name_the_window() {
10913        let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::ERROR);
10914        let supervisor = Supervisor::new_for_test(
10915            Arc::new(Registry::default()),
10916            RestartPolicy::new(2, Duration::ZERO),
10917        );
10918        let runtime = supervisor.runtime_config();
10919        let spec = windowed_crash_spec("crash-loop-in-window");
10920        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10921
10922        for attempt in 1..=2 {
10923            assert!(
10924                matches!(
10925                    on_child_exit(
10926                        &spec,
10927                        runtime.restart_policy,
10928                        &supervisor.registry,
10929                        &snapshot,
10930                        &runtime.terminal_ring,
10931                        &runtime.spawn_events,
10932                        &runtime.child_roster,
10933                        crash_exit_report(attempt),
10934                    )
10935                    .await,
10936                    NextAction::Restart { schedule: _ }
10937                ),
10938                "crash {attempt} is inside the budget and must respawn"
10939            );
10940        }
10941
10942        assert!(matches!(
10943            on_child_exit(
10944                &spec,
10945                runtime.restart_policy,
10946                &supervisor.registry,
10947                &snapshot,
10948                &runtime.terminal_ring,
10949                &runtime.spawn_events,
10950                &runtime.child_roster,
10951                crash_exit_report(3),
10952            )
10953            .await,
10954            NextAction::Stop { .. }
10955        ));
10956
10957        {
10958            let state = lock_snapshot(&snapshot).unwrap();
10959            assert_eq!(state.state, ModuleState::Failed);
10960            assert_eq!(state.crash_restarts.len(), 2);
10961            assert_eq!(state.lifetime_restarts, 2);
10962        }
10963
10964        let history = runtime
10965            .terminal_ring
10966            .lock()
10967            .expect("terminal ring is not poisoned")
10968            .snapshot();
10969        let last = history
10970            .entries
10971            .last()
10972            .expect("the refused crash is retained");
10973        assert_eq!(last.disposition, TerminalDisposition::Failed);
10974        assert_eq!(
10975            last.disposition_detail.as_deref(),
10976            Some("crash budget exhausted: max_restarts=2 within window_secs=600")
10977        );
10978
10979        let captured = crate::router::test_log::captured_logs(&logs);
10980        assert!(
10981            captured.contains("crash budget exhausted: max_restarts=2 within window_secs=600"),
10982            "the stop must be logged with its window: {captured}"
10983        );
10984    }
10985
10986    /// The rate, stated as a test: three crashes where the first has aged past
10987    /// the window are two crashes as far as the budget is concerned, so the
10988    /// third respawn is allowed and the ring holds only the two recent ones.
10989    ///
10990    /// This is the case a lifetime counter got wrong -- and the case the daemon
10991    /// now hits routinely, since a module exits non-zero every time its
10992    /// connection to the daemon drops.
10993    #[tokio::test]
10994    async fn a_crash_older_than_the_window_frees_its_slot_for_a_later_crash() {
10995        let supervisor = Supervisor::new_for_test(
10996            Arc::new(Registry::default()),
10997            RestartPolicy::new(2, Duration::ZERO),
10998        );
10999        let runtime = supervisor.runtime_config();
11000        let spec = windowed_crash_spec("crash-across-windows");
11001        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
11002
11003        for attempt in 1..=2 {
11004            assert!(matches!(
11005                on_child_exit(
11006                    &spec,
11007                    runtime.restart_policy,
11008                    &supervisor.registry,
11009                    &snapshot,
11010                    &runtime.terminal_ring,
11011                    &runtime.spawn_events,
11012                    &runtime.child_roster,
11013                    crash_exit_report(attempt),
11014                )
11015                .await,
11016                NextAction::Restart { schedule: _ }
11017            ));
11018        }
11019
11020        // The oldest crash moves out of the window; nothing else about the
11021        // module changes.
11022        update_snapshot(&snapshot, Some(&spec.module_id), |state| {
11023            age_oldest_crash_restart_out_of_window(state, runtime.restart_policy.window);
11024        })
11025        .unwrap();
11026
11027        assert!(
11028            matches!(
11029                on_child_exit(
11030                    &spec,
11031                    runtime.restart_policy,
11032                    &supervisor.registry,
11033                    &snapshot,
11034                    &runtime.terminal_ring,
11035                    &runtime.spawn_events,
11036                    &runtime.child_roster,
11037                    crash_exit_report(3),
11038                )
11039                .await,
11040                NextAction::Restart { schedule: _ }
11041            ),
11042            "a crash older than the window must not hold a budget slot"
11043        );
11044
11045        let state = lock_snapshot(&snapshot).unwrap();
11046        assert_eq!(state.state, ModuleState::Restarting);
11047        assert_eq!(
11048            state.crash_restarts.len(),
11049            2,
11050            "the aged instant is dropped and the new one takes its place"
11051        );
11052        assert_eq!(
11053            state.lifetime_restarts, 3,
11054            "the ledger counts every restart, including the ones the window forgot"
11055        );
11056    }
11057
11058    /// An operator restart hands the budget back whole, and the ledger keeps
11059    /// counting. Those are different questions -- "how close is this module to
11060    /// being stopped" and "how many times has it been replaced" -- and the
11061    /// operator action answers only the first.
11062    #[tokio::test]
11063    async fn an_operator_restart_clears_the_ring_and_leaves_the_ledger_alone() {
11064        let supervisor = Supervisor::new_for_test(
11065            Arc::new(Registry::default()),
11066            RestartPolicy::new(2, Duration::ZERO),
11067        );
11068        let runtime = supervisor.runtime_config();
11069        let spec = windowed_crash_spec("operator-cleared-budget");
11070        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
11071
11072        for attempt in 1..=2 {
11073            assert!(matches!(
11074                on_child_exit(
11075                    &spec,
11076                    runtime.restart_policy,
11077                    &supervisor.registry,
11078                    &snapshot,
11079                    &runtime.terminal_ring,
11080                    &runtime.spawn_events,
11081                    &runtime.child_roster,
11082                    crash_exit_report(attempt),
11083                )
11084                .await,
11085                NextAction::Restart { schedule: _ }
11086            ));
11087        }
11088
11089        reset_restart_count(&snapshot, &spec.module_id).unwrap();
11090        {
11091            let state = lock_snapshot(&snapshot).unwrap();
11092            assert!(
11093                state.crash_restarts.is_empty(),
11094                "an operator restart returns the full budget"
11095            );
11096            assert_eq!(
11097                state.lifetime_restarts, 2,
11098                "clearing the budget must not unmake the crashes"
11099            );
11100        }
11101
11102        assert!(
11103            matches!(
11104                on_child_exit(
11105                    &spec,
11106                    runtime.restart_policy,
11107                    &supervisor.registry,
11108                    &snapshot,
11109                    &runtime.terminal_ring,
11110                    &runtime.spawn_events,
11111                    &runtime.child_roster,
11112                    crash_exit_report(3),
11113                )
11114                .await,
11115                NextAction::Restart { schedule: _ }
11116            ),
11117            "the cleared budget must be spendable again"
11118        );
11119        let state = lock_snapshot(&snapshot).unwrap();
11120        assert_eq!(state.crash_restarts.len(), 1);
11121        assert_eq!(state.lifetime_restarts, 3);
11122    }
11123
11124    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
11125    async fn severance_marker_for_a_dead_child_does_not_label_its_successor() {
11126        let severed = ProcessIdentity {
11127            pid: 41,
11128            start_time: 101,
11129        };
11130        let successor = ProcessIdentity {
11131            pid: 41,
11132            start_time: 202,
11133        };
11134        let module = module_with_recovery_snapshot(ModuleState::Running, true, 0);
11135        update_snapshot(&module.inner.snapshot, Some("recovery-snapshot"), |state| {
11136            state.pid = Some(successor.pid);
11137            state.process_start_time = Some(successor.start_time);
11138        })
11139        .unwrap();
11140        assert!(!module.record_deliberate_severance(severed).unwrap());
11141
11142        let exit_report = apply_deliberate_severance_marker(
11143            &module.inner.snapshot,
11144            Some(successor),
11145            ExitReport {
11146                kind: ExitKind::Crash,
11147                code: Some(1),
11148                signal: None,
11149                at_ms: 1,
11150            },
11151        );
11152
11153        assert_eq!(exit_report.kind, ExitKind::Crash);
11154    }
11155
11156    #[tokio::test]
11157    async fn drain_reap_marks_deliberate_severance_and_records_lifetime_without_budget() {
11158        let registry = Registry::default();
11159        let supervisor = Supervisor::new_for_test(
11160            Arc::new(Registry::default()),
11161            RestartPolicy::new(3, Duration::ZERO),
11162        );
11163        let runtime = supervisor.runtime_config();
11164        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
11165        let spec = ModuleSpec {
11166            module_id: "drain-deliberate-severance".to_string(),
11167            program: fake_aft_stub_path(),
11168            args: Vec::new(),
11169            env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
11170            reserved: false,
11171            reserved_prefixes: Vec::new(),
11172            protocol: ModuleProtocol::Subc,
11173            overlap: Default::default(),
11174        };
11175        let mut child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
11176        let process = ProcessIdentity {
11177            pid: 41,
11178            start_time: 101,
11179        };
11180        child.process_identity = Some(process);
11181        update_snapshot(&snapshot, Some(&spec.module_id), |state| {
11182            state.pid = Some(process.pid);
11183            state.process_start_time = Some(process.start_time);
11184        })
11185        .unwrap();
11186        record_deliberate_severance(&snapshot, process).unwrap();
11187
11188        drain_child_to_state(
11189            &spec.module_id,
11190            spec.protocol,
11191            // The child exits on its own; no signal may change the exit this
11192            // test classifies.
11193            StopNotice::SentOverConnection,
11194            &registry,
11195            None,
11196            &snapshot,
11197            &runtime.terminal_ring,
11198            &runtime.spawn_events,
11199            child,
11200            Duration::from_secs(1),
11201            ModuleState::Stopped,
11202            Some(false),
11203        )
11204        .await
11205        .unwrap();
11206
11207        let state = lock_snapshot(&snapshot).unwrap();
11208        assert_eq!(
11209            state.last_exit.as_ref().map(|exit| exit.kind),
11210            Some(ExitKind::DeliberateSeverance)
11211        );
11212        assert_eq!(state.lifetime_restarts, 1);
11213        assert_eq!(state.crash_restarts.len(), 0);
11214        drop(state);
11215        let history = runtime.terminal_ring.lock().unwrap().snapshot();
11216        assert_eq!(
11217            history.entries[0].exit_kind,
11218            subc_control::TerminalExitKind::DeliberateSeverance
11219        );
11220    }
11221
11222    #[tokio::test]
11223    async fn ordinary_drain_reap_does_not_record_a_lifetime_restart() {
11224        let registry = Registry::default();
11225        let supervisor = Supervisor::new_for_test(
11226            Arc::new(Registry::default()),
11227            RestartPolicy::new(3, Duration::ZERO),
11228        );
11229        let runtime = supervisor.runtime_config();
11230        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
11231        let spec = ModuleSpec {
11232            module_id: "ordinary-drain".to_string(),
11233            program: fake_aft_stub_path(),
11234            args: Vec::new(),
11235            env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
11236            reserved: false,
11237            reserved_prefixes: Vec::new(),
11238            protocol: ModuleProtocol::Subc,
11239            overlap: Default::default(),
11240        };
11241        let child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
11242
11243        drain_child_to_state(
11244            &spec.module_id,
11245            spec.protocol,
11246            // The child exits on its own; no signal may change the exit this
11247            // test classifies.
11248            StopNotice::SentOverConnection,
11249            &registry,
11250            None,
11251            &snapshot,
11252            &runtime.terminal_ring,
11253            &runtime.spawn_events,
11254            child,
11255            Duration::from_secs(1),
11256            ModuleState::Stopped,
11257            Some(false),
11258        )
11259        .await
11260        .unwrap();
11261
11262        let state = lock_snapshot(&snapshot).unwrap();
11263        assert_eq!(
11264            state.last_exit.as_ref().map(|exit| exit.kind),
11265            Some(ExitKind::Crash)
11266        );
11267        assert_eq!(state.lifetime_restarts, 0);
11268        assert_eq!(state.crash_restarts.len(), 0);
11269    }
11270
11271    #[test]
11272    fn fatal_connection_teardown_cannot_arm_a_marker_for_a_surviving_process() {
11273        // The server's generic fatal-routing branch only knows that the
11274        // connection failed; it does not know that the daemon deliberately
11275        // initiated a process-killing severance. Keep this seam explicit so a
11276        // future connection error path cannot silently reintroduce the stale
11277        // exemption that mislabels a later genuine crash.
11278        assert!(!include_str!("server.rs")
11279            .contains("router.record_deliberate_connection_severance(ctx.connection_id)"));
11280    }
11281
11282    /// The `route.closed` `drained` value must be the quiescence wait's own
11283    /// measurement (`Ok`), never invented -- except on `Err`, where there is no
11284    /// measurement at all and `false` is the one honest constant. This is the exact
11285    /// logic `begin_forwarding_drain_with` now applies before sending `route.closed`
11286    /// on every return path, including the one that used to return early via `?`
11287    /// with `route.closing` already sent and no `route.closed` ever following.
11288    #[test]
11289    fn drained_after_quiescence_wait_passes_ok_through_and_forces_false_on_err() {
11290        assert!(drained_after_quiescence_wait(&Ok(true)));
11291        assert!(!drained_after_quiescence_wait(&Ok(false)));
11292        assert!(!drained_after_quiescence_wait(&Err(
11293            SuperviseError::StatePoisoned { module_id: None }
11294        )));
11295    }
11296
11297    /// `supervise_loop`'s `wait()`-error arm now calls `record_terminal` like every
11298    /// other exit path does, so a module whose child `wait()` itself errored (e.g.
11299    /// already reaped out-of-band) still leaves a terminal record rather than none
11300    /// at all. Triggering the real `wait()` I/O error from an integration test would
11301    /// need a genuine already-reaped-child race, which is OS-specific and not
11302    /// something this suite attempts elsewhere; this test instead verifies the
11303    /// record produced for that arm end-to-end through the real `TerminalRing`, and
11304    /// the call site itself is verified by inspection to sit in that exact arm.
11305    #[test]
11306    fn wait_error_exit_report_records_a_failed_terminal_with_no_code_or_signal() {
11307        let ring = Arc::new(Mutex::new(TerminalRing::new(
11308            TerminalRingConfig::default(),
11309            0,
11310        )));
11311        record_wait_error_terminal("wait-error", &ring, &super::SpawnEventFeed::default());
11312
11313        let snapshot = ring.lock().unwrap().snapshot();
11314        assert_eq!(snapshot.entries.len(), 1);
11315        let entry = &snapshot.entries[0];
11316        assert_eq!(entry.exit_code, None);
11317        assert_eq!(entry.exit_signal, None);
11318        assert_eq!(entry.disposition, TerminalDisposition::Failed);
11319    }
11320
11321    #[test]
11322    fn wait_error_exit_path_preserves_spawn_event_density() {
11323        let feed = super::SpawnEventFeed::default();
11324        feed.configure_incarnation("wait-error-density".to_string());
11325        feed.emit_spawned("wait-error", 41, 1);
11326        let ring = Arc::new(Mutex::new(TerminalRing::new(
11327            TerminalRingConfig::default(),
11328            0,
11329        )));
11330
11331        record_wait_error_terminal("wait-error", &ring, &feed);
11332        feed.emit_spawned("after-wait-error", 42, 2);
11333
11334        let state = feed.0.lock().unwrap();
11335        let sequences = state
11336            .events
11337            .iter()
11338            .map(|event| event.cursor.seq)
11339            .collect::<Vec<_>>();
11340        assert_eq!(sequences, vec![1, 2, 3]);
11341        assert_eq!(state.events[1].kind, SpawnEventKind::Exited);
11342        assert_eq!(state.events[1].exit_code, None);
11343        assert_eq!(state.events[1].exit_signal, None);
11344    }
11345
11346    /// Pins the report's `kind` too: the wait-error arm treats an unwaitable child
11347    /// as a crash (matching `fail_snapshot`'s `Failed` disposition for this arm),
11348    /// not a clean exit it never actually observed.
11349    #[test]
11350    fn wait_error_exit_report_is_classified_as_a_crash() {
11351        assert_eq!(wait_error_exit_report().kind, ExitKind::Crash);
11352    }
11353}
11354
11355#[cfg(all(test, unix))]
11356mod health_event_tests {
11357    use super::*;
11358    use subc_protocol::{manifest::Concurrency, session::ModuleControlResponse};
11359    use subc_test_support::TestTempDir;
11360
11361    struct Actor {
11362        module: SupervisedModule,
11363        registry: Arc<Registry>,
11364        forwarding: Arc<ForwardingTable>,
11365        spec: ModuleSpec,
11366        health: HealthConfig,
11367        rx: mpsc::Receiver<crate::router::OutboundFrame>,
11368        _home: TestTempDir,
11369    }
11370
11371    async fn wait_for<T>(reason: &str, mut observe: impl FnMut() -> Option<T>) -> T {
11372        // OS exec and socket readiness are real I/O. Stay runnable while waiting
11373        // for an observed condition, rather than letting paused time auto-advance
11374        // deadlines or assuming a fixed number of yields finishes the work.
11375        let deadline = std::time::Instant::now() + Duration::from_secs(10);
11376        let tick = Instant::now();
11377        loop {
11378            if let Some(value) = observe() {
11379                assert_eq!(Instant::now(), tick, "{reason} must not wait for a timer");
11380                return value;
11381            }
11382            assert!(
11383                std::time::Instant::now() < deadline,
11384                "timed out waiting for {reason}"
11385            );
11386            tokio::task::yield_now().await;
11387        }
11388    }
11389
11390    impl Actor {
11391        async fn start(protocol: ModuleProtocol, cadence: Duration) -> Self {
11392            let home = TestTempDir::new("health-events");
11393            let registry = Arc::new(Registry::default());
11394            let forwarding = Arc::new(ForwardingTable::default());
11395            let health = HealthConfig {
11396                cadence,
11397                deadline: Duration::from_secs(3600),
11398                ..HealthConfig::default()
11399            };
11400            let supervisor =
11401                Supervisor::new_for_test(registry.clone(), RestartPolicy::new(3, Duration::ZERO))
11402                    .with_forwarding(forwarding.clone())
11403                    .with_health_config(health.clone());
11404            let spec = ModuleSpec {
11405                module_id: "event-health-child".into(),
11406                program: PathBuf::from("/bin/sh"),
11407                // The process stays alive but has no real bus peer. Registrations
11408                // below exercise the same registry writes as accepted HELLOs.
11409                args: vec!["-c".into(), "exec sleep 600".into()],
11410                env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
11411                    .into_iter()
11412                    .map(|key| (key.into(), home.path().display().to_string()))
11413                    .collect(),
11414                reserved: false,
11415                reserved_prefixes: vec![],
11416                protocol,
11417                overlap: Default::default(),
11418            };
11419            let module = supervisor.spawn(spec.clone()).unwrap();
11420            let (_, rx) = mpsc::channel(8);
11421            let actor = Self {
11422                module,
11423                registry,
11424                forwarding,
11425                spec,
11426                health,
11427                rx,
11428                _home: home,
11429            };
11430            actor
11431                .wait_for_select(|checkpoint| checkpoint.generation > 0)
11432                .await;
11433            actor
11434        }
11435
11436        fn turns(&self) -> u64 {
11437            lock_snapshot(&self.module.inner.snapshot)
11438                .unwrap()
11439                .actor_turns
11440        }
11441
11442        async fn wait_for_select(
11443            &self,
11444            expected: impl Fn(&ActorSelectCheckpoint) -> bool,
11445        ) -> ActorSelectCheckpoint {
11446            wait_for("actor select checkpoint", || {
11447                let state = lock_snapshot(&self.module.inner.snapshot).unwrap();
11448                state.actor_select.filter(|checkpoint| {
11449                    checkpoint.generation == state.spawn_generation
11450                        && state.state == ModuleState::Running
11451                        && state.process_alive
11452                        && state.reported_pid().is_some()
11453                        && expected(checkpoint)
11454                })
11455            })
11456            .await
11457        }
11458
11459        fn hello(&mut self, connection: u64, health: bool) {
11460            let manifest =
11461                subc_protocol::manifest::ModuleManifest::builder(&self.spec.module_id, "1.0.0")
11462                    .protocol_ver(subc_protocol::PROTOCOL_VERSION)
11463                    .build();
11464            let connection = ConnectionId::new(connection);
11465            let (tx, rx) = mpsc::channel(8);
11466            self.rx = rx;
11467            self.forwarding
11468                .register_module_connection(
11469                    connection,
11470                    self.spec.module_id.clone(),
11471                    subc_protocol::PROTOCOL_VERSION,
11472                    Concurrency::ModuleManaged,
11473                    FrameSink::new(tx),
11474                )
11475                .unwrap();
11476            self.registry
11477                .register_with_control_ops(
11478                    manifest,
11479                    subc_protocol::PROTOCOL_VERSION,
11480                    connection,
11481                    if health {
11482                        vec![MODULE_CONTROL_OP_HEALTH_CHECK.into()]
11483                    } else {
11484                        vec![]
11485                    },
11486                )
11487                .unwrap();
11488        }
11489
11490        async fn probe(&mut self) -> crate::router::OutboundFrame {
11491            let frame = wait_for("outbound health probe", || match self.rx.try_recv() {
11492                Ok(frame) => Some(frame),
11493                Err(mpsc::error::TryRecvError::Empty) => None,
11494                Err(mpsc::error::TryRecvError::Disconnected) => panic!("health peer disconnected"),
11495            })
11496            .await;
11497            let body: Value = serde_json::from_slice(&frame.body).unwrap();
11498            assert_eq!(body["op"], MODULE_CONTROL_OP_HEALTH_CHECK);
11499            frame
11500        }
11501
11502        fn no_probe(&mut self) {
11503            assert!(matches!(
11504                self.rx.try_recv(),
11505                Err(mpsc::error::TryRecvError::Empty)
11506            ));
11507        }
11508
11509        fn answer(&self, connection: u64, frame: crate::router::OutboundFrame) {
11510            self.forwarding
11511                .complete_module_control_rpc(
11512                    ConnectionId::new(connection),
11513                    frame.header.corr,
11514                    Some(MODULE_CONTROL_OP_HEALTH_CHECK),
11515                    ModuleControlRpcOutcome::Response(ModuleControlResponse::HealthCheck {
11516                        status: HealthStatus::Ok,
11517                        detail: None,
11518                        metrics: None,
11519                    }),
11520                )
11521                .unwrap();
11522        }
11523    }
11524
11525    #[tokio::test(start_paused = true)]
11526    async fn no_health_children_park_for_sixty_seconds() {
11527        let none = Actor::start(ModuleProtocol::None, Duration::from_secs(30)).await;
11528        let unregistered = Actor::start(ModuleProtocol::Subc, Duration::from_secs(30)).await;
11529        let mut unadvertised = Actor::start(ModuleProtocol::Subc, Duration::from_secs(30)).await;
11530        unadvertised.hello(1, false);
11531        // Include any one-off registration wake in the budget; receiving a
11532        // no-health HELLO must not turn a parked child into a polling child.
11533        // Advance in 10 ms steps so the old poll really executes ~6,000 turns;
11534        // a single 60 s jump would only observe one expired timer.
11535        for _ in 0..6000 {
11536            tokio::time::advance(Duration::from_millis(10)).await;
11537            tokio::task::yield_now().await;
11538        }
11539        for actor in [&none, &unregistered, &unadvertised] {
11540            assert!(
11541                actor.turns() <= 5,
11542                "no-health actor woke {} times",
11543                actor.turns()
11544            );
11545            assert_eq!(actor.module.state().unwrap(), ModuleState::Running);
11546        }
11547    }
11548
11549    #[tokio::test(start_paused = true)]
11550    async fn late_hello_starts_a_due_probe_in_the_notification_tick() {
11551        // With a zero cadence the first probe is due at once, so this measures
11552        // only how fast the registration notification starts health probing,
11553        // not the normal 30-33 s wait before a first probe (checked below).
11554        let mut actor = Actor::start(ModuleProtocol::Subc, Duration::ZERO).await;
11555        let tick = Instant::now();
11556        actor.hello(1, true);
11557        actor.probe().await;
11558        assert_eq!(Instant::now(), tick, "HELLO must not wait for a poll");
11559    }
11560
11561    #[tokio::test(start_paused = true)]
11562    async fn hello_keeps_the_jittered_cadence_and_catalog_updates_do_not_reset_it() {
11563        let mut actor = Actor::start(ModuleProtocol::Subc, Duration::from_secs(30)).await;
11564        let hello_tick = Instant::now();
11565        actor.hello(1, true);
11566        let armed = actor
11567            .wait_for_select(|checkpoint| {
11568                checkpoint.registered_connection == Some(ConnectionId::new(1))
11569                    && checkpoint.next_probe_at.is_some()
11570            })
11571            .await;
11572        let delay = jittered_health_delay(&actor.spec.module_id, 0, actor.health.cadence);
11573        assert!((Duration::from_secs(30)..Duration::from_secs(33)).contains(&delay));
11574        let first_deadline = hello_tick + delay;
11575        assert_eq!(armed.next_probe_at, Some(first_deadline));
11576        tokio::time::advance(Duration::from_secs(29)).await;
11577        let turns = actor.turns();
11578        actor
11579            .registry
11580            .replace_catalog_for_connection(ConnectionId::new(1), vec![], None, Some(true))
11581            .unwrap();
11582        let updated = actor
11583            .wait_for_select(|checkpoint| checkpoint.turn > turns)
11584            .await;
11585        assert_eq!(
11586            updated.next_probe_at,
11587            Some(first_deadline),
11588            "catalog update must not reset cadence"
11589        );
11590        actor.no_probe();
11591        tokio::time::advance(first_deadline - Instant::now()).await;
11592        let frame = actor.probe().await;
11593        actor.answer(1, frame);
11594        let next = jittered_health_delay(&actor.spec.module_id, 1, actor.health.cadence);
11595        let next_deadline = Instant::now() + next;
11596        let rearmed = actor
11597            .wait_for_select(|checkpoint| {
11598                checkpoint
11599                    .next_probe_at
11600                    .is_some_and(|deadline| deadline > first_deadline)
11601            })
11602            .await;
11603        assert_eq!(rearmed.next_probe_at, Some(next_deadline));
11604        tokio::time::advance(next - Duration::from_nanos(1)).await;
11605        actor.no_probe();
11606        tokio::time::advance(Duration::from_nanos(1)).await;
11607        actor.probe().await;
11608    }
11609
11610    #[tokio::test(start_paused = true)]
11611    async fn disconnect_disarms_and_reconnect_rearms_health() {
11612        let mut actor = Actor::start(ModuleProtocol::Subc, Duration::from_secs(30)).await;
11613        actor.hello(1, true);
11614        actor
11615            .wait_for_select(|checkpoint| {
11616                checkpoint.registered_connection == Some(ConnectionId::new(1))
11617                    && checkpoint.next_probe_at.is_some()
11618            })
11619            .await;
11620        let turns = actor.turns();
11621        let tick = Instant::now();
11622        actor
11623            .registry
11624            .deregister_connection(ConnectionId::new(1))
11625            .unwrap();
11626        let disarmed = actor
11627            .wait_for_select(|checkpoint| {
11628                checkpoint.turn > turns && checkpoint.registered_connection.is_none()
11629            })
11630            .await;
11631        assert_eq!(disarmed.next_probe_at, None);
11632        assert_eq!(disarmed.wake_after, None);
11633        assert_eq!(
11634            Instant::now(),
11635            tick,
11636            "disconnect must disarm in the notification tick"
11637        );
11638        tokio::time::advance(Duration::from_secs(60)).await;
11639        actor.no_probe();
11640        let reconnect_tick = Instant::now();
11641        actor.hello(2, true);
11642        let rearmed = actor
11643            .wait_for_select(|checkpoint| {
11644                checkpoint.registered_connection == Some(ConnectionId::new(2))
11645                    && checkpoint.next_probe_at.is_some()
11646            })
11647            .await;
11648        let delay = jittered_health_delay(&actor.spec.module_id, 0, actor.health.cadence);
11649        assert_eq!(rearmed.next_probe_at, Some(reconnect_tick + delay));
11650        tokio::time::advance(delay).await;
11651        actor.probe().await;
11652    }
11653
11654    #[tokio::test(start_paused = true)]
11655    async fn reregistration_without_health_disarms_the_old_schedule() {
11656        let mut actor = Actor::start(ModuleProtocol::Subc, Duration::from_secs(30)).await;
11657        actor.hello(1, true);
11658        actor
11659            .wait_for_select(|checkpoint| {
11660                checkpoint.registered_connection == Some(ConnectionId::new(1))
11661                    && checkpoint.next_probe_at.is_some()
11662            })
11663            .await;
11664        actor
11665            .registry
11666            .deregister_connection(ConnectionId::new(1))
11667            .unwrap();
11668        actor.hello(2, false);
11669        let disarmed = actor
11670            .wait_for_select(|checkpoint| {
11671                checkpoint.registered_connection == Some(ConnectionId::new(2))
11672            })
11673            .await;
11674        assert_eq!(disarmed.next_probe_at, None);
11675        assert_eq!(disarmed.wake_after, None);
11676        tokio::time::advance(Duration::from_secs(60)).await;
11677        actor.no_probe();
11678    }
11679
11680    #[tokio::test(start_paused = true)]
11681    async fn restart_waits_for_its_own_hello_before_rearming_health() {
11682        let mut actor = Actor::start(ModuleProtocol::Subc, Duration::from_secs(30)).await;
11683        actor.hello(1, true);
11684        let armed = actor
11685            .wait_for_select(|checkpoint| {
11686                checkpoint.registered_connection == Some(ConnectionId::new(1))
11687                    && checkpoint.next_probe_at.is_some()
11688            })
11689            .await;
11690        // Simulate the old process's connection closing before the restart
11691        // tears the process down. The replacement process has started but has
11692        // not sent its HELLO, so it is not registered.
11693        actor
11694            .registry
11695            .deregister_connection(ConnectionId::new(1))
11696            .unwrap();
11697        actor.module.restart(Some(0)).await.unwrap();
11698        let parked = actor
11699            .wait_for_select(|checkpoint| {
11700                checkpoint.generation > armed.generation
11701                    && checkpoint.registered_connection.is_none()
11702                    && checkpoint.next_probe_at.is_none()
11703            })
11704            .await;
11705        assert_eq!(parked.wake_after, None);
11706        actor.rx.close();
11707        // Drain the stop notice the restart queued for the old process's connection,
11708        // so it isn't mistaken for traffic to the replacement.
11709        while actor.rx.try_recv().is_ok() {}
11710        let turns = actor.turns();
11711        tokio::time::advance(Duration::from_secs(60)).await;
11712        let after = actor.turns();
11713        eprintln!(
11714            "restart park turns: before={turns} after={after} delta={}",
11715            after - turns
11716        );
11717        assert_eq!(after, turns, "replacement must park before HELLO");
11718        let hello_tick = Instant::now();
11719        actor.hello(2, true);
11720        let rearmed = actor
11721            .wait_for_select(|checkpoint| {
11722                checkpoint.registered_connection == Some(ConnectionId::new(2))
11723                    && checkpoint.next_probe_at.is_some()
11724            })
11725            .await;
11726        let delay = jittered_health_delay(&actor.spec.module_id, 0, actor.health.cadence);
11727        assert_eq!(rearmed.next_probe_at, Some(hello_tick + delay));
11728        tokio::time::advance(delay).await;
11729        actor.probe().await;
11730    }
11731
11732    #[tokio::test(start_paused = true)]
11733    async fn rescan_adds_and_removes_http_health_without_polling() {
11734        use tokio::{
11735            io::{AsyncReadExt, AsyncWriteExt},
11736            net::TcpListener,
11737        };
11738        let mut actor = Actor::start(ModuleProtocol::None, Duration::from_secs(30)).await;
11739        let listener = TcpListener::bind("127.0.0.1:0").await.unwrap();
11740        actor.health.http = Some(format!("http://{}/health", listener.local_addr().unwrap()));
11741        // One millisecond separates successive probes without a polling delay.
11742        actor.health.cadence = Duration::from_millis(1);
11743        let (started_tx, mut started_rx) = mpsc::channel(8);
11744        let peer = tokio::spawn(async move {
11745            loop {
11746                let (mut socket, _) = listener.accept().await.unwrap();
11747                let mut request = [0; 1024];
11748                assert!(socket.read(&mut request).await.unwrap() > 0);
11749                started_tx.send(()).await.unwrap();
11750                socket
11751                    .write_all(b"HTTP/1.1 200 OK\r\nContent-Length: 0\r\n\r\n")
11752                    .await
11753                    .unwrap();
11754            }
11755        });
11756        let tick = Instant::now();
11757        let turns = actor.turns();
11758        actor
11759            .module
11760            .update_configuration(actor.spec.clone(), actor.health.clone(), None)
11761            .await
11762            .unwrap();
11763        let armed = actor
11764            .wait_for_select(|checkpoint| {
11765                checkpoint.turn > turns && checkpoint.next_probe_at.is_some()
11766            })
11767            .await;
11768        assert_eq!(armed.next_probe_at, Some(tick + Duration::from_millis(1)));
11769        assert_eq!(Instant::now(), tick);
11770        tokio::time::advance(Duration::from_millis(1)).await;
11771        let mut started = false;
11772        wait_for("successful HTTP probe", || {
11773            started |= started_rx.try_recv().is_ok();
11774            (started && actor.module.status().unwrap().health.status == SupervisorHealthStatus::Ok)
11775                .then_some(())
11776        })
11777        .await;
11778        assert!(started, "added HTTP check must start at its first deadline");
11779        assert_eq!(
11780            actor.module.status().unwrap().health.status,
11781            SupervisorHealthStatus::Ok
11782        );
11783        assert_eq!(Instant::now(), tick + Duration::from_millis(1));
11784        actor.health.http = None;
11785        let turns = actor.turns();
11786        actor
11787            .module
11788            .update_configuration(actor.spec.clone(), actor.health.clone(), None)
11789            .await
11790            .unwrap();
11791        let parked = actor
11792            .wait_for_select(|checkpoint| {
11793                checkpoint.turn > turns && checkpoint.next_probe_at.is_none()
11794            })
11795            .await;
11796        assert_eq!(parked.wake_after, None);
11797        let turns = actor.turns();
11798        tokio::time::advance(Duration::from_secs(60)).await;
11799        assert_eq!(
11800            actor.turns(),
11801            turns,
11802            "removed HTTP check must park the actor"
11803        );
11804        assert!(
11805            started_rx.try_recv().is_err(),
11806            "removed HTTP check must stay disarmed"
11807        );
11808        assert_eq!(
11809            actor.module.status().unwrap().health,
11810            ModuleHealthStatus::default()
11811        );
11812        peer.abort();
11813    }
11814}
11815
11816#[cfg(test)]
11817mod health_evidence_tests {
11818    use super::{HealthProbeError, HealthProbeEvidence};
11819    use std::collections::HashSet;
11820
11821    /// The evidential asymmetry, asserted rather than described.
11822    ///
11823    /// Exactly ONE observation is proof a module cannot serve, and the one that
11824    /// fires under CPU starvation is not it. Before the split, all fifteen
11825    /// construction sites collapsed into a single String, so a timeout carried the
11826    /// same weight as a dead lane -- which is how a healthy module was restarted
11827    /// three times in one day.
11828    #[test]
11829    fn only_a_dead_lane_is_proof_of_death() {
11830        assert!(HealthProbeError::lane_dead("gone").is_proof_of_death());
11831        // Three non-proof classes, each for a different reason: silence is
11832        // consistent with health, a bad answer proves the module ALIVE, and a
11833        // daemon-side fault never reached the module at all.
11834        assert!(!HealthProbeError::no_answer("timed out").is_proof_of_death());
11835        assert!(!HealthProbeError::bad_answer("garbage").is_proof_of_death());
11836        assert!(!HealthProbeError::misconfigured("no table").is_proof_of_death());
11837    }
11838
11839    /// Labels must be distinct, or the operator-facing distinction is cosmetic.
11840    ///
11841    /// A shared label renders two different observations identically in the line an
11842    /// operator reads after an unexplained restart -- the exact confusion this
11843    /// change removes.
11844    #[test]
11845    fn every_evidence_class_has_a_distinct_label() {
11846        let labels = [
11847            HealthProbeError::lane_dead("").label(),
11848            HealthProbeError::no_answer("").label(),
11849            HealthProbeError::bad_answer("").label(),
11850            HealthProbeError::misconfigured("").label(),
11851        ];
11852        let unique: HashSet<_> = labels.iter().collect();
11853        assert_eq!(unique.len(), labels.len(), "labels collided: {labels:?}");
11854    }
11855
11856    /// The class is additional information, not a replacement.
11857    ///
11858    /// An operator needs both "this was silence" and the specific text saying how
11859    /// long we waited; a classification that swallowed the message would trade one
11860    /// missing distinction for another.
11861    #[test]
11862    fn classification_preserves_the_original_message() {
11863        let err = HealthProbeError::no_answer("module did not answer within 5s");
11864        assert_eq!(err.to_string(), "module did not answer within 5s");
11865        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
11866    }
11867}
11868
11869#[cfg(test)]
11870mod health_tombstone_tests {
11871    use std::{path::PathBuf, sync::Arc, time::Duration};
11872
11873    use subc_protocol::{
11874        manifest::Concurrency,
11875        session::{HealthStatus, ModuleControlResponse},
11876    };
11877    use tokio::sync::mpsc;
11878
11879    use super::{
11880        probe_module_health, HealthAction, HealthConfig, HealthProbeEvidence, ModuleProtocol,
11881        ModuleSpec, RestartPolicy, Supervisor, SupervisorRuntimeConfig,
11882    };
11883    use crate::{
11884        control::ControlHandler,
11885        forwarding::{ForwardingTable, ModuleControlRpcCompletion, ModuleControlRpcOutcome},
11886        registry::{ConnectionId, Registry},
11887        router::FrameSink,
11888    };
11889
11890    struct ProbeHarness {
11891        spec: ModuleSpec,
11892        runtime: SupervisorRuntimeConfig,
11893        forwarding: Arc<ForwardingTable>,
11894        module_connection: ConnectionId,
11895        module_rx: mpsc::Receiver<crate::router::OutboundFrame>,
11896        handler: ControlHandler,
11897        module: super::SupervisedModule,
11898    }
11899
11900    fn probe_harness() -> ProbeHarness {
11901        let registry = Arc::new(Registry::default());
11902        let forwarding = Arc::new(ForwardingTable::default());
11903        let supervisor_handle = super::SupervisorHandle::new();
11904        let health = HealthConfig {
11905            http: None,
11906            cadence: Duration::from_secs(30),
11907            deadline: Duration::from_secs(5),
11908            failure_threshold: 3,
11909            on_degraded: HealthAction::Report,
11910            on_failing: HealthAction::Report,
11911            critical: false,
11912        };
11913        let supervisor = Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::default())
11914            .with_forwarding(Arc::clone(&forwarding))
11915            .with_handle(supervisor_handle.clone())
11916            .with_health_config(health);
11917        let spec = ModuleSpec {
11918            module_id: "late-health-module".to_string(),
11919            program: PathBuf::from("disabled-module"),
11920            args: Vec::new(),
11921            env: Vec::new(),
11922            reserved: false,
11923            reserved_prefixes: Vec::new(),
11924            protocol: ModuleProtocol::Subc,
11925            overlap: Default::default(),
11926        };
11927        let module = supervisor
11928            .supervise_configured(spec.clone(), false)
11929            .unwrap();
11930        let runtime = supervisor.runtime_config();
11931        let handler = ControlHandler::with_forwarding(registry, Arc::clone(&forwarding))
11932            .with_supervisor(supervisor_handle);
11933        let module_connection = ConnectionId::new(700);
11934        let (module_tx, module_rx) = mpsc::channel(8);
11935        forwarding
11936            .register_module_connection(
11937                module_connection,
11938                spec.module_id.clone(),
11939                subc_protocol::PROTOCOL_VERSION,
11940                Concurrency::ModuleManaged,
11941                FrameSink::new(module_tx),
11942            )
11943            .unwrap();
11944
11945        ProbeHarness {
11946            spec,
11947            runtime,
11948            forwarding,
11949            module_connection,
11950            module_rx,
11951            handler,
11952            module,
11953        }
11954    }
11955
11956    async fn finish_after(
11957        harness: &mut ProbeHarness,
11958        stall: Duration,
11959    ) -> ModuleControlRpcCompletion {
11960        assert!(stall > harness.runtime.health.deadline);
11961        let deadline = harness.runtime.health.deadline;
11962        let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
11963        let answer = async {
11964            let frame = harness.module_rx.recv().await.expect("health.check frame");
11965            tokio::time::advance(deadline).await;
11966            tokio::task::yield_now().await;
11967            tokio::time::advance(stall - deadline).await;
11968            harness
11969                .forwarding
11970                .complete_module_control_rpc(
11971                    harness.module_connection,
11972                    frame.header.corr,
11973                    Some("health.check"),
11974                    ModuleControlRpcOutcome::Response(ModuleControlResponse::HealthCheck {
11975                        status: HealthStatus::Ok,
11976                        detail: None,
11977                        metrics: None,
11978                    }),
11979                )
11980                .unwrap()
11981        };
11982        let (probe_result, completion) = tokio::join!(probe, answer);
11983        let err = probe_result.expect_err("probe must miss its deadline");
11984        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
11985        completion
11986    }
11987
11988    async fn time_out_without_answer(harness: &mut ProbeHarness) {
11989        let deadline = harness.runtime.health.deadline;
11990        let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
11991        let exhaust_deadline = async {
11992            let _frame = harness.module_rx.recv().await.expect("health.check frame");
11993            tokio::time::advance(deadline).await;
11994            tokio::task::yield_now().await;
11995        };
11996        let (probe_result, ()) = tokio::join!(probe, exhaust_deadline);
11997        let err = probe_result.expect_err("probe must miss its deadline");
11998        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
11999    }
12000
12001    async fn run_probe_cycle(harness: &mut ProbeHarness, answer: bool) {
12002        let registry = Arc::clone(&harness.module.inner.registry);
12003        let snapshot = Arc::clone(&harness.module.inner.snapshot);
12004        let process_liveness = super::SupervisorProcessLiveness::default();
12005        let mut child = None;
12006        let cycle = super::run_health_probe_cycle(
12007            &harness.spec,
12008            &harness.runtime,
12009            &registry,
12010            &process_liveness,
12011            &snapshot,
12012            &mut child,
12013        );
12014        let peer = async {
12015            let frame = harness.module_rx.recv().await.expect("health.check frame");
12016            if answer {
12017                harness
12018                    .forwarding
12019                    .complete_module_control_rpc(
12020                        harness.module_connection,
12021                        frame.header.corr,
12022                        Some("health.check"),
12023                        ModuleControlRpcOutcome::Response(ModuleControlResponse::HealthCheck {
12024                            status: HealthStatus::Ok,
12025                            detail: None,
12026                            metrics: Some(serde_json::json!({"ready": true})),
12027                        }),
12028                    )
12029                    .unwrap();
12030            } else {
12031                tokio::time::advance(harness.runtime.health.deadline).await;
12032                tokio::task::yield_now().await;
12033            }
12034        };
12035        tokio::join!(cycle, peer);
12036    }
12037
12038    #[tokio::test(start_paused = true)]
12039    async fn unanswered_probe_is_unknown_until_threshold_and_ok_report_recovers() {
12040        let mut harness = probe_harness();
12041        // Drive the probe cycle directly with an in-memory wire peer. Stop the
12042        // disabled module's monitor so only this test owns lifecycle transitions;
12043        // no OS process is launched, and a restart is observed at scheduling.
12044        let monitor = harness.module.inner.monitor.lock().unwrap().take().unwrap();
12045        monitor.abort();
12046        let _ = monitor.await;
12047        super::update_snapshot(&harness.module.inner.snapshot, None, |state| {
12048            state.enabled = true;
12049            state.state = super::ModuleState::Running;
12050            state.process_alive = true;
12051        })
12052        .unwrap();
12053
12054        run_probe_cycle(&mut harness, true).await;
12055        assert_eq!(
12056            harness.module.status().unwrap().health.status,
12057            super::SupervisorHealthStatus::Ok
12058        );
12059
12060        for failures in 1..harness.runtime.health.failure_threshold {
12061            run_probe_cycle(&mut harness, false).await;
12062            let status = harness.module.status().unwrap();
12063            assert_eq!(status.health.status, super::SupervisorHealthStatus::Unknown);
12064            assert_eq!(status.health.consecutive_failures, failures);
12065            assert!(status.health.last_probe_ms.is_some());
12066            assert!(status.health.detail.unwrap().starts_with("[no-answer]"));
12067            assert!(status.health.metrics.is_none());
12068            assert_eq!(status.state, super::ModuleState::Running);
12069            assert!(status.process_alive);
12070            assert_eq!(status.restart_count, 0);
12071            assert_eq!(status.lifetime_restarts, 0);
12072            assert!(status.health.last_action.is_none());
12073            assert!(harness.runtime.scheduled_respawn.lock().unwrap().is_none());
12074        }
12075
12076        run_probe_cycle(&mut harness, true).await;
12077        let recovered = harness.module.status().unwrap();
12078        assert_eq!(recovered.health.status, super::SupervisorHealthStatus::Ok);
12079        assert_eq!(recovered.health.consecutive_failures, 0);
12080        assert!(recovered.health.detail.is_none());
12081        assert_eq!(
12082            recovered.health.metrics,
12083            Some(serde_json::json!({"ready": true}))
12084        );
12085        assert_eq!(recovered.lifetime_restarts, 0);
12086
12087        for failures in 1..=harness.runtime.health.failure_threshold {
12088            run_probe_cycle(&mut harness, false).await;
12089            let status = harness.module.status().unwrap();
12090            assert_eq!(status.health.consecutive_failures, failures);
12091            if failures < harness.runtime.health.failure_threshold {
12092                assert_eq!(status.health.status, super::SupervisorHealthStatus::Unknown);
12093                assert_eq!(status.state, super::ModuleState::Running);
12094                assert_eq!(status.lifetime_restarts, 0);
12095                assert!(harness.runtime.scheduled_respawn.lock().unwrap().is_none());
12096            } else {
12097                assert_eq!(
12098                    status.health.status,
12099                    super::SupervisorHealthStatus::Unresponsive
12100                );
12101                assert_eq!(status.state, super::ModuleState::Restarting);
12102                assert_eq!(status.restart_count, 1);
12103                assert_eq!(status.lifetime_restarts, 1);
12104                assert_eq!(status.health.last_action.as_deref(), Some("restart"));
12105                assert!(harness.runtime.scheduled_respawn.lock().unwrap().is_some());
12106            }
12107        }
12108    }
12109
12110    #[tokio::test(start_paused = true)]
12111    async fn late_health_answers_record_start_anchored_latency_for_two_stalls() {
12112        let mut harness = probe_harness();
12113
12114        let first = finish_after(&mut harness, Duration::from_secs(8)).await;
12115        let first_latency = match &first {
12116            ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
12117            other => panic!("late answer was not retained: {other:?}"),
12118        };
12119        assert!(harness.handler.observe_module_control_completion(first));
12120
12121        let second = finish_after(&mut harness, Duration::from_secs(11)).await;
12122        let second_latency = match &second {
12123            ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
12124            other => panic!("late answer was not retained: {other:?}"),
12125        };
12126        assert!(harness.handler.observe_module_control_completion(second));
12127
12128        assert_eq!(first_latency, Duration::from_secs(8));
12129        assert_eq!(
12130            second_latency - first_latency,
12131            Duration::from_secs(3),
12132            "latency must grow linearly with the additional stall"
12133        );
12134        let health = harness.module.status().unwrap().health;
12135        assert_eq!(health.late_answer_count, 2);
12136        assert_eq!(health.last_late_answer_latency_ms, Some(11_000));
12137    }
12138
12139    /// A module that answers every probe late must never march to the kill
12140    /// threshold: the late answer proves it is alive, so it must clear the miss
12141    /// streak the timeout recorded. Without the reset, a CPU-starved module
12142    /// that serves every probe seconds past the deadline accumulates
12143    /// `consecutive_failures` to the threshold and is killed — the exact
12144    /// sequence from the 2026-08-14 aft disable, where the daemon logged
12145    /// "proves the module is alive" five times while counting five misses.
12146    #[tokio::test(start_paused = true)]
12147    async fn late_answer_clears_the_consecutive_failure_streak() {
12148        let mut harness = probe_harness();
12149
12150        // Timeout recorded first: the probe path saw no answer in time.
12151        time_out_without_answer(&mut harness).await;
12152        harness
12153            .module
12154            .record_health_probe_failure_for_test("[no-answer] test miss")
12155            .unwrap();
12156        assert_eq!(
12157            harness.module.status().unwrap().health.consecutive_failures,
12158            1,
12159            "precondition: the miss must be on the streak before the late answer"
12160        );
12161
12162        // The stalled reply then lands: proof of life.
12163        let late = finish_after(&mut harness, Duration::from_secs(9)).await;
12164        assert!(matches!(
12165            late,
12166            ModuleControlRpcCompletion::LateHealthAnswer { .. }
12167        ));
12168        assert!(harness.handler.observe_module_control_completion(late));
12169
12170        let health = harness.module.status().unwrap().health;
12171        assert_eq!(
12172            health.consecutive_failures, 0,
12173            "a late answer is an answer: the streak must reset"
12174        );
12175        assert_eq!(health.late_answer_count, 1);
12176    }
12177
12178    #[tokio::test(start_paused = true)]
12179    async fn repeated_serial_probe_cycles_keep_one_tombstone_per_endpoint() {
12180        let mut harness = probe_harness();
12181
12182        for _ in 0..20 {
12183            time_out_without_answer(&mut harness).await;
12184            assert_eq!(
12185                harness.forwarding.health_probe_tombstone_count().unwrap(),
12186                1
12187            );
12188        }
12189    }
12190}
12191
12192#[cfg(test)]
12193mod child_env_tests {
12194    use super::{
12195        apply_child_env, apply_spawn_role, apply_wire_spawn_args, ModuleProtocol, ModuleSpec,
12196        SpawnRole, SupervisorHandle, SPAWN_ROLE_SWAP_CANDIDATE, SUBC_ARG, SUBC_LAUNCH_NONCE_ENV,
12197        SUBC_MODULE_ID_ENV, SUBC_SPAWN_ROLE_ENV,
12198    };
12199    use std::{ffi::OsStr, path::PathBuf};
12200    use tokio::process::Command;
12201
12202    fn spec(env: Vec<(String, String)>) -> ModuleSpec {
12203        ModuleSpec {
12204            module_id: "env-plan".to_string(),
12205            program: PathBuf::from("/nonexistent"),
12206            args: Vec::new(),
12207            env,
12208            reserved: false,
12209            reserved_prefixes: Vec::new(),
12210            protocol: ModuleProtocol::Subc,
12211            overlap: Default::default(),
12212        }
12213    }
12214
12215    /// Ambient `CK_LOG` is REMOVED for an unconfigured module, and a configured
12216    /// one still gets its own.
12217    ///
12218    /// This is the narrow goal `env_clear()` was reached for, and the reason the
12219    /// fix is `env_remove` rather than deleting the line: an operator's ambient
12220    /// filter silently becoming an unconfigured module's log level is a real
12221    /// defect, just a much smaller one than clearing the environment.
12222    ///
12223    /// Asserted on the command plan rather than a spawned child because proving
12224    /// the ABSENCE of an inherited variable needs the parent's environment
12225    /// mutated, and `forbid(unsafe_code)` refuses that. `get_envs()` reports a
12226    /// removal as `(key, None)`, which is exactly the distinction wanted: not
12227    /// "absent because nobody set it" but "explicitly unset for the child".
12228    #[test]
12229    fn ambient_ck_log_is_removed_and_a_configured_one_survives() {
12230        let mut command = Command::new("/nonexistent");
12231        apply_child_env(&mut command, &spec(Vec::new()));
12232        let removed = command
12233            .as_std()
12234            .get_envs()
12235            .any(|(key, value)| key == OsStr::new("CK_LOG") && value.is_none());
12236        assert!(
12237            removed,
12238            "ambient CK_LOG must be explicitly removed for an unconfigured module"
12239        );
12240
12241        let mut configured = Command::new("/nonexistent");
12242        apply_child_env(
12243            &mut configured,
12244            &spec(vec![("CK_LOG".to_string(), "debug".to_string())]),
12245        );
12246        let effective = configured
12247            .as_std()
12248            .get_envs()
12249            .filter(|(key, _)| *key == OsStr::new("CK_LOG"))
12250            .last()
12251            .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
12252        assert_eq!(
12253            effective,
12254            Some(Some("debug".to_string())),
12255            "a module's configured CK_LOG must survive the ambient removal"
12256        );
12257    }
12258
12259    /// A `protocol: "none"` spawn carries NO `--subc` argument and NO launch
12260    /// nonce; a subc-wire spawn carries both. Asserted on the command plan for
12261    /// the same reason as the CK_LOG test above.
12262    ///
12263    /// The argument is the load-bearing half: a stock binary exits on an
12264    /// unknown flag before it listens, so with `--subc` appended the mode
12265    /// could not supervise the one process it exists for. Found by the first
12266    /// conformance run (nats-server: `flag provided but not defined: -subc`).
12267    #[test]
12268    fn protocol_none_spawn_carries_no_subc_argument_and_no_nonce() {
12269        let connection_file = std::path::Path::new("/run/subc-connection.json");
12270        let handle = SupervisorHandle::new();
12271
12272        let mut none_spec = spec(Vec::new());
12273        none_spec.protocol = ModuleProtocol::None;
12274        let mut none = Command::new("/nonexistent");
12275        let none_handoff =
12276            apply_wire_spawn_args(&mut none, &none_spec, Some(connection_file), Some(&handle))
12277                .expect("protocol-none spawn args apply");
12278        assert!(
12279            none_handoff.is_none(),
12280            "protocol:none spawn must not receive a nonce descriptor"
12281        );
12282        assert!(
12283            !none.as_std().get_envs().any(|(key, value)| key
12284                == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV)
12285                && value.is_some()),
12286            "protocol:none spawn must not name a nonce descriptor"
12287        );
12288        let none_args: Vec<String> = none
12289            .as_std()
12290            .get_args()
12291            .map(|a| a.to_string_lossy().into_owned())
12292            .collect();
12293        assert!(
12294            !none_args.iter().any(|a| a == SUBC_ARG),
12295            "protocol:none argv must not carry --subc; got {none_args:?}"
12296        );
12297        let none_has_nonce = none
12298            .as_std()
12299            .get_envs()
12300            .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some());
12301        assert!(
12302            !none_has_nonce,
12303            "protocol:none spawn must not receive a launch nonce"
12304        );
12305        let none_has_module_id = none
12306            .as_std()
12307            .get_envs()
12308            .any(|(key, value)| key == OsStr::new(SUBC_MODULE_ID_ENV) && value.is_some());
12309        assert!(
12310            none_has_module_id,
12311            "SUBC_MODULE_ID is inert and stays on every path"
12312        );
12313        assert!(
12314            handle.spawn_nonce(&none_spec.module_id).is_none(),
12315            "no nonce record for a process that will never present one"
12316        );
12317
12318        // Control: the subc-wire path is unchanged by the branch above.
12319        let wire_spec = spec(Vec::new());
12320        let mut wire = Command::new("/nonexistent");
12321        let wire_handoff =
12322            apply_wire_spawn_args(&mut wire, &wire_spec, Some(connection_file), Some(&handle))
12323                .expect("subc-wire spawn args apply");
12324        let wire_fd_env = wire
12325            .as_std()
12326            .get_envs()
12327            .find(|(key, _)| *key == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV))
12328            .and_then(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
12329        #[cfg(unix)]
12330        assert_eq!(
12331            wire_fd_env,
12332            Some(wire_handoff.expect("a descriptor handoff").fd_env_value()),
12333            "a subc-wire spawn names the pipe it will receive at descriptor 3"
12334        );
12335        #[cfg(not(unix))]
12336        assert!(wire_handoff.is_none() && wire_fd_env.is_none());
12337        let wire_args: Vec<String> = wire
12338            .as_std()
12339            .get_args()
12340            .map(|a| a.to_string_lossy().into_owned())
12341            .collect();
12342        assert_eq!(
12343            wire_args,
12344            vec![
12345                SUBC_ARG.to_string(),
12346                connection_file.to_string_lossy().into_owned()
12347            ],
12348            "a subc-wire spawn still carries --subc <path>"
12349        );
12350        assert_eq!(
12351            wire.as_std()
12352                .get_envs()
12353                .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some()),
12354            !cfg!(unix),
12355            "only Windows supplies the environment nonce"
12356        );
12357        assert!(handle.spawn_nonce(&wire_spec.module_id).is_some());
12358    }
12359
12360    /// A plain spawn EXPLICITLY REMOVES the spawn role, even when the module's
12361    /// spec tries to set it; only a swap candidate carries it.
12362    ///
12363    /// "Set it only on candidates" is not enough, because spawn applies the
12364    /// spec's env verbatim and the daemon's own environment is inherited: either
12365    /// could hand a plain restart the swap role, and a module reading it would
12366    /// warm on its long swap budget while callers wait. Asserted as an explicit
12367    /// removal (`(key, None)`), not mere absence, for the reason the `CK_LOG`
12368    /// test above gives.
12369    #[test]
12370    fn plain_spawn_removes_the_spawn_role_even_when_the_spec_sets_it() {
12371        let role = |command: &Command| {
12372            command
12373                .as_std()
12374                .get_envs()
12375                .filter(|(key, _)| *key == OsStr::new(SUBC_SPAWN_ROLE_ENV))
12376                .last()
12377                .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()))
12378        };
12379        let forged = spec(vec![(
12380            SUBC_SPAWN_ROLE_ENV.to_string(),
12381            SPAWN_ROLE_SWAP_CANDIDATE.to_string(),
12382        )]);
12383
12384        let mut plain = Command::new("/nonexistent");
12385        apply_child_env(&mut plain, &forged);
12386        apply_spawn_role(&mut plain, SpawnRole::Plain);
12387        assert_eq!(
12388            role(&plain),
12389            Some(None),
12390            "a plain spawn must remove SUBC_SPAWN_ROLE, whatever the spec says"
12391        );
12392
12393        let mut candidate = Command::new("/nonexistent");
12394        apply_child_env(&mut candidate, &spec(Vec::new()));
12395        apply_spawn_role(&mut candidate, SpawnRole::SwapCandidate);
12396        assert_eq!(
12397            role(&candidate),
12398            Some(Some(SPAWN_ROLE_SWAP_CANDIDATE.to_string()))
12399        );
12400    }
12401
12402    /// Daemon-private capture retention keys never reach the child.
12403    ///
12404    /// cortexkit-log exposes retention as a Rust struct with no environment
12405    /// names, so these entries are supervisor metadata. Passing them through
12406    /// would invent a public child-process contract by accident.
12407    #[test]
12408    fn daemon_private_capture_keys_are_not_passed_to_the_child() {
12409        let mut command = Command::new("/nonexistent");
12410        apply_child_env(
12411            &mut command,
12412            &spec(vec![
12413                (super::CAPTURE_KEEP_ENV.to_string(), "5".to_string()),
12414                ("KEPT".to_string(), "yes".to_string()),
12415            ]),
12416        );
12417        let keys: Vec<String> = command
12418            .as_std()
12419            .get_envs()
12420            .filter(|(_, value)| value.is_some())
12421            .map(|(key, _)| key.to_string_lossy().into_owned())
12422            .collect();
12423        assert!(keys.contains(&"KEPT".to_string()), "got {keys:?}");
12424        assert!(
12425            !keys.contains(&super::CAPTURE_KEEP_ENV.to_string()),
12426            "daemon-private capture key leaked to the child: {keys:?}"
12427        );
12428    }
12429}
12430
12431#[cfg(test)]
12432mod jitter_tests {
12433    use super::jittered_health_delay;
12434    use std::{collections::HashSet, time::Duration};
12435
12436    /// Module ids drawn from a real fleet, so the dispersal claim is about names
12437    /// that actually occur rather than invented ones.
12438    ///
12439    /// This is a SAMPLE, not a registry: the property under test is that distinct
12440    /// ids disperse, which holds for any set of distinct strings. Several entries
12441    /// are already historical (modules get renamed), and that costs nothing here --
12442    /// but it means a reader must not mistake this for the live module set, and a
12443    /// rename sweep will match it without there being anything to change.
12444    const FLEET: [&str; 14] = [
12445        "aft",
12446        "alfonso-core",
12447        "magic-context",
12448        "broca",
12449        "thalamus",
12450        "quota",
12451        "engram",
12452        "plexus",
12453        "cerebellum",
12454        "astrocyte",
12455        "synapse",
12456        "subc-mcp",
12457        "cortexkit-credentials",
12458        "subc-federation",
12459    ];
12460
12461    /// Probes must not converge after a fleet-wide restart.
12462    ///
12463    /// This is the property the jitter exists for: every module reconnects at
12464    /// once, and without dispersal all fourteen would then probe on the same
12465    /// tick forever. Nothing failed visibly when this went untested -- a
12466    /// convergent fleet still probes correctly, just in a burst, so the symptom
12467    /// is a periodic load spike that looks like whatever else is running.
12468    #[test]
12469    fn probe_delays_disperse_across_the_fleet() {
12470        let cadence = Duration::from_secs(30);
12471        let delays: HashSet<Duration> = FLEET
12472            .iter()
12473            .map(|id| jittered_health_delay(id, 0, cadence))
12474            .collect();
12475        assert_eq!(
12476            delays.len(),
12477            FLEET.len(),
12478            "every supervised module must land on its own probe offset"
12479        );
12480    }
12481
12482    /// The offset may only ever DELAY a probe, never bring it forward.
12483    ///
12484    /// A delay below the cadence would probe a module more often than
12485    /// configured, which is the opposite of what an operator asked for and
12486    /// would tighten the failure budget without anyone changing it.
12487    #[test]
12488    fn jitter_only_delays_and_stays_within_one_tenth_of_cadence() {
12489        let cadence = Duration::from_secs(30);
12490        let span = cadence / 10;
12491        for id in FLEET {
12492            for probe_index in 0..8 {
12493                let delay = jittered_health_delay(id, probe_index, cadence);
12494                assert!(
12495                    delay >= cadence,
12496                    "{id}#{probe_index}: jitter must not shorten the cadence"
12497                );
12498                assert!(
12499                    delay < cadence + span,
12500                    "{id}#{probe_index}: jitter must stay inside one tenth of the cadence"
12501                );
12502            }
12503        }
12504    }
12505
12506    /// A module keeps its offset across daemon restarts.
12507    ///
12508    /// The delay is derived rather than randomised precisely so a restart does
12509    /// not re-roll every module into a fresh chance of collision. A random
12510    /// source would satisfy the dispersal test above and quietly lose this.
12511    #[test]
12512    fn a_module_offset_is_stable_across_restarts() {
12513        let cadence = Duration::from_secs(30);
12514        for id in FLEET {
12515            assert_eq!(
12516                jittered_health_delay(id, 0, cadence),
12517                jittered_health_delay(id, 0, cadence),
12518                "{id}: the same module and probe index must produce the same offset"
12519            );
12520        }
12521    }
12522
12523    /// A zero cadence disables probing rather than producing a busy loop.
12524    #[test]
12525    fn zero_cadence_yields_zero_delay() {
12526        assert_eq!(
12527            jittered_health_delay("aft", 0, Duration::ZERO),
12528            Duration::ZERO
12529        );
12530    }
12531}
12532
12533#[cfg(all(test, target_os = "linux"))]
12534mod cgroup_placement_tests {
12535    use super::{
12536        apply_cgroup_placement, remove_module_cgroup, ModuleProtocol, ModuleSpec, SuperviseError,
12537        SupervisedChild,
12538    };
12539    use crate::stderr_tail::{StderrRing, StderrTailConfig};
12540    use std::{
12541        fs, io,
12542        path::{Path, PathBuf},
12543        sync::{Arc, Mutex},
12544    };
12545    use subc_test_support::TestTempDir;
12546    use tokio::process::Command;
12547
12548    #[tokio::test]
12549    async fn unique_spawn_cgroups_clean_up_across_restarts_and_shutdown() {
12550        use super::*;
12551        let dir = TestTempDir::new("unique-spawn-cgroups");
12552        let root = PathBuf::from(format!(
12553            "/sys/fs/cgroup/subc-unique-test-{}-{}",
12554            std::process::id(),
12555            unix_ms_now()
12556        ));
12557        if let Err(error) = fs::create_dir(&root) {
12558            assert!(
12559                std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
12560                "required cgroup test cannot execute: {error}"
12561            );
12562            eprintln!("SKIP unique_spawn_cgroups_clean_up_across_restarts_and_shutdown: {error}");
12563            return;
12564        }
12565        let placement = subc_cgroup::prepare_at(&root).unwrap().unwrap();
12566        let group_count = || {
12567            fs::read_dir(root.join("subc-modules"))
12568                .unwrap()
12569                .map(|entry| entry.unwrap().file_type().unwrap())
12570                .filter(|kind| kind.is_dir())
12571                .count()
12572        };
12573        let supervisor =
12574            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
12575                .with_cgroup_placement(Some(placement.clone()));
12576        let runtime = supervisor.runtime_config();
12577        let mut spec = ModuleSpec {
12578            module_id: "unique-spawn".into(),
12579            program: PathBuf::from("/bin/sleep"),
12580            args: vec!["60".into()],
12581            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
12582                .into_iter()
12583                .map(|key| (key.into(), dir.to_string_lossy().into_owned()))
12584                .collect(),
12585            reserved: false,
12586            reserved_prefixes: vec![],
12587            protocol: ModuleProtocol::None,
12588            overlap: Default::default(),
12589        };
12590        let spawn = |spec: &ModuleSpec| {
12591            spawn_child(
12592                spec,
12593                None,
12594                None,
12595                &runtime.stderr_ring,
12596                None,
12597                &runtime.child_roster,
12598                Some(&placement),
12599            )
12600            .unwrap()
12601        };
12602        let mut live = spawn(&spec);
12603        for _ in 0..3 {
12604            // A new process can enter the old slot while retirement is pending.
12605            let next = spawn(&spec);
12606            assert_ne!(live.module_id, next.module_id);
12607            live.start_kill().unwrap();
12608            live.wait().await.unwrap();
12609            live = next;
12610            assert!(
12611                live.child.try_wait().unwrap().is_none(),
12612                "retiring the old slot must not kill the replacement"
12613            );
12614            assert_eq!(
12615                group_count(),
12616                1,
12617                "only the live spawn's cgroup should remain"
12618            );
12619        }
12620        supervisor.begin_daemon_shutdown();
12621        let reap = tokio::spawn(async move {
12622            live.wait().await.unwrap();
12623        });
12624        supervisor
12625            .end_children_for_daemon_shutdown(false, std::future::pending())
12626            .await;
12627        reap.await.unwrap();
12628        assert_eq!(group_count(), 0);
12629        // A normal exit uses the same tree-cleanup path as a killed spawn.
12630        spec.program = PathBuf::from("/bin/true");
12631        spec.args.clear();
12632        let fresh_roster = ChildRoster::default();
12633        let mut short = spawn_child(
12634            &spec,
12635            None,
12636            None,
12637            &runtime.stderr_ring,
12638            None,
12639            &fresh_roster,
12640            Some(&placement),
12641        )
12642        .unwrap();
12643        short.wait().await.unwrap();
12644        assert_eq!(group_count(), 0);
12645        spec.module_id = "_".repeat(255);
12646        let mut long_id = spawn_child(
12647            &spec,
12648            None,
12649            None,
12650            &runtime.stderr_ring,
12651            None,
12652            &fresh_roster,
12653            Some(&placement),
12654        )
12655        .unwrap();
12656        long_id.wait().await.unwrap();
12657        assert_eq!(
12658            group_count(),
12659            0,
12660            "valid long module IDs must not exceed cgroup NAME_MAX"
12661        );
12662        fs::remove_dir(root.join("subc-modules")).unwrap();
12663        fs::remove_dir(root).unwrap();
12664        eprintln!("EXECUTED unique_spawn_cgroups_clean_up_across_restarts_and_shutdown");
12665    }
12666
12667    #[test]
12668    fn failed_parent_cgroup_open_is_a_cgroup_supervision_error() {
12669        let path = Path::new("/definitely-missing-subc-cgroup");
12670        let mut command = Command::new("true");
12671        let error = apply_cgroup_placement(
12672            &mut command,
12673            &ModuleSpec {
12674                module_id: "broken-cgroup".to_string(),
12675                program: PathBuf::from("true"),
12676                args: Vec::new(),
12677                env: Vec::new(),
12678                reserved: false,
12679                reserved_prefixes: Vec::new(),
12680                protocol: ModuleProtocol::Subc,
12681                overlap: Default::default(),
12682            },
12683            path,
12684        )
12685        .expect_err("a parent cgroup open failure must reject the supervised spawn");
12686        let reason = error.to_string();
12687
12688        assert!(
12689            matches!(error, SuperviseError::Cgroup { .. }),
12690            "parent cgroup open must be reported as a cgroup supervision error: {reason}"
12691        );
12692        assert!(
12693            reason.contains("/definitely-missing-subc-cgroup/cgroup.procs"),
12694            "parent cgroup open failure must name cgroup.procs: {reason}"
12695        );
12696    }
12697
12698    #[tokio::test]
12699    async fn reaping_a_child_removes_its_empty_module_cgroup() {
12700        let root = TestTempDir::new("supervisor-reap-cgroup");
12701        fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
12702        let placement = subc_cgroup::prepare_at(&root)
12703            .expect("prepare scratch cgroup root")
12704            .expect("scratch root has a cgroup.procs marker");
12705        let module_id = "reaped-module";
12706        let module = placement
12707            .module_path(module_id)
12708            .expect("create scratch module cgroup");
12709        let child = Command::new("true")
12710            .env("XDG_DATA_HOME", root.path())
12711            .env("XDG_RUNTIME_DIR", root.path())
12712            .env("XDG_CONFIG_HOME", root.path())
12713            .spawn()
12714            .expect("spawn short-lived child");
12715        let pid = child.id().expect("spawned child has pid");
12716        let mut child = SupervisedChild {
12717            child,
12718            protocol: ModuleProtocol::Subc,
12719            module_id: module_id.to_string(),
12720            cgroup_placement: Some(placement),
12721            stdout_pump: None,
12722            stderr_pump: None,
12723            stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
12724            spawned_at_ms: 0,
12725            spawned_from: PathBuf::from("true"),
12726            spawned_file_identity: None,
12727            process_start_time: None,
12728            process_identity: None,
12729            pid,
12730            roster_guard: None,
12731            #[cfg(target_os = "macos")]
12732            privacy_exec: None,
12733            spawn_failure: None,
12734        };
12735
12736        child.wait().await.expect("reap short-lived child");
12737
12738        assert!(
12739            !module.exists(),
12740            "reaping the supervised child must remove its empty cgroup"
12741        );
12742    }
12743
12744    #[test]
12745    fn non_empty_cgroup_removal_is_reported_without_blocking_teardown() {
12746        let root = TestTempDir::new("supervisor-non-empty-cgroup");
12747        fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
12748        let placement = subc_cgroup::prepare_at(&root)
12749            .expect("prepare scratch cgroup root")
12750            .expect("scratch root has a cgroup.procs marker");
12751        let module = placement
12752            .module_path("surviving-module")
12753            .expect("create scratch module cgroup");
12754        fs::write(module.join("surviving-process"), b"still present")
12755            .expect("make scratch cgroup non-empty");
12756        let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::WARN);
12757
12758        remove_module_cgroup(&placement, "surviving-module");
12759
12760        let logs = crate::router::test_log::captured_logs(&logs);
12761        assert!(
12762            module.exists(),
12763            "failed removal must leave the cgroup intact"
12764        );
12765        assert!(
12766            logs.contains("could not remove module cgroup after process exit; continuing teardown")
12767                && logs.contains("surviving-module"),
12768            "best-effort removal must report the failure without returning it: {logs}"
12769        );
12770    }
12771
12772    #[test]
12773    fn cgroup_pre_exec_spawn_failure_names_the_cgroup_path() {
12774        let cgroup_path = PathBuf::from("/sys/fs/cgroup/subc-modules/broken-module");
12775        let reason = SuperviseError::Spawn {
12776            program: PathBuf::from("/bin/true"),
12777            source: io::Error::from_raw_os_error(13),
12778            cgroup_path: Some(cgroup_path.clone()),
12779        }
12780        .to_string();
12781
12782        assert!(
12783            reason.contains(&cgroup_path.display().to_string()),
12784            "a pre_exec spawn failure must name the cgroup path: {reason}"
12785        );
12786    }
12787}
12788
12789#[cfg(test)]
12790mod spawn_subscriber_lag_tests {
12791    use super::*;
12792
12793    /// A subscriber whose connection stops draining is dropped once its frame
12794    /// channel fills. The client must learn that from a terminal Error frame
12795    /// after the frames already queued for it, not from a stream that simply
12796    /// goes quiet.
12797    #[tokio::test]
12798    async fn lagged_spawn_subscriber_receives_a_terminal_lagged_error_after_its_queued_frames() {
12799        let feed = SpawnEventFeed::default();
12800        feed.configure_incarnation("lag-incarnation".to_string());
12801        // A one-slot connection queue that nobody reads until the emits are
12802        // done: the forwarder parks on it and the subscriber channel fills.
12803        let (tx, mut rx) = mpsc::channel(1);
12804        feed.subscribe(ConnectionId::new(1), 7, 1, None, FrameSink::new(tx))
12805            .expect("subscribe");
12806        let emitted = SPAWN_SUBSCRIBER_BUFFER + 16;
12807        for index in 0..emitted {
12808            feed.emit_spawned(&format!("lag-module-{index}"), 1000, 0);
12809            // Let the forwarder take what it can so the fill point is the
12810            // subscriber channel, not a scheduling accident.
12811            tokio::task::yield_now().await;
12812        }
12813        assert_eq!(
12814            feed.subscriber_count(),
12815            0,
12816            "the lagged subscriber must be removed"
12817        );
12818
12819        let mut data = Vec::new();
12820        let mut last = None;
12821        loop {
12822            let next = tokio::time::timeout(Duration::from_secs(5), rx.recv())
12823                .await
12824                .expect("the forwarder must finish once the subscriber is dropped");
12825            let Some(outbound) = next else { break };
12826            let frame = outbound.frame;
12827            if frame.header.ty == FrameType::StreamData {
12828                assert!(last.is_none(), "no data may follow the terminal frame");
12829                let event: SpawnEvent = serde_json::from_slice(&frame.body).unwrap();
12830                data.push(event.cursor.seq);
12831            } else {
12832                assert!(last.is_none(), "exactly one terminal frame");
12833                last = Some(frame);
12834            }
12835        }
12836        assert!(!data.is_empty(), "queued frames drain before the terminal");
12837        for pair in data.windows(2) {
12838            assert_eq!(
12839                pair[1],
12840                pair[0] + 1,
12841                "queued frames arrive dense and in order"
12842            );
12843        }
12844        let terminal = last.expect("a lagged subscriber must receive a terminal frame");
12845        assert_eq!(terminal.header.ty, FrameType::Error);
12846        assert_eq!(terminal.header.corr, 7);
12847        let body: subc_protocol::ErrorBody = serde_json::from_slice(&terminal.body).unwrap();
12848        assert_eq!(body.code, SPAWN_SUBSCRIBER_LAGGED_CODE);
12849        let detail = body.detail.expect("lagged error carries detail");
12850        assert_eq!(
12851            detail["first_undelivered_cursor"]["seq"],
12852            data.last().unwrap() + 1,
12853            "the named cursor is the first event the subscriber did not receive"
12854        );
12855        assert_eq!(
12856            detail["first_undelivered_cursor"]["daemon_incarnation"],
12857            "lag-incarnation"
12858        );
12859    }
12860}
12861
12862#[cfg(test)]
12863mod terminal_history_read_concurrency_tests {
12864    use super::*;
12865    use crate::terminal_journal::read_pause;
12866    use std::sync::mpsc as std_mpsc;
12867    use subc_test_support::TestTempDir;
12868
12869    fn journaled_ring(
12870        journal: &Arc<crate::terminal_journal::TerminalJournal>,
12871    ) -> Arc<Mutex<TerminalRing>> {
12872        Arc::new(Mutex::new(
12873            TerminalRing::new(TerminalRingConfig::default(), 1)
12874                .with_journal(Some(Arc::clone(journal))),
12875        ))
12876    }
12877
12878    fn crash(at_ms: u64) -> ExitReport {
12879        ExitReport {
12880            kind: ExitKind::Crash,
12881            code: Some(1),
12882            signal: None,
12883            at_ms,
12884        }
12885    }
12886
12887    /// Record an exit on another thread and report whether it finished within
12888    /// `bound`. The recorder thread is left running if it did not.
12889    fn record_within(
12890        module_id: &'static str,
12891        ring: &Arc<Mutex<TerminalRing>>,
12892        at_ms: u64,
12893        bound: Duration,
12894    ) -> bool {
12895        let ring = Arc::clone(ring);
12896        let (done, done_rx) = std_mpsc::channel();
12897        std::thread::spawn(move || {
12898            record_terminal(
12899                module_id,
12900                &ring,
12901                &SpawnEventFeed::default(),
12902                &crash(at_ms),
12903                TerminalDisposition::Restarting,
12904            );
12905            let _ = done.send(());
12906        });
12907        done_rx.recv_timeout(bound).is_ok()
12908    }
12909
12910    /// A history read in progress must not hold the journal writer (which every
12911    /// module's exit recording needs) or the module's own ring. Exits recorded
12912    /// while the read is paused complete promptly; the paused read answers as of
12913    /// the moment it started, and the next read has each exit exactly once.
12914    #[test]
12915    fn exits_recorded_during_a_paused_history_read_are_not_blocked_or_half_merged() {
12916        let dir = TestTempDir::new("terminal-history-concurrent-read");
12917        let path = dir.join("terminals.jsonl");
12918        let journal = Arc::new(crate::terminal_journal::TerminalJournal::open(
12919            path.clone(),
12920            "daemon".into(),
12921        ));
12922        let reader_ring = journaled_ring(&journal);
12923        let other_ring = journaled_ring(&journal);
12924        assert!(record_within(
12925            "reader-module",
12926            &reader_ring,
12927            10,
12928            Duration::from_secs(5)
12929        ));
12930
12931        let (started, release) = read_pause::install(&path);
12932        let reading = {
12933            let ring = Arc::clone(&reader_ring);
12934            std::thread::spawn(move || durable_terminal_history_of(&ring, "reader-module"))
12935        };
12936        started
12937            .recv_timeout(Duration::from_secs(5))
12938            .expect("the history read reached its pause");
12939
12940        let bound = Duration::from_secs(1);
12941        assert!(
12942            record_within("other-module", &other_ring, 20, bound),
12943            "another module's exit waited on a history read (journal writer held)"
12944        );
12945        assert!(
12946            record_within("reader-module", &reader_ring, 30, bound),
12947            "the read module's own exit waited on its history read (ring held)"
12948        );
12949
12950        drop(release);
12951        let paused = reading.join().unwrap();
12952        assert_eq!(
12953            paused.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
12954            vec![10],
12955            "an exit recorded after the read began lands in neither half of it"
12956        );
12957        assert_eq!(paused.journal_skipped_lines, 0);
12958        assert_eq!(paused.journal_read_errors, 0);
12959
12960        let after = durable_terminal_history_of(&reader_ring, "reader-module");
12961        assert_eq!(
12962            after.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
12963            vec![10, 30],
12964            "the next read merges ring and journal with no duplicate"
12965        );
12966        assert_eq!(after.journal_skipped_lines, 0);
12967    }
12968}
12969
12970/// What a restart does with the exited process's stderr reader. These drive
12971/// the same `settle_stderr_pump` the supervisor calls, with a reader the test
12972/// holds, so a reader that has not been scheduled by the bound is a controlled
12973/// input rather than something only a loaded machine produces.
12974#[cfg(test)]
12975mod stderr_settle_tests {
12976    use std::{
12977        future::Future,
12978        io,
12979        pin::Pin,
12980        sync::{Arc, Mutex},
12981        task::{Context, Poll},
12982        time::Duration,
12983    };
12984
12985    use tokio::{
12986        io::{AsyncRead, ReadBuf},
12987        sync::oneshot,
12988        time::Instant,
12989    };
12990
12991    use super::{settle_stderr_pump, StderrPump};
12992    use crate::stderr_tail::{
12993        pump_stderr_to, untimed, CaptureState, OutputSink, StderrRing, StderrTailConfig, TailEntry,
12994    };
12995
12996    const BOUND: Duration = Duration::from_millis(250);
12997
12998    /// Yields `before`, then stays pending until the gate is released, then
12999    /// yields `after` and reaches EOF. The bytes after the gate were written
13000    /// by a process that has already exited; only the reader is behind.
13001    struct HeldReader {
13002        before: Option<Vec<u8>>,
13003        gate: Option<oneshot::Receiver<()>>,
13004        after: io::Cursor<Vec<u8>>,
13005    }
13006
13007    impl AsyncRead for HeldReader {
13008        fn poll_read(
13009            mut self: Pin<&mut Self>,
13010            cx: &mut Context<'_>,
13011            buf: &mut ReadBuf<'_>,
13012        ) -> Poll<io::Result<()>> {
13013            if let Some(bytes) = self.before.take() {
13014                buf.put_slice(&bytes);
13015                return Poll::Ready(Ok(()));
13016            }
13017            if let Some(gate) = self.gate.as_mut() {
13018                match Pin::new(gate).poll(cx) {
13019                    Poll::Pending => return Poll::Pending,
13020                    Poll::Ready(_) => self.gate = None,
13021                }
13022            }
13023            Pin::new(&mut self.after).poll_read(cx, buf)
13024        }
13025    }
13026
13027    struct DiscardSink;
13028
13029    impl OutputSink for DiscardSink {
13030        fn write_line(&mut self, _line: &[u8]) {}
13031    }
13032
13033    fn line(text: &str) -> TailEntry {
13034        TailEntry::Line {
13035            text: text.to_string(),
13036            truncated: false,
13037            at_ms: None,
13038        }
13039    }
13040
13041    fn lock(ring: &Arc<Mutex<StderrRing>>) -> std::sync::MutexGuard<'_, StderrRing> {
13042        ring.lock().unwrap()
13043    }
13044
13045    /// Start a reader for a new process generation that delivers `before`
13046    /// immediately and `after` only once the returned sender fires (or is
13047    /// dropped).
13048    fn held_pump(
13049        ring: &Arc<Mutex<StderrRing>>,
13050        before: &str,
13051        after: &str,
13052    ) -> (StderrPump, oneshot::Sender<()>) {
13053        let generation = lock(ring).begin_process();
13054        let (release, gate) = oneshot::channel();
13055        let reader = HeldReader {
13056            before: Some(before.as_bytes().to_vec()),
13057            gate: Some(gate),
13058            after: io::Cursor::new(after.as_bytes().to_vec()),
13059        };
13060        let task = tokio::spawn(pump_stderr_to(
13061            reader,
13062            Arc::clone(ring),
13063            generation,
13064            DiscardSink,
13065        ));
13066        (StderrPump { task, generation }, release)
13067    }
13068
13069    async fn wait_until(ring: &Arc<Mutex<StderrRing>>, done: impl Fn(&StderrRing) -> bool) {
13070        for _ in 0..1000 {
13071            if done(&lock(ring)) {
13072                return;
13073            }
13074            tokio::time::sleep(Duration::from_millis(1)).await;
13075        }
13076        panic!(
13077            "ring never reached the expected state: {:?}",
13078            lock(ring).snapshot(None, None)
13079        );
13080    }
13081
13082    #[tokio::test(start_paused = true)]
13083    async fn a_crash_line_the_reader_had_not_reached_by_the_bound_is_kept_before_the_restart() {
13084        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
13085        let (pump, release) = held_pump(&ring, "booting\n", "config error: missing storage\n");
13086
13087        settle_stderr_pump("crasher", &ring, pump, BOUND).await;
13088        let before_release = lock(&ring).snapshot(None, None);
13089        assert!(
13090            matches!(before_release.capture, CaptureState::Incomplete { .. }),
13091            "a reader that has not reached EOF cannot claim a whole tail: {before_release:?}"
13092        );
13093
13094        // The restart: the next process starts and writes before the old
13095        // reader catches up.
13096        let next = lock(&ring).begin_process();
13097        lock(&ring).push_line_from(next, "next process booting");
13098        release.send(()).unwrap();
13099        wait_until(&ring, |ring| {
13100            ring.snapshot(None, None).capture == CaptureState::Captured
13101        })
13102        .await;
13103
13104        assert_eq!(
13105            untimed(lock(&ring).snapshot(None, None).entries),
13106            vec![
13107                line("booting"),
13108                line("config error: missing storage"),
13109                TailEntry::ProcessStart,
13110                line("next process booting"),
13111            ],
13112            "the crash's last line must survive a slow reader and stay in the crashed process's section"
13113        );
13114    }
13115
13116    #[tokio::test(start_paused = true)]
13117    async fn a_pipe_held_open_by_a_descendant_reads_incomplete_without_delaying_the_restart_past_the_bound(
13118    ) {
13119        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
13120        // `_held` is never fired: a descendant keeps the pipe open for the
13121        // whole test.
13122        let (pump, _held) = held_pump(&ring, "parent exiting\n", "");
13123
13124        let started = Instant::now();
13125        settle_stderr_pump("orphaning", &ring, pump, BOUND).await;
13126        assert_eq!(
13127            started.elapsed(),
13128            BOUND,
13129            "the restart must wait exactly the bound for a pipe that stays open, no longer"
13130        );
13131
13132        let next = lock(&ring).begin_process();
13133        lock(&ring).push_line_from(next, "next process booting");
13134        tokio::time::sleep(Duration::from_secs(60)).await;
13135
13136        let snapshot = lock(&ring).snapshot(None, None);
13137        match &snapshot.capture {
13138            CaptureState::Incomplete { reason } => assert!(
13139                reason.contains("had not reached EOF") && reason.contains("250ms"),
13140                "the reason must say what is missing and after how long: {reason}"
13141            ),
13142            other => panic!("expected Incomplete while the pipe is held open, got {other:?}"),
13143        }
13144        assert_eq!(
13145            untimed(snapshot.entries),
13146            vec![
13147                line("parent exiting"),
13148                TailEntry::ProcessStart,
13149                line("next process booting"),
13150            ]
13151        );
13152    }
13153
13154    #[tokio::test(start_paused = true)]
13155    async fn a_reader_that_reaches_eof_within_the_bound_leaves_the_tail_captured() {
13156        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
13157        let (pump, release) = held_pump(&ring, "one\n", "two\n");
13158        release.send(()).unwrap();
13159
13160        settle_stderr_pump("clean", &ring, pump, BOUND).await;
13161
13162        let snapshot = lock(&ring).snapshot(None, None);
13163        assert_eq!(snapshot.capture, CaptureState::Captured);
13164        assert_eq!(untimed(snapshot.entries), vec![line("one"), line("two")]);
13165    }
13166}
13167
13168/// Containment of a module's process tree (issue #109).
13169///
13170/// The behaviour these defend against is a module helper surviving its module:
13171/// on a real machine the Synapse embedding module's CUDA worker holds ~2.2 GB of
13172/// VRAM, so a leaked grandchild is a leaked GPU allocation, and a day of restarts
13173/// compounds it.
13174///
13175/// They run against the SUPERVISOR rather than the job-object crate because the
13176/// claim is about teardown: a crate-level test proves a job can reap a tree, not
13177/// that the daemon's drain path reaches it.
13178///
13179/// Windows-only, like the mechanism. On Unix this arm compiles out; the cgroup
13180/// lane there is a separate containment path with its own tests.
13181#[cfg(all(test, windows))]
13182mod job_containment_tests {
13183    use super::*;
13184    use std::{
13185        path::{Path, PathBuf},
13186        sync::{Arc, Mutex},
13187        time::{Duration, Instant},
13188    };
13189    use subc_test_support::TestTempDir;
13190
13191    /// The stub, expected beside this test executable.
13192    ///
13193    /// The existence check is here for the reason its twin at `fake_aft_stub_path`
13194    /// documents: `--lib` does not build `[[bin]]` targets, and a bare spawn
13195    /// failure then reads as a broken test rather than an unbuilt dependency.
13196    fn stub_path() -> PathBuf {
13197        let mut path = std::env::current_exe().expect("current_exe available in tests");
13198        path.pop();
13199        path.pop();
13200        path.push("fake-aft-stub.exe");
13201        assert!(
13202            path.exists(),
13203            "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
13204             [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
13205            path.display()
13206        );
13207        path
13208    }
13209
13210    /// Poll for the grandchild pid the stub records, and parse it.
13211    fn read_grandchild_pid(path: &Path) -> u32 {
13212        let deadline = Instant::now() + Duration::from_secs(10);
13213        loop {
13214            if let Ok(contents) = std::fs::read_to_string(path) {
13215                if let Ok(pid) = contents.trim().parse() {
13216                    return pid;
13217                }
13218            }
13219            assert!(
13220                Instant::now() < deadline,
13221                "the stub never recorded a grandchild pid at {}",
13222                path.display()
13223            );
13224            std::thread::sleep(Duration::from_millis(10));
13225        }
13226    }
13227
13228    /// Everything one fixture run needs, so the two tests below differ in exactly
13229    /// one place: whether the child is contained.
13230    struct Fixture {
13231        _dir: TestTempDir,
13232        module_id: String,
13233        grandchild: u32,
13234        child: Option<SupervisedChild>,
13235        registry: Arc<Registry>,
13236        snapshot: Arc<Mutex<SupervisorSnapshot>>,
13237        terminal_ring: Arc<Mutex<TerminalRing>>,
13238        spawn_events: SpawnEventFeed,
13239    }
13240
13241    fn fixture(label: &str, module_id: &str) -> Fixture {
13242        let dir = TestTempDir::new(label);
13243        let pid_file = dir.join("grandchild.pid");
13244        let supervisor = Supervisor::new_for_test(
13245            Arc::new(Registry::default()),
13246            RestartPolicy::new(3, Duration::ZERO),
13247        );
13248        let runtime = supervisor.runtime_config();
13249        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
13250        let spec = ModuleSpec {
13251            module_id: module_id.to_string(),
13252            program: stub_path(),
13253            // Zero args deliberately: a `--subc` argument would make the stub dial
13254            // a daemon that is not there, and the failure would land in the same
13255            // stderr ring this fixture exists to keep quiet.
13256            args: Vec::new(),
13257            env: vec![
13258                ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
13259                (
13260                    "FAKE_AFT_GRANDCHILD_PID_FILE".to_string(),
13261                    pid_file.display().to_string(),
13262                ),
13263            ],
13264            reserved: false,
13265            reserved_prefixes: Vec::new(),
13266            protocol: ModuleProtocol::Subc,
13267            overlap: Default::default(),
13268        };
13269        let child = spawn_and_mark_running(&spec, &runtime, &snapshot)
13270            .expect("spawn the supervised fixture");
13271        let grandchild = read_grandchild_pid(&pid_file);
13272        Fixture {
13273            _dir: dir,
13274            module_id: module_id.to_string(),
13275            grandchild,
13276            child: Some(child),
13277            registry: Arc::new(Registry::default()),
13278            snapshot,
13279            terminal_ring: Arc::clone(&runtime.terminal_ring),
13280            spawn_events: SpawnEventFeed::default(),
13281        }
13282    }
13283
13284    impl Fixture {
13285        /// Drain through the supervisor's own teardown path.
13286        async fn drain(&mut self) {
13287            let child = self
13288                .child
13289                .take()
13290                .expect("the fixture child is still present");
13291            drain_child_to_state(
13292                &self.module_id,
13293                ModuleProtocol::Subc,
13294                // No forwarding table in this fixture, so nothing reaches the
13295                // child over a connection.
13296                StopNotice::NotSent,
13297                &self.registry,
13298                None,
13299                &self.snapshot,
13300                &self.terminal_ring,
13301                &self.spawn_events,
13302                child,
13303                Duration::from_millis(500),
13304                ModuleState::Stopped,
13305                Some(false),
13306            )
13307            .await
13308            .expect("drain the supervised fixture");
13309        }
13310    }
13311
13312    /// Teardown reaps the grandchild, not merely the direct child.
13313    ///
13314    /// This is the assertion the change exists for. Before containment the
13315    /// grandchild survived: it is a separate process, and `start_kill` is
13316    /// `TerminateProcess` scoped to one pid.
13317    #[tokio::test]
13318    async fn teardown_reaps_the_grandchild() {
13319        let mut fixture = fixture("teardown-grandchild", "tree-teardown");
13320        let grandchild = fixture.grandchild;
13321
13322        assert!(
13323            subc_jobobject::process_exists(grandchild),
13324            "grandchild {grandchild} must be alive before teardown, or this proves nothing"
13325        );
13326
13327        fixture.drain().await;
13328
13329        assert!(
13330            subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
13331            "grandchild {grandchild} outlived module teardown: the tree was not contained"
13332        );
13333    }
13334
13335    /// The mutation control: with containment withheld, the grandchild survives
13336    /// the same kill.
13337    ///
13338    /// This is the defect reproduction from #109 — a direct-child kill reaches
13339    /// one pid, and the grandchild is a different process. It spawns OUTSIDE the
13340    /// supervisor because `spawn_and_mark_running` now always contains on
13341    /// Windows, which is the point: there is no longer a path that spawns
13342    /// uncontained, so the control has to construct one.
13343    ///
13344    /// Its job is to keep `teardown_reaps_the_grandchild` honest. If the
13345    /// grandchild ever dies here, that test is passing for a reason unrelated to
13346    /// the job object and the containment claim is unproven.
13347    #[test]
13348    fn an_uncontained_grandchild_survives_a_direct_child_kill() {
13349        let dir = TestTempDir::new("teardown-uncontained");
13350        let pid_file = dir.join("grandchild.pid");
13351        let mut child = std::process::Command::new(stub_path())
13352            .env("FAKE_AFT_NEVER_CONNECT", "1")
13353            .env(
13354                "FAKE_AFT_GRANDCHILD_PID_FILE",
13355                pid_file.display().to_string(),
13356            )
13357            .stdin(std::process::Stdio::null())
13358            .stdout(std::process::Stdio::null())
13359            .stderr(std::process::Stdio::null())
13360            .spawn()
13361            .expect("spawn the uncontained fixture");
13362        let grandchild = read_grandchild_pid(&pid_file);
13363
13364        // Exactly what the pre-fix teardown did: kill the direct child.
13365        child.kill().expect("kill the direct child");
13366        let _ = child.wait();
13367
13368        assert!(
13369            subc_jobobject::process_exists(grandchild),
13370            "grandchild {grandchild} died with the direct child, so this control no longer \
13371             distinguishes contained from uncontained teardown and the regression test is \
13372             passing vacuously"
13373        );
13374
13375        // The orphan this control demonstrates is the leak the fix prevents, so
13376        // the control must not leave one behind.
13377        kill_tree(grandchild);
13378    }
13379
13380    /// Crash durability: closing the containment handle reaps the tree with no
13381    /// teardown code running at all.
13382    ///
13383    /// This is the case `taskkill /T` cannot cover — a daemon that dies cannot
13384    /// call anything — and it is why containment is a kernel property of the
13385    /// handle rather than a step in the drain. Discovered by getting the
13386    /// mutation control wrong: clearing `job` to "disable" containment instead
13387    /// killed the tree, which is the guarantee, not a mistake.
13388    #[tokio::test]
13389    async fn dropping_containment_reaps_the_grandchild() {
13390        let mut fixture = fixture("drop-containment", "tree-drop");
13391        let grandchild = fixture.grandchild;
13392
13393        assert!(subc_jobobject::process_exists(grandchild));
13394
13395        // No `drain` call, no kill: dropping the handle is the entire mechanism.
13396        fixture.child.as_mut().expect("child present").job = None;
13397
13398        assert!(
13399            subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
13400            "grandchild {grandchild} survived the containment handle closing, so a daemon \
13401             crash would leave the tree behind"
13402        );
13403    }
13404
13405    /// Kill a pid and its tree, then confirm it is gone.
13406    fn kill_tree(pid: u32) {
13407        let _ = std::process::Command::new("taskkill.exe")
13408            .args(["/PID", &pid.to_string(), "/T", "/F"])
13409            .stdin(std::process::Stdio::null())
13410            .stdout(std::process::Stdio::null())
13411            .stderr(std::process::Stdio::null())
13412            .status();
13413        assert!(
13414            subc_jobobject::wait_for_process_exit(pid, Duration::from_secs(10)),
13415            "could not clean up grandchild {pid}"
13416        );
13417    }
13418}
13419
13420#[cfg(test)]
13421mod privacy_trampoline_configuration_tests {
13422    #[tokio::test]
13423    #[cfg_attr(not(target_os = "macos"), ignore = "requires macOS privacy trampoline")]
13424    async fn macos_spawn_without_a_configured_trampoline_refuses_by_name() {
13425        #[cfg(target_os = "macos")]
13426        {
13427            let supervisor = super::Supervisor::new(
13428                std::sync::Arc::new(crate::Registry::default()),
13429                super::RestartPolicy::default(),
13430            );
13431            let error = supervisor.spawn(spec()).unwrap_err();
13432            assert!(
13433                error
13434                    .to_string()
13435                    .contains("no privacy trampoline configured"),
13436                "{error}"
13437            );
13438        }
13439    }
13440
13441    #[tokio::test]
13442    #[cfg_attr(not(target_os = "macos"), ignore = "requires macOS privacy trampoline")]
13443    async fn macos_wrong_trampoline_is_rejected_by_the_startup_probe() {
13444        #[cfg(target_os = "macos")]
13445        {
13446            let supervisor = super::Supervisor::new(
13447                std::sync::Arc::new(crate::Registry::default()),
13448                super::RestartPolicy::default(),
13449            )
13450            .with_privacy_trampoline(std::env::current_exe().unwrap());
13451            let error = supervisor.spawn(spec()).unwrap_err();
13452            assert!(
13453                error
13454                    .to_string()
13455                    .contains("binary does not implement the privacy trampoline protocol"),
13456                "{error}"
13457            );
13458        }
13459    }
13460
13461    #[cfg(target_os = "macos")]
13462    fn spec() -> super::ModuleSpec {
13463        super::ModuleSpec {
13464            module_id: "privacy-configuration".into(),
13465            program: "/bin/sleep".into(),
13466            args: vec!["30".into()],
13467            env: vec![],
13468            reserved: false,
13469            reserved_prefixes: vec![],
13470            protocol: subc_control::ModuleProtocol::None,
13471            overlap: super::ModuleOverlap::Exclusive,
13472        }
13473    }
13474}
13475
13476#[cfg(test)]
13477mod privacy_exec_boundary_tests {
13478    #[cfg(target_os = "macos")]
13479    use super::*;
13480    #[cfg(target_os = "macos")]
13481    use std::{
13482        io::{Read, Write},
13483        net::{TcpListener, TcpStream},
13484    };
13485
13486    /// Unit-test-only pause at the actual early image read, not at a later
13487    /// status read. Production supervisors never inspect this environment key.
13488    #[cfg(target_os = "macos")]
13489    pub(super) fn before_image_sample(spec: &ModuleSpec, pid: u32) {
13490        if let Some((_, path)) = spec
13491            .env
13492            .iter()
13493            .find(|(key, _)| key == "SUBC_TEST_PRIVACY_SAMPLE_BARRIER")
13494        {
13495            let mut barrier = TcpStream::connect(path).unwrap();
13496            barrier
13497                .set_read_timeout(Some(Duration::from_secs(30)))
13498                .unwrap();
13499            barrier.write_all(&pid.to_ne_bytes()).unwrap();
13500            let mut release = [0];
13501            barrier.read_exact(&mut release).unwrap();
13502            assert_eq!(&release, b"X");
13503        }
13504    }
13505
13506    #[cfg(target_os = "macos")]
13507    fn spec(program: &str, args: &[&str]) -> ModuleSpec {
13508        ModuleSpec {
13509            module_id: "privacy-boundary".into(),
13510            program: program.into(),
13511            args: args.iter().map(|arg| (*arg).into()).collect(),
13512            env: vec![],
13513            reserved: false,
13514            reserved_prefixes: vec![],
13515            protocol: ModuleProtocol::None,
13516            overlap: ModuleOverlap::Exclusive,
13517        }
13518    }
13519
13520    #[cfg(target_os = "macos")]
13521    async fn accept(listener: TcpListener) -> TcpStream {
13522        // Socket readiness, not elapsed time, establishes both pause points.
13523        let listener = tokio::net::TcpListener::from_std({
13524            listener.set_nonblocking(true).unwrap();
13525            listener
13526        })
13527        .unwrap();
13528        let (stream, _) = tokio::time::timeout(Duration::from_secs(30), listener.accept())
13529            .await
13530            .unwrap()
13531            .unwrap();
13532        let stream = stream.into_std().unwrap();
13533        stream.set_nonblocking(false).unwrap();
13534        stream
13535            .set_read_timeout(Some(Duration::from_secs(30)))
13536            .unwrap();
13537        stream
13538    }
13539
13540    #[tokio::test]
13541    #[cfg_attr(not(target_os = "macos"), ignore = "requires macOS privacy trampoline")]
13542    async fn macos_roster_withholds_a_nonnull_trampoline_image_until_exec_confirmation() {
13543        #[cfg(target_os = "macos")]
13544        {
13545            let root = subc_test_support::TestTempDir::new("privacy-roster-barrier");
13546            // Loopback sockets also work when the replay adapter's TMPDIR is
13547            // longer than Darwin's Unix-domain socket path limit.
13548            let exec_listener = TcpListener::bind("127.0.0.1:0").unwrap();
13549            let sample_listener = TcpListener::bind("127.0.0.1:0").unwrap();
13550            let record = root.join("live-children.json");
13551            let supervisor =
13552                Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
13553                    .with_live_children_record(&record);
13554            let runtime = supervisor.runtime_config();
13555            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
13556            let mut spec = spec("/bin/sleep", &["30"]);
13557            spec.env = vec![
13558                (
13559                    "SUBC_TEST_PRIVACY_EXEC_BARRIER".into(),
13560                    exec_listener.local_addr().unwrap().to_string(),
13561                ),
13562                (
13563                    "SUBC_TEST_PRIVACY_SAMPLE_BARRIER".into(),
13564                    sample_listener.local_addr().unwrap().to_string(),
13565                ),
13566            ];
13567            let spawn = tokio::task::spawn_blocking(move || {
13568                spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap()
13569            });
13570            let mut sample = accept(sample_listener).await;
13571            let mut pid = [0; 4];
13572            sample.read_exact(&mut pid).unwrap();
13573            let pid = u32::from_ne_bytes(pid);
13574            let mut exec = accept(exec_listener).await;
13575            let mut ready = [0];
13576            exec.read_exact(&mut ready).unwrap();
13577            assert_eq!(&ready, b"R");
13578            // The early read is guaranteed to see a real, non-null trampoline
13579            // image: the fixture has reached its barrier and cannot exec yet.
13580            let trampoline = subc_os::file_identity(&test_privacy_trampoline()).unwrap();
13581            assert_eq!(
13582                observe_spawned_image(pid).unwrap().executable,
13583                Some(trampoline)
13584            );
13585            sample.write_all(b"X").unwrap();
13586            let mut child = spawn.await.unwrap();
13587            let early = crate::live_children::read_record(&record).unwrap();
13588            assert_eq!(early.len(), 1);
13589            assert_eq!(early[0].pid, pid);
13590            assert_eq!(
13591                early[0].executable, None,
13592                "unconfirmed trampoline image entered the roster"
13593            );
13594            assert!(child.report_ready.get().is_none());
13595            // The barrier's duration is unrelated to the production five-second
13596            // exec budget. Start the test's confirmation budget upon release.
13597            child.privacy_exec.as_mut().unwrap().deadline =
13598                tokio::time::Instant::now() + Duration::from_secs(30);
13599            exec.write_all(b"X").unwrap();
13600            child.confirm_privacy_exec().await;
13601            assert_eq!(child.spawn_failure, None);
13602            assert!(child.report_ready.get().is_some());
13603            let confirmed = crate::live_children::read_record(&record).unwrap();
13604            let module = subc_os::file_identity(std::path::Path::new("/bin/sleep")).unwrap();
13605            assert_ne!(module, trampoline);
13606            assert_eq!(confirmed[0].executable, Some(module.into()));
13607            child.start_kill().unwrap();
13608            child.wait().await.unwrap();
13609            child.release_roster();
13610        }
13611    }
13612
13613    #[tokio::test]
13614    #[cfg_attr(not(target_os = "macos"), ignore = "requires macOS privacy trampoline")]
13615    async fn macos_already_exited_121_in_try_wait_is_a_module_exit_not_a_trampoline_refusal() {
13616        #[cfg(target_os = "macos")]
13617        {
13618            let registry = Arc::new(Registry::default());
13619            let policy = RestartPolicy::new(0, Duration::ZERO);
13620            let supervisor = Supervisor::new_for_test(Arc::clone(&registry), policy);
13621            let runtime = supervisor.runtime_config();
13622            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
13623            let spec = spec("/bin/sh", &["-c", "exit 121"]);
13624            // Drive spawn and confirmation separately instead of starting the
13625            // monitor. WNOWAIT observes a real exit without consuming its status,
13626            // so confirmation's first try_wait must take the already-exited arm.
13627            let mut child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
13628            let pid = child.pid;
13629            tokio::task::spawn_blocking(move || {
13630                subc_os::fork_exec_test::wait_for_child_exit_without_reaping(pid)
13631            })
13632            .await
13633            .unwrap()
13634            .unwrap();
13635            child.privacy_exec.as_mut().unwrap().deadline =
13636                tokio::time::Instant::now() + Duration::from_secs(30);
13637            let status = child.wait().await.unwrap();
13638            assert_eq!(status.code(), Some(121));
13639            assert!(child.privacy_exec.is_none());
13640            assert!(
13641                child.report_ready.get().is_none(),
13642                "an exited module must not publish a live pid"
13643            );
13644            let report = classify_reaped_child_exit(&snapshot, &child, &status);
13645            on_child_exit(
13646                &spec,
13647                policy,
13648                &registry,
13649                &snapshot,
13650                &runtime.terminal_ring,
13651                &runtime.spawn_events,
13652                &runtime.child_roster,
13653                report,
13654            )
13655            .await;
13656            let state = lock_snapshot(&snapshot).unwrap();
13657            assert_eq!(state.state, ModuleState::Failed);
13658            assert_eq!(state.last_exit.as_ref().unwrap().code, Some(121));
13659            assert_eq!(state.reported_pid(), None);
13660            drop(state);
13661            let history = runtime.terminal_ring.lock().unwrap().snapshot();
13662            assert_eq!(history.entries.len(), 1);
13663            let terminal = &history.entries[0];
13664            assert_eq!(terminal.exit_code, Some(121));
13665            assert_eq!(terminal.exit_kind, subc_control::TerminalExitKind::Crash);
13666            assert_eq!(terminal.disposition, TerminalDisposition::Failed);
13667            assert_eq!(
13668                terminal.disposition_detail.as_deref(),
13669                Some(policy.budget_exhausted_detail().as_str()),
13670                "module exit 121 was classified as a trampoline refusal: {terminal:?}"
13671            );
13672            assert_eq!(child.spawn_failure, None);
13673            child.release_roster();
13674        }
13675    }
13676}
13677
13678/// The daemon's real spawn path hands a subc-wire child its launch nonce on
13679/// descriptor 3, without an environment copy. The shell records the nonce
13680/// and its environment after exec so these tests observe the real handover.
13681#[cfg(all(test, unix))]
13682mod launch_nonce_descriptor_tests {
13683    use super::{ChildRoster, ModuleProtocol, ModuleSpec, SupervisorHandle};
13684    use crate::stderr_tail::{StderrRing, StderrTailConfig};
13685    use std::{
13686        path::PathBuf,
13687        sync::{Arc, Mutex},
13688        time::{Duration, Instant},
13689    };
13690    use subc_test_support::TestTempDir;
13691
13692    async fn probe(role: super::SpawnRole) {
13693        let scratch = TestTempDir::new("launch-nonce-descriptor");
13694        let fd_copy = scratch.join("from-descriptor");
13695        let env_copy = scratch.join("environment");
13696        let script = format!(
13697            "env > '{env}.tmp' && mv '{env}.tmp' '{env}'; cat <&3 > '{fd}.tmp' && mv '{fd}.tmp' '{fd}'; sleep 30",
13698            fd = fd_copy.display(), env = env_copy.display(),
13699        );
13700        let xdg = |name: &str| (name.to_string(), scratch.join(name).display().to_string());
13701        let spec = ModuleSpec {
13702            module_id: "nonce-descriptor-probe".to_string(),
13703            program: PathBuf::from("/bin/sh"),
13704            args: vec!["-c".to_string(), script],
13705            env: vec![
13706                xdg("XDG_DATA_HOME"),
13707                xdg("XDG_RUNTIME_DIR"),
13708                xdg("XDG_CONFIG_HOME"),
13709                ("SUBC_LAUNCH_NONCE".to_string(), "must-not-leak".to_string()),
13710            ],
13711            reserved: true,
13712            reserved_prefixes: Vec::new(),
13713            protocol: ModuleProtocol::Subc,
13714            overlap: Default::default(),
13715        };
13716        let handle = SupervisorHandle::new();
13717        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
13718        let roster = ChildRoster::default();
13719        #[cfg(target_os = "macos")]
13720        {
13721            let path = super::test_privacy_trampoline();
13722            roster.set_privacy_trampoline(super::probe_privacy_trampoline(&path).map(|()| path));
13723        }
13724        let child = super::spawn_child_in_slot(
13725            &spec,
13726            None,
13727            Some(&handle),
13728            &ring,
13729            None,
13730            &roster,
13731            #[cfg(target_os = "linux")]
13732            None,
13733            role,
13734            matches!(role, super::SpawnRole::SwapCandidate),
13735        )
13736        .expect("spawn probe");
13737        let deadline = Instant::now() + Duration::from_secs(10);
13738        while !(fd_copy.exists() && env_copy.exists()) {
13739            assert!(Instant::now() < deadline, "probe never wrote its copies");
13740            tokio::time::sleep(Duration::from_millis(20)).await;
13741        }
13742        let nonce = std::fs::read_to_string(fd_copy).unwrap();
13743        assert!(!nonce.is_empty());
13744        let environment = std::fs::read_to_string(env_copy).unwrap();
13745        assert!(environment
13746            .lines()
13747            .any(|line| line.starts_with("SUBC_LAUNCH_NONCE_FD=3:")));
13748        let copy = environment
13749            .lines()
13750            .find_map(|line| line.strip_prefix("SUBC_LAUNCH_NONCE="));
13751        assert_eq!(
13752            copy, None,
13753            "Unix children must never receive the environment nonce"
13754        );
13755        if matches!(role, super::SpawnRole::Plain) {
13756            assert_eq!(
13757                handle.spawn_nonce(&spec.module_id).as_deref(),
13758                Some(nonce.as_str())
13759            );
13760        }
13761        drop(child);
13762    }
13763
13764    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
13765    async fn a_spawned_module_receives_its_nonce_only_on_descriptor_3() {
13766        probe(super::SpawnRole::Plain).await;
13767    }
13768
13769    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
13770    async fn swap_candidate_receives_its_nonce_only_on_descriptor_3() {
13771        probe(super::SpawnRole::SwapCandidate).await;
13772    }
13773}
13774
13775#[cfg(all(test, target_os = "linux"))]
13776mod cgroup_containment_tests {
13777    use super::*;
13778    use subc_test_support::TestTempDir;
13779
13780    fn running(pid: u32) -> bool {
13781        // An orphan can remain a zombie until the container init reaps it.
13782        std::fs::read_to_string(format!("/proc/{pid}/stat"))
13783            .ok()
13784            .and_then(|stat| {
13785                stat.rsplit_once(") ")
13786                    .map(|(_, rest)| rest.starts_with('Z'))
13787            })
13788            .is_some_and(|zombie| !zombie)
13789    }
13790
13791    #[tokio::test]
13792    async fn linux_teardown_reaps_the_grandchild() {
13793        teardown_tree("linux_teardown_reaps_the_grandchild", false).await;
13794    }
13795
13796    #[tokio::test]
13797    async fn linux_shutdown_straggler_reaps_the_grandchild() {
13798        teardown_tree("linux_shutdown_straggler_reaps_the_grandchild", true).await;
13799    }
13800
13801    async fn teardown_tree(test_name: &str, shutdown: bool) {
13802        let dir = TestTempDir::new(test_name);
13803        let root = PathBuf::from(format!(
13804            "/sys/fs/cgroup/subc-tree-test-{test_name}-{}-{}",
13805            std::process::id(),
13806            unix_ms_now()
13807        ));
13808        if let Err(error) = std::fs::create_dir(&root) {
13809            assert!(
13810                std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
13811                "required cgroup test cannot execute: {error}"
13812            );
13813            eprintln!(
13814                "SKIP {test_name}: no writable delegated cgroup at {}: {error}",
13815                root.display()
13816            );
13817            return;
13818        }
13819        let placement = subc_cgroup::prepare_at(&root)
13820            .expect("prepare isolated kernel cgroup")
13821            .expect("isolated cgroup is delegated");
13822        let module_id = "tree-teardown";
13823        let module = placement
13824            .module_path(module_id)
13825            .expect("create isolated module cgroup");
13826        if !module.join("cgroup.kill").exists() {
13827            std::fs::remove_dir(&module).unwrap();
13828            std::fs::remove_dir(root.join("subc-modules")).unwrap();
13829            std::fs::remove_dir(&root).unwrap();
13830            assert!(
13831                std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
13832                "required cgroup.kill interface unavailable"
13833            );
13834            eprintln!("SKIP {test_name}: cgroup.kill unavailable (kernel < 5.14)");
13835            return;
13836        }
13837        let supervisor = Supervisor::new_for_test(
13838            Arc::new(Registry::default()),
13839            RestartPolicy::new(3, Duration::ZERO),
13840        )
13841        .with_cgroup_placement(Some(placement));
13842        let mut runtime = supervisor.runtime_config();
13843        runtime.child_roster = runtime
13844            .child_roster
13845            .for_module(Arc::new(Mutex::new(Duration::from_millis(100))));
13846        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
13847        let pid_file = dir.join("grandchild.pid");
13848        let spec = ModuleSpec {
13849            module_id: module_id.to_string(),
13850            program: PathBuf::from("/bin/sh"),
13851            args: vec![
13852                "-c".into(),
13853                "trap '' TERM; sleep 600 & echo $! > \"$1\"; wait".into(),
13854                "fixture".into(),
13855                pid_file.display().to_string(),
13856            ],
13857            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
13858                .into_iter()
13859                .map(|key| (key.to_string(), dir.display().to_string()))
13860                .collect(),
13861            reserved: false,
13862            reserved_prefixes: Vec::new(),
13863            protocol: ModuleProtocol::None,
13864            overlap: Default::default(),
13865        };
13866        let child =
13867            spawn_and_mark_running(&spec, &runtime, &snapshot).expect("spawn supervised tree");
13868        let deadline = tokio::time::Instant::now() + Duration::from_secs(10);
13869        let grandchild: u32 = loop {
13870            if let Ok(contents) = std::fs::read_to_string(&pid_file) {
13871                if let Ok(pid) = contents.trim().parse() {
13872                    break pid;
13873                }
13874            }
13875            assert!(
13876                tokio::time::Instant::now() < deadline,
13877                "grandchild pid was not recorded"
13878            );
13879            tokio::time::sleep(Duration::from_millis(10)).await;
13880        };
13881        assert!(
13882            running(grandchild),
13883            "grandchild must be alive before teardown"
13884        );
13885        if shutdown {
13886            let mut child = child;
13887            crate::child_roster::end_children_for_daemon_shutdown(
13888                &runtime.child_roster,
13889                false,
13890                std::future::pending(),
13891            )
13892            .await;
13893            child.wait().await.expect("reap shutdown straggler");
13894        } else {
13895            drain_child_to_state(
13896                module_id,
13897                ModuleProtocol::None,
13898                StopNotice::NotSent,
13899                &Registry::default(),
13900                None,
13901                &snapshot,
13902                &runtime.terminal_ring,
13903                &SpawnEventFeed::default(),
13904                child,
13905                Duration::from_millis(100),
13906                ModuleState::Stopped,
13907                Some(false),
13908            )
13909            .await
13910            .expect("real supervisor teardown");
13911        }
13912        let deadline = tokio::time::Instant::now() + Duration::from_secs(5);
13913        while running(grandchild) && tokio::time::Instant::now() < deadline {
13914            tokio::time::sleep(Duration::from_millis(10)).await;
13915        }
13916        let survived = running(grandchild);
13917        // Kill a surviving grandchild so a failed test does not leave it behind.
13918        if survived {
13919            let pid = rustix::process::Pid::from_raw(grandchild as i32).unwrap();
13920            let _ = rustix::process::kill_process(pid, rustix::process::Signal::KILL);
13921            tokio::time::sleep(Duration::from_millis(100)).await;
13922        }
13923        if module.exists() {
13924            std::fs::remove_dir(&module).expect("remove empty module cgroup");
13925        }
13926        std::fs::remove_dir(root.join("subc-modules")).unwrap();
13927        std::fs::remove_dir(&root).unwrap();
13928        assert!(
13929            !survived,
13930            "grandchild {grandchild} outlived module teardown"
13931        );
13932        eprintln!("EXECUTED {test_name}: grandchild {grandchild} killed in isolated cgroup");
13933    }
13934}