Skip to main content

subc_daemon/
supervise.rs

1use std::{
2    collections::{HashMap, HashSet, VecDeque},
3    error::Error,
4    fmt, io,
5    path::PathBuf,
6    process::{ExitStatus, Stdio},
7    sync::{Arc, Mutex, OnceLock},
8    time::{Duration, SystemTime, UNIX_EPOCH},
9};
10
11use cortexkit_log::Retention;
12use serde_json::Value;
13use subc_control::{
14    ClientControlPush, LiveSpawn, ModuleProtocol, RouteCloseReason, SpawnCursor, SpawnEvent,
15    SpawnEventKind, SpawnSnapshot, SupervisorHealthStatus, TerminalDisposition, TerminalExitKind,
16};
17use subc_protocol::{
18    manifest::{SelfSignalKind, SignalAnchor},
19    session::{
20        HealthReport, HealthStatus, ModuleControlCommand, ModuleControlRequest,
21        MODULE_CONTROL_OP_HEALTH_CHECK,
22    },
23    Flags, FrameType, Priority, SUBC_LAUNCH_NONCE_ENV, SUBC_MODULE_ID_ENV,
24};
25use tokio::{
26    process::{Child, Command},
27    sync::{mpsc, oneshot, watch, Mutex as AsyncMutex},
28    task::JoinHandle,
29    time::{sleep, sleep_until, timeout, timeout_at, Instant},
30};
31use tracing::{debug, error, info, warn};
32
33use crate::{
34    child_roster::ChildRoster,
35    daemon_config::{
36        CAPTURE_KEEP_ENV, CAPTURE_MAX_AGE_DAYS_ENV, CAPTURE_MAX_FILE_MB_ENV, CK_LOG_ENV,
37    },
38    forwarding::{
39        CloseReason, ForwardingError, ForwardingTable, GoodbyeTarget, ModuleControlRpcOutcome,
40        ModuleDrainTarget, PendingModuleControlRpc,
41    },
42    provenance::{spawned_file_identity, ExecutableIdentityProbe, SpawnedFileIdentity},
43    registry::{ConnectionId, RegistryError},
44    stderr_tail::{
45        pump_stderr_to, pump_stdout_to, ChildOutputSink, StderrRing, StderrTailConfig,
46        StderrTailSnapshot,
47    },
48    terminal_ring::{TerminalHistorySnapshot, TerminalRecord, TerminalRing, TerminalRingConfig},
49    Frame, FrameSink, Registry,
50};
51
52#[path = "supervise_swap.rs"]
53mod swap;
54
55/// Command-line flag used by supervised modules to find subc.
56///
57/// subc launches module-mode children as `<module> --subc <connection-file-path>`.
58/// The path points at the TCP+key connection file; it is not an ambient signal and
59/// is never inherited by standalone children.
60pub const SUBC_ARG: &str = "--subc";
61
62const DEFAULT_MAX_RESTARTS: u32 = 3;
63const DEFAULT_BACKOFF: Duration = Duration::from_millis(100);
64const DEFAULT_MAX_BACKOFF: Duration = Duration::from_secs(30);
65/// The span `DEFAULT_MAX_RESTARTS` is counted over. Ten minutes is long enough
66/// to contain a real crash loop (which respawns in seconds) and short enough
67/// that unrelated crashes hours apart never accumulate into a permanent stop.
68const DEFAULT_RESTART_WINDOW: Duration = Duration::from_secs(600);
69/// How long a drain waits for already-dispatched requests to finalize before
70/// the child is torn down. Sized for TOOL-SCALE work (bash, inspect, builds),
71/// not RPC-scale: the original 2s value silently cut nearly every real tool
72/// call at the fence, making the wait-for-finalize design decorative for the
73/// workloads it existed for. Quiescence short-circuits, so an idle module
74/// restarts immediately regardless of this value; the budget is spent only
75/// when a genuine in-flight request is worth finishing. Per-module override:
76/// `drain_timeout_ms` in subc.jsonc; per-restart override: the operator's
77/// `supervisor.restart{drain_timeout_ms}` (0 = cut now, for wedge bounces
78/// where a stuck request will never settle).
79pub const DEFAULT_DRAIN_TIMEOUT: Duration = Duration::from_secs(30);
80const REGISTRY_RELEASE_TIMEOUT: Duration = Duration::from_secs(1);
81const REGISTRY_RELEASE_POLL: Duration = Duration::from_millis(10);
82/// How long a restart waits for the exited process's output readers.
83///
84/// The restart does not depend on the stderr reader finishing. A reader still
85/// running at this bound is left running, and whatever it delivers later goes
86/// into the exited process's own section of the stderr ring (see
87/// `StderrRing::push_line_from`), ending naturally at EOF on its pipe. So the
88/// bound no longer decides whether a crash's last lines are kept: under load
89/// the reader may simply not have been scheduled yet, and cutting it there
90/// discarded exactly the lines that explained the crash.
91///
92/// What the bound still decides is when the tail starts reporting
93/// `Incomplete`: a pipe open past it usually means a descendant of the exited
94/// process still holds it, and the tail cannot claim to be whole until that
95/// pipe closes. That is also the only case in which the wait costs the restart
96/// anything, because a pipe with no other holder reaches EOF when the process
97/// exits. Under load a slow reader can show `Incomplete` briefly; it returns to
98/// `Captured` at EOF with nothing lost.
99///
100/// The stdout reader carries no ring, only the capture file, and is still
101/// stopped at this bound so an old process's stdout cannot trail into the file
102/// after its successor starts.
103const STDERR_PUMP_DRAIN_TIMEOUT: Duration = Duration::from_millis(250);
104/// Maximum number of supervised process spawn/exit facts retained per daemon incarnation.
105pub const SPAWN_EVENT_RING_CAPACITY: usize = 4096;
106const SPAWN_SUBSCRIBER_BUFFER: usize = SPAWN_EVENT_RING_CAPACITY + 1;
107/// Terminal Error code for a `supervisor.spawn_subscribe` stream the daemon
108/// dropped because the subscriber stopped draining its frames. The detail's
109/// `first_undelivered_cursor` names the first event it did not receive; the
110/// client resubscribes from the last cursor it did receive.
111pub(crate) const SPAWN_SUBSCRIBER_LAGGED_CODE: &str = "spawn_subscriber_lagged";
112
113#[cfg(target_os = "macos")]
114struct PrivacyExec {
115    reader: tokio::io::unix::AsyncFd<std::io::PipeReader>,
116    deadline: tokio::time::Instant,
117    expected: Option<subc_os::FileIdentity>,
118    trampoline: Option<subc_os::FileIdentity>,
119    script: bool,
120    module_id: String,
121}
122
123#[cfg(all(test, target_os = "macos"))]
124pub(crate) fn test_privacy_trampoline() -> PathBuf {
125    // Cargo's unit-test executable lives in <profile>/deps; its fixture bin
126    // lives beside that directory. This honors custom CARGO_TARGET_DIR too.
127    let path = std::env::current_exe()
128        .unwrap()
129        .parent()
130        .unwrap()
131        .parent()
132        .unwrap()
133        .join("privacy-trampoline-fixture");
134    // Without the fixture every macOS spawn is refused, and the tests that
135    // spawn fail later as a module in state Failed, which names the wrong
136    // cause. `cargo test -p subc-daemon --lib` alone does not build it.
137    assert!(
138        path.exists(),
139        "privacy-trampoline-fixture not built at {}: run `cargo build -p subc-daemon \
140         --bins --features test-support` or `cargo test -p subc-daemon` first",
141        path.display()
142    );
143    path
144}
145
146#[cfg(target_os = "macos")]
147fn probe_privacy_trampoline(path: &std::path::Path) -> Result<(), String> {
148    use std::io::Read;
149    let mut probe = std::process::Command::new(path)
150        .args(["__disclaim-exec", "--probe"])
151        .stdin(Stdio::null())
152        .stdout(Stdio::piped())
153        .stderr(Stdio::piped())
154        .spawn()
155        .map_err(|error| {
156            format!(
157                "privacy trampoline probe failed for {}: {error}",
158                path.display()
159            )
160        })?;
161    let deadline = std::time::Instant::now() + Duration::from_secs(5);
162    let status = loop {
163        match probe.try_wait() {
164            Ok(Some(status)) => break status,
165            Ok(None) if std::time::Instant::now() < deadline => {
166                std::thread::sleep(Duration::from_millis(5))
167            }
168            result => {
169                let _ = probe.kill();
170                let _ = probe.wait();
171                return Err(format!(
172                    "privacy trampoline probe failed or timed out for {}: {result:?}",
173                    path.display()
174                ));
175            }
176        }
177    };
178    let mut answer = String::new();
179    if let Some(stdout) = probe.stdout.take() {
180        let _ = stdout.take(256).read_to_string(&mut answer);
181    }
182    if status.success() && answer.trim() == subc_os::privacy_identity::TRAMPOLINE_PROBE {
183        return Ok(());
184    }
185    let mut diagnostic = String::new();
186    if let Some(stderr) = probe.stderr.take() {
187        let _ = stderr.take(1024).read_to_string(&mut diagnostic);
188    }
189    let cause = diagnostic
190        .trim()
191        .strip_prefix("ck-subc: own privacy identity refused: ")
192        .unwrap_or("binary does not implement the privacy trampoline protocol");
193    Err(format!(
194        "{cause}: probe of {} exited {status}",
195        path.display()
196    ))
197}
198
199#[cfg(target_os = "macos")]
200fn privacy_command(
201    spec: &ModuleSpec,
202    roster: &ChildRoster,
203) -> Result<
204    (
205        Command,
206        Option<PrivacyExec>,
207        subc_os::privacy_identity::ExecAcknowledgement,
208    ),
209    SuperviseError,
210> {
211    let failure = |cause: String| {
212        warn!(module_id = %spec.module_id, %cause, "privacy identity trampoline refused module spawn");
213        SuperviseError::Spawn {
214            program: spec.program.clone(),
215            source: io::Error::other(cause),
216            cgroup_path: None,
217        }
218    };
219    let trampoline = roster.privacy_trampoline().map_err(failure)?;
220    // Resolve PATH with the same environment the Command will receive. For
221    // scripts retain the existing orphan-identity rule: the kernel chooses
222    // the interpreter, and its observed image is the one recorded. Do not
223    // duplicate the kernel's shebang/PATH interpreter resolution in Rust.
224    let program = if spec.program.components().count() == 1 && !spec.program.is_absolute() {
225        let path = spec
226            .env
227            .iter()
228            .find(|(key, _)| key == "PATH")
229            .map(|(_, value)| std::ffi::OsString::from(value))
230            .or_else(|| std::env::var_os("PATH"))
231            .unwrap_or_else(|| "/usr/bin:/bin".into());
232        std::env::split_paths(&path)
233            .map(|dir| dir.join(&spec.program))
234            .find(|path| path.is_file())
235            .unwrap_or_else(|| spec.program.clone())
236    } else {
237        spec.program.clone()
238    };
239    let expected = subc_os::file_identity(&program);
240    let trampoline_image = subc_os::file_identity(&trampoline);
241    let script = {
242        use std::io::Read;
243        let mut prefix = [0u8; 2];
244        std::fs::File::open(&program)
245            .is_ok_and(|mut file| file.read_exact(&mut prefix).is_ok() && &prefix == b"#!")
246    };
247    if expected.is_none() || expected == subc_os::file_identity(&trampoline) {
248        return Err(failure(
249            "privacy identity module executable is missing or is the trampoline itself".to_string(),
250        ));
251    }
252    let (reader, ack) = subc_os::privacy_identity::ExecAcknowledgement::pipe()
253        .map_err(|error| failure(error.to_string()))?;
254    let reader =
255        tokio::io::unix::AsyncFd::new(reader).map_err(|error| failure(error.to_string()))?;
256    let mut command = Command::new(&trampoline);
257    command
258        .arg("__disclaim-exec")
259        .arg(ack.fd().to_string())
260        .arg(&program);
261    ack.install(command.as_std_mut());
262    Ok((
263        command,
264        Some(PrivacyExec {
265            reader,
266            deadline: tokio::time::Instant::now() + Duration::from_secs(5),
267            expected,
268            trampoline: trampoline_image,
269            script,
270            module_id: spec.module_id.clone(),
271        }),
272        ack,
273    ))
274}
275
276struct SupervisedChild {
277    child: Child,
278    #[cfg(target_os = "macos")]
279    privacy_exec: Option<PrivacyExec>,
280    /// Set once this launch's exec acknowledgement confirms the module image.
281    /// On macOS the pid first runs the `ck-subc` launch trampoline (see
282    /// `subc_os::privacy_identity`), which then replaces itself with the
283    /// module. The supervisor owns and can kill that pid from spawn, but
284    /// status readers report it only after this latch is set, so nothing
285    /// reports the trampoline's image as the module's.
286    #[cfg(target_os = "macos")]
287    report_ready: Arc<OnceLock<()>>,
288    /// Refusal before the module image was accepted, retained for terminal records.
289    spawn_failure: Option<String>,
290    /// The protocol this process was launched with. A reload can store a new
291    /// launch spec with a different protocol, but that takes effect only at the
292    /// next spawn, so this process keeps being handled by the protocol it
293    /// actually speaks.
294    protocol: ModuleProtocol,
295    /// This process's cgroup name: a bounded module/slot label followed by a
296    /// spawn suffix unique to this process (when cgroup placement is on). A
297    /// retired process in a slot may still be draining when a later one is
298    /// spawned into that slot, so the suffix keeps the later process out of
299    /// the retired one's cgroup, which is the domain a kill applies to.
300    #[cfg(target_os = "linux")]
301    module_id: String,
302    #[cfg(target_os = "linux")]
303    cgroup_placement: Option<subc_cgroup::Placement>,
304    /// The job that contains this child and every process it spawns (issue #109).
305    ///
306    /// Dropping this handle is what reaps a surviving tree when no supervisor
307    /// code runs — a daemon crash — because the job carries
308    /// `JOB_OBJECT_LIMIT_KILL_ON_JOB_CLOSE`.
309    ///
310    /// That limit is not crash-only, and the difference is worth knowing: a
311    /// Windows daemon *stop* is `taskkill` or the scheduler's `/End` — the
312    /// SIGTERM handler is `#[cfg(unix)]` — so the daemon dies with no stop
313    /// notice and the kernel closes the job handle, `TerminateProcess`ing every
314    /// module at once. Before this change they survived that, saw EOF on the
315    /// control socket, and ran their own teardown; Unix keeps that path
316    /// deliberately, so a module can seal a WAL or close a capture rather than
317    /// be killed mid-write. So this trades graceful teardown on every Windows
318    /// daemon stop for containment on a crash, which is the right way round
319    /// today: orphaned GPU workers are a reported, recurring problem, and the
320    /// modules that write most heavily do not run on Windows.
321    ///
322    /// The fix is a real Windows stop path — the daemon draining before it
323    /// exits, the twin of the Unix SIGTERM handler. Once it exists, this limit
324    /// reaches only what the drain left behind, which is what it should reach.
325    #[cfg(windows)]
326    job: Option<subc_jobobject::JobObject>,
327    stdout_pump: Option<JoinHandle<()>>,
328    stderr_pump: Option<StderrPump>,
329    stderr_ring: Arc<Mutex<StderrRing>>,
330    spawned_at_ms: u64,
331    spawned_from: PathBuf,
332    spawned_file_identity: Option<SpawnedFileIdentity>,
333    process_start_time: Option<u64>,
334    process_identity: Option<ProcessIdentity>,
335    pid: u32,
336    /// This process's entry in the daemon's child roster, released when the
337    /// process is reaped or this handle is dropped.
338    roster_guard: Option<crate::child_roster::RosterGuard>,
339}
340
341impl SupervisedChild {
342    fn id(&self) -> Option<u32> {
343        Some(self.pid)
344    }
345
346    fn process_identity(&self) -> Option<ProcessIdentity> {
347        self.process_identity
348    }
349
350    async fn wait(&mut self) -> io::Result<ExitStatus> {
351        #[cfg(target_os = "macos")]
352        self.confirm_privacy_exec().await;
353        // The roster entry is NOT released here. A daemon shutdown waits for the
354        // roster to empty and then exits the process, so releasing at the reap
355        // let it exit before the exit handler wrote this child's terminal record
356        // (the stderr drain and snapshot update sit in between), and the
357        // shutdown's own `daemon_shutdown` record was intermittently lost. The
358        // caller releases it after recording the exit (`release_roster`), and
359        // dropping the handle releases it too.
360        let result = self.child.wait().await;
361        #[cfg(target_os = "linux")]
362        if result.is_ok() {
363            if let Some(placement) = self.cgroup_placement.as_ref() {
364                cleanup_reaped_cgroup(placement, &self.module_id).await;
365                // Keep ownership while awaiting kernel population changes: a
366                // drain timeout may cancel this wait and then escalate/reap.
367                self.cgroup_placement = None;
368            }
369        }
370        result
371    }
372
373    #[cfg(target_os = "macos")]
374    async fn confirm_privacy_exec(&mut self) {
375        let Some(pending) = &mut self.privacy_exec else {
376            return;
377        };
378        let result = tokio::time::timeout_at(pending.deadline, async {
379            let mut record = Vec::new();
380            loop {
381                let mut ready = pending.reader.readable().await?;
382                let read = ready.try_io(|reader| {
383                    use std::io::Read;
384                    let mut reader = reader.get_ref();
385                    let mut buffer = [0u8; 256];
386                    reader.read(&mut buffer).map(|count| (count, buffer))
387                });
388                match read {
389                    Ok(Ok((0, _))) => return Ok::<_, io::Error>(record),
390                    Ok(Ok((count, buffer))) => {
391                        if record.len() + count > 1024 {
392                            return Err(io::Error::other(
393                                "privacy exec refusal record is too long",
394                            ));
395                        }
396                        record.extend_from_slice(&buffer[..count]);
397                    }
398                    Ok(Err(error)) => return Err(error),
399                    Err(_) => continue,
400                }
401            }
402        })
403        .await;
404        // Keep the reader in self across await: select cancellation must not
405        // discard the handshake or reset its original five-second deadline.
406        let pending = self.privacy_exec.as_ref().expect("pending exec");
407        let cause = match result {
408            Err(_) => Some("privacy identity trampoline did not exec within 5s".to_string()),
409            Ok(Err(error)) => Some(format!(
410                "privacy identity exec acknowledgement failed: {error}"
411            )),
412            Ok(Ok(record)) if !record.is_empty() => Some(
413                std::str::from_utf8(&record)
414                    .ok()
415                    .and_then(|record| {
416                        record
417                            .trim()
418                            .strip_prefix(subc_os::privacy_identity::EXEC_REFUSAL_TAG)
419                    })
420                    .filter(|cause| !cause.is_empty())
421                    .unwrap_or("invalid privacy exec refusal record")
422                    .to_string(),
423            ),
424            Ok(Ok(_)) => match self.child.try_wait() {
425                // Empty EOF is the exec acknowledgement. A real module may exit
426                // immediately, including with a reserved trampoline status; no
427                // image is admitted, and its ordinary exit contract stays intact.
428                Ok(Some(_status)) => None,
429                Err(error) => Some(format!("privacy identity trampoline wait failed: {error}")),
430                Ok(None) => {
431                    let image = observe_spawned_image(self.pid);
432                    if let Some(image) = image.filter(|image| {
433                        image.executable.is_some()
434                            && image.executable != pending.trampoline
435                            && (image.executable == pending.expected || pending.script)
436                    }) {
437                        if let Some(guard) = &self.roster_guard {
438                            guard.confirm_executable(image);
439                        }
440                        let _ = self.report_ready.set(());
441                        info!(module_id = %pending.module_id, pid = self.pid,
442                            "module spawned with own privacy identity (responsibility disclaimed)");
443                        None
444                    } else if image.is_none()
445                        || image.is_some_and(|image| image.executable.is_none())
446                    {
447                        // A process can exit between try_wait and the kernel
448                        // image read. Empty EOF already acknowledged exec, so
449                        // preserve that module's ordinary exit rather than
450                        // mislabel a disappearing image as trampoline refusal.
451                        // Keep pending in self across await so cancellation does
452                        // not discard validation or reset its original deadline.
453                        match tokio::time::timeout_at(pending.deadline, self.child.wait()).await {
454                            Ok(Ok(_status)) => None,
455                            Ok(Err(error)) => Some(format!("privacy identity trampoline wait failed: {error}")),
456                            Err(_) => Some("privacy identity module executable remained unreadable until the 5s exec deadline".to_string()),
457                        }
458                    } else {
459                        Some("privacy identity trampoline executable mismatch: expected the module program, not ck-subc".to_string())
460                    }
461                }
462            },
463        };
464        let pending = self.privacy_exec.take().expect("pending exec");
465        if let Some(cause) = cause {
466            warn!(module_id = %pending.module_id, pid = self.pid, %cause, "privacy identity trampoline refused module spawn");
467            self.spawn_failure = Some(cause);
468            // No image is admitted on failure. Reach the entire fresh process
469            // group, including a module which spawned a helper before refusal.
470            if let Some(pid) = rustix::process::Pid::from_raw(self.pid as i32) {
471                let _ = rustix::process::kill_process_group(pid, rustix::process::Signal::KILL);
472            }
473            let _ = self.child.start_kill();
474        }
475    }
476
477    /// Releases this child's daemon-shutdown roster entry once its exit has
478    /// been recorded. The pid is already reaped and free for reuse, so the
479    /// entry must not outlive the record any longer than that.
480    fn release_roster(&mut self) {
481        self.roster_guard = None;
482    }
483
484    /// Kill the child and, where containment is available, its process tree.
485    ///
486    /// `Child::kill` is `TerminateProcess` scoped to one pid, so a module with a
487    /// helper process leaked the helper — the Synapse embedding module's CUDA
488    /// worker holds the GPU allocation, so the leak cost VRAM until the next
489    /// restart of something else. Terminating the job reaches grandchildren that
490    /// a tree walk cannot, including one whose parent has already exited and
491    /// been reparented away.
492    ///
493    /// Best-effort like `request_graceful_stop`: a job failure is logged and the
494    /// direct-child kill still decides the outcome, so containment can never
495    /// change whether a module is reported as stopped.
496    fn start_kill(&mut self) -> io::Result<()> {
497        #[cfg(windows)]
498        if let Some(job) = &self.job {
499            if let Err(error) = job.terminate() {
500                debug!(
501                    error = %error,
502                    "job termination failed; the direct-child kill still owns the outcome"
503                );
504            }
505        }
506        #[cfg(target_os = "linux")]
507        kill_module_cgroup(self.cgroup_placement.as_ref(), &self.module_id);
508        self.child.start_kill()
509    }
510
511    async fn drain_stderr(&mut self, module_id: &str) {
512        if let Some(mut pump) = self.stdout_pump.take() {
513            match timeout(STDERR_PUMP_DRAIN_TIMEOUT, &mut pump).await {
514                Ok(Ok(())) => {}
515                Ok(Err(error)) => {
516                    warn!(module_id, error = %error, "stdout pump ended unexpectedly");
517                }
518                Err(_) => {
519                    pump.abort();
520                    warn!(
521                        module_id,
522                        waited = ?STDERR_PUMP_DRAIN_TIMEOUT,
523                        "stdout pump did not drain before restart; stopped it before the next process"
524                    );
525                }
526            }
527        }
528
529        let Some(pump) = self.stderr_pump.take() else {
530            return;
531        };
532        settle_stderr_pump(
533            module_id,
534            &self.stderr_ring,
535            pump,
536            STDERR_PUMP_DRAIN_TIMEOUT,
537        )
538        .await;
539    }
540}
541
542/// The reader task for one process's stderr, with the ring generation its
543/// lines are attributed to.
544struct StderrPump {
545    task: JoinHandle<()>,
546    generation: u64,
547}
548
549/// Retire an exited process's stderr reader and wait up to `bound` for it to
550/// reach EOF. A reader still running at the bound is detached, not stopped: it
551/// keeps filling the exited process's section of the ring until its pipe
552/// closes, and the tail reads `Incomplete` until then. See
553/// [`STDERR_PUMP_DRAIN_TIMEOUT`] for why.
554async fn settle_stderr_pump(
555    module_id: &str,
556    ring: &Arc<Mutex<StderrRing>>,
557    pump: StderrPump,
558    bound: Duration,
559) {
560    let lock = || ring.lock().unwrap_or_else(|poisoned| poisoned.into_inner());
561    let StderrPump {
562        mut task,
563        generation,
564    } = pump;
565    lock().retire_pump(generation);
566    match timeout(bound, &mut task).await {
567        Ok(Ok(())) => {}
568        Ok(Err(err)) => {
569            let mut ring = lock();
570            ring.mark_incomplete_from(generation, format!("stderr pump ended unexpectedly: {err}"));
571            ring.finish_pump(generation);
572            warn!(module_id, error = %err, "stderr pump ended before clean EOF");
573        }
574        Err(_) => {
575            // Dropping the handle detaches the task; it ends at EOF on its pipe.
576            drop(task);
577            lock().mark_pump_late(
578                generation,
579                format!(
580                    "stderr of the exited process had not reached EOF {bound:?} after it was \
581                     retired (a descendant may still hold the pipe open); lines it still \
582                     writes are kept in that process's section"
583                ),
584            );
585            warn!(
586                module_id,
587                waited = ?bound,
588                "stderr pipe of the exited process is still open; its reader keeps running without delaying the restart"
589            );
590        }
591    }
592}
593
594fn registration_release_events() -> &'static watch::Sender<u64> {
595    static EVENTS: OnceLock<watch::Sender<u64>> = OnceLock::new();
596    EVENTS.get_or_init(|| {
597        let (sender, _receiver) = watch::channel(0);
598        sender
599    })
600}
601
602pub(crate) fn notify_registration_release() {
603    let events = registration_release_events();
604    let next_generation = (*events.borrow()).wrapping_add(1);
605    events.send_replace(next_generation);
606}
607
608/// How to launch one singleton module process.
609#[derive(Debug, Clone, PartialEq, Eq)]
610pub struct ModuleSpec {
611    pub module_id: String,
612    pub program: PathBuf,
613    pub args: Vec<String>,
614    pub env: Vec<(String, String)>,
615    /// When true this is a reserved module: each spawn gets a fresh one-time launch
616    /// nonce that the child must echo in its HELLO, so only the daemon-spawned
617    /// process can register this module_id (a security-boundary module like the
618    /// credential vault must not be impersonable while it is down/restarting).
619    pub reserved: bool,
620    /// Module-id prefixes this supervised module owns for reserved HELLO checks.
621    /// Prefixes come from daemon config and must end in `:` before they reach the
622    /// supervisor; the owner module's current spawn nonce authorizes claims under
623    /// each prefix.
624    pub reserved_prefixes: Vec<String>,
625    /// The wire protocol this module speaks, as DECLARED in daemon config.
626    ///
627    /// [`ModuleProtocol::None`] changes five things and nothing else: wire health
628    /// probing is suppressed (an optional HTTP probe can replace it), teardown sends SIGTERM before waiting,
629    /// `route.open` is refused, the spawn passes NO `--subc <path>` argument
630    /// and NO launch nonce, and a clean exit the daemon did not request is
631    /// restarted as a crash rather than recorded as a stop (see `on_child_exit`:
632    /// a stock program exits 0 on a stray SIGTERM, and a stop would leave it
633    /// down for good). `SUBC_MODULE_ID` still goes into the environment,
634    /// because a process ignores an environment variable it does not read.
635    ///
636    /// The argument is the part that cannot be "harmless to a process that
637    /// ignores it": a stock binary exits on an unknown flag before it listens
638    /// (`nats-server`: "flag provided but not defined: -subc"), which is how the
639    /// first conformance run against this mode found it. The nonce is withheld
640    /// because a process that will never present it gains nothing from holding
641    /// it, and a secret in the environment of a process that does not need it is
642    /// a leak surface for no benefit.
643    pub protocol: ModuleProtocol,
644    /// Whether two processes of this module may run at once, which is what a
645    /// blue/green swap does for the length of its overlap. Declared in daemon
646    /// config because the daemon must be able to answer it while the module is
647    /// down, and so a module cannot talk itself into it after registering.
648    pub overlap: ModuleOverlap,
649}
650
651/// Whether a module tolerates a second process of itself running alongside.
652///
653/// Most modules are single-writer on their store (a WAL, a capture log, a
654/// resident index behind a writer barrier), and two processes on one store
655/// corrupt it. So a swap, which overlaps the old and new process by design,
656/// is refused unless the module's config opts in with `overlap: "safe"`.
657#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
658pub enum ModuleOverlap {
659    /// Never run two processes of this module at once. The default.
660    #[default]
661    Exclusive,
662    /// The module has said a second process of itself is harmless for the
663    /// length of a swap.
664    ///
665    /// Declare it only if a second instance can run for a few seconds without
666    /// touching ANY single-writer store: every database, WAL, index, projector
667    /// and scheduled job the module owns. A lease on part of that state is not
668    /// enough. broca's session lease guards WAL appends while its run index, its
669    /// store projector and its archive fold timer (which unlinks live WAL files)
670    /// stay single-writer, so broca is exclusive despite holding a lease. The
671    /// refusal only fires after this has been decided, so the decision is the
672    /// check.
673    Safe,
674}
675
676impl ModuleOverlap {
677    pub fn as_str(self) -> &'static str {
678        match self {
679            Self::Exclusive => "exclusive",
680            Self::Safe => "safe",
681        }
682    }
683}
684
685/// Environment variable telling a spawned module which case it was started
686/// for, before it sends HELLO. Only a swap candidate carries it, as
687/// [`SPAWN_ROLE_SWAP_CANDIDATE`]; every other spawn has it removed.
688///
689/// It chooses a warm-up budget, nothing else: a swap candidate can warm for
690/// longer because nobody waits on it, while a plain restart must flip ready
691/// quickly because callers see `module_warming` until it does. Absence means
692/// plain restart, the safe reading. The daemon trusts nothing about it; the
693/// candidate is proven by its launch nonce at HELLO.
694pub const SUBC_SPAWN_ROLE_ENV: &str = "SUBC_SPAWN_ROLE";
695/// The one value of [`SUBC_SPAWN_ROLE_ENV`] the daemon sets.
696pub const SPAWN_ROLE_SWAP_CANDIDATE: &str = "swap_candidate";
697/// How long a swap waits for its candidate to register and declare itself
698/// ready when the operator does not say. A module warming as a swap candidate
699/// may take up to 90 s (aft's ceiling, the largest in the fleet), so the
700/// daemon allows that plus time to start the process and send HELLO.
701pub const DEFAULT_SWAP_READY_TIMEOUT: Duration = Duration::from_secs(100);
702
703/// Bounded restart policy for crash exits.
704///
705/// `max_restarts` is the number of replacement processes allowed after the
706/// initial spawn WITHIN `window`. After that many crash restarts inside one
707/// window the module enters [`ModuleState::Failed`] and the supervisor stops
708/// the crash loop.
709///
710/// The budget is a RATE, not a lifetime total. It used to be a lifetime total,
711/// and that only survived because crashes were rare: a module that crashed
712/// three times across a week was disabled forever by crashes that had nothing
713/// to do with each other. That stopped being survivable once modules began
714/// exiting non-zero whenever the daemon's connection to them drops, because
715/// then every daemon-side connection drop spends a unit of the same budget and
716/// one flappy hour permanently stops a healthy module. Restarts older than
717/// `window` release their slot, so a module that crashed twice yesterday has a
718/// full budget today, while a genuine crash loop -- which is fast by
719/// definition -- still reaches the cap and stops.
720#[derive(Debug, Clone, Copy, PartialEq, Eq)]
721pub struct RestartPolicy {
722    pub max_restarts: u32,
723    /// Base delay before a crash replacement. The actual delay escalates with
724    /// the number of recent crash replacements and is capped by `max_backoff`.
725    pub backoff: Duration,
726    /// Maximum delay before a crash replacement.
727    pub max_backoff: Duration,
728    /// The span `max_restarts` is counted over. `Duration::ZERO` makes the
729    /// budget effectively infinite (nothing is ever in-window), which is why
730    /// daemon config refuses `window_secs: 0` rather than quietly accepting it.
731    pub window: Duration,
732}
733
734impl RestartPolicy {
735    /// A policy with the default crash window. Callers that care about the
736    /// window say so with [`Self::with_window`]; the ones that do not are
737    /// asking for the standard rate limit, not for no limit.
738    pub fn new(max_restarts: u32, backoff: Duration) -> Self {
739        Self {
740            max_restarts,
741            backoff,
742            max_backoff: DEFAULT_MAX_BACKOFF,
743            window: DEFAULT_RESTART_WINDOW,
744        }
745    }
746
747    pub fn with_max_backoff(mut self, max_backoff: Duration) -> Self {
748        self.max_backoff = max_backoff;
749        self
750    }
751
752    pub fn with_window(mut self, window: Duration) -> Self {
753        self.window = window;
754        self
755    }
756
757    /// Calculate the capped exponential delay for the next crash replacement.
758    /// `restart_in_window` is zero for the first replacement after an operator
759    /// action (restart, reload, re-enable) cleared the crash ring, or after all
760    /// older crash replacements have aged out of the window.
761    fn delay_for_restart(&self, restart_in_window: u32) -> Duration {
762        if self.backoff.is_zero() || self.max_backoff.is_zero() {
763            return Duration::ZERO;
764        }
765
766        let mut delay = self.backoff;
767        for _ in 0..restart_in_window {
768            if delay >= self.max_backoff {
769                return self.max_backoff;
770            }
771            delay = delay
772                .checked_mul(10)
773                .unwrap_or(self.max_backoff)
774                .min(self.max_backoff);
775        }
776        delay.min(self.max_backoff)
777    }
778
779    /// The one sentence that explains a budget-exhausted stop, used for both the
780    /// log line and the terminal record so the two cannot drift. It names the
781    /// window because `max_restarts=3` alone reads as a lifetime cap, which is
782    /// exactly what this budget is not.
783    fn budget_exhausted_detail(&self) -> String {
784        format!(
785            "crash budget exhausted: max_restarts={} within window_secs={}",
786            self.max_restarts,
787            self.window.as_secs()
788        )
789    }
790}
791
792impl Default for RestartPolicy {
793    fn default() -> Self {
794        Self {
795            max_restarts: DEFAULT_MAX_RESTARTS,
796            backoff: DEFAULT_BACKOFF,
797            max_backoff: DEFAULT_MAX_BACKOFF,
798            window: DEFAULT_RESTART_WINDOW,
799        }
800    }
801}
802
803#[derive(Debug, Clone, Copy, PartialEq, Eq)]
804struct CrashRestartSchedule {
805    restart_in_window: u32,
806    delay: Duration,
807}
808
809/// Whether the daemon itself will bring this module back after the exit being
810/// handled: it is enabled AND its in-window crash restarts are below the cap.
811///
812/// Takes `&mut` because reading the budget prunes it. Instants that fell out of
813/// the window are dropped here rather than by a timer, so the count is right
814/// the moment somebody asks and no bookkeeping runs for idle modules.
815fn daemon_will_restart(
816    state: &mut SupervisorSnapshot,
817    policy: &RestartPolicy,
818    now: Instant,
819) -> bool {
820    state.enabled && state.crash_restarts_in_window(policy.window, now) < policy.max_restarts
821}
822
823const DEFAULT_HEALTH_CADENCE: Duration = Duration::from_secs(30);
824const DEFAULT_HEALTH_DEADLINE: Duration = Duration::from_secs(5);
825const DEFAULT_HEALTH_FAILURE_THRESHOLD: u32 = 3;
826const MAX_HEALTH_METRICS_BYTES: usize = 16 * 1024;
827
828#[derive(Debug, Clone, Copy, PartialEq, Eq)]
829pub enum HealthAction {
830    Report,
831    Restart,
832    Alert,
833}
834
835impl fmt::Display for HealthAction {
836    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
837        f.write_str(match self {
838            Self::Report => "report",
839            Self::Restart => "restart",
840            Self::Alert => "alert",
841        })
842    }
843}
844
845#[derive(Debug, Clone, PartialEq, Eq)]
846pub struct HealthConfig {
847    /// Optional loopback HTTP endpoint for a managed non-wire process.
848    /// Changing it applies live on rescan; the process protocol changes only
849    /// at its next spawn.
850    pub http: Option<String>,
851    pub cadence: Duration,
852    pub deadline: Duration,
853    pub failure_threshold: u32,
854    pub on_degraded: HealthAction,
855    pub on_failing: HealthAction,
856    pub critical: bool,
857}
858
859impl Default for HealthConfig {
860    fn default() -> Self {
861        Self {
862            http: None,
863            cadence: DEFAULT_HEALTH_CADENCE,
864            deadline: DEFAULT_HEALTH_DEADLINE,
865            failure_threshold: DEFAULT_HEALTH_FAILURE_THRESHOLD,
866            on_degraded: HealthAction::Report,
867            on_failing: HealthAction::Report,
868            critical: false,
869        }
870    }
871}
872
873/// The supervisor's view of one module's health, relayed to clients over
874/// channel-0 and rendered by `ck health`.
875///
876/// THIS TYPE IS WHERE THE ABSENCE MEANINGS ARE CREATED, which is why they are
877/// stated here rather than only at the wire type a consumer reads. A reader can
878/// look up what `None` means; only a writer can silently change it, and the
879/// writer has no reason to go looking at a downstream contract before editing.
880///
881/// `last_probe_ms: None` MEANS NEVER PROBED, not probed-long-ago. It is cleared
882/// back to `None` on re-registration precisely so a respawned module does not
883/// carry its predecessor's timestamp — so an old value and an absent one call for
884/// opposite readings, and anything that defaulted this to a number would make a
885/// never-probed module indistinguishable from one probed at the epoch.
886///
887/// `detail` and `metrics` are `None` when the module published none on this
888/// probe, which does not mean it reported nothing wrong — it is also the shape
889/// when the probe never reached it. `last_probe_ms` is what separates those.
890#[derive(Debug, Clone, PartialEq)]
891pub struct ModuleHealthStatus {
892    pub status: SupervisorHealthStatus,
893    pub last_probe_ms: Option<u64>,
894    pub detail: Option<String>,
895    pub metrics: Option<Value>,
896    pub consecutive_failures: u32,
897    /// Number of replies received after a recurring health probe's deadline.
898    /// Unlike a timeout, every increment proves the module was alive.
899    pub late_answer_count: u64,
900    /// End-to-end latency of the newest late reply, measured from probe start.
901    pub last_late_answer_latency_ms: Option<u64>,
902    pub last_action: Option<String>,
903    /// Set together with `last_action`; the pair moves as one, and both being
904    /// absent means no escalation has ever been taken rather than that the last
905    /// one succeeded.
906    pub last_action_ms: Option<u64>,
907}
908
909impl Default for ModuleHealthStatus {
910    fn default() -> Self {
911        Self {
912            status: SupervisorHealthStatus::Unknown,
913            last_probe_ms: None,
914            detail: None,
915            metrics: None,
916            consecutive_failures: 0,
917            late_answer_count: 0,
918            last_late_answer_latency_ms: None,
919            last_action: None,
920            last_action_ms: None,
921        }
922    }
923}
924
925/// Typed lifecycle state for a supervised module.
926#[derive(Debug, Clone, Copy, PartialEq, Eq)]
927pub enum ModuleState {
928    Starting,
929    Running,
930    Unresponsive,
931    Restarting,
932    Draining,
933    Stopped,
934    Failed,
935    Disabled,
936}
937
938impl fmt::Display for ModuleState {
939    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
940        f.write_str(match self {
941            Self::Starting => "starting",
942            Self::Running => "running",
943            Self::Unresponsive => "unresponsive",
944            Self::Restarting => "restarting",
945            Self::Draining => "draining",
946            Self::Stopped => "stopped",
947            Self::Failed => "failed",
948            Self::Disabled => "disabled",
949        })
950    }
951}
952
953/// Supervisor classification of a child-process exit.
954#[derive(Debug, Clone, Copy, PartialEq, Eq)]
955pub enum ExitKind {
956    Clean,
957    Crash,
958    DeliberateSeverance,
959}
960
961impl From<ExitKind> for TerminalExitKind {
962    fn from(kind: ExitKind) -> Self {
963        match kind {
964            ExitKind::Clean => Self::Clean,
965            ExitKind::Crash => Self::Crash,
966            ExitKind::DeliberateSeverance => Self::DeliberateSeverance,
967        }
968    }
969}
970
971/// Exact process identity retained when a supervised module registers its
972/// connection. PID reuse makes a PID alone insufficient evidence of ownership.
973#[derive(Debug, Clone, Copy, PartialEq, Eq)]
974pub(crate) struct ProcessIdentity {
975    pub(crate) pid: u32,
976    pub(crate) start_time: u64,
977}
978
979/// Last observed child exit, if any.
980#[derive(Debug, Clone, PartialEq, Eq)]
981pub struct ExitReport {
982    pub kind: ExitKind,
983    pub code: Option<i32>,
984    pub signal: Option<i32>,
985    pub at_ms: u64,
986}
987
988/// Point-in-time module status answerable by subc without forwarding to the
989/// module process.
990#[derive(Debug, Clone, PartialEq)]
991pub struct ModuleStatus {
992    pub module_id: String,
993    pub state: ModuleState,
994    pub enabled: bool,
995    pub process_alive: bool,
996    pub registration_active: bool,
997    /// The module's declared wire protocol, carried beside `live` because it is
998    /// what makes `live` readable: the two fields answer one question together.
999    /// While a process is alive this is its launch declaration, not a later
1000    /// pending-reload edit. When down it is the configured next launch protocol.
1001    pub protocol: ModuleProtocol,
1002    /// Whether the module is serving, under the strongest definition the daemon
1003    /// can assert for its protocol.
1004    ///
1005    /// A subc module must also be REGISTERED: its process being alive says
1006    /// nothing about whether it can take a request. A `protocol: "none"` module
1007    /// never registers, so that term is dropped and this falls back to "enabled,
1008    /// running, and the process the daemon launched is alive" -- which is all
1009    /// the daemon observes about a process that speaks no subc wire. It stays a
1010    /// `bool` on the wire for compatibility; renderers pair it with `protocol`
1011    /// rather than printing it bare.
1012    pub live: bool,
1013    /// Crash restarts spent INSIDE `restart_window` as of this read. Older
1014    /// restarts have already released their slot, so this count can go down
1015    /// without anybody touching the module.
1016    pub restart_count: u32,
1017    /// Replacement processes spawned over this module's entire supervisor lifetime;
1018    /// unlike `restart_count`, this value is never reset by an operator action
1019    /// and never falls out of a window.
1020    pub lifetime_restarts: u32,
1021    pub spawn_generation: u64,
1022    /// The budget `restart_count` is spent against. Carried alongside the count
1023    /// because the count alone does not say how close the module is to being
1024    /// disabled, and reporting one without the other is what makes an
1025    /// about-to-be-retired module look ordinary.
1026    pub max_restarts: u32,
1027    /// The span `restart_count` is counted over. Carried with the pair above for
1028    /// the same reason they are carried together: "2 of 3" means one thing for a
1029    /// ten-minute window and something else entirely for a lifetime.
1030    pub restart_window: Duration,
1031    /// Effective drain and restart timing policy used by this running module.
1032    /// These values are carried together with the restart budget so status
1033    /// readers can compare configured intent with what the supervisor applied.
1034    pub drain_timeout: Duration,
1035    pub restart_backoff: Duration,
1036    pub restart_max_backoff: Duration,
1037    /// The module's process. On macOS this stays absent while the `ck-subc`
1038    /// launch trampoline is still running in that pid, and appears once the
1039    /// exec acknowledgement confirms the module image has replaced it. Launch
1040    /// time and the supervisor's own hold on the process are unaffected.
1041    pub pid: Option<u32>,
1042    pub spawned_at_ms: Option<u64>,
1043    pub spawned_from: Option<PathBuf>,
1044    pub process_start_time: Option<u64>,
1045    pub last_exit: Option<ExitReport>,
1046    pub health: ModuleHealthStatus,
1047}
1048
1049#[derive(Debug, Clone, PartialEq)]
1050struct SupervisorSnapshot {
1051    state: ModuleState,
1052    enabled: bool,
1053    process_alive: bool,
1054    spawned_protocol: Option<ModuleProtocol>,
1055    spawn_failure: Option<String>,
1056    /// When each crash restart was spent, oldest first. This IS the crash
1057    /// budget: its in-window length is the count an operator sees and the count
1058    /// the restart decision is made against, so there is no second counter that
1059    /// can disagree with it. Bounded by `max_restarts`, and cleared by the same
1060    /// operator actions that used to zero the old lifetime counter.
1061    crash_restarts: VecDeque<Instant>,
1062    lifetime_restarts: u32,
1063    /// Successful child spawns in this daemon incarnation.
1064    ///
1065    /// `lifetime_restarts` was considered and rejected: it starts at zero
1066    /// (line 640), successful initial/operator spawns in `set_running` do not
1067    /// increment it (lines 5264-5274), and crash/deliberate retry bookkeeping
1068    /// increments before a successful replacement exists (lines 604, 3846,
1069    /// and 3921), so a failed spawn can consume it. This counter moves only
1070    /// when a live PID is accepted below.
1071    spawn_generation: u64,
1072    pid: Option<u32>,
1073    #[cfg(target_os = "macos")]
1074    report_ready: Option<Arc<OnceLock<()>>>,
1075    /// Last reaped child, retained after current process facts are cleared.
1076    reaped_pid: Option<u32>,
1077    /// Whether the command-serving supervision loop has a scheduled respawn.
1078    respawn_pending: bool,
1079    /// A second restart is waiting for the replacement already scheduled.
1080    coalesced_restart_pending: bool,
1081    spawned_at_ms: Option<u64>,
1082    spawned_from: Option<PathBuf>,
1083    spawned_file_identity: Option<SpawnedFileIdentity>,
1084    process_start_time: Option<u64>,
1085    deliberate_severance: Option<ProcessIdentity>,
1086    last_exit: Option<ExitReport>,
1087    /// Diagnostic attached to the next drain's terminal record, if any.
1088    drain_disposition_detail: Option<String>,
1089    health: ModuleHealthStatus,
1090    /// Whether the current process was started as a swap candidate and so
1091    /// lives in the module's alternate cgroup. The next swap's candidate takes
1092    /// the other one, so the two processes of a swap never share a cgroup. A
1093    /// plain spawn always uses the primary cgroup.
1094    in_alternate_slot: bool,
1095    /// Whether the current `Draining` state ends in a replacement process
1096    /// (restart, reload, health restart) rather than a stop. Only meaningful
1097    /// while `state` is `Draining`; every entry into that state rewrites it.
1098    /// It is what lets route.open answer the retryable `module_reloading` to a
1099    /// consumer that reaches a still-registered process mid-restart, instead of
1100    /// the `supervisor_not_live` a stop or disable deserves.
1101    draining_to_replace: bool,
1102    /// Whether a configuration update has been applied since the current
1103    /// process was spawned, so that process runs an older spec than the one
1104    /// the supervisor now holds. A queued restart is only coalesced into a
1105    /// fresher process when this is false: a restart requested to pick up a
1106    /// new configuration must not be satisfied by a process that predates it.
1107    configuration_updated_since_spawn: bool,
1108}
1109
1110impl SupervisorSnapshot {
1111    /// The pid that status, provenance and resource readings may report. While
1112    /// the launch trampoline still runs in the pid, reading its executable or
1113    /// resource use would describe `ck-subc`, not the module, so none is
1114    /// reported until the exec acknowledgement confirms the module image.
1115    fn reported_pid(&self) -> Option<u32> {
1116        #[cfg(target_os = "macos")]
1117        if self
1118            .report_ready
1119            .as_ref()
1120            .is_some_and(|ready| ready.get().is_none())
1121        {
1122            return None;
1123        }
1124        self.pid
1125    }
1126
1127    fn starting() -> Self {
1128        Self::new(ModuleState::Starting, true)
1129    }
1130
1131    fn disabled() -> Self {
1132        Self::new(ModuleState::Disabled, false)
1133    }
1134
1135    fn failed() -> Self {
1136        Self::new(ModuleState::Failed, true)
1137    }
1138
1139    /// Crash restarts still inside `window`, having dropped the ones that are
1140    /// not. Pruning on read is what makes the budget a rate: an instant older
1141    /// than the window stops holding a slot the moment anybody counts.
1142    fn crash_restarts_in_window(&mut self, window: Duration, now: Instant) -> u32 {
1143        while let Some(oldest) = self.crash_restarts.front() {
1144            if now.duration_since(*oldest) > window {
1145                self.crash_restarts.pop_front();
1146            } else {
1147                break;
1148            }
1149        }
1150        u32::try_from(self.crash_restarts.len()).unwrap_or(u32::MAX)
1151    }
1152
1153    /// Spend one unit of the crash budget and record the restart in the ledger.
1154    ///
1155    /// The ring is bounded by the cap because more than `max_restarts` in-window
1156    /// instants can never be reached (the caller refuses the restart first), so
1157    /// anything beyond that is an unbounded queue waiting to happen.
1158    fn record_crash_restart(&mut self, policy: &RestartPolicy, now: Instant) {
1159        self.crash_restarts.push_back(now);
1160        while self.crash_restarts.len() > policy.max_restarts as usize {
1161            self.crash_restarts.pop_front();
1162        }
1163        self.lifetime_restarts += 1;
1164    }
1165
1166    /// Reserve one crash-restart slot and calculate the delay before respawning.
1167    /// The count is captured before recording this restart, so the first retry
1168    /// uses the base delay and each later in-window retry escalates once.
1169    fn next_crash_restart(
1170        &mut self,
1171        policy: &RestartPolicy,
1172        now: Instant,
1173    ) -> Option<CrashRestartSchedule> {
1174        let restart_in_window = self.crash_restarts_in_window(policy.window, now);
1175        if restart_in_window >= policy.max_restarts {
1176            return None;
1177        }
1178        self.record_crash_restart(policy, now);
1179        Some(CrashRestartSchedule {
1180            restart_in_window,
1181            delay: policy.delay_for_restart(restart_in_window),
1182        })
1183    }
1184
1185    /// Give the module its full budget back, as an operator restart, reload, or
1186    /// re-enable does. `lifetime_restarts` deliberately does not move: it is the
1187    /// ledger of what actually happened, and an operator action does not unmake
1188    /// the crashes.
1189    fn clear_crash_restarts(&mut self) {
1190        self.crash_restarts.clear();
1191    }
1192
1193    fn new(state: ModuleState, enabled: bool) -> Self {
1194        Self {
1195            state,
1196            enabled,
1197            process_alive: false,
1198            spawned_protocol: None,
1199            spawn_failure: None,
1200            crash_restarts: VecDeque::new(),
1201            lifetime_restarts: 0,
1202            spawn_generation: 0,
1203            pid: None,
1204            #[cfg(target_os = "macos")]
1205            report_ready: None,
1206            reaped_pid: None,
1207            respawn_pending: false,
1208            coalesced_restart_pending: false,
1209            spawned_at_ms: None,
1210            spawned_from: None,
1211            spawned_file_identity: None,
1212            process_start_time: None,
1213            deliberate_severance: None,
1214            last_exit: None,
1215            drain_disposition_detail: None,
1216            health: ModuleHealthStatus::default(),
1217            in_alternate_slot: false,
1218            draining_to_replace: false,
1219            configuration_updated_since_spawn: false,
1220        }
1221    }
1222}
1223
1224type SharedSnapshot = Arc<Mutex<SupervisorSnapshot>>;
1225
1226type SpawnSubscriberKey = (ConnectionId, u64);
1227
1228#[derive(Debug)]
1229struct SpawnSubscriber {
1230    version: u8,
1231    frames: mpsc::Sender<Frame>,
1232    /// Tells this subscriber's forwarder that it was dropped for lagging, and
1233    /// from which event. The full frame channel cannot carry that news, so it
1234    /// travels beside it; see `SpawnEventFeed::subscribe`.
1235    lagged: Option<oneshot::Sender<SpawnCursor>>,
1236}
1237
1238#[derive(Debug)]
1239struct SpawnEventState {
1240    daemon_incarnation: String,
1241    seq: u64,
1242    capacity: usize,
1243    live: HashMap<String, LiveSpawn>,
1244    generations: HashMap<String, u64>,
1245    events: VecDeque<SpawnEvent>,
1246    subscribers: HashMap<SpawnSubscriberKey, SpawnSubscriber>,
1247}
1248
1249impl Default for SpawnEventState {
1250    fn default() -> Self {
1251        Self {
1252            daemon_incarnation: "unconfigured".to_string(),
1253            seq: 0,
1254            capacity: SPAWN_EVENT_RING_CAPACITY,
1255            live: HashMap::new(),
1256            generations: HashMap::new(),
1257            events: VecDeque::new(),
1258            subscribers: HashMap::new(),
1259        }
1260    }
1261}
1262
1263#[derive(Debug, Clone, Default)]
1264struct SpawnEventFeed(Arc<Mutex<SpawnEventState>>);
1265
1266#[derive(Debug, Clone, PartialEq, Eq)]
1267pub(crate) enum SpawnSubscribeRefusal {
1268    ForeignIncarnation { current: String },
1269    TooOld { oldest: SpawnCursor },
1270    Frame(String),
1271}
1272
1273impl SpawnEventFeed {
1274    fn configure_incarnation(&self, daemon_incarnation: String) {
1275        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1276        state.daemon_incarnation = daemon_incarnation;
1277        state.seq = 0;
1278        state.live.clear();
1279        state.generations.clear();
1280        state.events.clear();
1281        state.subscribers.clear();
1282    }
1283
1284    fn cursor(state: &SpawnEventState) -> SpawnCursor {
1285        SpawnCursor {
1286            daemon_incarnation: state.daemon_incarnation.clone(),
1287            seq: state.seq,
1288        }
1289    }
1290
1291    fn snapshot(&self) -> SpawnSnapshot {
1292        let state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1293        let mut live = state.live.values().cloned().collect::<Vec<_>>();
1294        live.sort_by(|left, right| left.module_id.cmp(&right.module_id));
1295        SpawnSnapshot {
1296            cursor: Self::cursor(&state),
1297            ring_bound: state.capacity as u64,
1298            live,
1299        }
1300    }
1301
1302    fn emit_spawned(&self, module_id: &str, pid: u32, spawned_at_ms: u64) -> u64 {
1303        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1304        let generation = state
1305            .generations
1306            .get(module_id)
1307            .copied()
1308            .unwrap_or(0)
1309            .checked_add(1)
1310            .expect("spawn generation exhausted");
1311        state.generations.insert(module_id.to_string(), generation);
1312        let live = LiveSpawn {
1313            module_id: module_id.to_string(),
1314            spawn_generation: generation,
1315            pid,
1316            spawned_at_ms,
1317        };
1318        state.live.insert(module_id.to_string(), live);
1319        Self::emit_locked(
1320            &mut state,
1321            SpawnEventKind::Spawned,
1322            module_id.to_string(),
1323            generation,
1324            pid,
1325            None,
1326            None,
1327        );
1328        generation
1329    }
1330
1331    fn emit_exited(&self, module_id: &str, exit_code: Option<i32>, exit_signal: Option<i32>) {
1332        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1333        let Some(live) = state.live.remove(module_id) else {
1334            warn!(
1335                module_id,
1336                "terminal record had no live spawn event identity"
1337            );
1338            return;
1339        };
1340        Self::emit_locked(
1341            &mut state,
1342            SpawnEventKind::Exited,
1343            module_id.to_string(),
1344            live.spawn_generation,
1345            live.pid,
1346            exit_code,
1347            exit_signal,
1348        );
1349    }
1350
1351    /// Report the exit of a process that a swap has already replaced.
1352    ///
1353    /// `emit_exited` removes the module's live entry, which after a swap's
1354    /// cutover describes the promoted candidate, not the old process now
1355    /// exiting. This emits the old generation's exit and leaves the live entry
1356    /// alone unless it still names that generation.
1357    fn emit_superseded_exited(
1358        &self,
1359        module_id: &str,
1360        spawn_generation: u64,
1361        pid: u32,
1362        exit_code: Option<i32>,
1363        exit_signal: Option<i32>,
1364    ) {
1365        let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1366        if state
1367            .live
1368            .get(module_id)
1369            .is_some_and(|live| live.spawn_generation == spawn_generation)
1370        {
1371            state.live.remove(module_id);
1372        }
1373        Self::emit_locked(
1374            &mut state,
1375            SpawnEventKind::Exited,
1376            module_id.to_string(),
1377            spawn_generation,
1378            pid,
1379            exit_code,
1380            exit_signal,
1381        );
1382    }
1383
1384    #[allow(clippy::too_many_arguments)]
1385    fn emit_locked(
1386        state: &mut SpawnEventState,
1387        kind: SpawnEventKind,
1388        module_id: String,
1389        spawn_generation: u64,
1390        pid: u32,
1391        exit_code: Option<i32>,
1392        exit_signal: Option<i32>,
1393    ) {
1394        state.seq = state
1395            .seq
1396            .checked_add(1)
1397            .expect("spawn event sequence exhausted");
1398        let event = SpawnEvent {
1399            cursor: Self::cursor(state),
1400            kind,
1401            module_id,
1402            spawn_generation,
1403            pid,
1404            exit_code,
1405            exit_signal,
1406        };
1407        state.events.push_back(event.clone());
1408        while state.events.len() > state.capacity {
1409            state.events.pop_front();
1410        }
1411        let body = match serde_json::to_vec(&event) {
1412            Ok(body) => body,
1413            Err(error) => {
1414                error!(%error, "failed to serialize supervisor spawn event");
1415                return;
1416            }
1417        };
1418        state.subscribers.retain(|(connection_id, corr), subscriber| {
1419            let frame = Frame::build_with_version(
1420                subscriber.version,
1421                FrameType::StreamData,
1422                control_flags(),
1423                0,
1424                0,
1425                *corr,
1426                body.clone(),
1427            );
1428            match frame {
1429                Ok(frame) => {
1430                    if subscriber.frames.try_send(frame).is_ok() {
1431                        true
1432                    } else {
1433                        warn!(connection_id = connection_id.get(), corr, "dropping lagged supervisor spawn subscriber");
1434                        if let Some(lagged) = subscriber.lagged.take() {
1435                            let _ = lagged.send(event.cursor.clone());
1436                        }
1437                        false
1438                    }
1439                }
1440                Err(error) => {
1441                    warn!(connection_id = connection_id.get(), corr, %error, "dropping supervisor spawn subscriber after frame build failure");
1442                    false
1443                }
1444            }
1445        });
1446    }
1447
1448    fn subscribe(
1449        &self,
1450        connection_id: ConnectionId,
1451        corr: u64,
1452        version: u8,
1453        since: Option<SpawnCursor>,
1454        sink: FrameSink,
1455    ) -> Result<(), SpawnSubscribeRefusal> {
1456        let (frames, mut receiver) = mpsc::channel(SPAWN_SUBSCRIBER_BUFFER);
1457        let (lagged, mut lagged_rx) = oneshot::channel::<SpawnCursor>();
1458        {
1459            let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1460            let replay = if let Some(since) = since {
1461                if since.daemon_incarnation != state.daemon_incarnation {
1462                    return Err(SpawnSubscribeRefusal::ForeignIncarnation {
1463                        current: state.daemon_incarnation.clone(),
1464                    });
1465                }
1466                if let Some(oldest) = state.events.front().map(|event| event.cursor.clone()) {
1467                    if since.seq < oldest.seq.saturating_sub(1) {
1468                        return Err(SpawnSubscribeRefusal::TooOld { oldest });
1469                    }
1470                }
1471                state
1472                    .events
1473                    .iter()
1474                    .filter(|event| event.cursor.seq > since.seq)
1475                    .cloned()
1476                    .collect::<Vec<_>>()
1477            } else {
1478                Vec::new()
1479            };
1480            for event in replay {
1481                let body = serde_json::to_vec(&event)
1482                    .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1483                let frame = Frame::build_with_version(
1484                    version,
1485                    FrameType::StreamData,
1486                    control_flags(),
1487                    0,
1488                    0,
1489                    corr,
1490                    body,
1491                )
1492                .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1493                frames
1494                    .try_send(frame)
1495                    .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1496            }
1497            state.subscribers.insert(
1498                (connection_id, corr),
1499                SpawnSubscriber {
1500                    version,
1501                    frames,
1502                    lagged: Some(lagged),
1503                },
1504            );
1505        }
1506        // The lagged terminal is sent here, by the forwarder, rather than by
1507        // the emitter: at the moment of the drop the subscriber's own channel
1508        // is full, and writing to the connection sink directly from the emitter
1509        // would put the Error AHEAD of the events still queued in that channel
1510        // (and the emitter holds the feed lock, so it cannot await the sink).
1511        // Dropping the subscriber drops the only sender, so `recv` drains every
1512        // queued event and then returns `None`; only then is the Error sent, so
1513        // the client sees each event it can keep, then the reason it was cut.
1514        // Cancel and connection removal drop the oneshot unsent, so they end
1515        // the stream with no Error.
1516        tokio::spawn(async move {
1517            while let Some(frame) = receiver.recv().await {
1518                if sink.send(frame).await.is_err() {
1519                    return;
1520                }
1521            }
1522            let Ok(first_undelivered) = lagged_rx.try_recv() else {
1523                return;
1524            };
1525            match spawn_subscriber_lagged_frame(version, corr, first_undelivered) {
1526                Ok(frame) => {
1527                    let _ = sink.send(frame).await;
1528                }
1529                Err(error) => {
1530                    error!(%error, corr, "failed to build lagged spawn subscriber terminal frame");
1531                }
1532            }
1533        });
1534        Ok(())
1535    }
1536
1537    fn cancel(&self, connection_id: ConnectionId, corr: u64) -> bool {
1538        let Some(subscriber) = self
1539            .0
1540            .lock()
1541            .unwrap_or_else(|p| p.into_inner())
1542            .subscribers
1543            .remove(&(connection_id, corr))
1544        else {
1545            return false;
1546        };
1547        if let Ok(frame) = Frame::build_with_version(
1548            subscriber.version,
1549            FrameType::StreamEnd,
1550            control_flags(),
1551            0,
1552            0,
1553            corr,
1554            Vec::new(),
1555        ) {
1556            tokio::spawn(async move {
1557                let _ = subscriber.frames.send(frame).await;
1558            });
1559        }
1560        true
1561    }
1562
1563    fn remove_connection(&self, connection_id: ConnectionId) {
1564        self.0
1565            .lock()
1566            .unwrap_or_else(|p| p.into_inner())
1567            .subscribers
1568            .retain(|(subscriber_connection, _), _| *subscriber_connection != connection_id);
1569    }
1570
1571    #[cfg(any(test, feature = "test-support"))]
1572    fn set_capacity(&self, capacity: usize) {
1573        self.0.lock().unwrap_or_else(|p| p.into_inner()).capacity = capacity;
1574    }
1575
1576    #[cfg(any(test, feature = "test-support"))]
1577    fn subscriber_count(&self) -> usize {
1578        self.0
1579            .lock()
1580            .unwrap_or_else(|p| p.into_inner())
1581            .subscribers
1582            .len()
1583    }
1584}
1585
1586/// Narrow process-liveness signal published by supervisors and consumed by passive liveness polls.
1587/// The terminal Error a lagged spawn subscriber receives after its queued events.
1588fn spawn_subscriber_lagged_frame(
1589    version: u8,
1590    corr: u64,
1591    first_undelivered: SpawnCursor,
1592) -> Result<Frame, String> {
1593    let body = serde_json::to_vec(&subc_protocol::ErrorBody {
1594        code: SPAWN_SUBSCRIBER_LAGGED_CODE.to_string(),
1595        message: "spawn subscriber fell behind and was dropped; resubscribe from the last cursor received"
1596            .to_string(),
1597        detail: Some(serde_json::json!({
1598            "first_undelivered_cursor": first_undelivered
1599        })),
1600    })
1601    .map_err(|error| error.to_string())?;
1602    Frame::build_with_version(version, FrameType::Error, control_flags(), 0, 0, corr, body)
1603        .map_err(|error| error.to_string())
1604}
1605
1606pub trait ModuleProcessLiveness: Send + Sync {
1607    fn process_live(&self, module_id: &str) -> Option<bool>;
1608
1609    /// Whether the supervisor is replacing this module's process right now: an
1610    /// operator restart or reload, a health restart, or a crash respawn whose
1611    /// backoff is running. A module in that state is not live, but a consumer
1612    /// refused now should retry shortly rather than treat the target as gone.
1613    /// Stopped, failed, and disabled modules are not replacing.
1614    fn process_replacing(&self, _module_id: &str) -> bool {
1615        false
1616    }
1617}
1618
1619/// Shared process-liveness registry keyed by supervised `module_id`.
1620#[derive(Debug, Clone, Default)]
1621pub struct SupervisorProcessLiveness {
1622    snapshots: Arc<Mutex<HashMap<String, SharedSnapshot>>>,
1623}
1624
1625impl SupervisorProcessLiveness {
1626    pub fn new() -> Self {
1627        Self::default()
1628    }
1629
1630    fn track(&self, module_id: String, snapshot: SharedSnapshot) {
1631        let mut snapshots = self
1632            .snapshots
1633            .lock()
1634            .unwrap_or_else(|poisoned| poisoned.into_inner());
1635        snapshots.insert(module_id, snapshot);
1636    }
1637
1638    fn untrack_if_current(&self, module_id: &str, snapshot: &SharedSnapshot) {
1639        let mut snapshots = self
1640            .snapshots
1641            .lock()
1642            .unwrap_or_else(|poisoned| poisoned.into_inner());
1643        let is_current = snapshots
1644            .get(module_id)
1645            .map(|tracked| Arc::ptr_eq(tracked, snapshot))
1646            .unwrap_or(false);
1647        if is_current {
1648            snapshots.remove(module_id);
1649        }
1650    }
1651}
1652
1653impl ModuleProcessLiveness for SupervisorProcessLiveness {
1654    fn process_live(&self, module_id: &str) -> Option<bool> {
1655        let snapshot = {
1656            let snapshots = self
1657                .snapshots
1658                .lock()
1659                .unwrap_or_else(|poisoned| poisoned.into_inner());
1660            snapshots.get(module_id).cloned()
1661        }?;
1662        let snapshot = snapshot
1663            .lock()
1664            .unwrap_or_else(|poisoned| poisoned.into_inner());
1665        Some(snapshot.state == ModuleState::Running && snapshot.process_alive)
1666    }
1667
1668    fn process_replacing(&self, module_id: &str) -> bool {
1669        let Some(snapshot) = self
1670            .snapshots
1671            .lock()
1672            .unwrap_or_else(|poisoned| poisoned.into_inner())
1673            .get(module_id)
1674            .cloned()
1675        else {
1676            return false;
1677        };
1678        let snapshot = snapshot
1679            .lock()
1680            .unwrap_or_else(|poisoned| poisoned.into_inner());
1681        snapshot.enabled
1682            && match snapshot.state {
1683                ModuleState::Restarting => true,
1684                ModuleState::Draining => snapshot.draining_to_replace,
1685                ModuleState::Starting
1686                | ModuleState::Running
1687                | ModuleState::Unresponsive
1688                | ModuleState::Stopped
1689                | ModuleState::Failed
1690                | ModuleState::Disabled => false,
1691            }
1692    }
1693}
1694
1695#[cfg(test)]
1696#[derive(Debug, Default)]
1697struct ReloadExitRecordGate {
1698    reached: tokio::sync::Notify,
1699    resume: tokio::sync::Notify,
1700}
1701
1702#[derive(Debug, Clone, Copy)]
1703enum RespawnKind {
1704    Spawn,
1705    Reload,
1706}
1707
1708#[derive(Debug, Clone, Copy)]
1709struct PendingRespawn {
1710    deadline: Instant,
1711    kind: RespawnKind,
1712}
1713
1714type ReloadReply = oneshot::Sender<Result<(), SuperviseError>>;
1715
1716#[derive(Debug, Clone)]
1717struct SupervisorRuntimeConfig {
1718    /// Restart operations hand their backoff to the loop, which keeps serving commands.
1719    scheduled_respawn: Arc<Mutex<Option<PendingRespawn>>>,
1720    /// A reload acknowledges completion only after its replacement registers.
1721    deferred_reload_reply: Arc<Mutex<Option<ReloadReply>>>,
1722    restart_policy: RestartPolicy,
1723    /// This module's RESOLVED drain budget: per-module config when present,
1724    /// else `default_drain_timeout`.
1725    drain_timeout: Duration,
1726    /// Shared with the status handle so the attested value changes atomically
1727    /// when a rescan updates the running drain policy.
1728    effective_drain_timeout: Arc<Mutex<Duration>>,
1729    /// The supervisor-wide fallback, kept so a configuration update that
1730    /// REMOVES the per-module override can re-resolve to it.
1731    default_drain_timeout: Duration,
1732    health: HealthConfig,
1733    connection_file_path: Option<PathBuf>,
1734    capture_logs_dir: Option<PathBuf>,
1735    forwarding: Option<Arc<ForwardingTable>>,
1736    /// The shared handle, so every spawn path (initial, restart, reload) records the
1737    /// reserved-module launch nonce the HELLO verifier checks against.
1738    supervisor_handle: Option<SupervisorHandle>,
1739    /// This module's stderr tail, shared with the [`SupervisedModule`] that answers
1740    /// status queries.
1741    ///
1742    /// One ring per module, held across every respawn. The lines explaining an exit
1743    /// are written BEFORE that exit, so a ring recreated per process would be empty
1744    /// exactly when it is asked for.
1745    stderr_ring: Arc<Mutex<StderrRing>>,
1746    terminal_ring: Arc<Mutex<TerminalRing>>,
1747    spawn_events: SpawnEventFeed,
1748    child_roster: ChildRoster,
1749    #[cfg(target_os = "linux")]
1750    cgroup_placement: Option<subc_cgroup::Placement>,
1751    #[cfg(test)]
1752    test_seed_stale_facts_before_enable_spawn: bool,
1753    #[cfg(test)]
1754    test_reload_exit_record_gate: Option<Arc<ReloadExitRecordGate>>,
1755}
1756
1757#[derive(Debug, Clone, PartialEq, Eq)]
1758struct SupervisedConfiguration {
1759    spec: ModuleSpec,
1760    health: HealthConfig,
1761}
1762
1763/// Shared daemon lookup table for supervised module handles.
1764///
1765/// Shared by clone between the [`Supervisor`] (which spawns processes) and the
1766/// channel-0 control handler (which verifies HELLOs and consumer route opens), so
1767/// launch nonces recorded at spawn are checked by the same daemon instance.
1768#[derive(Debug, Clone, Default)]
1769pub struct SupervisorHandle {
1770    modules: Arc<Mutex<HashMap<String, SupervisedModule>>>,
1771    /// Module ids the supervisor has taken on. An id is added BEFORE the
1772    /// module's first process is spawned and removed only when the module
1773    /// leaves the roster (see [`Self::retire`]), so it is always a superset of
1774    /// the keys of `modules`.
1775    ///
1776    /// `modules` cannot answer "is this module configured?" on its own: a
1777    /// [`SupervisedModule`] only exists once its process has been spawned, and
1778    /// a fast child can connect, register, sync its scopes and ask about them
1779    /// before the supervisor has inserted it. Answering "not configured" in that
1780    /// gap makes scope admission refuse with the terminal "will never sync"
1781    /// instead of the retryable "has not synced yet".
1782    configured_ids: Arc<Mutex<HashSet<String>>>,
1783    spawn_events: SpawnEventFeed,
1784    /// The current expected launch nonce for each reserved module_id. Set when the
1785    /// supervisor spawns the reserved module; checked when a HELLO claims that id. A
1786    /// non-reserved module never has an entry here and is never nonce-checked.
1787    /// Reserved module ids and the nonce that authorizes their next HELLO.
1788    /// `None` means RESERVED WITH NO LEGITIMATE HOLDER — a reserved module that
1789    /// has never been spawned (e.g. configured `enabled: false`) — and refuses
1790    /// every HELLO. Before this was expressible, a reserved-but-never-spawned id
1791    /// had NO entry and admitted anyone: the reservation protected the nonce
1792    /// holder, not the NAME (found live by CKCRED's canary probe registering
1793    /// against a reserved scratch id).
1794    reserved_nonces: Arc<Mutex<HashMap<String, Option<String>>>>,
1795    /// Module ids removed by an executed rescan and the unix-millisecond removal time.
1796    ///
1797    /// This is deliberately in-memory only: subc is state-free across daemon
1798    /// restarts, and the tombstone only explains the hours-after-removal window
1799    /// while this executing daemon is still alive. Do not persist it in a store.
1800    removal_tombstones: Arc<Mutex<HashMap<String, u64>>>,
1801    /// The current launch nonce for every supervised spawn. This is separate from
1802    /// reserved_nonces because consumer route.open attestation applies to all spawned
1803    /// modules, while HELLO id-squatting protection remains opt-in via `reserved`.
1804    spawn_nonces: Arc<Mutex<HashMap<String, String>>>,
1805    /// Reserved namespace prefixes mapped to the supervised owner module whose
1806    /// current spawn nonce authorizes HELLO claims below the prefix.
1807    ///
1808    /// Per §2.6 this is not a same-user security barrier: a same-user process can
1809    /// read the key file and launch nonce env. Like exact reserved ids, it prevents
1810    /// accidental collisions and lower-trust processes from squatting protected
1811    /// namespaces.
1812    reserved_prefix_owners: Arc<Mutex<HashMap<String, String>>>,
1813    /// Blue/green swaps in progress, by module id. An entry exists from just
1814    /// before the candidate process is spawned until the swap has failed, or
1815    /// has cut over and the old process is gone. While it exists, HELLO for the
1816    /// id is gated on the swap token (see [`Self::swap_hello_admission`]) and
1817    /// consumer attestation accepts both processes' nonces.
1818    swaps: Arc<Mutex<HashMap<String, OpenSwap>>>,
1819    /// Told when a swap promotes its candidate; see [`SwapPromotionObserver`].
1820    promotion_observer: PromotionObserverSlot,
1821    /// Serializes module-set reconciliation with operator lifecycle commands. Without
1822    /// this daemon-wide ordering, a rescan could retire or update a module while a
1823    /// concurrent reload still held its old handle and launch specification.
1824    operation_lock: Arc<AsyncMutex<()>>,
1825}
1826
1827/// Told when a swap has promoted its candidate to be the module's active
1828/// registration.
1829///
1830/// An ordinary HELLO runs the control plane's registration side effects (the
1831/// capability cache, the deny census, the requirement recompute) as it
1832/// registers. A swap candidate's HELLO does not, because it is not routable;
1833/// promotion is when those must run instead, and promotion happens in the
1834/// supervisor, which has no other way into the control handler.
1835pub(crate) trait SwapPromotionObserver: Send + Sync {
1836    fn swap_promoted(&self, registration: &crate::registry::ModuleRegistration);
1837}
1838
1839/// The installed [`SwapPromotionObserver`], held weakly: the observer (the
1840/// control handler) owns this handle, so a strong reference back would be a
1841/// cycle that keeps both alive.
1842#[derive(Clone, Default)]
1843struct PromotionObserverSlot(Arc<Mutex<Option<std::sync::Weak<dyn SwapPromotionObserver>>>>);
1844
1845impl fmt::Debug for PromotionObserverSlot {
1846    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1847        f.write_str("PromotionObserverSlot")
1848    }
1849}
1850
1851/// The nonces of one open swap.
1852#[derive(Debug, Clone)]
1853struct OpenSwap {
1854    /// The launch nonce minted for the candidate process. It is the swap
1855    /// token: the only thing that admits a HELLO into the candidate slot.
1856    candidate_nonce: String,
1857    /// The incumbent's launch nonce, captured when the swap opened. It is kept
1858    /// here because cutover moves the module's recorded spawn nonce to the
1859    /// candidate while the incumbent is still draining and its consumers are
1860    /// still attesting with this one.
1861    incumbent_nonce: Option<String>,
1862    /// Set once a HELLO has been admitted with the swap token, so the token
1863    /// admits one registration and cannot be replayed after cutover empties
1864    /// the candidate slot.
1865    candidate_admitted: bool,
1866}
1867
1868/// What the swap gate says about a HELLO. See
1869/// [`SupervisorHandle::swap_hello_admission`].
1870#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1871pub(crate) enum SwapHelloAdmission {
1872    /// No swap is open for the id (or the HELLO carries the incumbent's own
1873    /// nonce); the ordinary gates decide.
1874    NotSwapping,
1875    /// The HELLO carries the swap token: register it into the candidate slot.
1876    Candidate,
1877    /// A swap is open and the HELLO carries a nonce the supervisor did not
1878    /// mint for this id, no nonce, or a token already used.
1879    Refused,
1880}
1881
1882#[derive(Debug, Clone, PartialEq, Eq)]
1883pub(crate) enum ReservedHelloRejection {
1884    Exact {
1885        module_id: String,
1886    },
1887    Prefix {
1888        prefix: String,
1889        owner_module_id: String,
1890    },
1891}
1892
1893impl SupervisorHandle {
1894    pub fn new() -> Self {
1895        Self::default()
1896    }
1897
1898    pub(crate) fn spawn_snapshot(&self) -> SpawnSnapshot {
1899        self.spawn_events.snapshot()
1900    }
1901
1902    pub(crate) fn subscribe_spawns(
1903        &self,
1904        connection_id: ConnectionId,
1905        corr: u64,
1906        version: u8,
1907        since: Option<SpawnCursor>,
1908        sink: FrameSink,
1909    ) -> Result<(), SpawnSubscribeRefusal> {
1910        self.spawn_events
1911            .subscribe(connection_id, corr, version, since, sink)
1912    }
1913
1914    pub(crate) fn cancel_spawn_subscription(&self, connection_id: ConnectionId, corr: u64) -> bool {
1915        self.spawn_events.cancel(connection_id, corr)
1916    }
1917
1918    pub(crate) fn remove_spawn_subscribers(&self, connection_id: ConnectionId) {
1919        self.spawn_events.remove_connection(connection_id);
1920    }
1921
1922    #[cfg(any(test, feature = "test-support"))]
1923    pub fn set_spawn_event_capacity_for_test(&self, capacity: usize) {
1924        assert!(capacity > 0, "spawn event capacity must be non-zero");
1925        self.spawn_events.set_capacity(capacity);
1926    }
1927
1928    #[cfg(any(test, feature = "test-support"))]
1929    pub fn spawn_subscriber_count_for_test(&self) -> usize {
1930        self.spawn_events.subscriber_count()
1931    }
1932
1933    /// Record the launch nonce from a supervised spawn, replacing any prior nonce so
1934    /// a respawn invalidates stale consumer identities.
1935    pub fn set_spawn_nonce(&self, module_id: &str, nonce: String) {
1936        self.spawn_nonces
1937            .lock()
1938            .unwrap_or_else(|poisoned| poisoned.into_inner())
1939            .insert(module_id.to_string(), nonce);
1940    }
1941
1942    /// Record the launch nonce expected from the next HELLO for a reserved module,
1943    /// replacing any prior nonce (a respawn invalidates the previous one).
1944    pub fn set_reserved_nonce(&self, module_id: &str, nonce: String) {
1945        self.reserved_nonces
1946            .lock()
1947            .unwrap_or_else(|poisoned| poisoned.into_inner())
1948            .insert(module_id.to_string(), Some(nonce));
1949    }
1950
1951    /// Record namespace prefixes owned by a supervised module.
1952    pub fn set_reserved_prefixes(&self, owner_module_id: &str, prefixes: &[String]) {
1953        let mut owners = self
1954            .reserved_prefix_owners
1955            .lock()
1956            .unwrap_or_else(|poisoned| poisoned.into_inner());
1957        owners.retain(|_, owner| owner != owner_module_id);
1958        for prefix in prefixes {
1959            owners.insert(prefix.clone(), owner_module_id.to_string());
1960        }
1961    }
1962
1963    /// The launch nonce most recently minted for a module's spawn, if any.
1964    #[cfg(test)]
1965    pub(crate) fn spawn_nonce(&self, module_id: &str) -> Option<String> {
1966        self.spawn_nonces
1967            .lock()
1968            .unwrap_or_else(|poisoned| poisoned.into_inner())
1969            .get(module_id)
1970            .cloned()
1971    }
1972
1973    fn apply_identity_configuration(&self, spec: &ModuleSpec) {
1974        self.set_reserved_prefixes(&spec.module_id, &spec.reserved_prefixes);
1975        let spawn_nonce = self
1976            .spawn_nonces
1977            .lock()
1978            .unwrap_or_else(|poisoned| poisoned.into_inner())
1979            .get(&spec.module_id)
1980            .cloned();
1981        let mut reserved_nonces = self
1982            .reserved_nonces
1983            .lock()
1984            .unwrap_or_else(|poisoned| poisoned.into_inner());
1985        if spec.reserved {
1986            // `None` (no spawn nonce minted) is INSERTED, not skipped: a
1987            // reserved name whose module has never spawned has no legitimate
1988            // holder, and the entry's absence is what used to leave the name
1989            // open to the first claimant.
1990            reserved_nonces.insert(spec.module_id.clone(), spawn_nonce);
1991        }
1992        drop(reserved_nonces);
1993        // A later unreserved declaration must not silently unreserve an id that
1994        // was retained after its reserved configuration was removed. The explicit
1995        // release ceremony is the only operation that retires that gate.
1996        self.removal_tombstones
1997            .lock()
1998            .unwrap_or_else(|poisoned| poisoned.into_inner())
1999            .remove(&spec.module_id);
2000    }
2001
2002    /// Whether a HELLO claiming `module_id` is authorized. An exact reserved id is
2003    /// authorized only by its expected nonce; otherwise a matching reserved prefix
2004    /// is authorized by the owner module's current spawn nonce. Non-reserved ids
2005    /// with no matching prefix are always authorized.
2006    pub fn reserved_hello_authorized(&self, module_id: &str, presented: Option<&str>) -> bool {
2007        self.reserved_hello_rejection(module_id, presented)
2008            .is_none()
2009    }
2010
2011    pub(crate) fn reserved_hello_rejection(
2012        &self,
2013        module_id: &str,
2014        presented: Option<&str>,
2015    ) -> Option<ReservedHelloRejection> {
2016        let nonces = self
2017            .reserved_nonces
2018            .lock()
2019            .unwrap_or_else(|poisoned| poisoned.into_inner());
2020        if let Some(expected) = nonces.get(module_id) {
2021            // `None` = reserved with no legitimate holder: refuse every
2022            // presentation, because no process can hold a nonce that was never
2023            // minted. Only a real minted nonce admits, in constant time.
2024            let authorized = match expected {
2025                Some(expected) => {
2026                    presented.is_some_and(|p| constant_time_eq(expected.as_bytes(), p.as_bytes()))
2027                }
2028                None => false,
2029            };
2030            if authorized {
2031                return None;
2032            }
2033            return Some(ReservedHelloRejection::Exact {
2034                module_id: module_id.to_string(),
2035            });
2036        }
2037        drop(nonces);
2038
2039        let matched_prefix = self
2040            .reserved_prefix_owners
2041            .lock()
2042            .unwrap_or_else(|poisoned| poisoned.into_inner())
2043            .iter()
2044            .filter(|(prefix, _)| module_id.starts_with(prefix.as_str()))
2045            .max_by_key(|(prefix, _)| prefix.len())
2046            .map(|(prefix, owner)| (prefix.clone(), owner.clone()));
2047        let (prefix, owner_module_id) = matched_prefix?;
2048
2049        let authorized = presented.is_some_and(|presented| {
2050            self.spawn_nonces
2051                .lock()
2052                .unwrap_or_else(|poisoned| poisoned.into_inner())
2053                .get(&owner_module_id)
2054                .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()))
2055                // While the owner is being swapped, children started by
2056                // either of its two processes hold that process's nonce.
2057                || self.swap_nonce_matches(&owner_module_id, presented)
2058        });
2059        if authorized {
2060            None
2061        } else {
2062            Some(ReservedHelloRejection::Prefix {
2063                prefix,
2064                owner_module_id,
2065            })
2066        }
2067    }
2068
2069    /// Whether a consumer connection proved it came from a daemon-spawned module.
2070    ///
2071    /// Absence of an expected spawn nonce is a hard failure: consumer_identity is
2072    /// accepted only for module ids the supervisor has spawned.
2073    pub fn spawned_consumer_authorized(&self, module_id: &str, presented: &str) -> bool {
2074        if presented.is_empty() {
2075            return false;
2076        }
2077        let nonces = self
2078            .spawn_nonces
2079            .lock()
2080            .unwrap_or_else(|poisoned| poisoned.into_inner());
2081        let current = nonces
2082            .get(module_id)
2083            .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()));
2084        drop(nonces);
2085        // During a swap two processes of the module are alive, and a consumer
2086        // started by either one presents that process's nonce. Accepting only
2087        // the recorded one would fail the incumbent's consumers for the whole
2088        // overlap once cutover moves the record to the candidate.
2089        current || self.swap_nonce_matches(module_id, presented)
2090    }
2091
2092    /// Whether `presented` is either nonce of an open swap for `module_id`.
2093    fn swap_nonce_matches(&self, module_id: &str, presented: &str) -> bool {
2094        let swaps = self
2095            .swaps
2096            .lock()
2097            .unwrap_or_else(|poisoned| poisoned.into_inner());
2098        swaps.get(module_id).is_some_and(|swap| {
2099            constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes())
2100                || swap.incumbent_nonce.as_deref().is_some_and(|incumbent| {
2101                    constant_time_eq(incumbent.as_bytes(), presented.as_bytes())
2102                })
2103        })
2104    }
2105
2106    /// Open a swap for `module_id` with the candidate's freshly minted nonce.
2107    /// Called before the candidate process exists.
2108    pub(crate) fn open_swap(&self, module_id: &str, candidate_nonce: String) {
2109        let incumbent_nonce = self
2110            .spawn_nonces
2111            .lock()
2112            .unwrap_or_else(|poisoned| poisoned.into_inner())
2113            .get(module_id)
2114            .cloned();
2115        self.swaps
2116            .lock()
2117            .unwrap_or_else(|poisoned| poisoned.into_inner())
2118            .insert(
2119                module_id.to_string(),
2120                OpenSwap {
2121                    candidate_nonce,
2122                    incumbent_nonce,
2123                    candidate_admitted: false,
2124                },
2125            );
2126    }
2127
2128    /// Close the swap for `module_id`, releasing whichever nonce is no longer
2129    /// the module's recorded one.
2130    pub(crate) fn close_swap(&self, module_id: &str) {
2131        self.swaps
2132            .lock()
2133            .unwrap_or_else(|poisoned| poisoned.into_inner())
2134            .remove(module_id);
2135    }
2136
2137    /// Install the observer told about swap promotions, replacing any earlier
2138    /// one.
2139    pub(crate) fn set_swap_promotion_observer(
2140        &self,
2141        observer: std::sync::Weak<dyn SwapPromotionObserver>,
2142    ) {
2143        *self
2144            .promotion_observer
2145            .0
2146            .lock()
2147            .unwrap_or_else(|poisoned| poisoned.into_inner()) = Some(observer);
2148    }
2149
2150    /// Tell the installed observer, if it is still alive, that a swap promoted
2151    /// `registration`.
2152    fn notify_swap_promoted(&self, registration: &crate::registry::ModuleRegistration) {
2153        let observer = self
2154            .promotion_observer
2155            .0
2156            .lock()
2157            .unwrap_or_else(|poisoned| poisoned.into_inner())
2158            .as_ref()
2159            .and_then(std::sync::Weak::upgrade);
2160        if let Some(observer) = observer {
2161            observer.swap_promoted(registration);
2162        }
2163    }
2164
2165    /// Whether a swap is open for `module_id`.
2166    pub(crate) fn swap_open(&self, module_id: &str) -> bool {
2167        self.swaps
2168            .lock()
2169            .unwrap_or_else(|poisoned| poisoned.into_inner())
2170            .contains_key(module_id)
2171    }
2172
2173    /// Make the candidate's nonce the module's recorded spawn nonce, as a plain
2174    /// respawn would, once cutover has made the candidate the module's process.
2175    /// The swap stays open so the incumbent's nonce keeps attesting until the
2176    /// incumbent has drained and exited.
2177    fn promote_swap_nonce(&self, module_id: &str, reserved: bool) {
2178        let candidate_nonce = self
2179            .swaps
2180            .lock()
2181            .unwrap_or_else(|poisoned| poisoned.into_inner())
2182            .get(module_id)
2183            .map(|swap| swap.candidate_nonce.clone());
2184        let Some(nonce) = candidate_nonce else {
2185            return;
2186        };
2187        self.set_spawn_nonce(module_id, nonce.clone());
2188        if reserved {
2189            self.set_reserved_nonce(module_id, nonce);
2190        }
2191    }
2192
2193    /// The swap gate for a HELLO claiming `module_id`.
2194    ///
2195    /// This runs BEFORE the reserved-module gate. A reserved module's candidate
2196    /// presents the candidate nonce, which the reserved gate (holding the
2197    /// incumbent's nonce) would refuse as `reserved_module` before swap
2198    /// admission was ever reached. And it applies to unreserved ids too: for an
2199    /// unreserved id the only thing that ever stopped a second process claiming
2200    /// a live id was the `duplicate_module_id` refusal, which is exactly the
2201    /// refusal a swap lifts for its candidate.
2202    ///
2203    /// The incumbent's own nonce falls through to the ordinary gates, which
2204    /// treat it as they always have (a live incumbent is refused as a
2205    /// duplicate). Anything else while a swap is open is refused, including an
2206    /// absent nonce.
2207    pub(crate) fn swap_hello_admission(
2208        &self,
2209        module_id: &str,
2210        presented: Option<&str>,
2211    ) -> SwapHelloAdmission {
2212        let swaps = self
2213            .swaps
2214            .lock()
2215            .unwrap_or_else(|poisoned| poisoned.into_inner());
2216        let Some(swap) = swaps.get(module_id) else {
2217            return SwapHelloAdmission::NotSwapping;
2218        };
2219        let Some(presented) = presented else {
2220            return SwapHelloAdmission::Refused;
2221        };
2222        if constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes()) {
2223            return if swap.candidate_admitted {
2224                SwapHelloAdmission::Refused
2225            } else {
2226                SwapHelloAdmission::Candidate
2227            };
2228        }
2229        if swap
2230            .incumbent_nonce
2231            .as_deref()
2232            .is_some_and(|incumbent| constant_time_eq(incumbent.as_bytes(), presented.as_bytes()))
2233        {
2234            return SwapHelloAdmission::NotSwapping;
2235        }
2236        SwapHelloAdmission::Refused
2237    }
2238
2239    /// Record that the swap token has registered a candidate, so it admits no
2240    /// second HELLO.
2241    pub(crate) fn mark_swap_candidate_admitted(&self, module_id: &str) {
2242        if let Some(swap) = self
2243            .swaps
2244            .lock()
2245            .unwrap_or_else(|poisoned| poisoned.into_inner())
2246            .get_mut(module_id)
2247        {
2248            swap.candidate_admitted = true;
2249        }
2250    }
2251
2252    /// Test/support lookup for the current launch nonce of a supervised spawn.
2253    pub fn spawn_launch_nonce_for(&self, module_id: &str) -> Option<String> {
2254        self.spawn_nonces
2255            .lock()
2256            .unwrap_or_else(|poisoned| poisoned.into_inner())
2257            .get(module_id)
2258            .cloned()
2259    }
2260
2261    /// Test/support lookup for the HELLO-gating nonce of a reserved module.
2262    pub fn reserved_launch_nonce_for(&self, module_id: &str) -> Option<String> {
2263        self.reserved_nonces
2264            .lock()
2265            .unwrap_or_else(|poisoned| poisoned.into_inner())
2266            .get(module_id)
2267            .cloned()
2268            .flatten()
2269    }
2270
2271    pub fn insert(&self, module: SupervisedModule) -> Option<SupervisedModule> {
2272        // Normally already marked before the process was spawned; marking here
2273        // too keeps `configured_ids` a superset of the roster for any caller
2274        // that inserts a module directly.
2275        self.mark_configured(module.module_id());
2276        let mut modules = self
2277            .modules
2278            .lock()
2279            .unwrap_or_else(|poisoned| poisoned.into_inner());
2280        modules.insert(module.module_id().to_string(), module)
2281    }
2282
2283    /// Record that the supervisor has taken on `module_id`. Called before the
2284    /// module's first process is spawned, so that by the time that process can
2285    /// register, [`Self::is_configured`] already answers true.
2286    fn mark_configured(&self, module_id: &str) {
2287        self.configured_ids
2288            .lock()
2289            .unwrap_or_else(|poisoned| poisoned.into_inner())
2290            .insert(module_id.to_string());
2291    }
2292
2293    /// Undo [`Self::mark_configured`] for a module whose first spawn failed
2294    /// before it was ever put on the roster. A module already on the roster
2295    /// keeps its mark: only [`Self::retire`] takes a rostered module off.
2296    fn unmark_configured_unless_rostered(&self, module_id: &str) {
2297        let modules = self
2298            .modules
2299            .lock()
2300            .unwrap_or_else(|poisoned| poisoned.into_inner());
2301        if !modules.contains_key(module_id) {
2302            self.configured_ids
2303                .lock()
2304                .unwrap_or_else(|poisoned| poisoned.into_inner())
2305                .remove(module_id);
2306        }
2307    }
2308
2309    /// Whether `module_id` is a module this daemon supervises: on the roster,
2310    /// or about to be (its process is being spawned right now).
2311    ///
2312    /// This, not `get(..).is_some()`, is what "the owner is configured" means
2313    /// for scopes: a supervised module's process can register and sync before
2314    /// [`Self::get`] can return it, and in that window it is still a module
2315    /// that will sync, not one that never will.
2316    pub(crate) fn is_configured(&self, module_id: &str) -> bool {
2317        self.configured_ids
2318            .lock()
2319            .unwrap_or_else(|poisoned| poisoned.into_inner())
2320            .contains(module_id)
2321    }
2322
2323    pub fn get(&self, module_id: &str) -> Option<SupervisedModule> {
2324        let modules = self
2325            .modules
2326            .lock()
2327            .unwrap_or_else(|poisoned| poisoned.into_inner());
2328        modules.get(module_id).cloned()
2329    }
2330
2331    pub(crate) fn record_late_health_answer(
2332        &self,
2333        module_id: &str,
2334        latency_ms: u64,
2335    ) -> Result<bool, SuperviseError> {
2336        let Some(module) = self.get(module_id) else {
2337            return Ok(false);
2338        };
2339        update_snapshot(&module.inner.snapshot, Some(module_id), |state| {
2340            state.health.late_answer_count = state.health.late_answer_count.saturating_add(1);
2341            state.health.last_late_answer_latency_ms = Some(latency_ms);
2342            // A late answer is an answer: the module served the probe, just past
2343            // the deadline. Leaving the miss streak in place while logging
2344            // "proves the module is alive" is how a CPU-starved module that
2345            // answers every probe a few seconds late still marches to the
2346            // threshold and gets killed — the exact kill class `NoAnswer` is
2347            // excluded from `is_proof_of_death` to prevent. Slow-but-answering
2348            // is degradation, and degradation reports; it does not restart.
2349            state.health.consecutive_failures = 0;
2350        })?;
2351        Ok(true)
2352    }
2353
2354    /// Arm the one-shot marker for the module process that this caller
2355    /// deliberately initiated severance against. Generic connection teardown
2356    /// must not call this:
2357    /// a surviving process would otherwise retain an exemption for a later
2358    /// genuine crash.
2359    pub fn record_deliberate_severance(&self, module_id: &str) -> Result<bool, SuperviseError> {
2360        let Some(module) = self.get(module_id) else {
2361            return Ok(false);
2362        };
2363        let snapshot = lock_snapshot(&module.inner.snapshot)?;
2364        let Some((pid, start_time)) = snapshot.pid.zip(snapshot.process_start_time) else {
2365            return Ok(false);
2366        };
2367        drop(snapshot);
2368        module.record_deliberate_severance(ProcessIdentity { pid, start_time })
2369    }
2370
2371    pub fn list(&self) -> Vec<SupervisedModule> {
2372        let modules = self
2373            .modules
2374            .lock()
2375            .unwrap_or_else(|poisoned| poisoned.into_inner());
2376        let mut modules = modules.values().cloned().collect::<Vec<_>>();
2377        modules.sort_by(|left, right| left.module_id().cmp(right.module_id()));
2378        modules
2379    }
2380
2381    pub(crate) fn retire(&self, module_id: &str) -> Option<SupervisedModule> {
2382        self.spawn_nonces
2383            .lock()
2384            .unwrap_or_else(|poisoned| poisoned.into_inner())
2385            .remove(module_id);
2386        self.close_swap(module_id);
2387        let mut reserved_nonces = self
2388            .reserved_nonces
2389            .lock()
2390            .unwrap_or_else(|poisoned| poisoned.into_inner());
2391        if reserved_nonces.contains_key(module_id) {
2392            // The old nonce must die with the removed process, but the exact-id
2393            // gate remains until an operator explicitly releases it.
2394            reserved_nonces.insert(module_id.to_string(), None);
2395        }
2396        drop(reserved_nonces);
2397        self.reserved_prefix_owners
2398            .lock()
2399            .unwrap_or_else(|poisoned| poisoned.into_inner())
2400            .retain(|_, owner| owner != module_id);
2401        let removed = self
2402            .modules
2403            .lock()
2404            .unwrap_or_else(|poisoned| poisoned.into_inner())
2405            .remove(module_id);
2406        self.configured_ids
2407            .lock()
2408            .unwrap_or_else(|poisoned| poisoned.into_inner())
2409            .remove(module_id);
2410        removed
2411    }
2412
2413    /// Remember a module removed by a non-preview rescan so route.open can
2414    /// distinguish that intentional removal from an unknown id.
2415    pub(crate) fn record_rescan_removal(&self, module_id: &str) {
2416        self.removal_tombstones
2417            .lock()
2418            .unwrap_or_else(|poisoned| poisoned.into_inner())
2419            .insert(module_id.to_string(), unix_ms_now());
2420    }
2421
2422    /// Return how long ago a rescan removed this module in milliseconds.
2423    pub(crate) fn removal_tombstone_age_ms(&self, module_id: &str) -> Option<u64> {
2424        self.removal_tombstones
2425            .lock()
2426            .unwrap_or_else(|poisoned| poisoned.into_inner())
2427            .get(module_id)
2428            .copied()
2429            .map(|removed_at_ms| unix_ms_now().saturating_sub(removed_at_ms))
2430    }
2431
2432    /// Retire a reserved-id gate only after its module has left supervision.
2433    ///
2434    /// A retained gate has no live nonce (`None`), so releasing any other entry
2435    /// would weaken a currently configured or otherwise active reservation.
2436    pub(crate) fn release_retained_reserved_gate(&self, module_id: &str) -> bool {
2437        if self.get(module_id).is_some() {
2438            return false;
2439        }
2440        let mut reserved_nonces = self
2441            .reserved_nonces
2442            .lock()
2443            .unwrap_or_else(|poisoned| poisoned.into_inner());
2444        if !matches!(reserved_nonces.get(module_id), Some(None)) {
2445            return false;
2446        }
2447        reserved_nonces.remove(module_id);
2448        true
2449    }
2450
2451    pub(crate) fn operation_lock(&self) -> Arc<AsyncMutex<()>> {
2452        Arc::clone(&self.operation_lock)
2453    }
2454}
2455
2456/// Process supervisor for subc-owned singleton modules.
2457#[derive(Debug, Clone)]
2458pub struct Supervisor {
2459    registry: Arc<Registry>,
2460    restart_policy: RestartPolicy,
2461    drain_timeout: Duration,
2462    connection_file_path: Option<PathBuf>,
2463    capture_logs_dir: Option<PathBuf>,
2464    forwarding: Option<Arc<ForwardingTable>>,
2465    process_liveness: Arc<SupervisorProcessLiveness>,
2466    supervisor_handle: Option<SupervisorHandle>,
2467    health: HealthConfig,
2468    daemon_start_clock: crate::clock::StartClock,
2469    terminal_journal: Option<Arc<crate::terminal_journal::TerminalJournal>>,
2470    spawn_events: SpawnEventFeed,
2471    provenance_probe: ExecutableIdentityProbe,
2472    /// Every process spawned through this supervisor (and its clones) and not
2473    /// yet reaped, so daemon shutdown can end them.
2474    child_roster: ChildRoster,
2475    #[cfg(target_os = "linux")]
2476    cgroup_placement: Option<subc_cgroup::Placement>,
2477    #[cfg(test)]
2478    test_after_first_spawn: AfterFirstSpawnHook,
2479}
2480
2481/// Test-only hook run on the path that takes on a new module, right after its
2482/// first `spawn_child` returns (the process exists and could already be
2483/// registering) and before that process is handed to the module's supervise
2484/// loop and put on the roster. Lets a test observe what a fast child would see
2485/// in that window without racing a real one.
2486#[cfg(test)]
2487#[derive(Clone, Default)]
2488struct AfterFirstSpawnHook(Option<AfterFirstSpawnFn>);
2489
2490#[cfg(test)]
2491type AfterFirstSpawnFn = Arc<dyn Fn(&str) + Send + Sync>;
2492
2493#[cfg(test)]
2494impl fmt::Debug for AfterFirstSpawnHook {
2495    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
2496        f.write_str("AfterFirstSpawnHook")
2497    }
2498}
2499
2500#[cfg(test)]
2501impl AfterFirstSpawnHook {
2502    fn run(&self, module_id: &str) {
2503        if let Some(hook) = &self.0 {
2504            hook(module_id);
2505        }
2506    }
2507}
2508
2509impl Supervisor {
2510    #[cfg(test)]
2511    pub(crate) fn new_for_test(registry: Arc<Registry>, policy: RestartPolicy) -> Self {
2512        let supervisor = Self::new(registry, policy);
2513        #[cfg(target_os = "macos")]
2514        let supervisor = supervisor.with_privacy_trampoline(test_privacy_trampoline());
2515        supervisor
2516    }
2517    /// Verify the trampoline once when configured. Missing private OS support
2518    /// refuses every macOS launch by name but does not stop the daemon's control
2519    /// server. Embedders must explicitly provide a binary with the subc-os hidden
2520    /// entry point; the library must not exec an arbitrary hosting program.
2521    pub fn with_privacy_trampoline(self, path: impl Into<PathBuf>) -> Self {
2522        let path = path.into();
2523        #[cfg(target_os = "macos")]
2524        {
2525            let result = probe_privacy_trampoline(&path).map(|()| path);
2526            if let Err(cause) = &result {
2527                error!(%cause, "macOS privacy identity unavailable; supervised launches will refuse");
2528            }
2529            self.child_roster.set_privacy_trampoline(result);
2530        }
2531        #[cfg(not(target_os = "macos"))]
2532        let _ = path;
2533        self
2534    }
2535    /// The first step of an announced daemon shutdown, before the notice and
2536    /// before any connection is closed.
2537    ///
2538    /// Sets the daemon-shutdown flag first: from here on no module is
2539    /// respawned (crash restart, operator restart, or swap), and every child
2540    /// exit is recorded as `daemon_shutdown` rather than as a crash, whether
2541    /// the module exits on the EOF this shutdown gives it or is signalled by a
2542    /// service manager that kills the whole cgroup. Then writes the journal's
2543    /// shutdown marker, which records the instant and closes this daemon
2544    /// incarnation's stretch of the journal.
2545    #[cfg(unix)]
2546    pub(crate) fn begin_daemon_shutdown(&self) {
2547        self.child_roster.close();
2548        if let Some(journal) = &self.terminal_journal {
2549            journal.stamp_shutdown();
2550        }
2551    }
2552
2553    /// Announce a cut while established connections can still carry replies.
2554    /// These budgets promise notice and a bounded wait, not child completion;
2555    /// they are local policy, not an estimate of launchd's unknown kill ceiling.
2556    #[cfg(unix)]
2557    pub(crate) async fn drain_for_daemon_shutdown(&self) -> Result<(), SuperviseError> {
2558        const NOTICE_BUDGET: Duration = Duration::from_millis(500);
2559        const DRAIN_BUDGET: Duration = Duration::from_secs(2);
2560        let Some(forwarding) = &self.forwarding else {
2561            return Ok(());
2562        };
2563        let module_ids = forwarding
2564            .begin_daemon_drain()
2565            .map_err(SuperviseError::Forwarding)?;
2566        let deadline_ms =
2567            unix_ms_now().saturating_add((NOTICE_BUDGET + DRAIN_BUDGET).as_millis() as u64);
2568        let mut notices = tokio::task::JoinSet::new();
2569        let mut drains = Vec::new();
2570        for module_id in module_ids {
2571            let Some(target) = forwarding
2572                .begin_module_drain(&module_id, RouteCloseReason::Restart)
2573                .map_err(SuperviseError::Forwarding)?
2574            else {
2575                continue;
2576            };
2577            let routes = forwarding
2578                .endpoint_routes(target.endpoint)
2579                .map_err(SuperviseError::Forwarding)?;
2580            // Restart allows deployed consumers to reopen after the new daemon
2581            // appears. The wire reason stays `restart`; what tells a daemon cut
2582            // apart from a module restart afterwards is the terminal record
2583            // itself, whose disposition is `daemon_shutdown` for every exit
2584            // observed once `begin_daemon_shutdown` has run.
2585            let command = serde_json::to_vec(&ModuleControlCommand::Draining {
2586                reason: RouteCloseReason::Restart,
2587                deadline_ms,
2588            })
2589            .expect("module draining serializes");
2590            let mut recipients = vec![(target.sink.clone(), target.negotiated_ver, command)];
2591            let mut clients: Vec<(crate::forwarding::GoodbyeTarget, Vec<u16>)> = Vec::new();
2592            for route in routes {
2593                let client = route.goodbye_target;
2594                if let Some((_, channels)) = clients
2595                    .iter_mut()
2596                    .find(|(existing, _)| existing.connection_id == client.connection_id)
2597                {
2598                    channels.push(client.channel);
2599                } else {
2600                    let channel = client.channel;
2601                    clients.push((client, vec![channel]));
2602                }
2603            }
2604            for (client, mut channels) in clients {
2605                channels.sort_unstable();
2606                channels.dedup();
2607                let closing = serde_json::to_vec(&ClientControlPush::RouteClosing {
2608                    module_id: module_id.clone(),
2609                    channels,
2610                    reason: RouteCloseReason::Restart,
2611                })
2612                .expect("route closing serializes");
2613                recipients.push((client.sink, client.negotiated_ver, closing));
2614            }
2615            for (sink, version, body) in recipients {
2616                notices.spawn(async move {
2617                    let frame = Frame::build_with_version(
2618                        version,
2619                        FrameType::Push,
2620                        control_flags(),
2621                        0,
2622                        0,
2623                        0,
2624                        body,
2625                    )
2626                    .expect("bounded lifecycle notice frame builds");
2627                    sink.send_flushed(frame).await
2628                });
2629            }
2630            let gauges = declared_busy_gauges(&self.registry, &module_id)?;
2631            drains.push((module_id, target.endpoint, gauges));
2632        }
2633        // A quiet forwarding table is not proof that queued notices reached the
2634        // socket. Wait for writer flush acknowledgements before testing quiescence.
2635        let notice_deadline = Instant::now() + NOTICE_BUDGET;
2636        while let Ok(Some(result)) = timeout_at(notice_deadline, notices.join_next()).await {
2637            if !matches!(result, Ok(Ok(()))) {
2638                warn!(?result, "daemon shutdown notice delivery failed");
2639            }
2640        }
2641        notices.abort_all();
2642        let deadline = Instant::now() + DRAIN_BUDGET;
2643        let mut waits = tokio::task::JoinSet::new();
2644        for (module_id, endpoint, gauges) in drains {
2645            let forwarding = Arc::clone(forwarding);
2646            let mut runtime = self.runtime_config();
2647            runtime.health.cadence = Duration::from_millis(100);
2648            waits.spawn(async move {
2649                wait_for_forwarding_quiescence(
2650                    &forwarding,
2651                    &module_id,
2652                    &runtime,
2653                    endpoint,
2654                    deadline,
2655                    &gauges,
2656                    DrainScope::Active,
2657                )
2658                .await
2659            });
2660        }
2661        while let Ok(Some(result)) = timeout_at(deadline, waits.join_next()).await {
2662            if !matches!(result, Ok(Ok(true))) {
2663                warn!(?result, "daemon shutdown drain did not reach quiescence");
2664            }
2665        }
2666        Ok(())
2667    }
2668
2669    /// The last step of an announced daemon shutdown, after the notice and the
2670    /// drain: send every registered module a module GOODBYE, the same planned
2671    /// stop signal `ck module stop` gives, then close every connection so each
2672    /// subc module sees EOF and starts its own teardown, then end every
2673    /// supervised child that has not exited
2674    /// by its own deadline (its drain budget, capped). Modules lead their own
2675    /// process groups, so a
2676    /// service manager's group kill no longer reaches them; without this a
2677    /// child that does not stop on EOF (every `protocol: "none"` child, which
2678    /// has no connection) would outlive the daemon. Every wait is bounded (see
2679    /// `child_roster`), and `escalate` resolving (a second SIGTERM) cuts them.
2680    #[cfg(unix)]
2681    pub(crate) async fn end_children_for_daemon_shutdown(
2682        &self,
2683        already_escalated: bool,
2684        escalate: impl std::future::Future<Output = ()>,
2685    ) {
2686        tokio::pin!(escalate);
2687        let mut escalated = already_escalated;
2688        if let Some(forwarding) = &self.forwarding {
2689            let reason = CloseReason::new(
2690                "daemon_shutdown",
2691                "the daemon is exiting after its shutdown notice and drain",
2692            );
2693            if escalated {
2694                // The operator asked to stop waiting: queue the GOODBYEs but
2695                // do not wait for them to be written.
2696                send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, false).await;
2697            } else {
2698                tokio::select! {
2699                    biased;
2700                    _ = escalate.as_mut() => {
2701                        info!("second SIGTERM: abandoning module GOODBYE delivery");
2702                        escalated = true;
2703                    }
2704                    _ = send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, true) => {}
2705                }
2706            }
2707            let closed = forwarding.close_all_connections(&reason);
2708            debug!(closed, "closed established connections for daemon shutdown");
2709        }
2710        // If the second SIGTERM arrived during GOODBYE delivery, `escalate` has
2711        // already completed and must not be polled again; the child shutdown
2712        // wait is told it is escalated and gets a future that never fires.
2713        let escalated_here = escalated && !already_escalated;
2714        let remaining_escalate = async move {
2715            if escalated_here {
2716                std::future::pending::<()>().await;
2717            } else {
2718                escalate.await;
2719            }
2720        };
2721        crate::child_roster::end_children_for_daemon_shutdown(
2722            &self.child_roster,
2723            escalated,
2724            remaining_escalate,
2725        )
2726        .await;
2727    }
2728
2729    pub fn new(registry: Arc<Registry>, restart_policy: RestartPolicy) -> Self {
2730        Self {
2731            registry,
2732            restart_policy,
2733            drain_timeout: DEFAULT_DRAIN_TIMEOUT,
2734            connection_file_path: None,
2735            capture_logs_dir: None,
2736            forwarding: None,
2737            process_liveness: Arc::new(SupervisorProcessLiveness::default()),
2738            supervisor_handle: None,
2739            health: HealthConfig::default(),
2740            daemon_start_clock: crate::clock::StartClock::capture(),
2741            terminal_journal: None,
2742            spawn_events: SpawnEventFeed::default(),
2743            provenance_probe: ExecutableIdentityProbe::default(),
2744            child_roster: ChildRoster::default(),
2745            #[cfg(target_os = "linux")]
2746            cgroup_placement: None,
2747            #[cfg(test)]
2748            test_after_first_spawn: AfterFirstSpawnHook::default(),
2749        }
2750    }
2751
2752    pub fn with_drain_timeout(mut self, drain_timeout: Duration) -> Self {
2753        self.drain_timeout = drain_timeout;
2754        self
2755    }
2756
2757    pub fn with_process_liveness(
2758        mut self,
2759        process_liveness: Arc<SupervisorProcessLiveness>,
2760    ) -> Self {
2761        self.process_liveness = process_liveness;
2762        self
2763    }
2764
2765    pub fn with_connection_file_path(mut self, connection_file_path: impl Into<PathBuf>) -> Self {
2766        self.connection_file_path = Some(connection_file_path.into());
2767        self
2768    }
2769
2770    /// Enables daemon-owned capture files for supervised stdout and stderr.
2771    pub fn with_capture_logs_dir(mut self, logs_dir: impl Into<PathBuf>) -> Self {
2772        self.capture_logs_dir = Some(logs_dir.into());
2773        self
2774    }
2775
2776    /// Names this daemon lifetime in spawn events, independently of whether a
2777    /// terminal journal is configured.
2778    pub fn with_daemon_incarnation(self, daemon_incarnation: String) -> Self {
2779        // A millisecond start stamp can repeat after clock rollback or a rapid
2780        // restart. Use the connection file's random daemon_id instead: it already
2781        // identifies this daemon lifetime independently of the wall clock.
2782        self.spawn_events.configure_incarnation(daemon_incarnation);
2783        self
2784    }
2785
2786    /// Enables best-effort history shared by every supervised module. Without
2787    /// it, terminal history is kept only in each module's in-memory ring.
2788    pub fn with_terminal_journal(self, path: PathBuf, daemon_incarnation: String) -> Self {
2789        let mut this = self.with_daemon_incarnation(daemon_incarnation.clone());
2790        this.terminal_journal = Some(Arc::new(crate::terminal_journal::TerminalJournal::open(
2791            path,
2792            daemon_incarnation,
2793        )));
2794        this
2795    }
2796
2797    pub fn with_forwarding(mut self, forwarding: Arc<ForwardingTable>) -> Self {
2798        self.forwarding = Some(forwarding);
2799        self
2800    }
2801
2802    pub fn with_handle(mut self, supervisor_handle: SupervisorHandle) -> Self {
2803        self.spawn_events = supervisor_handle.spawn_events.clone();
2804        self.supervisor_handle = Some(supervisor_handle);
2805        self
2806    }
2807
2808    pub fn with_health_config(mut self, health: HealthConfig) -> Self {
2809        self.health = health;
2810        self
2811    }
2812
2813    /// Keeps a record of every live child at `path`, rewritten on each spawn and
2814    /// reap, for the orphan sweep a later daemon runs at boot. Without it no
2815    /// record is kept.
2816    pub fn with_live_children_record(self, path: impl Into<PathBuf>) -> Self {
2817        self.child_roster.record_to(path.into());
2818        self
2819    }
2820
2821    #[cfg(target_os = "linux")]
2822    pub fn with_cgroup_placement(
2823        mut self,
2824        cgroup_placement: Option<subc_cgroup::Placement>,
2825    ) -> Self {
2826        self.cgroup_placement = cgroup_placement;
2827        self
2828    }
2829
2830    /// Spawn `spec.program` and start monitoring it.
2831    ///
2832    /// The child is expected to parse `--subc <connection-file-path>`, read the
2833    /// TCP+key connection file, authenticate to the already-running listener, and
2834    /// register with channel-0 `HELLO` using `spec.module_id` as its manifest id.
2835    pub fn spawn(&self, spec: ModuleSpec) -> Result<SupervisedModule, SuperviseError> {
2836        validate_spec(&spec)?;
2837        self.establish_identity(&spec);
2838
2839        let runtime = self.runtime_config();
2840        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2841        let spawned = spawn_child(
2842            &spec,
2843            runtime.connection_file_path.as_deref(),
2844            self.supervisor_handle.as_ref(),
2845            &runtime.stderr_ring,
2846            runtime.capture_logs_dir.as_deref(),
2847            &runtime.child_roster,
2848            #[cfg(target_os = "linux")]
2849            runtime.cgroup_placement.as_ref(),
2850        );
2851        #[cfg(test)]
2852        self.test_after_first_spawn.run(&spec.module_id);
2853        let child = match spawned {
2854            Ok(child) => child,
2855            Err(err) => {
2856                // Unlike the configured paths, a failed `spawn` leaves nothing
2857                // on the roster, so the module must not stay marked configured.
2858                self.abandon_unrostered(&spec.module_id);
2859                return Err(err);
2860            }
2861        };
2862        self.mark_first_process_running(&spec, &runtime, &snapshot, &child)?;
2863
2864        Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2865    }
2866
2867    /// Make `spec`'s module count as configured, with its identity gates
2868    /// (reserved id, reserved prefixes) in place, BEFORE any process of it
2869    /// exists.
2870    ///
2871    /// Every path that takes on a new module calls this before `spawn_child`.
2872    /// The order is the point: the child can connect, register, sync its
2873    /// scopes and ask about them as soon as it is spawned, and the module is
2874    /// only put on the roster after `spawn_child` returns. Were the mark set
2875    /// with the roster entry, a fast child would see its own owner reported
2876    /// as not configured, and a scoped `route.open` in that window would be
2877    /// refused as terminal `scope_not_live` ("will never sync") instead of
2878    /// retryable `scope_not_synced`.
2879    fn establish_identity(&self, spec: &ModuleSpec) {
2880        if let Some(supervisor_handle) = &self.supervisor_handle {
2881            supervisor_handle.apply_identity_configuration(spec);
2882            supervisor_handle.mark_configured(&spec.module_id);
2883        }
2884    }
2885
2886    /// Take back [`Self::establish_identity`]'s configured mark when the
2887    /// module will not be put on the roster after all.
2888    fn abandon_unrostered(&self, module_id: &str) {
2889        if let Some(supervisor_handle) = &self.supervisor_handle {
2890            supervisor_handle.unmark_configured_unless_rostered(module_id);
2891        }
2892    }
2893
2894    /// Record a freshly spawned first process as running. On failure the
2895    /// module never reaches the roster, so its configured mark is taken back.
2896    fn mark_first_process_running(
2897        &self,
2898        spec: &ModuleSpec,
2899        runtime: &SupervisorRuntimeConfig,
2900        snapshot: &SharedSnapshot,
2901        child: &SupervisedChild,
2902    ) -> Result<(), SuperviseError> {
2903        if let Err(err) = set_running(snapshot, child, &spec.module_id, &runtime.spawn_events) {
2904            self.abandon_unrostered(&spec.module_id);
2905            return Err(err);
2906        }
2907        self.process_liveness
2908            .track(spec.module_id.clone(), Arc::clone(snapshot));
2909        Ok(())
2910    }
2911
2912    /// Start supervising a module declared in daemon configuration.
2913    ///
2914    /// Unlike [`Self::spawn`], this records disabled modules and immediate spawn
2915    /// failures in the supervisor handle so operator-facing `supervisor.list`
2916    /// reflects every configured module while daemon startup continues.
2917    pub fn supervise_configured(
2918        &self,
2919        spec: ModuleSpec,
2920        enabled: bool,
2921    ) -> Result<SupervisedModule, SuperviseError> {
2922        validate_spec(&spec)?;
2923        self.establish_identity(&spec);
2924
2925        let runtime = self.runtime_config();
2926        if !enabled {
2927            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
2928            return Ok(self.supervised_module(spec, runtime, snapshot, None));
2929        }
2930
2931        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2932        let spawned = spawn_child(
2933            &spec,
2934            runtime.connection_file_path.as_deref(),
2935            self.supervisor_handle.as_ref(),
2936            &runtime.stderr_ring,
2937            runtime.capture_logs_dir.as_deref(),
2938            &runtime.child_roster,
2939            #[cfg(target_os = "linux")]
2940            runtime.cgroup_placement.as_ref(),
2941        );
2942        #[cfg(test)]
2943        self.test_after_first_spawn.run(&spec.module_id);
2944        match spawned {
2945            Ok(child) => {
2946                self.mark_first_process_running(&spec, &runtime, &snapshot, &child)?;
2947                Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2948            }
2949            Err(err) => {
2950                error!(
2951                    module_id = %spec.module_id,
2952                    program = %spec.program.display(),
2953                    error = %err,
2954                    "configured module failed to spawn; marking failed and continuing"
2955                );
2956                let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
2957                Ok(self.supervised_module(spec, runtime, snapshot, None))
2958            }
2959        }
2960    }
2961
2962    /// Supervise a configured module with its own health, drain, and crash
2963    /// budget. The restart policy is per-module because the config file is:
2964    /// `modules.<id>.restart` resolves to a full policy at parse time, and a
2965    /// module that is expensive to restart should not be forced onto the same
2966    /// budget as one that is cheap.
2967    pub fn supervise_configured_with_health(
2968        &self,
2969        spec: ModuleSpec,
2970        enabled: bool,
2971        health: HealthConfig,
2972        drain_timeout_ms: Option<u64>,
2973        restart_policy: RestartPolicy,
2974    ) -> Result<SupervisedModule, SuperviseError> {
2975        validate_spec(&spec)?;
2976        self.establish_identity(&spec);
2977
2978        let mut runtime = self.runtime_config();
2979        runtime.health = health.clone();
2980        runtime.restart_policy = restart_policy;
2981        if let Some(ms) = drain_timeout_ms {
2982            runtime.drain_timeout = Duration::from_millis(ms);
2983            *runtime
2984                .effective_drain_timeout
2985                .lock()
2986                .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
2987        }
2988        if !enabled {
2989            let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
2990            return Ok(self.supervised_module(spec, runtime, snapshot, None));
2991        }
2992
2993        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2994        let spawned = spawn_child(
2995            &spec,
2996            runtime.connection_file_path.as_deref(),
2997            self.supervisor_handle.as_ref(),
2998            &runtime.stderr_ring,
2999            runtime.capture_logs_dir.as_deref(),
3000            &runtime.child_roster,
3001            #[cfg(target_os = "linux")]
3002            runtime.cgroup_placement.as_ref(),
3003        );
3004        #[cfg(test)]
3005        self.test_after_first_spawn.run(&spec.module_id);
3006        match spawned {
3007            Ok(child) => {
3008                self.mark_first_process_running(&spec, &runtime, &snapshot, &child)?;
3009                Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
3010            }
3011            Err(err) => {
3012                if health.critical {
3013                    error!(
3014                        module_id = %spec.module_id,
3015                        program = %spec.program.display(),
3016                        error = %err,
3017                        "critical configured module failed to spawn; marking failed and alerting"
3018                    );
3019                } else {
3020                    error!(
3021                        module_id = %spec.module_id,
3022                        program = %spec.program.display(),
3023                        error = %err,
3024                        "configured module failed to spawn; marking failed and continuing"
3025                    );
3026                }
3027                let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
3028                Ok(self.supervised_module(spec, runtime, snapshot, None))
3029            }
3030        }
3031    }
3032
3033    fn runtime_config(&self) -> SupervisorRuntimeConfig {
3034        let effective_drain_timeout = Arc::new(Mutex::new(self.drain_timeout));
3035        SupervisorRuntimeConfig {
3036            scheduled_respawn: Arc::default(),
3037            deferred_reload_reply: Arc::default(),
3038            restart_policy: self.restart_policy,
3039            drain_timeout: self.drain_timeout,
3040            // Shared with this module's roster copy: daemon shutdown waits on
3041            // each child for the module's own drain budget, as resolved now.
3042            child_roster: self
3043                .child_roster
3044                .for_module(Arc::clone(&effective_drain_timeout)),
3045            effective_drain_timeout,
3046            default_drain_timeout: self.drain_timeout,
3047            health: self.health.clone(),
3048            connection_file_path: self.connection_file_path.clone(),
3049            capture_logs_dir: self.capture_logs_dir.clone(),
3050            forwarding: self.forwarding.clone(),
3051            supervisor_handle: self.supervisor_handle.clone(),
3052            stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
3053            terminal_ring: Arc::new(Mutex::new(
3054                TerminalRing::new(
3055                    TerminalRingConfig::default(),
3056                    self.daemon_start_clock.started_at_ms(),
3057                )
3058                .with_start_clock(self.daemon_start_clock)
3059                .with_journal(self.terminal_journal.clone())
3060                .with_daemon_shutdown(self.child_roster.shutdown_flag()),
3061            )),
3062            spawn_events: self.spawn_events.clone(),
3063            #[cfg(target_os = "linux")]
3064            cgroup_placement: self.cgroup_placement.clone(),
3065            #[cfg(test)]
3066            test_seed_stale_facts_before_enable_spawn: false,
3067            #[cfg(test)]
3068            test_reload_exit_record_gate: None,
3069        }
3070    }
3071
3072    fn supervised_module(
3073        &self,
3074        spec: ModuleSpec,
3075        runtime: SupervisorRuntimeConfig,
3076        snapshot: SharedSnapshot,
3077        child: Option<SupervisedChild>,
3078    ) -> SupervisedModule {
3079        let configuration = Arc::new(Mutex::new(SupervisedConfiguration {
3080            spec: spec.clone(),
3081            health: runtime.health.clone(),
3082        }));
3083        let stderr_ring = Arc::clone(&runtime.stderr_ring);
3084        let terminal_ring = Arc::clone(&runtime.terminal_ring);
3085        // The module's OWN policy, which may be its per-module config rather than
3086        // the supervisor-wide one; status must report the budget the supervise
3087        // loop actually enforces.
3088        let restart_policy = runtime.restart_policy;
3089        let effective_drain_timeout = Arc::clone(&runtime.effective_drain_timeout);
3090        let (tx, rx) = mpsc::channel(4);
3091        let monitor = tokio::spawn(supervise_loop(
3092            spec.clone(),
3093            runtime,
3094            Arc::clone(&self.registry),
3095            Arc::clone(&self.process_liveness),
3096            Arc::clone(&snapshot),
3097            child,
3098            rx,
3099        ));
3100
3101        let module_id = spec.module_id.clone();
3102        let module = SupervisedModule {
3103            inner: Arc::new(SupervisedModuleInner {
3104                module_id: module_id.clone(),
3105                registry: Arc::clone(&self.registry),
3106                snapshot,
3107                configuration,
3108                stderr_ring,
3109                terminal_ring,
3110                commands: tx,
3111                monitor: Mutex::new(Some(monitor)),
3112                restart_policy,
3113                effective_drain_timeout,
3114                provenance_probe: self.provenance_probe.clone(),
3115            }),
3116        };
3117        // The identity gates and the configured mark were set by
3118        // `establish_identity` before any process was spawned; only the roster
3119        // entry waits for the module handle, which needs the spawned child.
3120        if let Some(supervisor_handle) = &self.supervisor_handle {
3121            supervisor_handle.insert(module.clone());
3122        }
3123        module
3124    }
3125}
3126
3127impl Default for Supervisor {
3128    fn default() -> Self {
3129        Self::new(Arc::new(Registry::default()), RestartPolicy::default())
3130    }
3131}
3132
3133/// Handle to one supervised child process.
3134#[derive(Clone)]
3135pub struct SupervisedModule {
3136    inner: Arc<SupervisedModuleInner>,
3137}
3138
3139struct SupervisedModuleInner {
3140    module_id: String,
3141    registry: Arc<Registry>,
3142    snapshot: SharedSnapshot,
3143    configuration: Arc<Mutex<SupervisedConfiguration>>,
3144    stderr_ring: Arc<Mutex<StderrRing>>,
3145    terminal_ring: Arc<Mutex<TerminalRing>>,
3146    commands: mpsc::Sender<SupervisorCommand>,
3147    monitor: Mutex<Option<JoinHandle<()>>>,
3148    /// Copied from the supervisor's runtime config at spawn so `status()` can
3149    /// report the restart budget without reaching back into the supervisor. The
3150    /// policy is fixed for the process's lifetime, so a copy cannot drift.
3151    restart_policy: RestartPolicy,
3152    effective_drain_timeout: Arc<Mutex<Duration>>,
3153    provenance_probe: ExecutableIdentityProbe,
3154}
3155
3156impl fmt::Debug for SupervisedModule {
3157    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3158        f.debug_struct("SupervisedModule")
3159            .field("module_id", &self.inner.module_id)
3160            .field("status", &self.status())
3161            .finish_non_exhaustive()
3162    }
3163}
3164
3165impl SupervisedModule {
3166    pub fn module_id(&self) -> &str {
3167        &self.inner.module_id
3168    }
3169
3170    /// Test-only: put one probe miss on the streak, the way
3171    /// `handle_health_probe_failure` does, so tests can assert what a later
3172    /// event does to the streak without driving the whole probe loop.
3173    #[cfg(test)]
3174    pub(crate) fn record_health_probe_failure_for_test(
3175        &self,
3176        detail: &str,
3177    ) -> Result<(), SuperviseError> {
3178        update_snapshot(&self.inner.snapshot, Some(&self.inner.module_id), |state| {
3179            state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
3180            state.health.detail = Some(detail.to_string());
3181        })
3182    }
3183
3184    pub fn state(&self) -> Result<ModuleState, SuperviseError> {
3185        Ok(lock_snapshot(&self.inner.snapshot)?.state)
3186    }
3187
3188    /// The module's retained stderr, newest lines last.
3189    ///
3190    /// Deliberately NOT on [`Self::status`]: a bounded tail is kilobytes per
3191    /// module, `supervisor.list` renders every module, and putting it in the
3192    /// shared snapshot would make each status read carry a payload almost nobody
3193    /// asked for. Callers that want the text ask for it.
3194    pub fn stderr_tail(
3195        &self,
3196        max_lines: Option<usize>,
3197        max_bytes: Option<usize>,
3198    ) -> StderrTailSnapshot {
3199        self.inner
3200            .stderr_ring
3201            .lock()
3202            .unwrap_or_else(|poisoned| poisoned.into_inner())
3203            .snapshot(max_lines, max_bytes)
3204    }
3205
3206    /// The module's bounded terminal history, oldest retained exit first.
3207    ///
3208    /// The daemon-start stamp distinguishes a quiet supervisor from a replacement
3209    /// daemon whose in-memory history was necessarily reset.
3210    pub fn terminal_history(&self) -> TerminalHistorySnapshot {
3211        self.inner
3212            .terminal_ring
3213            .lock()
3214            .unwrap_or_else(|poisoned| poisoned.into_inner())
3215            .snapshot()
3216    }
3217
3218    /// Retained observations from the current ring and all journal generations.
3219    ///
3220    /// Blocking: this reads the journal files. Async callers use
3221    /// [`Self::read_durable_terminal_history`].
3222    pub fn durable_terminal_history(&self) -> subc_control::TerminalHistory {
3223        durable_terminal_history_of(&self.inner.terminal_ring, &self.inner.module_id)
3224    }
3225
3226    /// [`Self::durable_terminal_history`] on a blocking thread, so the journal
3227    /// read (up to every retained generation) never occupies a runtime worker.
3228    /// Fails only if the blocking task could not finish (runtime shutdown or a
3229    /// panic in the read).
3230    pub(crate) async fn read_durable_terminal_history(
3231        &self,
3232    ) -> Result<subc_control::TerminalHistory, tokio::task::JoinError> {
3233        let terminal_ring = Arc::clone(&self.inner.terminal_ring);
3234        let module_id = self.inner.module_id.clone();
3235        tokio::task::spawn_blocking(move || durable_terminal_history_of(&terminal_ring, &module_id))
3236            .await
3237    }
3238
3239    pub fn status(&self) -> Result<ModuleStatus, SuperviseError> {
3240        self.status_with_snapshot_lock(&self.inner.snapshot, None)
3241            .map(|(status, _)| status)
3242    }
3243
3244    pub(crate) fn record_deliberate_severance(
3245        &self,
3246        identity: ProcessIdentity,
3247    ) -> Result<bool, SuperviseError> {
3248        let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
3249        if snapshot.pid != Some(identity.pid)
3250            || snapshot.process_start_time != Some(identity.start_time)
3251        {
3252            return Ok(false);
3253        }
3254        snapshot.deliberate_severance = Some(identity);
3255        Ok(true)
3256    }
3257
3258    /// Read status for a channel-0 renderer and report a contended snapshot lock.
3259    ///
3260    /// Internal supervision callers use [`Self::status`] so writer-side machinery
3261    /// does not produce reader-observability logs.
3262    pub(crate) fn status_for_control(
3263        &self,
3264        caller: &'static str,
3265    ) -> Result<ModuleStatus, SuperviseError> {
3266        self.status_with_snapshot_lock(&self.inner.snapshot, Some(caller))
3267            .map(|(status, _)| status)
3268    }
3269
3270    fn status_with_snapshot_lock(
3271        &self,
3272        snapshot: &SharedSnapshot,
3273        caller: Option<&'static str>,
3274    ) -> Result<(ModuleStatus, Option<SpawnedFileIdentity>), SuperviseError> {
3275        let mut guard = match caller {
3276            Some(caller) => lock_snapshot_for_control(snapshot, &self.inner.module_id, caller)?,
3277            None => lock_snapshot(snapshot)?,
3278        };
3279        // Read the budget through the pruning path so a reader sees the same
3280        // in-window count the restart decision would use, not a stale total.
3281        let restart_count =
3282            guard.crash_restarts_in_window(self.inner.restart_policy.window, Instant::now());
3283        let snapshot = guard.clone();
3284        drop(guard);
3285        let drain_timeout = *self.inner.effective_drain_timeout.lock().map_err(|_| {
3286            SuperviseError::StatePoisoned {
3287                module_id: Some(self.inner.module_id.clone()),
3288            }
3289        })?;
3290        let registration_active = self
3291            .inner
3292            .registry
3293            .get_module(&self.inner.module_id)
3294            .map_err(SuperviseError::Registry)?
3295            .is_some();
3296        let protocol = snapshot
3297            .spawned_protocol
3298            .unwrap_or(self.declared_protocol()?);
3299        let running_process =
3300            snapshot.enabled && snapshot.state == ModuleState::Running && snapshot.process_alive;
3301        // Registration is the difference between the two protocols and the only
3302        // one: a subc module that has not registered cannot serve a request even
3303        // though its process is up, and a `none` module never registers at all,
3304        // so requiring it there would pin `live` to false for the whole life of
3305        // a perfectly healthy process.
3306        let live = match protocol {
3307            ModuleProtocol::Subc => running_process && registration_active,
3308            ModuleProtocol::None => running_process,
3309        };
3310
3311        Ok((
3312            ModuleStatus {
3313                module_id: self.inner.module_id.clone(),
3314                state: snapshot.state,
3315                enabled: snapshot.enabled,
3316                process_alive: snapshot.process_alive,
3317                registration_active,
3318                protocol,
3319                live,
3320                restart_count,
3321                lifetime_restarts: snapshot.lifetime_restarts,
3322                spawn_generation: snapshot.spawn_generation,
3323                max_restarts: self.inner.restart_policy.max_restarts,
3324                restart_window: self.inner.restart_policy.window,
3325                drain_timeout,
3326                restart_backoff: self.inner.restart_policy.backoff,
3327                restart_max_backoff: self.inner.restart_policy.max_backoff,
3328                pid: snapshot.reported_pid(),
3329                spawned_at_ms: snapshot.spawned_at_ms,
3330                spawned_from: snapshot.spawned_from,
3331                process_start_time: snapshot.process_start_time,
3332                last_exit: snapshot.last_exit,
3333                health: snapshot.health,
3334            },
3335            snapshot.spawned_file_identity,
3336        ))
3337    }
3338
3339    #[cfg(test)]
3340    pub(crate) fn hold_snapshot_for_test(
3341        &self,
3342        acquired: std::sync::mpsc::Sender<()>,
3343        hold: Duration,
3344    ) -> std::thread::JoinHandle<()> {
3345        let snapshot = Arc::clone(&self.inner.snapshot);
3346        std::thread::spawn(move || {
3347            let _guard = snapshot.lock().expect("test snapshot lock is not poisoned");
3348            acquired
3349                .send(())
3350                .expect("test receiver waits for snapshot lock");
3351            std::thread::sleep(hold);
3352        })
3353    }
3354
3355    /// The status and the running-image check for `supervisor.provenance`,
3356    /// taken from one status read. The exec acknowledgement can land between
3357    /// two separate reads, and the reply would then pair "no pid yet" with an
3358    /// image observed after the module started, which describes no single
3359    /// moment.
3360    pub(crate) async fn status_and_running_image_agreement(
3361        &self,
3362    ) -> Result<(ModuleStatus, subc_control::RunningImageAgreement), SuperviseError> {
3363        let (status, identity) = self.status_with_snapshot_lock(&self.inner.snapshot, None)?;
3364        let image = self
3365            .inner
3366            .provenance_probe
3367            .observe(
3368                status.pid,
3369                status.spawned_from.as_deref(),
3370                identity,
3371                status.process_start_time,
3372            )
3373            .await;
3374        Ok((status, image))
3375    }
3376
3377    pub(crate) async fn running_image_agreement(&self) -> subc_control::RunningImageAgreement {
3378        let snapshot = match lock_snapshot(&self.inner.snapshot) {
3379            Ok(snapshot) => snapshot.clone(),
3380            Err(_) => {
3381                return subc_control::RunningImageAgreement::Unavailable {
3382                    reason: subc_control::RunningImageUnavailableReason::NotRunning,
3383                };
3384            }
3385        };
3386        self.inner
3387            .provenance_probe
3388            .observe(
3389                snapshot.reported_pid(),
3390                snapshot.spawned_from.as_deref(),
3391                snapshot.spawned_file_identity,
3392                snapshot.process_start_time,
3393            )
3394            .await
3395    }
3396
3397    /// Memory and CPU time of the module's current process, read now. Only the
3398    /// process the supervisor spawned is read, not processes it has started.
3399    pub(crate) fn child_resource_usage(&self) -> subc_control::ChildResourceUsage {
3400        let (pid, start_time) = match lock_snapshot(&self.inner.snapshot) {
3401            Ok(snapshot) => (snapshot.reported_pid(), snapshot.process_start_time),
3402            Err(_) => {
3403                return subc_control::ChildResourceUsage::Unavailable {
3404                    reason: subc_control::ChildResourceUnavailableReason::Unreadable,
3405                }
3406            }
3407        };
3408        crate::child_resources::read(pid, start_time)
3409    }
3410
3411    pub(crate) fn will_recover_after_connection_loss(&self) -> Result<bool, SuperviseError> {
3412        let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
3413        Ok(match snapshot.state {
3414            ModuleState::Restarting => true,
3415            ModuleState::Failed | ModuleState::Disabled => false,
3416            _ => daemon_will_restart(&mut snapshot, &self.inner.restart_policy, Instant::now()),
3417        })
3418    }
3419
3420    #[cfg(test)]
3421    pub(crate) fn is_warming(&self) -> Result<bool, SuperviseError> {
3422        self.is_warming_with_snapshot_lock(None)
3423    }
3424
3425    pub(crate) fn is_warming_for_control(
3426        &self,
3427        caller: &'static str,
3428    ) -> Result<bool, SuperviseError> {
3429        self.is_warming_with_snapshot_lock(Some(caller))
3430    }
3431
3432    fn is_warming_with_snapshot_lock(
3433        &self,
3434        caller: Option<&'static str>,
3435    ) -> Result<bool, SuperviseError> {
3436        let snapshot = match caller {
3437            Some(caller) => {
3438                lock_snapshot_for_control(&self.inner.snapshot, &self.inner.module_id, caller)?
3439            }
3440            None => lock_snapshot(&self.inner.snapshot)?,
3441        }
3442        .clone();
3443        Ok(matches!(
3444            snapshot.state,
3445            ModuleState::Starting | ModuleState::Running | ModuleState::Restarting
3446        ))
3447    }
3448
3449    /// Drain the module and stop monitoring it.
3450    pub async fn drain(&self) -> Result<(), SuperviseError> {
3451        self.stop().await
3452    }
3453
3454    pub(crate) async fn retire(&self) -> Result<(), SuperviseError> {
3455        match self.state()? {
3456            ModuleState::Stopped | ModuleState::Failed => return Ok(()),
3457            ModuleState::Starting
3458            | ModuleState::Running
3459            | ModuleState::Unresponsive
3460            | ModuleState::Restarting
3461            | ModuleState::Draining
3462            | ModuleState::Disabled => {}
3463        }
3464
3465        let (reply_tx, reply_rx) = oneshot::channel();
3466        self.inner
3467            .commands
3468            .send(SupervisorCommand::Retire { reply: reply_tx })
3469            .await
3470            .map_err(|_| SuperviseError::CommandClosed {
3471                module_id: self.inner.module_id.clone(),
3472            })?;
3473        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3474            module_id: self.inner.module_id.clone(),
3475        })?
3476    }
3477
3478    pub async fn stop(&self) -> Result<(), SuperviseError> {
3479        match self.state()? {
3480            ModuleState::Stopped | ModuleState::Failed => return Ok(()),
3481            ModuleState::Starting
3482            | ModuleState::Running
3483            | ModuleState::Unresponsive
3484            | ModuleState::Restarting
3485            | ModuleState::Draining
3486            | ModuleState::Disabled => {}
3487        }
3488
3489        let (reply_tx, reply_rx) = oneshot::channel();
3490        self.inner
3491            .commands
3492            .send(SupervisorCommand::Drain { reply: reply_tx })
3493            .await
3494            .map_err(|_| SuperviseError::CommandClosed {
3495                module_id: self.inner.module_id.clone(),
3496            })?;
3497        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3498            module_id: self.inner.module_id.clone(),
3499        })?
3500    }
3501
3502    pub async fn restart(&self, drain_timeout_ms: Option<u64>) -> Result<(), SuperviseError> {
3503        let received_at_generation = lock_snapshot(&self.inner.snapshot)?.spawn_generation;
3504        let (reply_tx, reply_rx) = oneshot::channel();
3505        self.inner
3506            .commands
3507            .send(SupervisorCommand::Restart {
3508                drain_timeout_ms,
3509                received_at_generation,
3510                queued_at: Instant::now(),
3511                reply: reply_tx,
3512            })
3513            .await
3514            .map_err(|_| SuperviseError::CommandClosed {
3515                module_id: self.inner.module_id.clone(),
3516            })?;
3517        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3518            module_id: self.inner.module_id.clone(),
3519        })?
3520    }
3521
3522    /// Blue/green restart: see [`SupervisorCommand::Swap`] and the
3523    /// `supervisor_swap` module. Returns once the swap has cut over (the old
3524    /// process then drains in the background of the supervise loop) or has
3525    /// failed, leaving the old process serving.
3526    pub async fn swap(&self, ready_timeout: Option<Duration>) -> Result<(), SuperviseError> {
3527        let (reply_tx, reply_rx) = oneshot::channel();
3528        self.inner
3529            .commands
3530            .send(SupervisorCommand::Swap {
3531                ready_timeout,
3532                reply: reply_tx,
3533            })
3534            .await
3535            .map_err(|_| SuperviseError::CommandClosed {
3536                module_id: self.inner.module_id.clone(),
3537            })?;
3538        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3539            module_id: self.inner.module_id.clone(),
3540        })?
3541    }
3542
3543    pub async fn reload(&self) -> Result<(), SuperviseError> {
3544        let (reply_tx, reply_rx) = oneshot::channel();
3545        self.inner
3546            .commands
3547            .send(SupervisorCommand::Reload { reply: reply_tx })
3548            .await
3549            .map_err(|_| SuperviseError::CommandClosed {
3550                module_id: self.inner.module_id.clone(),
3551            })?;
3552        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3553            module_id: self.inner.module_id.clone(),
3554        })?
3555    }
3556
3557    pub async fn set_enabled(&self, enabled: bool) -> Result<bool, SuperviseError> {
3558        let (reply_tx, reply_rx) = oneshot::channel();
3559        self.inner
3560            .commands
3561            .send(SupervisorCommand::SetEnabled {
3562                enabled,
3563                reply: reply_tx,
3564            })
3565            .await
3566            .map_err(|_| SuperviseError::CommandClosed {
3567                module_id: self.inner.module_id.clone(),
3568            })?;
3569        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3570            module_id: self.inner.module_id.clone(),
3571        })?
3572    }
3573
3574    /// The current process's protocol, or the configured protocol when down.
3575    /// A rescan stores the next launch spec without changing how an existing
3576    /// process registers, serves routes, is probed, or exits.
3577    pub(crate) fn declared_protocol(&self) -> Result<ModuleProtocol, SuperviseError> {
3578        let configured = self
3579            .inner
3580            .configuration
3581            .lock()
3582            .map_err(|_| SuperviseError::StatePoisoned {
3583                module_id: Some(self.inner.module_id.clone()),
3584            })?
3585            .spec
3586            .protocol;
3587        let state = lock_snapshot(&self.inner.snapshot)?;
3588        Ok(state.spawned_protocol.unwrap_or(configured))
3589    }
3590
3591    pub(crate) fn configuration(&self) -> Result<(ModuleSpec, HealthConfig), SuperviseError> {
3592        let configuration =
3593            self.inner
3594                .configuration
3595                .lock()
3596                .map_err(|_| SuperviseError::StatePoisoned {
3597                    module_id: Some(self.inner.module_id.clone()),
3598                })?;
3599        Ok((configuration.spec.clone(), configuration.health.clone()))
3600    }
3601
3602    /// Replace this module's launch spec, keeping its health and drain policy,
3603    /// the way a rescan does for a changed config entry. The running process is
3604    /// untouched; the next spawn (a restart, or a swap's candidate) uses it.
3605    #[cfg(any(test, feature = "test-support"))]
3606    pub async fn update_spec_for_test(&self, spec: ModuleSpec) -> Result<(), SuperviseError> {
3607        let (_, health) = self.configuration()?;
3608        let drain_timeout_ms = u64::try_from(
3609            self.inner
3610                .effective_drain_timeout
3611                .lock()
3612                .unwrap_or_else(|poisoned| poisoned.into_inner())
3613                .as_millis(),
3614        )
3615        .ok();
3616        self.update_configuration(spec, health, drain_timeout_ms)
3617            .await
3618    }
3619
3620    pub(crate) async fn update_configuration(
3621        &self,
3622        spec: ModuleSpec,
3623        health: HealthConfig,
3624        drain_timeout_ms: Option<u64>,
3625    ) -> Result<(), SuperviseError> {
3626        if spec.module_id != self.inner.module_id {
3627            return Err(SuperviseError::InvalidSpec {
3628                reason: "a supervised module's module_id cannot be changed".to_string(),
3629            });
3630        }
3631        validate_spec(&spec)?;
3632        let (reply_tx, reply_rx) = oneshot::channel();
3633        self.inner
3634            .commands
3635            .send(SupervisorCommand::UpdateConfiguration {
3636                spec: spec.clone(),
3637                health: health.clone(),
3638                drain_timeout_ms,
3639                reply: reply_tx,
3640            })
3641            .await
3642            .map_err(|_| SuperviseError::CommandClosed {
3643                module_id: self.inner.module_id.clone(),
3644            })?;
3645        reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3646            module_id: self.inner.module_id.clone(),
3647        })?;
3648        let mut configuration =
3649            self.inner
3650                .configuration
3651                .lock()
3652                .map_err(|_| SuperviseError::StatePoisoned {
3653                    module_id: Some(self.inner.module_id.clone()),
3654                })?;
3655        configuration.spec = spec;
3656        configuration.health = health;
3657        Ok(())
3658    }
3659}
3660
3661impl Drop for SupervisedModuleInner {
3662    fn drop(&mut self) {
3663        let Ok(mut monitor) = self.monitor.lock() else {
3664            return;
3665        };
3666        if let Some(monitor) = monitor.as_ref().filter(|monitor| !monitor.is_finished()) {
3667            let _ = update_snapshot(&self.snapshot, Some(&self.module_id), |state| {
3668                state.state = ModuleState::Stopped;
3669                clear_current_process_facts(state);
3670            });
3671            monitor.abort();
3672        }
3673        let _ = monitor.take();
3674    }
3675}
3676
3677#[derive(Debug)]
3678enum SupervisorCommand {
3679    Drain {
3680        reply: oneshot::Sender<Result<(), SuperviseError>>,
3681    },
3682    Retire {
3683        reply: oneshot::Sender<Result<(), SuperviseError>>,
3684    },
3685    Restart {
3686        /// Operator override for this one restart's drain budget, in ms. `None`
3687        /// uses the module's configured/default budget; `Some(0)` cuts
3688        /// immediately (wedge bounce: a stuck request never settles, so
3689        /// waiting only delays recovery).
3690        drain_timeout_ms: Option<u64>,
3691        /// The module's `spawn_generation` when the request was received, before
3692        /// it waited in the command queue. A queued restart whose module has
3693        /// since spawned a newer process is already satisfied (see the handler).
3694        received_at_generation: u64,
3695        /// When the request entered the command queue, so the handler can log
3696        /// how long it waited behind the loop's other work.
3697        queued_at: Instant,
3698        reply: oneshot::Sender<Result<(), SuperviseError>>,
3699    },
3700    Reload {
3701        reply: oneshot::Sender<Result<(), SuperviseError>>,
3702    },
3703    SetEnabled {
3704        enabled: bool,
3705        reply: oneshot::Sender<Result<bool, SuperviseError>>,
3706    },
3707    UpdateConfiguration {
3708        spec: ModuleSpec,
3709        health: HealthConfig,
3710        /// Per-module drain override from the new config; `None` re-resolves to
3711        /// the supervisor-wide default.
3712        drain_timeout_ms: Option<u64>,
3713        reply: oneshot::Sender<()>,
3714    },
3715    Swap {
3716        /// How long the candidate may take to register and declare itself
3717        /// ready. `None` uses [`DEFAULT_SWAP_READY_TIMEOUT`].
3718        ready_timeout: Option<Duration>,
3719        /// Answered at cutover or failure; the incumbent's drain follows.
3720        reply: oneshot::Sender<Result<(), SuperviseError>>,
3721    },
3722}
3723
3724#[derive(Debug)]
3725pub enum SuperviseError {
3726    InvalidSpec {
3727        reason: String,
3728    },
3729    Spawn {
3730        program: PathBuf,
3731        source: io::Error,
3732        cgroup_path: Option<PathBuf>,
3733    },
3734    Cgroup {
3735        module_id: String,
3736        source: io::Error,
3737    },
3738    /// CSPRNG failure generating a reserved module's launch nonce. Fail loud rather
3739    /// than spawn a reserved module without its identity binding.
3740    LaunchNonce {
3741        reason: String,
3742    },
3743    Wait {
3744        module_id: String,
3745        source: io::Error,
3746    },
3747    Kill {
3748        module_id: String,
3749        source: io::Error,
3750    },
3751    Forwarding(ForwardingError),
3752    Registry(RegistryError),
3753    ReloadUnavailable {
3754        module_id: String,
3755        reason: String,
3756    },
3757    /// An operator restart/reload was requested for a module that is currently
3758    /// disabled. Restart/reload cycle a *running* module; a disabled module must
3759    /// be explicitly re-enabled (set_enabled(true)) rather than silently started
3760    /// by a restart, so these commands are rejected instead of re-enabling it.
3761    Disabled {
3762        module_id: String,
3763    },
3764    ReloadFailed {
3765        module_id: String,
3766        reason: String,
3767    },
3768    RegistrationStillActive {
3769        module_id: String,
3770        waited: Duration,
3771    },
3772    StatePoisoned {
3773        module_id: Option<String>,
3774    },
3775    CommandClosed {
3776        module_id: String,
3777    },
3778    /// A restart or reload arrived while a swap's candidate was warming. The
3779    /// swap owns the module until it cuts over or fails; a stop or disable
3780    /// would have aborted it instead.
3781    SwapInProgress {
3782        module_id: String,
3783    },
3784    /// A swap was refused before anything was spawned.
3785    SwapRefused {
3786        module_id: String,
3787        reason: SwapRefusal,
3788    },
3789    /// A swap spawned a candidate and gave up on it. The candidate has been
3790    /// killed and its slot freed; the incumbent was left serving and was never
3791    /// drained, except in the one `CutoverLost` case described on that arm.
3792    SwapFailed {
3793        module_id: String,
3794        arm: SwapFailureArm,
3795        detail: String,
3796        /// How the candidate exited, when it exited on its own before the
3797        /// supervisor gave up on it.
3798        candidate_exit: Option<ExitReport>,
3799    },
3800}
3801
3802/// Why a swap was refused before a candidate was spawned.
3803#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3804pub enum SwapRefusal {
3805    /// The module's config does not declare `overlap: "safe"`.
3806    OverlapExclusive,
3807    /// The module is not registered, so there is no incumbent to keep serving
3808    /// and nothing a swap would improve on; a plain restart is the tool.
3809    NotRegistered,
3810    /// The module does not speak the subc wire, so a candidate could never
3811    /// register or declare itself ready.
3812    ProtocolNone,
3813    /// The supervisor lacks the forwarding table (to cut routes over) or the
3814    /// shared handle (to admit the candidate's HELLO) that a swap needs.
3815    NotConfigured,
3816    /// A swap is already open for this module.
3817    AlreadySwapping,
3818}
3819
3820impl SwapRefusal {
3821    pub fn as_str(self) -> &'static str {
3822        match self {
3823            Self::OverlapExclusive => "overlap_exclusive",
3824            Self::NotRegistered => "not_registered",
3825            Self::ProtocolNone => "protocol_none",
3826            Self::NotConfigured => "not_configured",
3827            Self::AlreadySwapping => "already_swapping",
3828        }
3829    }
3830}
3831
3832/// Which failure arm ended a swap. Every arm but one leaves the incumbent
3833/// serving and undrained; see `CutoverLost`.
3834#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3835pub enum SwapFailureArm {
3836    /// The candidate process could not be started.
3837    SpawnFailed,
3838    /// The candidate did not register within the readiness budget.
3839    NeverRegistered,
3840    /// The candidate registered but did not declare itself ready in time.
3841    NeverReady,
3842    /// The candidate exited before cutover.
3843    CandidateExited,
3844    /// The candidate declared itself ready but failed its health probe.
3845    CandidateUnhealthy,
3846    /// An operator stop, disable or retire arrived while the candidate warmed.
3847    /// The candidate was killed and the operator's command then carried out on
3848    /// the incumbent.
3849    Interrupted,
3850    /// The candidate's connection closed at the moment of cutover. If it
3851    /// closed before forwarding moved, the incumbent is untouched. If it closed
3852    /// between the forwarding and registry halves of cutover, forwarding can no
3853    /// longer route to the incumbent, so the module is restarted plainly.
3854    CutoverLost,
3855}
3856
3857impl SwapFailureArm {
3858    pub fn as_str(self) -> &'static str {
3859        match self {
3860            Self::SpawnFailed => "spawn_failed",
3861            Self::NeverRegistered => "never_registered",
3862            Self::NeverReady => "never_ready",
3863            Self::CandidateExited => "candidate_exited",
3864            Self::CandidateUnhealthy => "candidate_unhealthy",
3865            Self::Interrupted => "interrupted",
3866            Self::CutoverLost => "cutover_lost",
3867        }
3868    }
3869}
3870
3871impl fmt::Display for SuperviseError {
3872    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3873        match self {
3874            Self::InvalidSpec { reason } => write!(f, "invalid module spec: {reason}"),
3875            Self::Spawn {
3876                program,
3877                source,
3878                cgroup_path: Some(cgroup_path),
3879            } => write!(
3880                f,
3881                "failed to place module in cgroup '{}' while spawning '{}': {source}",
3882                cgroup_path.display(),
3883                program.display()
3884            ),
3885            Self::Spawn {
3886                program,
3887                source,
3888                cgroup_path: None,
3889            } => write!(
3890                f,
3891                "failed to spawn module '{}': {source}",
3892                program.display()
3893            ),
3894            Self::Cgroup { module_id, source } => {
3895                write!(
3896                    f,
3897                    "failed to prepare cgroup for module '{module_id}': {source}"
3898                )
3899            }
3900            Self::LaunchNonce { reason } => {
3901                write!(
3902                    f,
3903                    "failed to generate reserved-module launch nonce: {reason}"
3904                )
3905            }
3906            Self::Wait { module_id, source } => {
3907                write!(f, "failed to wait for module '{module_id}': {source}")
3908            }
3909            Self::Kill { module_id, source } => {
3910                write!(f, "failed to kill module '{module_id}': {source}")
3911            }
3912            Self::Forwarding(err) => write!(f, "forwarding error: {err}"),
3913            Self::Registry(err) => write!(f, "registry error: {err}"),
3914            Self::ReloadUnavailable { module_id, reason } => {
3915                write!(f, "reload unavailable for module '{module_id}': {reason}")
3916            }
3917            Self::Disabled { module_id } => {
3918                write!(
3919                    f,
3920                    "module '{module_id}' is disabled; enable it before restart or reload"
3921                )
3922            }
3923            Self::ReloadFailed { module_id, reason } => {
3924                write!(f, "reload failed for module '{module_id}': {reason}")
3925            }
3926            Self::RegistrationStillActive { module_id, waited } => write!(
3927                f,
3928                "module '{module_id}' registration remained active after waiting {waited:?}"
3929            ),
3930            Self::StatePoisoned { module_id } => match module_id {
3931                Some(module_id) => {
3932                    write!(f, "supervisor state for module '{module_id}' was poisoned")
3933                }
3934                None => write!(f, "supervisor state was poisoned"),
3935            },
3936            Self::CommandClosed { module_id } => {
3937                write!(
3938                    f,
3939                    "supervisor command channel for module '{module_id}' is closed"
3940                )
3941            }
3942            Self::SwapInProgress { module_id } => write!(
3943                f,
3944                "module '{module_id}' is being swapped; retry once the swap has cut over or failed, or stop the module to abort the swap"
3945            ),
3946            Self::SwapRefused { module_id, reason } => match reason {
3947                SwapRefusal::OverlapExclusive => write!(
3948                    f,
3949                    "module '{module_id}' is declared overlap: \"exclusive\" (the default): two processes of it must not run at once, so it cannot be swapped; use a plain restart, or declare overlap: \"safe\" in its config if it really tolerates a second process"
3950                ),
3951                SwapRefusal::NotRegistered => write!(
3952                    f,
3953                    "module '{module_id}' is not registered, so there is no serving process to keep while a replacement warms; use a plain restart"
3954                ),
3955                SwapRefusal::ProtocolNone => write!(
3956                    f,
3957                    "module '{module_id}' is protocol: \"none\" and never registers, so a swap could never see its replacement become ready; use a plain restart"
3958                ),
3959                SwapRefusal::NotConfigured => write!(
3960                    f,
3961                    "module '{module_id}' cannot be swapped: the supervisor was built without the forwarding table or shared handle a swap needs"
3962                ),
3963                SwapRefusal::AlreadySwapping => {
3964                    write!(f, "module '{module_id}' is already being swapped")
3965                }
3966            },
3967            Self::SwapFailed {
3968                module_id,
3969                arm,
3970                detail,
3971                ..
3972            } => write!(
3973                f,
3974                "swap of module '{module_id}' failed ({}): {detail}; the running process was left serving",
3975                arm.as_str()
3976            ),
3977        }
3978    }
3979}
3980
3981impl Error for SuperviseError {
3982    fn source(&self) -> Option<&(dyn Error + 'static)> {
3983        match self {
3984            Self::Spawn { source, .. }
3985            | Self::Cgroup { source, .. }
3986            | Self::Wait { source, .. }
3987            | Self::Kill { source, .. } => Some(source),
3988            Self::Forwarding(err) => Some(err),
3989            Self::Registry(err) => Some(err),
3990            Self::LaunchNonce { .. }
3991            | Self::InvalidSpec { .. }
3992            | Self::ReloadUnavailable { .. }
3993            | Self::Disabled { .. }
3994            | Self::ReloadFailed { .. }
3995            | Self::RegistrationStillActive { .. }
3996            | Self::StatePoisoned { .. }
3997            | Self::CommandClosed { .. }
3998            | Self::SwapInProgress { .. }
3999            | Self::SwapRefused { .. }
4000            | Self::SwapFailed { .. } => None,
4001        }
4002    }
4003}
4004
4005pub(crate) fn validate_spec(spec: &ModuleSpec) -> Result<(), SuperviseError> {
4006    if spec.module_id.trim().is_empty() {
4007        return Err(SuperviseError::InvalidSpec {
4008            reason: "module_id must not be empty".to_string(),
4009        });
4010    }
4011
4012    Ok(())
4013}
4014
4015#[derive(Debug, Default)]
4016struct HealthProbeRuntime {
4017    configured_health: Option<HealthConfig>,
4018    registered_connection: Option<crate::ConnectionId>,
4019    advertised: bool,
4020    next_probe_at: Option<Instant>,
4021    probe_index: u64,
4022}
4023
4024fn running_protocol(spec: &ModuleSpec, snapshot: &SharedSnapshot) -> ModuleProtocol {
4025    lock_snapshot(snapshot)
4026        .ok()
4027        .and_then(|state| state.spawned_protocol)
4028        .unwrap_or(spec.protocol)
4029}
4030
4031impl HealthProbeRuntime {
4032    fn refresh_registration(
4033        &mut self,
4034        spec: &ModuleSpec,
4035        runtime: &SupervisorRuntimeConfig,
4036        registry: &Registry,
4037        snapshot: &SharedSnapshot,
4038    ) {
4039        if self.configured_health.as_ref() != Some(&runtime.health) {
4040            self.configured_health = Some(runtime.health.clone());
4041            self.next_probe_at = None;
4042            self.registered_connection = None;
4043            self.probe_index = 0;
4044        }
4045        // A non-wire process never registers. Only an explicitly configured
4046        // HTTP endpoint can arm its health probe; an absent HELLO is not a
4047        // health failure for that kind of process.
4048        if running_protocol(spec, snapshot) == ModuleProtocol::None {
4049            self.registered_connection = None;
4050            self.advertised = runtime.health.http.is_some();
4051            if !self.advertised {
4052                self.next_probe_at = None;
4053                let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4054                    state.health = ModuleHealthStatus::default();
4055                });
4056            } else if self.next_probe_at.is_none() {
4057                self.next_probe_at = Some(
4058                    Instant::now()
4059                        + jittered_health_delay(&spec.module_id, 0, runtime.health.cadence),
4060                );
4061            }
4062            return;
4063        }
4064
4065        let registration = match registry.get_module(&spec.module_id) {
4066            Ok(registration) => registration,
4067            Err(err) => {
4068                warn!(module_id = %spec.module_id, error = %err, "health prober could not read registry");
4069                self.advertised = false;
4070                self.next_probe_at = None;
4071                return;
4072            }
4073        };
4074
4075        let Some(registration) = registration else {
4076            self.registered_connection = None;
4077            self.advertised = false;
4078            self.next_probe_at = None;
4079            return;
4080        };
4081
4082        let advertised = registration
4083            .control_ops
4084            .iter()
4085            .any(|op| op == MODULE_CONTROL_OP_HEALTH_CHECK);
4086        if !advertised {
4087            self.registered_connection = Some(registration.connection_id);
4088            self.advertised = false;
4089            self.next_probe_at = None;
4090            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4091                state.health.status = SupervisorHealthStatus::Unknown;
4092                state.health.consecutive_failures = 0;
4093                state.health.last_probe_ms = None;
4094                state.health.detail = None;
4095                state.health.metrics = None;
4096            });
4097            return;
4098        }
4099
4100        let reregistered = self.registered_connection != Some(registration.connection_id);
4101        self.registered_connection = Some(registration.connection_id);
4102        self.advertised = true;
4103        if reregistered || self.next_probe_at.is_none() {
4104            self.probe_index = 0;
4105            self.next_probe_at = Some(
4106                Instant::now() + jittered_health_delay(&spec.module_id, 0, runtime.health.cadence),
4107            );
4108            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4109                state.health.status = SupervisorHealthStatus::Unknown;
4110                state.health.consecutive_failures = 0;
4111                state.health.detail = None;
4112                state.health.metrics = None;
4113            });
4114        }
4115    }
4116
4117    fn wake_after(&self) -> Duration {
4118        if !self.advertised {
4119            return REGISTRY_RELEASE_POLL;
4120        }
4121        self.next_probe_at
4122            .map(|next| next.saturating_duration_since(Instant::now()))
4123            .unwrap_or(REGISTRY_RELEASE_POLL)
4124    }
4125
4126    fn due(&self) -> bool {
4127        self.advertised
4128            && self
4129                .next_probe_at
4130                .is_some_and(|next| Instant::now() >= next)
4131    }
4132
4133    fn schedule_next(&mut self, spec: &ModuleSpec, cadence: Duration) {
4134        self.probe_index = self.probe_index.wrapping_add(1);
4135        self.next_probe_at = Some(
4136            Instant::now() + jittered_health_delay(&spec.module_id, self.probe_index, cadence),
4137        );
4138    }
4139}
4140
4141/// What a failed health probe actually OBSERVED, kept apart from how it reads.
4142///
4143/// This was a struct with a single `message: String`, and every one of the
4144/// fifteen construction sites collapsed into it. Each site knows exactly what it
4145/// saw -- the lane is gone, the module did not answer in time, the module
4146/// answered with the wrong thing -- and `handle_health_probe_failure` then
4147/// treated all of them identically: increment a counter, compare to a threshold,
4148/// restart the module. THE DISTINCTION EXISTED AT EVERY CALL SITE AND WAS
4149/// DESTROYED BEFORE THE DECISION THAT NEEDED IT.
4150///
4151/// The distinction that matters is not severity, it is EVIDENTIAL WEIGHT:
4152///
4153/// * `LaneDead` is PROOF. The module's control connection is gone; nothing will
4154///   answer on it again.
4155/// * `NoAnswer` is ABSENCE OF EVIDENCE. It is consistent with a wedged module
4156///   AND with a perfectly healthy one that lost a CPU race -- which is what
4157///   happens under machine load, and is how this supervisor killed a healthy
4158///   module three times in one day.
4159/// * `BadAnswer` proves the module is ALIVE. It replied; the reply was wrong.
4160///   Restarting on it is defensible, but it is not the silence case and should
4161///   never be counted as one.
4162/// * `Misconfigured` is a daemon-side fault. The module has not been asked
4163///   anything, so it cannot be evidence about the module at all.
4164///
4165/// The asymmetry is the whole point: under saturation the WEAKEST signal is the
4166/// one that fires most often, and while every variant collapsed into one string
4167/// it carried the same weight as the strongest.
4168///
4169/// LIVE BEHAVIOUR TODAY, stated here because this doc block describes the
4170/// DESIGN and a reader stopping at it gets the build backwards: the restart
4171/// decision does NOT yet consult this classification -- consecutive `NoAnswer`
4172/// probes still increment the failure streak and drive escalation at the
4173/// threshold (see `is_proof_of_death` below for why that is deliberate and
4174/// what gates the change). Absence of evidence restarts modules today.
4175#[derive(Debug)]
4176enum HealthProbeEvidence {
4177    /// The module's control lane is gone. Proof of death.
4178    LaneDead,
4179    /// No reply within the deadline. Proves nothing about the module's state.
4180    NoAnswer,
4181    /// The module replied, but not with a usable health report. Proves it is alive.
4182    BadAnswer,
4183    /// The daemon could not ask. Says nothing about the module.
4184    Misconfigured,
4185}
4186
4187#[derive(Debug)]
4188struct HealthProbeError {
4189    evidence: HealthProbeEvidence,
4190    message: String,
4191}
4192
4193impl HealthProbeError {
4194    fn lane_dead(message: impl Into<String>) -> Self {
4195        Self::with(HealthProbeEvidence::LaneDead, message)
4196    }
4197
4198    fn no_answer(message: impl Into<String>) -> Self {
4199        Self::with(HealthProbeEvidence::NoAnswer, message)
4200    }
4201
4202    fn bad_answer(message: impl Into<String>) -> Self {
4203        Self::with(HealthProbeEvidence::BadAnswer, message)
4204    }
4205
4206    fn misconfigured(message: impl Into<String>) -> Self {
4207        Self::with(HealthProbeEvidence::Misconfigured, message)
4208    }
4209
4210    fn with(evidence: HealthProbeEvidence, message: impl Into<String>) -> Self {
4211        Self {
4212            evidence,
4213            message: message.into(),
4214        }
4215    }
4216
4217    /// Whether this observation is proof the module cannot serve.
4218    ///
4219    /// Only `LaneDead` qualifies. `NoAnswer` is deliberately excluded: it is the
4220    /// variant that fires under CPU starvation, and treating it as proof is the
4221    /// defect this enum exists to make impossible to reintroduce silently.
4222    ///
4223    /// NOT YET CONSULTED BY THE RESTART DECISION, deliberately. Requiring proof
4224    /// to restart also needs a bound for the case it excludes -- a genuinely
4225    /// wedged module, alive but never answering -- and that bound must come from
4226    /// the distribution of real late-answer latencies, which nothing measures
4227    /// yet. Landing the classification first makes the later change a one-line
4228    /// decision against evidence that already exists, rather than two unproven
4229    /// changes at once.
4230    #[allow(dead_code)]
4231    fn is_proof_of_death(&self) -> bool {
4232        matches!(self.evidence, HealthProbeEvidence::LaneDead)
4233    }
4234
4235    /// Short stable label for logs and the health snapshot.
4236    ///
4237    /// An operator reading `ck health` currently cannot tell "the module is gone"
4238    /// from "the module did not answer in five seconds", because both render as
4239    /// prose in the same field. These labels are what make the two
4240    /// distinguishable at a glance, and they are what a later restart-policy
4241    /// change will be argued from.
4242    fn label(&self) -> &'static str {
4243        match self.evidence {
4244            HealthProbeEvidence::LaneDead => "lane-dead",
4245            HealthProbeEvidence::NoAnswer => "no-answer",
4246            HealthProbeEvidence::BadAnswer => "bad-answer",
4247            HealthProbeEvidence::Misconfigured => "daemon-misconfigured",
4248        }
4249    }
4250}
4251
4252impl fmt::Display for HealthProbeError {
4253    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
4254        f.write_str(&self.message)
4255    }
4256}
4257
4258async fn run_health_probe_cycle(
4259    spec: &ModuleSpec,
4260    runtime: &SupervisorRuntimeConfig,
4261    registry: &Registry,
4262    process_liveness: &SupervisorProcessLiveness,
4263    snapshot: &SharedSnapshot,
4264    child: &mut Option<SupervisedChild>,
4265) {
4266    let now_ms = unix_ms_now();
4267    let http = (running_protocol(spec, snapshot) == ModuleProtocol::None)
4268        .then_some(runtime.health.http.as_deref())
4269        .flatten();
4270    let result = match http {
4271        Some(url) => probe_http_health(url, runtime.health.deadline).await,
4272        None => probe_module_health(&spec.module_id, runtime, None).await,
4273    };
4274    match result {
4275        Ok(report) => {
4276            handle_health_report(
4277                spec,
4278                runtime,
4279                registry,
4280                process_liveness,
4281                snapshot,
4282                child,
4283                report,
4284                now_ms,
4285            )
4286            .await;
4287        }
4288        Err(err) => {
4289            if http.is_some() {
4290                let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4291                    state.health.status = SupervisorHealthStatus::Failing;
4292                });
4293            }
4294            handle_health_probe_failure(
4295                spec,
4296                runtime,
4297                registry,
4298                process_liveness,
4299                snapshot,
4300                child,
4301                err,
4302                now_ms,
4303            )
4304            .await;
4305        }
4306    }
4307}
4308
4309pub(crate) struct HttpProbeTarget<'a> {
4310    address: std::net::SocketAddr,
4311    localhost: bool,
4312    authority: &'a str,
4313    path: String,
4314}
4315
4316/// Resolve only literal loopback endpoints, without DNS, redirects, proxies,
4317/// or TLS. A URL cannot turn a local health check into an outbound connection.
4318pub(crate) fn parse_http_probe_url(url: &str) -> Result<HttpProbeTarget<'_>, String> {
4319    if url.bytes().any(|b| b <= b' ' || b == 127) || url.contains('#') {
4320        return Err("must not contain whitespace, controls, or a fragment".into());
4321    }
4322    let rest = url
4323        .strip_prefix("http://")
4324        .ok_or("must use plain http://")?;
4325    let split = rest.find(['/', '?']).unwrap_or(rest.len());
4326    let (authority, suffix) = rest.split_at(split);
4327    let (host, port) = if let Some(rest) = authority.strip_prefix("[::1]") {
4328        ("::1", rest)
4329    } else {
4330        let split = authority.find(':').unwrap_or(authority.len());
4331        authority.split_at(split)
4332    };
4333    let ip: std::net::IpAddr = match host {
4334        "127.0.0.1" | "localhost" => std::net::Ipv4Addr::LOCALHOST.into(),
4335        "::1" => std::net::Ipv6Addr::LOCALHOST.into(),
4336        _ => return Err("host must be 127.0.0.1, [::1], or localhost".into()),
4337    };
4338    let port = if port.is_empty() {
4339        80
4340    } else {
4341        port.strip_prefix(':')
4342            .and_then(|p| p.parse::<u16>().ok())
4343            .filter(|p| *p > 0)
4344            .ok_or("must have a valid nonzero TCP port")?
4345    };
4346    let path = if suffix.is_empty() {
4347        "/".into()
4348    } else if suffix.starts_with('?') {
4349        format!("/{suffix}")
4350    } else {
4351        suffix.into()
4352    };
4353    Ok(HttpProbeTarget {
4354        address: std::net::SocketAddr::new(ip, port),
4355        localhost: host == "localhost",
4356        authority,
4357        path,
4358    })
4359}
4360
4361async fn probe_http_health(
4362    url: &str,
4363    deadline: Duration,
4364) -> Result<HealthReport, HealthProbeError> {
4365    use tokio::io::{AsyncReadExt, AsyncWriteExt, BufReader};
4366    let target = parse_http_probe_url(url).map_err(HealthProbeError::misconfigured)?;
4367    // Keep partial diagnostics outside the timed future so cancellation does
4368    // not discard a status line or body bytes already received.
4369    let mut response_status = String::new();
4370    let mut body = Vec::new();
4371    let probe = async {
4372        // Resolve localhost ourselves so a hosts-file override cannot turn
4373        // this into an outbound request, while IPv6-only local servers work.
4374        let connection = match tokio::net::TcpStream::connect(target.address).await {
4375            Err(_) if target.localhost => {
4376                tokio::net::TcpStream::connect((
4377                    std::net::Ipv6Addr::LOCALHOST,
4378                    target.address.port(),
4379                ))
4380                .await
4381            }
4382            result => result,
4383        };
4384        let mut stream = connection.map_err(|error| {
4385            HealthProbeError::no_answer(format!("HTTP connect failed: {error}"))
4386        })?;
4387        stream.write_all(format!("GET {} HTTP/1.1\r\nHost: {}\r\nAccept-Encoding: identity\r\nConnection: close\r\n\r\n", target.path, target.authority).as_bytes()).await
4388            .map_err(|error| HealthProbeError::no_answer(format!("HTTP write failed: {error}")))?;
4389        let mut reader = BufReader::new(stream);
4390        let mut budget = 16 * 1024;
4391        let status = http_line(&mut reader, &mut budget).await?;
4392        let mut words = status.split_ascii_whitespace();
4393        let version = words.next();
4394        let code = words
4395            .next()
4396            .filter(|word| word.len() == 3)
4397            .and_then(|word| word.parse::<u16>().ok());
4398        if !matches!(version, Some("HTTP/1.1" | "HTTP/1.0"))
4399            || !code.is_some_and(|code| (100..600).contains(&code))
4400        {
4401            return Err(HealthProbeError::bad_answer(format!(
4402                "invalid HTTP status: {status}"
4403            )));
4404        }
4405        let code = code.expect("validated status code");
4406        response_status = status.clone();
4407        let mut length = None;
4408        let mut chunked = false;
4409        loop {
4410            let line = http_line(&mut reader, &mut budget).await?;
4411            if line.is_empty() {
4412                break;
4413            }
4414            if let Some((name, value)) = line.split_once(':') {
4415                if name.eq_ignore_ascii_case("content-length") {
4416                    length = Some(value.trim().parse::<u64>().map_err(|_| {
4417                        HealthProbeError::bad_answer("invalid HTTP Content-Length")
4418                    })?);
4419                } else if name.eq_ignore_ascii_case("transfer-encoding") {
4420                    chunked = value.trim().eq_ignore_ascii_case("chunked");
4421                }
4422            }
4423        }
4424        if chunked {
4425            while body.len() < 200 {
4426                let line = http_line(&mut reader, &mut budget).await?;
4427                let size = u64::from_str_radix(line.split(';').next().unwrap_or("").trim(), 16)
4428                    .map_err(|_| HealthProbeError::bad_answer("invalid HTTP chunk size"))?;
4429                if size == 0 {
4430                    break;
4431                }
4432                let count = size.min((200 - body.len()) as u64) as usize;
4433                let start = body.len();
4434                (&mut reader)
4435                    .take(count as u64)
4436                    .read_to_end(&mut body)
4437                    .await
4438                    .map_err(|error| {
4439                        HealthProbeError::no_answer(format!("HTTP body read failed: {error}"))
4440                    })?;
4441                if body.len() - start != count {
4442                    return Err(HealthProbeError::bad_answer("truncated HTTP chunk"));
4443                }
4444                if size > count as u64 || body.len() == 200 {
4445                    break;
4446                }
4447                if !http_line(&mut reader, &mut budget).await?.is_empty() {
4448                    return Err(HealthProbeError::bad_answer("invalid HTTP chunk delimiter"));
4449                }
4450            }
4451        } else {
4452            reader
4453                .take(length.unwrap_or(200).min(200))
4454                .read_to_end(&mut body)
4455                .await
4456                .map_err(|error| {
4457                    HealthProbeError::no_answer(format!("HTTP body read failed: {error}"))
4458                })?;
4459        }
4460        if (200..300).contains(&code) {
4461            Ok(HealthReport::ok())
4462        } else {
4463            Err(HealthProbeError::bad_answer(
4464                "HTTP health endpoint returned non-2xx",
4465            ))
4466        }
4467    };
4468    let mut result = timeout(deadline, probe).await.unwrap_or_else(|_| {
4469        Err(HealthProbeError::no_answer(format!(
4470            "HTTP probe timed out after {deadline:?}"
4471        )))
4472    });
4473    if let Err(error) = &mut result {
4474        if !response_status.is_empty() {
4475            error.message = format!(
4476                "{}; {response_status}: {}",
4477                error.message,
4478                String::from_utf8_lossy(&body)
4479            );
4480        }
4481    }
4482    result
4483}
4484
4485async fn http_line(
4486    reader: &mut tokio::io::BufReader<tokio::net::TcpStream>,
4487    remaining: &mut usize,
4488) -> Result<String, HealthProbeError> {
4489    use tokio::io::{AsyncBufReadExt, AsyncReadExt};
4490    let mut line = Vec::new();
4491    (&mut *reader)
4492        .take(*remaining as u64)
4493        .read_until(b'\n', &mut line)
4494        .await
4495        .map_err(|error| {
4496            HealthProbeError::no_answer(format!("HTTP header read failed: {error}"))
4497        })?;
4498    *remaining -= line.len();
4499    if !line.ends_with(b"\r\n") {
4500        return Err(HealthProbeError::bad_answer(
4501            "HTTP headers are incomplete or exceed 16 KiB",
4502        ));
4503    }
4504    line.truncate(line.len() - 2);
4505    String::from_utf8(line).map_err(|_| HealthProbeError::bad_answer("HTTP header is not UTF-8"))
4506}
4507
4508async fn probe_module_health(
4509    module_id: &str,
4510    runtime: &SupervisorRuntimeConfig,
4511    drain_deadline: Option<Instant>,
4512) -> Result<HealthReport, HealthProbeError> {
4513    let Some(forwarding) = runtime.forwarding.as_ref() else {
4514        return Err(HealthProbeError::misconfigured(
4515            "supervisor was not configured with a forwarding table",
4516        ));
4517    };
4518    let probe_started_at = Instant::now();
4519    let mut deadline = probe_started_at + runtime.health.deadline;
4520    if let Some(drain_deadline) = drain_deadline {
4521        deadline = deadline.min(drain_deadline);
4522    }
4523    let pending = if drain_deadline.is_some() {
4524        forwarding.begin_drain_health_probe_rpc_for(
4525            module_id,
4526            MODULE_CONTROL_OP_HEALTH_CHECK,
4527            probe_started_at,
4528            deadline,
4529        )
4530    } else {
4531        forwarding.begin_health_probe_rpc_for(
4532            module_id,
4533            MODULE_CONTROL_OP_HEALTH_CHECK,
4534            probe_started_at,
4535            deadline,
4536        )
4537    }
4538    .map_err(|err| {
4539        // The endpoint is not registered, so there is no live control lane to
4540        // ask. That is the module being absent, not slow.
4541        HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
4542    })?;
4543    await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
4544}
4545
4546/// [`probe_module_health`] for one endpoint rather than the id's active one.
4547///
4548/// A swap probes two processes that no by-id lookup reaches: its candidate
4549/// before cutover, and its superseded incumbent (for busy gauges) while the
4550/// incumbent drains. `deadline_cap` bounds the probe the way a drain deadline
4551/// bounds the by-id drain probe.
4552async fn probe_endpoint_health(
4553    endpoint: crate::ModuleEndpointId,
4554    runtime: &SupervisorRuntimeConfig,
4555    deadline_cap: Option<Instant>,
4556) -> Result<HealthReport, HealthProbeError> {
4557    let Some(forwarding) = runtime.forwarding.as_ref() else {
4558        return Err(HealthProbeError::misconfigured(
4559            "supervisor was not configured with a forwarding table",
4560        ));
4561    };
4562    let probe_started_at = Instant::now();
4563    let mut deadline = probe_started_at + runtime.health.deadline;
4564    if let Some(cap) = deadline_cap {
4565        deadline = deadline.min(cap);
4566    }
4567    let pending = forwarding
4568        .begin_endpoint_health_probe_rpc_for(
4569            endpoint,
4570            MODULE_CONTROL_OP_HEALTH_CHECK,
4571            probe_started_at,
4572            deadline,
4573        )
4574        .map_err(|err| {
4575            HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
4576        })?;
4577    await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
4578}
4579
4580/// Send a begun health probe and classify its answer.
4581async fn await_health_probe(
4582    forwarding: &ForwardingTable,
4583    pending: PendingModuleControlRpc,
4584    deadline: Instant,
4585    probe_budget: Duration,
4586) -> Result<HealthReport, HealthProbeError> {
4587    let PendingModuleControlRpc {
4588        endpoint,
4589        module_sink,
4590        negotiated_ver,
4591        corr,
4592        receiver,
4593    } = pending;
4594    let body = serde_json::to_vec(&ModuleControlRequest::HealthCheck {}).map_err(|err| {
4595        HealthProbeError::misconfigured(format!("failed to encode health.check: {err}"))
4596    })?;
4597    let frame = Frame::build_with_version(
4598        negotiated_ver,
4599        FrameType::Request,
4600        control_flags(),
4601        0,
4602        0,
4603        corr,
4604        body,
4605    )
4606    .map_err(|err| {
4607        HealthProbeError::misconfigured(format!("failed to build health.check frame: {err}"))
4608    })?;
4609
4610    // The enqueue itself must be bounded by the probe deadline: FrameSink.send
4611    // blocks waiting for capacity when the module's egress queue is full, and an
4612    // unbounded await here freezes the whole supervision actor (it stops polling
4613    // Child::wait and supervisor commands), making the module unrecoverable
4614    // in-band. On timeout the probe fails like any transport failure.
4615    match timeout_at(deadline, module_sink.send(frame)).await {
4616        Ok(Ok(())) => {}
4617        Ok(Err(err)) => {
4618            let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
4619            // A closed sink means the module's egress channel is gone -- the
4620            // receiving half is dropped when its connection tears down. Proof.
4621            return Err(HealthProbeError::lane_dead(format!(
4622                "failed to send health.check: {err}"
4623            )));
4624        }
4625        Err(_elapsed) => {
4626            let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
4627            // A full egress queue means the module is not draining its socket, which
4628            // is consistent with a wedged module AND with one whose reader is merely
4629            // starved. Silence, not proof.
4630            return Err(HealthProbeError::no_answer(
4631                "health.check send timed out before enqueue (module egress full)",
4632            ));
4633        }
4634    }
4635
4636    match timeout_at(deadline, receiver).await {
4637        // Each arm records WHAT WAS OBSERVED. Four of them are the module
4638        // demonstrably answering -- rejected, non-health, malformed, wrong op --
4639        // and those prove it is alive even though the probe failed.
4640        Ok(Ok(ModuleControlRpcOutcome::Response(response))) => {
4641            response.health_report().ok_or_else(|| {
4642                HealthProbeError::bad_answer("health.check RPC returned a non-health response")
4643            })
4644        }
4645        Ok(Ok(ModuleControlRpcOutcome::Rejected(body))) => Err(HealthProbeError::bad_answer(
4646            format!("health.check rejected: {}", body.message),
4647        )),
4648        Ok(Ok(ModuleControlRpcOutcome::ModuleGone(message))) => {
4649            Err(HealthProbeError::lane_dead(message))
4650        }
4651        Ok(Ok(ModuleControlRpcOutcome::MalformedResponse(message))) => {
4652            Err(HealthProbeError::bad_answer(message))
4653        }
4654        Ok(Ok(ModuleControlRpcOutcome::UnexpectedOp { expected, actual })) => {
4655            Err(HealthProbeError::bad_answer(format!(
4656                "expected module-control op '{expected}', got '{actual}'"
4657            )))
4658        }
4659        // A reply that crosses the deadline before this waiter observes it is
4660        // still proof of life. The forwarding path records its end-to-end latency
4661        // before delivering this classification.
4662        Ok(Ok(ModuleControlRpcOutcome::DeadlineElapsed)) => Err(HealthProbeError::bad_answer(
4663            "module answered health.check after its daemon deadline",
4664        )),
4665        Ok(Err(_)) => Err(HealthProbeError::misconfigured(
4666            "health.check waiter was canceled before the module responded",
4667        )),
4668        Err(_) => {
4669            let _ = forwarding.tombstone_health_probe_rpc(endpoint, corr);
4670            Err(HealthProbeError::no_answer(format!(
4671                "module did not answer health.check within {probe_budget:?}"
4672            )))
4673        }
4674    }
4675}
4676
4677#[allow(clippy::too_many_arguments)]
4678async fn handle_health_report(
4679    spec: &ModuleSpec,
4680    runtime: &SupervisorRuntimeConfig,
4681    registry: &Registry,
4682    process_liveness: &SupervisorProcessLiveness,
4683    snapshot: &SharedSnapshot,
4684    child: &mut Option<SupervisedChild>,
4685    report: HealthReport,
4686    now_ms: u64,
4687) {
4688    let status = supervisor_health_status(report.status);
4689    let detail = report.detail.clone();
4690    let metrics = truncate_health_metrics(report.metrics);
4691    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4692        state.health.status = status;
4693        state.health.last_probe_ms = Some(now_ms);
4694        state.health.detail = detail.clone();
4695        state.health.metrics = metrics.clone();
4696        state.health.consecutive_failures = 0;
4697    });
4698
4699    let action = match report.status {
4700        HealthStatus::Ok => return,
4701        HealthStatus::Degraded => runtime.health.on_degraded,
4702        HealthStatus::Failing => runtime.health.on_failing,
4703    };
4704    apply_l3_health_action(
4705        spec,
4706        runtime,
4707        registry,
4708        process_liveness,
4709        snapshot,
4710        child,
4711        status,
4712        detail.as_deref(),
4713        action,
4714        now_ms,
4715    )
4716    .await;
4717}
4718
4719#[allow(clippy::too_many_arguments)]
4720async fn handle_health_probe_failure(
4721    spec: &ModuleSpec,
4722    runtime: &SupervisorRuntimeConfig,
4723    registry: &Registry,
4724    process_liveness: &SupervisorProcessLiveness,
4725    snapshot: &SharedSnapshot,
4726    child: &mut Option<SupervisedChild>,
4727    err: HealthProbeError,
4728    now_ms: u64,
4729) {
4730    let threshold = runtime.health.failure_threshold.max(1);
4731    let mut failures = 0;
4732    // Carry the evidence class into the operator-visible detail. Without it,
4733    // "module did not answer within 5s" and "the control lane is gone" are two
4734    // prose strings in the same field, and the reader has to know the codebase to
4735    // tell which one is proof of anything.
4736    let detail = format!("[{}] {err}", err.label());
4737    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4738        state.health.last_probe_ms = Some(now_ms);
4739        state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
4740        state.health.detail = Some(detail.clone());
4741        state.health.metrics = None;
4742        failures = state.health.consecutive_failures;
4743    });
4744
4745    if failures < threshold {
4746        warn!(
4747            module_id = %spec.module_id,
4748            consecutive_failures = failures,
4749            threshold,
4750            evidence = err.label(),
4751            detail = %detail,
4752            "health.check probe failed"
4753        );
4754        return;
4755    }
4756
4757    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4758        state.state = ModuleState::Unresponsive;
4759        state.health.status = SupervisorHealthStatus::Unresponsive;
4760    });
4761    // The evidence class is logged at the kill site because this is the line an
4762    // operator reads after an unexplained restart. A streak of `no-answer` under
4763    // machine load is the known false-positive shape; a `lane-dead` is not.
4764    if runtime.health.critical {
4765        error!(
4766            module_id = %spec.module_id,
4767            status = "unresponsive",
4768            evidence = err.label(),
4769            detail = %detail,
4770            "critical module health alert"
4771        );
4772    } else {
4773        warn!(
4774            module_id = %spec.module_id,
4775            status = "unresponsive",
4776            evidence = err.label(),
4777            detail = %detail,
4778            "module health threshold breached"
4779        );
4780    }
4781    if let Err(err) = health_restart_child(
4782        spec,
4783        runtime,
4784        registry,
4785        process_liveness,
4786        snapshot,
4787        child,
4788        SupervisorHealthStatus::Unresponsive,
4789        Some(&detail),
4790        now_ms,
4791    )
4792    .await
4793    {
4794        error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
4795    }
4796}
4797
4798#[allow(clippy::too_many_arguments)]
4799async fn apply_l3_health_action(
4800    spec: &ModuleSpec,
4801    runtime: &SupervisorRuntimeConfig,
4802    registry: &Registry,
4803    process_liveness: &SupervisorProcessLiveness,
4804    snapshot: &SharedSnapshot,
4805    child: &mut Option<SupervisedChild>,
4806    status: SupervisorHealthStatus,
4807    detail: Option<&str>,
4808    action: HealthAction,
4809    now_ms: u64,
4810) {
4811    record_health_action(snapshot, &spec.module_id, action.to_string(), now_ms);
4812    match action {
4813        HealthAction::Report => {
4814            info!(
4815                module_id = %spec.module_id,
4816                status = ?status,
4817                detail,
4818                "module reported non-ok health"
4819            );
4820        }
4821        HealthAction::Alert => {
4822            error!(
4823                module_id = %spec.module_id,
4824                status = ?status,
4825                detail,
4826                "module health alert"
4827            );
4828        }
4829        HealthAction::Restart => {
4830            if let Err(err) = health_restart_child(
4831                spec,
4832                runtime,
4833                registry,
4834                process_liveness,
4835                snapshot,
4836                child,
4837                status,
4838                detail,
4839                now_ms,
4840            )
4841            .await
4842            {
4843                error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
4844            }
4845        }
4846    }
4847}
4848
4849#[allow(clippy::too_many_arguments)]
4850async fn health_restart_child(
4851    spec: &ModuleSpec,
4852    runtime: &SupervisorRuntimeConfig,
4853    registry: &Registry,
4854    process_liveness: &SupervisorProcessLiveness,
4855    snapshot: &SharedSnapshot,
4856    child: &mut Option<SupervisedChild>,
4857    status: SupervisorHealthStatus,
4858    detail: Option<&str>,
4859    now_ms: u64,
4860) -> Result<(), SuperviseError> {
4861    let (enabled, schedule) = {
4862        let mut state = lock_snapshot(snapshot)?;
4863        let enabled = state.enabled;
4864        let schedule = if enabled {
4865            state.next_crash_restart(&runtime.restart_policy, Instant::now())
4866        } else {
4867            None
4868        };
4869        (enabled, schedule)
4870    };
4871
4872    if !enabled {
4873        return Err(SuperviseError::Disabled {
4874            module_id: spec.module_id.clone(),
4875        });
4876    }
4877
4878    if schedule.is_none() {
4879        record_health_action(snapshot, &spec.module_id, "failed".to_string(), now_ms);
4880        error!(
4881            module_id = %spec.module_id,
4882            status = ?status,
4883            detail,
4884            max_restarts = runtime.restart_policy.max_restarts,
4885            window_secs = runtime.restart_policy.window.as_secs(),
4886            reason = %runtime.restart_policy.budget_exhausted_detail(),
4887            "health restart budget exhausted; marking module failed"
4888        );
4889        let stop_notice = begin_forwarding_drain_if_configured(
4890            spec,
4891            runtime,
4892            registry,
4893            snapshot,
4894            Some(true),
4895            RouteCloseReason::Disable,
4896        )
4897        .await?;
4898        update_snapshot(snapshot, Some(&spec.module_id), |state| {
4899            state.drain_disposition_detail = Some(runtime.restart_policy.budget_exhausted_detail());
4900        })?;
4901        drain_optional_child(
4902            &spec.module_id,
4903            spec.protocol,
4904            stop_notice,
4905            registry,
4906            runtime.forwarding.as_deref(),
4907            snapshot,
4908            &runtime.terminal_ring,
4909            &runtime.spawn_events,
4910            child,
4911            runtime.drain_timeout,
4912            ModuleState::Failed,
4913            Some(true),
4914        )
4915        .await?;
4916        process_liveness.untrack_if_current(&spec.module_id, snapshot);
4917        return Ok(());
4918    }
4919
4920    let schedule = schedule.expect("a health restart must have a crash-restart schedule");
4921    let mut restart_count = 0;
4922    update_snapshot(snapshot, Some(&spec.module_id), |state| {
4923        restart_count = state.crash_restarts.len();
4924        state.state = ModuleState::Unresponsive;
4925        state.health.status = status;
4926        state.health.last_action = Some(HealthAction::Restart.to_string());
4927        state.health.last_action_ms = Some(now_ms);
4928    })?;
4929    warn!(
4930        module_id = %spec.module_id,
4931        status = ?status,
4932        detail,
4933        restart_count,
4934        restart_in_window = schedule.restart_in_window,
4935        delay_ms = schedule.delay.as_millis() as u64,
4936        "health-triggered module restart"
4937    );
4938
4939    let stop_notice = begin_forwarding_drain_if_configured(
4940        spec,
4941        runtime,
4942        registry,
4943        snapshot,
4944        Some(true),
4945        RouteCloseReason::Restart,
4946    )
4947    .await?;
4948    drain_optional_child(
4949        &spec.module_id,
4950        spec.protocol,
4951        stop_notice,
4952        registry,
4953        runtime.forwarding.as_deref(),
4954        snapshot,
4955        &runtime.terminal_ring,
4956        &runtime.spawn_events,
4957        child,
4958        runtime.drain_timeout,
4959        ModuleState::Restarting,
4960        Some(true),
4961    )
4962    .await?;
4963    schedule_respawn(
4964        runtime,
4965        snapshot,
4966        &spec.module_id,
4967        schedule.delay,
4968        RespawnKind::Spawn,
4969    )
4970}
4971
4972fn cancel_deferred_reload(runtime: &SupervisorRuntimeConfig, module_id: &str, reason: &str) {
4973    if let Some(reply) = runtime
4974        .deferred_reload_reply
4975        .lock()
4976        .unwrap_or_else(|p| p.into_inner())
4977        .take()
4978    {
4979        let _ = reply.send(Err(SuperviseError::ReloadFailed {
4980            module_id: module_id.to_string(),
4981            reason: reason.to_string(),
4982        }));
4983    }
4984}
4985
4986fn schedule_respawn(
4987    runtime: &SupervisorRuntimeConfig,
4988    snapshot: &SharedSnapshot,
4989    module_id: &str,
4990    delay: Duration,
4991    kind: RespawnKind,
4992) -> Result<(), SuperviseError> {
4993    cancel_deferred_reload(runtime, module_id, "respawn superseded by another restart");
4994    update_snapshot(snapshot, Some(module_id), |state| {
4995        state.respawn_pending = true
4996    })?;
4997    *runtime
4998        .scheduled_respawn
4999        .lock()
5000        .unwrap_or_else(|p| p.into_inner()) = Some(PendingRespawn {
5001        deadline: Instant::now() + delay,
5002        kind,
5003    });
5004    Ok(())
5005}
5006
5007fn record_health_action(snapshot: &SharedSnapshot, module_id: &str, action: String, now_ms: u64) {
5008    let _ = update_snapshot(snapshot, Some(module_id), |state| {
5009        state.health.last_action = Some(action);
5010        state.health.last_action_ms = Some(now_ms);
5011    });
5012}
5013
5014fn supervisor_health_status(status: HealthStatus) -> SupervisorHealthStatus {
5015    match status {
5016        HealthStatus::Ok => SupervisorHealthStatus::Ok,
5017        HealthStatus::Degraded => SupervisorHealthStatus::Degraded,
5018        HealthStatus::Failing => SupervisorHealthStatus::Failing,
5019    }
5020}
5021
5022/// Caps the metrics blob stored in the cached supervisor snapshot, which is
5023/// returned to every `supervisor.list` and `supervisor.health` caller.
5024///
5025/// This cap is deliberately NOT applied on the one-shot `supervisor.health_probe`
5026/// path: that request exists to return a module's complete metrics object, and
5027/// `ck health <module-id>` documents it as the way to see what the cached view
5028/// truncates. The asymmetry is the feature.
5029///
5030/// So a new caller must decide which side it is on rather than assume the cap is
5031/// universal. Reaching for it on a fresh-probe path would silently reintroduce
5032/// the truncation that path exists to avoid.
5033fn truncate_health_metrics(metrics: Option<Value>) -> Option<Value> {
5034    let metrics = metrics?;
5035    match serde_json::to_vec(&metrics) {
5036        Ok(encoded) if encoded.len() > MAX_HEALTH_METRICS_BYTES => Some(serde_json::json!({
5037            "truncated": true,
5038            "original_bytes": encoded.len(),
5039        })),
5040        Ok(_) | Err(_) => Some(metrics),
5041    }
5042}
5043
5044/// Spread health probes so a fleet-wide restart does not converge them.
5045///
5046/// The delay is derived from the module id and probe index rather than a random
5047/// source, so it is deterministic per module: a module keeps its own offset
5048/// across daemon restarts instead of re-rolling into a collision.
5049fn jittered_health_delay(module_id: &str, probe_index: u64, cadence: Duration) -> Duration {
5050    if cadence.is_zero() {
5051        return Duration::ZERO;
5052    }
5053    let cadence_ms = cadence.as_millis() as u64;
5054    // This early return is REDUNDANT, deliberately, and a mutation run will show
5055    // it surviving removal. Recording why here so the next person to notice does
5056    // not have to re-derive it:
5057    //
5058    // - It is unreachable in practice. `positive_millis` in daemon_config rejects
5059    //   a zero cadence and builds the Duration from whole milliseconds, so a
5060    //   sub-millisecond cadence cannot come from config.
5061    // - Even if reached it changes no answer. The `.max(1)` below makes the span
5062    //   1, and `hash % 1` is 0, so the fall-through returns `cadence` unchanged
5063    //   -- exactly what this returns.
5064    //
5065    // Kept as a guard against a future widening of the config parser (accepting
5066    // microseconds, say), which would make the sub-millisecond case reachable.
5067    // The `.max(1)` is the load-bearing half TODAY: remove it and the modulo
5068    // divides by zero. Remove this and nothing changes.
5069    if cadence_ms == 0 {
5070        return cadence;
5071    }
5072    // Note that this never returns less than one cadence, including for the FIRST
5073    // probe. So a freshly registered module reports health `unknown` for a full
5074    // cadence plus jitter -- 30-33s at the default -- no matter how quickly it is
5075    // ready to answer.
5076    //
5077    // That is a property of the supervisor's schedule, not of any module: an
5078    // operator watching a restart sees `unknown` and cannot tell it from a module
5079    // that is slow to warm. Measured on two unrelated modules, both flipping to
5080    // `ok` between 22s and 32s after restart.
5081    //
5082    // Left as-is because spreading the first probe is what keeps a fleet-wide
5083    // restart from firing fourteen simultaneous probes into a cold machine. The
5084    // alternative -- probe at t+0 and jitter only from the second onward -- trades
5085    // that thundering herd for a faster first reading.
5086    let jitter_span = (cadence_ms / 10).max(1);
5087    let hash = module_id.as_bytes().iter().fold(
5088        probe_index.wrapping_mul(0x9E37_79B9_7F4A_7C15),
5089        |acc, byte| {
5090            acc.wrapping_mul(1099511628211)
5091                .wrapping_add(u64::from(*byte))
5092        },
5093    );
5094    cadence + Duration::from_millis(hash % jitter_span)
5095}
5096
5097#[cfg(test)]
5098mod tests {
5099    use super::*;
5100
5101    #[test]
5102    fn readding_a_module_clears_its_rescan_removal_tombstone() {
5103        let handle = SupervisorHandle::new();
5104        let module_id = "readded-tombstone";
5105        handle.record_rescan_removal(module_id);
5106        assert!(handle.removal_tombstone_age_ms(module_id).is_some());
5107
5108        handle.apply_identity_configuration(&ModuleSpec {
5109            module_id: module_id.to_string(),
5110            program: PathBuf::from("/test/module"),
5111            args: Vec::new(),
5112            env: Vec::new(),
5113            reserved: false,
5114            reserved_prefixes: Vec::new(),
5115            protocol: ModuleProtocol::Subc,
5116            overlap: Default::default(),
5117        });
5118
5119        assert!(
5120            handle.removal_tombstone_age_ms(module_id).is_none(),
5121            "a re-added module must not retain a stale removal tombstone"
5122        );
5123    }
5124
5125    /// What one module's owner looked like from the control plane at the
5126    /// instant after its first process was spawned.
5127    #[derive(Debug, PartialEq, Eq)]
5128    struct OwnerInSpawnWindow {
5129        module_id: String,
5130        configured: bool,
5131        on_roster: bool,
5132        admission_refusal: Option<&'static str>,
5133    }
5134
5135    /// A supervised module's process can connect, register, sync its scopes
5136    /// and describe them as soon as it is spawned, which is BEFORE the
5137    /// supervisor puts the module on the roster. In that window the owner must
5138    /// already read as configured, so a scoped `route.open` against it is
5139    /// refused as retryable `scope_not_synced` and not as terminal
5140    /// `scope_not_live` ("will never sync").
5141    ///
5142    /// The hook runs in exactly that window on every path that takes on a new
5143    /// module, so no race with a real child is needed: `on_roster: false`
5144    /// proves each observation was taken before the roster insert.
5145    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5146    async fn a_new_module_is_configured_before_its_first_process_can_register() {
5147        use crate::scopes::ScopeTable;
5148        use subc_protocol::{error_codes, scope::ScopeSelector, Principal};
5149
5150        let dir = subc_test_support::TestTempDir::new("configured-before-spawn");
5151        let stub = |module_id: &str, program: PathBuf| ModuleSpec {
5152            module_id: module_id.to_string(),
5153            program,
5154            args: Vec::new(),
5155            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
5156                .into_iter()
5157                .map(|key| (key.to_string(), dir.path().display().to_string()))
5158                .chain([("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string())])
5159                .collect(),
5160            reserved: false,
5161            reserved_prefixes: Vec::new(),
5162            protocol: ModuleProtocol::Subc,
5163            overlap: Default::default(),
5164        };
5165        let live = super::terminal_history_tests::fake_aft_stub_path();
5166        let missing = dir.path().join("definitely-missing-module");
5167
5168        let handle = SupervisorHandle::new();
5169        let mut supervisor =
5170            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5171                .with_handle(handle.clone());
5172        let observed = Arc::new(Mutex::new(Vec::<OwnerInSpawnWindow>::new()));
5173        let hook_handle = handle.clone();
5174        let hook_observed = Arc::clone(&observed);
5175        supervisor.test_after_first_spawn = AfterFirstSpawnHook(Some(Arc::new(move |module_id| {
5176            // Exactly what the control plane computes for a scoped route.open
5177            // naming this module as the owner of a scope it has not synced.
5178            let configured = hook_handle.is_configured(module_id);
5179            let selector = ScopeSelector {
5180                owner: Principal::Reserved {
5181                    module_id: module_id.to_string(),
5182                },
5183                scope_ref: "s".to_string(),
5184                scope_epoch: Some(1),
5185            };
5186            let carrier = Principal::Reserved {
5187                module_id: "carrier".to_string(),
5188            };
5189            let admission_refusal = match ScopeTable::new(Vec::<String>::new())
5190                .admit(&carrier, module_id, &selector, configured)
5191            {
5192                Ok(_) => None,
5193                Err(refusal) => Some(refusal.code),
5194            };
5195            hook_observed.lock().unwrap().push(OwnerInSpawnWindow {
5196                module_id: module_id.to_string(),
5197                configured,
5198                on_roster: hook_handle.get(module_id).is_some(),
5199                admission_refusal,
5200            });
5201        })));
5202
5203        let plain = supervisor.spawn(stub("plain", live.clone())).unwrap();
5204        let configured = supervisor
5205            .supervise_configured(stub("configured", live.clone()), true)
5206            .unwrap();
5207        let with_health = supervisor
5208            .supervise_configured_with_health(
5209                stub("with-health", live.clone()),
5210                true,
5211                HealthConfig::default(),
5212                None,
5213                RestartPolicy::default(),
5214            )
5215            .unwrap();
5216        // The failed-spawn path still puts the module on the roster (as
5217        // failed), so it is configured throughout.
5218        let failed = supervisor
5219            .supervise_configured_with_health(
5220                stub("failed-spawn", missing.clone()),
5221                true,
5222                HealthConfig::default(),
5223                None,
5224                RestartPolicy::default(),
5225            )
5226            .unwrap();
5227        // A failed plain `spawn` puts nothing on the roster, so its mark is
5228        // taken back once the spawn has failed.
5229        assert!(supervisor.spawn(stub("spawn-error", missing)).is_err());
5230
5231        let expected = [
5232            "plain",
5233            "configured",
5234            "with-health",
5235            "failed-spawn",
5236            "spawn-error",
5237        ]
5238        .into_iter()
5239        .map(|module_id| OwnerInSpawnWindow {
5240            module_id: module_id.to_string(),
5241            configured: true,
5242            on_roster: false,
5243            admission_refusal: Some(error_codes::SCOPE_NOT_SYNCED),
5244        })
5245        .collect::<Vec<_>>();
5246        assert_eq!(*observed.lock().unwrap(), expected);
5247
5248        for module_id in ["plain", "configured", "with-health", "failed-spawn"] {
5249            assert!(
5250                handle.get(module_id).is_some(),
5251                "{module_id} is on the roster"
5252            );
5253            assert!(
5254                handle.is_configured(module_id),
5255                "{module_id} stays configured"
5256            );
5257        }
5258        assert!(handle.get("spawn-error").is_none());
5259        assert!(
5260            !handle.is_configured("spawn-error"),
5261            "a plain spawn that failed must not leave its module marked configured"
5262        );
5263
5264        // Leaving the roster clears the mark with it.
5265        handle.retire("failed-spawn");
5266        assert!(!handle.is_configured("failed-spawn"));
5267
5268        for module in [plain, configured, with_health] {
5269            module.stop().await.unwrap();
5270        }
5271        drop(failed);
5272    }
5273
5274    fn stale_process_snapshot(state: ModuleState, enabled: bool) -> SharedSnapshot {
5275        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(state, enabled)));
5276        update_snapshot(&snapshot, Some("stale-process-facts"), |snapshot| {
5277            snapshot.process_alive = true;
5278            snapshot.pid = Some(41);
5279            snapshot.spawned_at_ms = Some(42);
5280            snapshot.spawned_from = Some(PathBuf::from("/spawned/module"));
5281            snapshot.spawned_file_identity = Some(SpawnedFileIdentity {
5282                device: 43,
5283                inode: 44,
5284            });
5285        })
5286        .unwrap();
5287        snapshot
5288    }
5289
5290    fn assert_snapshot_process_facts_cleared(snapshot: &SharedSnapshot) {
5291        let snapshot = lock_snapshot(snapshot).unwrap();
5292        assert!(!snapshot.process_alive);
5293        assert_eq!(snapshot.pid, None);
5294        assert_eq!(snapshot.spawned_at_ms, None);
5295        assert_eq!(snapshot.spawned_from, None);
5296        assert_eq!(snapshot.spawned_file_identity, None);
5297    }
5298
5299    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5300    async fn failed_enable_spawn_clears_preexisting_current_process_facts() {
5301        let supervisor =
5302            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5303        let mut runtime = supervisor.runtime_config();
5304        runtime.test_seed_stale_facts_before_enable_spawn = true;
5305        let snapshot = stale_process_snapshot(ModuleState::Disabled, false);
5306        let mut child = None;
5307        let spec = ModuleSpec {
5308            module_id: "failed-enable-clears-facts".to_string(),
5309            program: PathBuf::from("/definitely/missing/failed-enable-module"),
5310            args: Vec::new(),
5311            env: Vec::new(),
5312            reserved: false,
5313            reserved_prefixes: Vec::new(),
5314            protocol: ModuleProtocol::Subc,
5315            overlap: Default::default(),
5316        };
5317
5318        let result = set_child_enabled(
5319            &spec,
5320            &runtime,
5321            &supervisor.registry,
5322            &supervisor.process_liveness,
5323            &snapshot,
5324            &mut child,
5325            true,
5326        )
5327        .await;
5328
5329        assert!(matches!(result, Err(SuperviseError::Spawn { .. })));
5330        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
5331        assert_snapshot_process_facts_cleared(&snapshot);
5332    }
5333
5334    #[tokio::test]
5335    async fn start_revives_stranded_restarting_but_not_pending_backoff() {
5336        let supervisor =
5337            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5338        let runtime = supervisor.runtime_config();
5339        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(
5340            ModuleState::Restarting,
5341            true,
5342        )));
5343        let spec = ModuleSpec {
5344            module_id: "start-stranded-restarting".to_string(),
5345            program: super::terminal_history_tests::fake_aft_stub_path(),
5346            args: Vec::new(),
5347            env: vec![("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string())],
5348            reserved: false,
5349            reserved_prefixes: Vec::new(),
5350            protocol: ModuleProtocol::None,
5351            overlap: Default::default(),
5352        };
5353        let mut child = None;
5354        lock_snapshot(&snapshot).unwrap().respawn_pending = true;
5355        assert!(!super::set_child_enabled(
5356            &spec,
5357            &runtime,
5358            &Registry::default(),
5359            &supervisor.process_liveness,
5360            &snapshot,
5361            &mut child,
5362            true
5363        )
5364        .await
5365        .unwrap());
5366        assert!(child.is_none());
5367        lock_snapshot(&snapshot).unwrap().respawn_pending = false;
5368        assert!(super::set_child_enabled(
5369            &spec,
5370            &runtime,
5371            &Registry::default(),
5372            &supervisor.process_liveness,
5373            &snapshot,
5374            &mut child,
5375            true
5376        )
5377        .await
5378        .unwrap());
5379        assert_eq!(
5380            lock_snapshot(&snapshot).unwrap().state,
5381            ModuleState::Running
5382        );
5383        let mut child = child.unwrap();
5384        child.start_kill().unwrap();
5385        child.wait().await.unwrap();
5386    }
5387
5388    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5389    async fn failed_reload_spawn_clears_current_process_facts() {
5390        let supervisor =
5391            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5392        let mut runtime = supervisor.runtime_config();
5393        runtime.restart_policy = RestartPolicy::new(0, Duration::ZERO);
5394        let snapshot = stale_process_snapshot(ModuleState::Running, true);
5395        let mut child = None;
5396        let spec = ModuleSpec {
5397            module_id: "failed-reload-clears-facts".to_string(),
5398            program: PathBuf::from("/unused/failed-reload-module"),
5399            args: Vec::new(),
5400            env: Vec::new(),
5401            reserved: false,
5402            reserved_prefixes: Vec::new(),
5403            protocol: ModuleProtocol::Subc,
5404            overlap: Default::default(),
5405        };
5406
5407        let result = handle_reload_spawn_failure(
5408            &spec,
5409            &runtime,
5410            &supervisor.process_liveness,
5411            &snapshot,
5412            &mut child,
5413            "forced reload spawn failure".to_string(),
5414        )
5415        .await;
5416
5417        assert!(matches!(result, Err(SuperviseError::ReloadFailed { .. })));
5418        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
5419        assert_snapshot_process_facts_cleared(&snapshot);
5420    }
5421
5422    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5423    async fn dropping_a_module_with_an_active_monitor_clears_current_process_facts() {
5424        let supervisor =
5425            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5426        let snapshot = stale_process_snapshot(ModuleState::Running, true);
5427        let module = supervisor.supervised_module(
5428            ModuleSpec {
5429                module_id: "drop-clears-facts".to_string(),
5430                program: PathBuf::from("/unused/drop-module"),
5431                args: Vec::new(),
5432                env: Vec::new(),
5433                reserved: false,
5434                reserved_prefixes: Vec::new(),
5435                protocol: ModuleProtocol::Subc,
5436                overlap: Default::default(),
5437            },
5438            supervisor.runtime_config(),
5439            Arc::clone(&snapshot),
5440            None,
5441        );
5442        assert!(!module
5443            .inner
5444            .monitor
5445            .lock()
5446            .unwrap()
5447            .as_ref()
5448            .unwrap()
5449            .is_finished());
5450
5451        drop(module);
5452
5453        assert_eq!(
5454            lock_snapshot(&snapshot).unwrap().state,
5455            ModuleState::Stopped
5456        );
5457        assert_snapshot_process_facts_cleared(&snapshot);
5458    }
5459
5460    #[cfg(unix)]
5461    #[tokio::test]
5462    async fn rescan_preserves_running_protocol_until_respawn() {
5463        let dir = subc_test_support::TestTempDir::new("rescan-protocol");
5464        let initial = ModuleSpec {
5465            module_id: "rescan-protocol".into(),
5466            program: PathBuf::from("/bin/sleep"),
5467            args: vec!["60".into()],
5468            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
5469                .into_iter()
5470                .map(|key| (key.into(), dir.path().to_string_lossy().into_owned()))
5471                .collect(),
5472            reserved: false,
5473            reserved_prefixes: vec![],
5474            protocol: ModuleProtocol::None,
5475            overlap: Default::default(),
5476        };
5477        let supervisor =
5478            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5479        let module = supervisor.spawn(initial.clone()).unwrap();
5480        assert!(module.status().unwrap().live);
5481        let mut next = initial;
5482        next.protocol = ModuleProtocol::Subc;
5483        module
5484            .update_configuration(next.clone(), HealthConfig::default(), None)
5485            .await
5486            .unwrap();
5487        assert!(
5488            module.status().unwrap().live,
5489            "rescan must not require HELLO from the old non-wire process"
5490        );
5491        let runtime = supervisor.runtime_config();
5492        let action = on_child_exit(
5493            &next,
5494            RestartPolicy::default(),
5495            &supervisor.registry,
5496            &module.inner.snapshot,
5497            &runtime.terminal_ring,
5498            &runtime.spawn_events,
5499            &runtime.child_roster,
5500            ExitReport {
5501                kind: ExitKind::Clean,
5502                code: Some(0),
5503                signal: None,
5504                at_ms: unix_ms_now(),
5505            },
5506        )
5507        .await;
5508        assert!(
5509            matches!(action, NextAction::Restart { .. }),
5510            "the old non-wire process's clean exit must restart"
5511        );
5512        module.drain().await.unwrap();
5513    }
5514
5515    #[cfg(unix)]
5516    #[tokio::test]
5517    async fn orphan_identity_matches_path_names_and_shebang_interpreters() {
5518        use std::os::unix::fs::PermissionsExt;
5519        let dir = subc_test_support::TestTempDir::new("spawn-image-identity");
5520        let script = dir.join("module.sh");
5521        std::fs::write(&script, "#!/bin/sh\nwhile :; do sleep 0.1; done\n").unwrap();
5522        std::fs::set_permissions(&script, std::fs::Permissions::from_mode(0o700)).unwrap();
5523        let record_path = dir.join("live-children.json");
5524        let supervisor =
5525            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5526                .with_live_children_record(&record_path);
5527        for (program, args) in [
5528            (PathBuf::from("sleep"), vec!["60".into()]),
5529            (script, vec![]),
5530        ] {
5531            let spec = ModuleSpec {
5532                module_id: "image-identity".into(),
5533                program,
5534                args,
5535                env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
5536                    .into_iter()
5537                    .map(|key| (key.into(), dir.to_string_lossy().into_owned()))
5538                    .collect(),
5539                reserved: false,
5540                reserved_prefixes: vec![],
5541                protocol: ModuleProtocol::None,
5542                overlap: Default::default(),
5543            };
5544            let module = supervisor.spawn(spec).unwrap();
5545            #[cfg(target_os = "macos")]
5546            {
5547                // SETEXEC confirmation is asynchronous; the orphan record must
5548                // identify the final image, never the intermediate trampoline.
5549                let deadline = Instant::now() + Duration::from_secs(5);
5550                while crate::live_children::read_record(&record_path)
5551                    .unwrap()
5552                    .iter()
5553                    .all(|entry| entry.executable.is_none())
5554                {
5555                    assert!(Instant::now() < deadline, "module image was not confirmed");
5556                    tokio::time::sleep(Duration::from_millis(5)).await;
5557                }
5558            }
5559            let entry = crate::live_children::read_record(&record_path)
5560                .unwrap()
5561                .pop()
5562                .unwrap();
5563            let observed = subc_os::Process::open(entry.pid)
5564                .unwrap()
5565                .unwrap()
5566                .observe()
5567                .unwrap();
5568            let verdict = crate::live_children::identity_verdict(&entry, entry.pid, &observed);
5569            module.drain().await.unwrap();
5570            assert_eq!(verdict, crate::live_children::IdentityVerdict::Matches);
5571        }
5572    }
5573
5574    #[cfg(unix)]
5575    fn http_fixture(
5576        dir: &std::path::Path,
5577        url: &str,
5578        threshold: u32,
5579    ) -> crate::daemon_config::ConfiguredModule {
5580        let path = dir.join("subc.jsonc");
5581        std::fs::write(&path, serde_json::json!({"version": 1, "modules": {"http-process": {
5582            "program": "/bin/sleep", "args": ["60"], "protocol": "none",
5583            "env": {"XDG_DATA_HOME":dir,"XDG_RUNTIME_DIR":dir,"XDG_CONFIG_HOME":dir},
5584            "health": {"http":url,"cadence_ms":20,"deadline_ms":200,"failure_threshold":threshold},
5585            "restart": {"backoff_ms":1,"max_backoff_ms":1}, "drain_timeout_ms":10
5586        }}}).to_string()).unwrap();
5587        crate::daemon_config::load(&path)
5588            .unwrap()
5589            .unwrap()
5590            .modules
5591            .pop()
5592            .unwrap()
5593    }
5594
5595    #[cfg(unix)]
5596    async fn wait_http_health(module: &SupervisedModule, status: SupervisorHealthStatus) {
5597        timeout(Duration::from_secs(5), async {
5598            loop {
5599                if module.status().unwrap().health.status == status {
5600                    break;
5601                }
5602                sleep(Duration::from_millis(5)).await;
5603            }
5604        })
5605        .await
5606        .unwrap_or_else(|_| {
5607            panic!(
5608                "expected {status:?}, got {:?}",
5609                module.status().unwrap().health
5610            )
5611        });
5612    }
5613
5614    #[cfg(unix)]
5615    #[tokio::test]
5616    async fn http_health_status_flips_ok_failing_ok() {
5617        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5618        let dir = subc_test_support::TestTempDir::new("http-status-flips");
5619        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5620        let status = Arc::new(std::sync::atomic::AtomicU16::new(200));
5621        let serving_status = status.clone();
5622        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5623        let server = tokio::spawn(async move {
5624            loop {
5625                let (mut stream, _) = listener.accept().await.unwrap();
5626                let mut request = [0u8; 2048];
5627                let count = stream.read(&mut request).await.unwrap();
5628                assert!(count > 0, "a probe must send an HTTP request");
5629                let code = serving_status.load(std::sync::atomic::Ordering::SeqCst);
5630                let body = if code == 200 {
5631                    "ready"
5632                } else {
5633                    "{\"status\":\"unavailable\",\"error\":\"scratch failure\"}"
5634                };
5635                let response = format!("HTTP/1.1 {code} scratch\r\nContent-Length: {}\r\nConnection: close\r\n\r\n{body}", body.len());
5636                let _ = stream.write_all(response.as_bytes()).await;
5637            }
5638        });
5639        let configured = http_fixture(&dir, &url, 1000);
5640        let module =
5641            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5642                .supervise_configured_with_health(
5643                    configured.module_spec(),
5644                    true,
5645                    configured.health,
5646                    configured.drain_timeout_ms,
5647                    configured.restart,
5648                )
5649                .unwrap();
5650        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5651        status.store(503, std::sync::atomic::Ordering::SeqCst);
5652        wait_http_health(&module, SupervisorHealthStatus::Failing).await;
5653        assert!(module
5654            .status()
5655            .unwrap()
5656            .health
5657            .detail
5658            .unwrap()
5659            .contains("scratch failure"));
5660        status.store(200, std::sync::atomic::Ordering::SeqCst);
5661        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5662        assert_eq!(module.status().unwrap().health.consecutive_failures, 0);
5663        let before = module.status().unwrap();
5664        let (spec, mut health) = module.configuration().unwrap();
5665        health.http = None;
5666        module
5667            .update_configuration(spec.clone(), health.clone(), Some(10))
5668            .await
5669            .unwrap();
5670        wait_http_health(&module, SupervisorHealthStatus::Unknown).await;
5671        health.http = Some(url);
5672        module
5673            .update_configuration(spec, health, Some(10))
5674            .await
5675            .unwrap();
5676        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5677        assert_eq!(
5678            module.status().unwrap().pid,
5679            before.pid,
5680            "changing a probe must apply live, not restart its process"
5681        );
5682        let (spec, mut health) = module.configuration().unwrap();
5683        health.failure_threshold = 2;
5684        module
5685            .update_configuration(spec, health, Some(10))
5686            .await
5687            .unwrap();
5688        status.store(503, std::sync::atomic::Ordering::SeqCst);
5689        timeout(Duration::from_secs(5), async {
5690            while module.status().unwrap().spawn_generation == before.spawn_generation {
5691                sleep(Duration::from_millis(5)).await;
5692            }
5693        })
5694        .await
5695        .expect("sustained HTTP 503 responses must trigger the health restart policy");
5696        module.drain().await.unwrap();
5697        server.abort();
5698    }
5699
5700    #[cfg(unix)]
5701    #[tokio::test]
5702    async fn http_health_no_listener_restarts_after_consecutive_failures() {
5703        let dir = subc_test_support::TestTempDir::new("http-refused");
5704        let unused = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5705        let url = format!("http://{}/healthz", unused.local_addr().unwrap());
5706        drop(unused);
5707        let configured = http_fixture(&dir, &url, 2);
5708        let module =
5709            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5710                .supervise_configured_with_health(
5711                    configured.module_spec(),
5712                    true,
5713                    configured.health,
5714                    configured.drain_timeout_ms,
5715                    configured.restart,
5716                )
5717                .unwrap();
5718        let before = module.status().unwrap().spawn_generation;
5719        timeout(Duration::from_secs(5), async {
5720            loop {
5721                let status = module.status().unwrap();
5722                if status.spawn_generation > before {
5723                    assert!(status.lifetime_restarts > 0);
5724                    break;
5725                }
5726                sleep(Duration::from_millis(5)).await;
5727            }
5728        })
5729        .await
5730        .expect("sustained HTTP refusal must trigger the health restart policy");
5731        module.drain().await.unwrap();
5732    }
5733
5734    #[cfg(unix)]
5735    #[tokio::test]
5736    async fn http_health_timeout_honours_deadline() {
5737        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5738        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5739        let server = tokio::spawn(async move {
5740            let _held = listener.accept().await.unwrap();
5741            std::future::pending::<()>().await;
5742        });
5743        let error = timeout(
5744            Duration::from_secs(1),
5745            probe_http_health(&url, Duration::from_millis(10)),
5746        )
5747        .await
5748        .expect("the probe must enforce its own deadline")
5749        .unwrap_err();
5750        server.abort();
5751        assert!(error.to_string().contains("timed out"));
5752    }
5753
5754    #[cfg(unix)]
5755    #[tokio::test]
5756    async fn http_health_localhost_reaches_an_ipv6_only_listener() {
5757        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5758        let listener = tokio::net::TcpListener::bind("[::1]:0").await.unwrap();
5759        let url = format!(
5760            "http://localhost:{}/healthz",
5761            listener.local_addr().unwrap().port()
5762        );
5763        let server = tokio::spawn(async move {
5764            let (mut stream, _) = listener.accept().await.unwrap();
5765            let mut request = [0u8; 2048];
5766            assert!(stream.read(&mut request).await.unwrap() > 0);
5767            stream
5768                .write_all(b"HTTP/1.1 200 OK\r\nContent-Length: 2\r\n\r\nok")
5769                .await
5770                .unwrap();
5771        });
5772        // The deadline only bounds a hang. A probe that never tried the IPv6
5773        // address would be refused on 127.0.0.1 and fail at once, so a longer
5774        // deadline does not weaken the assertion; one second timed out under a
5775        // loaded parallel test run.
5776        assert_eq!(
5777            probe_http_health(&url, Duration::from_secs(10))
5778                .await
5779                .unwrap()
5780                .status,
5781            HealthStatus::Ok
5782        );
5783        server.await.unwrap();
5784    }
5785
5786    #[cfg(unix)]
5787    #[tokio::test]
5788    async fn http_health_timeout_keeps_partial_status_and_body() {
5789        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5790        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5791        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5792        let server = tokio::spawn(async move {
5793            let (mut stream, _) = listener.accept().await.unwrap();
5794            let mut request = [0u8; 2048];
5795            assert!(stream.read(&mut request).await.unwrap() > 0);
5796            stream
5797                .write_all(
5798                    b"HTTP/1.1 503 Unavailable\r\nContent-Length: 100\r\n\r\npartial diagnostic",
5799                )
5800                .await
5801                .unwrap();
5802            std::future::pending::<()>().await;
5803        });
5804        let error = probe_http_health(&url, Duration::from_secs(1))
5805            .await
5806            .unwrap_err()
5807            .to_string();
5808        server.abort();
5809        assert!(
5810            error.contains("timed out")
5811                && error.contains("503 Unavailable")
5812                && error.contains("partial diagnostic"),
5813            "{error}"
5814        );
5815    }
5816
5817    #[cfg(unix)]
5818    #[tokio::test]
5819    async fn http_health_diagnostic_body_is_bounded_to_200_bytes() {
5820        use tokio::io::{AsyncReadExt, AsyncWriteExt};
5821        let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5822        let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5823        let server = tokio::spawn(async move {
5824            let (mut stream, _) = listener.accept().await.unwrap();
5825            let mut request = [0u8; 2048];
5826            assert!(stream.read(&mut request).await.unwrap() > 0);
5827            let body = format!("{}not-in-diagnostic", "x".repeat(400));
5828            let response = format!(
5829                "HTTP/1.1 503 Unavailable\r\nContent-Length: {}\r\n\r\n{body}",
5830                body.len()
5831            );
5832            stream.write_all(response.as_bytes()).await.unwrap();
5833        });
5834        let error = probe_http_health(&url, Duration::from_secs(1))
5835            .await
5836            .unwrap_err()
5837            .to_string();
5838        server.await.unwrap();
5839        assert!(error.ends_with(&"x".repeat(200)), "{error}");
5840        assert_eq!(error.split(": ").last().unwrap().len(), 200);
5841        assert!(!error.contains("not-in-diagnostic"));
5842    }
5843
5844    #[cfg(unix)]
5845    #[tokio::test]
5846    async fn http_health_real_nats_server_monitoring() {
5847        if std::process::Command::new("nats-server")
5848            .arg("--version")
5849            .env("XDG_DATA_HOME", std::env::temp_dir())
5850            .env("XDG_RUNTIME_DIR", std::env::temp_dir())
5851            .env("XDG_CONFIG_HOME", std::env::temp_dir())
5852            .output()
5853            .is_err()
5854        {
5855            eprintln!(
5856                "skipped: http_health_real_nats_server_monitoring (nats-server is not on PATH)"
5857            );
5858            return;
5859        }
5860        let dir = subc_test_support::TestTempDir::new("nats-http-monitoring");
5861        let monitor = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5862        let port = monitor.local_addr().unwrap().port();
5863        let client = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5864        let client_port = client.local_addr().unwrap().port();
5865        let config = dir.join("server.conf");
5866        std::fs::write(&config, format!("listen: 127.0.0.1:{client_port}\nhttp: 127.0.0.1:{port}\njetstream {{ store_dir: \"{}\" }}\n", dir.join("jetstream").display())).unwrap();
5867        let mut configured = http_fixture(&dir, &format!("http://127.0.0.1:{port}/healthz"), 1000);
5868        configured.program = PathBuf::from("nats-server");
5869        configured.args = vec!["-c".into(), config.to_string_lossy().into_owned()];
5870        drop(monitor);
5871        drop(client);
5872        let module =
5873            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5874                .supervise_configured_with_health(
5875                    configured.module_spec(),
5876                    true,
5877                    configured.health,
5878                    configured.drain_timeout_ms,
5879                    configured.restart,
5880                )
5881                .unwrap();
5882        wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5883        module.drain().await.unwrap();
5884    }
5885
5886    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5887    async fn configuration_update_does_not_replace_captured_running_process_facts() {
5888        let supervisor =
5889            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5890        let snapshot = stale_process_snapshot(ModuleState::Running, true);
5891        let initial = ModuleSpec {
5892            module_id: "rescan-preserves-spawn-facts".to_string(),
5893            program: PathBuf::from("/spawned/module"),
5894            args: Vec::new(),
5895            env: Vec::new(),
5896            reserved: false,
5897            reserved_prefixes: Vec::new(),
5898            protocol: ModuleProtocol::Subc,
5899            overlap: Default::default(),
5900        };
5901        let module = supervisor.supervised_module(
5902            initial.clone(),
5903            supervisor.runtime_config(),
5904            snapshot,
5905            None,
5906        );
5907        let before = module.status().unwrap();
5908        let mut replacement = initial;
5909        replacement.program = PathBuf::from("/rescanned/replacement-module");
5910
5911        module
5912            .update_configuration(replacement, HealthConfig::default(), None)
5913            .await
5914            .unwrap();
5915
5916        let after = module.status().unwrap();
5917        assert_eq!(after.pid, before.pid);
5918        assert_eq!(after.spawned_at_ms, before.spawned_at_ms);
5919        assert_eq!(after.spawned_from, before.spawned_from);
5920        drop(module);
5921    }
5922}
5923
5924fn unix_ms_now() -> u64 {
5925    SystemTime::now()
5926        .duration_since(UNIX_EPOCH)
5927        .map(|duration| duration.as_millis().min(u128::from(u64::MAX)) as u64)
5928        .unwrap_or(0)
5929}
5930
5931async fn supervise_loop(
5932    mut spec: ModuleSpec,
5933    mut runtime: SupervisorRuntimeConfig,
5934    registry: Arc<Registry>,
5935    process_liveness: Arc<SupervisorProcessLiveness>,
5936    snapshot: SharedSnapshot,
5937    mut child: Option<SupervisedChild>,
5938    mut commands: mpsc::Receiver<SupervisorCommand>,
5939) {
5940    let mut health_probe = HealthProbeRuntime::default();
5941    // All restart backoffs run here, including health and operator requests.
5942    // While one is pending the loop serves commands, so disable or drain can
5943    // cancel the replacement without spawning a process just to stop it.
5944    let mut pending_respawn: Option<PendingRespawn> = None;
5945    // Commands a swap handed back to run next (see `swap::SwapEnd`). Served
5946    // before anything else so a stop that interrupted a swap runs at once.
5947    let mut requeued: VecDeque<SupervisorCommand> = VecDeque::new();
5948    loop {
5949        #[cfg(target_os = "macos")]
5950        if let Some(active) = child.as_mut() {
5951            active.confirm_privacy_exec().await;
5952        }
5953        if let Some(scheduled) = runtime
5954            .scheduled_respawn
5955            .lock()
5956            .unwrap_or_else(|p| p.into_inner())
5957            .take()
5958        {
5959            pending_respawn = Some(scheduled);
5960        }
5961        if pending_respawn.is_some() && (child.is_some() || !respawn_still_pending(&snapshot)) {
5962            pending_respawn = None;
5963            cancel_deferred_reload(
5964                &runtime,
5965                &spec.module_id,
5966                "respawn cancelled by a supervisor command",
5967            );
5968        }
5969        if child.is_none() && pending_respawn.is_none() {
5970            cancel_deferred_reload(
5971                &runtime,
5972                &spec.module_id,
5973                "respawn cancelled before a replacement was spawned",
5974            );
5975            let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
5976                state.respawn_pending = false;
5977                state.coalesced_restart_pending = false;
5978                if matches!(
5979                    state.state,
5980                    ModuleState::Restarting
5981                        | ModuleState::Starting
5982                        | ModuleState::Draining
5983                        | ModuleState::Unresponsive
5984                ) {
5985                    error!(module_id = %spec.module_id, state = ?state.state,
5986                        "supervision operation ended without a child or pending respawn; marking failed so start can retry");
5987                    state.state = ModuleState::Failed;
5988                    clear_current_process_facts(state);
5989                }
5990            });
5991        }
5992        if let Some(command) = requeued.pop_front() {
5993            if !handle_supervisor_command(
5994                command,
5995                &mut spec,
5996                &mut runtime,
5997                &registry,
5998                &process_liveness,
5999                &snapshot,
6000                &mut child,
6001                &mut commands,
6002                &mut requeued,
6003            )
6004            .await
6005            {
6006                return;
6007            }
6008            if child.is_some() || !respawn_still_pending(&snapshot) {
6009                pending_respawn = None;
6010                let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
6011                    state.respawn_pending = false
6012                });
6013            }
6014            continue;
6015        }
6016        if child.is_some() {
6017            health_probe.refresh_registration(&spec, &runtime, &registry, &snapshot);
6018            let probe_sleep = sleep(health_probe.wake_after());
6019            tokio::pin!(probe_sleep);
6020            let active_child = child.as_mut().expect("child checked above");
6021            tokio::select! {
6022                wait_result = active_child.wait() => {
6023                    // Every arm below that gives up on the CHILD must keep the
6024                    // supervision task itself alive (child = None, loop
6025                    // continues into command-serving mode). Returning here
6026                    // closes the command channel, which makes the module
6027                    // permanently unrestartable in-band: a clean child exit
6028                    // of an enabled module once wedged the fleet this way
6029                    // ('supervisor command channel is closed') and required a
6030                    // full daemon restart to recover.
6031                    let exit_report = match wait_result {
6032                        Ok(status) => classify_reaped_child_exit(&snapshot, active_child, &status),
6033                        Err(err) => {
6034                            active_child.drain_stderr(&spec.module_id).await;
6035                            fail_snapshot(&snapshot, Some(&spec.module_id), None);
6036                            // Every other exit path (on_child_exit's Clean/Crash arms,
6037                            // the reload-registration-failure path) records a terminal
6038                            // before moving on. Without one here, a module whose wait()
6039                            // itself errored (e.g. already reaped) leaves no terminal
6040                            // record at all -- an empty ring reads as "nothing died".
6041                            record_wait_error_terminal(
6042                                &spec.module_id,
6043                                &runtime.terminal_ring,
6044                                &runtime.spawn_events,
6045                            );
6046                            untrack_if_registration_released(
6047                                &process_liveness,
6048                                &registry,
6049                                &spec.module_id,
6050                                &snapshot,
6051                            );
6052                            error!(module_id = %spec.module_id, error = %err, "failed to wait for supervised module");
6053                            child = None;
6054                            continue;
6055                        }
6056                    };
6057                    active_child.drain_stderr(&spec.module_id).await;
6058
6059                    let next = on_child_exit(
6060                        &spec,
6061                        runtime.restart_policy,
6062                        &registry,
6063                        &snapshot,
6064                        &runtime.terminal_ring,
6065                        &runtime.spawn_events,
6066                        &runtime.child_roster,
6067                        exit_report,
6068                    ).await;
6069                    // The exit is recorded, so a daemon shutdown may stop
6070                    // waiting for this child (see `SupervisedChild::wait`).
6071                    active_child.release_roster();
6072                    match next {
6073                        NextAction::Stop { registration_released } => {
6074                            if registration_released {
6075                                process_liveness.untrack_if_current(&spec.module_id, &snapshot);
6076                            }
6077                            child = None;
6078                        }
6079                        NextAction::Restart { schedule } => {
6080                            let delay = schedule.map_or(
6081                                runtime.restart_policy.delay_for_restart(0),
6082                                |schedule| schedule.delay,
6083                            );
6084                            if let Some(schedule) = schedule {
6085                                log_crash_respawn(&spec.module_id, schedule);
6086                            }
6087                            // The exited child is fully recorded at this point,
6088                            // so release it and count the backoff down in the
6089                            // command-serving branch below rather than sleeping
6090                            // here: commands cannot be received from inside this
6091                            // select arm, and an operator disable or drain that
6092                            // arrives during the backoff must cancel the pending
6093                            // respawn instead of waiting for it to spawn first.
6094                            child = None;
6095                            pending_respawn = Some(PendingRespawn { deadline: Instant::now() + delay, kind: RespawnKind::Spawn });
6096                             let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = true);
6097                        }
6098                    }
6099                }
6100                command = commands.recv() => {
6101                    let Some(command) = command else {
6102                        return;
6103                    };
6104                    if !handle_supervisor_command(
6105                        command,
6106                        &mut spec,
6107                        &mut runtime,
6108                        &registry,
6109                        &process_liveness,
6110                        &snapshot,
6111                        &mut child,
6112                        &mut commands,
6113                        &mut requeued,
6114                    ).await {
6115                        return;
6116                    }
6117                }
6118                _ = &mut probe_sleep => {
6119                    if health_probe.due() {
6120                        run_health_probe_cycle(
6121                            &spec,
6122                            &runtime,
6123                            &registry,
6124                            &process_liveness,
6125                            &snapshot,
6126                            &mut child,
6127                        ).await;
6128                        if child.is_some() {
6129                            health_probe.schedule_next(&spec, runtime.health.cadence);
6130                        }
6131                    }
6132                }
6133            }
6134        } else if let Some(pending) = pending_respawn {
6135            tokio::select! {
6136                _ = sleep_until(pending.deadline) => {
6137                    pending_respawn = None;
6138                let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = false);
6139                    // A command handled below while the backoff elapsed may
6140                    // have stopped the module; never respawn past an operator's
6141                    // disable or drain.
6142                    if !respawn_still_pending(&snapshot) {
6143                        continue;
6144                    }
6145                    // The daemon began shutting down during the backoff: the
6146                    // spawn would be refused anyway, and refusing it here
6147                    // leaves the module stopped instead of reporting a
6148                    // failed restart.
6149                    if runtime.child_roster.is_closed() {
6150                        let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
6151                            state.state = ModuleState::Stopped;
6152                        });
6153                        debug!(module_id = %spec.module_id, "crash respawn cancelled by daemon shutdown");
6154                        continue;
6155                    }
6156                    if let Err(err) = release_dead_registration(
6157                        &registry,
6158                        runtime.forwarding.as_deref(),
6159                        &snapshot,
6160                        &spec.module_id,
6161                    ).await {
6162                        fail_snapshot(&snapshot, Some(&spec.module_id), None);
6163                        error!(module_id = %spec.module_id, error = %err, "registration did not release before restart");
6164                        continue;
6165                    }
6166
6167                    if matches!(pending.kind, RespawnKind::Reload) {
6168                        let reply = runtime.deferred_reload_reply.lock().unwrap_or_else(|p| p.into_inner()).take();
6169                        let result = finish_reload_child(&spec, &runtime, &registry, &process_liveness, &snapshot, &mut child).await;
6170                        if let Some(reply) = reply { let _ = reply.send(result); }
6171                        continue;
6172                    }
6173                    process_liveness.track(spec.module_id.clone(), Arc::clone(&snapshot));
6174                    match spawn_and_mark_running(&spec, &runtime, &snapshot) {
6175                        Ok(next_child) => {
6176                            child = Some(next_child);
6177                            debug!(module_id = %spec.module_id, "supervised module restarted after crash");
6178                        }
6179                        Err(err) => {
6180                            fail_snapshot(&snapshot, Some(&spec.module_id), None);
6181                            process_liveness.untrack_if_current(&spec.module_id, &snapshot);
6182                            error!(module_id = %spec.module_id, error = %err, "failed to restart supervised module");
6183                        }
6184                    }
6185                }
6186                command = commands.recv() => {
6187                    let Some(command) = command else {
6188                        return;
6189                    };
6190                    if !handle_supervisor_command(
6191                        command,
6192                        &mut spec,
6193                        &mut runtime,
6194                        &registry,
6195                        &process_liveness,
6196                        &snapshot,
6197                        &mut child,
6198                        &mut commands,
6199                        &mut requeued,
6200                    ).await {
6201                        return;
6202                    }
6203                    // Reconcile the pending respawn with what the command did:
6204                    // a start may already have spawned a fresh child,
6205                    // while a disable or drain moved the snapshot out of the
6206                    // state the respawn was counting down from.
6207                    if child.is_some() || !respawn_still_pending(&snapshot) {
6208                        pending_respawn = None;
6209                let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = false);
6210                    }
6211                }
6212            }
6213        } else {
6214            let Some(command) = commands.recv().await else {
6215                return;
6216            };
6217            if !handle_supervisor_command(
6218                command,
6219                &mut spec,
6220                &mut runtime,
6221                &registry,
6222                &process_liveness,
6223                &snapshot,
6224                &mut child,
6225                &mut commands,
6226                &mut requeued,
6227            )
6228            .await
6229            {
6230                return;
6231            }
6232        }
6233    }
6234}
6235
6236fn log_crash_respawn(module_id: &str, schedule: CrashRestartSchedule) {
6237    info!(
6238        module_id,
6239        restart_in_window = schedule.restart_in_window,
6240        delay_ms = schedule.delay.as_millis() as u64,
6241        "respawning after crash"
6242    );
6243}
6244
6245/// Whether the respawn a backoff was counting down to is still wanted. A
6246/// disable or drain handled while the backoff elapsed moves the snapshot out
6247/// of `Restarting`, and the operator's stop must win over the pending respawn,
6248/// so every sleep-then-spawn path re-validates against the live snapshot
6249/// instead of assuming the state it left behind still holds.
6250fn respawn_still_pending(snapshot: &SharedSnapshot) -> bool {
6251    matches!(
6252        lock_snapshot(snapshot),
6253        Ok(state) if state.enabled && state.state == ModuleState::Restarting
6254    )
6255}
6256
6257enum NextAction {
6258    Stop {
6259        registration_released: bool,
6260    },
6261    Restart {
6262        schedule: Option<CrashRestartSchedule>,
6263    },
6264}
6265
6266#[allow(clippy::too_many_arguments)]
6267async fn handle_supervisor_command(
6268    command: SupervisorCommand,
6269    spec: &mut ModuleSpec,
6270    runtime: &mut SupervisorRuntimeConfig,
6271    registry: &Arc<Registry>,
6272    process_liveness: &SupervisorProcessLiveness,
6273    snapshot: &SharedSnapshot,
6274    child: &mut Option<SupervisedChild>,
6275    commands: &mut mpsc::Receiver<SupervisorCommand>,
6276    requeued: &mut VecDeque<SupervisorCommand>,
6277) -> bool {
6278    match command {
6279        SupervisorCommand::Drain { reply } => {
6280            // A plain stop runs no forwarding drain, so nothing reaches the
6281            // module over its connection before the wait: ask by signal.
6282            let result = drain_optional_child(
6283                &spec.module_id,
6284                spec.protocol,
6285                StopNotice::NotSent,
6286                registry,
6287                runtime.forwarding.as_deref(),
6288                snapshot,
6289                &runtime.terminal_ring,
6290                &runtime.spawn_events,
6291                child,
6292                runtime.drain_timeout,
6293                ModuleState::Stopped,
6294                None,
6295            )
6296            .await;
6297            let registration_released = result.is_ok();
6298            let _ = reply.send(result);
6299            if registration_released {
6300                process_liveness.untrack_if_current(&spec.module_id, snapshot);
6301            }
6302            false
6303        }
6304        SupervisorCommand::Retire { reply } => {
6305            let result = async {
6306                let stop_notice = begin_forwarding_drain_if_configured(
6307                    spec,
6308                    runtime,
6309                    registry,
6310                    snapshot,
6311                    None,
6312                    RouteCloseReason::Disable,
6313                )
6314                .await?;
6315                drain_optional_child(
6316                    &spec.module_id,
6317                    spec.protocol,
6318                    stop_notice,
6319                    registry,
6320                    runtime.forwarding.as_deref(),
6321                    snapshot,
6322                    &runtime.terminal_ring,
6323                    &runtime.spawn_events,
6324                    child,
6325                    runtime.drain_timeout,
6326                    ModuleState::Stopped,
6327                    None,
6328                )
6329                .await
6330            }
6331            .await;
6332            let registration_released = result.is_ok();
6333            let _ = reply.send(result);
6334            if registration_released {
6335                process_liveness.untrack_if_current(&spec.module_id, snapshot);
6336            }
6337            false
6338        }
6339        SupervisorCommand::Restart {
6340            drain_timeout_ms,
6341            received_at_generation,
6342            queued_at,
6343            reply,
6344        } => {
6345            // Without this line a restart that waited in the queue (behind a
6346            // health probe cycle or another command) was invisible: the log
6347            // showed only the drain timing out, minutes after the operator's call.
6348            info!(
6349                module_id = %spec.module_id,
6350                queued_ms = u64::try_from(queued_at.elapsed().as_millis()).unwrap_or(u64::MAX),
6351                "restart command dequeued"
6352            );
6353            // ACK AT INITIATION, not completion. The blocking form deadlocked any
6354            // caller whose own request lane rides the module being restarted: the
6355            // caller's in-flight request keeps the drain from quiescing, the drain
6356            // keeps the restart from completing, and the completion keeps the reply
6357            // from releasing the caller — so the drain always timed out and cut the
6358            // initiator with a GOODBYE, even on a healthy module. Replying once the
6359            // restart is validated lets a self-lane caller settle, which is exactly
6360            // what makes the drain succeed. Completion is observable via
6361            // supervisor.list / module status; a post-ack failure lands the module
6362            // in a visible terminal state below rather than in a reply nobody can
6363            // receive.
6364            let validation = match lock_snapshot(snapshot) {
6365                Ok(state) if !state.enabled => Err(SuperviseError::Disabled {
6366                    module_id: spec.module_id.clone(),
6367                }),
6368                Ok(_) => Ok(()),
6369                Err(err) => Err(err),
6370            };
6371            let initiated = validation.is_ok();
6372            let _ = reply.send(validation);
6373            // A restart asks for a fresh process. Commands run one at a time,
6374            // so a restart queued behind another restart (two operator calls
6375            // in quick succession) is dequeued the moment the first one has
6376            // spawned its replacement -- before that process has sent HELLO.
6377            // Running it would drain and kill the process the first restart
6378            // just produced, which is the opposite of what both callers asked
6379            // for. If a process spawned after this request was received is
6380            // still supervised, the request is already satisfied. Not when the
6381            // configuration changed since that spawn: then the newer process
6382            // predates the spec this restart may exist to apply.
6383            let satisfied_by_generation = if initiated && child.is_some() {
6384                lock_snapshot(snapshot).ok().and_then(|state| {
6385                    (state.spawn_generation > received_at_generation
6386                        && !state.configuration_updated_since_spawn)
6387                        .then_some(state.spawn_generation)
6388                })
6389            } else {
6390                None
6391            };
6392            let satisfied_by_pending = initiated
6393                && child.is_none()
6394                && lock_snapshot(snapshot).ok().is_some_and(|mut state| {
6395                    let pending = state.respawn_pending && !state.configuration_updated_since_spawn;
6396                    if pending {
6397                        state.coalesced_restart_pending = true;
6398                    }
6399                    pending
6400                });
6401            if satisfied_by_pending {
6402                debug!(module_id = %spec.module_id, "restart coalesced into the pending replacement");
6403            } else if let Some(generation) = satisfied_by_generation {
6404                info!(
6405                    module_id = %spec.module_id,
6406                    received_at_generation,
6407                    "restart already satisfied by generation {generation}; not restarting again"
6408                );
6409            } else if initiated {
6410                // Precedence: this restart's operator override, else the module's
6411                // configured budget (already resolved into the runtime).
6412                let drain_timeout = drain_timeout_ms
6413                    .map(Duration::from_millis)
6414                    .unwrap_or(runtime.drain_timeout);
6415                if let Err(err) = restart_child(
6416                    spec,
6417                    runtime,
6418                    registry,
6419                    process_liveness,
6420                    snapshot,
6421                    child,
6422                    drain_timeout,
6423                )
6424                .await
6425                {
6426                    warn!(
6427                        module_id = %spec.module_id,
6428                        error = %err,
6429                        "operator restart failed after initiation ack; module state carries the outcome"
6430                    );
6431                    let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6432                        state.state = ModuleState::Failed;
6433                        clear_current_process_facts(state);
6434                    });
6435                }
6436            }
6437            true
6438        }
6439        SupervisorCommand::Reload { reply } => {
6440            let result =
6441                reload_child(spec, runtime, registry, process_liveness, snapshot, child).await;
6442            if result.is_ok()
6443                && runtime
6444                    .scheduled_respawn
6445                    .lock()
6446                    .unwrap_or_else(|p| p.into_inner())
6447                    .as_ref()
6448                    .is_some_and(|pending| matches!(pending.kind, RespawnKind::Reload))
6449            {
6450                *runtime
6451                    .deferred_reload_reply
6452                    .lock()
6453                    .unwrap_or_else(|p| p.into_inner()) = Some(reply);
6454            } else {
6455                let _ = reply.send(result);
6456            }
6457            true
6458        }
6459        SupervisorCommand::SetEnabled { enabled, reply } => {
6460            let result = set_child_enabled(
6461                spec,
6462                runtime,
6463                registry,
6464                process_liveness,
6465                snapshot,
6466                child,
6467                enabled,
6468            )
6469            .await;
6470            let _ = reply.send(result);
6471            true
6472        }
6473        SupervisorCommand::UpdateConfiguration {
6474            spec: next_spec,
6475            health,
6476            drain_timeout_ms,
6477            reply,
6478        } => {
6479            if let Some(handle) = &runtime.supervisor_handle {
6480                handle.apply_identity_configuration(&next_spec);
6481            }
6482            *spec = next_spec;
6483            let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6484                state.configuration_updated_since_spawn = true;
6485            });
6486            let health_changed = runtime.health != health;
6487            runtime.health = health;
6488            // Reset the cadence and old endpoint's failure streak on a live
6489            // health-policy change rather than waiting for its old deadline.
6490            if health_changed {
6491                let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6492                    state.health = ModuleHealthStatus::default();
6493                });
6494            }
6495            runtime.drain_timeout = drain_timeout_ms
6496                .map(Duration::from_millis)
6497                .unwrap_or(runtime.default_drain_timeout);
6498            *runtime
6499                .effective_drain_timeout
6500                .lock()
6501                .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
6502            let _ = reply.send(());
6503            true
6504        }
6505        SupervisorCommand::Swap {
6506            ready_timeout,
6507            reply,
6508        } => {
6509            let end = swap::run_swap(
6510                spec,
6511                runtime,
6512                registry,
6513                process_liveness,
6514                snapshot,
6515                child,
6516                commands,
6517                ready_timeout.unwrap_or(DEFAULT_SWAP_READY_TIMEOUT),
6518                reply,
6519            )
6520            .await;
6521            requeued.extend(end.requeue);
6522            true
6523        }
6524    }
6525}
6526
6527async fn restart_child(
6528    spec: &ModuleSpec,
6529    runtime: &SupervisorRuntimeConfig,
6530    registry: &Registry,
6531    process_liveness: &SupervisorProcessLiveness,
6532    snapshot: &SharedSnapshot,
6533    child: &mut Option<SupervisedChild>,
6534    drain_timeout: Duration,
6535) -> Result<(), SuperviseError> {
6536    // Restart cycles a running module; it must not silently start a disabled one.
6537    if !lock_snapshot(snapshot)?.enabled {
6538        return Err(SuperviseError::Disabled {
6539            module_id: spec.module_id.clone(),
6540        });
6541    }
6542    let stop_notice = begin_forwarding_drain_with_timeout(
6543        spec,
6544        runtime,
6545        registry,
6546        snapshot,
6547        None,
6548        RouteCloseReason::Restart,
6549        drain_timeout,
6550    )
6551    .await?;
6552
6553    if child.is_some() {
6554        drain_optional_child(
6555            &spec.module_id,
6556            spec.protocol,
6557            stop_notice,
6558            registry,
6559            runtime.forwarding.as_deref(),
6560            snapshot,
6561            &runtime.terminal_ring,
6562            &runtime.spawn_events,
6563            child,
6564            drain_timeout,
6565            ModuleState::Restarting,
6566            Some(true),
6567        )
6568        .await?;
6569    } else {
6570        update_snapshot(snapshot, Some(&spec.module_id), |state| {
6571            state.enabled = true;
6572            state.state = ModuleState::Restarting;
6573            clear_current_process_facts(state);
6574        })?;
6575        release_dead_registration(
6576            registry,
6577            runtime.forwarding.as_deref(),
6578            snapshot,
6579            &spec.module_id,
6580        )
6581        .await?;
6582    }
6583
6584    reset_restart_count(snapshot, &spec.module_id)?;
6585    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6586    schedule_respawn(
6587        runtime,
6588        snapshot,
6589        &spec.module_id,
6590        runtime.restart_policy.backoff,
6591        RespawnKind::Spawn,
6592    )
6593}
6594
6595async fn reload_child(
6596    spec: &ModuleSpec,
6597    runtime: &SupervisorRuntimeConfig,
6598    registry: &Registry,
6599    process_liveness: &SupervisorProcessLiveness,
6600    snapshot: &SharedSnapshot,
6601    child: &mut Option<SupervisedChild>,
6602) -> Result<(), SuperviseError> {
6603    // Reload cycles a running module; it must not silently start a disabled one.
6604    if !lock_snapshot(snapshot)?.enabled {
6605        return Err(SuperviseError::Disabled {
6606            module_id: spec.module_id.clone(),
6607        });
6608    }
6609    let stop_notice = begin_forwarding_drain(
6610        spec,
6611        runtime,
6612        registry,
6613        snapshot,
6614        Some(true),
6615        RouteCloseReason::Reload,
6616    )
6617    .await?;
6618
6619    if child.is_some() {
6620        drain_optional_child(
6621            &spec.module_id,
6622            spec.protocol,
6623            stop_notice,
6624            registry,
6625            runtime.forwarding.as_deref(),
6626            snapshot,
6627            &runtime.terminal_ring,
6628            &runtime.spawn_events,
6629            child,
6630            runtime.drain_timeout,
6631            ModuleState::Restarting,
6632            Some(true),
6633        )
6634        .await?;
6635    } else {
6636        update_snapshot(snapshot, Some(&spec.module_id), |state| {
6637            state.enabled = true;
6638            state.state = ModuleState::Restarting;
6639            clear_current_process_facts(state);
6640        })?;
6641        release_dead_registration(
6642            registry,
6643            runtime.forwarding.as_deref(),
6644            snapshot,
6645            &spec.module_id,
6646        )
6647        .await?;
6648    }
6649
6650    reset_restart_count(snapshot, &spec.module_id)?;
6651    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6652    schedule_respawn(
6653        runtime,
6654        snapshot,
6655        &spec.module_id,
6656        runtime.restart_policy.backoff,
6657        RespawnKind::Reload,
6658    )
6659}
6660
6661async fn finish_reload_child(
6662    spec: &ModuleSpec,
6663    runtime: &SupervisorRuntimeConfig,
6664    registry: &Registry,
6665    process_liveness: &SupervisorProcessLiveness,
6666    snapshot: &SharedSnapshot,
6667    child: &mut Option<SupervisedChild>,
6668) -> Result<(), SuperviseError> {
6669    process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6670    let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
6671        Ok(next_child) => next_child,
6672        Err(err) => {
6673            return handle_reload_spawn_failure(
6674                spec,
6675                runtime,
6676                process_liveness,
6677                snapshot,
6678                child,
6679                format!("new child failed to spawn: {err}"),
6680            )
6681            .await;
6682        }
6683    };
6684    *child = Some(next_child);
6685
6686    let wait_outcome = {
6687        let active_child = child.as_mut().expect("new reload child was just stored");
6688        wait_for_registration_after_reload(
6689            registry,
6690            &spec.module_id,
6691            snapshot,
6692            active_child,
6693            REGISTRY_RELEASE_TIMEOUT,
6694        )
6695        .await?
6696    };
6697
6698    match wait_outcome {
6699        RegistrationWaitOutcome::Registered => {
6700            debug!(module_id = %spec.module_id, "supervised module reloaded and registered");
6701            Ok(())
6702        }
6703        RegistrationWaitOutcome::Exited(exit_report) => {
6704            if let Some(active_child) = child.as_mut() {
6705                active_child.drain_stderr(&spec.module_id).await;
6706            }
6707            // Keep the reaped child's roster guard until its terminal is written.
6708            // Shutdown waits on that guard, not on the child Option used for respawn.
6709            let mut exited_child = child.take().expect("exited reload child is still stored");
6710            #[cfg(test)]
6711            if let Some(gate) = &runtime.test_reload_exit_record_gate {
6712                gate.reached.notify_one();
6713                gate.resume.notified().await;
6714            }
6715            let result = handle_reload_child_registration_failure(
6716                spec,
6717                runtime,
6718                registry,
6719                process_liveness,
6720                snapshot,
6721                child,
6722                ReloadRegistrationFailure {
6723                    exit_report: registration_failure_exit_report(exit_report),
6724                    reason: exited_child
6725                        .spawn_failure
6726                        .clone()
6727                        .unwrap_or_else(|| "new child exited before registering".to_string()),
6728                },
6729            )
6730            .await;
6731            exited_child.release_roster();
6732            result
6733        }
6734        RegistrationWaitOutcome::TimedOut => {
6735            let mut timed_out_child = child
6736                .take()
6737                .expect("timed-out reload child is still running");
6738            timed_out_child
6739                .start_kill()
6740                .map_err(|source| SuperviseError::Kill {
6741                    module_id: spec.module_id.clone(),
6742                    source,
6743                })?;
6744            let status = timed_out_child
6745                .wait()
6746                .await
6747                .map_err(|source| SuperviseError::Wait {
6748                    module_id: spec.module_id.clone(),
6749                    source,
6750                })?;
6751            timed_out_child.drain_stderr(&spec.module_id).await;
6752            handle_reload_child_registration_failure(
6753                spec,
6754                runtime,
6755                registry,
6756                process_liveness,
6757                snapshot,
6758                child,
6759                ReloadRegistrationFailure {
6760                    exit_report: registration_failure_exit_report(classify_reaped_child_exit(
6761                        snapshot,
6762                        &timed_out_child,
6763                        &status,
6764                    )),
6765                    reason: format!(
6766                        "new child did not register within {:?}",
6767                        REGISTRY_RELEASE_TIMEOUT
6768                    ),
6769                },
6770            )
6771            .await
6772        }
6773    }
6774}
6775
6776async fn set_child_enabled(
6777    spec: &ModuleSpec,
6778    runtime: &SupervisorRuntimeConfig,
6779    registry: &Registry,
6780    process_liveness: &SupervisorProcessLiveness,
6781    snapshot: &SharedSnapshot,
6782    child: &mut Option<SupervisedChild>,
6783    enabled: bool,
6784) -> Result<bool, SuperviseError> {
6785    let (current_enabled, current_state, respawn_pending) = {
6786        let state = lock_snapshot(snapshot)?;
6787        (state.enabled, state.state, state.respawn_pending)
6788    };
6789    // `start` (enable on an already-enabled module) heals TERMINAL states instead
6790    // of no-op'ing: a module whose restart budget exhausted (Failed) or that exited
6791    // clean (Stopped) has no live process and no other in-band recovery — the
6792    // operator's start is the explicit recovery act and resets the budget. Without
6793    // this arm the only revival was subc-probe --supervisor-restart in a terminal,
6794    // which the 2026-07-14 aft outage proved is a trap when the failed module is
6795    // the one providing every agent's shell.
6796    let revive_terminal = enabled
6797        && current_enabled
6798        && child.is_none()
6799        && (matches!(current_state, ModuleState::Failed | ModuleState::Stopped)
6800            || (current_state == ModuleState::Restarting && !respawn_pending));
6801    if current_enabled == enabled && !revive_terminal {
6802        return Ok(false);
6803    }
6804
6805    if enabled {
6806        update_snapshot(snapshot, Some(&spec.module_id), |state| {
6807            state.enabled = true;
6808            state.state = ModuleState::Starting;
6809            clear_current_process_facts(state);
6810        })?;
6811        #[cfg(test)]
6812        if runtime.test_seed_stale_facts_before_enable_spawn {
6813            update_snapshot(snapshot, Some(&spec.module_id), |state| {
6814                state.process_alive = true;
6815                state.pid = Some(41);
6816                state.spawned_at_ms = Some(42);
6817                state.spawned_from = Some(PathBuf::from("/spawned/module"));
6818                state.spawned_file_identity = Some(SpawnedFileIdentity {
6819                    device: 43,
6820                    inode: 44,
6821                });
6822            })?;
6823        }
6824        release_dead_registration(
6825            registry,
6826            runtime.forwarding.as_deref(),
6827            snapshot,
6828            &spec.module_id,
6829        )
6830        .await?;
6831        reset_restart_count(snapshot, &spec.module_id)?;
6832        process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6833        let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
6834            Ok(next_child) => next_child,
6835            Err(err) => {
6836                if let Err(state_err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6837                    state.state = ModuleState::Failed;
6838                    clear_current_process_facts(state);
6839                }) {
6840                    error!(module_id = %spec.module_id, error = %state_err, "failed to record enable spawn failure");
6841                }
6842                process_liveness.untrack_if_current(&spec.module_id, snapshot);
6843                return Err(err);
6844            }
6845        };
6846        *child = Some(next_child);
6847        debug!(module_id = %spec.module_id, "supervised module enabled");
6848        Ok(true)
6849    } else {
6850        let stop_notice = begin_forwarding_drain_if_configured(
6851            spec,
6852            runtime,
6853            registry,
6854            snapshot,
6855            Some(false),
6856            RouteCloseReason::Disable,
6857        )
6858        .await?;
6859        drain_optional_child(
6860            &spec.module_id,
6861            spec.protocol,
6862            stop_notice,
6863            registry,
6864            runtime.forwarding.as_deref(),
6865            snapshot,
6866            &runtime.terminal_ring,
6867            &runtime.spawn_events,
6868            child,
6869            runtime.drain_timeout,
6870            ModuleState::Disabled,
6871            Some(false),
6872        )
6873        .await?;
6874        debug!(module_id = %spec.module_id, "supervised module disabled");
6875        Ok(true)
6876    }
6877}
6878
6879#[allow(clippy::too_many_arguments)]
6880async fn on_child_exit(
6881    spec: &ModuleSpec,
6882    policy: RestartPolicy,
6883    registry: &Registry,
6884    snapshot: &SharedSnapshot,
6885    terminal_ring: &Arc<Mutex<TerminalRing>>,
6886    spawn_events: &SpawnEventFeed,
6887    roster: &ChildRoster,
6888    exit_report: ExitReport,
6889) -> NextAction {
6890    // Once the daemon has begun shutting down, no exit is a crash to recover
6891    // from: the module is exiting because the daemon is going away (EOF on its
6892    // connection, or a service manager signalling the whole cgroup). Record it
6893    // as such and never schedule a respawn, which would only start a process
6894    // for the shutdown to end again.
6895    if roster.is_closed() {
6896        return on_child_exit_during_daemon_shutdown(
6897            spec,
6898            registry,
6899            snapshot,
6900            terminal_ring,
6901            spawn_events,
6902            exit_report,
6903        )
6904        .await;
6905    }
6906    // Every stop the supervisor itself asks for (operator stop, disable,
6907    // restart, reload, swap, a health restart, a drain that runs out of budget)
6908    // takes the child out of the supervise loop and reaps it in
6909    // `drain_child_to_state`, and daemon shutdown is handled above. So an exit
6910    // that reaches this point was not requested by the daemon.
6911    //
6912    // For a subc-wire module a clean exit is still a stop: those modules are
6913    // written to re-raise SIGTERM, so a stray outside signal already reads as a
6914    // crash, and exiting 0 is a deliberate choice the module made. A
6915    // `protocol: "none"` module is a stock program we cannot change, and many
6916    // of them (nats-server among them) exit 0 on SIGTERM. Treating that as a
6917    // stop would leave the module down for good after any stray signal, so it
6918    // goes through the crash path instead: it spends restart budget, respawns
6919    // with the crash backoff, and ends `failed` when the budget runs out.
6920    let unrequested_clean_exit_of_protocol_none = exit_report.kind == ExitKind::Clean
6921        && running_protocol(spec, snapshot) == ModuleProtocol::None;
6922    match exit_report.kind {
6923        ExitKind::Clean if !unrequested_clean_exit_of_protocol_none => {
6924            info!(
6925                module_id = %spec.module_id,
6926                exit_code = ?exit_report.code,
6927                exit_signal = ?exit_report.signal,
6928                "supervised module exited cleanly"
6929            );
6930            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6931                state.state = ModuleState::Stopped;
6932                clear_current_process_facts(state);
6933                state.last_exit = Some(exit_report.clone());
6934            }) {
6935                error!(module_id = %spec.module_id, error = %err, "failed to record clean module exit");
6936            }
6937            record_terminal(
6938                &spec.module_id,
6939                terminal_ring,
6940                spawn_events,
6941                &exit_report,
6942                TerminalDisposition::Stopped,
6943            );
6944            let registration_released = match wait_for_registration_release(
6945                registry,
6946                &spec.module_id,
6947                REGISTRY_RELEASE_TIMEOUT,
6948            )
6949            .await
6950            {
6951                Ok(()) => true,
6952                Err(err) => {
6953                    warn!(module_id = %spec.module_id, error = %err, "registration still active after clean exit");
6954                    false
6955                }
6956            };
6957            NextAction::Stop {
6958                registration_released,
6959            }
6960        }
6961        ExitKind::Clean | ExitKind::Crash => {
6962            if unrequested_clean_exit_of_protocol_none {
6963                warn!(
6964                    module_id = %spec.module_id,
6965                    exit_code = ?exit_report.code,
6966                    exit_signal = ?exit_report.signal,
6967                    "protocol-none module exited cleanly without a stop request; handling it as a crash"
6968                );
6969            } else {
6970                warn!(
6971                    module_id = %spec.module_id,
6972                    exit_code = ?exit_report.code,
6973                    exit_signal = ?exit_report.signal,
6974                    "supervised module exited abnormally (crash)"
6975                );
6976            }
6977            let mut restart_schedule = None;
6978            let mut disposition = TerminalDisposition::Disabled;
6979            // Set only when the budget is what stopped the module, so the
6980            // terminal record says which limit was hit rather than leaving
6981            // `failed` to be read as "crashed once, badly".
6982            let mut disposition_detail = lock_snapshot(snapshot)
6983                .ok()
6984                .and_then(|mut state| state.spawn_failure.take());
6985            let now = Instant::now();
6986            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6987                clear_current_process_facts(state);
6988                state.last_exit = Some(exit_report.clone());
6989                if state.enabled {
6990                    if let Some(schedule) = state.next_crash_restart(&policy, now) {
6991                        state.state = ModuleState::Restarting;
6992                        restart_schedule = Some(schedule);
6993                        disposition = TerminalDisposition::Restarting;
6994                    } else {
6995                        state.state = ModuleState::Failed;
6996                        disposition = TerminalDisposition::Failed;
6997                        let budget = policy.budget_exhausted_detail();
6998                        disposition_detail =
6999                            Some(disposition_detail.take().map_or_else(
7000                                || budget.clone(),
7001                                |cause| format!("{cause}; {budget}"),
7002                            ));
7003                    }
7004                } else {
7005                    state.state = ModuleState::Disabled;
7006                    disposition = TerminalDisposition::Disabled;
7007                }
7008            }) {
7009                error!(module_id = %spec.module_id, error = %err, "failed to record crashed module exit");
7010                return NextAction::Stop {
7011                    registration_released: false,
7012                };
7013            }
7014            if disposition == TerminalDisposition::Failed {
7015                // The window is in the message, not only in the fields: this line
7016                // is read in a scrollback where a bare `max_restarts=3` reads as a
7017                // lifetime cap and sends the operator looking for three crashes
7018                // that never happened together.
7019                error!(
7020                    module_id = %spec.module_id,
7021                    max_restarts = policy.max_restarts,
7022                    window_secs = policy.window.as_secs(),
7023                    "module stopped: {}",
7024                    policy.budget_exhausted_detail()
7025                );
7026            }
7027            record_terminal_with_detail(
7028                &spec.module_id,
7029                terminal_ring,
7030                spawn_events,
7031                &exit_report,
7032                disposition,
7033                disposition_detail,
7034            );
7035
7036            if let Some(schedule) = restart_schedule {
7037                NextAction::Restart {
7038                    schedule: Some(schedule),
7039                }
7040            } else {
7041                let registration_released = match wait_for_registration_release(
7042                    registry,
7043                    &spec.module_id,
7044                    REGISTRY_RELEASE_TIMEOUT,
7045                )
7046                .await
7047                {
7048                    Ok(()) => true,
7049                    Err(err) => {
7050                        warn!(module_id = %spec.module_id, error = %err, "registration still active after failed module");
7051                        false
7052                    }
7053                };
7054                NextAction::Stop {
7055                    registration_released,
7056                }
7057            }
7058        }
7059        ExitKind::DeliberateSeverance => {
7060            warn!(
7061                module_id = %spec.module_id,
7062                exit_code = ?exit_report.code,
7063                exit_signal = ?exit_report.signal,
7064                "supervised module exited after deliberate connection severance"
7065            );
7066            let mut should_restart = false;
7067            let mut disposition = TerminalDisposition::Disabled;
7068            if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
7069                clear_current_process_facts(state);
7070                state.last_exit = Some(exit_report.clone());
7071                state.lifetime_restarts += 1;
7072                if state.enabled {
7073                    state.state = ModuleState::Restarting;
7074                    should_restart = true;
7075                    disposition = TerminalDisposition::Restarting;
7076                } else {
7077                    state.state = ModuleState::Disabled;
7078                }
7079            }) {
7080                error!(module_id = %spec.module_id, error = %err, "failed to record deliberately severed module exit");
7081                return NextAction::Stop {
7082                    registration_released: false,
7083                };
7084            }
7085            record_terminal(
7086                &spec.module_id,
7087                terminal_ring,
7088                spawn_events,
7089                &exit_report,
7090                disposition,
7091            );
7092
7093            if should_restart {
7094                NextAction::Restart { schedule: None }
7095            } else {
7096                let registration_released = match wait_for_registration_release(
7097                    registry,
7098                    &spec.module_id,
7099                    REGISTRY_RELEASE_TIMEOUT,
7100                )
7101                .await
7102                {
7103                    Ok(()) => true,
7104                    Err(err) => {
7105                        warn!(module_id = %spec.module_id, error = %err, "registration still active after deliberately severed module exit");
7106                        false
7107                    }
7108                };
7109                NextAction::Stop {
7110                    registration_released,
7111                }
7112            }
7113        }
7114    }
7115}
7116
7117async fn on_child_exit_during_daemon_shutdown(
7118    spec: &ModuleSpec,
7119    registry: &Registry,
7120    snapshot: &SharedSnapshot,
7121    terminal_ring: &Arc<Mutex<TerminalRing>>,
7122    spawn_events: &SpawnEventFeed,
7123    exit_report: ExitReport,
7124) -> NextAction {
7125    info!(
7126        module_id = %spec.module_id,
7127        exit_code = ?exit_report.code,
7128        exit_signal = ?exit_report.signal,
7129        exit_kind = ?exit_report.kind,
7130        "supervised module exited during daemon shutdown; not restarting it"
7131    );
7132    if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
7133        state.state = ModuleState::Stopped;
7134        clear_current_process_facts(state);
7135        state.last_exit = Some(exit_report.clone());
7136    }) {
7137        error!(module_id = %spec.module_id, error = %err, "failed to record module exit during daemon shutdown");
7138    }
7139    record_terminal(
7140        &spec.module_id,
7141        terminal_ring,
7142        spawn_events,
7143        &exit_report,
7144        TerminalDisposition::DaemonShutdown,
7145    );
7146    let registration_released =
7147        wait_for_registration_release(registry, &spec.module_id, REGISTRY_RELEASE_TIMEOUT)
7148            .await
7149            .is_ok();
7150    NextAction::Stop {
7151        registration_released,
7152    }
7153}
7154
7155fn record_wait_error_terminal(
7156    module_id: &str,
7157    terminal_ring: &Arc<Mutex<TerminalRing>>,
7158    spawn_events: &SpawnEventFeed,
7159) {
7160    record_terminal(
7161        module_id,
7162        terminal_ring,
7163        spawn_events,
7164        &wait_error_exit_report(),
7165        TerminalDisposition::Failed,
7166    );
7167}
7168
7169fn record_terminal(
7170    module_id: &str,
7171    terminal_ring: &Arc<Mutex<TerminalRing>>,
7172    spawn_events: &SpawnEventFeed,
7173    exit_report: &ExitReport,
7174    disposition: TerminalDisposition,
7175) {
7176    record_terminal_with_detail(
7177        module_id,
7178        terminal_ring,
7179        spawn_events,
7180        exit_report,
7181        disposition,
7182        None,
7183    );
7184}
7185
7186/// The ring lock is held only to capture the read (see
7187/// `TerminalJournal::capture_read`), so this module's exits keep recording
7188/// while the journal files are read. Blocking: it reads files.
7189fn durable_terminal_history_of(
7190    terminal_ring: &Mutex<TerminalRing>,
7191    module_id: &str,
7192) -> subc_control::TerminalHistory {
7193    let read = terminal_ring
7194        .lock()
7195        .unwrap_or_else(|p| p.into_inner())
7196        .capture_durable_history();
7197    read.read(module_id)
7198}
7199
7200fn record_terminal_with_detail(
7201    module_id: &str,
7202    terminal_ring: &Arc<Mutex<TerminalRing>>,
7203    spawn_events: &SpawnEventFeed,
7204    exit_report: &ExitReport,
7205    disposition: TerminalDisposition,
7206    disposition_detail: Option<String>,
7207) {
7208    spawn_events.emit_exited(module_id, exit_report.code, exit_report.signal);
7209    let record = TerminalRecord {
7210        exit_code: exit_report.code,
7211        exit_signal: exit_report.signal,
7212        at_ms: exit_report.at_ms,
7213        disposition,
7214        exit_kind: exit_report.kind.into(),
7215        disposition_detail,
7216    };
7217    terminal_ring
7218        .lock()
7219        .unwrap_or_else(|poisoned| poisoned.into_inner())
7220        .record_exit(module_id, record);
7221}
7222
7223fn untrack_if_registration_released(
7224    process_liveness: &SupervisorProcessLiveness,
7225    registry: &Registry,
7226    module_id: &str,
7227    snapshot: &SharedSnapshot,
7228) {
7229    match registry.get_module(module_id) {
7230        Ok(None) => process_liveness.untrack_if_current(module_id, snapshot),
7231        Ok(Some(_)) => {}
7232        Err(err) => {
7233            warn!(module_id, error = %err, "could not determine whether supervisor liveness can be untracked");
7234        }
7235    }
7236}
7237
7238/// The child's environment plan: inherit the parent's, drop ambient `CK_LOG`,
7239/// then apply the module's configured entries minus daemon-private capture keys.
7240///
7241/// Separated from `spawn_child` only so it can be asserted without spawning a
7242/// process — a duplicate of this logic in a test would pass while the real one
7243/// drifted, which is the defect class this function exists to avoid.
7244/// The subc-wire half of a spawn: `--subc <connection file>` and the launch
7245/// nonce. A `protocol: "none"` module gets neither, because it cannot use
7246/// either and the argument would stop a stock binary from starting at all.
7247/// `SUBC_MODULE_ID` is set on every path since an unread variable is inert.
7248///
7249/// The plain-spawn form, kept for the tests that assert its plan; spawns go
7250/// through [`apply_wire_spawn_args_for_role`].
7251#[cfg(test)]
7252fn apply_wire_spawn_args(
7253    command: &mut Command,
7254    spec: &ModuleSpec,
7255    connection_file_path: Option<&std::path::Path>,
7256    handle: Option<&SupervisorHandle>,
7257) -> Result<Option<NonceHandoff>, SuperviseError> {
7258    apply_wire_spawn_args_for_role(
7259        command,
7260        spec,
7261        connection_file_path,
7262        handle,
7263        SpawnRole::Plain,
7264    )
7265}
7266
7267/// The read end of a spawn's launch-nonce pipe, prepared by
7268/// [`apply_wire_spawn_args_for_role`] and installed as the child's descriptor 3
7269/// by the last pre-exec step, just before `spawn()`. Windows has no descriptor
7270/// handoff and keeps only the environment copy.
7271#[cfg(unix)]
7272type NonceHandoff = subc_os::LaunchNonceHandoff;
7273#[cfg(not(unix))]
7274type NonceHandoff = std::convert::Infallible;
7275
7276/// Prepare wire identity for a plain spawn or a swap candidate.
7277///
7278/// A plain spawn replaces the module's recorded nonce. A swap candidate records
7279/// a separate candidate token so the still-serving incumbent and its consumers
7280/// keep their nonce. Both records are installed before the process exists, so
7281/// the child's initial HELLO registration cannot arrive ahead of its nonce.
7282///
7283/// On Unix the nonce is delivered only through a pipe. It is written into
7284/// a pipe whose read end the child gets as descriptor 3, named by
7285/// `SUBC_LAUNCH_NONCE_FD=3:<pipe inode>`: unlike the environment, another
7286/// process of the same user cannot read it with `ps eww`. That handoff is
7287/// returned rather than installed here, because installing it replaces
7288/// whatever the child has at descriptor 3 and so must be the last pre-exec
7289/// step, after the Linux cgroup placement that the caller registers later.
7290/// Windows retains the environment handoff until restricted handle inheritance
7291/// can be implemented outside std's process primitives.
7292fn apply_wire_spawn_args_for_role(
7293    command: &mut Command,
7294    spec: &ModuleSpec,
7295    connection_file_path: Option<&std::path::Path>,
7296    handle: Option<&SupervisorHandle>,
7297    role: SpawnRole,
7298) -> Result<Option<NonceHandoff>, SuperviseError> {
7299    command.env(SUBC_MODULE_ID_ENV, &spec.module_id);
7300    // SUBC_LAUNCH_NONCE_FD is removed for every child, `protocol: "none"`
7301    // included: a daemon started from a module's process tree inherits it,
7302    // and passing it on would point the child at a descriptor it does not
7303    // have.
7304    command.env_remove(subc_os::LAUNCH_NONCE_FD_ENV);
7305    // Remove inherited or configured copies too: withholding must mean absent.
7306    command.env_remove(SUBC_LAUNCH_NONCE_ENV);
7307    if spec.protocol == ModuleProtocol::None {
7308        return Ok(None);
7309    }
7310    if let Some(connection_file_path) = connection_file_path {
7311        command.arg(SUBC_ARG).arg(connection_file_path);
7312    }
7313
7314    // Every subc-wire spawn receives a fresh one-time launch nonce for consumer
7315    // route.open attestation. Reserved modules additionally use the same nonce
7316    // for HELLO id-squatting protection. A respawn rotates both records.
7317    let nonce = generate_launch_nonce()?;
7318    if let Some(handle) = handle {
7319        match role {
7320            SpawnRole::Plain => {
7321                handle.set_spawn_nonce(&spec.module_id, nonce.clone());
7322                if spec.reserved {
7323                    handle.set_reserved_nonce(&spec.module_id, nonce.clone());
7324                }
7325            }
7326            SpawnRole::SwapCandidate => handle.open_swap(&spec.module_id, nonce.clone()),
7327        }
7328    }
7329    #[cfg(unix)]
7330    let handoff = {
7331        let handoff =
7332            subc_os::LaunchNonceHandoff::new(&nonce).map_err(|source| SuperviseError::Spawn {
7333                program: spec.program.clone(),
7334                source,
7335                cgroup_path: None,
7336            })?;
7337        command.env(subc_os::LAUNCH_NONCE_FD_ENV, handoff.fd_env_value());
7338        Some(handoff)
7339    };
7340    #[cfg(not(unix))]
7341    let handoff = None;
7342    // Windows keeps the environment copy: std cannot restrict an inherited pipe
7343    // handle to this child without leaking it to concurrently spawned processes.
7344    #[cfg(not(unix))]
7345    command.env(SUBC_LAUNCH_NONCE_ENV, nonce);
7346    Ok(handoff)
7347}
7348
7349fn apply_child_env(command: &mut Command, spec: &ModuleSpec) {
7350    command.env_remove(CK_LOG_ENV);
7351    // The spawn role is the supervisor's to set, and only on a swap candidate
7352    // (see `apply_spawn_role`). Removing it here, rather than just not setting
7353    // it, is what makes it absent on a plain spawn: the daemon's own
7354    // environment could carry it, and so could a spec built outside daemon
7355    // config (config refuses it as an `env` key). A module reading it on a
7356    // plain restart would pick the long swap budget and leave callers waiting.
7357    command.env_remove(SUBC_SPAWN_ROLE_ENV);
7358    for (key, value) in &spec.env {
7359        // cortexkit-log currently exposes retention only as a Rust struct, not
7360        // environment names. These values are daemon-private sink metadata and
7361        // must never become a public child-process contract by being inherited.
7362        if matches!(
7363            key.as_str(),
7364            CAPTURE_MAX_FILE_MB_ENV | CAPTURE_KEEP_ENV | CAPTURE_MAX_AGE_DAYS_ENV
7365        ) || key == SUBC_SPAWN_ROLE_ENV
7366        {
7367            continue;
7368        }
7369        command.env(key, value);
7370    }
7371}
7372
7373/// Which slot a spawn fills: the module's ordinary one, or the candidate slot
7374/// of a blue/green swap.
7375#[derive(Debug, Clone, Copy, PartialEq, Eq)]
7376enum SpawnRole {
7377    Plain,
7378    SwapCandidate,
7379}
7380
7381/// Set the spawn role for a swap candidate. A plain spawn gets nothing here;
7382/// `apply_child_env` has already removed the variable for every spawn.
7383fn apply_spawn_role(command: &mut Command, role: SpawnRole) {
7384    if role == SpawnRole::SwapCandidate {
7385        command.env(SUBC_SPAWN_ROLE_ENV, SPAWN_ROLE_SWAP_CANDIDATE);
7386    }
7387}
7388
7389fn spawn_child(
7390    spec: &ModuleSpec,
7391    connection_file_path: Option<&std::path::Path>,
7392    handle: Option<&SupervisorHandle>,
7393    ring: &Arc<Mutex<StderrRing>>,
7394    capture_logs_dir: Option<&std::path::Path>,
7395    roster: &ChildRoster,
7396    #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
7397) -> Result<SupervisedChild, SuperviseError> {
7398    spawn_child_in_slot(
7399        spec,
7400        connection_file_path,
7401        handle,
7402        ring,
7403        capture_logs_dir,
7404        roster,
7405        #[cfg(target_os = "linux")]
7406        cgroup_placement,
7407        SpawnRole::Plain,
7408        false,
7409    )
7410}
7411
7412/// Spawn one process of `spec` into a slot.
7413///
7414/// `alternate_slot` picks the process's cgroup label on Linux (see `swap::cgroup_name`).
7415/// A swap candidate needs a different cgroup from the process it is replacing,
7416/// which is still alive: in the same cgroup the two would be one kill domain,
7417/// and killing a failed candidate could take the incumbent with it.
7418///
7419/// The stderr capture file is `<module_id>.stderr.log` for every process of
7420/// the module, whichever slot it is in, because that is the one file
7421/// `ck module logs` reads. During a swap's overlap both processes append to it;
7422/// the daemon writes whole lines, so the two interleave by line, which is also
7423/// the merged view an operator wants while a swap runs.
7424#[allow(clippy::too_many_arguments)]
7425fn spawn_child_in_slot(
7426    spec: &ModuleSpec,
7427    connection_file_path: Option<&std::path::Path>,
7428    handle: Option<&SupervisorHandle>,
7429    ring: &Arc<Mutex<StderrRing>>,
7430    capture_logs_dir: Option<&std::path::Path>,
7431    roster: &ChildRoster,
7432    #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
7433    role: SpawnRole,
7434    alternate_slot: bool,
7435) -> Result<SupervisedChild, SuperviseError> {
7436    if roster.is_closed() {
7437        return Err(SuperviseError::Spawn {
7438            program: spec.program.clone(),
7439            source: io::Error::other("the daemon is shutting down; not starting a new process"),
7440            cgroup_path: None,
7441        });
7442    }
7443    #[cfg(target_os = "linux")]
7444    let cgroup_name = {
7445        // Slot names alone are not kill domains: a retired incumbent may still
7446        // be draining when a later enable/restart spawns into the same slot.
7447        // Decimal entropy keeps the suffix unambiguous; Placement performs
7448        // the module-id escaping and constructs the filesystem path.
7449        if cgroup_placement.is_none() {
7450            swap::cgroup_name(&spec.module_id, alternate_slot)
7451        } else {
7452            let nonce = generate_launch_nonce()?;
7453            let suffix = u128::from_str_radix(&nonce[..32], 16).expect("hex launch nonce");
7454            // Leave room for byte escaping and the suffix under NAME_MAX. The
7455            // label is only for humans; the nonce identifies the kill domain.
7456            let mut end = spec.module_id.len().min(64);
7457            while !spec.module_id.is_char_boundary(end) {
7458                end -= 1;
7459            }
7460            format!(
7461                "{}_{suffix}",
7462                swap::cgroup_name(&spec.module_id[..end], alternate_slot)
7463            )
7464        }
7465    };
7466    #[cfg(not(target_os = "linux"))]
7467    let _ = alternate_slot;
7468    #[cfg(target_os = "macos")]
7469    let (mut command, privacy_exec, exec_ack) = privacy_command(spec, roster)?;
7470    #[cfg(not(target_os = "macos"))]
7471    let mut command = Command::new(&spec.program);
7472    command.args(&spec.args);
7473    // AMBIENT `CK_LOG` MUST NOT LEAK INTO AN OTHERWISE UNCONFIGURED MODULE — but
7474    // that is the whole of the intent, so remove that one key rather than the
7475    // environment.
7476    //
7477    // This was `env_clear()` from 0.17.41 until 0.18.3, which achieved the goal
7478    // and took the POSIX environment with it. Modules spawned that way had no
7479    // HOME, XDG_RUNTIME_DIR, TMPDIR or USER, and the consequences ran past
7480    // logging:
7481    //
7482    //   * `connection_file::discover` reads XDG_RUNTIME_DIR and HOME, so with
7483    //     both unset it fell back to the temp dir alone and `ck` could not find
7484    //     a daemon running on the same machine from inside any module's process
7485    //     tree — reporting a path the file has never lived at, which reads as
7486    //     "the daemon did not write its file".
7487    //   * `default_data_home()` with HOME and XDG_DATA_HOME both unset returns
7488    //     the RELATIVE `.local/share`, so a module deriving its own store path
7489    //     resolved it against its own CWD. That is the store-fragmentation
7490    //     defect the daemon already refuses in config (`parse_doc` rejects a
7491    //     relative `storage.data_home`) arriving by derivation instead.
7492    //   * anything a module spawns inherited it: git without ~/.gitconfig,
7493    //     cargo without CARGO_HOME, ssh, python user dirs — all degrading
7494    //     quietly rather than erroring.
7495    //
7496    // Reported by iceteaSA as #104 after deploying 0.18.2, where `ck daemon`
7497    // offered one candidate under /tmp while the file sat in /run/user/1000.
7498    //
7499    // A configured module is unaffected either way: `module_spec()` puts the
7500    // resolved CK_LOG into `spec.env`, which is applied below and therefore
7501    // wins over anything ambient.
7502    apply_child_env(&mut command, spec);
7503    apply_spawn_role(&mut command, role);
7504    let nonce_handoff =
7505        apply_wire_spawn_args_for_role(&mut command, spec, connection_file_path, handle, role)?;
7506
7507    #[cfg(target_os = "linux")]
7508    let cgroup_path = cgroup_placement
7509        .map(|placement| placement.module_path(&cgroup_name))
7510        .transpose()
7511        .map_err(|source| SuperviseError::Cgroup {
7512            module_id: spec.module_id.clone(),
7513            source,
7514        })?;
7515    #[cfg(not(target_os = "linux"))]
7516    let cgroup_path: Option<PathBuf> = None;
7517    #[cfg(target_os = "linux")]
7518    if let Some(path) = &cgroup_path {
7519        if let Err(error) = apply_cgroup_placement(&mut command, spec, path) {
7520            if let Some(placement) = cgroup_placement {
7521                remove_module_cgroup(placement, &cgroup_name);
7522            }
7523            return Err(error);
7524        }
7525    }
7526
7527    let output_sink = if let Some(logs_dir) = capture_logs_dir {
7528        let path = logs_dir.join(format!("{}.stderr.log", spec.module_id));
7529        match ChildOutputSink::open(&path, capture_retention(spec)) {
7530            Ok(sink) => sink,
7531            Err(error) => {
7532                warn!(
7533                    module_id = %spec.module_id,
7534                    path = %path.display(),
7535                    error = %error,
7536                    "could not open child output capture file; forwarding to stderr"
7537                );
7538                ChildOutputSink::Stderr
7539            }
7540        }
7541    } else {
7542        ChildOutputSink::Stderr
7543    };
7544
7545    command.stdout(Stdio::piped());
7546    command.stderr(Stdio::piped());
7547    command.kill_on_drop(true);
7548    // EACH MODULE LEADS ITS OWN PROCESS GROUP (the child calls setpgid(0, 0)
7549    // before exec). In the daemon's group, a service manager that kills the
7550    // job's process group when the daemon exits (launchd's default) killed
7551    // every module at the same moment its control connection closed, so no
7552    // module ever ran its EOF teardown on a daemon stop. Outside that group a
7553    // module is reached only by the daemon: the EOF it sees when its
7554    // connection closes, and the bounded stop in `child_roster` for anything
7555    // still running after that. On Linux this composes with the cgroup
7556    // placement above: that is a pre_exec write to cgroup.procs, std performs
7557    // setpgid in the child before running pre_exec callbacks, and the two
7558    // change independent process attributes.
7559    //
7560    // stdin is /dev/null because a process outside the terminal's foreground
7561    // group is stopped (SIGTTIN) if it reads the terminal, which a daemon run
7562    // by hand would otherwise hand down. Under a service manager stdin is
7563    // already /dev/null.
7564    #[cfg(unix)]
7565    command.process_group(0);
7566    command.stdin(Stdio::null());
7567    // The LAST pre-exec step, after the cgroup placement above: installing the
7568    // nonce at descriptor 3 replaces whatever the child had there, which could
7569    // be the descriptor an earlier step writes through.
7570    #[cfg(unix)]
7571    if let Some(handoff) = nonce_handoff {
7572        handoff.install_last(command.as_std_mut());
7573    }
7574    #[cfg(not(unix))]
7575    let _ = nonce_handoff;
7576
7577    // Containment, step 1 of 3 (issue #109): create the child suspended so it
7578    // cannot run a single instruction -- and therefore cannot spawn a
7579    // grandchild -- before it is in the job. See `contain_spawned_child` for the
7580    // other two steps and why the window matters.
7581    #[cfg(windows)]
7582    subc_jobobject::suspend_on_create_async(&mut command);
7583    let mut child = match command.spawn() {
7584        Ok(child) => child,
7585        Err(source) => {
7586            #[cfg(target_os = "linux")]
7587            if let Some(placement) = cgroup_placement {
7588                remove_module_cgroup(placement, &cgroup_name);
7589            }
7590            return Err(SuperviseError::Spawn {
7591                program: spec.program.clone(),
7592                source,
7593                cgroup_path,
7594            });
7595        }
7596    };
7597    // The parent must close its writer now: the acknowledgement pipe reports EOF
7598    // only when every writer is gone, and the child's copy closes when the
7599    // trampoline replaces itself with the module. Command holds only an integer
7600    // in its pre_exec callback, not another writer.
7601    #[cfg(target_os = "macos")]
7602    drop(exec_ack);
7603
7604    // Containment, steps 2 and 3: assign while suspended, then resume.
7605    #[cfg(windows)]
7606    let job = contain_spawned_child(&child, spec)?;
7607    let spawned_at_ms = unix_ms_now();
7608    let spawned_from = spec.program.clone();
7609    let spawned_file_identity = spawned_file_identity(&spawned_from);
7610    let pid = child.id().ok_or_else(|| SuperviseError::Spawn {
7611        program: spec.program.clone(),
7612        source: io::Error::other("spawned child exposed no live pid"),
7613        cgroup_path: cgroup_path.clone(),
7614    })?;
7615    let process_start_time = crate::provenance::process_start_time(pid);
7616    let process_identity = process_start_time.map(|start_time| ProcessIdentity { pid, start_time });
7617    // Unix spawn returns after exec's error pipe closes. The kernel image is
7618    // therefore the executable to compare during a future orphan sweep: PATH
7619    // lookup and shebang interpretation may select a different file from the
7620    // configured program. Keep the literal program's identity for provenance,
7621    // but never use it as proof that a recorded pid may be signalled.
7622    let recorded_image = observe_spawned_image(pid);
7623    // spawn() confirms only the first exec, into the trampoline. Never persist
7624    // the trampoline image; the asynchronous acknowledgement publishes the
7625    // module image once the trampoline has replaced itself with the module.
7626    #[cfg(target_os = "macos")]
7627    let recorded_image = if privacy_exec.is_some() {
7628        None
7629    } else {
7630        recorded_image
7631    };
7632    #[cfg(target_os = "linux")]
7633    let recorded_cgroup_name = cgroup_path.as_ref().map(|_| cgroup_name.clone());
7634    #[cfg(not(target_os = "linux"))]
7635    let recorded_cgroup_name = None;
7636    let roster_guard = roster.admit(
7637        spec.module_id.clone(),
7638        pid,
7639        spec.protocol,
7640        process_start_time,
7641        crate::child_roster::RecordedIdentity {
7642            start_time: recorded_image.map(|image| image.start_time),
7643            executable: recorded_image
7644                .and_then(|image| image.executable)
7645                .map(crate::live_children::ExecutableIdentity::from),
7646            cgroup_name: recorded_cgroup_name,
7647            #[cfg(target_os = "linux")]
7648            cgroup_placement: cgroup_placement.cloned(),
7649        },
7650    );
7651    // The check at the top of this function can pass just before daemon
7652    // shutdown begins, and the process is only in the roster from here on.
7653    // The shutdown stop returns as soon as it finds the roster empty, so a
7654    // process admitted after that look would outlive the daemon. The roster
7655    // is closed before the stop first reads it and admission happens under
7656    // the roster's lock, so either the stop sees this process or this check
7657    // sees the roster closed: end the process now rather than start a module
7658    // the daemon is about to stop.
7659    if roster.is_closed() {
7660        // This child was never admitted, so there is no module protocol shutdown to wait for.
7661        #[cfg(target_os = "linux")]
7662        kill_module_cgroup(cgroup_placement, &cgroup_name);
7663        if let Err(error) = child.start_kill() {
7664            debug!(module_id = %spec.module_id, pid, %error, "kill of a process spawned during daemon shutdown failed; it may already have exited");
7665        }
7666        #[cfg(target_os = "linux")]
7667        if let Some(placement) = cgroup_placement {
7668            // This spawn was never admitted, so shutdown has no roster entry
7669            // to await. Do not detach its cleanup: the runtime could exit
7670            // before that task reaps the rejected child and removes its group.
7671            while matches!(child.try_wait(), Ok(None)) {
7672                std::thread::yield_now();
7673            }
7674            if matches!(
7675                subc_cgroup::kill_module(Some(placement), &cgroup_name),
7676                subc_cgroup::KillOutcome::Killed
7677            ) {
7678                if let Ok(path) = placement.module_path(&cgroup_name) {
7679                    while std::fs::read_to_string(path.join("cgroup.events"))
7680                        .ok()
7681                        .is_some_and(|events| events.lines().any(|line| line == "populated 1"))
7682                    {
7683                        std::thread::yield_now();
7684                    }
7685                }
7686            }
7687            remove_module_cgroup(placement, &cgroup_name);
7688        }
7689        drop(roster_guard);
7690        return Err(SuperviseError::Spawn {
7691            program: spec.program.clone(),
7692            source: io::Error::other(
7693                "the daemon began shutting down while this process was starting; ended it",
7694            ),
7695            cgroup_path,
7696        });
7697    }
7698
7699    let stdout_pump = match child.stdout.take() {
7700        Some(stdout) => Some(tokio::spawn(pump_stdout_to(stdout, output_sink.clone()))),
7701        None => {
7702            warn!(
7703                module_id = %spec.module_id,
7704                "spawned child exposed no stdout pipe; file capture will be incomplete"
7705            );
7706            None
7707        }
7708    };
7709    let stderr_pump = match child.stderr.take() {
7710        Some(stderr) => {
7711            let generation = ring
7712                .lock()
7713                .unwrap_or_else(|poisoned| poisoned.into_inner())
7714                .begin_process();
7715            Some(StderrPump {
7716                task: tokio::spawn(pump_stderr_to(
7717                    stderr,
7718                    Arc::clone(ring),
7719                    generation,
7720                    output_sink,
7721                )),
7722                generation,
7723            })
7724        }
7725        None => {
7726            // Spawning succeeded but the pipe did not materialise. Recording it as
7727            // uncaptured keeps the tail honest: the alternative is an empty tail
7728            // that reads as a module which printed nothing.
7729            ring.lock()
7730                .unwrap_or_else(|poisoned| poisoned.into_inner())
7731                .mark_not_captured("stderr pipe was not available on spawn");
7732            warn!(
7733                module_id = %spec.module_id,
7734                "spawned child exposed no stderr pipe; tail will be unavailable"
7735            );
7736            None
7737        }
7738    };
7739
7740    Ok(SupervisedChild {
7741        child,
7742        protocol: spec.protocol,
7743        #[cfg(target_os = "linux")]
7744        module_id: cgroup_name,
7745        #[cfg(target_os = "linux")]
7746        cgroup_placement: cgroup_placement.cloned(),
7747        #[cfg(windows)]
7748        job,
7749        stdout_pump,
7750        stderr_pump,
7751        stderr_ring: Arc::clone(ring),
7752        spawned_at_ms,
7753        spawned_from,
7754        spawned_file_identity,
7755        process_start_time,
7756        process_identity,
7757        pid,
7758        roster_guard: Some(roster_guard),
7759        #[cfg(target_os = "macos")]
7760        privacy_exec,
7761        #[cfg(target_os = "macos")]
7762        report_ready: Arc::new(OnceLock::new()),
7763        spawn_failure: None,
7764    })
7765}
7766
7767#[cfg(target_os = "linux")]
7768pub(crate) fn kill_module_cgroup(placement: Option<&subc_cgroup::Placement>, module_id: &str) {
7769    use subc_cgroup::KillOutcome;
7770    match subc_cgroup::kill_module(placement, module_id) {
7771        KillOutcome::Killed => {}
7772        KillOutcome::NotPlaced | KillOutcome::Unsupported => {
7773            debug!(
7774                module_id,
7775                "cgroup tree kill unavailable; using direct-child kill"
7776            );
7777        }
7778        KillOutcome::IoError { path, error } => {
7779            warn!(module_id, path = %path.display(), %error, "cgroup tree kill failed; using direct-child kill");
7780        }
7781    }
7782}
7783
7784/// Contain a freshly spawned Windows child and start it.
7785///
7786/// Steps 2 and 3 of the suspended-create contract: the job is created and the
7787/// child assigned **while it is still suspended** (step 1 is
7788/// `suspend_on_create_async` at the spawn site), then the child is resumed.
7789///
7790/// A child that is never resumed hangs forever holding a pid, so a resume
7791/// failure kills the child and fails the spawn rather than returning a
7792/// `SupervisedChild` that can never run.
7793///
7794/// An assignment failure is NOT fatal: an uncontained module behaves exactly as
7795/// it did before this existed, whereas refusing to start one would be a new
7796/// outage. It is logged at warn because it means a helper process could leak.
7797#[cfg(windows)]
7798fn contain_spawned_child(
7799    child: &Child,
7800    spec: &ModuleSpec,
7801) -> Result<Option<subc_jobobject::JobObject>, SuperviseError> {
7802    let module_id = spec.module_id.as_str();
7803    let Some(pid) = child.id() else {
7804        // The child exited between spawn and here. Its tree, if it made one,
7805        // needs no containment: nothing is left to contain.
7806        warn!(
7807            module_id,
7808            "spawned child had already exited before containment; no job object attached"
7809        );
7810        return Ok(None);
7811    };
7812
7813    let job = match subc_jobobject::JobObject::new() {
7814        Ok(job) => job,
7815        Err(source) => {
7816            warn!(
7817                module_id,
7818                error = %source,
7819                "could not create a job object; this module's helper processes will not be \
7820                 reaped on teardown"
7821            );
7822            // Resume regardless: leaving the child suspended would turn a
7823            // containment gap into a hung module.
7824            resume_suspended_child(pid, spec)?;
7825            return Ok(None);
7826        }
7827    };
7828
7829    if let Err(source) = job.assign(child) {
7830        warn!(
7831            module_id,
7832            error = %source,
7833            "could not assign the child to its job object; this module's helper processes \
7834             will not be reaped on teardown"
7835        );
7836        resume_suspended_child(pid, spec)?;
7837        return Ok(None);
7838    }
7839
7840    resume_suspended_child(pid, spec)?;
7841    Ok(Some(job))
7842}
7843
7844/// Resume a suspended child, killing it if it cannot be started.
7845///
7846/// A suspended process holds a pid and does nothing, so there is no useful
7847/// state to return: the caller gets an error and the spawn fails.
7848#[cfg(windows)]
7849fn resume_suspended_child(pid: u32, spec: &ModuleSpec) -> Result<(), SuperviseError> {
7850    if let Err(source) = subc_jobobject::resume_main_thread(pid) {
7851        // Kill it here rather than leaving a suspended process for the caller
7852        // to notice; `kill_on_drop` would eventually do this, but the module
7853        // would have been reported as running in between.
7854        let _ = std::process::Command::new("taskkill.exe")
7855            .args(["/PID", &pid.to_string(), "/T", "/F"])
7856            .stdin(Stdio::null())
7857            .stdout(Stdio::null())
7858            .stderr(Stdio::null())
7859            .status();
7860        return Err(SuperviseError::Spawn {
7861            program: spec.program.clone(),
7862            source,
7863            cgroup_path: None,
7864        });
7865    }
7866    Ok(())
7867}
7868
7869#[cfg(target_os = "linux")]
7870fn remove_module_cgroup(placement: &subc_cgroup::Placement, module_id: &str) {
7871    match placement.remove_module(module_id) {
7872        Ok(()) => debug!(module_id, "removed module cgroup after process exit"),
7873        Err(error) => warn!(
7874            module_id,
7875            error = %error,
7876            "could not remove module cgroup after process exit; continuing teardown"
7877        ),
7878    }
7879}
7880
7881#[cfg(target_os = "linux")]
7882async fn cleanup_reaped_cgroup(placement: &subc_cgroup::Placement, module_id: &str) {
7883    // Reaping the direct child is not proof its descendants exited. End the
7884    // residual tree and wait for the kernel's population fact before rmdir;
7885    // otherwise a successful parent wait leaks a directory on each restart.
7886    if matches!(
7887        subc_cgroup::kill_module(Some(placement), module_id),
7888        subc_cgroup::KillOutcome::Killed
7889    ) {
7890        if let Ok(path) = placement.module_path(module_id) {
7891            while std::fs::read_to_string(path.join("cgroup.events"))
7892                .ok()
7893                .is_some_and(|events| events.lines().any(|line| line == "populated 1"))
7894            {
7895                sleep(Duration::from_millis(1)).await;
7896            }
7897        }
7898    }
7899    remove_module_cgroup(placement, module_id);
7900}
7901
7902#[cfg(target_os = "linux")]
7903fn apply_cgroup_placement(
7904    command: &mut Command,
7905    spec: &ModuleSpec,
7906    path: &std::path::Path,
7907) -> Result<(), SuperviseError> {
7908    subc_cgroup::apply(command, path).map_err(|source| SuperviseError::Cgroup {
7909        module_id: spec.module_id.clone(),
7910        source,
7911    })
7912}
7913
7914fn capture_retention(spec: &ModuleSpec) -> Retention {
7915    let defaults = Retention::default();
7916    let value = |name: &str| {
7917        spec.env
7918            .iter()
7919            .rev()
7920            .find_map(|(key, value)| (key == name).then_some(value.as_str()))
7921    };
7922    Retention {
7923        max_file_mb: value(CAPTURE_MAX_FILE_MB_ENV)
7924            .and_then(|value| value.parse().ok())
7925            .unwrap_or(defaults.max_file_mb),
7926        keep: value(CAPTURE_KEEP_ENV)
7927            .and_then(|value| value.parse().ok())
7928            .unwrap_or(defaults.keep),
7929        max_age_days: value(CAPTURE_MAX_AGE_DAYS_ENV)
7930            .and_then(|value| value.parse().ok())
7931            .unwrap_or(defaults.max_age_days),
7932    }
7933}
7934
7935/// A fresh 256-bit CSPRNG launch nonce, lowercase hex. Used to bind a reserved
7936/// module's registration to the exact process the supervisor spawned.
7937fn generate_launch_nonce() -> Result<String, SuperviseError> {
7938    let mut bytes = [0u8; 32];
7939    getrandom::getrandom(&mut bytes).map_err(|source| SuperviseError::LaunchNonce {
7940        reason: source.to_string(),
7941    })?;
7942    let mut hex = String::with_capacity(64);
7943    for b in bytes {
7944        use std::fmt::Write;
7945        let _ = write!(hex, "{b:02x}");
7946    }
7947    Ok(hex)
7948}
7949
7950/// Constant-time byte comparison so a reserved-nonce mismatch leaks no timing
7951/// signal about how many leading bytes matched.
7952fn constant_time_eq(a: &[u8], b: &[u8]) -> bool {
7953    if a.len() != b.len() {
7954        return false;
7955    }
7956    let mut diff = 0u8;
7957    for (x, y) in a.iter().zip(b.iter()) {
7958        diff |= x ^ y;
7959    }
7960    diff == 0
7961}
7962
7963/// The kernel's image after an acknowledged exec, shared by ordinary launches
7964/// and privacy trampolines. PATH and shebang interpretation are kernel facts,
7965/// not identities inferred from a configured pathname.
7966fn observe_spawned_image(pid: u32) -> Option<subc_os::Observation> {
7967    subc_os::Process::open(pid)
7968        .ok()
7969        .flatten()
7970        .and_then(|process| process.observe())
7971}
7972
7973fn spawn_and_mark_running(
7974    spec: &ModuleSpec,
7975    runtime: &SupervisorRuntimeConfig,
7976    snapshot: &SharedSnapshot,
7977) -> Result<SupervisedChild, SuperviseError> {
7978    let child = spawn_child(
7979        spec,
7980        runtime.connection_file_path.as_deref(),
7981        runtime.supervisor_handle.as_ref(),
7982        &runtime.stderr_ring,
7983        runtime.capture_logs_dir.as_deref(),
7984        &runtime.child_roster,
7985        #[cfg(target_os = "linux")]
7986        runtime.cgroup_placement.as_ref(),
7987    )?;
7988    set_running(snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
7989    Ok(child)
7990}
7991
7992enum RegistrationWaitOutcome {
7993    Registered,
7994    Exited(ExitReport),
7995    TimedOut,
7996}
7997
7998struct ReloadRegistrationFailure {
7999    exit_report: ExitReport,
8000    reason: String,
8001}
8002
8003#[derive(Debug, Clone, Copy, PartialEq, Eq)]
8004enum BusyGaugeObservation {
8005    Quiescent,
8006    Busy,
8007    Omitted,
8008}
8009
8010fn busy_gauge_observation(metrics: Option<&Value>, gauges: &[String]) -> BusyGaugeObservation {
8011    let Some(metrics) = metrics.and_then(Value::as_object) else {
8012        return BusyGaugeObservation::Omitted;
8013    };
8014    let mut sum = 0u128;
8015    for gauge in gauges {
8016        let Some(value) = metrics.get(gauge) else {
8017            return BusyGaugeObservation::Omitted;
8018        };
8019        let Some(value) = value.as_u64() else {
8020            return BusyGaugeObservation::Busy;
8021        };
8022        sum = sum.saturating_add(u128::from(value));
8023    }
8024    if sum == 0 {
8025        BusyGaugeObservation::Quiescent
8026    } else {
8027        BusyGaugeObservation::Busy
8028    }
8029}
8030
8031fn declared_busy_gauges(
8032    registry: &Registry,
8033    module_id: &str,
8034) -> Result<Vec<String>, SuperviseError> {
8035    busy_gauges_of(
8036        registry
8037            .get_module(module_id)
8038            .map_err(SuperviseError::Registry)?,
8039    )
8040}
8041
8042/// [`declared_busy_gauges`] for the registration a connection holds, in any
8043/// slot: after cutover the incumbent is no longer the id's active
8044/// registration, and its own manifest is the one that names its gauges.
8045fn declared_busy_gauges_for_connection(
8046    registry: &Registry,
8047    connection_id: ConnectionId,
8048) -> Result<Vec<String>, SuperviseError> {
8049    busy_gauges_of(
8050        registry
8051            .get_module_by_connection(connection_id)
8052            .map_err(SuperviseError::Registry)?,
8053    )
8054}
8055
8056fn busy_gauges_of(
8057    registration: Option<crate::registry::ModuleRegistration>,
8058) -> Result<Vec<String>, SuperviseError> {
8059    let Some(registration) = registration else {
8060        return Ok(Vec::new());
8061    };
8062    let Some(self_signals) = registration.manifest.self_signals else {
8063        return Ok(Vec::new());
8064    };
8065
8066    let mut gauges = Vec::new();
8067    for declaration in self_signals {
8068        if declaration.kind != SelfSignalKind::Busy {
8069            continue;
8070        }
8071        match declaration.anchored_to {
8072            SignalAnchor::HealthGauges { gauges: declared } if !declared.is_empty() => {
8073                gauges.extend(declared)
8074            }
8075            _ => {
8076                // An invalid Busy anchor is fail-safe: the empty name cannot be
8077                // present in a conforming health report, so this drain stays busy.
8078                gauges.push(String::new());
8079            }
8080        }
8081    }
8082    Ok(gauges)
8083}
8084
8085/// Wait for `endpoint` to have nothing in flight and, when the module declares
8086/// busy gauges, for a health probe to report them quiet. The probe is addressed
8087/// by `scope`: a swap's superseded incumbent must be asked about its own
8088/// gauges, and by module id the probe would reach the promoted candidate.
8089async fn wait_for_forwarding_quiescence(
8090    forwarding: &ForwardingTable,
8091    module_id: &str,
8092    runtime: &SupervisorRuntimeConfig,
8093    endpoint: crate::ModuleEndpointId,
8094    deadline: Instant,
8095    busy_gauges: &[String],
8096    scope: DrainScope,
8097) -> Result<bool, SuperviseError> {
8098    let mut gauges_quiescent = busy_gauges.is_empty();
8099    let mut next_probe_at = Instant::now();
8100    let mut omission_counted = false;
8101
8102    loop {
8103        let now = Instant::now();
8104        if !busy_gauges.is_empty() && now >= next_probe_at && now < deadline {
8105            let report = match scope {
8106                DrainScope::Active => probe_module_health(module_id, runtime, Some(deadline)).await,
8107                DrainScope::Endpoint(endpoint) => {
8108                    probe_endpoint_health(endpoint, runtime, Some(deadline)).await
8109                }
8110            };
8111            gauges_quiescent = match report {
8112                Ok(report) => match busy_gauge_observation(report.metrics.as_ref(), busy_gauges) {
8113                    BusyGaugeObservation::Quiescent => true,
8114                    BusyGaugeObservation::Busy => false,
8115                    BusyGaugeObservation::Omitted => {
8116                        if !omission_counted {
8117                            forwarding
8118                                .counters()
8119                                .increment_drains_with_undeclared_gauge();
8120                            omission_counted = true;
8121                        }
8122                        false
8123                    }
8124                },
8125                Err(err) => {
8126                    warn!(
8127                        module_id,
8128                        error = %err,
8129                        "drain health.check did not produce declared busy gauges; treating module as busy"
8130                    );
8131                    false
8132                }
8133            };
8134            next_probe_at = Instant::now() + runtime.health.cadence.max(REGISTRY_RELEASE_POLL);
8135        }
8136
8137        let in_flight = forwarding
8138            .endpoint_in_flight_count(endpoint)
8139            .map_err(SuperviseError::Forwarding)?;
8140        if in_flight == 0 && gauges_quiescent {
8141            return Ok(true);
8142        }
8143
8144        let now = Instant::now();
8145        if now >= deadline {
8146            return Ok(false);
8147        }
8148        let mut wait = deadline
8149            .saturating_duration_since(now)
8150            .min(REGISTRY_RELEASE_POLL);
8151        if !busy_gauges.is_empty() {
8152            wait = wait.min(next_probe_at.saturating_duration_since(now));
8153        }
8154        sleep(wait).await;
8155    }
8156}
8157
8158/// The `route.closed` `drained` value implied by a quiescence-wait outcome.
8159///
8160/// `Ok` is always honest and passed straight through -- the wait actually measured
8161/// in-flight state. `Err` means the wait produced no measurement at all (the
8162/// forwarding table's lock was poisoned), so `false` is reported as the one honest
8163/// constant: the drain did not complete. Never recomputed from route state, never a
8164/// third "unknown" value -- the caller must still send a well-formed `route.closed`.
8165fn drained_after_quiescence_wait(wait_result: &Result<bool, SuperviseError>) -> bool {
8166    match wait_result {
8167        Ok(drained) => *drained,
8168        Err(_) => false,
8169    }
8170}
8171
8172fn send_route_goodbyes(forwarding: &ForwardingTable, released_routes: Vec<GoodbyeTarget>) {
8173    for released in released_routes {
8174        let frame = match Frame::build_with_version(
8175            released.negotiated_ver,
8176            FrameType::Goodbye,
8177            control_flags(),
8178            released.channel,
8179            released.epoch,
8180            0,
8181            Vec::new(),
8182        ) {
8183            Ok(frame) => frame,
8184            Err(err) => {
8185                warn!(
8186                    route_channel = released.channel,
8187                    error = %err,
8188                    "failed to build supervisor drain route GOODBYE frame"
8189                );
8190                continue;
8191            }
8192        };
8193        if !released.close_on_delivery_failure() {
8194            crate::forwarding::send_module_route_goodbye(
8195                &forwarding.counters(),
8196                &released.sink,
8197                frame,
8198                released.module_id.as_deref(),
8199                "supervisor drain",
8200            );
8201            continue;
8202        }
8203        if let Err(err) = released.sink.try_send(frame) {
8204            warn!(
8205                target_connection_id = released.connection_id.get(),
8206                route_channel = released.channel,
8207                error = %err,
8208                "supervisor drain route GOODBYE was not delivered to client; closing target connection"
8209            );
8210            let _ = forwarding.escalate_client_delivery_failure(
8211                released.connection_id,
8212                released.channel,
8213                released.epoch,
8214                CloseReason::new(
8215                    "route_goodbye_delivery_failed",
8216                    format!(
8217                        "failed to enqueue supervisor drain route GOODBYE for channel {}: {err}",
8218                        released.channel
8219                    ),
8220                ),
8221                crate::forwarding::UndeliveredFrame {
8222                    module_id: released.module_id.as_deref(),
8223                    sink: &released.sink,
8224                },
8225            );
8226        }
8227    }
8228}
8229
8230fn send_module_draining(
8231    module_id: &str,
8232    reason: RouteCloseReason,
8233    deadline_ms: u64,
8234    target: &ModuleDrainTarget,
8235) {
8236    let body = match serde_json::to_vec(&ModuleControlCommand::Draining {
8237        reason,
8238        deadline_ms,
8239    }) {
8240        Ok(body) => body,
8241        Err(err) => {
8242            warn!(
8243                module_id,
8244                error = %err,
8245                "failed to encode module draining command"
8246            );
8247            return;
8248        }
8249    };
8250    let frame = match Frame::build_with_version(
8251        target.negotiated_ver,
8252        FrameType::Push,
8253        control_flags(),
8254        0,
8255        0,
8256        0,
8257        body,
8258    ) {
8259        Ok(frame) => frame,
8260        Err(err) => {
8261            warn!(
8262                module_id,
8263                error = %err,
8264                "failed to build module draining command frame"
8265            );
8266            return;
8267        }
8268    };
8269    if let Err(err) = target.sink.try_send(frame) {
8270        warn!(
8271            module_id,
8272            target_connection_id = target.endpoint.connection_id.get(),
8273            error = %err,
8274            "module draining command was not delivered to peer"
8275        );
8276    }
8277}
8278
8279/// The channel-0 GOODBYE that tells a module its stop is planned.
8280fn module_goodbye_frame(module_id: &str, negotiated_ver: u8) -> Option<Frame> {
8281    match Frame::build_with_version(
8282        negotiated_ver,
8283        FrameType::Goodbye,
8284        control_flags(),
8285        0,
8286        0,
8287        0,
8288        Vec::new(),
8289    ) {
8290        Ok(frame) => Some(frame),
8291        Err(err) => {
8292            warn!(
8293                module_id,
8294                error = %err,
8295                "failed to build module GOODBYE frame"
8296            );
8297            None
8298        }
8299    }
8300}
8301
8302/// Send every registered module connection its module GOODBYE at daemon
8303/// shutdown, then request that connection's close.
8304///
8305/// A module tells a planned stop from a lost daemon by whether a GOODBYE came
8306/// before EOF, so the GOODBYE must reach the socket before the close. A close
8307/// request does not wait for the connection's queued frames: its writer gets a
8308/// bounded grace after the close, is aborted if it overruns it, and the daemon
8309/// process may exit before that grace ends. So with `wait_for_flush`, each
8310/// connection is closed only after its writer has acknowledged writing the
8311/// GOODBYE, or once a short shared budget runs out, so one module that is not
8312/// reading cannot hold up the others or the shutdown. Without it the GOODBYEs
8313/// are only queued, for a shutdown the operator has told to stop waiting.
8314/// A connection that is already gone is skipped.
8315#[cfg(unix)]
8316async fn send_module_goodbyes_for_daemon_shutdown(
8317    forwarding: &Arc<ForwardingTable>,
8318    reason: &CloseReason,
8319    wait_for_flush: bool,
8320) {
8321    const GOODBYE_BUDGET: Duration = Duration::from_millis(500);
8322    let targets = match forwarding.module_connections() {
8323        Ok(targets) => targets,
8324        Err(err) => {
8325            warn!(error = %err, "could not list module connections for shutdown GOODBYE");
8326            return;
8327        }
8328    };
8329    let deadline = Instant::now() + GOODBYE_BUDGET;
8330    let mut sends = tokio::task::JoinSet::new();
8331    for target in targets {
8332        let Some(frame) = module_goodbye_frame(&target.module_id, target.negotiated_ver) else {
8333            continue;
8334        };
8335        if !wait_for_flush {
8336            if let Err(err) = target.sink.try_send(frame) {
8337                debug!(
8338                    module_id = %target.module_id,
8339                    error = %err,
8340                    "shutdown module GOODBYE was not queued"
8341                );
8342            }
8343            continue;
8344        }
8345        let forwarding = Arc::clone(forwarding);
8346        let reason = reason.clone();
8347        sends.spawn(async move {
8348            match timeout_at(deadline, target.sink.send_flushed(frame)).await {
8349                Ok(Ok(())) => {}
8350                Ok(Err(err)) => debug!(
8351                    module_id = %target.module_id,
8352                    error = %err,
8353                    "module connection closed before its shutdown GOODBYE was written"
8354                ),
8355                Err(_) => warn!(
8356                    module_id = %target.module_id,
8357                    budget = ?GOODBYE_BUDGET,
8358                    "shutdown module GOODBYE was not written within its budget; closing anyway"
8359                ),
8360            }
8361            forwarding.request_connection_close(target.endpoint.connection_id, reason);
8362        });
8363    }
8364    // Every task ends by the shared deadline, so this wait is bounded too.
8365    while sends.join_next().await.is_some() {}
8366}
8367
8368fn send_module_goodbye(module_id: &str, forwarding: &ForwardingTable, target: &ModuleDrainTarget) {
8369    let Some(frame) = module_goodbye_frame(module_id, target.negotiated_ver) else {
8370        return;
8371    };
8372    if let Err(err) = target.sink.try_send(frame) {
8373        warn!(
8374            module_id,
8375            target_connection_id = target.endpoint.connection_id.get(),
8376            error = %err,
8377            "supervisor drain module GOODBYE was not delivered to peer; closing module connection"
8378        );
8379        forwarding.request_connection_close(
8380            target.endpoint.connection_id,
8381            CloseReason::new(
8382                "module_goodbye_delivery_failed",
8383                format!("failed to enqueue supervisor drain module GOODBYE for module '{module_id}': {err}"),
8384            ),
8385        );
8386    }
8387}
8388
8389#[derive(Clone, Copy)]
8390struct ForwardingDrainContext<'a> {
8391    spec: &'a ModuleSpec,
8392    runtime: &'a SupervisorRuntimeConfig,
8393    registry: &'a Registry,
8394    scope: DrainScope,
8395}
8396
8397/// Which process a forwarding drain addresses.
8398#[derive(Debug, Clone, Copy, PartialEq, Eq)]
8399enum DrainScope {
8400    /// Whatever endpoint is active for the module id: every plain stop,
8401    /// restart and reload. Also moves the module's state to `Draining`.
8402    Active,
8403    /// One specific endpoint: a swap's incumbent after cutover. Draining it by
8404    /// module id would resolve to the promoted candidate and leave neither
8405    /// process routable. The module's state is left alone, since the promoted
8406    /// candidate is what it describes and that process is running.
8407    Endpoint(crate::ModuleEndpointId),
8408}
8409
8410/// Whether a child being drained has already been asked to stop by the time
8411/// its drain wait starts.
8412///
8413/// The drain wait is the same budget whatever this says. What it decides is
8414/// whether the supervisor must ask by signal before that wait begins: a child
8415/// that nobody asked will sit out the whole budget and then be SIGKILLed,
8416/// healthy or not.
8417#[derive(Debug, Clone, Copy, PartialEq, Eq)]
8418enum StopNotice {
8419    /// The module was sent `module.draining` and a module GOODBYE over its own
8420    /// registered connection, and stops itself.
8421    SentOverConnection,
8422    /// The forwarding drain found no registered connection for the module: a
8423    /// subc child spawned moments ago that has not sent HELLO yet, or a
8424    /// `protocol: "none"` child, which never registers.
8425    NoConnection,
8426    /// This path sends nothing over the module's connection: the supervisor has
8427    /// no forwarding table, or the caller stops the child without a forwarding
8428    /// drain.
8429    NotSent,
8430}
8431
8432async fn begin_forwarding_drain(
8433    spec: &ModuleSpec,
8434    runtime: &SupervisorRuntimeConfig,
8435    registry: &Registry,
8436    snapshot: &SharedSnapshot,
8437    enabled: Option<bool>,
8438    reason: RouteCloseReason,
8439) -> Result<StopNotice, SuperviseError> {
8440    let Some(forwarding) = runtime.forwarding.as_ref() else {
8441        return Err(SuperviseError::ReloadUnavailable {
8442            module_id: spec.module_id.clone(),
8443            reason: "supervisor was not configured with a forwarding table".to_string(),
8444        });
8445    };
8446
8447    begin_forwarding_drain_with(
8448        forwarding,
8449        ForwardingDrainContext {
8450            spec,
8451            runtime,
8452            registry,
8453            scope: DrainScope::Active,
8454        },
8455        snapshot,
8456        enabled,
8457        reason,
8458        runtime.drain_timeout,
8459    )
8460    .await
8461}
8462
8463async fn begin_forwarding_drain_if_configured(
8464    spec: &ModuleSpec,
8465    runtime: &SupervisorRuntimeConfig,
8466    registry: &Registry,
8467    snapshot: &SharedSnapshot,
8468    enabled: Option<bool>,
8469    reason: RouteCloseReason,
8470) -> Result<StopNotice, SuperviseError> {
8471    begin_forwarding_drain_with_timeout(
8472        spec,
8473        runtime,
8474        registry,
8475        snapshot,
8476        enabled,
8477        reason,
8478        runtime.drain_timeout,
8479    )
8480    .await
8481}
8482
8483/// Like [`begin_forwarding_drain_if_configured`] but with an explicit drain
8484/// budget, for paths where the operator overrides the module's configured one
8485/// (`supervisor.restart{drain_timeout_ms}`).
8486async fn begin_forwarding_drain_with_timeout(
8487    spec: &ModuleSpec,
8488    runtime: &SupervisorRuntimeConfig,
8489    registry: &Registry,
8490    snapshot: &SharedSnapshot,
8491    enabled: Option<bool>,
8492    reason: RouteCloseReason,
8493    drain_timeout: Duration,
8494) -> Result<StopNotice, SuperviseError> {
8495    let Some(forwarding) = runtime.forwarding.as_ref() else {
8496        return Ok(StopNotice::NotSent);
8497    };
8498
8499    begin_forwarding_drain_with(
8500        forwarding,
8501        ForwardingDrainContext {
8502            spec,
8503            runtime,
8504            registry,
8505            scope: DrainScope::Active,
8506        },
8507        snapshot,
8508        enabled,
8509        reason,
8510        drain_timeout,
8511    )
8512    .await
8513}
8514
8515async fn begin_forwarding_drain_with(
8516    forwarding: &ForwardingTable,
8517    context: ForwardingDrainContext<'_>,
8518    snapshot: &SharedSnapshot,
8519    enabled: Option<bool>,
8520    reason: RouteCloseReason,
8521    drain_timeout: Duration,
8522) -> Result<StopNotice, SuperviseError> {
8523    let ForwardingDrainContext {
8524        spec,
8525        runtime,
8526        registry,
8527        scope,
8528    } = context;
8529    debug_assert_ne!(reason, RouteCloseReason::Crash);
8530    let terminal = matches!(reason, RouteCloseReason::Disable);
8531    let drain_started_at = Instant::now();
8532    let drain_deadline = drain_started_at + drain_timeout;
8533    let deadline_ms =
8534        unix_ms_now().saturating_add(u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX));
8535    let busy_gauges = match scope {
8536        DrainScope::Active => declared_busy_gauges(registry, &spec.module_id)?,
8537        DrainScope::Endpoint(endpoint) => {
8538            declared_busy_gauges_for_connection(registry, endpoint.connection_id)?
8539        }
8540    };
8541
8542    // Admission gate first: route.open/commit and route REQUEST admission are closed
8543    // before the first quiescence check, so the outstanding count can only fall.
8544    let gate_started = Instant::now();
8545    let drain_target = match scope {
8546        DrainScope::Active => forwarding.begin_module_drain(&spec.module_id, reason),
8547        DrainScope::Endpoint(endpoint) => forwarding.begin_endpoint_drain(endpoint, reason),
8548    }
8549    .map_err(SuperviseError::Forwarding)?;
8550    // The instant admission closed, and how long taking the forwarding write
8551    // lock to close it took. The timeout line reports only the quiescence
8552    // wait, so without this a drain that started late looked like one that
8553    // started on time.
8554    info!(
8555        module_id = %spec.module_id,
8556        ?reason,
8557        gate_ms = u64::try_from(gate_started.elapsed().as_millis()).unwrap_or(u64::MAX),
8558        connected = drain_target.is_some(),
8559        "module drain began; route admission closed"
8560    );
8561    if scope == DrainScope::Active {
8562        update_snapshot(snapshot, Some(&spec.module_id), |state| {
8563            state.state = ModuleState::Draining;
8564            state.draining_to_replace =
8565                matches!(reason, RouteCloseReason::Restart | RouteCloseReason::Reload);
8566            if let Some(enabled) = enabled {
8567                state.enabled = enabled;
8568            }
8569        })?;
8570    }
8571
8572    let Some(target) = drain_target.as_ref() else {
8573        // Nothing was sent: the module has no registered connection to carry
8574        // `module.draining` or a GOODBYE. The caller must not assume the child
8575        // was asked to stop.
8576        return Ok(StopNotice::NoConnection);
8577    };
8578    {
8579        send_module_draining(&spec.module_id, reason, deadline_ms, target);
8580        let routes = forwarding
8581            .endpoint_routes(target.endpoint)
8582            .map_err(SuperviseError::Forwarding)?;
8583        let routes_notified = routes.len();
8584        crate::control::send_route_control_pushes(
8585            forwarding,
8586            routes.clone(),
8587            ClientControlPush::RouteClosing {
8588                module_id: spec.module_id.clone(),
8589                channels: Vec::new(),
8590                reason,
8591            },
8592        );
8593        send_route_goodbyes(forwarding, target.abandoned_bindings.clone());
8594
8595        // `route.closing` was just sent above: from here on every return path,
8596        // including an early one, MUST send `route.closed` before propagating
8597        // anything else. A client holds `closing` as a promise that a verdict is
8598        // coming; leaving early without `closed` strands it waiting forever, since
8599        // `closing` carries no timeout of its own.
8600        let wait_result = wait_for_forwarding_quiescence(
8601            forwarding,
8602            &spec.module_id,
8603            runtime,
8604            target.endpoint,
8605            drain_deadline,
8606            &busy_gauges,
8607            scope,
8608        )
8609        .await;
8610        let drained = drained_after_quiescence_wait(&wait_result);
8611        if let Err(err) = &wait_result {
8612            error!(
8613                module_id = %spec.module_id,
8614                ?reason,
8615                error = %err,
8616                "forwarding quiescence wait failed after route.closing; forcing route.closed(drained: false) so the client is not left waiting on an unfulfilled promise"
8617            );
8618        } else if !drained {
8619            // Name what the drain waited on. Without it the line says only that
8620            // something did not settle, and "one wedged call" and "every
8621            // session's held stream" read the same; the first is a module bug,
8622            // the second is a module that should end its streams on
8623            // module.draining. Read before teardown releases the routes.
8624            let holdouts = forwarding
8625                .endpoint_drain_holdouts(target.endpoint)
8626                .unwrap_or_default();
8627            warn!(
8628                module_id = %spec.module_id,
8629                waited = ?drain_timeout,
8630                ?reason,
8631                held_requests = holdouts.requests,
8632                held_routes = holdouts.routes,
8633                total_routes = holdouts.total_routes,
8634                top_connections = ?holdouts.top_connections,
8635                // `module_channel:corr`, so the module can find each held request
8636                // in its own log; capped, so `held_requests` is the full count.
8637                held = %holdouts
8638                    .held
8639                    .iter()
8640                    .map(|(channel, corr)| format!("{channel}:{corr}"))
8641                    .collect::<Vec<_>>()
8642                    .join(","),
8643                "route drain timed out before request quiescence; forcing teardown"
8644            );
8645        }
8646        crate::control::send_route_control_pushes(
8647            forwarding,
8648            routes,
8649            ClientControlPush::RouteClosed {
8650                module_id: spec.module_id.clone(),
8651                channels: Vec::new(),
8652                reason,
8653                drained,
8654                abandoned: target.abandoned_bindings.len() as u32,
8655                excluded_subscriptions: target.excluded_subscriptions,
8656                terminal: Some(terminal),
8657            },
8658        );
8659        wait_result?;
8660
8661        // `route.closed` has now been sent unconditionally above. From here the
8662        // remaining steps are cleanup (route + module GOODBYE) rather than a
8663        // promise the client is waiting on, but a lock-poisoned
8664        // `release_module_endpoint_routes` would otherwise skip the module
8665        // GOODBYE silently too -- send it before propagating the error.
8666        let released_routes = match forwarding.release_module_endpoint_routes(target.endpoint) {
8667            Ok(routes) => routes,
8668            Err(err) => {
8669                warn!(
8670                    module_id = %spec.module_id,
8671                    ?reason,
8672                    error = %err,
8673                    "failed to release module endpoint routes after route.closed; module GOODBYE will still be sent"
8674                );
8675                send_module_goodbye(&spec.module_id, forwarding, target);
8676                return Err(SuperviseError::Forwarding(err));
8677            }
8678        };
8679        let route_goodbye_count = released_routes.len();
8680        send_route_goodbyes(forwarding, released_routes);
8681        send_module_goodbye(&spec.module_id, forwarding, target);
8682
8683        // The drain's happy path was previously silent: every emission above is
8684        // best-effort with only its failure arm logged, so "were consumers told"
8685        // was unprovable from the daemon log (surfaced by a 30-minute consumer
8686        // hang where the open question was exactly whether teardown notice went
8687        // out). One summary line makes that class decidable in one grep.
8688        info!(
8689            module_id = %spec.module_id,
8690            ?reason,
8691            routes_notified,
8692            route_goodbyes = route_goodbye_count,
8693            abandoned_reservations = target.abandoned_bindings.len(),
8694            excluded_subscriptions = target.excluded_subscriptions,
8695            drained,
8696            "module drain complete; consumers notified via route.closing/route.closed pushes and per-route GOODBYE frames"
8697        );
8698    }
8699
8700    Ok(StopNotice::SentOverConnection)
8701}
8702
8703/// Wait for the freshly spawned child to take the ACTIVE slot for `module_id`,
8704/// the only slot a plain (non-swap) spawn can register into.
8705async fn wait_for_registration_after_reload(
8706    registry: &Registry,
8707    module_id: &str,
8708    snapshot: &SharedSnapshot,
8709    child: &mut SupervisedChild,
8710    wait: Duration,
8711) -> Result<RegistrationWaitOutcome, SuperviseError> {
8712    wait_for_slot_registration(
8713        registry,
8714        crate::registry::RegistrationSlot::Active(module_id),
8715        module_id,
8716        snapshot,
8717        child,
8718        wait,
8719    )
8720    .await
8721}
8722
8723/// Wait for `child` to register into `slot`, or to exit, or for `wait` to pass.
8724///
8725/// Keyed on the slot rather than the bare module id because during a swap the
8726/// id's active slot is already held by the incumbent: an id-keyed wait would
8727/// report the incumbent's registration as the candidate's and a candidate that
8728/// never registers would look registered. A swap candidate waits on
8729/// `crate::registry::RegistrationSlot::Candidate`.
8730async fn wait_for_slot_registration(
8731    registry: &Registry,
8732    slot: crate::registry::RegistrationSlot<'_>,
8733    module_id: &str,
8734    snapshot: &SharedSnapshot,
8735    child: &mut SupervisedChild,
8736    wait: Duration,
8737) -> Result<RegistrationWaitOutcome, SuperviseError> {
8738    let deadline = Instant::now() + wait;
8739    loop {
8740        if registry
8741            .registration(slot)
8742            .map_err(SuperviseError::Registry)?
8743            .is_some()
8744        {
8745            return Ok(RegistrationWaitOutcome::Registered);
8746        }
8747
8748        let now = Instant::now();
8749        if now >= deadline {
8750            return Ok(RegistrationWaitOutcome::TimedOut);
8751        }
8752        let remaining = deadline.saturating_duration_since(now);
8753        let poll = remaining.min(REGISTRY_RELEASE_POLL);
8754
8755        tokio::select! {
8756            wait_result = child.wait() => {
8757                let status = wait_result.map_err(|source| SuperviseError::Wait {
8758                    module_id: module_id.to_string(),
8759                    source,
8760                })?;
8761                return Ok(RegistrationWaitOutcome::Exited(classify_reaped_child_exit(
8762                    snapshot,
8763                    child,
8764                    &status,
8765                )));
8766            }
8767            _ = sleep(poll) => {}
8768        }
8769    }
8770}
8771
8772fn registration_failure_exit_report(mut exit_report: ExitReport) -> ExitReport {
8773    // A replacement process that exits before HELLO did not provide service, even
8774    // if it used status 0. Count it against the restart cap as a new-binary failure.
8775    if exit_report.kind != ExitKind::DeliberateSeverance {
8776        exit_report.kind = ExitKind::Crash;
8777    }
8778    exit_report
8779}
8780
8781async fn handle_reload_child_registration_failure(
8782    spec: &ModuleSpec,
8783    runtime: &SupervisorRuntimeConfig,
8784    registry: &Registry,
8785    process_liveness: &SupervisorProcessLiveness,
8786    snapshot: &SharedSnapshot,
8787    _child: &mut Option<SupervisedChild>,
8788    failure: ReloadRegistrationFailure,
8789) -> Result<(), SuperviseError> {
8790    let ReloadRegistrationFailure {
8791        exit_report,
8792        reason,
8793    } = failure;
8794    match on_child_exit(
8795        spec,
8796        runtime.restart_policy,
8797        registry,
8798        snapshot,
8799        &runtime.terminal_ring,
8800        &runtime.spawn_events,
8801        &runtime.child_roster,
8802        exit_report,
8803    )
8804    .await
8805    {
8806        NextAction::Stop {
8807            registration_released,
8808        } => {
8809            if registration_released {
8810                process_liveness.untrack_if_current(&spec.module_id, snapshot);
8811            }
8812        }
8813        NextAction::Restart { schedule } => {
8814            let delay = schedule.map_or(runtime.restart_policy.delay_for_restart(0), |schedule| {
8815                schedule.delay
8816            });
8817            if let Some(schedule) = schedule {
8818                log_crash_respawn(&spec.module_id, schedule);
8819            }
8820            schedule_respawn(
8821                runtime,
8822                snapshot,
8823                &spec.module_id,
8824                delay,
8825                RespawnKind::Spawn,
8826            )?;
8827        }
8828    }
8829    Err(SuperviseError::ReloadFailed {
8830        module_id: spec.module_id.clone(),
8831        reason,
8832    })
8833}
8834
8835async fn handle_reload_spawn_failure(
8836    spec: &ModuleSpec,
8837    runtime: &SupervisorRuntimeConfig,
8838    process_liveness: &SupervisorProcessLiveness,
8839    snapshot: &SharedSnapshot,
8840    _child: &mut Option<SupervisedChild>,
8841    reason: String,
8842) -> Result<(), SuperviseError> {
8843    let now = Instant::now();
8844    let mut schedule = None;
8845    update_snapshot(snapshot, Some(&spec.module_id), |state| {
8846        clear_current_process_facts(state);
8847        if state.enabled {
8848            schedule = state.next_crash_restart(&runtime.restart_policy, now);
8849            state.state = if schedule.is_some() {
8850                ModuleState::Restarting
8851            } else {
8852                ModuleState::Failed
8853            };
8854        } else {
8855            state.state = ModuleState::Disabled;
8856        }
8857    })?;
8858    if let Some(schedule) = schedule {
8859        schedule_respawn(
8860            runtime,
8861            snapshot,
8862            &spec.module_id,
8863            schedule.delay,
8864            RespawnKind::Spawn,
8865        )?;
8866    } else {
8867        process_liveness.untrack_if_current(&spec.module_id, snapshot);
8868    }
8869    Err(SuperviseError::ReloadFailed {
8870        module_id: spec.module_id.clone(),
8871        reason,
8872    })
8873}
8874
8875fn control_flags() -> Flags {
8876    Flags::new(false, Priority::Passive, false)
8877}
8878
8879#[allow(clippy::too_many_arguments)]
8880async fn drain_optional_child(
8881    module_id: &str,
8882    protocol: ModuleProtocol,
8883    stop_notice: StopNotice,
8884    registry: &Registry,
8885    forwarding: Option<&ForwardingTable>,
8886    snapshot: &SharedSnapshot,
8887    terminal_ring: &Arc<Mutex<TerminalRing>>,
8888    spawn_events: &SpawnEventFeed,
8889    child: &mut Option<SupervisedChild>,
8890    drain_timeout: Duration,
8891    final_state: ModuleState,
8892    enabled: Option<bool>,
8893) -> Result<(), SuperviseError> {
8894    if let Some(child) = child.take() {
8895        drain_child_to_state(
8896            module_id,
8897            protocol,
8898            stop_notice,
8899            registry,
8900            forwarding,
8901            snapshot,
8902            terminal_ring,
8903            spawn_events,
8904            child,
8905            drain_timeout,
8906            final_state,
8907            enabled,
8908        )
8909        .await
8910    } else {
8911        update_snapshot(snapshot, Some(module_id), |state| {
8912            state.state = final_state;
8913            if let Some(enabled) = enabled {
8914                state.enabled = enabled;
8915            }
8916            clear_current_process_facts(state);
8917        })?;
8918        release_dead_registration(registry, forwarding, snapshot, module_id).await
8919    }
8920}
8921
8922#[allow(clippy::too_many_arguments)]
8923async fn drain_child_to_state(
8924    module_id: &str,
8925    _protocol: ModuleProtocol,
8926    stop_notice: StopNotice,
8927    registry: &Registry,
8928    forwarding: Option<&ForwardingTable>,
8929    snapshot: &SharedSnapshot,
8930    terminal_ring: &Arc<Mutex<TerminalRing>>,
8931    spawn_events: &SpawnEventFeed,
8932    mut child: SupervisedChild,
8933    drain_timeout: Duration,
8934    final_state: ModuleState,
8935    enabled: Option<bool>,
8936) -> Result<(), SuperviseError> {
8937    let protocol = child.protocol;
8938    update_snapshot(snapshot, Some(module_id), |state| {
8939        state.state = ModuleState::Draining;
8940        state.draining_to_replace = final_state == ModuleState::Restarting;
8941        if let Some(enabled) = enabled {
8942            state.enabled = enabled;
8943        }
8944    })?;
8945
8946    // The wait below is the same budget in every case; what differs is
8947    // whether anything has ASKED the child to stop before it starts. Only a
8948    // forwarding drain that reached the module's registered connection has
8949    // (`module.draining`, then a module GOODBYE). Every other child was told
8950    // nothing: a `protocol: "none"` module, which never registers; a subc
8951    // module spawned moments ago that has not sent HELLO yet; or a stop that
8952    // runs no forwarding drain. Without a signal the budget is only a delay
8953    // in front of SIGKILL -- and the not-yet-registered child is the worst
8954    // case, because it registers into a module that is already draining,
8955    // is never told, and is killed while healthy.
8956    if stop_notice != StopNotice::SentOverConnection {
8957        if protocol == ModuleProtocol::Subc && stop_notice == StopNotice::NoConnection {
8958            info!(
8959                module_id,
8960                pid = child.pid,
8961                budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
8962                "module has no connection yet; requesting stop by signal"
8963            );
8964        }
8965        request_graceful_stop(module_id, &child);
8966    }
8967
8968    let exit_report = match timeout(drain_timeout, child.wait()).await {
8969        Ok(Ok(status)) => classify_reaped_child_exit(snapshot, &child, &status),
8970        Ok(Err(source)) => {
8971            fail_snapshot(snapshot, Some(module_id), None);
8972            return Err(SuperviseError::Wait {
8973                module_id: module_id.to_string(),
8974                source,
8975            });
8976        }
8977        Err(_) => {
8978            // Mirror the sibling arm above: state is already `Draining`, and an
8979            // error propagated from here would strand it there -- a state
8980            // `set_enabled(true)` cannot heal (`revive_terminal` matches only
8981            // `Failed | Stopped`), leaving an operator Restart as the only exit.
8982            // `Failed` before `?` keeps the module operator-visible and
8983            // revivable. Trigger is an ESRCH race (process exits between the
8984            // drain timeout firing and the kill) or a post-kill wait failure
8985            // (issue #34).
8986            //
8987            // Logged because the kill is otherwise visible only as signal 9 in
8988            // the terminal ring, and the budget it follows can be long enough
8989            // that consumers see a stretch of refusals with no stated cause.
8990            warn!(
8991                module_id,
8992                pid = child.pid,
8993                budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
8994                reason = ?final_state,
8995                ?stop_notice,
8996                "drain budget expired before the module exited; killing it"
8997            );
8998            child.start_kill().map_err(|source| {
8999                fail_snapshot(snapshot, Some(module_id), None);
9000                SuperviseError::Kill {
9001                    module_id: module_id.to_string(),
9002                    source,
9003                }
9004            })?;
9005            let status = child.wait().await.map_err(|source| {
9006                fail_snapshot(snapshot, Some(module_id), None);
9007                SuperviseError::Wait {
9008                    module_id: module_id.to_string(),
9009                    source,
9010                }
9011            })?;
9012            classify_reaped_child_exit(snapshot, &child, &status)
9013        }
9014    };
9015
9016    update_snapshot(snapshot, Some(module_id), |state| {
9017        state.state = final_state;
9018        if let Some(enabled) = enabled {
9019            state.enabled = enabled;
9020        }
9021        clear_current_process_facts(state);
9022        state.last_exit = Some(exit_report.clone());
9023        if exit_report.kind == ExitKind::DeliberateSeverance {
9024            state.lifetime_restarts += 1;
9025        }
9026    })?;
9027    let detail = lock_snapshot(snapshot)?.drain_disposition_detail.take();
9028    record_terminal_with_detail(
9029        module_id,
9030        terminal_ring,
9031        spawn_events,
9032        &exit_report,
9033        terminal_disposition(final_state),
9034        detail,
9035    );
9036    child.drain_stderr(module_id).await;
9037
9038    release_dead_registration(registry, forwarding, snapshot, module_id).await
9039}
9040
9041/// Ask a child that nothing else has asked to stop, by signal.
9042///
9043/// A registered subc module is asked over its own connection: the drain sends
9044/// `route.closing`/`route.closed` to its consumers, a GOODBYE per route, then a
9045/// module GOODBYE, and the module stops itself. A module that speaks no subc
9046/// wire receives none of that, and neither does a subc module that has not
9047/// registered yet, so for them the drain budget would be pure delay in front of
9048/// a SIGKILL -- and for a process with a store to flush (JetStream is the
9049/// reason `protocol: "none"` exists) a SIGKILL turns every ordinary teardown
9050/// into a recovery on the next start.
9051///
9052/// NEVER CALLED FOR A MODULE THAT WAS TOLD OVER ITS CONNECTION, and that is a
9053/// rule rather than an optimisation: that module's graceful stop is already
9054/// running by the time its child is drained, and a signal would race it.
9055///
9056/// Best-effort by construction. A child that has already exited is the ordinary
9057/// case rather than an error (the kill lands on a reaped or exiting pid), so a
9058/// failure is logged at debug and the wait-then-kill below still decides the
9059/// outcome.
9060#[cfg(unix)]
9061fn request_graceful_stop(module_id: &str, child: &SupervisedChild) {
9062    let Some(pid) = child
9063        .id()
9064        .and_then(|pid| i32::try_from(pid).ok())
9065        .and_then(rustix::process::Pid::from_raw)
9066    else {
9067        debug!(
9068            module_id,
9069            "no pid to signal for teardown; falling through to the drain wait"
9070        );
9071        return;
9072    };
9073    match rustix::process::kill_process(pid, rustix::process::Signal::TERM) {
9074        Ok(()) => debug!(
9075            module_id,
9076            "sent SIGTERM to a module nothing else asked to stop"
9077        ),
9078        Err(err) => debug!(
9079            module_id,
9080            error = %err,
9081            "SIGTERM to module failed; the drain wait and kill still apply"
9082        ),
9083    }
9084}
9085
9086/// Windows has no SIGTERM and no portable stand-in for one. The graceful stops
9087/// Windows does offer need cooperation this supervisor cannot assume: a console
9088/// control event requires sharing a console with the child, and `WM_CLOSE`
9089/// requires the child to pump a message loop. A supervised server process does
9090/// neither, so there is nothing to send and teardown is the wait followed by the
9091/// kill. Emulating a signal here would mean inventing a stop protocol, which is
9092/// the thing `protocol: "none"` exists to avoid.
9093#[cfg(not(unix))]
9094fn request_graceful_stop(module_id: &str, _child: &SupervisedChild) {
9095    debug!(
9096        module_id,
9097        "no graceful stop signal exists on this platform; teardown of a module nothing asked to stop waits, then kills"
9098    );
9099}
9100
9101fn terminal_disposition(final_state: ModuleState) -> TerminalDisposition {
9102    match final_state {
9103        ModuleState::Stopped => TerminalDisposition::Stopped,
9104        ModuleState::Disabled => TerminalDisposition::Disabled,
9105        ModuleState::Restarting => TerminalDisposition::Restarting,
9106        ModuleState::Failed => TerminalDisposition::Failed,
9107        ModuleState::Starting
9108        | ModuleState::Running
9109        | ModuleState::Unresponsive
9110        | ModuleState::Draining => {
9111            unreachable!("terminal exits only finish in terminal or restarting states")
9112        }
9113    }
9114}
9115
9116/// Release a reaped child's registration before allowing another spawn.
9117///
9118/// EOF is not a process-lifetime signal: an inherited socket can stay open
9119/// indefinitely, and serial frame dispatch can be waiting on egress instead of
9120/// reading EOF. After the normal release grace, request connection close (which
9121/// cancels both reads and dispatch), then allow one more release grace for the
9122/// connection guard's forwarding cleanup. Never evict a different connection.
9123async fn release_dead_registration(
9124    registry: &Registry,
9125    forwarding: Option<&ForwardingTable>,
9126    snapshot: &SharedSnapshot,
9127    module_id: &str,
9128) -> Result<(), SuperviseError> {
9129    let result = async {
9130        let registration = registry
9131            .get_module(module_id)
9132            .map_err(SuperviseError::Registry)?;
9133        match wait_for_registration_release(registry, module_id, REGISTRY_RELEASE_TIMEOUT).await {
9134            Ok(()) => return Ok(()),
9135            Err(SuperviseError::RegistrationStillActive { .. }) => {}
9136            Err(err) => return Err(err),
9137        }
9138        let pid = lock_snapshot(snapshot)?.reaped_pid;
9139        if let (Some(registration), Some(forwarding), Some(pid)) = (registration, forwarding, pid) {
9140            warn!(
9141                module_id,
9142                pid,
9143                connection_id = registration.connection_id.get(),
9144                "reaped module registration outlived release grace; closing dead connection"
9145            );
9146            forwarding.request_connection_close(
9147                registration.connection_id,
9148                CloseReason::new(
9149                    "supervised_process_reaped",
9150                    format!("module '{module_id}' pid {pid} exited"),
9151                ),
9152            );
9153            wait_for_slot_registration_release(
9154                registry,
9155                crate::registry::RegistrationSlot::Connection(registration.connection_id),
9156                REGISTRY_RELEASE_TIMEOUT,
9157            )
9158            .await?;
9159        }
9160        wait_for_registration_release(registry, module_id, Duration::ZERO).await
9161    }
9162    .await;
9163    if let Err(err) = &result {
9164        fail_snapshot(snapshot, Some(module_id), None);
9165        error!(module_id, error = %err, "registration release failed after child exit; module is failed and start can retry");
9166    }
9167    result
9168}
9169
9170/// Wait for the ACTIVE registration of `module_id` to go away, which is what a
9171/// plain stop or restart waits for before it spawns a replacement.
9172async fn wait_for_registration_release(
9173    registry: &Registry,
9174    module_id: &str,
9175    wait: Duration,
9176) -> Result<(), SuperviseError> {
9177    wait_for_slot_registration_release(
9178        registry,
9179        crate::registry::RegistrationSlot::Active(module_id),
9180        wait,
9181    )
9182    .await
9183}
9184
9185/// Wait for the registration in `slot` to go away.
9186///
9187/// Keyed on the slot rather than the bare module id because a successful swap
9188/// never empties the id's active slot (the promoted candidate is in it), so an
9189/// id-keyed wait for the incumbent's release would always time out. Draining a
9190/// swap's incumbent waits on `crate::registry::RegistrationSlot::Connection` with the
9191/// incumbent's connection instead.
9192async fn wait_for_slot_registration_release(
9193    registry: &Registry,
9194    slot: crate::registry::RegistrationSlot<'_>,
9195    wait: Duration,
9196) -> Result<(), SuperviseError> {
9197    let deadline = Instant::now() + wait;
9198    let mut release_events = registration_release_events().subscribe();
9199    let still_active = |registration: &crate::registry::ModuleRegistration| {
9200        SuperviseError::RegistrationStillActive {
9201            module_id: registration.manifest.module_id.clone(),
9202            waited: wait,
9203        }
9204    };
9205    loop {
9206        let _observed_generation = *release_events.borrow_and_update();
9207        let Some(registration) = registry
9208            .registration(slot)
9209            .map_err(SuperviseError::Registry)?
9210        else {
9211            return Ok(());
9212        };
9213
9214        let now = Instant::now();
9215        if now >= deadline {
9216            return Err(still_active(&registration));
9217        }
9218
9219        let remaining = deadline.saturating_duration_since(now);
9220        match timeout(remaining, release_events.changed()).await {
9221            Ok(Ok(())) | Ok(Err(_)) => {}
9222            Err(_) => return Err(still_active(&registration)),
9223        }
9224    }
9225}
9226
9227#[cfg(test)]
9228mod slot_registration_wait_tests {
9229    use super::*;
9230    use crate::registry::{ConnectionId, RegistrationSlot};
9231    use subc_protocol::manifest::ModuleManifest;
9232
9233    #[tokio::test]
9234    async fn enable_release_failure_is_failed_and_a_second_enable_retries() {
9235        let registry = Arc::new(Registry::default());
9236        let supervisor = Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::default());
9237        let runtime = supervisor.runtime_config();
9238        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
9239        let spec = ModuleSpec {
9240            module_id: "enable-stale-registration".to_string(),
9241            program: PathBuf::from("/missing/enable-retry-test"),
9242            args: Vec::new(),
9243            env: Vec::new(),
9244            reserved: false,
9245            reserved_prefixes: Vec::new(),
9246            protocol: ModuleProtocol::Subc,
9247            overlap: Default::default(),
9248        };
9249        let connection = ConnectionId::new(90);
9250        registry
9251            .register_with_control_ops(
9252                ModuleManifest::builder(&spec.module_id, "0.1.0").build(),
9253                1,
9254                connection,
9255                Vec::new(),
9256            )
9257            .unwrap();
9258        let mut child = None;
9259        let err = set_child_enabled(
9260            &spec,
9261            &runtime,
9262            &registry,
9263            &supervisor.process_liveness,
9264            &snapshot,
9265            &mut child,
9266            true,
9267        )
9268        .await
9269        .unwrap_err();
9270        assert!(matches!(
9271            err,
9272            SuperviseError::RegistrationStillActive { .. }
9273        ));
9274        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
9275        assert!(child.is_none());
9276        registry.deregister_connection(connection).unwrap();
9277        let err = set_child_enabled(
9278            &spec,
9279            &runtime,
9280            &registry,
9281            &supervisor.process_liveness,
9282            &snapshot,
9283            &mut child,
9284            true,
9285        )
9286        .await
9287        .unwrap_err();
9288        assert!(
9289            matches!(err, SuperviseError::Spawn { .. }),
9290            "second enable must attempt a spawn: {err}"
9291        );
9292        assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
9293    }
9294
9295    const INCUMBENT: u64 = 1;
9296    const CANDIDATE: u64 = 2;
9297
9298    fn swapped_registry() -> Arc<Registry> {
9299        let registry = Arc::new(Registry::default());
9300        let manifest = ModuleManifest::builder("m", "0.1.0").build();
9301        registry
9302            .register_with_control_ops(
9303                manifest.clone(),
9304                1,
9305                ConnectionId::new(INCUMBENT),
9306                Vec::new(),
9307            )
9308            .unwrap();
9309        registry
9310            .register_candidate_with_control_ops(
9311                manifest,
9312                1,
9313                ConnectionId::new(CANDIDATE),
9314                Vec::new(),
9315            )
9316            .unwrap();
9317        registry
9318    }
9319
9320    /// After a promotion the id's active slot is held by the new process, so an
9321    /// id-keyed wait for the incumbent's release can never succeed; the
9322    /// connection-keyed wait completes as soon as the incumbent deregisters.
9323    #[tokio::test]
9324    async fn incumbent_release_is_awaited_by_connection_not_by_module_id() {
9325        let registry = swapped_registry();
9326        registry.promote_candidate("m").unwrap().unwrap();
9327
9328        assert!(matches!(
9329            wait_for_registration_release(&registry, "m", Duration::from_millis(50)).await,
9330            Err(SuperviseError::RegistrationStillActive { .. })
9331        ));
9332
9333        // Still held while the incumbent's connection has not deregistered.
9334        assert!(matches!(
9335            wait_for_slot_registration_release(
9336                &registry,
9337                RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
9338                Duration::from_millis(50),
9339            )
9340            .await,
9341            Err(SuperviseError::RegistrationStillActive { .. })
9342        ));
9343
9344        let releaser = Arc::clone(&registry);
9345        let release = tokio::spawn(async move {
9346            sleep(Duration::from_millis(20)).await;
9347            releaser
9348                .deregister_connection(ConnectionId::new(INCUMBENT))
9349                .unwrap();
9350            notify_registration_release();
9351        });
9352        wait_for_slot_registration_release(
9353            &registry,
9354            RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
9355            Duration::from_secs(5),
9356        )
9357        .await
9358        .expect("the incumbent's own registration is released");
9359        release.await.unwrap();
9360        assert!(registry.get_module("m").unwrap().is_some());
9361    }
9362
9363    /// The candidate slot is waited on separately from the active slot: the
9364    /// incumbent's registration neither holds up nor stands in for it.
9365    #[tokio::test]
9366    async fn candidate_slot_wait_ignores_the_incumbents_registration() {
9367        let registry = swapped_registry();
9368        assert!(matches!(
9369            wait_for_slot_registration_release(
9370                &registry,
9371                RegistrationSlot::Candidate("m"),
9372                Duration::from_millis(50),
9373            )
9374            .await,
9375            Err(SuperviseError::RegistrationStillActive { .. })
9376        ));
9377        registry
9378            .deregister_connection(ConnectionId::new(CANDIDATE))
9379            .unwrap();
9380        wait_for_slot_registration_release(
9381            &registry,
9382            RegistrationSlot::Candidate("m"),
9383            Duration::from_millis(50),
9384        )
9385        .await
9386        .expect("a candidate slot with no candidate is released");
9387        assert!(registry
9388            .registration(RegistrationSlot::Active("m"))
9389            .unwrap()
9390            .is_some());
9391    }
9392}
9393
9394fn classify_exit(status: &ExitStatus) -> ExitReport {
9395    ExitReport {
9396        kind: if status.success() {
9397            ExitKind::Clean
9398        } else {
9399            ExitKind::Crash
9400        },
9401        code: status.code(),
9402        signal: exit_signal(status),
9403        at_ms: unix_ms_now(),
9404    }
9405}
9406
9407/// The terminal record for a module whose `wait()` call itself errored (e.g. the
9408/// child was already reaped out-of-band). There is no `ExitStatus` to read a code
9409/// or signal from -- `None`/`None` is the honest shape, not a guess -- but the
9410/// disposition still must be `Failed` so the terminal ring is not silently missing
9411/// an entry, matching what `fail_snapshot` records for this same arm.
9412fn wait_error_exit_report() -> ExitReport {
9413    ExitReport {
9414        kind: ExitKind::Crash,
9415        code: None,
9416        signal: None,
9417        at_ms: unix_ms_now(),
9418    }
9419}
9420
9421#[cfg(unix)]
9422fn exit_signal(status: &ExitStatus) -> Option<i32> {
9423    use std::os::unix::process::ExitStatusExt;
9424
9425    status.signal()
9426}
9427
9428#[cfg(not(unix))]
9429fn exit_signal(_status: &ExitStatus) -> Option<i32> {
9430    None
9431}
9432
9433/// Give an operator-touched module its full crash budget back.
9434///
9435/// Named for the counter it used to zero; it now empties the in-window ring,
9436/// which is the same act. `lifetime_restarts` is untouched on purpose -- the
9437/// ledger of what happened survives every operator action.
9438fn reset_restart_count(snapshot: &SharedSnapshot, module_id: &str) -> Result<(), SuperviseError> {
9439    update_snapshot(snapshot, Some(module_id), |state| {
9440        state.clear_crash_restarts();
9441    })
9442}
9443
9444fn set_running(
9445    snapshot: &SharedSnapshot,
9446    child: &SupervisedChild,
9447    module_id: &str,
9448    spawn_events: &SpawnEventFeed,
9449) -> Result<(), SuperviseError> {
9450    let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
9451        module_id: Some(module_id.to_string()),
9452    })?;
9453    state.spawn_generation = spawn_events.emit_spawned(module_id, child.pid, child.spawned_at_ms);
9454    if std::mem::take(&mut state.coalesced_restart_pending) {
9455        let generation = state.spawn_generation;
9456        info!(module_id, "restart already satisfied by generation {generation}; coalesced pending request completed");
9457    }
9458    state.drain_disposition_detail = None;
9459    state.spawn_failure = None;
9460    // Every caller of this is a plain spawn, which always uses the primary key;
9461    // a promoted swap candidate sets the flag itself after this returns.
9462    state.in_alternate_slot = false;
9463    state.configuration_updated_since_spawn = false;
9464    state.spawned_protocol = Some(child.protocol);
9465    state.state = ModuleState::Running;
9466    state.enabled = true;
9467    state.process_alive = true;
9468    state.pid = child.id();
9469    #[cfg(target_os = "macos")]
9470    {
9471        state.report_ready = Some(Arc::clone(&child.report_ready));
9472    }
9473    state.spawned_at_ms = Some(child.spawned_at_ms);
9474    state.spawned_from = Some(child.spawned_from.clone());
9475    state.spawned_file_identity = child.spawned_file_identity;
9476    state.process_start_time = child.process_start_time;
9477    Ok(())
9478}
9479
9480fn clear_current_process_facts(state: &mut SupervisorSnapshot) {
9481    state.process_alive = false;
9482    state.spawned_protocol = None;
9483    state.pid = None;
9484    #[cfg(target_os = "macos")]
9485    {
9486        state.report_ready = None;
9487    }
9488    state.spawned_at_ms = None;
9489    state.spawned_from = None;
9490    state.spawned_file_identity = None;
9491    state.process_start_time = None;
9492    state.deliberate_severance = None;
9493}
9494
9495#[cfg(test)]
9496fn record_deliberate_severance(
9497    snapshot: &SharedSnapshot,
9498    identity: ProcessIdentity,
9499) -> Result<(), SuperviseError> {
9500    update_snapshot(snapshot, None, |state| {
9501        state.deliberate_severance = Some(identity);
9502    })
9503}
9504
9505fn apply_deliberate_severance_marker(
9506    snapshot: &SharedSnapshot,
9507    exited_identity: Option<ProcessIdentity>,
9508    mut exit_report: ExitReport,
9509) -> ExitReport {
9510    let marker = lock_snapshot(snapshot)
9511        .ok()
9512        .and_then(|mut state| state.deliberate_severance.take());
9513    if marker.is_some() && marker == exited_identity {
9514        exit_report.kind = ExitKind::DeliberateSeverance;
9515    }
9516    exit_report
9517}
9518
9519fn classify_reaped_child_exit(
9520    snapshot: &SharedSnapshot,
9521    child: &SupervisedChild,
9522    status: &ExitStatus,
9523) -> ExitReport {
9524    let _ = update_snapshot(snapshot, None, |state| {
9525        state.reaped_pid = Some(child.pid);
9526        state.spawn_failure = child.spawn_failure.clone();
9527    });
9528    apply_deliberate_severance_marker(snapshot, child.process_identity(), classify_exit(status))
9529}
9530
9531fn fail_snapshot(
9532    snapshot: &SharedSnapshot,
9533    module_id: Option<&str>,
9534    last_exit: Option<ExitReport>,
9535) {
9536    if let Err(err) = update_snapshot(snapshot, module_id, |state| {
9537        state.state = ModuleState::Failed;
9538        clear_current_process_facts(state);
9539        if let Some(last_exit) = last_exit {
9540            state.last_exit = Some(last_exit);
9541        }
9542    }) {
9543        error!(error = %err, "failed to mark supervisor state failed");
9544    }
9545}
9546
9547fn update_snapshot(
9548    snapshot: &SharedSnapshot,
9549    module_id: Option<&str>,
9550    update: impl FnOnce(&mut SupervisorSnapshot),
9551) -> Result<(), SuperviseError> {
9552    let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
9553        module_id: module_id.map(ToOwned::to_owned),
9554    })?;
9555    update(&mut state);
9556    Ok(())
9557}
9558
9559const SLOW_SNAPSHOT_LOCK_THRESHOLD: Duration = Duration::from_millis(250);
9560
9561fn lock_snapshot_for_control<'a>(
9562    snapshot: &'a SharedSnapshot,
9563    module_id: &str,
9564    caller: &'static str,
9565) -> Result<std::sync::MutexGuard<'a, SupervisorSnapshot>, SuperviseError> {
9566    let started_at = Instant::now();
9567    let guard = lock_snapshot(snapshot)?;
9568    let waited = started_at.elapsed();
9569    if waited >= SLOW_SNAPSHOT_LOCK_THRESHOLD {
9570        warn!(
9571            module_id = %module_id,
9572            waited_ms = waited.as_millis() as u64,
9573            caller = %caller,
9574            "slow snapshot lock"
9575        );
9576    }
9577    Ok(guard)
9578}
9579
9580fn lock_snapshot(
9581    snapshot: &SharedSnapshot,
9582) -> Result<std::sync::MutexGuard<'_, SupervisorSnapshot>, SuperviseError> {
9583    snapshot
9584        .lock()
9585        .map_err(|_| SuperviseError::StatePoisoned { module_id: None })
9586}
9587
9588#[cfg(test)]
9589mod terminal_history_tests {
9590    use std::{
9591        path::PathBuf,
9592        sync::Arc,
9593        time::{Duration, Instant},
9594    };
9595
9596    use tokio::time::sleep;
9597
9598    use super::{
9599        apply_deliberate_severance_marker, daemon_will_restart, drain_child_to_state,
9600        drained_after_quiescence_wait, handle_reload_spawn_failure, health_restart_child,
9601        lock_snapshot, on_child_exit, record_deliberate_severance, record_wait_error_terminal,
9602        reset_restart_count, spawn_and_mark_running, update_snapshot, wait_error_exit_report,
9603        ExitKind, ExitReport, ModuleProtocol, ModuleSpec, ModuleState, NextAction, ProcessIdentity,
9604        RestartPolicy, SpawnEventKind, StopNotice, SuperviseError, SupervisedModule, Supervisor,
9605        SupervisorHandle, SupervisorHealthStatus, SupervisorSnapshot,
9606    };
9607    // The supervisor's clock, distinct from the `std::time::Instant` these tests
9608    // use for their own wall-clock deadlines: crash-restart instants must be on
9609    // the same clock the production code stamps them with, which is tokio's (and
9610    // is what `start_paused` tests can move).
9611    use super::Instant as ClockInstant;
9612    use crate::{
9613        registry::Registry,
9614        terminal_ring::{TerminalRing, TerminalRingConfig},
9615    };
9616    use std::sync::Mutex;
9617    use subc_control::TerminalDisposition;
9618
9619    /// See the twin in `control.rs` for why this derives the path from
9620    /// `current_exe()` and why the existence check is here: `--lib` alone does
9621    /// not build `[[bin]]` targets, and a bare spawn then fails with a raw
9622    /// `NotFound` that reads as a broken test rather than an unbuilt dependency.
9623    pub(super) fn fake_aft_stub_path() -> PathBuf {
9624        let mut path = std::env::current_exe().expect("current_exe available in tests");
9625        path.pop();
9626        path.pop();
9627        path.push(if cfg!(windows) {
9628            "fake-aft-stub.exe"
9629        } else {
9630            "fake-aft-stub"
9631        });
9632        assert!(
9633            path.exists(),
9634            "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
9635             [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
9636            path.display()
9637        );
9638        path
9639    }
9640
9641    #[test]
9642    fn reserved_never_spawned_refuses_every_hello() {
9643        // The canary hole: a reserved id whose module has never spawned had NO
9644        // gate entry and admitted anyone -- the reservation protected the nonce
9645        // holder, not the NAME. Now the entry is present with no legitimate
9646        // holder and refuses all comers.
9647        let supervisor = SupervisorHandle::default();
9648        supervisor.apply_identity_configuration(&ModuleSpec {
9649            module_id: "never-spawned".to_string(),
9650            program: PathBuf::from("/usr/bin/false"),
9651            args: Vec::new(),
9652            env: Vec::new(),
9653            reserved: true,
9654            reserved_prefixes: Vec::new(),
9655            protocol: ModuleProtocol::Subc,
9656            overlap: Default::default(),
9657        });
9658        assert!(
9659            supervisor
9660                .reserved_hello_rejection("never-spawned", Some("any-forged-nonce"))
9661                .is_some(),
9662            "forged nonce must refuse on a reserved never-spawned id"
9663        );
9664        assert!(
9665            supervisor
9666                .reserved_hello_rejection("never-spawned", None)
9667                .is_some(),
9668            "absent nonce must refuse on a reserved never-spawned id"
9669        );
9670        // And a real spawn nonce minted later admits exactly that nonce.
9671        supervisor.set_spawn_nonce("never-spawned", "minted".to_string());
9672        supervisor.apply_identity_configuration(&ModuleSpec {
9673            module_id: "never-spawned".to_string(),
9674            program: PathBuf::from("/usr/bin/false"),
9675            args: Vec::new(),
9676            env: Vec::new(),
9677            reserved: true,
9678            reserved_prefixes: Vec::new(),
9679            protocol: ModuleProtocol::Subc,
9680            overlap: Default::default(),
9681        });
9682        assert!(supervisor
9683            .reserved_hello_rejection("never-spawned", Some("minted"))
9684            .is_none());
9685        assert!(supervisor
9686            .reserved_hello_rejection("never-spawned", Some("forged"))
9687            .is_some());
9688    }
9689
9690    /// Put `count` crash restarts on a snapshot's ring as if they had all just
9691    /// happened, which is what "spent budget" looks like to every reader.
9692    fn seed_crash_restarts(state: &mut SupervisorSnapshot, count: u32) {
9693        let now = ClockInstant::now();
9694        for _ in 0..count {
9695            state.crash_restarts.push_back(now);
9696        }
9697    }
9698
9699    /// Age the oldest recorded restart out of `window`, standing in for the hours
9700    /// that would otherwise have to pass. Injecting the instant is the point: a
9701    /// test that slept a real window would take ten minutes and still prove less.
9702    fn age_oldest_crash_restart_out_of_window(state: &mut SupervisorSnapshot, window: Duration) {
9703        let aged = state
9704            .crash_restarts
9705            .front()
9706            .expect("a crash restart must be recorded before it can be aged")
9707            .checked_sub(window + Duration::from_secs(1))
9708            .expect("the test clock is far enough from its origin to age an instant");
9709        state.crash_restarts[0] = aged;
9710    }
9711
9712    fn snapshot_with_restarts(enabled: bool, count: u32) -> SupervisorSnapshot {
9713        let mut state = SupervisorSnapshot::new(ModuleState::Running, enabled);
9714        seed_crash_restarts(&mut state, count);
9715        state
9716    }
9717
9718    #[test]
9719    fn daemon_owned_recovery_predicate_uses_the_pre_increment_budget() {
9720        let policy = RestartPolicy::new(3, Duration::ZERO);
9721        let now = ClockInstant::now();
9722        assert!(daemon_will_restart(
9723            &mut snapshot_with_restarts(true, 2),
9724            &policy,
9725            now
9726        ));
9727        assert!(!daemon_will_restart(
9728            &mut snapshot_with_restarts(true, 3),
9729            &policy,
9730            now
9731        ));
9732        assert!(!daemon_will_restart(
9733            &mut snapshot_with_restarts(false, 0),
9734            &policy,
9735            now
9736        ));
9737    }
9738
9739    #[test]
9740    fn crash_restart_backoff_escalates_with_in_window_count() {
9741        let policy = RestartPolicy::new(4, Duration::from_millis(100))
9742            .with_max_backoff(Duration::from_secs(30));
9743        let now = ClockInstant::now();
9744        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
9745        let schedules = (0..4)
9746            .map(|_| {
9747                state
9748                    .next_crash_restart(&policy, now)
9749                    .expect("the test policy allows four crash restarts")
9750            })
9751            .collect::<Vec<_>>();
9752
9753        assert_eq!(
9754            schedules
9755                .iter()
9756                .map(|schedule| schedule.restart_in_window)
9757                .collect::<Vec<_>>(),
9758            vec![0, 1, 2, 3]
9759        );
9760        assert_eq!(
9761            schedules
9762                .iter()
9763                .map(|schedule| schedule.delay)
9764                .collect::<Vec<_>>(),
9765            vec![
9766                Duration::from_millis(100),
9767                Duration::from_secs(1),
9768                Duration::from_secs(10),
9769                Duration::from_secs(30),
9770            ]
9771        );
9772    }
9773
9774    #[test]
9775    fn crash_restart_backoff_resets_after_ring_clear() {
9776        let policy = RestartPolicy::new(3, Duration::from_millis(100));
9777        let now = ClockInstant::now();
9778        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
9779        assert_eq!(
9780            state.next_crash_restart(&policy, now).unwrap().delay,
9781            Duration::from_millis(100)
9782        );
9783        assert_eq!(
9784            state.next_crash_restart(&policy, now).unwrap().delay,
9785            Duration::from_secs(1)
9786        );
9787
9788        state.clear_crash_restarts();
9789        let schedule = state
9790            .next_crash_restart(&policy, now)
9791            .expect("a cleared ring must allow another restart");
9792        assert_eq!(schedule.restart_in_window, 0);
9793        assert_eq!(schedule.delay, Duration::from_millis(100));
9794    }
9795
9796    #[test]
9797    fn crash_restart_backoff_ignores_aged_restarts() {
9798        let policy = RestartPolicy::new(3, Duration::from_millis(100));
9799        let now = ClockInstant::now();
9800        let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
9801        state
9802            .next_crash_restart(&policy, now)
9803            .expect("the first restart is allowed");
9804        state
9805            .next_crash_restart(&policy, now)
9806            .expect("the second restart is allowed");
9807        state.crash_restarts[0] = now
9808            .checked_sub(policy.window + Duration::from_secs(1))
9809            .expect("the fake clock can age a restart past the window");
9810
9811        let schedule = state
9812            .next_crash_restart(&policy, now)
9813            .expect("an aged restart must release its slot");
9814        assert_eq!(schedule.restart_in_window, 1);
9815        assert_eq!(schedule.delay, Duration::from_secs(1));
9816        assert_eq!(state.crash_restarts.len(), 2);
9817    }
9818
9819    /// The budget is a rate: the same three spent restarts refuse a respawn
9820    /// while they are recent and allow one once they have aged past the window.
9821    /// Nothing about the module changed in between, which is the whole point.
9822    #[test]
9823    fn a_budget_spent_before_the_window_no_longer_refuses() {
9824        let policy = RestartPolicy::new(3, Duration::ZERO);
9825        let mut state = snapshot_with_restarts(true, 3);
9826        let now = ClockInstant::now();
9827        assert!(!daemon_will_restart(&mut state, &policy, now));
9828
9829        assert!(daemon_will_restart(
9830            &mut state,
9831            &policy,
9832            now + policy.window + Duration::from_secs(1)
9833        ));
9834        assert!(
9835            state.crash_restarts.is_empty(),
9836            "reading the budget must drop the instants that left the window"
9837        );
9838    }
9839
9840    fn module_with_recovery_snapshot(
9841        state: ModuleState,
9842        enabled: bool,
9843        restart_count: u32,
9844    ) -> SupervisedModule {
9845        let registry = Arc::new(Registry::default());
9846        let supervisor =
9847            Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::new(3, Duration::ZERO));
9848        let module = supervisor
9849            .spawn(ModuleSpec {
9850                module_id: "recovery-snapshot".to_string(),
9851                program: fake_aft_stub_path(),
9852                args: Vec::new(),
9853                env: Vec::new(),
9854                reserved: false,
9855                reserved_prefixes: Vec::new(),
9856                protocol: ModuleProtocol::Subc,
9857                overlap: Default::default(),
9858            })
9859            .unwrap();
9860        update_snapshot(
9861            &module.inner.snapshot,
9862            Some("recovery-snapshot"),
9863            |snapshot| {
9864                snapshot.state = state;
9865                snapshot.enabled = enabled;
9866                seed_crash_restarts(snapshot, restart_count);
9867            },
9868        )
9869        .unwrap();
9870        module
9871    }
9872
9873    #[cfg(target_os = "linux")]
9874    #[tokio::test]
9875    async fn no_cgroup_placement_does_not_block_fake_aft_stub_spawn() {
9876        let supervisor =
9877            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
9878                .with_cgroup_placement(None);
9879        let result = supervisor.spawn(ModuleSpec {
9880            module_id: "no-cgroup-placement".to_string(),
9881            program: fake_aft_stub_path(),
9882            args: Vec::new(),
9883            env: Vec::new(),
9884            reserved: false,
9885            reserved_prefixes: Vec::new(),
9886            protocol: ModuleProtocol::Subc,
9887            overlap: Default::default(),
9888        });
9889
9890        assert!(
9891            result.is_ok(),
9892            "no delegation must not turn an otherwise valid spawn into a failure: {result:?}"
9893        );
9894    }
9895
9896    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9897    async fn undecided_snapshot_uses_shared_restart_predicate() {
9898        assert!(module_with_recovery_snapshot(ModuleState::Running, true, 2)
9899            .will_recover_after_connection_loss()
9900            .unwrap());
9901        assert!(
9902            !module_with_recovery_snapshot(ModuleState::Running, true, 3)
9903                .will_recover_after_connection_loss()
9904                .unwrap()
9905        );
9906    }
9907
9908    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9909    async fn restarting_snapshot_at_exhausted_budget_is_non_terminal() {
9910        assert!(
9911            module_with_recovery_snapshot(ModuleState::Restarting, true, 3)
9912                .will_recover_after_connection_loss()
9913                .unwrap()
9914        );
9915    }
9916
9917    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9918    async fn terminal_phase_snapshots_are_terminal_before_budget_exhaustion() {
9919        assert!(!module_with_recovery_snapshot(ModuleState::Failed, true, 0)
9920            .will_recover_after_connection_loss()
9921            .unwrap());
9922        assert!(
9923            !module_with_recovery_snapshot(ModuleState::Disabled, true, 0)
9924                .will_recover_after_connection_loss()
9925                .unwrap()
9926        );
9927    }
9928
9929    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9930    async fn warming_snapshot_is_limited_to_startup_phases() {
9931        for state in [
9932            ModuleState::Starting,
9933            ModuleState::Running,
9934            ModuleState::Restarting,
9935        ] {
9936            assert!(
9937                module_with_recovery_snapshot(state, true, 0)
9938                    .is_warming()
9939                    .unwrap(),
9940                "{state:?} should be warming"
9941            );
9942        }
9943        for state in [
9944            ModuleState::Unresponsive,
9945            ModuleState::Draining,
9946            ModuleState::Stopped,
9947            ModuleState::Failed,
9948            ModuleState::Disabled,
9949        ] {
9950            assert!(
9951                !module_with_recovery_snapshot(state, true, 0)
9952                    .is_warming()
9953                    .unwrap(),
9954                "{state:?} should not be warming"
9955            );
9956        }
9957    }
9958
9959    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9960    async fn terminal_history_survives_respawn_and_keeps_both_crashes_in_order() {
9961        let registry = Arc::new(Registry::default());
9962        let supervisor =
9963            Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::new(1, Duration::ZERO));
9964        let module = supervisor
9965            .spawn(ModuleSpec {
9966                module_id: "terminal-history".to_string(),
9967                program: fake_aft_stub_path(),
9968                args: Vec::new(),
9969                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
9970                reserved: false,
9971                reserved_prefixes: Vec::new(),
9972                protocol: ModuleProtocol::Subc,
9973                overlap: Default::default(),
9974            })
9975            .unwrap();
9976
9977        let deadline = Instant::now() + Duration::from_secs(5);
9978        loop {
9979            let history = module.terminal_history();
9980            if history.entries.len() == 2 {
9981                assert_eq!(module.status().unwrap().state, ModuleState::Failed);
9982                assert_eq!(history.dropped, 0);
9983                assert_eq!(
9984                    history
9985                        .entries
9986                        .iter()
9987                        .map(|entry| entry.exit_code)
9988                        .collect::<Vec<_>>(),
9989                    vec![Some(23), Some(23)]
9990                );
9991                assert!(history.entries[0].at_ms <= history.entries[1].at_ms);
9992                return;
9993            }
9994            assert!(
9995                Instant::now() < deadline,
9996                "module did not retain two terminal exits: {history:?}"
9997            );
9998            sleep(Duration::from_millis(10)).await;
9999        }
10000    }
10001
10002    /// A disable issued while a crash respawn is still backing off must preempt
10003    /// that respawn: the operator's stop wins, the disable must not queue behind
10004    /// the backoff, and the module must never come back up afterwards.
10005    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10006    async fn disable_during_crash_backoff_cancels_pending_respawn() {
10007        let backoff = Duration::from_secs(2);
10008        let supervisor = Supervisor::new_for_test(
10009            Arc::new(Registry::default()),
10010            RestartPolicy::new(10, backoff),
10011        );
10012        let module = supervisor
10013            .spawn(ModuleSpec {
10014                module_id: "disable-during-backoff".to_string(),
10015                program: fake_aft_stub_path(),
10016                args: Vec::new(),
10017                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
10018                reserved: false,
10019                reserved_prefixes: Vec::new(),
10020                protocol: ModuleProtocol::Subc,
10021                overlap: Default::default(),
10022            })
10023            .unwrap();
10024
10025        // Wait for the first crash to put the module into its backoff window.
10026        let deadline = Instant::now() + Duration::from_secs(5);
10027        loop {
10028            if module.status().unwrap().state == ModuleState::Restarting {
10029                break;
10030            }
10031            assert!(
10032                Instant::now() < deadline,
10033                "module never entered the crash backoff"
10034            );
10035            sleep(Duration::from_millis(10)).await;
10036        }
10037
10038        let started = Instant::now();
10039        module.set_enabled(false).await.unwrap();
10040        let waited = started.elapsed();
10041
10042        assert!(
10043            waited < backoff / 2,
10044            "disable waited {waited:?} behind the {backoff:?} crash backoff; the operator command must preempt the pending respawn"
10045        );
10046        assert_eq!(module.status().unwrap().state, ModuleState::Disabled);
10047
10048        // Outlast the backoff: the respawn it was counting down to must never run.
10049        sleep(backoff + Duration::from_millis(500)).await;
10050        let status = module.status().unwrap();
10051        assert_eq!(status.state, ModuleState::Disabled);
10052        assert_eq!(
10053            status.spawn_generation, 1,
10054            "module respawned after the operator disabled it"
10055        );
10056    }
10057
10058    /// A `protocol: "none"` child that stays parked and exits 0 on SIGTERM:
10059    /// the shape of nats-server, the program this rule exists for.
10060    #[cfg(unix)]
10061    fn protocol_none_sigterm_exits_clean_spec(
10062        module_id: &str,
10063        dir: &std::path::Path,
10064    ) -> (ModuleSpec, PathBuf, PathBuf) {
10065        let ready = dir.join("ready");
10066        let marker = dir.join("sigterm");
10067        let spec = ModuleSpec {
10068            module_id: module_id.to_string(),
10069            program: fake_aft_stub_path(),
10070            args: Vec::new(),
10071            env: vec![
10072                ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
10073                (
10074                    "FAKE_AFT_SIGTERM_MARKER_PATH".to_string(),
10075                    marker.display().to_string(),
10076                ),
10077                (
10078                    "FAKE_AFT_NEVER_CONNECT_READY_PATH".to_string(),
10079                    ready.display().to_string(),
10080                ),
10081            ],
10082            reserved: false,
10083            reserved_prefixes: Vec::new(),
10084            protocol: ModuleProtocol::None,
10085            overlap: Default::default(),
10086        };
10087        (spec, ready, marker)
10088    }
10089
10090    /// Wait for a file the child writes, so a signal is never sent before the
10091    /// child's SIGTERM handler is installed (the default disposition would
10092    /// kill it by signal and the exit would not be clean).
10093    #[cfg(unix)]
10094    async fn wait_for_file(path: &std::path::Path) {
10095        let deadline = Instant::now() + Duration::from_secs(10);
10096        while !path.exists() {
10097            assert!(
10098                Instant::now() < deadline,
10099                "{} never appeared",
10100                path.display()
10101            );
10102            sleep(Duration::from_millis(10)).await;
10103        }
10104    }
10105
10106    /// A protocol-none module that exits 0 because something OUTSIDE the
10107    /// supervisor sent it SIGTERM is respawned, and the terminal record carries
10108    /// the crash-path disposition rather than `stopped`.
10109    #[cfg(unix)]
10110    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10111    async fn protocol_none_unrequested_clean_exit_restarts_as_a_crash() {
10112        let dir = subc_test_support::TestTempDir::new("none-unrequested-clean-exit");
10113        let (spec, ready, marker) =
10114            protocol_none_sigterm_exits_clean_spec("none-unrequested-clean-exit", dir.path());
10115        let supervisor = Supervisor::new_for_test(
10116            Arc::new(Registry::default()),
10117            RestartPolicy::new(3, Duration::ZERO),
10118        );
10119        let module = supervisor.spawn(spec).unwrap();
10120        wait_for_file(&ready).await;
10121        let first_pid = module
10122            .status()
10123            .unwrap()
10124            .pid
10125            .expect("a running module reports its pid");
10126
10127        rustix::process::kill_process(
10128            rustix::process::Pid::from_raw(i32::try_from(first_pid).unwrap()).unwrap(),
10129            rustix::process::Signal::TERM,
10130        )
10131        .unwrap();
10132
10133        let deadline = Instant::now() + Duration::from_secs(10);
10134        let respawned = loop {
10135            let status = module.status().unwrap();
10136            if status.state == ModuleState::Running
10137                && status.pid.is_some_and(|pid| pid != first_pid)
10138            {
10139                break status;
10140            }
10141            assert!(
10142                Instant::now() < deadline,
10143                "protocol-none module was not respawned after an unrequested clean exit: {status:?}"
10144            );
10145            sleep(Duration::from_millis(10)).await;
10146        };
10147        assert_eq!(respawned.spawn_generation, 2);
10148        assert!(
10149            marker.exists(),
10150            "the child must have exited through its SIGTERM handler (exit 0), or this proves nothing about clean exits"
10151        );
10152
10153        let history = module.terminal_history();
10154        assert_eq!(history.entries.len(), 1, "{history:?}");
10155        let entry = &history.entries[0];
10156        assert_eq!(entry.exit_code, Some(0));
10157        assert_eq!(entry.exit_kind, subc_control::TerminalExitKind::Clean);
10158        assert_eq!(entry.disposition, TerminalDisposition::Restarting);
10159
10160        module.stop().await.unwrap();
10161    }
10162
10163    /// Repeated unrequested clean exits of a protocol-none module spend the
10164    /// restart budget exactly as crashes do, and the module ends `failed` with
10165    /// the budget named.
10166    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10167    async fn protocol_none_repeated_clean_exits_exhaust_the_restart_budget() {
10168        let supervisor = Supervisor::new_for_test(
10169            Arc::new(Registry::default()),
10170            RestartPolicy::new(1, Duration::ZERO),
10171        );
10172        let module = supervisor
10173            .spawn(ModuleSpec {
10174                module_id: "none-clean-exit-budget".to_string(),
10175                program: fake_aft_stub_path(),
10176                args: Vec::new(),
10177                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
10178                reserved: false,
10179                reserved_prefixes: Vec::new(),
10180                protocol: ModuleProtocol::None,
10181                overlap: Default::default(),
10182            })
10183            .unwrap();
10184
10185        let deadline = Instant::now() + Duration::from_secs(10);
10186        loop {
10187            let status = module.status().unwrap();
10188            if status.state == ModuleState::Failed {
10189                break;
10190            }
10191            assert!(
10192                Instant::now() < deadline,
10193                "module never exhausted its budget: {status:?} {:?}",
10194                module.terminal_history()
10195            );
10196            sleep(Duration::from_millis(10)).await;
10197        }
10198        let history = module.terminal_history();
10199        assert_eq!(
10200            history
10201                .entries
10202                .iter()
10203                .map(|entry| (entry.exit_code, entry.disposition.clone()))
10204                .collect::<Vec<_>>(),
10205            vec![
10206                (Some(0), TerminalDisposition::Restarting),
10207                (Some(0), TerminalDisposition::Failed),
10208            ]
10209        );
10210        let detail = history.entries[1]
10211            .disposition_detail
10212            .as_deref()
10213            .expect("a budget failure names the budget");
10214        assert!(detail.contains("max_restarts=1"), "{detail}");
10215        assert_eq!(module.status().unwrap().spawn_generation, 2);
10216    }
10217
10218    /// A stop the supervisor itself requests still stops a protocol-none
10219    /// module, even though the child answers the SIGTERM with exit 0.
10220    #[cfg(unix)]
10221    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10222    async fn protocol_none_requested_stop_and_disable_do_not_respawn() {
10223        for disable in [false, true] {
10224            let label = if disable {
10225                "none-requested-disable"
10226            } else {
10227                "none-requested-stop"
10228            };
10229            let dir = subc_test_support::TestTempDir::new(label);
10230            let (spec, ready, marker) = protocol_none_sigterm_exits_clean_spec(label, dir.path());
10231            let supervisor = Supervisor::new_for_test(
10232                Arc::new(Registry::default()),
10233                RestartPolicy::new(3, Duration::ZERO),
10234            );
10235            let module = supervisor.spawn(spec).unwrap();
10236            wait_for_file(&ready).await;
10237
10238            if disable {
10239                module.set_enabled(false).await.unwrap();
10240            } else {
10241                module.stop().await.unwrap();
10242            }
10243            assert!(
10244                marker.exists(),
10245                "{label}: the child must have left through its SIGTERM handler with exit 0"
10246            );
10247
10248            // Long enough for a zero-backoff respawn to have happened if the
10249            // exit had been treated as a crash.
10250            sleep(Duration::from_millis(500)).await;
10251            let status = module.status().unwrap();
10252            let expected = if disable {
10253                ModuleState::Disabled
10254            } else {
10255                ModuleState::Stopped
10256            };
10257            assert_eq!(status.state, expected, "{label}");
10258            assert_eq!(
10259                status.spawn_generation, 1,
10260                "{label}: respawned after a requested stop"
10261            );
10262            let history = module.terminal_history();
10263            assert_eq!(history.entries.len(), 1, "{label}: {history:?}");
10264            assert_eq!(history.entries[0].exit_code, Some(0), "{label}");
10265            assert_ne!(
10266                history.entries[0].disposition,
10267                TerminalDisposition::Restarting,
10268                "{label}"
10269            );
10270        }
10271    }
10272
10273    /// A subc-wire module that exits 0 on its own is still a stop: the
10274    /// protocol-none rule must not reach it.
10275    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10276    async fn subc_wire_clean_exit_is_still_a_stop() {
10277        let supervisor = Supervisor::new_for_test(
10278            Arc::new(Registry::default()),
10279            RestartPolicy::new(3, Duration::ZERO),
10280        );
10281        let module = supervisor
10282            .spawn(ModuleSpec {
10283                module_id: "wire-clean-exit".to_string(),
10284                program: fake_aft_stub_path(),
10285                args: Vec::new(),
10286                env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
10287                reserved: false,
10288                reserved_prefixes: Vec::new(),
10289                protocol: ModuleProtocol::Subc,
10290                overlap: Default::default(),
10291            })
10292            .unwrap();
10293
10294        let deadline = Instant::now() + Duration::from_secs(10);
10295        while module.terminal_history().entries.is_empty() {
10296            assert!(Instant::now() < deadline, "module never exited");
10297            sleep(Duration::from_millis(10)).await;
10298        }
10299        // Long enough for a zero-backoff respawn to have happened.
10300        sleep(Duration::from_millis(500)).await;
10301        let status = module.status().unwrap();
10302        assert_eq!(status.state, ModuleState::Stopped);
10303        assert_eq!(status.spawn_generation, 1);
10304        let history = module.terminal_history();
10305        assert_eq!(history.entries.len(), 1, "{history:?}");
10306        assert_eq!(history.entries[0].exit_code, Some(0));
10307        assert_eq!(history.entries[0].disposition, TerminalDisposition::Stopped);
10308    }
10309
10310    #[cfg(unix)]
10311    #[tokio::test]
10312    async fn reload_exit_keeps_roster_until_terminal_is_recorded_during_shutdown() {
10313        let dir = subc_test_support::TestTempDir::new("reload-roster-terminal-order");
10314        let record = dir.join("live-children.json");
10315        let supervisor = Supervisor::new_for_test(
10316            Arc::new(Registry::default()),
10317            RestartPolicy::new(0, Duration::ZERO),
10318        );
10319        let mut runtime = supervisor.runtime_config();
10320        runtime.child_roster.record_to(record.clone());
10321        let gate = Arc::new(super::ReloadExitRecordGate::default());
10322        runtime.test_reload_exit_record_gate = Some(Arc::clone(&gate));
10323        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10324        let spec = ModuleSpec {
10325            module_id: "reload-exit-roster".into(),
10326            program: fake_aft_stub_path(),
10327            args: Vec::new(),
10328            env: vec![("FAKE_AFT_EXIT_CODE".into(), "23".into())],
10329            reserved: false,
10330            reserved_prefixes: Vec::new(),
10331            protocol: ModuleProtocol::Subc,
10332            overlap: Default::default(),
10333        };
10334        let mut child = None;
10335        let reload = super::finish_reload_child(
10336            &spec,
10337            &runtime,
10338            &supervisor.registry,
10339            &supervisor.process_liveness,
10340            &snapshot,
10341            &mut child,
10342        );
10343        tokio::pin!(reload);
10344        tokio::select! {
10345            result = &mut reload => panic!("reload missed the before-record gate: {result:?}"),
10346            _ = gate.reached.notified() => {}
10347        }
10348        assert!(runtime
10349            .terminal_ring
10350            .lock()
10351            .unwrap()
10352            .snapshot()
10353            .entries
10354            .is_empty());
10355        assert_eq!(
10356            crate::live_children::read_record(&record).unwrap().len(),
10357            1,
10358            "shutdown must still wait for the reaped child until its terminal record exists"
10359        );
10360        runtime.child_roster.close();
10361        gate.resume.notify_one();
10362        assert!(reload.await.is_err());
10363        assert!(crate::live_children::read_record(&record)
10364            .unwrap()
10365            .is_empty());
10366        let history = runtime.terminal_ring.lock().unwrap().snapshot();
10367        assert_eq!(history.entries.len(), 1);
10368        assert_eq!(
10369            history.entries[0].disposition,
10370            TerminalDisposition::DaemonShutdown
10371        );
10372    }
10373
10374    /// Each restart-producing arm has its own state transition. Keeping their
10375    /// lifetime count assertions adjacent prevents a later new arm from silently
10376    /// spending budget without recording the historical restart.
10377    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10378    async fn every_restart_increment_path_advances_lifetime_count() {
10379        let supervisor = Supervisor::new_for_test(
10380            Arc::new(Registry::default()),
10381            RestartPolicy::new(1, Duration::ZERO),
10382        );
10383        let runtime = supervisor.runtime_config();
10384        let spec = ModuleSpec {
10385            module_id: "lifetime-increment-path".to_string(),
10386            program: PathBuf::from("/unused/lifetime-increment-path"),
10387            args: Vec::new(),
10388            env: Vec::new(),
10389            reserved: false,
10390            reserved_prefixes: Vec::new(),
10391            protocol: ModuleProtocol::Subc,
10392            overlap: Default::default(),
10393        };
10394
10395        let crash_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10396        assert!(matches!(
10397            on_child_exit(
10398                &spec,
10399                runtime.restart_policy,
10400                &supervisor.registry,
10401                &crash_snapshot,
10402                &runtime.terminal_ring,
10403                &runtime.spawn_events,
10404                &runtime.child_roster,
10405                ExitReport {
10406                    kind: ExitKind::Crash,
10407                    code: Some(1),
10408                    signal: None,
10409                    at_ms: 1,
10410                },
10411            )
10412            .await,
10413            NextAction::Restart { schedule: _ }
10414        ));
10415        let (crash_restarts, crash_lifetime) = {
10416            let state = lock_snapshot(&crash_snapshot).unwrap();
10417            (state.crash_restarts.len(), state.lifetime_restarts)
10418        };
10419        assert_eq!(crash_restarts, 1);
10420        assert_eq!(crash_lifetime, 1);
10421
10422        let health_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10423        let mut health_child = None;
10424        assert!(matches!(
10425            health_restart_child(
10426                &spec,
10427                &runtime,
10428                &supervisor.registry,
10429                &supervisor.process_liveness,
10430                &health_snapshot,
10431                &mut health_child,
10432                SupervisorHealthStatus::Failing,
10433                None,
10434                2,
10435            )
10436            .await,
10437            Ok(())
10438        ));
10439        assert!(health_child.is_none());
10440        assert!(lock_snapshot(&health_snapshot).unwrap().respawn_pending);
10441        assert!(runtime.scheduled_respawn.lock().unwrap().take().is_some());
10442        let (health_restarts, health_lifetime) = {
10443            let state = lock_snapshot(&health_snapshot).unwrap();
10444            (state.crash_restarts.len(), state.lifetime_restarts)
10445        };
10446        assert_eq!(health_restarts, 1);
10447        assert_eq!(health_lifetime, 1);
10448
10449        let reload_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10450        let mut reload_child = None;
10451        assert!(matches!(
10452            handle_reload_spawn_failure(
10453                &spec,
10454                &runtime,
10455                &supervisor.process_liveness,
10456                &reload_snapshot,
10457                &mut reload_child,
10458                "forced reload spawn failure".to_string(),
10459            )
10460            .await,
10461            Err(SuperviseError::ReloadFailed { .. })
10462        ));
10463        let (reload_restarts, reload_lifetime) = {
10464            let state = lock_snapshot(&reload_snapshot).unwrap();
10465            (state.crash_restarts.len(), state.lifetime_restarts)
10466        };
10467        assert_eq!(reload_restarts, 1);
10468        assert_eq!(reload_lifetime, 1);
10469    }
10470
10471    #[tokio::test]
10472    async fn deliberately_severed_live_child_records_lifetime_without_spending_restart_budget() {
10473        let supervisor = Supervisor::new_for_test(
10474            Arc::new(Registry::default()),
10475            RestartPolicy::new(3, Duration::ZERO),
10476        );
10477        let runtime = supervisor.runtime_config();
10478        let spec = ModuleSpec {
10479            module_id: "deliberately-severed".to_string(),
10480            program: PathBuf::from("/unused/deliberately-severed"),
10481            args: Vec::new(),
10482            env: Vec::new(),
10483            reserved: false,
10484            reserved_prefixes: Vec::new(),
10485            protocol: ModuleProtocol::Subc,
10486            overlap: Default::default(),
10487        };
10488        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10489        let process = ProcessIdentity {
10490            pid: 41,
10491            start_time: 101,
10492        };
10493        record_deliberate_severance(&snapshot, process).unwrap();
10494        let exit_report = apply_deliberate_severance_marker(
10495            &snapshot,
10496            Some(process),
10497            ExitReport {
10498                kind: ExitKind::Crash,
10499                code: Some(1),
10500                signal: None,
10501                at_ms: 1,
10502            },
10503        );
10504        assert_eq!(exit_report.kind, ExitKind::DeliberateSeverance);
10505
10506        assert!(matches!(
10507            on_child_exit(
10508                &spec,
10509                runtime.restart_policy,
10510                &supervisor.registry,
10511                &snapshot,
10512                &runtime.terminal_ring,
10513                &runtime.spawn_events,
10514                &runtime.child_roster,
10515                exit_report,
10516            )
10517            .await,
10518            NextAction::Restart { schedule: _ }
10519        ));
10520        let state = lock_snapshot(&snapshot).unwrap();
10521        assert_eq!(state.lifetime_restarts, 1);
10522        assert_eq!(state.crash_restarts.len(), 0);
10523    }
10524
10525    #[tokio::test]
10526    async fn genuine_crash_spends_restart_budget_and_records_lifetime() {
10527        let supervisor = Supervisor::new_for_test(
10528            Arc::new(Registry::default()),
10529            RestartPolicy::new(3, Duration::ZERO),
10530        );
10531        let runtime = supervisor.runtime_config();
10532        let spec = ModuleSpec {
10533            module_id: "genuine-crash".to_string(),
10534            program: PathBuf::from("/unused/genuine-crash"),
10535            args: Vec::new(),
10536            env: Vec::new(),
10537            reserved: false,
10538            reserved_prefixes: Vec::new(),
10539            protocol: ModuleProtocol::Subc,
10540            overlap: Default::default(),
10541        };
10542        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10543
10544        assert!(matches!(
10545            on_child_exit(
10546                &spec,
10547                runtime.restart_policy,
10548                &supervisor.registry,
10549                &snapshot,
10550                &runtime.terminal_ring,
10551                &runtime.spawn_events,
10552                &runtime.child_roster,
10553                ExitReport {
10554                    kind: ExitKind::Crash,
10555                    code: Some(1),
10556                    signal: None,
10557                    at_ms: 1,
10558                },
10559            )
10560            .await,
10561            NextAction::Restart { schedule: _ }
10562        ));
10563        let state = lock_snapshot(&snapshot).unwrap();
10564        assert_eq!(state.lifetime_restarts, 1);
10565        assert_eq!(state.crash_restarts.len(), 1);
10566    }
10567
10568    fn crash_exit_report(at_ms: u64) -> ExitReport {
10569        ExitReport {
10570            kind: ExitKind::Crash,
10571            code: Some(1),
10572            signal: None,
10573            at_ms,
10574        }
10575    }
10576
10577    fn windowed_crash_spec(module_id: &str) -> ModuleSpec {
10578        ModuleSpec {
10579            module_id: module_id.to_string(),
10580            program: PathBuf::from("/unused").join(module_id),
10581            args: Vec::new(),
10582            env: Vec::new(),
10583            reserved: false,
10584            reserved_prefixes: Vec::new(),
10585            protocol: ModuleProtocol::Subc,
10586            overlap: Default::default(),
10587        }
10588    }
10589
10590    /// A real crash loop still stops. Three crashes with nothing aging out spend
10591    /// a budget of two and the third respawn is refused, and both surfaces an
10592    /// operator has -- the log line and the retained terminal record -- name the
10593    /// window rather than only the cap, because `max_restarts=2` alone is what
10594    /// this budget used to mean.
10595    #[tokio::test]
10596    async fn three_crashes_inside_the_window_stop_the_module_and_name_the_window() {
10597        let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::ERROR);
10598        let supervisor = Supervisor::new_for_test(
10599            Arc::new(Registry::default()),
10600            RestartPolicy::new(2, Duration::ZERO),
10601        );
10602        let runtime = supervisor.runtime_config();
10603        let spec = windowed_crash_spec("crash-loop-in-window");
10604        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10605
10606        for attempt in 1..=2 {
10607            assert!(
10608                matches!(
10609                    on_child_exit(
10610                        &spec,
10611                        runtime.restart_policy,
10612                        &supervisor.registry,
10613                        &snapshot,
10614                        &runtime.terminal_ring,
10615                        &runtime.spawn_events,
10616                        &runtime.child_roster,
10617                        crash_exit_report(attempt),
10618                    )
10619                    .await,
10620                    NextAction::Restart { schedule: _ }
10621                ),
10622                "crash {attempt} is inside the budget and must respawn"
10623            );
10624        }
10625
10626        assert!(matches!(
10627            on_child_exit(
10628                &spec,
10629                runtime.restart_policy,
10630                &supervisor.registry,
10631                &snapshot,
10632                &runtime.terminal_ring,
10633                &runtime.spawn_events,
10634                &runtime.child_roster,
10635                crash_exit_report(3),
10636            )
10637            .await,
10638            NextAction::Stop { .. }
10639        ));
10640
10641        {
10642            let state = lock_snapshot(&snapshot).unwrap();
10643            assert_eq!(state.state, ModuleState::Failed);
10644            assert_eq!(state.crash_restarts.len(), 2);
10645            assert_eq!(state.lifetime_restarts, 2);
10646        }
10647
10648        let history = runtime
10649            .terminal_ring
10650            .lock()
10651            .expect("terminal ring is not poisoned")
10652            .snapshot();
10653        let last = history
10654            .entries
10655            .last()
10656            .expect("the refused crash is retained");
10657        assert_eq!(last.disposition, TerminalDisposition::Failed);
10658        assert_eq!(
10659            last.disposition_detail.as_deref(),
10660            Some("crash budget exhausted: max_restarts=2 within window_secs=600")
10661        );
10662
10663        let captured = crate::router::test_log::captured_logs(&logs);
10664        assert!(
10665            captured.contains("crash budget exhausted: max_restarts=2 within window_secs=600"),
10666            "the stop must be logged with its window: {captured}"
10667        );
10668    }
10669
10670    /// The rate, stated as a test: three crashes where the first has aged past
10671    /// the window are two crashes as far as the budget is concerned, so the
10672    /// third respawn is allowed and the ring holds only the two recent ones.
10673    ///
10674    /// This is the case a lifetime counter got wrong -- and the case the daemon
10675    /// now hits routinely, since a module exits non-zero every time its
10676    /// connection to the daemon drops.
10677    #[tokio::test]
10678    async fn a_crash_older_than_the_window_frees_its_slot_for_a_later_crash() {
10679        let supervisor = Supervisor::new_for_test(
10680            Arc::new(Registry::default()),
10681            RestartPolicy::new(2, Duration::ZERO),
10682        );
10683        let runtime = supervisor.runtime_config();
10684        let spec = windowed_crash_spec("crash-across-windows");
10685        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10686
10687        for attempt in 1..=2 {
10688            assert!(matches!(
10689                on_child_exit(
10690                    &spec,
10691                    runtime.restart_policy,
10692                    &supervisor.registry,
10693                    &snapshot,
10694                    &runtime.terminal_ring,
10695                    &runtime.spawn_events,
10696                    &runtime.child_roster,
10697                    crash_exit_report(attempt),
10698                )
10699                .await,
10700                NextAction::Restart { schedule: _ }
10701            ));
10702        }
10703
10704        // The oldest crash moves out of the window; nothing else about the
10705        // module changes.
10706        update_snapshot(&snapshot, Some(&spec.module_id), |state| {
10707            age_oldest_crash_restart_out_of_window(state, runtime.restart_policy.window);
10708        })
10709        .unwrap();
10710
10711        assert!(
10712            matches!(
10713                on_child_exit(
10714                    &spec,
10715                    runtime.restart_policy,
10716                    &supervisor.registry,
10717                    &snapshot,
10718                    &runtime.terminal_ring,
10719                    &runtime.spawn_events,
10720                    &runtime.child_roster,
10721                    crash_exit_report(3),
10722                )
10723                .await,
10724                NextAction::Restart { schedule: _ }
10725            ),
10726            "a crash older than the window must not hold a budget slot"
10727        );
10728
10729        let state = lock_snapshot(&snapshot).unwrap();
10730        assert_eq!(state.state, ModuleState::Restarting);
10731        assert_eq!(
10732            state.crash_restarts.len(),
10733            2,
10734            "the aged instant is dropped and the new one takes its place"
10735        );
10736        assert_eq!(
10737            state.lifetime_restarts, 3,
10738            "the ledger counts every restart, including the ones the window forgot"
10739        );
10740    }
10741
10742    /// An operator restart hands the budget back whole, and the ledger keeps
10743    /// counting. Those are different questions -- "how close is this module to
10744    /// being stopped" and "how many times has it been replaced" -- and the
10745    /// operator action answers only the first.
10746    #[tokio::test]
10747    async fn an_operator_restart_clears_the_ring_and_leaves_the_ledger_alone() {
10748        let supervisor = Supervisor::new_for_test(
10749            Arc::new(Registry::default()),
10750            RestartPolicy::new(2, Duration::ZERO),
10751        );
10752        let runtime = supervisor.runtime_config();
10753        let spec = windowed_crash_spec("operator-cleared-budget");
10754        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10755
10756        for attempt in 1..=2 {
10757            assert!(matches!(
10758                on_child_exit(
10759                    &spec,
10760                    runtime.restart_policy,
10761                    &supervisor.registry,
10762                    &snapshot,
10763                    &runtime.terminal_ring,
10764                    &runtime.spawn_events,
10765                    &runtime.child_roster,
10766                    crash_exit_report(attempt),
10767                )
10768                .await,
10769                NextAction::Restart { schedule: _ }
10770            ));
10771        }
10772
10773        reset_restart_count(&snapshot, &spec.module_id).unwrap();
10774        {
10775            let state = lock_snapshot(&snapshot).unwrap();
10776            assert!(
10777                state.crash_restarts.is_empty(),
10778                "an operator restart returns the full budget"
10779            );
10780            assert_eq!(
10781                state.lifetime_restarts, 2,
10782                "clearing the budget must not unmake the crashes"
10783            );
10784        }
10785
10786        assert!(
10787            matches!(
10788                on_child_exit(
10789                    &spec,
10790                    runtime.restart_policy,
10791                    &supervisor.registry,
10792                    &snapshot,
10793                    &runtime.terminal_ring,
10794                    &runtime.spawn_events,
10795                    &runtime.child_roster,
10796                    crash_exit_report(3),
10797                )
10798                .await,
10799                NextAction::Restart { schedule: _ }
10800            ),
10801            "the cleared budget must be spendable again"
10802        );
10803        let state = lock_snapshot(&snapshot).unwrap();
10804        assert_eq!(state.crash_restarts.len(), 1);
10805        assert_eq!(state.lifetime_restarts, 3);
10806    }
10807
10808    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10809    async fn severance_marker_for_a_dead_child_does_not_label_its_successor() {
10810        let severed = ProcessIdentity {
10811            pid: 41,
10812            start_time: 101,
10813        };
10814        let successor = ProcessIdentity {
10815            pid: 41,
10816            start_time: 202,
10817        };
10818        let module = module_with_recovery_snapshot(ModuleState::Running, true, 0);
10819        update_snapshot(&module.inner.snapshot, Some("recovery-snapshot"), |state| {
10820            state.pid = Some(successor.pid);
10821            state.process_start_time = Some(successor.start_time);
10822        })
10823        .unwrap();
10824        assert!(!module.record_deliberate_severance(severed).unwrap());
10825
10826        let exit_report = apply_deliberate_severance_marker(
10827            &module.inner.snapshot,
10828            Some(successor),
10829            ExitReport {
10830                kind: ExitKind::Crash,
10831                code: Some(1),
10832                signal: None,
10833                at_ms: 1,
10834            },
10835        );
10836
10837        assert_eq!(exit_report.kind, ExitKind::Crash);
10838    }
10839
10840    #[tokio::test]
10841    async fn drain_reap_marks_deliberate_severance_and_records_lifetime_without_budget() {
10842        let registry = Registry::default();
10843        let supervisor = Supervisor::new_for_test(
10844            Arc::new(Registry::default()),
10845            RestartPolicy::new(3, Duration::ZERO),
10846        );
10847        let runtime = supervisor.runtime_config();
10848        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10849        let spec = ModuleSpec {
10850            module_id: "drain-deliberate-severance".to_string(),
10851            program: fake_aft_stub_path(),
10852            args: Vec::new(),
10853            env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
10854            reserved: false,
10855            reserved_prefixes: Vec::new(),
10856            protocol: ModuleProtocol::Subc,
10857            overlap: Default::default(),
10858        };
10859        let mut child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
10860        let process = ProcessIdentity {
10861            pid: 41,
10862            start_time: 101,
10863        };
10864        child.process_identity = Some(process);
10865        update_snapshot(&snapshot, Some(&spec.module_id), |state| {
10866            state.pid = Some(process.pid);
10867            state.process_start_time = Some(process.start_time);
10868        })
10869        .unwrap();
10870        record_deliberate_severance(&snapshot, process).unwrap();
10871
10872        drain_child_to_state(
10873            &spec.module_id,
10874            spec.protocol,
10875            // The child exits on its own; no signal may change the exit this
10876            // test classifies.
10877            StopNotice::SentOverConnection,
10878            &registry,
10879            None,
10880            &snapshot,
10881            &runtime.terminal_ring,
10882            &runtime.spawn_events,
10883            child,
10884            Duration::from_secs(1),
10885            ModuleState::Stopped,
10886            Some(false),
10887        )
10888        .await
10889        .unwrap();
10890
10891        let state = lock_snapshot(&snapshot).unwrap();
10892        assert_eq!(
10893            state.last_exit.as_ref().map(|exit| exit.kind),
10894            Some(ExitKind::DeliberateSeverance)
10895        );
10896        assert_eq!(state.lifetime_restarts, 1);
10897        assert_eq!(state.crash_restarts.len(), 0);
10898        drop(state);
10899        let history = runtime.terminal_ring.lock().unwrap().snapshot();
10900        assert_eq!(
10901            history.entries[0].exit_kind,
10902            subc_control::TerminalExitKind::DeliberateSeverance
10903        );
10904    }
10905
10906    #[tokio::test]
10907    async fn ordinary_drain_reap_does_not_record_a_lifetime_restart() {
10908        let registry = Registry::default();
10909        let supervisor = Supervisor::new_for_test(
10910            Arc::new(Registry::default()),
10911            RestartPolicy::new(3, Duration::ZERO),
10912        );
10913        let runtime = supervisor.runtime_config();
10914        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10915        let spec = ModuleSpec {
10916            module_id: "ordinary-drain".to_string(),
10917            program: fake_aft_stub_path(),
10918            args: Vec::new(),
10919            env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
10920            reserved: false,
10921            reserved_prefixes: Vec::new(),
10922            protocol: ModuleProtocol::Subc,
10923            overlap: Default::default(),
10924        };
10925        let child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
10926
10927        drain_child_to_state(
10928            &spec.module_id,
10929            spec.protocol,
10930            // The child exits on its own; no signal may change the exit this
10931            // test classifies.
10932            StopNotice::SentOverConnection,
10933            &registry,
10934            None,
10935            &snapshot,
10936            &runtime.terminal_ring,
10937            &runtime.spawn_events,
10938            child,
10939            Duration::from_secs(1),
10940            ModuleState::Stopped,
10941            Some(false),
10942        )
10943        .await
10944        .unwrap();
10945
10946        let state = lock_snapshot(&snapshot).unwrap();
10947        assert_eq!(
10948            state.last_exit.as_ref().map(|exit| exit.kind),
10949            Some(ExitKind::Crash)
10950        );
10951        assert_eq!(state.lifetime_restarts, 0);
10952        assert_eq!(state.crash_restarts.len(), 0);
10953    }
10954
10955    #[test]
10956    fn fatal_connection_teardown_cannot_arm_a_marker_for_a_surviving_process() {
10957        // The server's generic fatal-routing branch only knows that the
10958        // connection failed; it does not know that the daemon deliberately
10959        // initiated a process-killing severance. Keep this seam explicit so a
10960        // future connection error path cannot silently reintroduce the stale
10961        // exemption that mislabels a later genuine crash.
10962        assert!(!include_str!("server.rs")
10963            .contains("router.record_deliberate_connection_severance(ctx.connection_id)"));
10964    }
10965
10966    /// The `route.closed` `drained` value must be the quiescence wait's own
10967    /// measurement (`Ok`), never invented -- except on `Err`, where there is no
10968    /// measurement at all and `false` is the one honest constant. This is the exact
10969    /// logic `begin_forwarding_drain_with` now applies before sending `route.closed`
10970    /// on every return path, including the one that used to return early via `?`
10971    /// with `route.closing` already sent and no `route.closed` ever following.
10972    #[test]
10973    fn drained_after_quiescence_wait_passes_ok_through_and_forces_false_on_err() {
10974        assert!(drained_after_quiescence_wait(&Ok(true)));
10975        assert!(!drained_after_quiescence_wait(&Ok(false)));
10976        assert!(!drained_after_quiescence_wait(&Err(
10977            SuperviseError::StatePoisoned { module_id: None }
10978        )));
10979    }
10980
10981    /// `supervise_loop`'s `wait()`-error arm now calls `record_terminal` like every
10982    /// other exit path does, so a module whose child `wait()` itself errored (e.g.
10983    /// already reaped out-of-band) still leaves a terminal record rather than none
10984    /// at all. Triggering the real `wait()` I/O error from an integration test would
10985    /// need a genuine already-reaped-child race, which is OS-specific and not
10986    /// something this suite attempts elsewhere; this test instead verifies the
10987    /// record produced for that arm end-to-end through the real `TerminalRing`, and
10988    /// the call site itself is verified by inspection to sit in that exact arm.
10989    #[test]
10990    fn wait_error_exit_report_records_a_failed_terminal_with_no_code_or_signal() {
10991        let ring = Arc::new(Mutex::new(TerminalRing::new(
10992            TerminalRingConfig::default(),
10993            0,
10994        )));
10995        record_wait_error_terminal("wait-error", &ring, &super::SpawnEventFeed::default());
10996
10997        let snapshot = ring.lock().unwrap().snapshot();
10998        assert_eq!(snapshot.entries.len(), 1);
10999        let entry = &snapshot.entries[0];
11000        assert_eq!(entry.exit_code, None);
11001        assert_eq!(entry.exit_signal, None);
11002        assert_eq!(entry.disposition, TerminalDisposition::Failed);
11003    }
11004
11005    #[test]
11006    fn wait_error_exit_path_preserves_spawn_event_density() {
11007        let feed = super::SpawnEventFeed::default();
11008        feed.configure_incarnation("wait-error-density".to_string());
11009        feed.emit_spawned("wait-error", 41, 1);
11010        let ring = Arc::new(Mutex::new(TerminalRing::new(
11011            TerminalRingConfig::default(),
11012            0,
11013        )));
11014
11015        record_wait_error_terminal("wait-error", &ring, &feed);
11016        feed.emit_spawned("after-wait-error", 42, 2);
11017
11018        let state = feed.0.lock().unwrap();
11019        let sequences = state
11020            .events
11021            .iter()
11022            .map(|event| event.cursor.seq)
11023            .collect::<Vec<_>>();
11024        assert_eq!(sequences, vec![1, 2, 3]);
11025        assert_eq!(state.events[1].kind, SpawnEventKind::Exited);
11026        assert_eq!(state.events[1].exit_code, None);
11027        assert_eq!(state.events[1].exit_signal, None);
11028    }
11029
11030    /// Pins the report's `kind` too: the wait-error arm treats an unwaitable child
11031    /// as a crash (matching `fail_snapshot`'s `Failed` disposition for this arm),
11032    /// not a clean exit it never actually observed.
11033    #[test]
11034    fn wait_error_exit_report_is_classified_as_a_crash() {
11035        assert_eq!(wait_error_exit_report().kind, ExitKind::Crash);
11036    }
11037}
11038
11039#[cfg(test)]
11040mod health_evidence_tests {
11041    use super::{HealthProbeError, HealthProbeEvidence};
11042    use std::collections::HashSet;
11043
11044    /// The evidential asymmetry, asserted rather than described.
11045    ///
11046    /// Exactly ONE observation is proof a module cannot serve, and the one that
11047    /// fires under CPU starvation is not it. Before the split, all fifteen
11048    /// construction sites collapsed into a single String, so a timeout carried the
11049    /// same weight as a dead lane -- which is how a healthy module was restarted
11050    /// three times in one day.
11051    #[test]
11052    fn only_a_dead_lane_is_proof_of_death() {
11053        assert!(HealthProbeError::lane_dead("gone").is_proof_of_death());
11054        // Three non-proof classes, each for a different reason: silence is
11055        // consistent with health, a bad answer proves the module ALIVE, and a
11056        // daemon-side fault never reached the module at all.
11057        assert!(!HealthProbeError::no_answer("timed out").is_proof_of_death());
11058        assert!(!HealthProbeError::bad_answer("garbage").is_proof_of_death());
11059        assert!(!HealthProbeError::misconfigured("no table").is_proof_of_death());
11060    }
11061
11062    /// Labels must be distinct, or the operator-facing distinction is cosmetic.
11063    ///
11064    /// A shared label renders two different observations identically in the line an
11065    /// operator reads after an unexplained restart -- the exact confusion this
11066    /// change removes.
11067    #[test]
11068    fn every_evidence_class_has_a_distinct_label() {
11069        let labels = [
11070            HealthProbeError::lane_dead("").label(),
11071            HealthProbeError::no_answer("").label(),
11072            HealthProbeError::bad_answer("").label(),
11073            HealthProbeError::misconfigured("").label(),
11074        ];
11075        let unique: HashSet<_> = labels.iter().collect();
11076        assert_eq!(unique.len(), labels.len(), "labels collided: {labels:?}");
11077    }
11078
11079    /// The class is additional information, not a replacement.
11080    ///
11081    /// An operator needs both "this was silence" and the specific text saying how
11082    /// long we waited; a classification that swallowed the message would trade one
11083    /// missing distinction for another.
11084    #[test]
11085    fn classification_preserves_the_original_message() {
11086        let err = HealthProbeError::no_answer("module did not answer within 5s");
11087        assert_eq!(err.to_string(), "module did not answer within 5s");
11088        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
11089    }
11090}
11091
11092#[cfg(test)]
11093mod health_tombstone_tests {
11094    use std::{path::PathBuf, sync::Arc, time::Duration};
11095
11096    use subc_protocol::{
11097        manifest::Concurrency,
11098        session::{HealthStatus, ModuleControlResponse},
11099    };
11100    use tokio::sync::mpsc;
11101
11102    use super::{
11103        probe_module_health, HealthAction, HealthConfig, HealthProbeEvidence, ModuleProtocol,
11104        ModuleSpec, RestartPolicy, Supervisor, SupervisorRuntimeConfig,
11105    };
11106    use crate::{
11107        control::ControlHandler,
11108        forwarding::{ForwardingTable, ModuleControlRpcCompletion, ModuleControlRpcOutcome},
11109        registry::{ConnectionId, Registry},
11110        router::FrameSink,
11111    };
11112
11113    struct ProbeHarness {
11114        spec: ModuleSpec,
11115        runtime: SupervisorRuntimeConfig,
11116        forwarding: Arc<ForwardingTable>,
11117        module_connection: ConnectionId,
11118        module_rx: mpsc::Receiver<crate::router::OutboundFrame>,
11119        handler: ControlHandler,
11120        module: super::SupervisedModule,
11121    }
11122
11123    fn probe_harness() -> ProbeHarness {
11124        let registry = Arc::new(Registry::default());
11125        let forwarding = Arc::new(ForwardingTable::default());
11126        let supervisor_handle = super::SupervisorHandle::new();
11127        let health = HealthConfig {
11128            http: None,
11129            cadence: Duration::from_secs(30),
11130            deadline: Duration::from_secs(5),
11131            failure_threshold: 3,
11132            on_degraded: HealthAction::Report,
11133            on_failing: HealthAction::Report,
11134            critical: false,
11135        };
11136        let supervisor = Supervisor::new_for_test(Arc::clone(&registry), RestartPolicy::default())
11137            .with_forwarding(Arc::clone(&forwarding))
11138            .with_handle(supervisor_handle.clone())
11139            .with_health_config(health);
11140        let spec = ModuleSpec {
11141            module_id: "late-health-module".to_string(),
11142            program: PathBuf::from("disabled-module"),
11143            args: Vec::new(),
11144            env: Vec::new(),
11145            reserved: false,
11146            reserved_prefixes: Vec::new(),
11147            protocol: ModuleProtocol::Subc,
11148            overlap: Default::default(),
11149        };
11150        let module = supervisor
11151            .supervise_configured(spec.clone(), false)
11152            .unwrap();
11153        let runtime = supervisor.runtime_config();
11154        let handler = ControlHandler::with_forwarding(registry, Arc::clone(&forwarding))
11155            .with_supervisor(supervisor_handle);
11156        let module_connection = ConnectionId::new(700);
11157        let (module_tx, module_rx) = mpsc::channel(8);
11158        forwarding
11159            .register_module_connection(
11160                module_connection,
11161                spec.module_id.clone(),
11162                subc_protocol::PROTOCOL_VERSION,
11163                Concurrency::ModuleManaged,
11164                FrameSink::new(module_tx),
11165            )
11166            .unwrap();
11167
11168        ProbeHarness {
11169            spec,
11170            runtime,
11171            forwarding,
11172            module_connection,
11173            module_rx,
11174            handler,
11175            module,
11176        }
11177    }
11178
11179    async fn finish_after(
11180        harness: &mut ProbeHarness,
11181        stall: Duration,
11182    ) -> ModuleControlRpcCompletion {
11183        assert!(stall > harness.runtime.health.deadline);
11184        let deadline = harness.runtime.health.deadline;
11185        let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
11186        let answer = async {
11187            let frame = harness.module_rx.recv().await.expect("health.check frame");
11188            tokio::time::advance(deadline).await;
11189            tokio::task::yield_now().await;
11190            tokio::time::advance(stall - deadline).await;
11191            harness
11192                .forwarding
11193                .complete_module_control_rpc(
11194                    harness.module_connection,
11195                    frame.header.corr,
11196                    Some("health.check"),
11197                    ModuleControlRpcOutcome::Response(ModuleControlResponse::HealthCheck {
11198                        status: HealthStatus::Ok,
11199                        detail: None,
11200                        metrics: None,
11201                    }),
11202                )
11203                .unwrap()
11204        };
11205        let (probe_result, completion) = tokio::join!(probe, answer);
11206        let err = probe_result.expect_err("probe must miss its deadline");
11207        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
11208        completion
11209    }
11210
11211    async fn time_out_without_answer(harness: &mut ProbeHarness) {
11212        let deadline = harness.runtime.health.deadline;
11213        let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
11214        let exhaust_deadline = async {
11215            let _frame = harness.module_rx.recv().await.expect("health.check frame");
11216            tokio::time::advance(deadline).await;
11217            tokio::task::yield_now().await;
11218        };
11219        let (probe_result, ()) = tokio::join!(probe, exhaust_deadline);
11220        let err = probe_result.expect_err("probe must miss its deadline");
11221        assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
11222    }
11223
11224    #[tokio::test(start_paused = true)]
11225    async fn late_health_answers_record_start_anchored_latency_for_two_stalls() {
11226        let mut harness = probe_harness();
11227
11228        let first = finish_after(&mut harness, Duration::from_secs(8)).await;
11229        let first_latency = match &first {
11230            ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
11231            other => panic!("late answer was not retained: {other:?}"),
11232        };
11233        assert!(harness.handler.observe_module_control_completion(first));
11234
11235        let second = finish_after(&mut harness, Duration::from_secs(11)).await;
11236        let second_latency = match &second {
11237            ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
11238            other => panic!("late answer was not retained: {other:?}"),
11239        };
11240        assert!(harness.handler.observe_module_control_completion(second));
11241
11242        assert_eq!(first_latency, Duration::from_secs(8));
11243        assert_eq!(
11244            second_latency - first_latency,
11245            Duration::from_secs(3),
11246            "latency must grow linearly with the additional stall"
11247        );
11248        let health = harness.module.status().unwrap().health;
11249        assert_eq!(health.late_answer_count, 2);
11250        assert_eq!(health.last_late_answer_latency_ms, Some(11_000));
11251    }
11252
11253    /// A module that answers every probe late must never march to the kill
11254    /// threshold: the late answer proves it is alive, so it must clear the miss
11255    /// streak the timeout recorded. Without the reset, a CPU-starved module
11256    /// that serves every probe seconds past the deadline accumulates
11257    /// `consecutive_failures` to the threshold and is killed — the exact
11258    /// sequence from the 2026-08-14 aft disable, where the daemon logged
11259    /// "proves the module is alive" five times while counting five misses.
11260    #[tokio::test(start_paused = true)]
11261    async fn late_answer_clears_the_consecutive_failure_streak() {
11262        let mut harness = probe_harness();
11263
11264        // Timeout recorded first: the probe path saw no answer in time.
11265        time_out_without_answer(&mut harness).await;
11266        harness
11267            .module
11268            .record_health_probe_failure_for_test("[no-answer] test miss")
11269            .unwrap();
11270        assert_eq!(
11271            harness.module.status().unwrap().health.consecutive_failures,
11272            1,
11273            "precondition: the miss must be on the streak before the late answer"
11274        );
11275
11276        // The stalled reply then lands: proof of life.
11277        let late = finish_after(&mut harness, Duration::from_secs(9)).await;
11278        assert!(matches!(
11279            late,
11280            ModuleControlRpcCompletion::LateHealthAnswer { .. }
11281        ));
11282        assert!(harness.handler.observe_module_control_completion(late));
11283
11284        let health = harness.module.status().unwrap().health;
11285        assert_eq!(
11286            health.consecutive_failures, 0,
11287            "a late answer is an answer: the streak must reset"
11288        );
11289        assert_eq!(health.late_answer_count, 1);
11290    }
11291
11292    #[tokio::test(start_paused = true)]
11293    async fn repeated_serial_probe_cycles_keep_one_tombstone_per_endpoint() {
11294        let mut harness = probe_harness();
11295
11296        for _ in 0..20 {
11297            time_out_without_answer(&mut harness).await;
11298            assert_eq!(
11299                harness.forwarding.health_probe_tombstone_count().unwrap(),
11300                1
11301            );
11302        }
11303    }
11304}
11305
11306#[cfg(test)]
11307mod child_env_tests {
11308    use super::{
11309        apply_child_env, apply_spawn_role, apply_wire_spawn_args, ModuleProtocol, ModuleSpec,
11310        SpawnRole, SupervisorHandle, SPAWN_ROLE_SWAP_CANDIDATE, SUBC_ARG, SUBC_LAUNCH_NONCE_ENV,
11311        SUBC_MODULE_ID_ENV, SUBC_SPAWN_ROLE_ENV,
11312    };
11313    use std::{ffi::OsStr, path::PathBuf};
11314    use tokio::process::Command;
11315
11316    fn spec(env: Vec<(String, String)>) -> ModuleSpec {
11317        ModuleSpec {
11318            module_id: "env-plan".to_string(),
11319            program: PathBuf::from("/nonexistent"),
11320            args: Vec::new(),
11321            env,
11322            reserved: false,
11323            reserved_prefixes: Vec::new(),
11324            protocol: ModuleProtocol::Subc,
11325            overlap: Default::default(),
11326        }
11327    }
11328
11329    /// Ambient `CK_LOG` is REMOVED for an unconfigured module, and a configured
11330    /// one still gets its own.
11331    ///
11332    /// This is the narrow goal `env_clear()` was reached for, and the reason the
11333    /// fix is `env_remove` rather than deleting the line: an operator's ambient
11334    /// filter silently becoming an unconfigured module's log level is a real
11335    /// defect, just a much smaller one than clearing the environment.
11336    ///
11337    /// Asserted on the command plan rather than a spawned child because proving
11338    /// the ABSENCE of an inherited variable needs the parent's environment
11339    /// mutated, and `forbid(unsafe_code)` refuses that. `get_envs()` reports a
11340    /// removal as `(key, None)`, which is exactly the distinction wanted: not
11341    /// "absent because nobody set it" but "explicitly unset for the child".
11342    #[test]
11343    fn ambient_ck_log_is_removed_and_a_configured_one_survives() {
11344        let mut command = Command::new("/nonexistent");
11345        apply_child_env(&mut command, &spec(Vec::new()));
11346        let removed = command
11347            .as_std()
11348            .get_envs()
11349            .any(|(key, value)| key == OsStr::new("CK_LOG") && value.is_none());
11350        assert!(
11351            removed,
11352            "ambient CK_LOG must be explicitly removed for an unconfigured module"
11353        );
11354
11355        let mut configured = Command::new("/nonexistent");
11356        apply_child_env(
11357            &mut configured,
11358            &spec(vec![("CK_LOG".to_string(), "debug".to_string())]),
11359        );
11360        let effective = configured
11361            .as_std()
11362            .get_envs()
11363            .filter(|(key, _)| *key == OsStr::new("CK_LOG"))
11364            .last()
11365            .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
11366        assert_eq!(
11367            effective,
11368            Some(Some("debug".to_string())),
11369            "a module's configured CK_LOG must survive the ambient removal"
11370        );
11371    }
11372
11373    /// A `protocol: "none"` spawn carries NO `--subc` argument and NO launch
11374    /// nonce; a subc-wire spawn carries both. Asserted on the command plan for
11375    /// the same reason as the CK_LOG test above.
11376    ///
11377    /// The argument is the load-bearing half: a stock binary exits on an
11378    /// unknown flag before it listens, so with `--subc` appended the mode
11379    /// could not supervise the one process it exists for. Found by the first
11380    /// conformance run (nats-server: `flag provided but not defined: -subc`).
11381    #[test]
11382    fn protocol_none_spawn_carries_no_subc_argument_and_no_nonce() {
11383        let connection_file = std::path::Path::new("/run/subc-connection.json");
11384        let handle = SupervisorHandle::new();
11385
11386        let mut none_spec = spec(Vec::new());
11387        none_spec.protocol = ModuleProtocol::None;
11388        let mut none = Command::new("/nonexistent");
11389        let none_handoff =
11390            apply_wire_spawn_args(&mut none, &none_spec, Some(connection_file), Some(&handle))
11391                .expect("protocol-none spawn args apply");
11392        assert!(
11393            none_handoff.is_none(),
11394            "protocol:none spawn must not receive a nonce descriptor"
11395        );
11396        assert!(
11397            !none.as_std().get_envs().any(|(key, value)| key
11398                == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV)
11399                && value.is_some()),
11400            "protocol:none spawn must not name a nonce descriptor"
11401        );
11402        let none_args: Vec<String> = none
11403            .as_std()
11404            .get_args()
11405            .map(|a| a.to_string_lossy().into_owned())
11406            .collect();
11407        assert!(
11408            !none_args.iter().any(|a| a == SUBC_ARG),
11409            "protocol:none argv must not carry --subc; got {none_args:?}"
11410        );
11411        let none_has_nonce = none
11412            .as_std()
11413            .get_envs()
11414            .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some());
11415        assert!(
11416            !none_has_nonce,
11417            "protocol:none spawn must not receive a launch nonce"
11418        );
11419        let none_has_module_id = none
11420            .as_std()
11421            .get_envs()
11422            .any(|(key, value)| key == OsStr::new(SUBC_MODULE_ID_ENV) && value.is_some());
11423        assert!(
11424            none_has_module_id,
11425            "SUBC_MODULE_ID is inert and stays on every path"
11426        );
11427        assert!(
11428            handle.spawn_nonce(&none_spec.module_id).is_none(),
11429            "no nonce record for a process that will never present one"
11430        );
11431
11432        // Control: the subc-wire path is unchanged by the branch above.
11433        let wire_spec = spec(Vec::new());
11434        let mut wire = Command::new("/nonexistent");
11435        let wire_handoff =
11436            apply_wire_spawn_args(&mut wire, &wire_spec, Some(connection_file), Some(&handle))
11437                .expect("subc-wire spawn args apply");
11438        let wire_fd_env = wire
11439            .as_std()
11440            .get_envs()
11441            .find(|(key, _)| *key == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV))
11442            .and_then(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
11443        #[cfg(unix)]
11444        assert_eq!(
11445            wire_fd_env,
11446            Some(wire_handoff.expect("a descriptor handoff").fd_env_value()),
11447            "a subc-wire spawn names the pipe it will receive at descriptor 3"
11448        );
11449        #[cfg(not(unix))]
11450        assert!(wire_handoff.is_none() && wire_fd_env.is_none());
11451        let wire_args: Vec<String> = wire
11452            .as_std()
11453            .get_args()
11454            .map(|a| a.to_string_lossy().into_owned())
11455            .collect();
11456        assert_eq!(
11457            wire_args,
11458            vec![
11459                SUBC_ARG.to_string(),
11460                connection_file.to_string_lossy().into_owned()
11461            ],
11462            "a subc-wire spawn still carries --subc <path>"
11463        );
11464        assert_eq!(
11465            wire.as_std()
11466                .get_envs()
11467                .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some()),
11468            !cfg!(unix),
11469            "only Windows supplies the environment nonce"
11470        );
11471        assert!(handle.spawn_nonce(&wire_spec.module_id).is_some());
11472    }
11473
11474    /// A plain spawn EXPLICITLY REMOVES the spawn role, even when the module's
11475    /// spec tries to set it; only a swap candidate carries it.
11476    ///
11477    /// "Set it only on candidates" is not enough, because spawn applies the
11478    /// spec's env verbatim and the daemon's own environment is inherited: either
11479    /// could hand a plain restart the swap role, and a module reading it would
11480    /// warm on its long swap budget while callers wait. Asserted as an explicit
11481    /// removal (`(key, None)`), not mere absence, for the reason the `CK_LOG`
11482    /// test above gives.
11483    #[test]
11484    fn plain_spawn_removes_the_spawn_role_even_when_the_spec_sets_it() {
11485        let role = |command: &Command| {
11486            command
11487                .as_std()
11488                .get_envs()
11489                .filter(|(key, _)| *key == OsStr::new(SUBC_SPAWN_ROLE_ENV))
11490                .last()
11491                .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()))
11492        };
11493        let forged = spec(vec![(
11494            SUBC_SPAWN_ROLE_ENV.to_string(),
11495            SPAWN_ROLE_SWAP_CANDIDATE.to_string(),
11496        )]);
11497
11498        let mut plain = Command::new("/nonexistent");
11499        apply_child_env(&mut plain, &forged);
11500        apply_spawn_role(&mut plain, SpawnRole::Plain);
11501        assert_eq!(
11502            role(&plain),
11503            Some(None),
11504            "a plain spawn must remove SUBC_SPAWN_ROLE, whatever the spec says"
11505        );
11506
11507        let mut candidate = Command::new("/nonexistent");
11508        apply_child_env(&mut candidate, &spec(Vec::new()));
11509        apply_spawn_role(&mut candidate, SpawnRole::SwapCandidate);
11510        assert_eq!(
11511            role(&candidate),
11512            Some(Some(SPAWN_ROLE_SWAP_CANDIDATE.to_string()))
11513        );
11514    }
11515
11516    /// Daemon-private capture retention keys never reach the child.
11517    ///
11518    /// cortexkit-log exposes retention as a Rust struct with no environment
11519    /// names, so these entries are supervisor metadata. Passing them through
11520    /// would invent a public child-process contract by accident.
11521    #[test]
11522    fn daemon_private_capture_keys_are_not_passed_to_the_child() {
11523        let mut command = Command::new("/nonexistent");
11524        apply_child_env(
11525            &mut command,
11526            &spec(vec![
11527                (super::CAPTURE_KEEP_ENV.to_string(), "5".to_string()),
11528                ("KEPT".to_string(), "yes".to_string()),
11529            ]),
11530        );
11531        let keys: Vec<String> = command
11532            .as_std()
11533            .get_envs()
11534            .filter(|(_, value)| value.is_some())
11535            .map(|(key, _)| key.to_string_lossy().into_owned())
11536            .collect();
11537        assert!(keys.contains(&"KEPT".to_string()), "got {keys:?}");
11538        assert!(
11539            !keys.contains(&super::CAPTURE_KEEP_ENV.to_string()),
11540            "daemon-private capture key leaked to the child: {keys:?}"
11541        );
11542    }
11543}
11544
11545#[cfg(test)]
11546mod jitter_tests {
11547    use super::jittered_health_delay;
11548    use std::{collections::HashSet, time::Duration};
11549
11550    /// Module ids drawn from a real fleet, so the dispersal claim is about names
11551    /// that actually occur rather than invented ones.
11552    ///
11553    /// This is a SAMPLE, not a registry: the property under test is that distinct
11554    /// ids disperse, which holds for any set of distinct strings. Several entries
11555    /// are already historical (modules get renamed), and that costs nothing here --
11556    /// but it means a reader must not mistake this for the live module set, and a
11557    /// rename sweep will match it without there being anything to change.
11558    const FLEET: [&str; 14] = [
11559        "aft",
11560        "alfonso-core",
11561        "magic-context",
11562        "broca",
11563        "thalamus",
11564        "quota",
11565        "engram",
11566        "plexus",
11567        "cerebellum",
11568        "astrocyte",
11569        "synapse",
11570        "subc-mcp",
11571        "cortexkit-credentials",
11572        "subc-federation",
11573    ];
11574
11575    /// Probes must not converge after a fleet-wide restart.
11576    ///
11577    /// This is the property the jitter exists for: every module reconnects at
11578    /// once, and without dispersal all fourteen would then probe on the same
11579    /// tick forever. Nothing failed visibly when this went untested -- a
11580    /// convergent fleet still probes correctly, just in a burst, so the symptom
11581    /// is a periodic load spike that looks like whatever else is running.
11582    #[test]
11583    fn probe_delays_disperse_across_the_fleet() {
11584        let cadence = Duration::from_secs(30);
11585        let delays: HashSet<Duration> = FLEET
11586            .iter()
11587            .map(|id| jittered_health_delay(id, 0, cadence))
11588            .collect();
11589        assert_eq!(
11590            delays.len(),
11591            FLEET.len(),
11592            "every supervised module must land on its own probe offset"
11593        );
11594    }
11595
11596    /// The offset may only ever DELAY a probe, never bring it forward.
11597    ///
11598    /// A delay below the cadence would probe a module more often than
11599    /// configured, which is the opposite of what an operator asked for and
11600    /// would tighten the failure budget without anyone changing it.
11601    #[test]
11602    fn jitter_only_delays_and_stays_within_one_tenth_of_cadence() {
11603        let cadence = Duration::from_secs(30);
11604        let span = cadence / 10;
11605        for id in FLEET {
11606            for probe_index in 0..8 {
11607                let delay = jittered_health_delay(id, probe_index, cadence);
11608                assert!(
11609                    delay >= cadence,
11610                    "{id}#{probe_index}: jitter must not shorten the cadence"
11611                );
11612                assert!(
11613                    delay < cadence + span,
11614                    "{id}#{probe_index}: jitter must stay inside one tenth of the cadence"
11615                );
11616            }
11617        }
11618    }
11619
11620    /// A module keeps its offset across daemon restarts.
11621    ///
11622    /// The delay is derived rather than randomised precisely so a restart does
11623    /// not re-roll every module into a fresh chance of collision. A random
11624    /// source would satisfy the dispersal test above and quietly lose this.
11625    #[test]
11626    fn a_module_offset_is_stable_across_restarts() {
11627        let cadence = Duration::from_secs(30);
11628        for id in FLEET {
11629            assert_eq!(
11630                jittered_health_delay(id, 0, cadence),
11631                jittered_health_delay(id, 0, cadence),
11632                "{id}: the same module and probe index must produce the same offset"
11633            );
11634        }
11635    }
11636
11637    /// A zero cadence disables probing rather than producing a busy loop.
11638    #[test]
11639    fn zero_cadence_yields_zero_delay() {
11640        assert_eq!(
11641            jittered_health_delay("aft", 0, Duration::ZERO),
11642            Duration::ZERO
11643        );
11644    }
11645}
11646
11647#[cfg(all(test, target_os = "linux"))]
11648mod cgroup_placement_tests {
11649    use super::{
11650        apply_cgroup_placement, remove_module_cgroup, ModuleProtocol, ModuleSpec, SuperviseError,
11651        SupervisedChild,
11652    };
11653    use crate::stderr_tail::{StderrRing, StderrTailConfig};
11654    use std::{
11655        fs, io,
11656        path::{Path, PathBuf},
11657        sync::{Arc, Mutex},
11658    };
11659    use subc_test_support::TestTempDir;
11660    use tokio::process::Command;
11661
11662    #[tokio::test]
11663    async fn unique_spawn_cgroups_clean_up_across_restarts_and_shutdown() {
11664        use super::*;
11665        let dir = TestTempDir::new("unique-spawn-cgroups");
11666        let root = PathBuf::from(format!(
11667            "/sys/fs/cgroup/subc-unique-test-{}-{}",
11668            std::process::id(),
11669            unix_ms_now()
11670        ));
11671        if let Err(error) = fs::create_dir(&root) {
11672            assert!(
11673                std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
11674                "required cgroup test cannot execute: {error}"
11675            );
11676            eprintln!("SKIP unique_spawn_cgroups_clean_up_across_restarts_and_shutdown: {error}");
11677            return;
11678        }
11679        let placement = subc_cgroup::prepare_at(&root).unwrap().unwrap();
11680        let group_count = || {
11681            fs::read_dir(root.join("subc-modules"))
11682                .unwrap()
11683                .map(|entry| entry.unwrap().file_type().unwrap())
11684                .filter(|kind| kind.is_dir())
11685                .count()
11686        };
11687        let supervisor =
11688            Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
11689                .with_cgroup_placement(Some(placement.clone()));
11690        let runtime = supervisor.runtime_config();
11691        let mut spec = ModuleSpec {
11692            module_id: "unique-spawn".into(),
11693            program: PathBuf::from("/bin/sleep"),
11694            args: vec!["60".into()],
11695            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
11696                .into_iter()
11697                .map(|key| (key.into(), dir.to_string_lossy().into_owned()))
11698                .collect(),
11699            reserved: false,
11700            reserved_prefixes: vec![],
11701            protocol: ModuleProtocol::None,
11702            overlap: Default::default(),
11703        };
11704        let spawn = |spec: &ModuleSpec| {
11705            spawn_child(
11706                spec,
11707                None,
11708                None,
11709                &runtime.stderr_ring,
11710                None,
11711                &runtime.child_roster,
11712                Some(&placement),
11713            )
11714            .unwrap()
11715        };
11716        let mut live = spawn(&spec);
11717        for _ in 0..3 {
11718            // A new process can enter the old slot while retirement is pending.
11719            let next = spawn(&spec);
11720            assert_ne!(live.module_id, next.module_id);
11721            live.start_kill().unwrap();
11722            live.wait().await.unwrap();
11723            live = next;
11724            assert!(
11725                live.child.try_wait().unwrap().is_none(),
11726                "retiring the old slot must not kill the replacement"
11727            );
11728            assert_eq!(
11729                group_count(),
11730                1,
11731                "only the live spawn's cgroup should remain"
11732            );
11733        }
11734        supervisor.begin_daemon_shutdown();
11735        let reap = tokio::spawn(async move {
11736            live.wait().await.unwrap();
11737        });
11738        supervisor
11739            .end_children_for_daemon_shutdown(false, std::future::pending())
11740            .await;
11741        reap.await.unwrap();
11742        assert_eq!(group_count(), 0);
11743        // A normal exit uses the same tree-cleanup path as a killed spawn.
11744        spec.program = PathBuf::from("/bin/true");
11745        spec.args.clear();
11746        let fresh_roster = ChildRoster::default();
11747        let mut short = spawn_child(
11748            &spec,
11749            None,
11750            None,
11751            &runtime.stderr_ring,
11752            None,
11753            &fresh_roster,
11754            Some(&placement),
11755        )
11756        .unwrap();
11757        short.wait().await.unwrap();
11758        assert_eq!(group_count(), 0);
11759        spec.module_id = "_".repeat(255);
11760        let mut long_id = spawn_child(
11761            &spec,
11762            None,
11763            None,
11764            &runtime.stderr_ring,
11765            None,
11766            &fresh_roster,
11767            Some(&placement),
11768        )
11769        .unwrap();
11770        long_id.wait().await.unwrap();
11771        assert_eq!(
11772            group_count(),
11773            0,
11774            "valid long module IDs must not exceed cgroup NAME_MAX"
11775        );
11776        fs::remove_dir(root.join("subc-modules")).unwrap();
11777        fs::remove_dir(root).unwrap();
11778        eprintln!("EXECUTED unique_spawn_cgroups_clean_up_across_restarts_and_shutdown");
11779    }
11780
11781    #[test]
11782    fn failed_parent_cgroup_open_is_a_cgroup_supervision_error() {
11783        let path = Path::new("/definitely-missing-subc-cgroup");
11784        let mut command = Command::new("true");
11785        let error = apply_cgroup_placement(
11786            &mut command,
11787            &ModuleSpec {
11788                module_id: "broken-cgroup".to_string(),
11789                program: PathBuf::from("true"),
11790                args: Vec::new(),
11791                env: Vec::new(),
11792                reserved: false,
11793                reserved_prefixes: Vec::new(),
11794                protocol: ModuleProtocol::Subc,
11795                overlap: Default::default(),
11796            },
11797            path,
11798        )
11799        .expect_err("a parent cgroup open failure must reject the supervised spawn");
11800        let reason = error.to_string();
11801
11802        assert!(
11803            matches!(error, SuperviseError::Cgroup { .. }),
11804            "parent cgroup open must be reported as a cgroup supervision error: {reason}"
11805        );
11806        assert!(
11807            reason.contains("/definitely-missing-subc-cgroup/cgroup.procs"),
11808            "parent cgroup open failure must name cgroup.procs: {reason}"
11809        );
11810    }
11811
11812    #[tokio::test]
11813    async fn reaping_a_child_removes_its_empty_module_cgroup() {
11814        let root = TestTempDir::new("supervisor-reap-cgroup");
11815        fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
11816        let placement = subc_cgroup::prepare_at(&root)
11817            .expect("prepare scratch cgroup root")
11818            .expect("scratch root has a cgroup.procs marker");
11819        let module_id = "reaped-module";
11820        let module = placement
11821            .module_path(module_id)
11822            .expect("create scratch module cgroup");
11823        let child = Command::new("true")
11824            .env("XDG_DATA_HOME", root.path())
11825            .env("XDG_RUNTIME_DIR", root.path())
11826            .env("XDG_CONFIG_HOME", root.path())
11827            .spawn()
11828            .expect("spawn short-lived child");
11829        let pid = child.id().expect("spawned child has pid");
11830        let mut child = SupervisedChild {
11831            child,
11832            protocol: ModuleProtocol::Subc,
11833            module_id: module_id.to_string(),
11834            cgroup_placement: Some(placement),
11835            stdout_pump: None,
11836            stderr_pump: None,
11837            stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
11838            spawned_at_ms: 0,
11839            spawned_from: PathBuf::from("true"),
11840            spawned_file_identity: None,
11841            process_start_time: None,
11842            process_identity: None,
11843            pid,
11844            roster_guard: None,
11845            #[cfg(target_os = "macos")]
11846            privacy_exec: None,
11847            spawn_failure: None,
11848        };
11849
11850        child.wait().await.expect("reap short-lived child");
11851
11852        assert!(
11853            !module.exists(),
11854            "reaping the supervised child must remove its empty cgroup"
11855        );
11856    }
11857
11858    #[test]
11859    fn non_empty_cgroup_removal_is_reported_without_blocking_teardown() {
11860        let root = TestTempDir::new("supervisor-non-empty-cgroup");
11861        fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
11862        let placement = subc_cgroup::prepare_at(&root)
11863            .expect("prepare scratch cgroup root")
11864            .expect("scratch root has a cgroup.procs marker");
11865        let module = placement
11866            .module_path("surviving-module")
11867            .expect("create scratch module cgroup");
11868        fs::write(module.join("surviving-process"), b"still present")
11869            .expect("make scratch cgroup non-empty");
11870        let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::WARN);
11871
11872        remove_module_cgroup(&placement, "surviving-module");
11873
11874        let logs = crate::router::test_log::captured_logs(&logs);
11875        assert!(
11876            module.exists(),
11877            "failed removal must leave the cgroup intact"
11878        );
11879        assert!(
11880            logs.contains("could not remove module cgroup after process exit; continuing teardown")
11881                && logs.contains("surviving-module"),
11882            "best-effort removal must report the failure without returning it: {logs}"
11883        );
11884    }
11885
11886    #[test]
11887    fn cgroup_pre_exec_spawn_failure_names_the_cgroup_path() {
11888        let cgroup_path = PathBuf::from("/sys/fs/cgroup/subc-modules/broken-module");
11889        let reason = SuperviseError::Spawn {
11890            program: PathBuf::from("/bin/true"),
11891            source: io::Error::from_raw_os_error(13),
11892            cgroup_path: Some(cgroup_path.clone()),
11893        }
11894        .to_string();
11895
11896        assert!(
11897            reason.contains(&cgroup_path.display().to_string()),
11898            "a pre_exec spawn failure must name the cgroup path: {reason}"
11899        );
11900    }
11901}
11902
11903#[cfg(test)]
11904mod spawn_subscriber_lag_tests {
11905    use super::*;
11906
11907    /// A subscriber whose connection stops draining is dropped once its frame
11908    /// channel fills. The client must learn that from a terminal Error frame
11909    /// after the frames already queued for it, not from a stream that simply
11910    /// goes quiet.
11911    #[tokio::test]
11912    async fn lagged_spawn_subscriber_receives_a_terminal_lagged_error_after_its_queued_frames() {
11913        let feed = SpawnEventFeed::default();
11914        feed.configure_incarnation("lag-incarnation".to_string());
11915        // A one-slot connection queue that nobody reads until the emits are
11916        // done: the forwarder parks on it and the subscriber channel fills.
11917        let (tx, mut rx) = mpsc::channel(1);
11918        feed.subscribe(ConnectionId::new(1), 7, 1, None, FrameSink::new(tx))
11919            .expect("subscribe");
11920        let emitted = SPAWN_SUBSCRIBER_BUFFER + 16;
11921        for index in 0..emitted {
11922            feed.emit_spawned(&format!("lag-module-{index}"), 1000, 0);
11923            // Let the forwarder take what it can so the fill point is the
11924            // subscriber channel, not a scheduling accident.
11925            tokio::task::yield_now().await;
11926        }
11927        assert_eq!(
11928            feed.subscriber_count(),
11929            0,
11930            "the lagged subscriber must be removed"
11931        );
11932
11933        let mut data = Vec::new();
11934        let mut last = None;
11935        loop {
11936            let next = tokio::time::timeout(Duration::from_secs(5), rx.recv())
11937                .await
11938                .expect("the forwarder must finish once the subscriber is dropped");
11939            let Some(outbound) = next else { break };
11940            let frame = outbound.frame;
11941            if frame.header.ty == FrameType::StreamData {
11942                assert!(last.is_none(), "no data may follow the terminal frame");
11943                let event: SpawnEvent = serde_json::from_slice(&frame.body).unwrap();
11944                data.push(event.cursor.seq);
11945            } else {
11946                assert!(last.is_none(), "exactly one terminal frame");
11947                last = Some(frame);
11948            }
11949        }
11950        assert!(!data.is_empty(), "queued frames drain before the terminal");
11951        for pair in data.windows(2) {
11952            assert_eq!(
11953                pair[1],
11954                pair[0] + 1,
11955                "queued frames arrive dense and in order"
11956            );
11957        }
11958        let terminal = last.expect("a lagged subscriber must receive a terminal frame");
11959        assert_eq!(terminal.header.ty, FrameType::Error);
11960        assert_eq!(terminal.header.corr, 7);
11961        let body: subc_protocol::ErrorBody = serde_json::from_slice(&terminal.body).unwrap();
11962        assert_eq!(body.code, SPAWN_SUBSCRIBER_LAGGED_CODE);
11963        let detail = body.detail.expect("lagged error carries detail");
11964        assert_eq!(
11965            detail["first_undelivered_cursor"]["seq"],
11966            data.last().unwrap() + 1,
11967            "the named cursor is the first event the subscriber did not receive"
11968        );
11969        assert_eq!(
11970            detail["first_undelivered_cursor"]["daemon_incarnation"],
11971            "lag-incarnation"
11972        );
11973    }
11974}
11975
11976#[cfg(test)]
11977mod terminal_history_read_concurrency_tests {
11978    use super::*;
11979    use crate::terminal_journal::read_pause;
11980    use std::sync::mpsc as std_mpsc;
11981    use subc_test_support::TestTempDir;
11982
11983    fn journaled_ring(
11984        journal: &Arc<crate::terminal_journal::TerminalJournal>,
11985    ) -> Arc<Mutex<TerminalRing>> {
11986        Arc::new(Mutex::new(
11987            TerminalRing::new(TerminalRingConfig::default(), 1)
11988                .with_journal(Some(Arc::clone(journal))),
11989        ))
11990    }
11991
11992    fn crash(at_ms: u64) -> ExitReport {
11993        ExitReport {
11994            kind: ExitKind::Crash,
11995            code: Some(1),
11996            signal: None,
11997            at_ms,
11998        }
11999    }
12000
12001    /// Record an exit on another thread and report whether it finished within
12002    /// `bound`. The recorder thread is left running if it did not.
12003    fn record_within(
12004        module_id: &'static str,
12005        ring: &Arc<Mutex<TerminalRing>>,
12006        at_ms: u64,
12007        bound: Duration,
12008    ) -> bool {
12009        let ring = Arc::clone(ring);
12010        let (done, done_rx) = std_mpsc::channel();
12011        std::thread::spawn(move || {
12012            record_terminal(
12013                module_id,
12014                &ring,
12015                &SpawnEventFeed::default(),
12016                &crash(at_ms),
12017                TerminalDisposition::Restarting,
12018            );
12019            let _ = done.send(());
12020        });
12021        done_rx.recv_timeout(bound).is_ok()
12022    }
12023
12024    /// A history read in progress must not hold the journal writer (which every
12025    /// module's exit recording needs) or the module's own ring. Exits recorded
12026    /// while the read is paused complete promptly; the paused read answers as of
12027    /// the moment it started, and the next read has each exit exactly once.
12028    #[test]
12029    fn exits_recorded_during_a_paused_history_read_are_not_blocked_or_half_merged() {
12030        let dir = TestTempDir::new("terminal-history-concurrent-read");
12031        let path = dir.join("terminals.jsonl");
12032        let journal = Arc::new(crate::terminal_journal::TerminalJournal::open(
12033            path.clone(),
12034            "daemon".into(),
12035        ));
12036        let reader_ring = journaled_ring(&journal);
12037        let other_ring = journaled_ring(&journal);
12038        assert!(record_within(
12039            "reader-module",
12040            &reader_ring,
12041            10,
12042            Duration::from_secs(5)
12043        ));
12044
12045        let (started, release) = read_pause::install(&path);
12046        let reading = {
12047            let ring = Arc::clone(&reader_ring);
12048            std::thread::spawn(move || durable_terminal_history_of(&ring, "reader-module"))
12049        };
12050        started
12051            .recv_timeout(Duration::from_secs(5))
12052            .expect("the history read reached its pause");
12053
12054        let bound = Duration::from_secs(1);
12055        assert!(
12056            record_within("other-module", &other_ring, 20, bound),
12057            "another module's exit waited on a history read (journal writer held)"
12058        );
12059        assert!(
12060            record_within("reader-module", &reader_ring, 30, bound),
12061            "the read module's own exit waited on its history read (ring held)"
12062        );
12063
12064        drop(release);
12065        let paused = reading.join().unwrap();
12066        assert_eq!(
12067            paused.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
12068            vec![10],
12069            "an exit recorded after the read began lands in neither half of it"
12070        );
12071        assert_eq!(paused.journal_skipped_lines, 0);
12072        assert_eq!(paused.journal_read_errors, 0);
12073
12074        let after = durable_terminal_history_of(&reader_ring, "reader-module");
12075        assert_eq!(
12076            after.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
12077            vec![10, 30],
12078            "the next read merges ring and journal with no duplicate"
12079        );
12080        assert_eq!(after.journal_skipped_lines, 0);
12081    }
12082}
12083
12084/// What a restart does with the exited process's stderr reader. These drive
12085/// the same `settle_stderr_pump` the supervisor calls, with a reader the test
12086/// holds, so a reader that has not been scheduled by the bound is a controlled
12087/// input rather than something only a loaded machine produces.
12088#[cfg(test)]
12089mod stderr_settle_tests {
12090    use std::{
12091        future::Future,
12092        io,
12093        pin::Pin,
12094        sync::{Arc, Mutex},
12095        task::{Context, Poll},
12096        time::Duration,
12097    };
12098
12099    use tokio::{
12100        io::{AsyncRead, ReadBuf},
12101        sync::oneshot,
12102        time::Instant,
12103    };
12104
12105    use super::{settle_stderr_pump, StderrPump};
12106    use crate::stderr_tail::{
12107        pump_stderr_to, untimed, CaptureState, OutputSink, StderrRing, StderrTailConfig, TailEntry,
12108    };
12109
12110    const BOUND: Duration = Duration::from_millis(250);
12111
12112    /// Yields `before`, then stays pending until the gate is released, then
12113    /// yields `after` and reaches EOF. The bytes after the gate were written
12114    /// by a process that has already exited; only the reader is behind.
12115    struct HeldReader {
12116        before: Option<Vec<u8>>,
12117        gate: Option<oneshot::Receiver<()>>,
12118        after: io::Cursor<Vec<u8>>,
12119    }
12120
12121    impl AsyncRead for HeldReader {
12122        fn poll_read(
12123            mut self: Pin<&mut Self>,
12124            cx: &mut Context<'_>,
12125            buf: &mut ReadBuf<'_>,
12126        ) -> Poll<io::Result<()>> {
12127            if let Some(bytes) = self.before.take() {
12128                buf.put_slice(&bytes);
12129                return Poll::Ready(Ok(()));
12130            }
12131            if let Some(gate) = self.gate.as_mut() {
12132                match Pin::new(gate).poll(cx) {
12133                    Poll::Pending => return Poll::Pending,
12134                    Poll::Ready(_) => self.gate = None,
12135                }
12136            }
12137            Pin::new(&mut self.after).poll_read(cx, buf)
12138        }
12139    }
12140
12141    struct DiscardSink;
12142
12143    impl OutputSink for DiscardSink {
12144        fn write_line(&mut self, _line: &[u8]) {}
12145    }
12146
12147    fn line(text: &str) -> TailEntry {
12148        TailEntry::Line {
12149            text: text.to_string(),
12150            truncated: false,
12151            at_ms: None,
12152        }
12153    }
12154
12155    fn lock(ring: &Arc<Mutex<StderrRing>>) -> std::sync::MutexGuard<'_, StderrRing> {
12156        ring.lock().unwrap()
12157    }
12158
12159    /// Start a reader for a new process generation that delivers `before`
12160    /// immediately and `after` only once the returned sender fires (or is
12161    /// dropped).
12162    fn held_pump(
12163        ring: &Arc<Mutex<StderrRing>>,
12164        before: &str,
12165        after: &str,
12166    ) -> (StderrPump, oneshot::Sender<()>) {
12167        let generation = lock(ring).begin_process();
12168        let (release, gate) = oneshot::channel();
12169        let reader = HeldReader {
12170            before: Some(before.as_bytes().to_vec()),
12171            gate: Some(gate),
12172            after: io::Cursor::new(after.as_bytes().to_vec()),
12173        };
12174        let task = tokio::spawn(pump_stderr_to(
12175            reader,
12176            Arc::clone(ring),
12177            generation,
12178            DiscardSink,
12179        ));
12180        (StderrPump { task, generation }, release)
12181    }
12182
12183    async fn wait_until(ring: &Arc<Mutex<StderrRing>>, done: impl Fn(&StderrRing) -> bool) {
12184        for _ in 0..1000 {
12185            if done(&lock(ring)) {
12186                return;
12187            }
12188            tokio::time::sleep(Duration::from_millis(1)).await;
12189        }
12190        panic!(
12191            "ring never reached the expected state: {:?}",
12192            lock(ring).snapshot(None, None)
12193        );
12194    }
12195
12196    #[tokio::test(start_paused = true)]
12197    async fn a_crash_line_the_reader_had_not_reached_by_the_bound_is_kept_before_the_restart() {
12198        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
12199        let (pump, release) = held_pump(&ring, "booting\n", "config error: missing storage\n");
12200
12201        settle_stderr_pump("crasher", &ring, pump, BOUND).await;
12202        let before_release = lock(&ring).snapshot(None, None);
12203        assert!(
12204            matches!(before_release.capture, CaptureState::Incomplete { .. }),
12205            "a reader that has not reached EOF cannot claim a whole tail: {before_release:?}"
12206        );
12207
12208        // The restart: the next process starts and writes before the old
12209        // reader catches up.
12210        let next = lock(&ring).begin_process();
12211        lock(&ring).push_line_from(next, "next process booting");
12212        release.send(()).unwrap();
12213        wait_until(&ring, |ring| {
12214            ring.snapshot(None, None).capture == CaptureState::Captured
12215        })
12216        .await;
12217
12218        assert_eq!(
12219            untimed(lock(&ring).snapshot(None, None).entries),
12220            vec![
12221                line("booting"),
12222                line("config error: missing storage"),
12223                TailEntry::ProcessStart,
12224                line("next process booting"),
12225            ],
12226            "the crash's last line must survive a slow reader and stay in the crashed process's section"
12227        );
12228    }
12229
12230    #[tokio::test(start_paused = true)]
12231    async fn a_pipe_held_open_by_a_descendant_reads_incomplete_without_delaying_the_restart_past_the_bound(
12232    ) {
12233        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
12234        // `_held` is never fired: a descendant keeps the pipe open for the
12235        // whole test.
12236        let (pump, _held) = held_pump(&ring, "parent exiting\n", "");
12237
12238        let started = Instant::now();
12239        settle_stderr_pump("orphaning", &ring, pump, BOUND).await;
12240        assert_eq!(
12241            started.elapsed(),
12242            BOUND,
12243            "the restart must wait exactly the bound for a pipe that stays open, no longer"
12244        );
12245
12246        let next = lock(&ring).begin_process();
12247        lock(&ring).push_line_from(next, "next process booting");
12248        tokio::time::sleep(Duration::from_secs(60)).await;
12249
12250        let snapshot = lock(&ring).snapshot(None, None);
12251        match &snapshot.capture {
12252            CaptureState::Incomplete { reason } => assert!(
12253                reason.contains("had not reached EOF") && reason.contains("250ms"),
12254                "the reason must say what is missing and after how long: {reason}"
12255            ),
12256            other => panic!("expected Incomplete while the pipe is held open, got {other:?}"),
12257        }
12258        assert_eq!(
12259            untimed(snapshot.entries),
12260            vec![
12261                line("parent exiting"),
12262                TailEntry::ProcessStart,
12263                line("next process booting"),
12264            ]
12265        );
12266    }
12267
12268    #[tokio::test(start_paused = true)]
12269    async fn a_reader_that_reaches_eof_within_the_bound_leaves_the_tail_captured() {
12270        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
12271        let (pump, release) = held_pump(&ring, "one\n", "two\n");
12272        release.send(()).unwrap();
12273
12274        settle_stderr_pump("clean", &ring, pump, BOUND).await;
12275
12276        let snapshot = lock(&ring).snapshot(None, None);
12277        assert_eq!(snapshot.capture, CaptureState::Captured);
12278        assert_eq!(untimed(snapshot.entries), vec![line("one"), line("two")]);
12279    }
12280}
12281
12282/// Containment of a module's process tree (issue #109).
12283///
12284/// The behaviour these defend against is a module helper surviving its module:
12285/// on a real machine the Synapse embedding module's CUDA worker holds ~2.2 GB of
12286/// VRAM, so a leaked grandchild is a leaked GPU allocation, and a day of restarts
12287/// compounds it.
12288///
12289/// They run against the SUPERVISOR rather than the job-object crate because the
12290/// claim is about teardown: a crate-level test proves a job can reap a tree, not
12291/// that the daemon's drain path reaches it.
12292///
12293/// Windows-only, like the mechanism. On Unix this arm compiles out; the cgroup
12294/// lane there is a separate containment path with its own tests.
12295#[cfg(all(test, windows))]
12296mod job_containment_tests {
12297    use super::*;
12298    use std::{
12299        path::{Path, PathBuf},
12300        sync::{Arc, Mutex},
12301        time::{Duration, Instant},
12302    };
12303    use subc_test_support::TestTempDir;
12304
12305    /// The stub, expected beside this test executable.
12306    ///
12307    /// The existence check is here for the reason its twin at `fake_aft_stub_path`
12308    /// documents: `--lib` does not build `[[bin]]` targets, and a bare spawn
12309    /// failure then reads as a broken test rather than an unbuilt dependency.
12310    fn stub_path() -> PathBuf {
12311        let mut path = std::env::current_exe().expect("current_exe available in tests");
12312        path.pop();
12313        path.pop();
12314        path.push("fake-aft-stub.exe");
12315        assert!(
12316            path.exists(),
12317            "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
12318             [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
12319            path.display()
12320        );
12321        path
12322    }
12323
12324    /// Poll for the grandchild pid the stub records, and parse it.
12325    fn read_grandchild_pid(path: &Path) -> u32 {
12326        let deadline = Instant::now() + Duration::from_secs(10);
12327        loop {
12328            if let Ok(contents) = std::fs::read_to_string(path) {
12329                if let Ok(pid) = contents.trim().parse() {
12330                    return pid;
12331                }
12332            }
12333            assert!(
12334                Instant::now() < deadline,
12335                "the stub never recorded a grandchild pid at {}",
12336                path.display()
12337            );
12338            std::thread::sleep(Duration::from_millis(10));
12339        }
12340    }
12341
12342    /// Everything one fixture run needs, so the two tests below differ in exactly
12343    /// one place: whether the child is contained.
12344    struct Fixture {
12345        _dir: TestTempDir,
12346        module_id: String,
12347        grandchild: u32,
12348        child: Option<SupervisedChild>,
12349        registry: Arc<Registry>,
12350        snapshot: Arc<Mutex<SupervisorSnapshot>>,
12351        terminal_ring: Arc<Mutex<TerminalRing>>,
12352        spawn_events: SpawnEventFeed,
12353    }
12354
12355    fn fixture(label: &str, module_id: &str) -> Fixture {
12356        let dir = TestTempDir::new(label);
12357        let pid_file = dir.join("grandchild.pid");
12358        let supervisor = Supervisor::new_for_test(
12359            Arc::new(Registry::default()),
12360            RestartPolicy::new(3, Duration::ZERO),
12361        );
12362        let runtime = supervisor.runtime_config();
12363        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
12364        let spec = ModuleSpec {
12365            module_id: module_id.to_string(),
12366            program: stub_path(),
12367            // Zero args deliberately: a `--subc` argument would make the stub dial
12368            // a daemon that is not there, and the failure would land in the same
12369            // stderr ring this fixture exists to keep quiet.
12370            args: Vec::new(),
12371            env: vec![
12372                ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
12373                (
12374                    "FAKE_AFT_GRANDCHILD_PID_FILE".to_string(),
12375                    pid_file.display().to_string(),
12376                ),
12377            ],
12378            reserved: false,
12379            reserved_prefixes: Vec::new(),
12380            protocol: ModuleProtocol::Subc,
12381            overlap: Default::default(),
12382        };
12383        let child = spawn_and_mark_running(&spec, &runtime, &snapshot)
12384            .expect("spawn the supervised fixture");
12385        let grandchild = read_grandchild_pid(&pid_file);
12386        Fixture {
12387            _dir: dir,
12388            module_id: module_id.to_string(),
12389            grandchild,
12390            child: Some(child),
12391            registry: Arc::new(Registry::default()),
12392            snapshot,
12393            terminal_ring: Arc::clone(&runtime.terminal_ring),
12394            spawn_events: SpawnEventFeed::default(),
12395        }
12396    }
12397
12398    impl Fixture {
12399        /// Drain through the supervisor's own teardown path.
12400        async fn drain(&mut self) {
12401            let child = self
12402                .child
12403                .take()
12404                .expect("the fixture child is still present");
12405            drain_child_to_state(
12406                &self.module_id,
12407                ModuleProtocol::Subc,
12408                // No forwarding table in this fixture, so nothing reaches the
12409                // child over a connection.
12410                StopNotice::NotSent,
12411                &self.registry,
12412                None,
12413                &self.snapshot,
12414                &self.terminal_ring,
12415                &self.spawn_events,
12416                child,
12417                Duration::from_millis(500),
12418                ModuleState::Stopped,
12419                Some(false),
12420            )
12421            .await
12422            .expect("drain the supervised fixture");
12423        }
12424    }
12425
12426    /// Teardown reaps the grandchild, not merely the direct child.
12427    ///
12428    /// This is the assertion the change exists for. Before containment the
12429    /// grandchild survived: it is a separate process, and `start_kill` is
12430    /// `TerminateProcess` scoped to one pid.
12431    #[tokio::test]
12432    async fn teardown_reaps_the_grandchild() {
12433        let mut fixture = fixture("teardown-grandchild", "tree-teardown");
12434        let grandchild = fixture.grandchild;
12435
12436        assert!(
12437            subc_jobobject::process_exists(grandchild),
12438            "grandchild {grandchild} must be alive before teardown, or this proves nothing"
12439        );
12440
12441        fixture.drain().await;
12442
12443        assert!(
12444            subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
12445            "grandchild {grandchild} outlived module teardown: the tree was not contained"
12446        );
12447    }
12448
12449    /// The mutation control: with containment withheld, the grandchild survives
12450    /// the same kill.
12451    ///
12452    /// This is the defect reproduction from #109 — a direct-child kill reaches
12453    /// one pid, and the grandchild is a different process. It spawns OUTSIDE the
12454    /// supervisor because `spawn_and_mark_running` now always contains on
12455    /// Windows, which is the point: there is no longer a path that spawns
12456    /// uncontained, so the control has to construct one.
12457    ///
12458    /// Its job is to keep `teardown_reaps_the_grandchild` honest. If the
12459    /// grandchild ever dies here, that test is passing for a reason unrelated to
12460    /// the job object and the containment claim is unproven.
12461    #[test]
12462    fn an_uncontained_grandchild_survives_a_direct_child_kill() {
12463        let dir = TestTempDir::new("teardown-uncontained");
12464        let pid_file = dir.join("grandchild.pid");
12465        let mut child = std::process::Command::new(stub_path())
12466            .env("FAKE_AFT_NEVER_CONNECT", "1")
12467            .env(
12468                "FAKE_AFT_GRANDCHILD_PID_FILE",
12469                pid_file.display().to_string(),
12470            )
12471            .stdin(std::process::Stdio::null())
12472            .stdout(std::process::Stdio::null())
12473            .stderr(std::process::Stdio::null())
12474            .spawn()
12475            .expect("spawn the uncontained fixture");
12476        let grandchild = read_grandchild_pid(&pid_file);
12477
12478        // Exactly what the pre-fix teardown did: kill the direct child.
12479        child.kill().expect("kill the direct child");
12480        let _ = child.wait();
12481
12482        assert!(
12483            subc_jobobject::process_exists(grandchild),
12484            "grandchild {grandchild} died with the direct child, so this control no longer \
12485             distinguishes contained from uncontained teardown and the regression test is \
12486             passing vacuously"
12487        );
12488
12489        // The orphan this control demonstrates is the leak the fix prevents, so
12490        // the control must not leave one behind.
12491        kill_tree(grandchild);
12492    }
12493
12494    /// Crash durability: closing the containment handle reaps the tree with no
12495    /// teardown code running at all.
12496    ///
12497    /// This is the case `taskkill /T` cannot cover — a daemon that dies cannot
12498    /// call anything — and it is why containment is a kernel property of the
12499    /// handle rather than a step in the drain. Discovered by getting the
12500    /// mutation control wrong: clearing `job` to "disable" containment instead
12501    /// killed the tree, which is the guarantee, not a mistake.
12502    #[tokio::test]
12503    async fn dropping_containment_reaps_the_grandchild() {
12504        let mut fixture = fixture("drop-containment", "tree-drop");
12505        let grandchild = fixture.grandchild;
12506
12507        assert!(subc_jobobject::process_exists(grandchild));
12508
12509        // No `drain` call, no kill: dropping the handle is the entire mechanism.
12510        fixture.child.as_mut().expect("child present").job = None;
12511
12512        assert!(
12513            subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
12514            "grandchild {grandchild} survived the containment handle closing, so a daemon \
12515             crash would leave the tree behind"
12516        );
12517    }
12518
12519    /// Kill a pid and its tree, then confirm it is gone.
12520    fn kill_tree(pid: u32) {
12521        let _ = std::process::Command::new("taskkill.exe")
12522            .args(["/PID", &pid.to_string(), "/T", "/F"])
12523            .stdin(std::process::Stdio::null())
12524            .stdout(std::process::Stdio::null())
12525            .stderr(std::process::Stdio::null())
12526            .status();
12527        assert!(
12528            subc_jobobject::wait_for_process_exit(pid, Duration::from_secs(10)),
12529            "could not clean up grandchild {pid}"
12530        );
12531    }
12532}
12533
12534#[cfg(test)]
12535mod privacy_trampoline_configuration_tests {
12536    #[tokio::test]
12537    #[cfg_attr(not(target_os = "macos"), ignore = "requires macOS privacy trampoline")]
12538    async fn macos_spawn_without_a_configured_trampoline_refuses_by_name() {
12539        #[cfg(target_os = "macos")]
12540        {
12541            let supervisor = super::Supervisor::new(
12542                std::sync::Arc::new(crate::Registry::default()),
12543                super::RestartPolicy::default(),
12544            );
12545            let error = supervisor.spawn(spec()).unwrap_err();
12546            assert!(
12547                error
12548                    .to_string()
12549                    .contains("no privacy trampoline configured"),
12550                "{error}"
12551            );
12552        }
12553    }
12554
12555    #[tokio::test]
12556    #[cfg_attr(not(target_os = "macos"), ignore = "requires macOS privacy trampoline")]
12557    async fn macos_wrong_trampoline_is_rejected_by_the_startup_probe() {
12558        #[cfg(target_os = "macos")]
12559        {
12560            let supervisor = super::Supervisor::new(
12561                std::sync::Arc::new(crate::Registry::default()),
12562                super::RestartPolicy::default(),
12563            )
12564            .with_privacy_trampoline(std::env::current_exe().unwrap());
12565            let error = supervisor.spawn(spec()).unwrap_err();
12566            assert!(
12567                error
12568                    .to_string()
12569                    .contains("binary does not implement the privacy trampoline protocol"),
12570                "{error}"
12571            );
12572        }
12573    }
12574
12575    #[cfg(target_os = "macos")]
12576    fn spec() -> super::ModuleSpec {
12577        super::ModuleSpec {
12578            module_id: "privacy-configuration".into(),
12579            program: "/bin/sleep".into(),
12580            args: vec!["30".into()],
12581            env: vec![],
12582            reserved: false,
12583            reserved_prefixes: vec![],
12584            protocol: subc_control::ModuleProtocol::None,
12585            overlap: super::ModuleOverlap::Exclusive,
12586        }
12587    }
12588}
12589
12590/// The daemon's real spawn path hands a subc-wire child its launch nonce on
12591/// descriptor 3, without an environment copy. The shell records the nonce
12592/// and its environment after exec so these tests observe the real handover.
12593#[cfg(all(test, unix))]
12594mod launch_nonce_descriptor_tests {
12595    use super::{ChildRoster, ModuleProtocol, ModuleSpec, SupervisorHandle};
12596    use crate::stderr_tail::{StderrRing, StderrTailConfig};
12597    use std::{
12598        path::PathBuf,
12599        sync::{Arc, Mutex},
12600        time::{Duration, Instant},
12601    };
12602    use subc_test_support::TestTempDir;
12603
12604    async fn probe(role: super::SpawnRole) {
12605        let scratch = TestTempDir::new("launch-nonce-descriptor");
12606        let fd_copy = scratch.join("from-descriptor");
12607        let env_copy = scratch.join("environment");
12608        let script = format!(
12609            "env > '{env}.tmp' && mv '{env}.tmp' '{env}'; cat <&3 > '{fd}.tmp' && mv '{fd}.tmp' '{fd}'; sleep 30",
12610            fd = fd_copy.display(), env = env_copy.display(),
12611        );
12612        let xdg = |name: &str| (name.to_string(), scratch.join(name).display().to_string());
12613        let spec = ModuleSpec {
12614            module_id: "nonce-descriptor-probe".to_string(),
12615            program: PathBuf::from("/bin/sh"),
12616            args: vec!["-c".to_string(), script],
12617            env: vec![
12618                xdg("XDG_DATA_HOME"),
12619                xdg("XDG_RUNTIME_DIR"),
12620                xdg("XDG_CONFIG_HOME"),
12621                ("SUBC_LAUNCH_NONCE".to_string(), "must-not-leak".to_string()),
12622            ],
12623            reserved: true,
12624            reserved_prefixes: Vec::new(),
12625            protocol: ModuleProtocol::Subc,
12626            overlap: Default::default(),
12627        };
12628        let handle = SupervisorHandle::new();
12629        let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
12630        let roster = ChildRoster::default();
12631        #[cfg(target_os = "macos")]
12632        {
12633            let path = super::test_privacy_trampoline();
12634            roster.set_privacy_trampoline(super::probe_privacy_trampoline(&path).map(|()| path));
12635        }
12636        let child = super::spawn_child_in_slot(
12637            &spec,
12638            None,
12639            Some(&handle),
12640            &ring,
12641            None,
12642            &roster,
12643            #[cfg(target_os = "linux")]
12644            None,
12645            role,
12646            matches!(role, super::SpawnRole::SwapCandidate),
12647        )
12648        .expect("spawn probe");
12649        let deadline = Instant::now() + Duration::from_secs(10);
12650        while !(fd_copy.exists() && env_copy.exists()) {
12651            assert!(Instant::now() < deadline, "probe never wrote its copies");
12652            tokio::time::sleep(Duration::from_millis(20)).await;
12653        }
12654        let nonce = std::fs::read_to_string(fd_copy).unwrap();
12655        assert!(!nonce.is_empty());
12656        let environment = std::fs::read_to_string(env_copy).unwrap();
12657        assert!(environment
12658            .lines()
12659            .any(|line| line.starts_with("SUBC_LAUNCH_NONCE_FD=3:")));
12660        let copy = environment
12661            .lines()
12662            .find_map(|line| line.strip_prefix("SUBC_LAUNCH_NONCE="));
12663        assert_eq!(
12664            copy, None,
12665            "Unix children must never receive the environment nonce"
12666        );
12667        if matches!(role, super::SpawnRole::Plain) {
12668            assert_eq!(
12669                handle.spawn_nonce(&spec.module_id).as_deref(),
12670                Some(nonce.as_str())
12671            );
12672        }
12673        drop(child);
12674    }
12675
12676    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
12677    async fn a_spawned_module_receives_its_nonce_only_on_descriptor_3() {
12678        probe(super::SpawnRole::Plain).await;
12679    }
12680
12681    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
12682    async fn swap_candidate_receives_its_nonce_only_on_descriptor_3() {
12683        probe(super::SpawnRole::SwapCandidate).await;
12684    }
12685}
12686
12687#[cfg(all(test, target_os = "linux"))]
12688mod cgroup_containment_tests {
12689    use super::*;
12690    use subc_test_support::TestTempDir;
12691
12692    fn running(pid: u32) -> bool {
12693        // An orphan can remain a zombie until the container init reaps it.
12694        std::fs::read_to_string(format!("/proc/{pid}/stat"))
12695            .ok()
12696            .and_then(|stat| {
12697                stat.rsplit_once(") ")
12698                    .map(|(_, rest)| rest.starts_with('Z'))
12699            })
12700            .is_some_and(|zombie| !zombie)
12701    }
12702
12703    #[tokio::test]
12704    async fn linux_teardown_reaps_the_grandchild() {
12705        teardown_tree("linux_teardown_reaps_the_grandchild", false).await;
12706    }
12707
12708    #[tokio::test]
12709    async fn linux_shutdown_straggler_reaps_the_grandchild() {
12710        teardown_tree("linux_shutdown_straggler_reaps_the_grandchild", true).await;
12711    }
12712
12713    async fn teardown_tree(test_name: &str, shutdown: bool) {
12714        let dir = TestTempDir::new(test_name);
12715        let root = PathBuf::from(format!(
12716            "/sys/fs/cgroup/subc-tree-test-{test_name}-{}-{}",
12717            std::process::id(),
12718            unix_ms_now()
12719        ));
12720        if let Err(error) = std::fs::create_dir(&root) {
12721            assert!(
12722                std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
12723                "required cgroup test cannot execute: {error}"
12724            );
12725            eprintln!(
12726                "SKIP {test_name}: no writable delegated cgroup at {}: {error}",
12727                root.display()
12728            );
12729            return;
12730        }
12731        let placement = subc_cgroup::prepare_at(&root)
12732            .expect("prepare isolated kernel cgroup")
12733            .expect("isolated cgroup is delegated");
12734        let module_id = "tree-teardown";
12735        let module = placement
12736            .module_path(module_id)
12737            .expect("create isolated module cgroup");
12738        if !module.join("cgroup.kill").exists() {
12739            std::fs::remove_dir(&module).unwrap();
12740            std::fs::remove_dir(root.join("subc-modules")).unwrap();
12741            std::fs::remove_dir(&root).unwrap();
12742            assert!(
12743                std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
12744                "required cgroup.kill interface unavailable"
12745            );
12746            eprintln!("SKIP {test_name}: cgroup.kill unavailable (kernel < 5.14)");
12747            return;
12748        }
12749        let supervisor = Supervisor::new_for_test(
12750            Arc::new(Registry::default()),
12751            RestartPolicy::new(3, Duration::ZERO),
12752        )
12753        .with_cgroup_placement(Some(placement));
12754        let mut runtime = supervisor.runtime_config();
12755        runtime.child_roster = runtime
12756            .child_roster
12757            .for_module(Arc::new(Mutex::new(Duration::from_millis(100))));
12758        let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
12759        let pid_file = dir.join("grandchild.pid");
12760        let spec = ModuleSpec {
12761            module_id: module_id.to_string(),
12762            program: PathBuf::from("/bin/sh"),
12763            args: vec![
12764                "-c".into(),
12765                "trap '' TERM; sleep 600 & echo $! > \"$1\"; wait".into(),
12766                "fixture".into(),
12767                pid_file.display().to_string(),
12768            ],
12769            env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
12770                .into_iter()
12771                .map(|key| (key.to_string(), dir.display().to_string()))
12772                .collect(),
12773            reserved: false,
12774            reserved_prefixes: Vec::new(),
12775            protocol: ModuleProtocol::None,
12776            overlap: Default::default(),
12777        };
12778        let child =
12779            spawn_and_mark_running(&spec, &runtime, &snapshot).expect("spawn supervised tree");
12780        let deadline = tokio::time::Instant::now() + Duration::from_secs(10);
12781        let grandchild: u32 = loop {
12782            if let Ok(contents) = std::fs::read_to_string(&pid_file) {
12783                if let Ok(pid) = contents.trim().parse() {
12784                    break pid;
12785                }
12786            }
12787            assert!(
12788                tokio::time::Instant::now() < deadline,
12789                "grandchild pid was not recorded"
12790            );
12791            tokio::time::sleep(Duration::from_millis(10)).await;
12792        };
12793        assert!(
12794            running(grandchild),
12795            "grandchild must be alive before teardown"
12796        );
12797        if shutdown {
12798            let mut child = child;
12799            crate::child_roster::end_children_for_daemon_shutdown(
12800                &runtime.child_roster,
12801                false,
12802                std::future::pending(),
12803            )
12804            .await;
12805            child.wait().await.expect("reap shutdown straggler");
12806        } else {
12807            drain_child_to_state(
12808                module_id,
12809                ModuleProtocol::None,
12810                StopNotice::NotSent,
12811                &Registry::default(),
12812                None,
12813                &snapshot,
12814                &runtime.terminal_ring,
12815                &SpawnEventFeed::default(),
12816                child,
12817                Duration::from_millis(100),
12818                ModuleState::Stopped,
12819                Some(false),
12820            )
12821            .await
12822            .expect("real supervisor teardown");
12823        }
12824        let deadline = tokio::time::Instant::now() + Duration::from_secs(5);
12825        while running(grandchild) && tokio::time::Instant::now() < deadline {
12826            tokio::time::sleep(Duration::from_millis(10)).await;
12827        }
12828        let survived = running(grandchild);
12829        // Kill a surviving grandchild so a failed test does not leave it behind.
12830        if survived {
12831            let pid = rustix::process::Pid::from_raw(grandchild as i32).unwrap();
12832            let _ = rustix::process::kill_process(pid, rustix::process::Signal::KILL);
12833            tokio::time::sleep(Duration::from_millis(100)).await;
12834        }
12835        if module.exists() {
12836            std::fs::remove_dir(&module).expect("remove empty module cgroup");
12837        }
12838        std::fs::remove_dir(root.join("subc-modules")).unwrap();
12839        std::fs::remove_dir(&root).unwrap();
12840        assert!(
12841            !survived,
12842            "grandchild {grandchild} outlived module teardown"
12843        );
12844        eprintln!("EXECUTED {test_name}: grandchild {grandchild} killed in isolated cgroup");
12845    }
12846}