1use std::{
2 collections::{HashMap, HashSet, VecDeque},
3 error::Error,
4 fmt, io,
5 path::PathBuf,
6 process::{ExitStatus, Stdio},
7 sync::{Arc, Mutex, OnceLock},
8 time::{Duration, SystemTime, UNIX_EPOCH},
9};
10
11use cortexkit_log::Retention;
12use serde_json::Value;
13use subc_control::{
14 ClientControlPush, LiveSpawn, ModuleProtocol, RouteCloseReason, SpawnCursor, SpawnEvent,
15 SpawnEventKind, SpawnSnapshot, SupervisorHealthStatus, TerminalDisposition, TerminalExitKind,
16};
17use subc_protocol::{
18 manifest::{SelfSignalKind, SignalAnchor},
19 session::{
20 HealthReport, HealthStatus, ModuleControlCommand, ModuleControlRequest,
21 MODULE_CONTROL_OP_HEALTH_CHECK,
22 },
23 Flags, FrameType, Priority, SUBC_LAUNCH_NONCE_ENV, SUBC_MODULE_ID_ENV,
24};
25use tokio::{
26 process::{Child, Command},
27 sync::{mpsc, oneshot, watch, Mutex as AsyncMutex},
28 task::JoinHandle,
29 time::{sleep, sleep_until, timeout, timeout_at, Instant},
30};
31use tracing::{debug, error, info, warn};
32
33use crate::{
34 child_roster::ChildRoster,
35 daemon_config::{
36 CAPTURE_KEEP_ENV, CAPTURE_MAX_AGE_DAYS_ENV, CAPTURE_MAX_FILE_MB_ENV, CK_LOG_ENV,
37 },
38 forwarding::{
39 CloseReason, ForwardingError, ForwardingTable, GoodbyeTarget, ModuleControlRpcOutcome,
40 ModuleDrainTarget, PendingModuleControlRpc,
41 },
42 provenance::{spawned_file_identity, ExecutableIdentityProbe, SpawnedFileIdentity},
43 registry::{ConnectionId, RegistryError},
44 stderr_tail::{
45 pump_stderr_to, pump_stdout_to, ChildOutputSink, StderrRing, StderrTailConfig,
46 StderrTailSnapshot,
47 },
48 terminal_ring::{TerminalHistorySnapshot, TerminalRecord, TerminalRing, TerminalRingConfig},
49 Frame, FrameSink, Registry,
50};
51
52#[path = "supervise_swap.rs"]
53mod swap;
54
55pub const SUBC_ARG: &str = "--subc";
61
62const DEFAULT_MAX_RESTARTS: u32 = 3;
63const DEFAULT_BACKOFF: Duration = Duration::from_millis(100);
64const DEFAULT_MAX_BACKOFF: Duration = Duration::from_secs(30);
65const DEFAULT_RESTART_WINDOW: Duration = Duration::from_secs(600);
69pub const DEFAULT_DRAIN_TIMEOUT: Duration = Duration::from_secs(30);
80const REGISTRY_RELEASE_TIMEOUT: Duration = Duration::from_secs(1);
81const REGISTRY_RELEASE_POLL: Duration = Duration::from_millis(10);
82const STDERR_PUMP_DRAIN_TIMEOUT: Duration = Duration::from_millis(250);
104pub const SPAWN_EVENT_RING_CAPACITY: usize = 4096;
106const SPAWN_SUBSCRIBER_BUFFER: usize = SPAWN_EVENT_RING_CAPACITY + 1;
107pub(crate) const SPAWN_SUBSCRIBER_LAGGED_CODE: &str = "spawn_subscriber_lagged";
112
113#[cfg(target_os = "macos")]
114struct PrivacyExec {
115 reader: tokio::io::unix::AsyncFd<std::io::PipeReader>,
116 deadline: tokio::time::Instant,
117 expected: Option<subc_os::FileIdentity>,
118 trampoline: Option<subc_os::FileIdentity>,
119 script: bool,
120 module_id: String,
121}
122
123#[cfg(all(test, target_os = "macos"))]
124pub(crate) fn test_privacy_trampoline() -> PathBuf {
125 let path = std::env::current_exe()
128 .unwrap()
129 .parent()
130 .unwrap()
131 .parent()
132 .unwrap()
133 .join("privacy-trampoline-fixture");
134 assert!(
138 path.exists(),
139 "privacy-trampoline-fixture not built at {}: run `cargo build -p subc-daemon \
140 --bins --features test-support` or `cargo test -p subc-daemon` first",
141 path.display()
142 );
143 path
144}
145
146#[cfg(target_os = "macos")]
147fn probe_privacy_trampoline(path: &std::path::Path) -> Result<(), String> {
148 use std::io::Read;
149 let mut probe = std::process::Command::new(path)
150 .args(["__disclaim-exec", "--probe"])
151 .stdin(Stdio::null())
152 .stdout(Stdio::piped())
153 .stderr(Stdio::piped())
154 .spawn()
155 .map_err(|error| {
156 format!(
157 "privacy trampoline probe failed for {}: {error}",
158 path.display()
159 )
160 })?;
161 let deadline = std::time::Instant::now() + Duration::from_secs(5);
162 let status = loop {
163 match probe.try_wait() {
164 Ok(Some(status)) => break status,
165 Ok(None) if std::time::Instant::now() < deadline => {
166 std::thread::sleep(Duration::from_millis(5))
167 }
168 result => {
169 let _ = probe.kill();
170 let _ = probe.wait();
171 return Err(format!(
172 "privacy trampoline probe failed or timed out for {}: {result:?}",
173 path.display()
174 ));
175 }
176 }
177 };
178 let mut answer = String::new();
179 if let Some(stdout) = probe.stdout.take() {
180 let _ = stdout.take(256).read_to_string(&mut answer);
181 }
182 if status.success() && answer.trim() == subc_os::privacy_identity::TRAMPOLINE_PROBE {
183 return Ok(());
184 }
185 let mut diagnostic = String::new();
186 if let Some(stderr) = probe.stderr.take() {
187 let _ = stderr.take(1024).read_to_string(&mut diagnostic);
188 }
189 let cause = diagnostic
190 .trim()
191 .strip_prefix("ck-subc: own privacy identity refused: ")
192 .unwrap_or("binary does not implement the privacy trampoline protocol");
193 Err(format!(
194 "{cause}: probe of {} exited {status}",
195 path.display()
196 ))
197}
198
199#[cfg(target_os = "macos")]
200fn privacy_command(
201 spec: &ModuleSpec,
202 roster: &ChildRoster,
203) -> Result<
204 (
205 Command,
206 Option<PrivacyExec>,
207 subc_os::privacy_identity::ExecAcknowledgement,
208 ),
209 SuperviseError,
210> {
211 let failure = |cause: String| {
212 warn!(module_id = %spec.module_id, %cause, "privacy identity trampoline refused module spawn");
213 SuperviseError::Spawn {
214 program: spec.program.clone(),
215 source: io::Error::other(cause),
216 cgroup_path: None,
217 }
218 };
219 let trampoline = roster.privacy_trampoline().map_err(failure)?;
220 let program = if spec.program.components().count() == 1 && !spec.program.is_absolute() {
225 let path = spec
226 .env
227 .iter()
228 .find(|(key, _)| key == "PATH")
229 .map(|(_, value)| std::ffi::OsString::from(value))
230 .or_else(|| std::env::var_os("PATH"))
231 .unwrap_or_else(|| "/usr/bin:/bin".into());
232 std::env::split_paths(&path)
233 .map(|dir| dir.join(&spec.program))
234 .find(|path| path.is_file())
235 .unwrap_or_else(|| spec.program.clone())
236 } else {
237 spec.program.clone()
238 };
239 let expected = subc_os::file_identity(&program);
240 let trampoline_image = subc_os::file_identity(&trampoline);
241 let script = {
242 use std::io::Read;
243 let mut prefix = [0u8; 2];
244 std::fs::File::open(&program)
245 .is_ok_and(|mut file| file.read_exact(&mut prefix).is_ok() && &prefix == b"#!")
246 };
247 if expected.is_none() || expected == subc_os::file_identity(&trampoline) {
248 return Err(failure(
249 "privacy identity module executable is missing or is the trampoline itself".to_string(),
250 ));
251 }
252 let (reader, ack) = subc_os::privacy_identity::ExecAcknowledgement::pipe()
253 .map_err(|error| failure(error.to_string()))?;
254 let reader =
255 tokio::io::unix::AsyncFd::new(reader).map_err(|error| failure(error.to_string()))?;
256 let mut command = Command::new(&trampoline);
257 command
258 .arg("__disclaim-exec")
259 .arg(ack.fd().to_string())
260 .arg(&program);
261 ack.install(command.as_std_mut());
262 Ok((
263 command,
264 Some(PrivacyExec {
265 reader,
266 deadline: tokio::time::Instant::now() + Duration::from_secs(5),
267 expected,
268 trampoline: trampoline_image,
269 script,
270 module_id: spec.module_id.clone(),
271 }),
272 ack,
273 ))
274}
275
276struct SupervisedChild {
277 child: Child,
278 #[cfg(target_os = "macos")]
279 privacy_exec: Option<PrivacyExec>,
280 #[cfg(target_os = "macos")]
287 report_ready: Arc<OnceLock<()>>,
288 spawn_failure: Option<String>,
290 protocol: ModuleProtocol,
295 #[cfg(target_os = "linux")]
301 module_id: String,
302 #[cfg(target_os = "linux")]
303 cgroup_placement: Option<subc_cgroup::Placement>,
304 #[cfg(windows)]
326 job: Option<subc_jobobject::JobObject>,
327 stdout_pump: Option<JoinHandle<()>>,
328 stderr_pump: Option<StderrPump>,
329 stderr_ring: Arc<Mutex<StderrRing>>,
330 spawned_at_ms: u64,
331 spawned_from: PathBuf,
332 spawned_file_identity: Option<SpawnedFileIdentity>,
333 process_start_time: Option<u64>,
334 process_identity: Option<ProcessIdentity>,
335 pid: u32,
336 roster_guard: Option<crate::child_roster::RosterGuard>,
339}
340
341impl SupervisedChild {
342 fn id(&self) -> Option<u32> {
343 Some(self.pid)
344 }
345
346 fn process_identity(&self) -> Option<ProcessIdentity> {
347 self.process_identity
348 }
349
350 async fn wait(&mut self) -> io::Result<ExitStatus> {
351 #[cfg(target_os = "macos")]
352 self.confirm_privacy_exec().await;
353 let result = self.child.wait().await;
361 #[cfg(target_os = "linux")]
362 if result.is_ok() {
363 if let Some(placement) = self.cgroup_placement.as_ref() {
364 cleanup_reaped_cgroup(placement, &self.module_id).await;
365 self.cgroup_placement = None;
368 }
369 }
370 result
371 }
372
373 #[cfg(target_os = "macos")]
374 async fn confirm_privacy_exec(&mut self) {
375 let Some(pending) = &mut self.privacy_exec else {
376 return;
377 };
378 let result = tokio::time::timeout_at(pending.deadline, async {
379 let mut record = Vec::new();
380 loop {
381 let mut ready = pending.reader.readable().await?;
382 let read = ready.try_io(|reader| {
383 use std::io::Read;
384 let mut reader = reader.get_ref();
385 let mut buffer = [0u8; 256];
386 reader.read(&mut buffer).map(|count| (count, buffer))
387 });
388 match read {
389 Ok(Ok((0, _))) => return Ok::<_, io::Error>(record),
390 Ok(Ok((count, buffer))) => {
391 if record.len() + count > 1024 {
392 return Err(io::Error::other(
393 "privacy exec refusal record is too long",
394 ));
395 }
396 record.extend_from_slice(&buffer[..count]);
397 }
398 Ok(Err(error)) => return Err(error),
399 Err(_) => continue,
400 }
401 }
402 })
403 .await;
404 let pending = self.privacy_exec.as_ref().expect("pending exec");
407 let cause = match result {
408 Err(_) => Some("privacy identity trampoline did not exec within 5s".to_string()),
409 Ok(Err(error)) => Some(format!(
410 "privacy identity exec acknowledgement failed: {error}"
411 )),
412 Ok(Ok(record)) if !record.is_empty() => Some(
413 std::str::from_utf8(&record)
414 .ok()
415 .and_then(|record| {
416 record
417 .trim()
418 .strip_prefix(subc_os::privacy_identity::EXEC_REFUSAL_TAG)
419 })
420 .filter(|cause| !cause.is_empty())
421 .unwrap_or("invalid privacy exec refusal record")
422 .to_string(),
423 ),
424 Ok(Ok(_)) => match self.child.try_wait() {
425 Ok(Some(_status)) => None,
429 Err(error) => Some(format!("privacy identity trampoline wait failed: {error}")),
430 Ok(None) => {
431 let image = observe_spawned_image(self.pid);
432 if let Some(image) = image.filter(|image| {
433 image.executable.is_some()
434 && image.executable != pending.trampoline
435 && (image.executable == pending.expected || pending.script)
436 }) {
437 if let Some(guard) = &self.roster_guard {
438 guard.confirm_executable(image);
439 }
440 let _ = self.report_ready.set(());
441 info!(module_id = %pending.module_id, pid = self.pid,
442 "module spawned with own privacy identity (responsibility disclaimed)");
443 None
444 } else if image.is_none()
445 || image.is_some_and(|image| image.executable.is_none())
446 {
447 match tokio::time::timeout_at(pending.deadline, self.child.wait()).await {
454 Ok(Ok(_status)) => None,
455 Ok(Err(error)) => Some(format!("privacy identity trampoline wait failed: {error}")),
456 Err(_) => Some("privacy identity module executable remained unreadable until the 5s exec deadline".to_string()),
457 }
458 } else {
459 Some("privacy identity trampoline executable mismatch: expected the module program, not ck-subc".to_string())
460 }
461 }
462 },
463 };
464 let pending = self.privacy_exec.take().expect("pending exec");
465 if let Some(cause) = cause {
466 warn!(module_id = %pending.module_id, pid = self.pid, %cause, "privacy identity trampoline refused module spawn");
467 self.spawn_failure = Some(cause);
468 if let Some(pid) = rustix::process::Pid::from_raw(self.pid as i32) {
471 let _ = rustix::process::kill_process_group(pid, rustix::process::Signal::KILL);
472 }
473 let _ = self.child.start_kill();
474 }
475 }
476
477 fn release_roster(&mut self) {
481 self.roster_guard = None;
482 }
483
484 fn start_kill(&mut self) -> io::Result<()> {
497 #[cfg(windows)]
498 if let Some(job) = &self.job {
499 if let Err(error) = job.terminate() {
500 debug!(
501 error = %error,
502 "job termination failed; the direct-child kill still owns the outcome"
503 );
504 }
505 }
506 #[cfg(target_os = "linux")]
507 kill_module_cgroup(self.cgroup_placement.as_ref(), &self.module_id);
508 self.child.start_kill()
509 }
510
511 async fn drain_stderr(&mut self, module_id: &str) {
512 if let Some(mut pump) = self.stdout_pump.take() {
513 match timeout(STDERR_PUMP_DRAIN_TIMEOUT, &mut pump).await {
514 Ok(Ok(())) => {}
515 Ok(Err(error)) => {
516 warn!(module_id, error = %error, "stdout pump ended unexpectedly");
517 }
518 Err(_) => {
519 pump.abort();
520 warn!(
521 module_id,
522 waited = ?STDERR_PUMP_DRAIN_TIMEOUT,
523 "stdout pump did not drain before restart; stopped it before the next process"
524 );
525 }
526 }
527 }
528
529 let Some(pump) = self.stderr_pump.take() else {
530 return;
531 };
532 settle_stderr_pump(
533 module_id,
534 &self.stderr_ring,
535 pump,
536 STDERR_PUMP_DRAIN_TIMEOUT,
537 )
538 .await;
539 }
540}
541
542struct StderrPump {
545 task: JoinHandle<()>,
546 generation: u64,
547}
548
549async fn settle_stderr_pump(
555 module_id: &str,
556 ring: &Arc<Mutex<StderrRing>>,
557 pump: StderrPump,
558 bound: Duration,
559) {
560 let lock = || ring.lock().unwrap_or_else(|poisoned| poisoned.into_inner());
561 let StderrPump {
562 mut task,
563 generation,
564 } = pump;
565 lock().retire_pump(generation);
566 match timeout(bound, &mut task).await {
567 Ok(Ok(())) => {}
568 Ok(Err(err)) => {
569 let mut ring = lock();
570 ring.mark_incomplete_from(generation, format!("stderr pump ended unexpectedly: {err}"));
571 ring.finish_pump(generation);
572 warn!(module_id, error = %err, "stderr pump ended before clean EOF");
573 }
574 Err(_) => {
575 drop(task);
577 lock().mark_pump_late(
578 generation,
579 format!(
580 "stderr of the exited process had not reached EOF {bound:?} after it was \
581 retired (a descendant may still hold the pipe open); lines it still \
582 writes are kept in that process's section"
583 ),
584 );
585 warn!(
586 module_id,
587 waited = ?bound,
588 "stderr pipe of the exited process is still open; its reader keeps running without delaying the restart"
589 );
590 }
591 }
592}
593
594fn registration_release_events() -> &'static watch::Sender<u64> {
595 static EVENTS: OnceLock<watch::Sender<u64>> = OnceLock::new();
596 EVENTS.get_or_init(|| {
597 let (sender, _receiver) = watch::channel(0);
598 sender
599 })
600}
601
602pub(crate) fn notify_registration_release() {
603 let events = registration_release_events();
604 let next_generation = (*events.borrow()).wrapping_add(1);
605 events.send_replace(next_generation);
606}
607
608#[derive(Debug, Clone, PartialEq, Eq)]
610pub struct ModuleSpec {
611 pub module_id: String,
612 pub program: PathBuf,
613 pub args: Vec<String>,
614 pub env: Vec<(String, String)>,
615 pub reserved: bool,
620 pub reserved_prefixes: Vec<String>,
625 pub protocol: ModuleProtocol,
644 pub overlap: ModuleOverlap,
649}
650
651#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
658pub enum ModuleOverlap {
659 #[default]
661 Exclusive,
662 Safe,
674}
675
676impl ModuleOverlap {
677 pub fn as_str(self) -> &'static str {
678 match self {
679 Self::Exclusive => "exclusive",
680 Self::Safe => "safe",
681 }
682 }
683}
684
685pub const SUBC_SPAWN_ROLE_ENV: &str = "SUBC_SPAWN_ROLE";
695pub const SPAWN_ROLE_SWAP_CANDIDATE: &str = "swap_candidate";
697pub const DEFAULT_SWAP_READY_TIMEOUT: Duration = Duration::from_secs(100);
702
703#[derive(Debug, Clone, Copy, PartialEq, Eq)]
721pub struct RestartPolicy {
722 pub max_restarts: u32,
723 pub backoff: Duration,
726 pub max_backoff: Duration,
728 pub window: Duration,
732}
733
734impl RestartPolicy {
735 pub fn new(max_restarts: u32, backoff: Duration) -> Self {
739 Self {
740 max_restarts,
741 backoff,
742 max_backoff: DEFAULT_MAX_BACKOFF,
743 window: DEFAULT_RESTART_WINDOW,
744 }
745 }
746
747 pub fn with_max_backoff(mut self, max_backoff: Duration) -> Self {
748 self.max_backoff = max_backoff;
749 self
750 }
751
752 pub fn with_window(mut self, window: Duration) -> Self {
753 self.window = window;
754 self
755 }
756
757 fn delay_for_restart(&self, restart_in_window: u32) -> Duration {
762 if self.backoff.is_zero() || self.max_backoff.is_zero() {
763 return Duration::ZERO;
764 }
765
766 let mut delay = self.backoff;
767 for _ in 0..restart_in_window {
768 if delay >= self.max_backoff {
769 return self.max_backoff;
770 }
771 delay = delay
772 .checked_mul(10)
773 .unwrap_or(self.max_backoff)
774 .min(self.max_backoff);
775 }
776 delay.min(self.max_backoff)
777 }
778
779 fn budget_exhausted_detail(&self) -> String {
784 format!(
785 "crash budget exhausted: max_restarts={} within window_secs={}",
786 self.max_restarts,
787 self.window.as_secs()
788 )
789 }
790}
791
792impl Default for RestartPolicy {
793 fn default() -> Self {
794 Self {
795 max_restarts: DEFAULT_MAX_RESTARTS,
796 backoff: DEFAULT_BACKOFF,
797 max_backoff: DEFAULT_MAX_BACKOFF,
798 window: DEFAULT_RESTART_WINDOW,
799 }
800 }
801}
802
803#[derive(Debug, Clone, Copy, PartialEq, Eq)]
804struct CrashRestartSchedule {
805 restart_in_window: u32,
806 delay: Duration,
807}
808
809fn daemon_will_restart(
816 state: &mut SupervisorSnapshot,
817 policy: &RestartPolicy,
818 now: Instant,
819) -> bool {
820 state.enabled && state.crash_restarts_in_window(policy.window, now) < policy.max_restarts
821}
822
823const DEFAULT_HEALTH_CADENCE: Duration = Duration::from_secs(30);
824const DEFAULT_HEALTH_DEADLINE: Duration = Duration::from_secs(5);
825const DEFAULT_HEALTH_FAILURE_THRESHOLD: u32 = 3;
826const MAX_HEALTH_METRICS_BYTES: usize = 16 * 1024;
827
828#[derive(Debug, Clone, Copy, PartialEq, Eq)]
829pub enum HealthAction {
830 Report,
831 Restart,
832 Alert,
833}
834
835impl fmt::Display for HealthAction {
836 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
837 f.write_str(match self {
838 Self::Report => "report",
839 Self::Restart => "restart",
840 Self::Alert => "alert",
841 })
842 }
843}
844
845#[derive(Debug, Clone, PartialEq, Eq)]
846pub struct HealthConfig {
847 pub http: Option<String>,
851 pub cadence: Duration,
852 pub deadline: Duration,
853 pub failure_threshold: u32,
854 pub on_degraded: HealthAction,
855 pub on_failing: HealthAction,
856 pub critical: bool,
857}
858
859impl Default for HealthConfig {
860 fn default() -> Self {
861 Self {
862 http: None,
863 cadence: DEFAULT_HEALTH_CADENCE,
864 deadline: DEFAULT_HEALTH_DEADLINE,
865 failure_threshold: DEFAULT_HEALTH_FAILURE_THRESHOLD,
866 on_degraded: HealthAction::Report,
867 on_failing: HealthAction::Report,
868 critical: false,
869 }
870 }
871}
872
873#[derive(Debug, Clone, PartialEq)]
891pub struct ModuleHealthStatus {
892 pub status: SupervisorHealthStatus,
893 pub last_probe_ms: Option<u64>,
894 pub detail: Option<String>,
895 pub metrics: Option<Value>,
896 pub consecutive_failures: u32,
897 pub late_answer_count: u64,
900 pub last_late_answer_latency_ms: Option<u64>,
902 pub last_action: Option<String>,
903 pub last_action_ms: Option<u64>,
907}
908
909impl Default for ModuleHealthStatus {
910 fn default() -> Self {
911 Self {
912 status: SupervisorHealthStatus::Unknown,
913 last_probe_ms: None,
914 detail: None,
915 metrics: None,
916 consecutive_failures: 0,
917 late_answer_count: 0,
918 last_late_answer_latency_ms: None,
919 last_action: None,
920 last_action_ms: None,
921 }
922 }
923}
924
925#[derive(Debug, Clone, Copy, PartialEq, Eq)]
927pub enum ModuleState {
928 Starting,
929 Running,
930 Unresponsive,
931 Restarting,
932 Draining,
933 Stopped,
934 Failed,
935 Disabled,
936}
937
938impl fmt::Display for ModuleState {
939 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
940 f.write_str(match self {
941 Self::Starting => "starting",
942 Self::Running => "running",
943 Self::Unresponsive => "unresponsive",
944 Self::Restarting => "restarting",
945 Self::Draining => "draining",
946 Self::Stopped => "stopped",
947 Self::Failed => "failed",
948 Self::Disabled => "disabled",
949 })
950 }
951}
952
953#[derive(Debug, Clone, Copy, PartialEq, Eq)]
955pub enum ExitKind {
956 Clean,
957 Crash,
958 DeliberateSeverance,
959}
960
961impl From<ExitKind> for TerminalExitKind {
962 fn from(kind: ExitKind) -> Self {
963 match kind {
964 ExitKind::Clean => Self::Clean,
965 ExitKind::Crash => Self::Crash,
966 ExitKind::DeliberateSeverance => Self::DeliberateSeverance,
967 }
968 }
969}
970
971#[derive(Debug, Clone, Copy, PartialEq, Eq)]
974pub(crate) struct ProcessIdentity {
975 pub(crate) pid: u32,
976 pub(crate) start_time: u64,
977}
978
979#[derive(Debug, Clone, PartialEq, Eq)]
981pub struct ExitReport {
982 pub kind: ExitKind,
983 pub code: Option<i32>,
984 pub signal: Option<i32>,
985 pub at_ms: u64,
986}
987
988#[derive(Debug, Clone, PartialEq)]
991pub struct ModuleStatus {
992 pub module_id: String,
993 pub state: ModuleState,
994 pub enabled: bool,
995 pub process_alive: bool,
996 pub registration_active: bool,
997 pub protocol: ModuleProtocol,
1002 pub live: bool,
1013 pub restart_count: u32,
1017 pub lifetime_restarts: u32,
1021 pub spawn_generation: u64,
1022 pub max_restarts: u32,
1027 pub restart_window: Duration,
1031 pub drain_timeout: Duration,
1035 pub restart_backoff: Duration,
1036 pub restart_max_backoff: Duration,
1037 pub pid: Option<u32>,
1042 pub spawned_at_ms: Option<u64>,
1043 pub spawned_from: Option<PathBuf>,
1044 pub process_start_time: Option<u64>,
1045 pub last_exit: Option<ExitReport>,
1046 pub health: ModuleHealthStatus,
1047}
1048
1049#[derive(Debug, Clone, PartialEq)]
1050struct SupervisorSnapshot {
1051 state: ModuleState,
1052 enabled: bool,
1053 process_alive: bool,
1054 spawned_protocol: Option<ModuleProtocol>,
1055 spawn_failure: Option<String>,
1056 crash_restarts: VecDeque<Instant>,
1062 lifetime_restarts: u32,
1063 spawn_generation: u64,
1072 pid: Option<u32>,
1073 #[cfg(target_os = "macos")]
1074 report_ready: Option<Arc<OnceLock<()>>>,
1075 reaped_pid: Option<u32>,
1077 respawn_pending: bool,
1079 coalesced_restart_pending: bool,
1081 spawned_at_ms: Option<u64>,
1082 spawned_from: Option<PathBuf>,
1083 spawned_file_identity: Option<SpawnedFileIdentity>,
1084 process_start_time: Option<u64>,
1085 deliberate_severance: Option<ProcessIdentity>,
1086 last_exit: Option<ExitReport>,
1087 drain_disposition_detail: Option<String>,
1089 health: ModuleHealthStatus,
1090 in_alternate_slot: bool,
1095 draining_to_replace: bool,
1102 configuration_updated_since_spawn: bool,
1108}
1109
1110impl SupervisorSnapshot {
1111 fn reported_pid(&self) -> Option<u32> {
1116 #[cfg(target_os = "macos")]
1117 if self
1118 .report_ready
1119 .as_ref()
1120 .is_some_and(|ready| ready.get().is_none())
1121 {
1122 return None;
1123 }
1124 self.pid
1125 }
1126
1127 fn starting() -> Self {
1128 Self::new(ModuleState::Starting, true)
1129 }
1130
1131 fn disabled() -> Self {
1132 Self::new(ModuleState::Disabled, false)
1133 }
1134
1135 fn failed() -> Self {
1136 Self::new(ModuleState::Failed, true)
1137 }
1138
1139 fn crash_restarts_in_window(&mut self, window: Duration, now: Instant) -> u32 {
1143 while let Some(oldest) = self.crash_restarts.front() {
1144 if now.duration_since(*oldest) > window {
1145 self.crash_restarts.pop_front();
1146 } else {
1147 break;
1148 }
1149 }
1150 u32::try_from(self.crash_restarts.len()).unwrap_or(u32::MAX)
1151 }
1152
1153 fn record_crash_restart(&mut self, policy: &RestartPolicy, now: Instant) {
1159 self.crash_restarts.push_back(now);
1160 while self.crash_restarts.len() > policy.max_restarts as usize {
1161 self.crash_restarts.pop_front();
1162 }
1163 self.lifetime_restarts += 1;
1164 }
1165
1166 fn next_crash_restart(
1170 &mut self,
1171 policy: &RestartPolicy,
1172 now: Instant,
1173 ) -> Option<CrashRestartSchedule> {
1174 let restart_in_window = self.crash_restarts_in_window(policy.window, now);
1175 if restart_in_window >= policy.max_restarts {
1176 return None;
1177 }
1178 self.record_crash_restart(policy, now);
1179 Some(CrashRestartSchedule {
1180 restart_in_window,
1181 delay: policy.delay_for_restart(restart_in_window),
1182 })
1183 }
1184
1185 fn clear_crash_restarts(&mut self) {
1190 self.crash_restarts.clear();
1191 }
1192
1193 fn new(state: ModuleState, enabled: bool) -> Self {
1194 Self {
1195 state,
1196 enabled,
1197 process_alive: false,
1198 spawned_protocol: None,
1199 spawn_failure: None,
1200 crash_restarts: VecDeque::new(),
1201 lifetime_restarts: 0,
1202 spawn_generation: 0,
1203 pid: None,
1204 #[cfg(target_os = "macos")]
1205 report_ready: None,
1206 reaped_pid: None,
1207 respawn_pending: false,
1208 coalesced_restart_pending: false,
1209 spawned_at_ms: None,
1210 spawned_from: None,
1211 spawned_file_identity: None,
1212 process_start_time: None,
1213 deliberate_severance: None,
1214 last_exit: None,
1215 drain_disposition_detail: None,
1216 health: ModuleHealthStatus::default(),
1217 in_alternate_slot: false,
1218 draining_to_replace: false,
1219 configuration_updated_since_spawn: false,
1220 }
1221 }
1222}
1223
1224type SharedSnapshot = Arc<Mutex<SupervisorSnapshot>>;
1225
1226type SpawnSubscriberKey = (ConnectionId, u64);
1227
1228#[derive(Debug)]
1229struct SpawnSubscriber {
1230 version: u8,
1231 frames: mpsc::Sender<Frame>,
1232 lagged: Option<oneshot::Sender<SpawnCursor>>,
1236}
1237
1238#[derive(Debug)]
1239struct SpawnEventState {
1240 daemon_incarnation: String,
1241 seq: u64,
1242 capacity: usize,
1243 live: HashMap<String, LiveSpawn>,
1244 generations: HashMap<String, u64>,
1245 events: VecDeque<SpawnEvent>,
1246 subscribers: HashMap<SpawnSubscriberKey, SpawnSubscriber>,
1247}
1248
1249impl Default for SpawnEventState {
1250 fn default() -> Self {
1251 Self {
1252 daemon_incarnation: "unconfigured".to_string(),
1253 seq: 0,
1254 capacity: SPAWN_EVENT_RING_CAPACITY,
1255 live: HashMap::new(),
1256 generations: HashMap::new(),
1257 events: VecDeque::new(),
1258 subscribers: HashMap::new(),
1259 }
1260 }
1261}
1262
1263#[derive(Debug, Clone, Default)]
1264struct SpawnEventFeed(Arc<Mutex<SpawnEventState>>);
1265
1266#[derive(Debug, Clone, PartialEq, Eq)]
1267pub(crate) enum SpawnSubscribeRefusal {
1268 ForeignIncarnation { current: String },
1269 TooOld { oldest: SpawnCursor },
1270 Frame(String),
1271}
1272
1273impl SpawnEventFeed {
1274 fn configure_incarnation(&self, daemon_incarnation: String) {
1275 let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1276 state.daemon_incarnation = daemon_incarnation;
1277 state.seq = 0;
1278 state.live.clear();
1279 state.generations.clear();
1280 state.events.clear();
1281 state.subscribers.clear();
1282 }
1283
1284 fn cursor(state: &SpawnEventState) -> SpawnCursor {
1285 SpawnCursor {
1286 daemon_incarnation: state.daemon_incarnation.clone(),
1287 seq: state.seq,
1288 }
1289 }
1290
1291 fn snapshot(&self) -> SpawnSnapshot {
1292 let state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1293 let mut live = state.live.values().cloned().collect::<Vec<_>>();
1294 live.sort_by(|left, right| left.module_id.cmp(&right.module_id));
1295 SpawnSnapshot {
1296 cursor: Self::cursor(&state),
1297 ring_bound: state.capacity as u64,
1298 live,
1299 }
1300 }
1301
1302 fn emit_spawned(&self, module_id: &str, pid: u32, spawned_at_ms: u64) -> u64 {
1303 let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1304 let generation = state
1305 .generations
1306 .get(module_id)
1307 .copied()
1308 .unwrap_or(0)
1309 .checked_add(1)
1310 .expect("spawn generation exhausted");
1311 state.generations.insert(module_id.to_string(), generation);
1312 let live = LiveSpawn {
1313 module_id: module_id.to_string(),
1314 spawn_generation: generation,
1315 pid,
1316 spawned_at_ms,
1317 };
1318 state.live.insert(module_id.to_string(), live);
1319 Self::emit_locked(
1320 &mut state,
1321 SpawnEventKind::Spawned,
1322 module_id.to_string(),
1323 generation,
1324 pid,
1325 None,
1326 None,
1327 );
1328 generation
1329 }
1330
1331 fn emit_exited(&self, module_id: &str, exit_code: Option<i32>, exit_signal: Option<i32>) {
1332 let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1333 let Some(live) = state.live.remove(module_id) else {
1334 warn!(
1335 module_id,
1336 "terminal record had no live spawn event identity"
1337 );
1338 return;
1339 };
1340 Self::emit_locked(
1341 &mut state,
1342 SpawnEventKind::Exited,
1343 module_id.to_string(),
1344 live.spawn_generation,
1345 live.pid,
1346 exit_code,
1347 exit_signal,
1348 );
1349 }
1350
1351 fn emit_superseded_exited(
1358 &self,
1359 module_id: &str,
1360 spawn_generation: u64,
1361 pid: u32,
1362 exit_code: Option<i32>,
1363 exit_signal: Option<i32>,
1364 ) {
1365 let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1366 if state
1367 .live
1368 .get(module_id)
1369 .is_some_and(|live| live.spawn_generation == spawn_generation)
1370 {
1371 state.live.remove(module_id);
1372 }
1373 Self::emit_locked(
1374 &mut state,
1375 SpawnEventKind::Exited,
1376 module_id.to_string(),
1377 spawn_generation,
1378 pid,
1379 exit_code,
1380 exit_signal,
1381 );
1382 }
1383
1384 #[allow(clippy::too_many_arguments)]
1385 fn emit_locked(
1386 state: &mut SpawnEventState,
1387 kind: SpawnEventKind,
1388 module_id: String,
1389 spawn_generation: u64,
1390 pid: u32,
1391 exit_code: Option<i32>,
1392 exit_signal: Option<i32>,
1393 ) {
1394 state.seq = state
1395 .seq
1396 .checked_add(1)
1397 .expect("spawn event sequence exhausted");
1398 let event = SpawnEvent {
1399 cursor: Self::cursor(state),
1400 kind,
1401 module_id,
1402 spawn_generation,
1403 pid,
1404 exit_code,
1405 exit_signal,
1406 };
1407 state.events.push_back(event.clone());
1408 while state.events.len() > state.capacity {
1409 state.events.pop_front();
1410 }
1411 let body = match serde_json::to_vec(&event) {
1412 Ok(body) => body,
1413 Err(error) => {
1414 error!(%error, "failed to serialize supervisor spawn event");
1415 return;
1416 }
1417 };
1418 state.subscribers.retain(|(connection_id, corr), subscriber| {
1419 let frame = Frame::build_with_version(
1420 subscriber.version,
1421 FrameType::StreamData,
1422 control_flags(),
1423 0,
1424 0,
1425 *corr,
1426 body.clone(),
1427 );
1428 match frame {
1429 Ok(frame) => {
1430 if subscriber.frames.try_send(frame).is_ok() {
1431 true
1432 } else {
1433 warn!(connection_id = connection_id.get(), corr, "dropping lagged supervisor spawn subscriber");
1434 if let Some(lagged) = subscriber.lagged.take() {
1435 let _ = lagged.send(event.cursor.clone());
1436 }
1437 false
1438 }
1439 }
1440 Err(error) => {
1441 warn!(connection_id = connection_id.get(), corr, %error, "dropping supervisor spawn subscriber after frame build failure");
1442 false
1443 }
1444 }
1445 });
1446 }
1447
1448 fn subscribe(
1449 &self,
1450 connection_id: ConnectionId,
1451 corr: u64,
1452 version: u8,
1453 since: Option<SpawnCursor>,
1454 sink: FrameSink,
1455 ) -> Result<(), SpawnSubscribeRefusal> {
1456 let (frames, mut receiver) = mpsc::channel(SPAWN_SUBSCRIBER_BUFFER);
1457 let (lagged, mut lagged_rx) = oneshot::channel::<SpawnCursor>();
1458 {
1459 let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1460 let replay = if let Some(since) = since {
1461 if since.daemon_incarnation != state.daemon_incarnation {
1462 return Err(SpawnSubscribeRefusal::ForeignIncarnation {
1463 current: state.daemon_incarnation.clone(),
1464 });
1465 }
1466 if let Some(oldest) = state.events.front().map(|event| event.cursor.clone()) {
1467 if since.seq < oldest.seq.saturating_sub(1) {
1468 return Err(SpawnSubscribeRefusal::TooOld { oldest });
1469 }
1470 }
1471 state
1472 .events
1473 .iter()
1474 .filter(|event| event.cursor.seq > since.seq)
1475 .cloned()
1476 .collect::<Vec<_>>()
1477 } else {
1478 Vec::new()
1479 };
1480 for event in replay {
1481 let body = serde_json::to_vec(&event)
1482 .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1483 let frame = Frame::build_with_version(
1484 version,
1485 FrameType::StreamData,
1486 control_flags(),
1487 0,
1488 0,
1489 corr,
1490 body,
1491 )
1492 .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1493 frames
1494 .try_send(frame)
1495 .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1496 }
1497 state.subscribers.insert(
1498 (connection_id, corr),
1499 SpawnSubscriber {
1500 version,
1501 frames,
1502 lagged: Some(lagged),
1503 },
1504 );
1505 }
1506 tokio::spawn(async move {
1517 while let Some(frame) = receiver.recv().await {
1518 if sink.send(frame).await.is_err() {
1519 return;
1520 }
1521 }
1522 let Ok(first_undelivered) = lagged_rx.try_recv() else {
1523 return;
1524 };
1525 match spawn_subscriber_lagged_frame(version, corr, first_undelivered) {
1526 Ok(frame) => {
1527 let _ = sink.send(frame).await;
1528 }
1529 Err(error) => {
1530 error!(%error, corr, "failed to build lagged spawn subscriber terminal frame");
1531 }
1532 }
1533 });
1534 Ok(())
1535 }
1536
1537 fn cancel(&self, connection_id: ConnectionId, corr: u64) -> bool {
1538 let Some(subscriber) = self
1539 .0
1540 .lock()
1541 .unwrap_or_else(|p| p.into_inner())
1542 .subscribers
1543 .remove(&(connection_id, corr))
1544 else {
1545 return false;
1546 };
1547 if let Ok(frame) = Frame::build_with_version(
1548 subscriber.version,
1549 FrameType::StreamEnd,
1550 control_flags(),
1551 0,
1552 0,
1553 corr,
1554 Vec::new(),
1555 ) {
1556 tokio::spawn(async move {
1557 let _ = subscriber.frames.send(frame).await;
1558 });
1559 }
1560 true
1561 }
1562
1563 fn remove_connection(&self, connection_id: ConnectionId) {
1564 self.0
1565 .lock()
1566 .unwrap_or_else(|p| p.into_inner())
1567 .subscribers
1568 .retain(|(subscriber_connection, _), _| *subscriber_connection != connection_id);
1569 }
1570
1571 #[cfg(any(test, feature = "test-support"))]
1572 fn set_capacity(&self, capacity: usize) {
1573 self.0.lock().unwrap_or_else(|p| p.into_inner()).capacity = capacity;
1574 }
1575
1576 #[cfg(any(test, feature = "test-support"))]
1577 fn subscriber_count(&self) -> usize {
1578 self.0
1579 .lock()
1580 .unwrap_or_else(|p| p.into_inner())
1581 .subscribers
1582 .len()
1583 }
1584}
1585
1586fn spawn_subscriber_lagged_frame(
1589 version: u8,
1590 corr: u64,
1591 first_undelivered: SpawnCursor,
1592) -> Result<Frame, String> {
1593 let body = serde_json::to_vec(&subc_protocol::ErrorBody {
1594 code: SPAWN_SUBSCRIBER_LAGGED_CODE.to_string(),
1595 message: "spawn subscriber fell behind and was dropped; resubscribe from the last cursor received"
1596 .to_string(),
1597 detail: Some(serde_json::json!({
1598 "first_undelivered_cursor": first_undelivered
1599 })),
1600 })
1601 .map_err(|error| error.to_string())?;
1602 Frame::build_with_version(version, FrameType::Error, control_flags(), 0, 0, corr, body)
1603 .map_err(|error| error.to_string())
1604}
1605
1606pub trait ModuleProcessLiveness: Send + Sync {
1607 fn process_live(&self, module_id: &str) -> Option<bool>;
1608
1609 fn process_replacing(&self, _module_id: &str) -> bool {
1615 false
1616 }
1617}
1618
1619#[derive(Debug, Clone, Default)]
1621pub struct SupervisorProcessLiveness {
1622 snapshots: Arc<Mutex<HashMap<String, SharedSnapshot>>>,
1623}
1624
1625impl SupervisorProcessLiveness {
1626 pub fn new() -> Self {
1627 Self::default()
1628 }
1629
1630 fn track(&self, module_id: String, snapshot: SharedSnapshot) {
1631 let mut snapshots = self
1632 .snapshots
1633 .lock()
1634 .unwrap_or_else(|poisoned| poisoned.into_inner());
1635 snapshots.insert(module_id, snapshot);
1636 }
1637
1638 fn untrack_if_current(&self, module_id: &str, snapshot: &SharedSnapshot) {
1639 let mut snapshots = self
1640 .snapshots
1641 .lock()
1642 .unwrap_or_else(|poisoned| poisoned.into_inner());
1643 let is_current = snapshots
1644 .get(module_id)
1645 .map(|tracked| Arc::ptr_eq(tracked, snapshot))
1646 .unwrap_or(false);
1647 if is_current {
1648 snapshots.remove(module_id);
1649 }
1650 }
1651}
1652
1653impl ModuleProcessLiveness for SupervisorProcessLiveness {
1654 fn process_live(&self, module_id: &str) -> Option<bool> {
1655 let snapshot = {
1656 let snapshots = self
1657 .snapshots
1658 .lock()
1659 .unwrap_or_else(|poisoned| poisoned.into_inner());
1660 snapshots.get(module_id).cloned()
1661 }?;
1662 let snapshot = snapshot
1663 .lock()
1664 .unwrap_or_else(|poisoned| poisoned.into_inner());
1665 Some(snapshot.state == ModuleState::Running && snapshot.process_alive)
1666 }
1667
1668 fn process_replacing(&self, module_id: &str) -> bool {
1669 let Some(snapshot) = self
1670 .snapshots
1671 .lock()
1672 .unwrap_or_else(|poisoned| poisoned.into_inner())
1673 .get(module_id)
1674 .cloned()
1675 else {
1676 return false;
1677 };
1678 let snapshot = snapshot
1679 .lock()
1680 .unwrap_or_else(|poisoned| poisoned.into_inner());
1681 snapshot.enabled
1682 && match snapshot.state {
1683 ModuleState::Restarting => true,
1684 ModuleState::Draining => snapshot.draining_to_replace,
1685 ModuleState::Starting
1686 | ModuleState::Running
1687 | ModuleState::Unresponsive
1688 | ModuleState::Stopped
1689 | ModuleState::Failed
1690 | ModuleState::Disabled => false,
1691 }
1692 }
1693}
1694
1695#[cfg(test)]
1696#[derive(Debug, Default)]
1697struct ReloadExitRecordGate {
1698 reached: tokio::sync::Notify,
1699 resume: tokio::sync::Notify,
1700}
1701
1702#[derive(Debug, Clone, Copy)]
1703enum RespawnKind {
1704 Spawn,
1705 Reload,
1706}
1707
1708#[derive(Debug, Clone, Copy)]
1709struct PendingRespawn {
1710 deadline: Instant,
1711 kind: RespawnKind,
1712}
1713
1714type ReloadReply = oneshot::Sender<Result<(), SuperviseError>>;
1715
1716#[derive(Debug, Clone)]
1717struct SupervisorRuntimeConfig {
1718 scheduled_respawn: Arc<Mutex<Option<PendingRespawn>>>,
1720 deferred_reload_reply: Arc<Mutex<Option<ReloadReply>>>,
1722 restart_policy: RestartPolicy,
1723 drain_timeout: Duration,
1726 effective_drain_timeout: Arc<Mutex<Duration>>,
1729 default_drain_timeout: Duration,
1732 health: HealthConfig,
1733 connection_file_path: Option<PathBuf>,
1734 capture_logs_dir: Option<PathBuf>,
1735 forwarding: Option<Arc<ForwardingTable>>,
1736 supervisor_handle: Option<SupervisorHandle>,
1739 stderr_ring: Arc<Mutex<StderrRing>>,
1746 terminal_ring: Arc<Mutex<TerminalRing>>,
1747 spawn_events: SpawnEventFeed,
1748 child_roster: ChildRoster,
1749 #[cfg(target_os = "linux")]
1750 cgroup_placement: Option<subc_cgroup::Placement>,
1751 #[cfg(test)]
1752 test_seed_stale_facts_before_enable_spawn: bool,
1753 #[cfg(test)]
1754 test_reload_exit_record_gate: Option<Arc<ReloadExitRecordGate>>,
1755}
1756
1757#[derive(Debug, Clone, PartialEq, Eq)]
1758struct SupervisedConfiguration {
1759 spec: ModuleSpec,
1760 health: HealthConfig,
1761}
1762
1763#[derive(Debug, Clone, Default)]
1769pub struct SupervisorHandle {
1770 modules: Arc<Mutex<HashMap<String, SupervisedModule>>>,
1771 configured_ids: Arc<Mutex<HashSet<String>>>,
1783 spawn_events: SpawnEventFeed,
1784 reserved_nonces: Arc<Mutex<HashMap<String, Option<String>>>>,
1795 removal_tombstones: Arc<Mutex<HashMap<String, u64>>>,
1801 spawn_nonces: Arc<Mutex<HashMap<String, String>>>,
1805 reserved_prefix_owners: Arc<Mutex<HashMap<String, String>>>,
1813 swaps: Arc<Mutex<HashMap<String, OpenSwap>>>,
1819 promotion_observer: PromotionObserverSlot,
1821 operation_lock: Arc<AsyncMutex<()>>,
1825}
1826
1827pub(crate) trait SwapPromotionObserver: Send + Sync {
1836 fn swap_promoted(&self, registration: &crate::registry::ModuleRegistration);
1837}
1838
1839#[derive(Clone, Default)]
1843struct PromotionObserverSlot(Arc<Mutex<Option<std::sync::Weak<dyn SwapPromotionObserver>>>>);
1844
1845impl fmt::Debug for PromotionObserverSlot {
1846 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1847 f.write_str("PromotionObserverSlot")
1848 }
1849}
1850
1851#[derive(Debug, Clone)]
1853struct OpenSwap {
1854 candidate_nonce: String,
1857 incumbent_nonce: Option<String>,
1862 candidate_admitted: bool,
1866}
1867
1868#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1871pub(crate) enum SwapHelloAdmission {
1872 NotSwapping,
1875 Candidate,
1877 Refused,
1880}
1881
1882#[derive(Debug, Clone, PartialEq, Eq)]
1883pub(crate) enum ReservedHelloRejection {
1884 Exact {
1885 module_id: String,
1886 },
1887 Prefix {
1888 prefix: String,
1889 owner_module_id: String,
1890 },
1891}
1892
1893impl SupervisorHandle {
1894 pub fn new() -> Self {
1895 Self::default()
1896 }
1897
1898 pub(crate) fn spawn_snapshot(&self) -> SpawnSnapshot {
1899 self.spawn_events.snapshot()
1900 }
1901
1902 pub(crate) fn subscribe_spawns(
1903 &self,
1904 connection_id: ConnectionId,
1905 corr: u64,
1906 version: u8,
1907 since: Option<SpawnCursor>,
1908 sink: FrameSink,
1909 ) -> Result<(), SpawnSubscribeRefusal> {
1910 self.spawn_events
1911 .subscribe(connection_id, corr, version, since, sink)
1912 }
1913
1914 pub(crate) fn cancel_spawn_subscription(&self, connection_id: ConnectionId, corr: u64) -> bool {
1915 self.spawn_events.cancel(connection_id, corr)
1916 }
1917
1918 pub(crate) fn remove_spawn_subscribers(&self, connection_id: ConnectionId) {
1919 self.spawn_events.remove_connection(connection_id);
1920 }
1921
1922 #[cfg(any(test, feature = "test-support"))]
1923 pub fn set_spawn_event_capacity_for_test(&self, capacity: usize) {
1924 assert!(capacity > 0, "spawn event capacity must be non-zero");
1925 self.spawn_events.set_capacity(capacity);
1926 }
1927
1928 #[cfg(any(test, feature = "test-support"))]
1929 pub fn spawn_subscriber_count_for_test(&self) -> usize {
1930 self.spawn_events.subscriber_count()
1931 }
1932
1933 pub fn set_spawn_nonce(&self, module_id: &str, nonce: String) {
1936 self.spawn_nonces
1937 .lock()
1938 .unwrap_or_else(|poisoned| poisoned.into_inner())
1939 .insert(module_id.to_string(), nonce);
1940 }
1941
1942 pub fn set_reserved_nonce(&self, module_id: &str, nonce: String) {
1945 self.reserved_nonces
1946 .lock()
1947 .unwrap_or_else(|poisoned| poisoned.into_inner())
1948 .insert(module_id.to_string(), Some(nonce));
1949 }
1950
1951 pub fn set_reserved_prefixes(&self, owner_module_id: &str, prefixes: &[String]) {
1953 let mut owners = self
1954 .reserved_prefix_owners
1955 .lock()
1956 .unwrap_or_else(|poisoned| poisoned.into_inner());
1957 owners.retain(|_, owner| owner != owner_module_id);
1958 for prefix in prefixes {
1959 owners.insert(prefix.clone(), owner_module_id.to_string());
1960 }
1961 }
1962
1963 #[cfg(test)]
1965 pub(crate) fn spawn_nonce(&self, module_id: &str) -> Option<String> {
1966 self.spawn_nonces
1967 .lock()
1968 .unwrap_or_else(|poisoned| poisoned.into_inner())
1969 .get(module_id)
1970 .cloned()
1971 }
1972
1973 fn apply_identity_configuration(&self, spec: &ModuleSpec) {
1974 self.set_reserved_prefixes(&spec.module_id, &spec.reserved_prefixes);
1975 let spawn_nonce = self
1976 .spawn_nonces
1977 .lock()
1978 .unwrap_or_else(|poisoned| poisoned.into_inner())
1979 .get(&spec.module_id)
1980 .cloned();
1981 let mut reserved_nonces = self
1982 .reserved_nonces
1983 .lock()
1984 .unwrap_or_else(|poisoned| poisoned.into_inner());
1985 if spec.reserved {
1986 reserved_nonces.insert(spec.module_id.clone(), spawn_nonce);
1991 }
1992 drop(reserved_nonces);
1993 self.removal_tombstones
1997 .lock()
1998 .unwrap_or_else(|poisoned| poisoned.into_inner())
1999 .remove(&spec.module_id);
2000 }
2001
2002 pub fn reserved_hello_authorized(&self, module_id: &str, presented: Option<&str>) -> bool {
2007 self.reserved_hello_rejection(module_id, presented)
2008 .is_none()
2009 }
2010
2011 pub(crate) fn reserved_hello_rejection(
2012 &self,
2013 module_id: &str,
2014 presented: Option<&str>,
2015 ) -> Option<ReservedHelloRejection> {
2016 let nonces = self
2017 .reserved_nonces
2018 .lock()
2019 .unwrap_or_else(|poisoned| poisoned.into_inner());
2020 if let Some(expected) = nonces.get(module_id) {
2021 let authorized = match expected {
2025 Some(expected) => {
2026 presented.is_some_and(|p| constant_time_eq(expected.as_bytes(), p.as_bytes()))
2027 }
2028 None => false,
2029 };
2030 if authorized {
2031 return None;
2032 }
2033 return Some(ReservedHelloRejection::Exact {
2034 module_id: module_id.to_string(),
2035 });
2036 }
2037 drop(nonces);
2038
2039 let matched_prefix = self
2040 .reserved_prefix_owners
2041 .lock()
2042 .unwrap_or_else(|poisoned| poisoned.into_inner())
2043 .iter()
2044 .filter(|(prefix, _)| module_id.starts_with(prefix.as_str()))
2045 .max_by_key(|(prefix, _)| prefix.len())
2046 .map(|(prefix, owner)| (prefix.clone(), owner.clone()));
2047 let (prefix, owner_module_id) = matched_prefix?;
2048
2049 let authorized = presented.is_some_and(|presented| {
2050 self.spawn_nonces
2051 .lock()
2052 .unwrap_or_else(|poisoned| poisoned.into_inner())
2053 .get(&owner_module_id)
2054 .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()))
2055 || self.swap_nonce_matches(&owner_module_id, presented)
2058 });
2059 if authorized {
2060 None
2061 } else {
2062 Some(ReservedHelloRejection::Prefix {
2063 prefix,
2064 owner_module_id,
2065 })
2066 }
2067 }
2068
2069 pub fn spawned_consumer_authorized(&self, module_id: &str, presented: &str) -> bool {
2074 if presented.is_empty() {
2075 return false;
2076 }
2077 let nonces = self
2078 .spawn_nonces
2079 .lock()
2080 .unwrap_or_else(|poisoned| poisoned.into_inner());
2081 let current = nonces
2082 .get(module_id)
2083 .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()));
2084 drop(nonces);
2085 current || self.swap_nonce_matches(module_id, presented)
2090 }
2091
2092 fn swap_nonce_matches(&self, module_id: &str, presented: &str) -> bool {
2094 let swaps = self
2095 .swaps
2096 .lock()
2097 .unwrap_or_else(|poisoned| poisoned.into_inner());
2098 swaps.get(module_id).is_some_and(|swap| {
2099 constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes())
2100 || swap.incumbent_nonce.as_deref().is_some_and(|incumbent| {
2101 constant_time_eq(incumbent.as_bytes(), presented.as_bytes())
2102 })
2103 })
2104 }
2105
2106 pub(crate) fn open_swap(&self, module_id: &str, candidate_nonce: String) {
2109 let incumbent_nonce = self
2110 .spawn_nonces
2111 .lock()
2112 .unwrap_or_else(|poisoned| poisoned.into_inner())
2113 .get(module_id)
2114 .cloned();
2115 self.swaps
2116 .lock()
2117 .unwrap_or_else(|poisoned| poisoned.into_inner())
2118 .insert(
2119 module_id.to_string(),
2120 OpenSwap {
2121 candidate_nonce,
2122 incumbent_nonce,
2123 candidate_admitted: false,
2124 },
2125 );
2126 }
2127
2128 pub(crate) fn close_swap(&self, module_id: &str) {
2131 self.swaps
2132 .lock()
2133 .unwrap_or_else(|poisoned| poisoned.into_inner())
2134 .remove(module_id);
2135 }
2136
2137 pub(crate) fn set_swap_promotion_observer(
2140 &self,
2141 observer: std::sync::Weak<dyn SwapPromotionObserver>,
2142 ) {
2143 *self
2144 .promotion_observer
2145 .0
2146 .lock()
2147 .unwrap_or_else(|poisoned| poisoned.into_inner()) = Some(observer);
2148 }
2149
2150 fn notify_swap_promoted(&self, registration: &crate::registry::ModuleRegistration) {
2153 let observer = self
2154 .promotion_observer
2155 .0
2156 .lock()
2157 .unwrap_or_else(|poisoned| poisoned.into_inner())
2158 .as_ref()
2159 .and_then(std::sync::Weak::upgrade);
2160 if let Some(observer) = observer {
2161 observer.swap_promoted(registration);
2162 }
2163 }
2164
2165 pub(crate) fn swap_open(&self, module_id: &str) -> bool {
2167 self.swaps
2168 .lock()
2169 .unwrap_or_else(|poisoned| poisoned.into_inner())
2170 .contains_key(module_id)
2171 }
2172
2173 fn promote_swap_nonce(&self, module_id: &str, reserved: bool) {
2178 let candidate_nonce = self
2179 .swaps
2180 .lock()
2181 .unwrap_or_else(|poisoned| poisoned.into_inner())
2182 .get(module_id)
2183 .map(|swap| swap.candidate_nonce.clone());
2184 let Some(nonce) = candidate_nonce else {
2185 return;
2186 };
2187 self.set_spawn_nonce(module_id, nonce.clone());
2188 if reserved {
2189 self.set_reserved_nonce(module_id, nonce);
2190 }
2191 }
2192
2193 pub(crate) fn swap_hello_admission(
2208 &self,
2209 module_id: &str,
2210 presented: Option<&str>,
2211 ) -> SwapHelloAdmission {
2212 let swaps = self
2213 .swaps
2214 .lock()
2215 .unwrap_or_else(|poisoned| poisoned.into_inner());
2216 let Some(swap) = swaps.get(module_id) else {
2217 return SwapHelloAdmission::NotSwapping;
2218 };
2219 let Some(presented) = presented else {
2220 return SwapHelloAdmission::Refused;
2221 };
2222 if constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes()) {
2223 return if swap.candidate_admitted {
2224 SwapHelloAdmission::Refused
2225 } else {
2226 SwapHelloAdmission::Candidate
2227 };
2228 }
2229 if swap
2230 .incumbent_nonce
2231 .as_deref()
2232 .is_some_and(|incumbent| constant_time_eq(incumbent.as_bytes(), presented.as_bytes()))
2233 {
2234 return SwapHelloAdmission::NotSwapping;
2235 }
2236 SwapHelloAdmission::Refused
2237 }
2238
2239 pub(crate) fn mark_swap_candidate_admitted(&self, module_id: &str) {
2242 if let Some(swap) = self
2243 .swaps
2244 .lock()
2245 .unwrap_or_else(|poisoned| poisoned.into_inner())
2246 .get_mut(module_id)
2247 {
2248 swap.candidate_admitted = true;
2249 }
2250 }
2251
2252 pub fn spawn_launch_nonce_for(&self, module_id: &str) -> Option<String> {
2254 self.spawn_nonces
2255 .lock()
2256 .unwrap_or_else(|poisoned| poisoned.into_inner())
2257 .get(module_id)
2258 .cloned()
2259 }
2260
2261 pub fn reserved_launch_nonce_for(&self, module_id: &str) -> Option<String> {
2263 self.reserved_nonces
2264 .lock()
2265 .unwrap_or_else(|poisoned| poisoned.into_inner())
2266 .get(module_id)
2267 .cloned()
2268 .flatten()
2269 }
2270
2271 pub fn insert(&self, module: SupervisedModule) -> Option<SupervisedModule> {
2272 self.mark_configured(module.module_id());
2276 let mut modules = self
2277 .modules
2278 .lock()
2279 .unwrap_or_else(|poisoned| poisoned.into_inner());
2280 modules.insert(module.module_id().to_string(), module)
2281 }
2282
2283 fn mark_configured(&self, module_id: &str) {
2287 self.configured_ids
2288 .lock()
2289 .unwrap_or_else(|poisoned| poisoned.into_inner())
2290 .insert(module_id.to_string());
2291 }
2292
2293 fn unmark_configured_unless_rostered(&self, module_id: &str) {
2297 let modules = self
2298 .modules
2299 .lock()
2300 .unwrap_or_else(|poisoned| poisoned.into_inner());
2301 if !modules.contains_key(module_id) {
2302 self.configured_ids
2303 .lock()
2304 .unwrap_or_else(|poisoned| poisoned.into_inner())
2305 .remove(module_id);
2306 }
2307 }
2308
2309 pub(crate) fn is_configured(&self, module_id: &str) -> bool {
2317 self.configured_ids
2318 .lock()
2319 .unwrap_or_else(|poisoned| poisoned.into_inner())
2320 .contains(module_id)
2321 }
2322
2323 pub fn get(&self, module_id: &str) -> Option<SupervisedModule> {
2324 let modules = self
2325 .modules
2326 .lock()
2327 .unwrap_or_else(|poisoned| poisoned.into_inner());
2328 modules.get(module_id).cloned()
2329 }
2330
2331 pub(crate) fn record_late_health_answer(
2332 &self,
2333 module_id: &str,
2334 latency_ms: u64,
2335 ) -> Result<bool, SuperviseError> {
2336 let Some(module) = self.get(module_id) else {
2337 return Ok(false);
2338 };
2339 update_snapshot(&module.inner.snapshot, Some(module_id), |state| {
2340 state.health.late_answer_count = state.health.late_answer_count.saturating_add(1);
2341 state.health.last_late_answer_latency_ms = Some(latency_ms);
2342 state.health.consecutive_failures = 0;
2350 })?;
2351 Ok(true)
2352 }
2353
2354 pub fn record_deliberate_severance(&self, module_id: &str) -> Result<bool, SuperviseError> {
2360 let Some(module) = self.get(module_id) else {
2361 return Ok(false);
2362 };
2363 let snapshot = lock_snapshot(&module.inner.snapshot)?;
2364 let Some((pid, start_time)) = snapshot.pid.zip(snapshot.process_start_time) else {
2365 return Ok(false);
2366 };
2367 drop(snapshot);
2368 module.record_deliberate_severance(ProcessIdentity { pid, start_time })
2369 }
2370
2371 pub fn list(&self) -> Vec<SupervisedModule> {
2372 let modules = self
2373 .modules
2374 .lock()
2375 .unwrap_or_else(|poisoned| poisoned.into_inner());
2376 let mut modules = modules.values().cloned().collect::<Vec<_>>();
2377 modules.sort_by(|left, right| left.module_id().cmp(right.module_id()));
2378 modules
2379 }
2380
2381 pub(crate) fn retire(&self, module_id: &str) -> Option<SupervisedModule> {
2382 self.spawn_nonces
2383 .lock()
2384 .unwrap_or_else(|poisoned| poisoned.into_inner())
2385 .remove(module_id);
2386 self.close_swap(module_id);
2387 let mut reserved_nonces = self
2388 .reserved_nonces
2389 .lock()
2390 .unwrap_or_else(|poisoned| poisoned.into_inner());
2391 if reserved_nonces.contains_key(module_id) {
2392 reserved_nonces.insert(module_id.to_string(), None);
2395 }
2396 drop(reserved_nonces);
2397 self.reserved_prefix_owners
2398 .lock()
2399 .unwrap_or_else(|poisoned| poisoned.into_inner())
2400 .retain(|_, owner| owner != module_id);
2401 let removed = self
2402 .modules
2403 .lock()
2404 .unwrap_or_else(|poisoned| poisoned.into_inner())
2405 .remove(module_id);
2406 self.configured_ids
2407 .lock()
2408 .unwrap_or_else(|poisoned| poisoned.into_inner())
2409 .remove(module_id);
2410 removed
2411 }
2412
2413 pub(crate) fn record_rescan_removal(&self, module_id: &str) {
2416 self.removal_tombstones
2417 .lock()
2418 .unwrap_or_else(|poisoned| poisoned.into_inner())
2419 .insert(module_id.to_string(), unix_ms_now());
2420 }
2421
2422 pub(crate) fn removal_tombstone_age_ms(&self, module_id: &str) -> Option<u64> {
2424 self.removal_tombstones
2425 .lock()
2426 .unwrap_or_else(|poisoned| poisoned.into_inner())
2427 .get(module_id)
2428 .copied()
2429 .map(|removed_at_ms| unix_ms_now().saturating_sub(removed_at_ms))
2430 }
2431
2432 pub(crate) fn release_retained_reserved_gate(&self, module_id: &str) -> bool {
2437 if self.get(module_id).is_some() {
2438 return false;
2439 }
2440 let mut reserved_nonces = self
2441 .reserved_nonces
2442 .lock()
2443 .unwrap_or_else(|poisoned| poisoned.into_inner());
2444 if !matches!(reserved_nonces.get(module_id), Some(None)) {
2445 return false;
2446 }
2447 reserved_nonces.remove(module_id);
2448 true
2449 }
2450
2451 pub(crate) fn operation_lock(&self) -> Arc<AsyncMutex<()>> {
2452 Arc::clone(&self.operation_lock)
2453 }
2454}
2455
2456#[derive(Debug, Clone)]
2458pub struct Supervisor {
2459 registry: Arc<Registry>,
2460 restart_policy: RestartPolicy,
2461 drain_timeout: Duration,
2462 connection_file_path: Option<PathBuf>,
2463 capture_logs_dir: Option<PathBuf>,
2464 forwarding: Option<Arc<ForwardingTable>>,
2465 process_liveness: Arc<SupervisorProcessLiveness>,
2466 supervisor_handle: Option<SupervisorHandle>,
2467 health: HealthConfig,
2468 daemon_start_clock: crate::clock::StartClock,
2469 terminal_journal: Option<Arc<crate::terminal_journal::TerminalJournal>>,
2470 spawn_events: SpawnEventFeed,
2471 provenance_probe: ExecutableIdentityProbe,
2472 child_roster: ChildRoster,
2475 #[cfg(target_os = "linux")]
2476 cgroup_placement: Option<subc_cgroup::Placement>,
2477 #[cfg(test)]
2478 test_after_first_spawn: AfterFirstSpawnHook,
2479}
2480
2481#[cfg(test)]
2487#[derive(Clone, Default)]
2488struct AfterFirstSpawnHook(Option<AfterFirstSpawnFn>);
2489
2490#[cfg(test)]
2491type AfterFirstSpawnFn = Arc<dyn Fn(&str) + Send + Sync>;
2492
2493#[cfg(test)]
2494impl fmt::Debug for AfterFirstSpawnHook {
2495 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
2496 f.write_str("AfterFirstSpawnHook")
2497 }
2498}
2499
2500#[cfg(test)]
2501impl AfterFirstSpawnHook {
2502 fn run(&self, module_id: &str) {
2503 if let Some(hook) = &self.0 {
2504 hook(module_id);
2505 }
2506 }
2507}
2508
2509impl Supervisor {
2510 #[cfg(test)]
2511 pub(crate) fn new_for_test(registry: Arc<Registry>, policy: RestartPolicy) -> Self {
2512 let supervisor = Self::new(registry, policy);
2513 #[cfg(target_os = "macos")]
2514 let supervisor = supervisor.with_privacy_trampoline(test_privacy_trampoline());
2515 supervisor
2516 }
2517 pub fn with_privacy_trampoline(self, path: impl Into<PathBuf>) -> Self {
2522 let path = path.into();
2523 #[cfg(target_os = "macos")]
2524 {
2525 let result = probe_privacy_trampoline(&path).map(|()| path);
2526 if let Err(cause) = &result {
2527 error!(%cause, "macOS privacy identity unavailable; supervised launches will refuse");
2528 }
2529 self.child_roster.set_privacy_trampoline(result);
2530 }
2531 #[cfg(not(target_os = "macos"))]
2532 let _ = path;
2533 self
2534 }
2535 #[cfg(unix)]
2546 pub(crate) fn begin_daemon_shutdown(&self) {
2547 self.child_roster.close();
2548 if let Some(journal) = &self.terminal_journal {
2549 journal.stamp_shutdown();
2550 }
2551 }
2552
2553 #[cfg(unix)]
2557 pub(crate) async fn drain_for_daemon_shutdown(&self) -> Result<(), SuperviseError> {
2558 const NOTICE_BUDGET: Duration = Duration::from_millis(500);
2559 const DRAIN_BUDGET: Duration = Duration::from_secs(2);
2560 let Some(forwarding) = &self.forwarding else {
2561 return Ok(());
2562 };
2563 let module_ids = forwarding
2564 .begin_daemon_drain()
2565 .map_err(SuperviseError::Forwarding)?;
2566 let deadline_ms =
2567 unix_ms_now().saturating_add((NOTICE_BUDGET + DRAIN_BUDGET).as_millis() as u64);
2568 let mut notices = tokio::task::JoinSet::new();
2569 let mut drains = Vec::new();
2570 for module_id in module_ids {
2571 let Some(target) = forwarding
2572 .begin_module_drain(&module_id, RouteCloseReason::Restart)
2573 .map_err(SuperviseError::Forwarding)?
2574 else {
2575 continue;
2576 };
2577 let routes = forwarding
2578 .endpoint_routes(target.endpoint)
2579 .map_err(SuperviseError::Forwarding)?;
2580 let command = serde_json::to_vec(&ModuleControlCommand::Draining {
2586 reason: RouteCloseReason::Restart,
2587 deadline_ms,
2588 })
2589 .expect("module draining serializes");
2590 let mut recipients = vec![(target.sink.clone(), target.negotiated_ver, command)];
2591 let mut clients: Vec<(crate::forwarding::GoodbyeTarget, Vec<u16>)> = Vec::new();
2592 for route in routes {
2593 let client = route.goodbye_target;
2594 if let Some((_, channels)) = clients
2595 .iter_mut()
2596 .find(|(existing, _)| existing.connection_id == client.connection_id)
2597 {
2598 channels.push(client.channel);
2599 } else {
2600 let channel = client.channel;
2601 clients.push((client, vec![channel]));
2602 }
2603 }
2604 for (client, mut channels) in clients {
2605 channels.sort_unstable();
2606 channels.dedup();
2607 let closing = serde_json::to_vec(&ClientControlPush::RouteClosing {
2608 module_id: module_id.clone(),
2609 channels,
2610 reason: RouteCloseReason::Restart,
2611 })
2612 .expect("route closing serializes");
2613 recipients.push((client.sink, client.negotiated_ver, closing));
2614 }
2615 for (sink, version, body) in recipients {
2616 notices.spawn(async move {
2617 let frame = Frame::build_with_version(
2618 version,
2619 FrameType::Push,
2620 control_flags(),
2621 0,
2622 0,
2623 0,
2624 body,
2625 )
2626 .expect("bounded lifecycle notice frame builds");
2627 sink.send_flushed(frame).await
2628 });
2629 }
2630 let gauges = declared_busy_gauges(&self.registry, &module_id)?;
2631 drains.push((module_id, target.endpoint, gauges));
2632 }
2633 let notice_deadline = Instant::now() + NOTICE_BUDGET;
2636 while let Ok(Some(result)) = timeout_at(notice_deadline, notices.join_next()).await {
2637 if !matches!(result, Ok(Ok(()))) {
2638 warn!(?result, "daemon shutdown notice delivery failed");
2639 }
2640 }
2641 notices.abort_all();
2642 let deadline = Instant::now() + DRAIN_BUDGET;
2643 let mut waits = tokio::task::JoinSet::new();
2644 for (module_id, endpoint, gauges) in drains {
2645 let forwarding = Arc::clone(forwarding);
2646 let mut runtime = self.runtime_config();
2647 runtime.health.cadence = Duration::from_millis(100);
2648 waits.spawn(async move {
2649 wait_for_forwarding_quiescence(
2650 &forwarding,
2651 &module_id,
2652 &runtime,
2653 endpoint,
2654 deadline,
2655 &gauges,
2656 DrainScope::Active,
2657 )
2658 .await
2659 });
2660 }
2661 while let Ok(Some(result)) = timeout_at(deadline, waits.join_next()).await {
2662 if !matches!(result, Ok(Ok(true))) {
2663 warn!(?result, "daemon shutdown drain did not reach quiescence");
2664 }
2665 }
2666 Ok(())
2667 }
2668
2669 #[cfg(unix)]
2681 pub(crate) async fn end_children_for_daemon_shutdown(
2682 &self,
2683 already_escalated: bool,
2684 escalate: impl std::future::Future<Output = ()>,
2685 ) {
2686 tokio::pin!(escalate);
2687 let mut escalated = already_escalated;
2688 if let Some(forwarding) = &self.forwarding {
2689 let reason = CloseReason::new(
2690 "daemon_shutdown",
2691 "the daemon is exiting after its shutdown notice and drain",
2692 );
2693 if escalated {
2694 send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, false).await;
2697 } else {
2698 tokio::select! {
2699 biased;
2700 _ = escalate.as_mut() => {
2701 info!("second SIGTERM: abandoning module GOODBYE delivery");
2702 escalated = true;
2703 }
2704 _ = send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, true) => {}
2705 }
2706 }
2707 let closed = forwarding.close_all_connections(&reason);
2708 debug!(closed, "closed established connections for daemon shutdown");
2709 }
2710 let escalated_here = escalated && !already_escalated;
2714 let remaining_escalate = async move {
2715 if escalated_here {
2716 std::future::pending::<()>().await;
2717 } else {
2718 escalate.await;
2719 }
2720 };
2721 crate::child_roster::end_children_for_daemon_shutdown(
2722 &self.child_roster,
2723 escalated,
2724 remaining_escalate,
2725 )
2726 .await;
2727 }
2728
2729 pub fn new(registry: Arc<Registry>, restart_policy: RestartPolicy) -> Self {
2730 Self {
2731 registry,
2732 restart_policy,
2733 drain_timeout: DEFAULT_DRAIN_TIMEOUT,
2734 connection_file_path: None,
2735 capture_logs_dir: None,
2736 forwarding: None,
2737 process_liveness: Arc::new(SupervisorProcessLiveness::default()),
2738 supervisor_handle: None,
2739 health: HealthConfig::default(),
2740 daemon_start_clock: crate::clock::StartClock::capture(),
2741 terminal_journal: None,
2742 spawn_events: SpawnEventFeed::default(),
2743 provenance_probe: ExecutableIdentityProbe::default(),
2744 child_roster: ChildRoster::default(),
2745 #[cfg(target_os = "linux")]
2746 cgroup_placement: None,
2747 #[cfg(test)]
2748 test_after_first_spawn: AfterFirstSpawnHook::default(),
2749 }
2750 }
2751
2752 pub fn with_drain_timeout(mut self, drain_timeout: Duration) -> Self {
2753 self.drain_timeout = drain_timeout;
2754 self
2755 }
2756
2757 pub fn with_process_liveness(
2758 mut self,
2759 process_liveness: Arc<SupervisorProcessLiveness>,
2760 ) -> Self {
2761 self.process_liveness = process_liveness;
2762 self
2763 }
2764
2765 pub fn with_connection_file_path(mut self, connection_file_path: impl Into<PathBuf>) -> Self {
2766 self.connection_file_path = Some(connection_file_path.into());
2767 self
2768 }
2769
2770 pub fn with_capture_logs_dir(mut self, logs_dir: impl Into<PathBuf>) -> Self {
2772 self.capture_logs_dir = Some(logs_dir.into());
2773 self
2774 }
2775
2776 pub fn with_daemon_incarnation(self, daemon_incarnation: String) -> Self {
2779 self.spawn_events.configure_incarnation(daemon_incarnation);
2783 self
2784 }
2785
2786 pub fn with_terminal_journal(self, path: PathBuf, daemon_incarnation: String) -> Self {
2789 let mut this = self.with_daemon_incarnation(daemon_incarnation.clone());
2790 this.terminal_journal = Some(Arc::new(crate::terminal_journal::TerminalJournal::open(
2791 path,
2792 daemon_incarnation,
2793 )));
2794 this
2795 }
2796
2797 pub fn with_forwarding(mut self, forwarding: Arc<ForwardingTable>) -> Self {
2798 self.forwarding = Some(forwarding);
2799 self
2800 }
2801
2802 pub fn with_handle(mut self, supervisor_handle: SupervisorHandle) -> Self {
2803 self.spawn_events = supervisor_handle.spawn_events.clone();
2804 self.supervisor_handle = Some(supervisor_handle);
2805 self
2806 }
2807
2808 pub fn with_health_config(mut self, health: HealthConfig) -> Self {
2809 self.health = health;
2810 self
2811 }
2812
2813 pub fn with_live_children_record(self, path: impl Into<PathBuf>) -> Self {
2817 self.child_roster.record_to(path.into());
2818 self
2819 }
2820
2821 #[cfg(target_os = "linux")]
2822 pub fn with_cgroup_placement(
2823 mut self,
2824 cgroup_placement: Option<subc_cgroup::Placement>,
2825 ) -> Self {
2826 self.cgroup_placement = cgroup_placement;
2827 self
2828 }
2829
2830 pub fn spawn(&self, spec: ModuleSpec) -> Result<SupervisedModule, SuperviseError> {
2836 validate_spec(&spec)?;
2837 self.establish_identity(&spec);
2838
2839 let runtime = self.runtime_config();
2840 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2841 let spawned = spawn_child(
2842 &spec,
2843 runtime.connection_file_path.as_deref(),
2844 self.supervisor_handle.as_ref(),
2845 &runtime.stderr_ring,
2846 runtime.capture_logs_dir.as_deref(),
2847 &runtime.child_roster,
2848 #[cfg(target_os = "linux")]
2849 runtime.cgroup_placement.as_ref(),
2850 );
2851 #[cfg(test)]
2852 self.test_after_first_spawn.run(&spec.module_id);
2853 let child = match spawned {
2854 Ok(child) => child,
2855 Err(err) => {
2856 self.abandon_unrostered(&spec.module_id);
2859 return Err(err);
2860 }
2861 };
2862 self.mark_first_process_running(&spec, &runtime, &snapshot, &child)?;
2863
2864 Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2865 }
2866
2867 fn establish_identity(&self, spec: &ModuleSpec) {
2880 if let Some(supervisor_handle) = &self.supervisor_handle {
2881 supervisor_handle.apply_identity_configuration(spec);
2882 supervisor_handle.mark_configured(&spec.module_id);
2883 }
2884 }
2885
2886 fn abandon_unrostered(&self, module_id: &str) {
2889 if let Some(supervisor_handle) = &self.supervisor_handle {
2890 supervisor_handle.unmark_configured_unless_rostered(module_id);
2891 }
2892 }
2893
2894 fn mark_first_process_running(
2897 &self,
2898 spec: &ModuleSpec,
2899 runtime: &SupervisorRuntimeConfig,
2900 snapshot: &SharedSnapshot,
2901 child: &SupervisedChild,
2902 ) -> Result<(), SuperviseError> {
2903 if let Err(err) = set_running(snapshot, child, &spec.module_id, &runtime.spawn_events) {
2904 self.abandon_unrostered(&spec.module_id);
2905 return Err(err);
2906 }
2907 self.process_liveness
2908 .track(spec.module_id.clone(), Arc::clone(snapshot));
2909 Ok(())
2910 }
2911
2912 pub fn supervise_configured(
2918 &self,
2919 spec: ModuleSpec,
2920 enabled: bool,
2921 ) -> Result<SupervisedModule, SuperviseError> {
2922 validate_spec(&spec)?;
2923 self.establish_identity(&spec);
2924
2925 let runtime = self.runtime_config();
2926 if !enabled {
2927 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
2928 return Ok(self.supervised_module(spec, runtime, snapshot, None));
2929 }
2930
2931 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2932 let spawned = spawn_child(
2933 &spec,
2934 runtime.connection_file_path.as_deref(),
2935 self.supervisor_handle.as_ref(),
2936 &runtime.stderr_ring,
2937 runtime.capture_logs_dir.as_deref(),
2938 &runtime.child_roster,
2939 #[cfg(target_os = "linux")]
2940 runtime.cgroup_placement.as_ref(),
2941 );
2942 #[cfg(test)]
2943 self.test_after_first_spawn.run(&spec.module_id);
2944 match spawned {
2945 Ok(child) => {
2946 self.mark_first_process_running(&spec, &runtime, &snapshot, &child)?;
2947 Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2948 }
2949 Err(err) => {
2950 error!(
2951 module_id = %spec.module_id,
2952 program = %spec.program.display(),
2953 error = %err,
2954 "configured module failed to spawn; marking failed and continuing"
2955 );
2956 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
2957 Ok(self.supervised_module(spec, runtime, snapshot, None))
2958 }
2959 }
2960 }
2961
2962 pub fn supervise_configured_with_health(
2968 &self,
2969 spec: ModuleSpec,
2970 enabled: bool,
2971 health: HealthConfig,
2972 drain_timeout_ms: Option<u64>,
2973 restart_policy: RestartPolicy,
2974 ) -> Result<SupervisedModule, SuperviseError> {
2975 validate_spec(&spec)?;
2976 self.establish_identity(&spec);
2977
2978 let mut runtime = self.runtime_config();
2979 runtime.health = health.clone();
2980 runtime.restart_policy = restart_policy;
2981 if let Some(ms) = drain_timeout_ms {
2982 runtime.drain_timeout = Duration::from_millis(ms);
2983 *runtime
2984 .effective_drain_timeout
2985 .lock()
2986 .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
2987 }
2988 if !enabled {
2989 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
2990 return Ok(self.supervised_module(spec, runtime, snapshot, None));
2991 }
2992
2993 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2994 let spawned = spawn_child(
2995 &spec,
2996 runtime.connection_file_path.as_deref(),
2997 self.supervisor_handle.as_ref(),
2998 &runtime.stderr_ring,
2999 runtime.capture_logs_dir.as_deref(),
3000 &runtime.child_roster,
3001 #[cfg(target_os = "linux")]
3002 runtime.cgroup_placement.as_ref(),
3003 );
3004 #[cfg(test)]
3005 self.test_after_first_spawn.run(&spec.module_id);
3006 match spawned {
3007 Ok(child) => {
3008 self.mark_first_process_running(&spec, &runtime, &snapshot, &child)?;
3009 Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
3010 }
3011 Err(err) => {
3012 if health.critical {
3013 error!(
3014 module_id = %spec.module_id,
3015 program = %spec.program.display(),
3016 error = %err,
3017 "critical configured module failed to spawn; marking failed and alerting"
3018 );
3019 } else {
3020 error!(
3021 module_id = %spec.module_id,
3022 program = %spec.program.display(),
3023 error = %err,
3024 "configured module failed to spawn; marking failed and continuing"
3025 );
3026 }
3027 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
3028 Ok(self.supervised_module(spec, runtime, snapshot, None))
3029 }
3030 }
3031 }
3032
3033 fn runtime_config(&self) -> SupervisorRuntimeConfig {
3034 let effective_drain_timeout = Arc::new(Mutex::new(self.drain_timeout));
3035 SupervisorRuntimeConfig {
3036 scheduled_respawn: Arc::default(),
3037 deferred_reload_reply: Arc::default(),
3038 restart_policy: self.restart_policy,
3039 drain_timeout: self.drain_timeout,
3040 child_roster: self
3043 .child_roster
3044 .for_module(Arc::clone(&effective_drain_timeout)),
3045 effective_drain_timeout,
3046 default_drain_timeout: self.drain_timeout,
3047 health: self.health.clone(),
3048 connection_file_path: self.connection_file_path.clone(),
3049 capture_logs_dir: self.capture_logs_dir.clone(),
3050 forwarding: self.forwarding.clone(),
3051 supervisor_handle: self.supervisor_handle.clone(),
3052 stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
3053 terminal_ring: Arc::new(Mutex::new(
3054 TerminalRing::new(
3055 TerminalRingConfig::default(),
3056 self.daemon_start_clock.started_at_ms(),
3057 )
3058 .with_start_clock(self.daemon_start_clock)
3059 .with_journal(self.terminal_journal.clone())
3060 .with_daemon_shutdown(self.child_roster.shutdown_flag()),
3061 )),
3062 spawn_events: self.spawn_events.clone(),
3063 #[cfg(target_os = "linux")]
3064 cgroup_placement: self.cgroup_placement.clone(),
3065 #[cfg(test)]
3066 test_seed_stale_facts_before_enable_spawn: false,
3067 #[cfg(test)]
3068 test_reload_exit_record_gate: None,
3069 }
3070 }
3071
3072 fn supervised_module(
3073 &self,
3074 spec: ModuleSpec,
3075 runtime: SupervisorRuntimeConfig,
3076 snapshot: SharedSnapshot,
3077 child: Option<SupervisedChild>,
3078 ) -> SupervisedModule {
3079 let configuration = Arc::new(Mutex::new(SupervisedConfiguration {
3080 spec: spec.clone(),
3081 health: runtime.health.clone(),
3082 }));
3083 let stderr_ring = Arc::clone(&runtime.stderr_ring);
3084 let terminal_ring = Arc::clone(&runtime.terminal_ring);
3085 let restart_policy = runtime.restart_policy;
3089 let effective_drain_timeout = Arc::clone(&runtime.effective_drain_timeout);
3090 let (tx, rx) = mpsc::channel(4);
3091 let monitor = tokio::spawn(supervise_loop(
3092 spec.clone(),
3093 runtime,
3094 Arc::clone(&self.registry),
3095 Arc::clone(&self.process_liveness),
3096 Arc::clone(&snapshot),
3097 child,
3098 rx,
3099 ));
3100
3101 let module_id = spec.module_id.clone();
3102 let module = SupervisedModule {
3103 inner: Arc::new(SupervisedModuleInner {
3104 module_id: module_id.clone(),
3105 registry: Arc::clone(&self.registry),
3106 snapshot,
3107 configuration,
3108 stderr_ring,
3109 terminal_ring,
3110 commands: tx,
3111 monitor: Mutex::new(Some(monitor)),
3112 restart_policy,
3113 effective_drain_timeout,
3114 provenance_probe: self.provenance_probe.clone(),
3115 }),
3116 };
3117 if let Some(supervisor_handle) = &self.supervisor_handle {
3121 supervisor_handle.insert(module.clone());
3122 }
3123 module
3124 }
3125}
3126
3127impl Default for Supervisor {
3128 fn default() -> Self {
3129 Self::new(Arc::new(Registry::default()), RestartPolicy::default())
3130 }
3131}
3132
3133#[derive(Clone)]
3135pub struct SupervisedModule {
3136 inner: Arc<SupervisedModuleInner>,
3137}
3138
3139struct SupervisedModuleInner {
3140 module_id: String,
3141 registry: Arc<Registry>,
3142 snapshot: SharedSnapshot,
3143 configuration: Arc<Mutex<SupervisedConfiguration>>,
3144 stderr_ring: Arc<Mutex<StderrRing>>,
3145 terminal_ring: Arc<Mutex<TerminalRing>>,
3146 commands: mpsc::Sender<SupervisorCommand>,
3147 monitor: Mutex<Option<JoinHandle<()>>>,
3148 restart_policy: RestartPolicy,
3152 effective_drain_timeout: Arc<Mutex<Duration>>,
3153 provenance_probe: ExecutableIdentityProbe,
3154}
3155
3156impl fmt::Debug for SupervisedModule {
3157 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3158 f.debug_struct("SupervisedModule")
3159 .field("module_id", &self.inner.module_id)
3160 .field("status", &self.status())
3161 .finish_non_exhaustive()
3162 }
3163}
3164
3165impl SupervisedModule {
3166 pub fn module_id(&self) -> &str {
3167 &self.inner.module_id
3168 }
3169
3170 #[cfg(test)]
3174 pub(crate) fn record_health_probe_failure_for_test(
3175 &self,
3176 detail: &str,
3177 ) -> Result<(), SuperviseError> {
3178 update_snapshot(&self.inner.snapshot, Some(&self.inner.module_id), |state| {
3179 state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
3180 state.health.detail = Some(detail.to_string());
3181 })
3182 }
3183
3184 pub fn state(&self) -> Result<ModuleState, SuperviseError> {
3185 Ok(lock_snapshot(&self.inner.snapshot)?.state)
3186 }
3187
3188 pub fn stderr_tail(
3195 &self,
3196 max_lines: Option<usize>,
3197 max_bytes: Option<usize>,
3198 ) -> StderrTailSnapshot {
3199 self.inner
3200 .stderr_ring
3201 .lock()
3202 .unwrap_or_else(|poisoned| poisoned.into_inner())
3203 .snapshot(max_lines, max_bytes)
3204 }
3205
3206 pub fn terminal_history(&self) -> TerminalHistorySnapshot {
3211 self.inner
3212 .terminal_ring
3213 .lock()
3214 .unwrap_or_else(|poisoned| poisoned.into_inner())
3215 .snapshot()
3216 }
3217
3218 pub fn durable_terminal_history(&self) -> subc_control::TerminalHistory {
3223 durable_terminal_history_of(&self.inner.terminal_ring, &self.inner.module_id)
3224 }
3225
3226 pub(crate) async fn read_durable_terminal_history(
3231 &self,
3232 ) -> Result<subc_control::TerminalHistory, tokio::task::JoinError> {
3233 let terminal_ring = Arc::clone(&self.inner.terminal_ring);
3234 let module_id = self.inner.module_id.clone();
3235 tokio::task::spawn_blocking(move || durable_terminal_history_of(&terminal_ring, &module_id))
3236 .await
3237 }
3238
3239 pub fn status(&self) -> Result<ModuleStatus, SuperviseError> {
3240 self.status_with_snapshot_lock(&self.inner.snapshot, None)
3241 .map(|(status, _)| status)
3242 }
3243
3244 pub(crate) fn record_deliberate_severance(
3245 &self,
3246 identity: ProcessIdentity,
3247 ) -> Result<bool, SuperviseError> {
3248 let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
3249 if snapshot.pid != Some(identity.pid)
3250 || snapshot.process_start_time != Some(identity.start_time)
3251 {
3252 return Ok(false);
3253 }
3254 snapshot.deliberate_severance = Some(identity);
3255 Ok(true)
3256 }
3257
3258 pub(crate) fn status_for_control(
3263 &self,
3264 caller: &'static str,
3265 ) -> Result<ModuleStatus, SuperviseError> {
3266 self.status_with_snapshot_lock(&self.inner.snapshot, Some(caller))
3267 .map(|(status, _)| status)
3268 }
3269
3270 fn status_with_snapshot_lock(
3271 &self,
3272 snapshot: &SharedSnapshot,
3273 caller: Option<&'static str>,
3274 ) -> Result<(ModuleStatus, Option<SpawnedFileIdentity>), SuperviseError> {
3275 let mut guard = match caller {
3276 Some(caller) => lock_snapshot_for_control(snapshot, &self.inner.module_id, caller)?,
3277 None => lock_snapshot(snapshot)?,
3278 };
3279 let restart_count =
3282 guard.crash_restarts_in_window(self.inner.restart_policy.window, Instant::now());
3283 let snapshot = guard.clone();
3284 drop(guard);
3285 let drain_timeout = *self.inner.effective_drain_timeout.lock().map_err(|_| {
3286 SuperviseError::StatePoisoned {
3287 module_id: Some(self.inner.module_id.clone()),
3288 }
3289 })?;
3290 let registration_active = self
3291 .inner
3292 .registry
3293 .get_module(&self.inner.module_id)
3294 .map_err(SuperviseError::Registry)?
3295 .is_some();
3296 let protocol = snapshot
3297 .spawned_protocol
3298 .unwrap_or(self.declared_protocol()?);
3299 let running_process =
3300 snapshot.enabled && snapshot.state == ModuleState::Running && snapshot.process_alive;
3301 let live = match protocol {
3307 ModuleProtocol::Subc => running_process && registration_active,
3308 ModuleProtocol::None => running_process,
3309 };
3310
3311 Ok((
3312 ModuleStatus {
3313 module_id: self.inner.module_id.clone(),
3314 state: snapshot.state,
3315 enabled: snapshot.enabled,
3316 process_alive: snapshot.process_alive,
3317 registration_active,
3318 protocol,
3319 live,
3320 restart_count,
3321 lifetime_restarts: snapshot.lifetime_restarts,
3322 spawn_generation: snapshot.spawn_generation,
3323 max_restarts: self.inner.restart_policy.max_restarts,
3324 restart_window: self.inner.restart_policy.window,
3325 drain_timeout,
3326 restart_backoff: self.inner.restart_policy.backoff,
3327 restart_max_backoff: self.inner.restart_policy.max_backoff,
3328 pid: snapshot.reported_pid(),
3329 spawned_at_ms: snapshot.spawned_at_ms,
3330 spawned_from: snapshot.spawned_from,
3331 process_start_time: snapshot.process_start_time,
3332 last_exit: snapshot.last_exit,
3333 health: snapshot.health,
3334 },
3335 snapshot.spawned_file_identity,
3336 ))
3337 }
3338
3339 #[cfg(test)]
3340 pub(crate) fn hold_snapshot_for_test(
3341 &self,
3342 acquired: std::sync::mpsc::Sender<()>,
3343 hold: Duration,
3344 ) -> std::thread::JoinHandle<()> {
3345 let snapshot = Arc::clone(&self.inner.snapshot);
3346 std::thread::spawn(move || {
3347 let _guard = snapshot.lock().expect("test snapshot lock is not poisoned");
3348 acquired
3349 .send(())
3350 .expect("test receiver waits for snapshot lock");
3351 std::thread::sleep(hold);
3352 })
3353 }
3354
3355 pub(crate) async fn status_and_running_image_agreement(
3361 &self,
3362 ) -> Result<(ModuleStatus, subc_control::RunningImageAgreement), SuperviseError> {
3363 let (status, identity) = self.status_with_snapshot_lock(&self.inner.snapshot, None)?;
3364 let image = self
3365 .inner
3366 .provenance_probe
3367 .observe(
3368 status.pid,
3369 status.spawned_from.as_deref(),
3370 identity,
3371 status.process_start_time,
3372 )
3373 .await;
3374 Ok((status, image))
3375 }
3376
3377 pub(crate) async fn running_image_agreement(&self) -> subc_control::RunningImageAgreement {
3378 let snapshot = match lock_snapshot(&self.inner.snapshot) {
3379 Ok(snapshot) => snapshot.clone(),
3380 Err(_) => {
3381 return subc_control::RunningImageAgreement::Unavailable {
3382 reason: subc_control::RunningImageUnavailableReason::NotRunning,
3383 };
3384 }
3385 };
3386 self.inner
3387 .provenance_probe
3388 .observe(
3389 snapshot.reported_pid(),
3390 snapshot.spawned_from.as_deref(),
3391 snapshot.spawned_file_identity,
3392 snapshot.process_start_time,
3393 )
3394 .await
3395 }
3396
3397 pub(crate) fn child_resource_usage(&self) -> subc_control::ChildResourceUsage {
3400 let (pid, start_time) = match lock_snapshot(&self.inner.snapshot) {
3401 Ok(snapshot) => (snapshot.reported_pid(), snapshot.process_start_time),
3402 Err(_) => {
3403 return subc_control::ChildResourceUsage::Unavailable {
3404 reason: subc_control::ChildResourceUnavailableReason::Unreadable,
3405 }
3406 }
3407 };
3408 crate::child_resources::read(pid, start_time)
3409 }
3410
3411 pub(crate) fn will_recover_after_connection_loss(&self) -> Result<bool, SuperviseError> {
3412 let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
3413 Ok(match snapshot.state {
3414 ModuleState::Restarting => true,
3415 ModuleState::Failed | ModuleState::Disabled => false,
3416 _ => daemon_will_restart(&mut snapshot, &self.inner.restart_policy, Instant::now()),
3417 })
3418 }
3419
3420 #[cfg(test)]
3421 pub(crate) fn is_warming(&self) -> Result<bool, SuperviseError> {
3422 self.is_warming_with_snapshot_lock(None)
3423 }
3424
3425 pub(crate) fn is_warming_for_control(
3426 &self,
3427 caller: &'static str,
3428 ) -> Result<bool, SuperviseError> {
3429 self.is_warming_with_snapshot_lock(Some(caller))
3430 }
3431
3432 fn is_warming_with_snapshot_lock(
3433 &self,
3434 caller: Option<&'static str>,
3435 ) -> Result<bool, SuperviseError> {
3436 let snapshot = match caller {
3437 Some(caller) => {
3438 lock_snapshot_for_control(&self.inner.snapshot, &self.inner.module_id, caller)?
3439 }
3440 None => lock_snapshot(&self.inner.snapshot)?,
3441 }
3442 .clone();
3443 Ok(matches!(
3444 snapshot.state,
3445 ModuleState::Starting | ModuleState::Running | ModuleState::Restarting
3446 ))
3447 }
3448
3449 pub async fn drain(&self) -> Result<(), SuperviseError> {
3451 self.stop().await
3452 }
3453
3454 pub(crate) async fn retire(&self) -> Result<(), SuperviseError> {
3455 match self.state()? {
3456 ModuleState::Stopped | ModuleState::Failed => return Ok(()),
3457 ModuleState::Starting
3458 | ModuleState::Running
3459 | ModuleState::Unresponsive
3460 | ModuleState::Restarting
3461 | ModuleState::Draining
3462 | ModuleState::Disabled => {}
3463 }
3464
3465 let (reply_tx, reply_rx) = oneshot::channel();
3466 self.inner
3467 .commands
3468 .send(SupervisorCommand::Retire { reply: reply_tx })
3469 .await
3470 .map_err(|_| SuperviseError::CommandClosed {
3471 module_id: self.inner.module_id.clone(),
3472 })?;
3473 reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3474 module_id: self.inner.module_id.clone(),
3475 })?
3476 }
3477
3478 pub async fn stop(&self) -> Result<(), SuperviseError> {
3479 match self.state()? {
3480 ModuleState::Stopped | ModuleState::Failed => return Ok(()),
3481 ModuleState::Starting
3482 | ModuleState::Running
3483 | ModuleState::Unresponsive
3484 | ModuleState::Restarting
3485 | ModuleState::Draining
3486 | ModuleState::Disabled => {}
3487 }
3488
3489 let (reply_tx, reply_rx) = oneshot::channel();
3490 self.inner
3491 .commands
3492 .send(SupervisorCommand::Drain { reply: reply_tx })
3493 .await
3494 .map_err(|_| SuperviseError::CommandClosed {
3495 module_id: self.inner.module_id.clone(),
3496 })?;
3497 reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3498 module_id: self.inner.module_id.clone(),
3499 })?
3500 }
3501
3502 pub async fn restart(&self, drain_timeout_ms: Option<u64>) -> Result<(), SuperviseError> {
3503 let received_at_generation = lock_snapshot(&self.inner.snapshot)?.spawn_generation;
3504 let (reply_tx, reply_rx) = oneshot::channel();
3505 self.inner
3506 .commands
3507 .send(SupervisorCommand::Restart {
3508 drain_timeout_ms,
3509 received_at_generation,
3510 queued_at: Instant::now(),
3511 reply: reply_tx,
3512 })
3513 .await
3514 .map_err(|_| SuperviseError::CommandClosed {
3515 module_id: self.inner.module_id.clone(),
3516 })?;
3517 reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3518 module_id: self.inner.module_id.clone(),
3519 })?
3520 }
3521
3522 pub async fn swap(&self, ready_timeout: Option<Duration>) -> Result<(), SuperviseError> {
3527 let (reply_tx, reply_rx) = oneshot::channel();
3528 self.inner
3529 .commands
3530 .send(SupervisorCommand::Swap {
3531 ready_timeout,
3532 reply: reply_tx,
3533 })
3534 .await
3535 .map_err(|_| SuperviseError::CommandClosed {
3536 module_id: self.inner.module_id.clone(),
3537 })?;
3538 reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3539 module_id: self.inner.module_id.clone(),
3540 })?
3541 }
3542
3543 pub async fn reload(&self) -> Result<(), SuperviseError> {
3544 let (reply_tx, reply_rx) = oneshot::channel();
3545 self.inner
3546 .commands
3547 .send(SupervisorCommand::Reload { reply: reply_tx })
3548 .await
3549 .map_err(|_| SuperviseError::CommandClosed {
3550 module_id: self.inner.module_id.clone(),
3551 })?;
3552 reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3553 module_id: self.inner.module_id.clone(),
3554 })?
3555 }
3556
3557 pub async fn set_enabled(&self, enabled: bool) -> Result<bool, SuperviseError> {
3558 let (reply_tx, reply_rx) = oneshot::channel();
3559 self.inner
3560 .commands
3561 .send(SupervisorCommand::SetEnabled {
3562 enabled,
3563 reply: reply_tx,
3564 })
3565 .await
3566 .map_err(|_| SuperviseError::CommandClosed {
3567 module_id: self.inner.module_id.clone(),
3568 })?;
3569 reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3570 module_id: self.inner.module_id.clone(),
3571 })?
3572 }
3573
3574 pub(crate) fn declared_protocol(&self) -> Result<ModuleProtocol, SuperviseError> {
3578 let configured = self
3579 .inner
3580 .configuration
3581 .lock()
3582 .map_err(|_| SuperviseError::StatePoisoned {
3583 module_id: Some(self.inner.module_id.clone()),
3584 })?
3585 .spec
3586 .protocol;
3587 let state = lock_snapshot(&self.inner.snapshot)?;
3588 Ok(state.spawned_protocol.unwrap_or(configured))
3589 }
3590
3591 pub(crate) fn configuration(&self) -> Result<(ModuleSpec, HealthConfig), SuperviseError> {
3592 let configuration =
3593 self.inner
3594 .configuration
3595 .lock()
3596 .map_err(|_| SuperviseError::StatePoisoned {
3597 module_id: Some(self.inner.module_id.clone()),
3598 })?;
3599 Ok((configuration.spec.clone(), configuration.health.clone()))
3600 }
3601
3602 #[cfg(any(test, feature = "test-support"))]
3606 pub async fn update_spec_for_test(&self, spec: ModuleSpec) -> Result<(), SuperviseError> {
3607 let (_, health) = self.configuration()?;
3608 let drain_timeout_ms = u64::try_from(
3609 self.inner
3610 .effective_drain_timeout
3611 .lock()
3612 .unwrap_or_else(|poisoned| poisoned.into_inner())
3613 .as_millis(),
3614 )
3615 .ok();
3616 self.update_configuration(spec, health, drain_timeout_ms)
3617 .await
3618 }
3619
3620 pub(crate) async fn update_configuration(
3621 &self,
3622 spec: ModuleSpec,
3623 health: HealthConfig,
3624 drain_timeout_ms: Option<u64>,
3625 ) -> Result<(), SuperviseError> {
3626 if spec.module_id != self.inner.module_id {
3627 return Err(SuperviseError::InvalidSpec {
3628 reason: "a supervised module's module_id cannot be changed".to_string(),
3629 });
3630 }
3631 validate_spec(&spec)?;
3632 let (reply_tx, reply_rx) = oneshot::channel();
3633 self.inner
3634 .commands
3635 .send(SupervisorCommand::UpdateConfiguration {
3636 spec: spec.clone(),
3637 health: health.clone(),
3638 drain_timeout_ms,
3639 reply: reply_tx,
3640 })
3641 .await
3642 .map_err(|_| SuperviseError::CommandClosed {
3643 module_id: self.inner.module_id.clone(),
3644 })?;
3645 reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3646 module_id: self.inner.module_id.clone(),
3647 })?;
3648 let mut configuration =
3649 self.inner
3650 .configuration
3651 .lock()
3652 .map_err(|_| SuperviseError::StatePoisoned {
3653 module_id: Some(self.inner.module_id.clone()),
3654 })?;
3655 configuration.spec = spec;
3656 configuration.health = health;
3657 Ok(())
3658 }
3659}
3660
3661impl Drop for SupervisedModuleInner {
3662 fn drop(&mut self) {
3663 let Ok(mut monitor) = self.monitor.lock() else {
3664 return;
3665 };
3666 if let Some(monitor) = monitor.as_ref().filter(|monitor| !monitor.is_finished()) {
3667 let _ = update_snapshot(&self.snapshot, Some(&self.module_id), |state| {
3668 state.state = ModuleState::Stopped;
3669 clear_current_process_facts(state);
3670 });
3671 monitor.abort();
3672 }
3673 let _ = monitor.take();
3674 }
3675}
3676
3677#[derive(Debug)]
3678enum SupervisorCommand {
3679 Drain {
3680 reply: oneshot::Sender<Result<(), SuperviseError>>,
3681 },
3682 Retire {
3683 reply: oneshot::Sender<Result<(), SuperviseError>>,
3684 },
3685 Restart {
3686 drain_timeout_ms: Option<u64>,
3691 received_at_generation: u64,
3695 queued_at: Instant,
3698 reply: oneshot::Sender<Result<(), SuperviseError>>,
3699 },
3700 Reload {
3701 reply: oneshot::Sender<Result<(), SuperviseError>>,
3702 },
3703 SetEnabled {
3704 enabled: bool,
3705 reply: oneshot::Sender<Result<bool, SuperviseError>>,
3706 },
3707 UpdateConfiguration {
3708 spec: ModuleSpec,
3709 health: HealthConfig,
3710 drain_timeout_ms: Option<u64>,
3713 reply: oneshot::Sender<()>,
3714 },
3715 Swap {
3716 ready_timeout: Option<Duration>,
3719 reply: oneshot::Sender<Result<(), SuperviseError>>,
3721 },
3722}
3723
3724#[derive(Debug)]
3725pub enum SuperviseError {
3726 InvalidSpec {
3727 reason: String,
3728 },
3729 Spawn {
3730 program: PathBuf,
3731 source: io::Error,
3732 cgroup_path: Option<PathBuf>,
3733 },
3734 Cgroup {
3735 module_id: String,
3736 source: io::Error,
3737 },
3738 LaunchNonce {
3741 reason: String,
3742 },
3743 Wait {
3744 module_id: String,
3745 source: io::Error,
3746 },
3747 Kill {
3748 module_id: String,
3749 source: io::Error,
3750 },
3751 Forwarding(ForwardingError),
3752 Registry(RegistryError),
3753 ReloadUnavailable {
3754 module_id: String,
3755 reason: String,
3756 },
3757 Disabled {
3762 module_id: String,
3763 },
3764 ReloadFailed {
3765 module_id: String,
3766 reason: String,
3767 },
3768 RegistrationStillActive {
3769 module_id: String,
3770 waited: Duration,
3771 },
3772 StatePoisoned {
3773 module_id: Option<String>,
3774 },
3775 CommandClosed {
3776 module_id: String,
3777 },
3778 SwapInProgress {
3782 module_id: String,
3783 },
3784 SwapRefused {
3786 module_id: String,
3787 reason: SwapRefusal,
3788 },
3789 SwapFailed {
3793 module_id: String,
3794 arm: SwapFailureArm,
3795 detail: String,
3796 candidate_exit: Option<ExitReport>,
3799 },
3800}
3801
3802#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3804pub enum SwapRefusal {
3805 OverlapExclusive,
3807 NotRegistered,
3810 ProtocolNone,
3813 NotConfigured,
3816 AlreadySwapping,
3818}
3819
3820impl SwapRefusal {
3821 pub fn as_str(self) -> &'static str {
3822 match self {
3823 Self::OverlapExclusive => "overlap_exclusive",
3824 Self::NotRegistered => "not_registered",
3825 Self::ProtocolNone => "protocol_none",
3826 Self::NotConfigured => "not_configured",
3827 Self::AlreadySwapping => "already_swapping",
3828 }
3829 }
3830}
3831
3832#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3835pub enum SwapFailureArm {
3836 SpawnFailed,
3838 NeverRegistered,
3840 NeverReady,
3842 CandidateExited,
3844 CandidateUnhealthy,
3846 Interrupted,
3850 CutoverLost,
3855}
3856
3857impl SwapFailureArm {
3858 pub fn as_str(self) -> &'static str {
3859 match self {
3860 Self::SpawnFailed => "spawn_failed",
3861 Self::NeverRegistered => "never_registered",
3862 Self::NeverReady => "never_ready",
3863 Self::CandidateExited => "candidate_exited",
3864 Self::CandidateUnhealthy => "candidate_unhealthy",
3865 Self::Interrupted => "interrupted",
3866 Self::CutoverLost => "cutover_lost",
3867 }
3868 }
3869}
3870
3871impl fmt::Display for SuperviseError {
3872 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3873 match self {
3874 Self::InvalidSpec { reason } => write!(f, "invalid module spec: {reason}"),
3875 Self::Spawn {
3876 program,
3877 source,
3878 cgroup_path: Some(cgroup_path),
3879 } => write!(
3880 f,
3881 "failed to place module in cgroup '{}' while spawning '{}': {source}",
3882 cgroup_path.display(),
3883 program.display()
3884 ),
3885 Self::Spawn {
3886 program,
3887 source,
3888 cgroup_path: None,
3889 } => write!(
3890 f,
3891 "failed to spawn module '{}': {source}",
3892 program.display()
3893 ),
3894 Self::Cgroup { module_id, source } => {
3895 write!(
3896 f,
3897 "failed to prepare cgroup for module '{module_id}': {source}"
3898 )
3899 }
3900 Self::LaunchNonce { reason } => {
3901 write!(
3902 f,
3903 "failed to generate reserved-module launch nonce: {reason}"
3904 )
3905 }
3906 Self::Wait { module_id, source } => {
3907 write!(f, "failed to wait for module '{module_id}': {source}")
3908 }
3909 Self::Kill { module_id, source } => {
3910 write!(f, "failed to kill module '{module_id}': {source}")
3911 }
3912 Self::Forwarding(err) => write!(f, "forwarding error: {err}"),
3913 Self::Registry(err) => write!(f, "registry error: {err}"),
3914 Self::ReloadUnavailable { module_id, reason } => {
3915 write!(f, "reload unavailable for module '{module_id}': {reason}")
3916 }
3917 Self::Disabled { module_id } => {
3918 write!(
3919 f,
3920 "module '{module_id}' is disabled; enable it before restart or reload"
3921 )
3922 }
3923 Self::ReloadFailed { module_id, reason } => {
3924 write!(f, "reload failed for module '{module_id}': {reason}")
3925 }
3926 Self::RegistrationStillActive { module_id, waited } => write!(
3927 f,
3928 "module '{module_id}' registration remained active after waiting {waited:?}"
3929 ),
3930 Self::StatePoisoned { module_id } => match module_id {
3931 Some(module_id) => {
3932 write!(f, "supervisor state for module '{module_id}' was poisoned")
3933 }
3934 None => write!(f, "supervisor state was poisoned"),
3935 },
3936 Self::CommandClosed { module_id } => {
3937 write!(
3938 f,
3939 "supervisor command channel for module '{module_id}' is closed"
3940 )
3941 }
3942 Self::SwapInProgress { module_id } => write!(
3943 f,
3944 "module '{module_id}' is being swapped; retry once the swap has cut over or failed, or stop the module to abort the swap"
3945 ),
3946 Self::SwapRefused { module_id, reason } => match reason {
3947 SwapRefusal::OverlapExclusive => write!(
3948 f,
3949 "module '{module_id}' is declared overlap: \"exclusive\" (the default): two processes of it must not run at once, so it cannot be swapped; use a plain restart, or declare overlap: \"safe\" in its config if it really tolerates a second process"
3950 ),
3951 SwapRefusal::NotRegistered => write!(
3952 f,
3953 "module '{module_id}' is not registered, so there is no serving process to keep while a replacement warms; use a plain restart"
3954 ),
3955 SwapRefusal::ProtocolNone => write!(
3956 f,
3957 "module '{module_id}' is protocol: \"none\" and never registers, so a swap could never see its replacement become ready; use a plain restart"
3958 ),
3959 SwapRefusal::NotConfigured => write!(
3960 f,
3961 "module '{module_id}' cannot be swapped: the supervisor was built without the forwarding table or shared handle a swap needs"
3962 ),
3963 SwapRefusal::AlreadySwapping => {
3964 write!(f, "module '{module_id}' is already being swapped")
3965 }
3966 },
3967 Self::SwapFailed {
3968 module_id,
3969 arm,
3970 detail,
3971 ..
3972 } => write!(
3973 f,
3974 "swap of module '{module_id}' failed ({}): {detail}; the running process was left serving",
3975 arm.as_str()
3976 ),
3977 }
3978 }
3979}
3980
3981impl Error for SuperviseError {
3982 fn source(&self) -> Option<&(dyn Error + 'static)> {
3983 match self {
3984 Self::Spawn { source, .. }
3985 | Self::Cgroup { source, .. }
3986 | Self::Wait { source, .. }
3987 | Self::Kill { source, .. } => Some(source),
3988 Self::Forwarding(err) => Some(err),
3989 Self::Registry(err) => Some(err),
3990 Self::LaunchNonce { .. }
3991 | Self::InvalidSpec { .. }
3992 | Self::ReloadUnavailable { .. }
3993 | Self::Disabled { .. }
3994 | Self::ReloadFailed { .. }
3995 | Self::RegistrationStillActive { .. }
3996 | Self::StatePoisoned { .. }
3997 | Self::CommandClosed { .. }
3998 | Self::SwapInProgress { .. }
3999 | Self::SwapRefused { .. }
4000 | Self::SwapFailed { .. } => None,
4001 }
4002 }
4003}
4004
4005pub(crate) fn validate_spec(spec: &ModuleSpec) -> Result<(), SuperviseError> {
4006 if spec.module_id.trim().is_empty() {
4007 return Err(SuperviseError::InvalidSpec {
4008 reason: "module_id must not be empty".to_string(),
4009 });
4010 }
4011
4012 Ok(())
4013}
4014
4015#[derive(Debug, Default)]
4016struct HealthProbeRuntime {
4017 configured_health: Option<HealthConfig>,
4018 registered_connection: Option<crate::ConnectionId>,
4019 advertised: bool,
4020 next_probe_at: Option<Instant>,
4021 probe_index: u64,
4022}
4023
4024fn running_protocol(spec: &ModuleSpec, snapshot: &SharedSnapshot) -> ModuleProtocol {
4025 lock_snapshot(snapshot)
4026 .ok()
4027 .and_then(|state| state.spawned_protocol)
4028 .unwrap_or(spec.protocol)
4029}
4030
4031impl HealthProbeRuntime {
4032 fn refresh_registration(
4033 &mut self,
4034 spec: &ModuleSpec,
4035 runtime: &SupervisorRuntimeConfig,
4036 registry: &Registry,
4037 snapshot: &SharedSnapshot,
4038 ) {
4039 if self.configured_health.as_ref() != Some(&runtime.health) {
4040 self.configured_health = Some(runtime.health.clone());
4041 self.next_probe_at = None;
4042 self.registered_connection = None;
4043 self.probe_index = 0;
4044 }
4045 if running_protocol(spec, snapshot) == ModuleProtocol::None {
4049 self.registered_connection = None;
4050 self.advertised = runtime.health.http.is_some();
4051 if !self.advertised {
4052 self.next_probe_at = None;
4053 let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4054 state.health = ModuleHealthStatus::default();
4055 });
4056 } else if self.next_probe_at.is_none() {
4057 self.next_probe_at = Some(
4058 Instant::now()
4059 + jittered_health_delay(&spec.module_id, 0, runtime.health.cadence),
4060 );
4061 }
4062 return;
4063 }
4064
4065 let registration = match registry.get_module(&spec.module_id) {
4066 Ok(registration) => registration,
4067 Err(err) => {
4068 warn!(module_id = %spec.module_id, error = %err, "health prober could not read registry");
4069 self.advertised = false;
4070 self.next_probe_at = None;
4071 return;
4072 }
4073 };
4074
4075 let Some(registration) = registration else {
4076 self.registered_connection = None;
4077 self.advertised = false;
4078 self.next_probe_at = None;
4079 return;
4080 };
4081
4082 let advertised = registration
4083 .control_ops
4084 .iter()
4085 .any(|op| op == MODULE_CONTROL_OP_HEALTH_CHECK);
4086 if !advertised {
4087 self.registered_connection = Some(registration.connection_id);
4088 self.advertised = false;
4089 self.next_probe_at = None;
4090 let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4091 state.health.status = SupervisorHealthStatus::Unknown;
4092 state.health.consecutive_failures = 0;
4093 state.health.last_probe_ms = None;
4094 state.health.detail = None;
4095 state.health.metrics = None;
4096 });
4097 return;
4098 }
4099
4100 let reregistered = self.registered_connection != Some(registration.connection_id);
4101 self.registered_connection = Some(registration.connection_id);
4102 self.advertised = true;
4103 if reregistered || self.next_probe_at.is_none() {
4104 self.probe_index = 0;
4105 self.next_probe_at = Some(
4106 Instant::now() + jittered_health_delay(&spec.module_id, 0, runtime.health.cadence),
4107 );
4108 let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4109 state.health.status = SupervisorHealthStatus::Unknown;
4110 state.health.consecutive_failures = 0;
4111 state.health.detail = None;
4112 state.health.metrics = None;
4113 });
4114 }
4115 }
4116
4117 fn wake_after(&self) -> Duration {
4118 if !self.advertised {
4119 return REGISTRY_RELEASE_POLL;
4120 }
4121 self.next_probe_at
4122 .map(|next| next.saturating_duration_since(Instant::now()))
4123 .unwrap_or(REGISTRY_RELEASE_POLL)
4124 }
4125
4126 fn due(&self) -> bool {
4127 self.advertised
4128 && self
4129 .next_probe_at
4130 .is_some_and(|next| Instant::now() >= next)
4131 }
4132
4133 fn schedule_next(&mut self, spec: &ModuleSpec, cadence: Duration) {
4134 self.probe_index = self.probe_index.wrapping_add(1);
4135 self.next_probe_at = Some(
4136 Instant::now() + jittered_health_delay(&spec.module_id, self.probe_index, cadence),
4137 );
4138 }
4139}
4140
4141#[derive(Debug)]
4176enum HealthProbeEvidence {
4177 LaneDead,
4179 NoAnswer,
4181 BadAnswer,
4183 Misconfigured,
4185}
4186
4187#[derive(Debug)]
4188struct HealthProbeError {
4189 evidence: HealthProbeEvidence,
4190 message: String,
4191}
4192
4193impl HealthProbeError {
4194 fn lane_dead(message: impl Into<String>) -> Self {
4195 Self::with(HealthProbeEvidence::LaneDead, message)
4196 }
4197
4198 fn no_answer(message: impl Into<String>) -> Self {
4199 Self::with(HealthProbeEvidence::NoAnswer, message)
4200 }
4201
4202 fn bad_answer(message: impl Into<String>) -> Self {
4203 Self::with(HealthProbeEvidence::BadAnswer, message)
4204 }
4205
4206 fn misconfigured(message: impl Into<String>) -> Self {
4207 Self::with(HealthProbeEvidence::Misconfigured, message)
4208 }
4209
4210 fn with(evidence: HealthProbeEvidence, message: impl Into<String>) -> Self {
4211 Self {
4212 evidence,
4213 message: message.into(),
4214 }
4215 }
4216
4217 #[allow(dead_code)]
4231 fn is_proof_of_death(&self) -> bool {
4232 matches!(self.evidence, HealthProbeEvidence::LaneDead)
4233 }
4234
4235 fn label(&self) -> &'static str {
4243 match self.evidence {
4244 HealthProbeEvidence::LaneDead => "lane-dead",
4245 HealthProbeEvidence::NoAnswer => "no-answer",
4246 HealthProbeEvidence::BadAnswer => "bad-answer",
4247 HealthProbeEvidence::Misconfigured => "daemon-misconfigured",
4248 }
4249 }
4250}
4251
4252impl fmt::Display for HealthProbeError {
4253 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
4254 f.write_str(&self.message)
4255 }
4256}
4257
4258async fn run_health_probe_cycle(
4259 spec: &ModuleSpec,
4260 runtime: &SupervisorRuntimeConfig,
4261 registry: &Registry,
4262 process_liveness: &SupervisorProcessLiveness,
4263 snapshot: &SharedSnapshot,
4264 child: &mut Option<SupervisedChild>,
4265) {
4266 let now_ms = unix_ms_now();
4267 let http = (running_protocol(spec, snapshot) == ModuleProtocol::None)
4268 .then_some(runtime.health.http.as_deref())
4269 .flatten();
4270 let result = match http {
4271 Some(url) => probe_http_health(url, runtime.health.deadline).await,
4272 None => probe_module_health(&spec.module_id, runtime, None).await,
4273 };
4274 match result {
4275 Ok(report) => {
4276 handle_health_report(
4277 spec,
4278 runtime,
4279 registry,
4280 process_liveness,
4281 snapshot,
4282 child,
4283 report,
4284 now_ms,
4285 )
4286 .await;
4287 }
4288 Err(err) => {
4289 if http.is_some() {
4290 let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4291 state.health.status = SupervisorHealthStatus::Failing;
4292 });
4293 }
4294 handle_health_probe_failure(
4295 spec,
4296 runtime,
4297 registry,
4298 process_liveness,
4299 snapshot,
4300 child,
4301 err,
4302 now_ms,
4303 )
4304 .await;
4305 }
4306 }
4307}
4308
4309pub(crate) struct HttpProbeTarget<'a> {
4310 address: std::net::SocketAddr,
4311 localhost: bool,
4312 authority: &'a str,
4313 path: String,
4314}
4315
4316pub(crate) fn parse_http_probe_url(url: &str) -> Result<HttpProbeTarget<'_>, String> {
4319 if url.bytes().any(|b| b <= b' ' || b == 127) || url.contains('#') {
4320 return Err("must not contain whitespace, controls, or a fragment".into());
4321 }
4322 let rest = url
4323 .strip_prefix("http://")
4324 .ok_or("must use plain http://")?;
4325 let split = rest.find(['/', '?']).unwrap_or(rest.len());
4326 let (authority, suffix) = rest.split_at(split);
4327 let (host, port) = if let Some(rest) = authority.strip_prefix("[::1]") {
4328 ("::1", rest)
4329 } else {
4330 let split = authority.find(':').unwrap_or(authority.len());
4331 authority.split_at(split)
4332 };
4333 let ip: std::net::IpAddr = match host {
4334 "127.0.0.1" | "localhost" => std::net::Ipv4Addr::LOCALHOST.into(),
4335 "::1" => std::net::Ipv6Addr::LOCALHOST.into(),
4336 _ => return Err("host must be 127.0.0.1, [::1], or localhost".into()),
4337 };
4338 let port = if port.is_empty() {
4339 80
4340 } else {
4341 port.strip_prefix(':')
4342 .and_then(|p| p.parse::<u16>().ok())
4343 .filter(|p| *p > 0)
4344 .ok_or("must have a valid nonzero TCP port")?
4345 };
4346 let path = if suffix.is_empty() {
4347 "/".into()
4348 } else if suffix.starts_with('?') {
4349 format!("/{suffix}")
4350 } else {
4351 suffix.into()
4352 };
4353 Ok(HttpProbeTarget {
4354 address: std::net::SocketAddr::new(ip, port),
4355 localhost: host == "localhost",
4356 authority,
4357 path,
4358 })
4359}
4360
4361async fn probe_http_health(
4362 url: &str,
4363 deadline: Duration,
4364) -> Result<HealthReport, HealthProbeError> {
4365 use tokio::io::{AsyncReadExt, AsyncWriteExt, BufReader};
4366 let target = parse_http_probe_url(url).map_err(HealthProbeError::misconfigured)?;
4367 let mut response_status = String::new();
4370 let mut body = Vec::new();
4371 let probe = async {
4372 let connection = match tokio::net::TcpStream::connect(target.address).await {
4375 Err(_) if target.localhost => {
4376 tokio::net::TcpStream::connect((
4377 std::net::Ipv6Addr::LOCALHOST,
4378 target.address.port(),
4379 ))
4380 .await
4381 }
4382 result => result,
4383 };
4384 let mut stream = connection.map_err(|error| {
4385 HealthProbeError::no_answer(format!("HTTP connect failed: {error}"))
4386 })?;
4387 stream.write_all(format!("GET {} HTTP/1.1\r\nHost: {}\r\nAccept-Encoding: identity\r\nConnection: close\r\n\r\n", target.path, target.authority).as_bytes()).await
4388 .map_err(|error| HealthProbeError::no_answer(format!("HTTP write failed: {error}")))?;
4389 let mut reader = BufReader::new(stream);
4390 let mut budget = 16 * 1024;
4391 let status = http_line(&mut reader, &mut budget).await?;
4392 let mut words = status.split_ascii_whitespace();
4393 let version = words.next();
4394 let code = words
4395 .next()
4396 .filter(|word| word.len() == 3)
4397 .and_then(|word| word.parse::<u16>().ok());
4398 if !matches!(version, Some("HTTP/1.1" | "HTTP/1.0"))
4399 || !code.is_some_and(|code| (100..600).contains(&code))
4400 {
4401 return Err(HealthProbeError::bad_answer(format!(
4402 "invalid HTTP status: {status}"
4403 )));
4404 }
4405 let code = code.expect("validated status code");
4406 response_status = status.clone();
4407 let mut length = None;
4408 let mut chunked = false;
4409 loop {
4410 let line = http_line(&mut reader, &mut budget).await?;
4411 if line.is_empty() {
4412 break;
4413 }
4414 if let Some((name, value)) = line.split_once(':') {
4415 if name.eq_ignore_ascii_case("content-length") {
4416 length = Some(value.trim().parse::<u64>().map_err(|_| {
4417 HealthProbeError::bad_answer("invalid HTTP Content-Length")
4418 })?);
4419 } else if name.eq_ignore_ascii_case("transfer-encoding") {
4420 chunked = value.trim().eq_ignore_ascii_case("chunked");
4421 }
4422 }
4423 }
4424 if chunked {
4425 while body.len() < 200 {
4426 let line = http_line(&mut reader, &mut budget).await?;
4427 let size = u64::from_str_radix(line.split(';').next().unwrap_or("").trim(), 16)
4428 .map_err(|_| HealthProbeError::bad_answer("invalid HTTP chunk size"))?;
4429 if size == 0 {
4430 break;
4431 }
4432 let count = size.min((200 - body.len()) as u64) as usize;
4433 let start = body.len();
4434 (&mut reader)
4435 .take(count as u64)
4436 .read_to_end(&mut body)
4437 .await
4438 .map_err(|error| {
4439 HealthProbeError::no_answer(format!("HTTP body read failed: {error}"))
4440 })?;
4441 if body.len() - start != count {
4442 return Err(HealthProbeError::bad_answer("truncated HTTP chunk"));
4443 }
4444 if size > count as u64 || body.len() == 200 {
4445 break;
4446 }
4447 if !http_line(&mut reader, &mut budget).await?.is_empty() {
4448 return Err(HealthProbeError::bad_answer("invalid HTTP chunk delimiter"));
4449 }
4450 }
4451 } else {
4452 reader
4453 .take(length.unwrap_or(200).min(200))
4454 .read_to_end(&mut body)
4455 .await
4456 .map_err(|error| {
4457 HealthProbeError::no_answer(format!("HTTP body read failed: {error}"))
4458 })?;
4459 }
4460 if (200..300).contains(&code) {
4461 Ok(HealthReport::ok())
4462 } else {
4463 Err(HealthProbeError::bad_answer(
4464 "HTTP health endpoint returned non-2xx",
4465 ))
4466 }
4467 };
4468 let mut result = timeout(deadline, probe).await.unwrap_or_else(|_| {
4469 Err(HealthProbeError::no_answer(format!(
4470 "HTTP probe timed out after {deadline:?}"
4471 )))
4472 });
4473 if let Err(error) = &mut result {
4474 if !response_status.is_empty() {
4475 error.message = format!(
4476 "{}; {response_status}: {}",
4477 error.message,
4478 String::from_utf8_lossy(&body)
4479 );
4480 }
4481 }
4482 result
4483}
4484
4485async fn http_line(
4486 reader: &mut tokio::io::BufReader<tokio::net::TcpStream>,
4487 remaining: &mut usize,
4488) -> Result<String, HealthProbeError> {
4489 use tokio::io::{AsyncBufReadExt, AsyncReadExt};
4490 let mut line = Vec::new();
4491 (&mut *reader)
4492 .take(*remaining as u64)
4493 .read_until(b'\n', &mut line)
4494 .await
4495 .map_err(|error| {
4496 HealthProbeError::no_answer(format!("HTTP header read failed: {error}"))
4497 })?;
4498 *remaining -= line.len();
4499 if !line.ends_with(b"\r\n") {
4500 return Err(HealthProbeError::bad_answer(
4501 "HTTP headers are incomplete or exceed 16 KiB",
4502 ));
4503 }
4504 line.truncate(line.len() - 2);
4505 String::from_utf8(line).map_err(|_| HealthProbeError::bad_answer("HTTP header is not UTF-8"))
4506}
4507
4508async fn probe_module_health(
4509 module_id: &str,
4510 runtime: &SupervisorRuntimeConfig,
4511 drain_deadline: Option<Instant>,
4512) -> Result<HealthReport, HealthProbeError> {
4513 let Some(forwarding) = runtime.forwarding.as_ref() else {
4514 return Err(HealthProbeError::misconfigured(
4515 "supervisor was not configured with a forwarding table",
4516 ));
4517 };
4518 let probe_started_at = Instant::now();
4519 let mut deadline = probe_started_at + runtime.health.deadline;
4520 if let Some(drain_deadline) = drain_deadline {
4521 deadline = deadline.min(drain_deadline);
4522 }
4523 let pending = if drain_deadline.is_some() {
4524 forwarding.begin_drain_health_probe_rpc_for(
4525 module_id,
4526 MODULE_CONTROL_OP_HEALTH_CHECK,
4527 probe_started_at,
4528 deadline,
4529 )
4530 } else {
4531 forwarding.begin_health_probe_rpc_for(
4532 module_id,
4533 MODULE_CONTROL_OP_HEALTH_CHECK,
4534 probe_started_at,
4535 deadline,
4536 )
4537 }
4538 .map_err(|err| {
4539 HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
4542 })?;
4543 await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
4544}
4545
4546async fn probe_endpoint_health(
4553 endpoint: crate::ModuleEndpointId,
4554 runtime: &SupervisorRuntimeConfig,
4555 deadline_cap: Option<Instant>,
4556) -> Result<HealthReport, HealthProbeError> {
4557 let Some(forwarding) = runtime.forwarding.as_ref() else {
4558 return Err(HealthProbeError::misconfigured(
4559 "supervisor was not configured with a forwarding table",
4560 ));
4561 };
4562 let probe_started_at = Instant::now();
4563 let mut deadline = probe_started_at + runtime.health.deadline;
4564 if let Some(cap) = deadline_cap {
4565 deadline = deadline.min(cap);
4566 }
4567 let pending = forwarding
4568 .begin_endpoint_health_probe_rpc_for(
4569 endpoint,
4570 MODULE_CONTROL_OP_HEALTH_CHECK,
4571 probe_started_at,
4572 deadline,
4573 )
4574 .map_err(|err| {
4575 HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
4576 })?;
4577 await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
4578}
4579
4580async fn await_health_probe(
4582 forwarding: &ForwardingTable,
4583 pending: PendingModuleControlRpc,
4584 deadline: Instant,
4585 probe_budget: Duration,
4586) -> Result<HealthReport, HealthProbeError> {
4587 let PendingModuleControlRpc {
4588 endpoint,
4589 module_sink,
4590 negotiated_ver,
4591 corr,
4592 receiver,
4593 } = pending;
4594 let body = serde_json::to_vec(&ModuleControlRequest::HealthCheck {}).map_err(|err| {
4595 HealthProbeError::misconfigured(format!("failed to encode health.check: {err}"))
4596 })?;
4597 let frame = Frame::build_with_version(
4598 negotiated_ver,
4599 FrameType::Request,
4600 control_flags(),
4601 0,
4602 0,
4603 corr,
4604 body,
4605 )
4606 .map_err(|err| {
4607 HealthProbeError::misconfigured(format!("failed to build health.check frame: {err}"))
4608 })?;
4609
4610 match timeout_at(deadline, module_sink.send(frame)).await {
4616 Ok(Ok(())) => {}
4617 Ok(Err(err)) => {
4618 let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
4619 return Err(HealthProbeError::lane_dead(format!(
4622 "failed to send health.check: {err}"
4623 )));
4624 }
4625 Err(_elapsed) => {
4626 let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
4627 return Err(HealthProbeError::no_answer(
4631 "health.check send timed out before enqueue (module egress full)",
4632 ));
4633 }
4634 }
4635
4636 match timeout_at(deadline, receiver).await {
4637 Ok(Ok(ModuleControlRpcOutcome::Response(response))) => {
4641 response.health_report().ok_or_else(|| {
4642 HealthProbeError::bad_answer("health.check RPC returned a non-health response")
4643 })
4644 }
4645 Ok(Ok(ModuleControlRpcOutcome::Rejected(body))) => Err(HealthProbeError::bad_answer(
4646 format!("health.check rejected: {}", body.message),
4647 )),
4648 Ok(Ok(ModuleControlRpcOutcome::ModuleGone(message))) => {
4649 Err(HealthProbeError::lane_dead(message))
4650 }
4651 Ok(Ok(ModuleControlRpcOutcome::MalformedResponse(message))) => {
4652 Err(HealthProbeError::bad_answer(message))
4653 }
4654 Ok(Ok(ModuleControlRpcOutcome::UnexpectedOp { expected, actual })) => {
4655 Err(HealthProbeError::bad_answer(format!(
4656 "expected module-control op '{expected}', got '{actual}'"
4657 )))
4658 }
4659 Ok(Ok(ModuleControlRpcOutcome::DeadlineElapsed)) => Err(HealthProbeError::bad_answer(
4663 "module answered health.check after its daemon deadline",
4664 )),
4665 Ok(Err(_)) => Err(HealthProbeError::misconfigured(
4666 "health.check waiter was canceled before the module responded",
4667 )),
4668 Err(_) => {
4669 let _ = forwarding.tombstone_health_probe_rpc(endpoint, corr);
4670 Err(HealthProbeError::no_answer(format!(
4671 "module did not answer health.check within {probe_budget:?}"
4672 )))
4673 }
4674 }
4675}
4676
4677#[allow(clippy::too_many_arguments)]
4678async fn handle_health_report(
4679 spec: &ModuleSpec,
4680 runtime: &SupervisorRuntimeConfig,
4681 registry: &Registry,
4682 process_liveness: &SupervisorProcessLiveness,
4683 snapshot: &SharedSnapshot,
4684 child: &mut Option<SupervisedChild>,
4685 report: HealthReport,
4686 now_ms: u64,
4687) {
4688 let status = supervisor_health_status(report.status);
4689 let detail = report.detail.clone();
4690 let metrics = truncate_health_metrics(report.metrics);
4691 let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4692 state.health.status = status;
4693 state.health.last_probe_ms = Some(now_ms);
4694 state.health.detail = detail.clone();
4695 state.health.metrics = metrics.clone();
4696 state.health.consecutive_failures = 0;
4697 });
4698
4699 let action = match report.status {
4700 HealthStatus::Ok => return,
4701 HealthStatus::Degraded => runtime.health.on_degraded,
4702 HealthStatus::Failing => runtime.health.on_failing,
4703 };
4704 apply_l3_health_action(
4705 spec,
4706 runtime,
4707 registry,
4708 process_liveness,
4709 snapshot,
4710 child,
4711 status,
4712 detail.as_deref(),
4713 action,
4714 now_ms,
4715 )
4716 .await;
4717}
4718
4719#[allow(clippy::too_many_arguments)]
4720async fn handle_health_probe_failure(
4721 spec: &ModuleSpec,
4722 runtime: &SupervisorRuntimeConfig,
4723 registry: &Registry,
4724 process_liveness: &SupervisorProcessLiveness,
4725 snapshot: &SharedSnapshot,
4726 child: &mut Option<SupervisedChild>,
4727 err: HealthProbeError,
4728 now_ms: u64,
4729) {
4730 let threshold = runtime.health.failure_threshold.max(1);
4731 let mut failures = 0;
4732 let detail = format!("[{}] {err}", err.label());
4737 let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4738 if state.spawned_protocol.unwrap_or(spec.protocol) == ModuleProtocol::Subc {
4741 state.health.status = SupervisorHealthStatus::Unknown;
4742 }
4743 state.health.last_probe_ms = Some(now_ms);
4744 state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
4745 state.health.detail = Some(detail.clone());
4746 state.health.metrics = None;
4747 failures = state.health.consecutive_failures;
4748 });
4749
4750 if failures < threshold {
4751 warn!(
4752 module_id = %spec.module_id,
4753 consecutive_failures = failures,
4754 threshold,
4755 evidence = err.label(),
4756 detail = %detail,
4757 "health.check probe failed"
4758 );
4759 return;
4760 }
4761
4762 let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4763 state.state = ModuleState::Unresponsive;
4764 state.health.status = SupervisorHealthStatus::Unresponsive;
4765 });
4766 if runtime.health.critical {
4770 error!(
4771 module_id = %spec.module_id,
4772 status = "unresponsive",
4773 evidence = err.label(),
4774 detail = %detail,
4775 "critical module health alert"
4776 );
4777 } else {
4778 warn!(
4779 module_id = %spec.module_id,
4780 status = "unresponsive",
4781 evidence = err.label(),
4782 detail = %detail,
4783 "module health threshold breached"
4784 );
4785 }
4786 if let Err(err) = health_restart_child(
4787 spec,
4788 runtime,
4789 registry,
4790 process_liveness,
4791 snapshot,
4792 child,
4793 SupervisorHealthStatus::Unresponsive,
4794 Some(&detail),
4795 now_ms,
4796 )
4797 .await
4798 {
4799 error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
4800 }
4801}
4802
4803#[allow(clippy::too_many_arguments)]
4804async fn apply_l3_health_action(
4805 spec: &ModuleSpec,
4806 runtime: &SupervisorRuntimeConfig,
4807 registry: &Registry,
4808 process_liveness: &SupervisorProcessLiveness,
4809 snapshot: &SharedSnapshot,
4810 child: &mut Option<SupervisedChild>,
4811 status: SupervisorHealthStatus,
4812 detail: Option<&str>,
4813 action: HealthAction,
4814 now_ms: u64,
4815) {
4816 record_health_action(snapshot, &spec.module_id, action.to_string(), now_ms);
4817 match action {
4818 HealthAction::Report => {
4819 info!(
4820 module_id = %spec.module_id,
4821 status = ?status,
4822 detail,
4823 "module reported non-ok health"
4824 );
4825 }
4826 HealthAction::Alert => {
4827 error!(
4828 module_id = %spec.module_id,
4829 status = ?status,
4830 detail,
4831 "module health alert"
4832 );
4833 }
4834 HealthAction::Restart => {
4835 if let Err(err) = health_restart_child(
4836 spec,
4837 runtime,
4838 registry,
4839 process_liveness,
4840 snapshot,
4841 child,
4842 status,
4843 detail,
4844 now_ms,
4845 )
4846 .await
4847 {
4848 error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
4849 }
4850 }
4851 }
4852}
4853
4854#[allow(clippy::too_many_arguments)]
4855async fn health_restart_child(
4856 spec: &ModuleSpec,
4857 runtime: &SupervisorRuntimeConfig,
4858 registry: &Registry,
4859 process_liveness: &SupervisorProcessLiveness,
4860 snapshot: &SharedSnapshot,
4861 child: &mut Option<SupervisedChild>,
4862 status: SupervisorHealthStatus,
4863 detail: Option<&str>,
4864 now_ms: u64,
4865) -> Result<(), SuperviseError> {
4866 let (enabled, schedule) = {
4867 let mut state = lock_snapshot(snapshot)?;
4868 let enabled = state.enabled;
4869 let schedule = if enabled {
4870 state.next_crash_restart(&runtime.restart_policy, Instant::now())
4871 } else {
4872 None
4873 };
4874 (enabled, schedule)
4875 };
4876
4877 if !enabled {
4878 return Err(SuperviseError::Disabled {
4879 module_id: spec.module_id.clone(),
4880 });
4881 }
4882
4883 if schedule.is_none() {
4884 record_health_action(snapshot, &spec.module_id, "failed".to_string(), now_ms);
4885 error!(
4886 module_id = %spec.module_id,
4887 status = ?status,
4888 detail,
4889 max_restarts = runtime.restart_policy.max_restarts,
4890 window_secs = runtime.restart_policy.window.as_secs(),
4891 reason = %runtime.restart_policy.budget_exhausted_detail(),
4892 "health restart budget exhausted; marking module failed"
4893 );
4894 let stop_notice = begin_forwarding_drain_if_configured(
4895 spec,
4896 runtime,
4897 registry,
4898 snapshot,
4899 Some(true),
4900 RouteCloseReason::Disable,
4901 )
4902 .await?;
4903 update_snapshot(snapshot, Some(&spec.module_id), |state| {
4904 state.drain_disposition_detail = Some(runtime.restart_policy.budget_exhausted_detail());
4905 })?;
4906 drain_optional_child(
4907 &spec.module_id,
4908 spec.protocol,
4909 stop_notice,
4910 registry,
4911 runtime.forwarding.as_deref(),
4912 snapshot,
4913 &runtime.terminal_ring,
4914 &runtime.spawn_events,
4915 child,
4916 runtime.drain_timeout,
4917 ModuleState::Failed,
4918 Some(true),
4919 )
4920 .await?;
4921 process_liveness.untrack_if_current(&spec.module_id, snapshot);
4922 return Ok(());
4923 }
4924
4925 let schedule = schedule.expect("a health restart must have a crash-restart schedule");
4926 let mut restart_count = 0;
4927 update_snapshot(snapshot, Some(&spec.module_id), |state| {
4928 restart_count = state.crash_restarts.len();
4929 state.state = ModuleState::Unresponsive;
4930 state.health.status = status;
4931 state.health.last_action = Some(HealthAction::Restart.to_string());
4932 state.health.last_action_ms = Some(now_ms);
4933 })?;
4934 warn!(
4935 module_id = %spec.module_id,
4936 status = ?status,
4937 detail,
4938 restart_count,
4939 restart_in_window = schedule.restart_in_window,
4940 delay_ms = schedule.delay.as_millis() as u64,
4941 "health-triggered module restart"
4942 );
4943
4944 let stop_notice = begin_forwarding_drain_if_configured(
4945 spec,
4946 runtime,
4947 registry,
4948 snapshot,
4949 Some(true),
4950 RouteCloseReason::Restart,
4951 )
4952 .await?;
4953 drain_optional_child(
4954 &spec.module_id,
4955 spec.protocol,
4956 stop_notice,
4957 registry,
4958 runtime.forwarding.as_deref(),
4959 snapshot,
4960 &runtime.terminal_ring,
4961 &runtime.spawn_events,
4962 child,
4963 runtime.drain_timeout,
4964 ModuleState::Restarting,
4965 Some(true),
4966 )
4967 .await?;
4968 schedule_respawn(
4969 runtime,
4970 snapshot,
4971 &spec.module_id,
4972 schedule.delay,
4973 RespawnKind::Spawn,
4974 )
4975}
4976
4977fn cancel_deferred_reload(runtime: &SupervisorRuntimeConfig, module_id: &str, reason: &str) {
4978 if let Some(reply) = runtime
4979 .deferred_reload_reply
4980 .lock()
4981 .unwrap_or_else(|p| p.into_inner())
4982 .take()
4983 {
4984 let _ = reply.send(Err(SuperviseError::ReloadFailed {
4985 module_id: module_id.to_string(),
4986 reason: reason.to_string(),
4987 }));
4988 }
4989}
4990
4991fn schedule_respawn(
4992 runtime: &SupervisorRuntimeConfig,
4993 snapshot: &SharedSnapshot,
4994 module_id: &str,
4995 delay: Duration,
4996 kind: RespawnKind,
4997) -> Result<(), SuperviseError> {
4998 cancel_deferred_reload(runtime, module_id, "respawn superseded by another restart");
4999 update_snapshot(snapshot, Some(module_id), |state| {
5000 state.respawn_pending = true
5001 })?;
5002 *runtime
5003 .scheduled_respawn
5004 .lock()
5005 .unwrap_or_else(|p| p.into_inner()) = Some(PendingRespawn {
5006 deadline: Instant::now() + delay,
5007 kind,
5008 });
5009 Ok(())
5010}
5011
5012fn record_health_action(snapshot: &SharedSnapshot, module_id: &str, action: String, now_ms: u64) {
5013 let _ = update_snapshot(snapshot, Some(module_id), |state| {
5014 state.health.last_action = Some(action);
5015 state.health.last_action_ms = Some(now_ms);
5016 });
5017}
5018
5019fn supervisor_health_status(status: HealthStatus) -> SupervisorHealthStatus {
5020 match status {
5021 HealthStatus::Ok => SupervisorHealthStatus::Ok,
5022 HealthStatus::Degraded => SupervisorHealthStatus::Degraded,
5023 HealthStatus::Failing => SupervisorHealthStatus::Failing,
5024 }
5025}
5026
5027fn truncate_health_metrics(metrics: Option<Value>) -> Option<Value> {
5039 let metrics = metrics?;
5040 match serde_json::to_vec(&metrics) {
5041 Ok(encoded) if encoded.len() > MAX_HEALTH_METRICS_BYTES => Some(serde_json::json!({
5042 "truncated": true,
5043 "original_bytes": encoded.len(),
5044 })),
5045 Ok(_) | Err(_) => Some(metrics),
5046 }
5047}
5048
5049fn jittered_health_delay(module_id: &str, probe_index: u64, cadence: Duration) -> Duration {
5055 if cadence.is_zero() {
5056 return Duration::ZERO;
5057 }
5058 let cadence_ms = cadence.as_millis() as u64;
5059 if cadence_ms == 0 {
5075 return cadence;
5076 }
5077 let jitter_span = (cadence_ms / 10).max(1);
5092 let hash = module_id.as_bytes().iter().fold(
5093 probe_index.wrapping_mul(0x9E37_79B9_7F4A_7C15),
5094 |acc, byte| {
5095 acc.wrapping_mul(1099511628211)
5096 .wrapping_add(u64::from(*byte))
5097 },
5098 );
5099 cadence + Duration::from_millis(hash % jitter_span)
5100}
5101
5102#[cfg(test)]
5103mod tests {
5104 use super::*;
5105
5106 #[test]
5107 fn readding_a_module_clears_its_rescan_removal_tombstone() {
5108 let handle = SupervisorHandle::new();
5109 let module_id = "readded-tombstone";
5110 handle.record_rescan_removal(module_id);
5111 assert!(handle.removal_tombstone_age_ms(module_id).is_some());
5112
5113 handle.apply_identity_configuration(&ModuleSpec {
5114 module_id: module_id.to_string(),
5115 program: PathBuf::from("/test/module"),
5116 args: Vec::new(),
5117 env: Vec::new(),
5118 reserved: false,
5119 reserved_prefixes: Vec::new(),
5120 protocol: ModuleProtocol::Subc,
5121 overlap: Default::default(),
5122 });
5123
5124 assert!(
5125 handle.removal_tombstone_age_ms(module_id).is_none(),
5126 "a re-added module must not retain a stale removal tombstone"
5127 );
5128 }
5129
5130 #[derive(Debug, PartialEq, Eq)]
5133 struct OwnerInSpawnWindow {
5134 module_id: String,
5135 configured: bool,
5136 on_roster: bool,
5137 admission_refusal: Option<&'static str>,
5138 }
5139
5140 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5151 async fn a_new_module_is_configured_before_its_first_process_can_register() {
5152 use crate::scopes::ScopeTable;
5153 use subc_protocol::{error_codes, scope::ScopeSelector, Principal};
5154
5155 let dir = subc_test_support::TestTempDir::new("configured-before-spawn");
5156 let stub = |module_id: &str, program: PathBuf| ModuleSpec {
5157 module_id: module_id.to_string(),
5158 program,
5159 args: Vec::new(),
5160 env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
5161 .into_iter()
5162 .map(|key| (key.to_string(), dir.path().display().to_string()))
5163 .chain([("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string())])
5164 .collect(),
5165 reserved: false,
5166 reserved_prefixes: Vec::new(),
5167 protocol: ModuleProtocol::Subc,
5168 overlap: Default::default(),
5169 };
5170 let live = super::terminal_history_tests::fake_aft_stub_path();
5171 let missing = dir.path().join("definitely-missing-module");
5172
5173 let handle = SupervisorHandle::new();
5174 let mut supervisor =
5175 Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5176 .with_handle(handle.clone());
5177 let observed = Arc::new(Mutex::new(Vec::<OwnerInSpawnWindow>::new()));
5178 let hook_handle = handle.clone();
5179 let hook_observed = Arc::clone(&observed);
5180 supervisor.test_after_first_spawn = AfterFirstSpawnHook(Some(Arc::new(move |module_id| {
5181 let configured = hook_handle.is_configured(module_id);
5184 let selector = ScopeSelector {
5185 owner: Principal::Reserved {
5186 module_id: module_id.to_string(),
5187 },
5188 scope_ref: "s".to_string(),
5189 scope_epoch: Some(1),
5190 };
5191 let carrier = Principal::Reserved {
5192 module_id: "carrier".to_string(),
5193 };
5194 let admission_refusal = match ScopeTable::new(Vec::<String>::new())
5195 .admit(&carrier, module_id, &selector, configured)
5196 {
5197 Ok(_) => None,
5198 Err(refusal) => Some(refusal.code),
5199 };
5200 hook_observed.lock().unwrap().push(OwnerInSpawnWindow {
5201 module_id: module_id.to_string(),
5202 configured,
5203 on_roster: hook_handle.get(module_id).is_some(),
5204 admission_refusal,
5205 });
5206 })));
5207
5208 let plain = supervisor.spawn(stub("plain", live.clone())).unwrap();
5209 let configured = supervisor
5210 .supervise_configured(stub("configured", live.clone()), true)
5211 .unwrap();
5212 let with_health = supervisor
5213 .supervise_configured_with_health(
5214 stub("with-health", live.clone()),
5215 true,
5216 HealthConfig::default(),
5217 None,
5218 RestartPolicy::default(),
5219 )
5220 .unwrap();
5221 let failed = supervisor
5224 .supervise_configured_with_health(
5225 stub("failed-spawn", missing.clone()),
5226 true,
5227 HealthConfig::default(),
5228 None,
5229 RestartPolicy::default(),
5230 )
5231 .unwrap();
5232 assert!(supervisor.spawn(stub("spawn-error", missing)).is_err());
5235
5236 let expected = [
5237 "plain",
5238 "configured",
5239 "with-health",
5240 "failed-spawn",
5241 "spawn-error",
5242 ]
5243 .into_iter()
5244 .map(|module_id| OwnerInSpawnWindow {
5245 module_id: module_id.to_string(),
5246 configured: true,
5247 on_roster: false,
5248 admission_refusal: Some(error_codes::SCOPE_NOT_SYNCED),
5249 })
5250 .collect::<Vec<_>>();
5251 assert_eq!(*observed.lock().unwrap(), expected);
5252
5253 for module_id in ["plain", "configured", "with-health", "failed-spawn"] {
5254 assert!(
5255 handle.get(module_id).is_some(),
5256 "{module_id} is on the roster"
5257 );
5258 assert!(
5259 handle.is_configured(module_id),
5260 "{module_id} stays configured"
5261 );
5262 }
5263 assert!(handle.get("spawn-error").is_none());
5264 assert!(
5265 !handle.is_configured("spawn-error"),
5266 "a plain spawn that failed must not leave its module marked configured"
5267 );
5268
5269 handle.retire("failed-spawn");
5271 assert!(!handle.is_configured("failed-spawn"));
5272
5273 for module in [plain, configured, with_health] {
5274 module.stop().await.unwrap();
5275 }
5276 drop(failed);
5277 }
5278
5279 fn stale_process_snapshot(state: ModuleState, enabled: bool) -> SharedSnapshot {
5280 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(state, enabled)));
5281 update_snapshot(&snapshot, Some("stale-process-facts"), |snapshot| {
5282 snapshot.process_alive = true;
5283 snapshot.pid = Some(41);
5284 snapshot.spawned_at_ms = Some(42);
5285 snapshot.spawned_from = Some(PathBuf::from("/spawned/module"));
5286 snapshot.spawned_file_identity = Some(SpawnedFileIdentity {
5287 device: 43,
5288 inode: 44,
5289 });
5290 })
5291 .unwrap();
5292 snapshot
5293 }
5294
5295 fn assert_snapshot_process_facts_cleared(snapshot: &SharedSnapshot) {
5296 let snapshot = lock_snapshot(snapshot).unwrap();
5297 assert!(!snapshot.process_alive);
5298 assert_eq!(snapshot.pid, None);
5299 assert_eq!(snapshot.spawned_at_ms, None);
5300 assert_eq!(snapshot.spawned_from, None);
5301 assert_eq!(snapshot.spawned_file_identity, None);
5302 }
5303
5304 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5305 async fn failed_enable_spawn_clears_preexisting_current_process_facts() {
5306 let supervisor =
5307 Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5308 let mut runtime = supervisor.runtime_config();
5309 runtime.test_seed_stale_facts_before_enable_spawn = true;
5310 let snapshot = stale_process_snapshot(ModuleState::Disabled, false);
5311 let mut child = None;
5312 let spec = ModuleSpec {
5313 module_id: "failed-enable-clears-facts".to_string(),
5314 program: PathBuf::from("/definitely/missing/failed-enable-module"),
5315 args: Vec::new(),
5316 env: Vec::new(),
5317 reserved: false,
5318 reserved_prefixes: Vec::new(),
5319 protocol: ModuleProtocol::Subc,
5320 overlap: Default::default(),
5321 };
5322
5323 let result = set_child_enabled(
5324 &spec,
5325 &runtime,
5326 &supervisor.registry,
5327 &supervisor.process_liveness,
5328 &snapshot,
5329 &mut child,
5330 true,
5331 )
5332 .await;
5333
5334 assert!(matches!(result, Err(SuperviseError::Spawn { .. })));
5335 assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
5336 assert_snapshot_process_facts_cleared(&snapshot);
5337 }
5338
5339 #[tokio::test]
5340 async fn start_revives_stranded_restarting_but_not_pending_backoff() {
5341 let supervisor =
5342 Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5343 let runtime = supervisor.runtime_config();
5344 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(
5345 ModuleState::Restarting,
5346 true,
5347 )));
5348 let spec = ModuleSpec {
5349 module_id: "start-stranded-restarting".to_string(),
5350 program: super::terminal_history_tests::fake_aft_stub_path(),
5351 args: Vec::new(),
5352 env: vec![("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string())],
5353 reserved: false,
5354 reserved_prefixes: Vec::new(),
5355 protocol: ModuleProtocol::None,
5356 overlap: Default::default(),
5357 };
5358 let mut child = None;
5359 lock_snapshot(&snapshot).unwrap().respawn_pending = true;
5360 assert!(!super::set_child_enabled(
5361 &spec,
5362 &runtime,
5363 &Registry::default(),
5364 &supervisor.process_liveness,
5365 &snapshot,
5366 &mut child,
5367 true
5368 )
5369 .await
5370 .unwrap());
5371 assert!(child.is_none());
5372 lock_snapshot(&snapshot).unwrap().respawn_pending = false;
5373 assert!(super::set_child_enabled(
5374 &spec,
5375 &runtime,
5376 &Registry::default(),
5377 &supervisor.process_liveness,
5378 &snapshot,
5379 &mut child,
5380 true
5381 )
5382 .await
5383 .unwrap());
5384 assert_eq!(
5385 lock_snapshot(&snapshot).unwrap().state,
5386 ModuleState::Running
5387 );
5388 let mut child = child.unwrap();
5389 child.start_kill().unwrap();
5390 child.wait().await.unwrap();
5391 }
5392
5393 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5394 async fn failed_reload_spawn_clears_current_process_facts() {
5395 let supervisor =
5396 Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5397 let mut runtime = supervisor.runtime_config();
5398 runtime.restart_policy = RestartPolicy::new(0, Duration::ZERO);
5399 let snapshot = stale_process_snapshot(ModuleState::Running, true);
5400 let mut child = None;
5401 let spec = ModuleSpec {
5402 module_id: "failed-reload-clears-facts".to_string(),
5403 program: PathBuf::from("/unused/failed-reload-module"),
5404 args: Vec::new(),
5405 env: Vec::new(),
5406 reserved: false,
5407 reserved_prefixes: Vec::new(),
5408 protocol: ModuleProtocol::Subc,
5409 overlap: Default::default(),
5410 };
5411
5412 let result = handle_reload_spawn_failure(
5413 &spec,
5414 &runtime,
5415 &supervisor.process_liveness,
5416 &snapshot,
5417 &mut child,
5418 "forced reload spawn failure".to_string(),
5419 )
5420 .await;
5421
5422 assert!(matches!(result, Err(SuperviseError::ReloadFailed { .. })));
5423 assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
5424 assert_snapshot_process_facts_cleared(&snapshot);
5425 }
5426
5427 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5428 async fn dropping_a_module_with_an_active_monitor_clears_current_process_facts() {
5429 let supervisor =
5430 Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5431 let snapshot = stale_process_snapshot(ModuleState::Running, true);
5432 let module = supervisor.supervised_module(
5433 ModuleSpec {
5434 module_id: "drop-clears-facts".to_string(),
5435 program: PathBuf::from("/unused/drop-module"),
5436 args: Vec::new(),
5437 env: Vec::new(),
5438 reserved: false,
5439 reserved_prefixes: Vec::new(),
5440 protocol: ModuleProtocol::Subc,
5441 overlap: Default::default(),
5442 },
5443 supervisor.runtime_config(),
5444 Arc::clone(&snapshot),
5445 None,
5446 );
5447 assert!(!module
5448 .inner
5449 .monitor
5450 .lock()
5451 .unwrap()
5452 .as_ref()
5453 .unwrap()
5454 .is_finished());
5455
5456 drop(module);
5457
5458 assert_eq!(
5459 lock_snapshot(&snapshot).unwrap().state,
5460 ModuleState::Stopped
5461 );
5462 assert_snapshot_process_facts_cleared(&snapshot);
5463 }
5464
5465 #[cfg(unix)]
5466 #[tokio::test]
5467 async fn rescan_preserves_running_protocol_until_respawn() {
5468 let dir = subc_test_support::TestTempDir::new("rescan-protocol");
5469 let initial = ModuleSpec {
5470 module_id: "rescan-protocol".into(),
5471 program: PathBuf::from("/bin/sleep"),
5472 args: vec!["60".into()],
5473 env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
5474 .into_iter()
5475 .map(|key| (key.into(), dir.path().to_string_lossy().into_owned()))
5476 .collect(),
5477 reserved: false,
5478 reserved_prefixes: vec![],
5479 protocol: ModuleProtocol::None,
5480 overlap: Default::default(),
5481 };
5482 let supervisor =
5483 Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5484 let module = supervisor.spawn(initial.clone()).unwrap();
5485 assert!(module.status().unwrap().live);
5486 let mut next = initial;
5487 next.protocol = ModuleProtocol::Subc;
5488 module
5489 .update_configuration(next.clone(), HealthConfig::default(), None)
5490 .await
5491 .unwrap();
5492 assert!(
5493 module.status().unwrap().live,
5494 "rescan must not require HELLO from the old non-wire process"
5495 );
5496 let runtime = supervisor.runtime_config();
5497 let action = on_child_exit(
5498 &next,
5499 RestartPolicy::default(),
5500 &supervisor.registry,
5501 &module.inner.snapshot,
5502 &runtime.terminal_ring,
5503 &runtime.spawn_events,
5504 &runtime.child_roster,
5505 ExitReport {
5506 kind: ExitKind::Clean,
5507 code: Some(0),
5508 signal: None,
5509 at_ms: unix_ms_now(),
5510 },
5511 )
5512 .await;
5513 assert!(
5514 matches!(action, NextAction::Restart { .. }),
5515 "the old non-wire process's clean exit must restart"
5516 );
5517 module.drain().await.unwrap();
5518 }
5519
5520 #[cfg(unix)]
5521 #[tokio::test]
5522 async fn orphan_identity_matches_path_names_and_shebang_interpreters() {
5523 use std::os::unix::fs::PermissionsExt;
5524 let dir = subc_test_support::TestTempDir::new("spawn-image-identity");
5525 let script = dir.join("module.sh");
5526 std::fs::write(&script, "#!/bin/sh\nwhile :; do sleep 0.1; done\n").unwrap();
5527 std::fs::set_permissions(&script, std::fs::Permissions::from_mode(0o700)).unwrap();
5528 let record_path = dir.join("live-children.json");
5529 let supervisor =
5530 Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5531 .with_live_children_record(&record_path);
5532 for (program, args) in [
5533 (PathBuf::from("sleep"), vec!["60".into()]),
5534 (script, vec![]),
5535 ] {
5536 let spec = ModuleSpec {
5537 module_id: "image-identity".into(),
5538 program,
5539 args,
5540 env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
5541 .into_iter()
5542 .map(|key| (key.into(), dir.to_string_lossy().into_owned()))
5543 .collect(),
5544 reserved: false,
5545 reserved_prefixes: vec![],
5546 protocol: ModuleProtocol::None,
5547 overlap: Default::default(),
5548 };
5549 let module = supervisor.spawn(spec).unwrap();
5550 #[cfg(target_os = "macos")]
5551 {
5552 let deadline = Instant::now() + Duration::from_secs(5);
5555 while crate::live_children::read_record(&record_path)
5556 .unwrap()
5557 .iter()
5558 .all(|entry| entry.executable.is_none())
5559 {
5560 assert!(Instant::now() < deadline, "module image was not confirmed");
5561 tokio::time::sleep(Duration::from_millis(5)).await;
5562 }
5563 }
5564 let entry = crate::live_children::read_record(&record_path)
5565 .unwrap()
5566 .pop()
5567 .unwrap();
5568 let observed = subc_os::Process::open(entry.pid)
5569 .unwrap()
5570 .unwrap()
5571 .observe()
5572 .unwrap();
5573 let verdict = crate::live_children::identity_verdict(&entry, entry.pid, &observed);
5574 module.drain().await.unwrap();
5575 assert_eq!(verdict, crate::live_children::IdentityVerdict::Matches);
5576 }
5577 }
5578
5579 #[cfg(unix)]
5580 fn http_fixture(
5581 dir: &std::path::Path,
5582 url: &str,
5583 threshold: u32,
5584 ) -> crate::daemon_config::ConfiguredModule {
5585 let path = dir.join("subc.jsonc");
5586 std::fs::write(&path, serde_json::json!({"version": 1, "modules": {"http-process": {
5587 "program": "/bin/sleep", "args": ["60"], "protocol": "none",
5588 "env": {"XDG_DATA_HOME":dir,"XDG_RUNTIME_DIR":dir,"XDG_CONFIG_HOME":dir},
5589 "health": {"http":url,"cadence_ms":20,"deadline_ms":200,"failure_threshold":threshold},
5590 "restart": {"backoff_ms":1,"max_backoff_ms":1}, "drain_timeout_ms":10
5591 }}}).to_string()).unwrap();
5592 crate::daemon_config::load(&path)
5593 .unwrap()
5594 .unwrap()
5595 .modules
5596 .pop()
5597 .unwrap()
5598 }
5599
5600 #[cfg(unix)]
5601 async fn wait_http_health(module: &SupervisedModule, status: SupervisorHealthStatus) {
5602 timeout(Duration::from_secs(5), async {
5603 loop {
5604 if module.status().unwrap().health.status == status {
5605 break;
5606 }
5607 sleep(Duration::from_millis(5)).await;
5608 }
5609 })
5610 .await
5611 .unwrap_or_else(|_| {
5612 panic!(
5613 "expected {status:?}, got {:?}",
5614 module.status().unwrap().health
5615 )
5616 });
5617 }
5618
5619 #[cfg(unix)]
5620 #[tokio::test]
5621 async fn http_health_status_flips_ok_failing_ok() {
5622 use tokio::io::{AsyncReadExt, AsyncWriteExt};
5623 let dir = subc_test_support::TestTempDir::new("http-status-flips");
5624 let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5625 let status = Arc::new(std::sync::atomic::AtomicU16::new(200));
5626 let serving_status = status.clone();
5627 let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5628 let server = tokio::spawn(async move {
5629 loop {
5630 let (mut stream, _) = listener.accept().await.unwrap();
5631 let mut request = [0u8; 2048];
5632 let count = stream.read(&mut request).await.unwrap();
5633 assert!(count > 0, "a probe must send an HTTP request");
5634 let code = serving_status.load(std::sync::atomic::Ordering::SeqCst);
5635 let body = if code == 200 {
5636 "ready"
5637 } else {
5638 "{\"status\":\"unavailable\",\"error\":\"scratch failure\"}"
5639 };
5640 let response = format!("HTTP/1.1 {code} scratch\r\nContent-Length: {}\r\nConnection: close\r\n\r\n{body}", body.len());
5641 let _ = stream.write_all(response.as_bytes()).await;
5642 }
5643 });
5644 let configured = http_fixture(&dir, &url, 1000);
5645 let module =
5646 Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5647 .supervise_configured_with_health(
5648 configured.module_spec(),
5649 true,
5650 configured.health,
5651 configured.drain_timeout_ms,
5652 configured.restart,
5653 )
5654 .unwrap();
5655 wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5656 status.store(503, std::sync::atomic::Ordering::SeqCst);
5657 wait_http_health(&module, SupervisorHealthStatus::Failing).await;
5658 assert!(module
5659 .status()
5660 .unwrap()
5661 .health
5662 .detail
5663 .unwrap()
5664 .contains("scratch failure"));
5665 status.store(200, std::sync::atomic::Ordering::SeqCst);
5666 wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5667 assert_eq!(module.status().unwrap().health.consecutive_failures, 0);
5668 let before = module.status().unwrap();
5669 let (spec, mut health) = module.configuration().unwrap();
5670 health.http = None;
5671 module
5672 .update_configuration(spec.clone(), health.clone(), Some(10))
5673 .await
5674 .unwrap();
5675 wait_http_health(&module, SupervisorHealthStatus::Unknown).await;
5676 health.http = Some(url);
5677 module
5678 .update_configuration(spec, health, Some(10))
5679 .await
5680 .unwrap();
5681 wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5682 assert_eq!(
5683 module.status().unwrap().pid,
5684 before.pid,
5685 "changing a probe must apply live, not restart its process"
5686 );
5687 let (spec, mut health) = module.configuration().unwrap();
5688 health.failure_threshold = 2;
5689 module
5690 .update_configuration(spec, health, Some(10))
5691 .await
5692 .unwrap();
5693 status.store(503, std::sync::atomic::Ordering::SeqCst);
5694 timeout(Duration::from_secs(5), async {
5695 while module.status().unwrap().spawn_generation == before.spawn_generation {
5696 sleep(Duration::from_millis(5)).await;
5697 }
5698 })
5699 .await
5700 .expect("sustained HTTP 503 responses must trigger the health restart policy");
5701 module.drain().await.unwrap();
5702 server.abort();
5703 }
5704
5705 #[cfg(unix)]
5706 #[tokio::test]
5707 async fn http_health_no_listener_restarts_after_consecutive_failures() {
5708 let dir = subc_test_support::TestTempDir::new("http-refused");
5709 let unused = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5710 let url = format!("http://{}/healthz", unused.local_addr().unwrap());
5711 drop(unused);
5712 let configured = http_fixture(&dir, &url, 2);
5713 let module =
5714 Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5715 .supervise_configured_with_health(
5716 configured.module_spec(),
5717 true,
5718 configured.health,
5719 configured.drain_timeout_ms,
5720 configured.restart,
5721 )
5722 .unwrap();
5723 let before = module.status().unwrap().spawn_generation;
5724 timeout(Duration::from_secs(5), async {
5725 loop {
5726 let status = module.status().unwrap();
5727 if status.spawn_generation > before {
5728 assert!(status.lifetime_restarts > 0);
5729 break;
5730 }
5731 sleep(Duration::from_millis(5)).await;
5732 }
5733 })
5734 .await
5735 .expect("sustained HTTP refusal must trigger the health restart policy");
5736 module.drain().await.unwrap();
5737 }
5738
5739 #[cfg(unix)]
5740 #[tokio::test]
5741 async fn http_health_timeout_honours_deadline() {
5742 let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5743 let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5744 let server = tokio::spawn(async move {
5745 let _held = listener.accept().await.unwrap();
5746 std::future::pending::<()>().await;
5747 });
5748 let error = timeout(
5749 Duration::from_secs(1),
5750 probe_http_health(&url, Duration::from_millis(10)),
5751 )
5752 .await
5753 .expect("the probe must enforce its own deadline")
5754 .unwrap_err();
5755 server.abort();
5756 assert!(error.to_string().contains("timed out"));
5757 }
5758
5759 #[cfg(unix)]
5760 #[tokio::test]
5761 async fn http_health_localhost_reaches_an_ipv6_only_listener() {
5762 use tokio::io::{AsyncReadExt, AsyncWriteExt};
5763 let listener = tokio::net::TcpListener::bind("[::1]:0").await.unwrap();
5764 let url = format!(
5765 "http://localhost:{}/healthz",
5766 listener.local_addr().unwrap().port()
5767 );
5768 let server = tokio::spawn(async move {
5769 let (mut stream, _) = listener.accept().await.unwrap();
5770 let mut request = [0u8; 2048];
5771 assert!(stream.read(&mut request).await.unwrap() > 0);
5772 stream
5773 .write_all(b"HTTP/1.1 200 OK\r\nContent-Length: 2\r\n\r\nok")
5774 .await
5775 .unwrap();
5776 });
5777 assert_eq!(
5782 probe_http_health(&url, Duration::from_secs(10))
5783 .await
5784 .unwrap()
5785 .status,
5786 HealthStatus::Ok
5787 );
5788 server.await.unwrap();
5789 }
5790
5791 #[cfg(unix)]
5792 #[tokio::test]
5793 async fn http_health_timeout_keeps_partial_status_and_body() {
5794 use tokio::io::{AsyncReadExt, AsyncWriteExt};
5795 let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5796 let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5797 let server = tokio::spawn(async move {
5798 let (mut stream, _) = listener.accept().await.unwrap();
5799 let mut request = [0u8; 2048];
5800 assert!(stream.read(&mut request).await.unwrap() > 0);
5801 stream
5802 .write_all(
5803 b"HTTP/1.1 503 Unavailable\r\nContent-Length: 100\r\n\r\npartial diagnostic",
5804 )
5805 .await
5806 .unwrap();
5807 std::future::pending::<()>().await;
5808 });
5809 let error = probe_http_health(&url, Duration::from_secs(1))
5810 .await
5811 .unwrap_err()
5812 .to_string();
5813 server.abort();
5814 assert!(
5815 error.contains("timed out")
5816 && error.contains("503 Unavailable")
5817 && error.contains("partial diagnostic"),
5818 "{error}"
5819 );
5820 }
5821
5822 #[cfg(unix)]
5823 #[tokio::test]
5824 async fn http_health_diagnostic_body_is_bounded_to_200_bytes() {
5825 use tokio::io::{AsyncReadExt, AsyncWriteExt};
5826 let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5827 let url = format!("http://{}/healthz", listener.local_addr().unwrap());
5828 let server = tokio::spawn(async move {
5829 let (mut stream, _) = listener.accept().await.unwrap();
5830 let mut request = [0u8; 2048];
5831 assert!(stream.read(&mut request).await.unwrap() > 0);
5832 let body = format!("{}not-in-diagnostic", "x".repeat(400));
5833 let response = format!(
5834 "HTTP/1.1 503 Unavailable\r\nContent-Length: {}\r\n\r\n{body}",
5835 body.len()
5836 );
5837 stream.write_all(response.as_bytes()).await.unwrap();
5838 });
5839 let error = probe_http_health(&url, Duration::from_secs(1))
5840 .await
5841 .unwrap_err()
5842 .to_string();
5843 server.await.unwrap();
5844 assert!(error.ends_with(&"x".repeat(200)), "{error}");
5845 assert_eq!(error.split(": ").last().unwrap().len(), 200);
5846 assert!(!error.contains("not-in-diagnostic"));
5847 }
5848
5849 #[cfg(unix)]
5850 #[tokio::test]
5851 async fn http_health_real_nats_server_monitoring() {
5852 if std::process::Command::new("nats-server")
5853 .arg("--version")
5854 .env("XDG_DATA_HOME", std::env::temp_dir())
5855 .env("XDG_RUNTIME_DIR", std::env::temp_dir())
5856 .env("XDG_CONFIG_HOME", std::env::temp_dir())
5857 .output()
5858 .is_err()
5859 {
5860 eprintln!(
5861 "skipped: http_health_real_nats_server_monitoring (nats-server is not on PATH)"
5862 );
5863 return;
5864 }
5865 let dir = subc_test_support::TestTempDir::new("nats-http-monitoring");
5866 let monitor = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5867 let port = monitor.local_addr().unwrap().port();
5868 let client = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
5869 let client_port = client.local_addr().unwrap().port();
5870 let config = dir.join("server.conf");
5871 std::fs::write(&config, format!("listen: 127.0.0.1:{client_port}\nhttp: 127.0.0.1:{port}\njetstream {{ store_dir: \"{}\" }}\n", dir.join("jetstream").display())).unwrap();
5872 let mut configured = http_fixture(&dir, &format!("http://127.0.0.1:{port}/healthz"), 1000);
5873 configured.program = PathBuf::from("nats-server");
5874 configured.args = vec!["-c".into(), config.to_string_lossy().into_owned()];
5875 drop(monitor);
5876 drop(client);
5877 let module =
5878 Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
5879 .supervise_configured_with_health(
5880 configured.module_spec(),
5881 true,
5882 configured.health,
5883 configured.drain_timeout_ms,
5884 configured.restart,
5885 )
5886 .unwrap();
5887 wait_http_health(&module, SupervisorHealthStatus::Ok).await;
5888 module.drain().await.unwrap();
5889 }
5890
5891 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
5892 async fn configuration_update_does_not_replace_captured_running_process_facts() {
5893 let supervisor =
5894 Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default());
5895 let snapshot = stale_process_snapshot(ModuleState::Running, true);
5896 let initial = ModuleSpec {
5897 module_id: "rescan-preserves-spawn-facts".to_string(),
5898 program: PathBuf::from("/spawned/module"),
5899 args: Vec::new(),
5900 env: Vec::new(),
5901 reserved: false,
5902 reserved_prefixes: Vec::new(),
5903 protocol: ModuleProtocol::Subc,
5904 overlap: Default::default(),
5905 };
5906 let module = supervisor.supervised_module(
5907 initial.clone(),
5908 supervisor.runtime_config(),
5909 snapshot,
5910 None,
5911 );
5912 let before = module.status().unwrap();
5913 let mut replacement = initial;
5914 replacement.program = PathBuf::from("/rescanned/replacement-module");
5915
5916 module
5917 .update_configuration(replacement, HealthConfig::default(), None)
5918 .await
5919 .unwrap();
5920
5921 let after = module.status().unwrap();
5922 assert_eq!(after.pid, before.pid);
5923 assert_eq!(after.spawned_at_ms, before.spawned_at_ms);
5924 assert_eq!(after.spawned_from, before.spawned_from);
5925 drop(module);
5926 }
5927}
5928
5929fn unix_ms_now() -> u64 {
5930 SystemTime::now()
5931 .duration_since(UNIX_EPOCH)
5932 .map(|duration| duration.as_millis().min(u128::from(u64::MAX)) as u64)
5933 .unwrap_or(0)
5934}
5935
5936async fn supervise_loop(
5937 mut spec: ModuleSpec,
5938 mut runtime: SupervisorRuntimeConfig,
5939 registry: Arc<Registry>,
5940 process_liveness: Arc<SupervisorProcessLiveness>,
5941 snapshot: SharedSnapshot,
5942 mut child: Option<SupervisedChild>,
5943 mut commands: mpsc::Receiver<SupervisorCommand>,
5944) {
5945 let mut health_probe = HealthProbeRuntime::default();
5946 let mut pending_respawn: Option<PendingRespawn> = None;
5950 let mut requeued: VecDeque<SupervisorCommand> = VecDeque::new();
5953 loop {
5954 #[cfg(target_os = "macos")]
5955 if let Some(active) = child.as_mut() {
5956 active.confirm_privacy_exec().await;
5957 }
5958 if let Some(scheduled) = runtime
5959 .scheduled_respawn
5960 .lock()
5961 .unwrap_or_else(|p| p.into_inner())
5962 .take()
5963 {
5964 pending_respawn = Some(scheduled);
5965 }
5966 if pending_respawn.is_some() && (child.is_some() || !respawn_still_pending(&snapshot)) {
5967 pending_respawn = None;
5968 cancel_deferred_reload(
5969 &runtime,
5970 &spec.module_id,
5971 "respawn cancelled by a supervisor command",
5972 );
5973 }
5974 if child.is_none() && pending_respawn.is_none() {
5975 cancel_deferred_reload(
5976 &runtime,
5977 &spec.module_id,
5978 "respawn cancelled before a replacement was spawned",
5979 );
5980 let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
5981 state.respawn_pending = false;
5982 state.coalesced_restart_pending = false;
5983 if matches!(
5984 state.state,
5985 ModuleState::Restarting
5986 | ModuleState::Starting
5987 | ModuleState::Draining
5988 | ModuleState::Unresponsive
5989 ) {
5990 error!(module_id = %spec.module_id, state = ?state.state,
5991 "supervision operation ended without a child or pending respawn; marking failed so start can retry");
5992 state.state = ModuleState::Failed;
5993 clear_current_process_facts(state);
5994 }
5995 });
5996 }
5997 if let Some(command) = requeued.pop_front() {
5998 if !handle_supervisor_command(
5999 command,
6000 &mut spec,
6001 &mut runtime,
6002 ®istry,
6003 &process_liveness,
6004 &snapshot,
6005 &mut child,
6006 &mut commands,
6007 &mut requeued,
6008 )
6009 .await
6010 {
6011 return;
6012 }
6013 if child.is_some() || !respawn_still_pending(&snapshot) {
6014 pending_respawn = None;
6015 let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
6016 state.respawn_pending = false
6017 });
6018 }
6019 continue;
6020 }
6021 if child.is_some() {
6022 health_probe.refresh_registration(&spec, &runtime, ®istry, &snapshot);
6023 let probe_sleep = sleep(health_probe.wake_after());
6024 tokio::pin!(probe_sleep);
6025 let active_child = child.as_mut().expect("child checked above");
6026 tokio::select! {
6027 wait_result = active_child.wait() => {
6028 let exit_report = match wait_result {
6037 Ok(status) => classify_reaped_child_exit(&snapshot, active_child, &status),
6038 Err(err) => {
6039 active_child.drain_stderr(&spec.module_id).await;
6040 fail_snapshot(&snapshot, Some(&spec.module_id), None);
6041 record_wait_error_terminal(
6047 &spec.module_id,
6048 &runtime.terminal_ring,
6049 &runtime.spawn_events,
6050 );
6051 untrack_if_registration_released(
6052 &process_liveness,
6053 ®istry,
6054 &spec.module_id,
6055 &snapshot,
6056 );
6057 error!(module_id = %spec.module_id, error = %err, "failed to wait for supervised module");
6058 child = None;
6059 continue;
6060 }
6061 };
6062 active_child.drain_stderr(&spec.module_id).await;
6063
6064 let next = on_child_exit(
6065 &spec,
6066 runtime.restart_policy,
6067 ®istry,
6068 &snapshot,
6069 &runtime.terminal_ring,
6070 &runtime.spawn_events,
6071 &runtime.child_roster,
6072 exit_report,
6073 ).await;
6074 active_child.release_roster();
6077 match next {
6078 NextAction::Stop { registration_released } => {
6079 if registration_released {
6080 process_liveness.untrack_if_current(&spec.module_id, &snapshot);
6081 }
6082 child = None;
6083 }
6084 NextAction::Restart { schedule } => {
6085 let delay = schedule.map_or(
6086 runtime.restart_policy.delay_for_restart(0),
6087 |schedule| schedule.delay,
6088 );
6089 if let Some(schedule) = schedule {
6090 log_crash_respawn(&spec.module_id, schedule);
6091 }
6092 child = None;
6100 pending_respawn = Some(PendingRespawn { deadline: Instant::now() + delay, kind: RespawnKind::Spawn });
6101 let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = true);
6102 }
6103 }
6104 }
6105 command = commands.recv() => {
6106 let Some(command) = command else {
6107 return;
6108 };
6109 if !handle_supervisor_command(
6110 command,
6111 &mut spec,
6112 &mut runtime,
6113 ®istry,
6114 &process_liveness,
6115 &snapshot,
6116 &mut child,
6117 &mut commands,
6118 &mut requeued,
6119 ).await {
6120 return;
6121 }
6122 }
6123 _ = &mut probe_sleep => {
6124 if health_probe.due() {
6125 run_health_probe_cycle(
6126 &spec,
6127 &runtime,
6128 ®istry,
6129 &process_liveness,
6130 &snapshot,
6131 &mut child,
6132 ).await;
6133 if child.is_some() {
6134 health_probe.schedule_next(&spec, runtime.health.cadence);
6135 }
6136 }
6137 }
6138 }
6139 } else if let Some(pending) = pending_respawn {
6140 tokio::select! {
6141 _ = sleep_until(pending.deadline) => {
6142 pending_respawn = None;
6143 let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = false);
6144 if !respawn_still_pending(&snapshot) {
6148 continue;
6149 }
6150 if runtime.child_roster.is_closed() {
6155 let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
6156 state.state = ModuleState::Stopped;
6157 });
6158 debug!(module_id = %spec.module_id, "crash respawn cancelled by daemon shutdown");
6159 continue;
6160 }
6161 if let Err(err) = release_dead_registration(
6162 ®istry,
6163 runtime.forwarding.as_deref(),
6164 &snapshot,
6165 &spec.module_id,
6166 ).await {
6167 fail_snapshot(&snapshot, Some(&spec.module_id), None);
6168 error!(module_id = %spec.module_id, error = %err, "registration did not release before restart");
6169 continue;
6170 }
6171
6172 if matches!(pending.kind, RespawnKind::Reload) {
6173 let reply = runtime.deferred_reload_reply.lock().unwrap_or_else(|p| p.into_inner()).take();
6174 let result = finish_reload_child(&spec, &runtime, ®istry, &process_liveness, &snapshot, &mut child).await;
6175 if let Some(reply) = reply { let _ = reply.send(result); }
6176 continue;
6177 }
6178 process_liveness.track(spec.module_id.clone(), Arc::clone(&snapshot));
6179 match spawn_and_mark_running(&spec, &runtime, &snapshot) {
6180 Ok(next_child) => {
6181 child = Some(next_child);
6182 debug!(module_id = %spec.module_id, "supervised module restarted after crash");
6183 }
6184 Err(err) => {
6185 fail_snapshot(&snapshot, Some(&spec.module_id), None);
6186 process_liveness.untrack_if_current(&spec.module_id, &snapshot);
6187 error!(module_id = %spec.module_id, error = %err, "failed to restart supervised module");
6188 }
6189 }
6190 }
6191 command = commands.recv() => {
6192 let Some(command) = command else {
6193 return;
6194 };
6195 if !handle_supervisor_command(
6196 command,
6197 &mut spec,
6198 &mut runtime,
6199 ®istry,
6200 &process_liveness,
6201 &snapshot,
6202 &mut child,
6203 &mut commands,
6204 &mut requeued,
6205 ).await {
6206 return;
6207 }
6208 if child.is_some() || !respawn_still_pending(&snapshot) {
6213 pending_respawn = None;
6214 let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = false);
6215 }
6216 }
6217 }
6218 } else {
6219 let Some(command) = commands.recv().await else {
6220 return;
6221 };
6222 if !handle_supervisor_command(
6223 command,
6224 &mut spec,
6225 &mut runtime,
6226 ®istry,
6227 &process_liveness,
6228 &snapshot,
6229 &mut child,
6230 &mut commands,
6231 &mut requeued,
6232 )
6233 .await
6234 {
6235 return;
6236 }
6237 }
6238 }
6239}
6240
6241fn log_crash_respawn(module_id: &str, schedule: CrashRestartSchedule) {
6242 info!(
6243 module_id,
6244 restart_in_window = schedule.restart_in_window,
6245 delay_ms = schedule.delay.as_millis() as u64,
6246 "respawning after crash"
6247 );
6248}
6249
6250fn respawn_still_pending(snapshot: &SharedSnapshot) -> bool {
6256 matches!(
6257 lock_snapshot(snapshot),
6258 Ok(state) if state.enabled && state.state == ModuleState::Restarting
6259 )
6260}
6261
6262enum NextAction {
6263 Stop {
6264 registration_released: bool,
6265 },
6266 Restart {
6267 schedule: Option<CrashRestartSchedule>,
6268 },
6269}
6270
6271#[allow(clippy::too_many_arguments)]
6272async fn handle_supervisor_command(
6273 command: SupervisorCommand,
6274 spec: &mut ModuleSpec,
6275 runtime: &mut SupervisorRuntimeConfig,
6276 registry: &Arc<Registry>,
6277 process_liveness: &SupervisorProcessLiveness,
6278 snapshot: &SharedSnapshot,
6279 child: &mut Option<SupervisedChild>,
6280 commands: &mut mpsc::Receiver<SupervisorCommand>,
6281 requeued: &mut VecDeque<SupervisorCommand>,
6282) -> bool {
6283 match command {
6284 SupervisorCommand::Drain { reply } => {
6285 let result = drain_optional_child(
6288 &spec.module_id,
6289 spec.protocol,
6290 StopNotice::NotSent,
6291 registry,
6292 runtime.forwarding.as_deref(),
6293 snapshot,
6294 &runtime.terminal_ring,
6295 &runtime.spawn_events,
6296 child,
6297 runtime.drain_timeout,
6298 ModuleState::Stopped,
6299 None,
6300 )
6301 .await;
6302 let registration_released = result.is_ok();
6303 let _ = reply.send(result);
6304 if registration_released {
6305 process_liveness.untrack_if_current(&spec.module_id, snapshot);
6306 }
6307 false
6308 }
6309 SupervisorCommand::Retire { reply } => {
6310 let result = async {
6311 let stop_notice = begin_forwarding_drain_if_configured(
6312 spec,
6313 runtime,
6314 registry,
6315 snapshot,
6316 None,
6317 RouteCloseReason::Disable,
6318 )
6319 .await?;
6320 drain_optional_child(
6321 &spec.module_id,
6322 spec.protocol,
6323 stop_notice,
6324 registry,
6325 runtime.forwarding.as_deref(),
6326 snapshot,
6327 &runtime.terminal_ring,
6328 &runtime.spawn_events,
6329 child,
6330 runtime.drain_timeout,
6331 ModuleState::Stopped,
6332 None,
6333 )
6334 .await
6335 }
6336 .await;
6337 let registration_released = result.is_ok();
6338 let _ = reply.send(result);
6339 if registration_released {
6340 process_liveness.untrack_if_current(&spec.module_id, snapshot);
6341 }
6342 false
6343 }
6344 SupervisorCommand::Restart {
6345 drain_timeout_ms,
6346 received_at_generation,
6347 queued_at,
6348 reply,
6349 } => {
6350 info!(
6354 module_id = %spec.module_id,
6355 queued_ms = u64::try_from(queued_at.elapsed().as_millis()).unwrap_or(u64::MAX),
6356 "restart command dequeued"
6357 );
6358 let validation = match lock_snapshot(snapshot) {
6370 Ok(state) if !state.enabled => Err(SuperviseError::Disabled {
6371 module_id: spec.module_id.clone(),
6372 }),
6373 Ok(_) => Ok(()),
6374 Err(err) => Err(err),
6375 };
6376 let initiated = validation.is_ok();
6377 let _ = reply.send(validation);
6378 let satisfied_by_generation = if initiated && child.is_some() {
6389 lock_snapshot(snapshot).ok().and_then(|state| {
6390 (state.spawn_generation > received_at_generation
6391 && !state.configuration_updated_since_spawn)
6392 .then_some(state.spawn_generation)
6393 })
6394 } else {
6395 None
6396 };
6397 let satisfied_by_pending = initiated
6398 && child.is_none()
6399 && lock_snapshot(snapshot).ok().is_some_and(|mut state| {
6400 let pending = state.respawn_pending && !state.configuration_updated_since_spawn;
6401 if pending {
6402 state.coalesced_restart_pending = true;
6403 }
6404 pending
6405 });
6406 if satisfied_by_pending {
6407 debug!(module_id = %spec.module_id, "restart coalesced into the pending replacement");
6408 } else if let Some(generation) = satisfied_by_generation {
6409 info!(
6410 module_id = %spec.module_id,
6411 received_at_generation,
6412 "restart already satisfied by generation {generation}; not restarting again"
6413 );
6414 } else if initiated {
6415 let drain_timeout = drain_timeout_ms
6418 .map(Duration::from_millis)
6419 .unwrap_or(runtime.drain_timeout);
6420 if let Err(err) = restart_child(
6421 spec,
6422 runtime,
6423 registry,
6424 process_liveness,
6425 snapshot,
6426 child,
6427 drain_timeout,
6428 )
6429 .await
6430 {
6431 warn!(
6432 module_id = %spec.module_id,
6433 error = %err,
6434 "operator restart failed after initiation ack; module state carries the outcome"
6435 );
6436 let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6437 state.state = ModuleState::Failed;
6438 clear_current_process_facts(state);
6439 });
6440 }
6441 }
6442 true
6443 }
6444 SupervisorCommand::Reload { reply } => {
6445 let result =
6446 reload_child(spec, runtime, registry, process_liveness, snapshot, child).await;
6447 if result.is_ok()
6448 && runtime
6449 .scheduled_respawn
6450 .lock()
6451 .unwrap_or_else(|p| p.into_inner())
6452 .as_ref()
6453 .is_some_and(|pending| matches!(pending.kind, RespawnKind::Reload))
6454 {
6455 *runtime
6456 .deferred_reload_reply
6457 .lock()
6458 .unwrap_or_else(|p| p.into_inner()) = Some(reply);
6459 } else {
6460 let _ = reply.send(result);
6461 }
6462 true
6463 }
6464 SupervisorCommand::SetEnabled { enabled, reply } => {
6465 let result = set_child_enabled(
6466 spec,
6467 runtime,
6468 registry,
6469 process_liveness,
6470 snapshot,
6471 child,
6472 enabled,
6473 )
6474 .await;
6475 let _ = reply.send(result);
6476 true
6477 }
6478 SupervisorCommand::UpdateConfiguration {
6479 spec: next_spec,
6480 health,
6481 drain_timeout_ms,
6482 reply,
6483 } => {
6484 if let Some(handle) = &runtime.supervisor_handle {
6485 handle.apply_identity_configuration(&next_spec);
6486 }
6487 *spec = next_spec;
6488 let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6489 state.configuration_updated_since_spawn = true;
6490 });
6491 let health_changed = runtime.health != health;
6492 runtime.health = health;
6493 if health_changed {
6496 let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6497 state.health = ModuleHealthStatus::default();
6498 });
6499 }
6500 runtime.drain_timeout = drain_timeout_ms
6501 .map(Duration::from_millis)
6502 .unwrap_or(runtime.default_drain_timeout);
6503 *runtime
6504 .effective_drain_timeout
6505 .lock()
6506 .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
6507 let _ = reply.send(());
6508 true
6509 }
6510 SupervisorCommand::Swap {
6511 ready_timeout,
6512 reply,
6513 } => {
6514 let end = swap::run_swap(
6515 spec,
6516 runtime,
6517 registry,
6518 process_liveness,
6519 snapshot,
6520 child,
6521 commands,
6522 ready_timeout.unwrap_or(DEFAULT_SWAP_READY_TIMEOUT),
6523 reply,
6524 )
6525 .await;
6526 requeued.extend(end.requeue);
6527 true
6528 }
6529 }
6530}
6531
6532async fn restart_child(
6533 spec: &ModuleSpec,
6534 runtime: &SupervisorRuntimeConfig,
6535 registry: &Registry,
6536 process_liveness: &SupervisorProcessLiveness,
6537 snapshot: &SharedSnapshot,
6538 child: &mut Option<SupervisedChild>,
6539 drain_timeout: Duration,
6540) -> Result<(), SuperviseError> {
6541 if !lock_snapshot(snapshot)?.enabled {
6543 return Err(SuperviseError::Disabled {
6544 module_id: spec.module_id.clone(),
6545 });
6546 }
6547 let stop_notice = begin_forwarding_drain_with_timeout(
6548 spec,
6549 runtime,
6550 registry,
6551 snapshot,
6552 None,
6553 RouteCloseReason::Restart,
6554 drain_timeout,
6555 )
6556 .await?;
6557
6558 if child.is_some() {
6559 drain_optional_child(
6560 &spec.module_id,
6561 spec.protocol,
6562 stop_notice,
6563 registry,
6564 runtime.forwarding.as_deref(),
6565 snapshot,
6566 &runtime.terminal_ring,
6567 &runtime.spawn_events,
6568 child,
6569 drain_timeout,
6570 ModuleState::Restarting,
6571 Some(true),
6572 )
6573 .await?;
6574 } else {
6575 update_snapshot(snapshot, Some(&spec.module_id), |state| {
6576 state.enabled = true;
6577 state.state = ModuleState::Restarting;
6578 clear_current_process_facts(state);
6579 })?;
6580 release_dead_registration(
6581 registry,
6582 runtime.forwarding.as_deref(),
6583 snapshot,
6584 &spec.module_id,
6585 )
6586 .await?;
6587 }
6588
6589 reset_restart_count(snapshot, &spec.module_id)?;
6590 process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6591 schedule_respawn(
6592 runtime,
6593 snapshot,
6594 &spec.module_id,
6595 runtime.restart_policy.backoff,
6596 RespawnKind::Spawn,
6597 )
6598}
6599
6600async fn reload_child(
6601 spec: &ModuleSpec,
6602 runtime: &SupervisorRuntimeConfig,
6603 registry: &Registry,
6604 process_liveness: &SupervisorProcessLiveness,
6605 snapshot: &SharedSnapshot,
6606 child: &mut Option<SupervisedChild>,
6607) -> Result<(), SuperviseError> {
6608 if !lock_snapshot(snapshot)?.enabled {
6610 return Err(SuperviseError::Disabled {
6611 module_id: spec.module_id.clone(),
6612 });
6613 }
6614 let stop_notice = begin_forwarding_drain(
6615 spec,
6616 runtime,
6617 registry,
6618 snapshot,
6619 Some(true),
6620 RouteCloseReason::Reload,
6621 )
6622 .await?;
6623
6624 if child.is_some() {
6625 drain_optional_child(
6626 &spec.module_id,
6627 spec.protocol,
6628 stop_notice,
6629 registry,
6630 runtime.forwarding.as_deref(),
6631 snapshot,
6632 &runtime.terminal_ring,
6633 &runtime.spawn_events,
6634 child,
6635 runtime.drain_timeout,
6636 ModuleState::Restarting,
6637 Some(true),
6638 )
6639 .await?;
6640 } else {
6641 update_snapshot(snapshot, Some(&spec.module_id), |state| {
6642 state.enabled = true;
6643 state.state = ModuleState::Restarting;
6644 clear_current_process_facts(state);
6645 })?;
6646 release_dead_registration(
6647 registry,
6648 runtime.forwarding.as_deref(),
6649 snapshot,
6650 &spec.module_id,
6651 )
6652 .await?;
6653 }
6654
6655 reset_restart_count(snapshot, &spec.module_id)?;
6656 process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6657 schedule_respawn(
6658 runtime,
6659 snapshot,
6660 &spec.module_id,
6661 runtime.restart_policy.backoff,
6662 RespawnKind::Reload,
6663 )
6664}
6665
6666async fn finish_reload_child(
6667 spec: &ModuleSpec,
6668 runtime: &SupervisorRuntimeConfig,
6669 registry: &Registry,
6670 process_liveness: &SupervisorProcessLiveness,
6671 snapshot: &SharedSnapshot,
6672 child: &mut Option<SupervisedChild>,
6673) -> Result<(), SuperviseError> {
6674 process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6675 let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
6676 Ok(next_child) => next_child,
6677 Err(err) => {
6678 return handle_reload_spawn_failure(
6679 spec,
6680 runtime,
6681 process_liveness,
6682 snapshot,
6683 child,
6684 format!("new child failed to spawn: {err}"),
6685 )
6686 .await;
6687 }
6688 };
6689 *child = Some(next_child);
6690
6691 let wait_outcome = {
6692 let active_child = child.as_mut().expect("new reload child was just stored");
6693 wait_for_registration_after_reload(
6694 registry,
6695 &spec.module_id,
6696 snapshot,
6697 active_child,
6698 REGISTRY_RELEASE_TIMEOUT,
6699 )
6700 .await?
6701 };
6702
6703 match wait_outcome {
6704 RegistrationWaitOutcome::Registered => {
6705 debug!(module_id = %spec.module_id, "supervised module reloaded and registered");
6706 Ok(())
6707 }
6708 RegistrationWaitOutcome::Exited(exit_report) => {
6709 if let Some(active_child) = child.as_mut() {
6710 active_child.drain_stderr(&spec.module_id).await;
6711 }
6712 let mut exited_child = child.take().expect("exited reload child is still stored");
6715 #[cfg(test)]
6716 if let Some(gate) = &runtime.test_reload_exit_record_gate {
6717 gate.reached.notify_one();
6718 gate.resume.notified().await;
6719 }
6720 let result = handle_reload_child_registration_failure(
6721 spec,
6722 runtime,
6723 registry,
6724 process_liveness,
6725 snapshot,
6726 child,
6727 ReloadRegistrationFailure {
6728 exit_report: registration_failure_exit_report(exit_report),
6729 reason: exited_child
6730 .spawn_failure
6731 .clone()
6732 .unwrap_or_else(|| "new child exited before registering".to_string()),
6733 },
6734 )
6735 .await;
6736 exited_child.release_roster();
6737 result
6738 }
6739 RegistrationWaitOutcome::TimedOut => {
6740 let mut timed_out_child = child
6741 .take()
6742 .expect("timed-out reload child is still running");
6743 timed_out_child
6744 .start_kill()
6745 .map_err(|source| SuperviseError::Kill {
6746 module_id: spec.module_id.clone(),
6747 source,
6748 })?;
6749 let status = timed_out_child
6750 .wait()
6751 .await
6752 .map_err(|source| SuperviseError::Wait {
6753 module_id: spec.module_id.clone(),
6754 source,
6755 })?;
6756 timed_out_child.drain_stderr(&spec.module_id).await;
6757 handle_reload_child_registration_failure(
6758 spec,
6759 runtime,
6760 registry,
6761 process_liveness,
6762 snapshot,
6763 child,
6764 ReloadRegistrationFailure {
6765 exit_report: registration_failure_exit_report(classify_reaped_child_exit(
6766 snapshot,
6767 &timed_out_child,
6768 &status,
6769 )),
6770 reason: format!(
6771 "new child did not register within {:?}",
6772 REGISTRY_RELEASE_TIMEOUT
6773 ),
6774 },
6775 )
6776 .await
6777 }
6778 }
6779}
6780
6781async fn set_child_enabled(
6782 spec: &ModuleSpec,
6783 runtime: &SupervisorRuntimeConfig,
6784 registry: &Registry,
6785 process_liveness: &SupervisorProcessLiveness,
6786 snapshot: &SharedSnapshot,
6787 child: &mut Option<SupervisedChild>,
6788 enabled: bool,
6789) -> Result<bool, SuperviseError> {
6790 let (current_enabled, current_state, respawn_pending) = {
6791 let state = lock_snapshot(snapshot)?;
6792 (state.enabled, state.state, state.respawn_pending)
6793 };
6794 let revive_terminal = enabled
6802 && current_enabled
6803 && child.is_none()
6804 && (matches!(current_state, ModuleState::Failed | ModuleState::Stopped)
6805 || (current_state == ModuleState::Restarting && !respawn_pending));
6806 if current_enabled == enabled && !revive_terminal {
6807 return Ok(false);
6808 }
6809
6810 if enabled {
6811 update_snapshot(snapshot, Some(&spec.module_id), |state| {
6812 state.enabled = true;
6813 state.state = ModuleState::Starting;
6814 clear_current_process_facts(state);
6815 })?;
6816 #[cfg(test)]
6817 if runtime.test_seed_stale_facts_before_enable_spawn {
6818 update_snapshot(snapshot, Some(&spec.module_id), |state| {
6819 state.process_alive = true;
6820 state.pid = Some(41);
6821 state.spawned_at_ms = Some(42);
6822 state.spawned_from = Some(PathBuf::from("/spawned/module"));
6823 state.spawned_file_identity = Some(SpawnedFileIdentity {
6824 device: 43,
6825 inode: 44,
6826 });
6827 })?;
6828 }
6829 release_dead_registration(
6830 registry,
6831 runtime.forwarding.as_deref(),
6832 snapshot,
6833 &spec.module_id,
6834 )
6835 .await?;
6836 reset_restart_count(snapshot, &spec.module_id)?;
6837 process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
6838 let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
6839 Ok(next_child) => next_child,
6840 Err(err) => {
6841 if let Err(state_err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6842 state.state = ModuleState::Failed;
6843 clear_current_process_facts(state);
6844 }) {
6845 error!(module_id = %spec.module_id, error = %state_err, "failed to record enable spawn failure");
6846 }
6847 process_liveness.untrack_if_current(&spec.module_id, snapshot);
6848 return Err(err);
6849 }
6850 };
6851 *child = Some(next_child);
6852 debug!(module_id = %spec.module_id, "supervised module enabled");
6853 Ok(true)
6854 } else {
6855 let stop_notice = begin_forwarding_drain_if_configured(
6856 spec,
6857 runtime,
6858 registry,
6859 snapshot,
6860 Some(false),
6861 RouteCloseReason::Disable,
6862 )
6863 .await?;
6864 drain_optional_child(
6865 &spec.module_id,
6866 spec.protocol,
6867 stop_notice,
6868 registry,
6869 runtime.forwarding.as_deref(),
6870 snapshot,
6871 &runtime.terminal_ring,
6872 &runtime.spawn_events,
6873 child,
6874 runtime.drain_timeout,
6875 ModuleState::Disabled,
6876 Some(false),
6877 )
6878 .await?;
6879 debug!(module_id = %spec.module_id, "supervised module disabled");
6880 Ok(true)
6881 }
6882}
6883
6884#[allow(clippy::too_many_arguments)]
6885async fn on_child_exit(
6886 spec: &ModuleSpec,
6887 policy: RestartPolicy,
6888 registry: &Registry,
6889 snapshot: &SharedSnapshot,
6890 terminal_ring: &Arc<Mutex<TerminalRing>>,
6891 spawn_events: &SpawnEventFeed,
6892 roster: &ChildRoster,
6893 exit_report: ExitReport,
6894) -> NextAction {
6895 if roster.is_closed() {
6901 return on_child_exit_during_daemon_shutdown(
6902 spec,
6903 registry,
6904 snapshot,
6905 terminal_ring,
6906 spawn_events,
6907 exit_report,
6908 )
6909 .await;
6910 }
6911 let unrequested_clean_exit_of_protocol_none = exit_report.kind == ExitKind::Clean
6926 && running_protocol(spec, snapshot) == ModuleProtocol::None;
6927 match exit_report.kind {
6928 ExitKind::Clean if !unrequested_clean_exit_of_protocol_none => {
6929 info!(
6930 module_id = %spec.module_id,
6931 exit_code = ?exit_report.code,
6932 exit_signal = ?exit_report.signal,
6933 "supervised module exited cleanly"
6934 );
6935 if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6936 state.state = ModuleState::Stopped;
6937 clear_current_process_facts(state);
6938 state.last_exit = Some(exit_report.clone());
6939 }) {
6940 error!(module_id = %spec.module_id, error = %err, "failed to record clean module exit");
6941 }
6942 record_terminal(
6943 &spec.module_id,
6944 terminal_ring,
6945 spawn_events,
6946 &exit_report,
6947 TerminalDisposition::Stopped,
6948 );
6949 let registration_released = match wait_for_registration_release(
6950 registry,
6951 &spec.module_id,
6952 REGISTRY_RELEASE_TIMEOUT,
6953 )
6954 .await
6955 {
6956 Ok(()) => true,
6957 Err(err) => {
6958 warn!(module_id = %spec.module_id, error = %err, "registration still active after clean exit");
6959 false
6960 }
6961 };
6962 NextAction::Stop {
6963 registration_released,
6964 }
6965 }
6966 ExitKind::Clean | ExitKind::Crash => {
6967 if unrequested_clean_exit_of_protocol_none {
6968 warn!(
6969 module_id = %spec.module_id,
6970 exit_code = ?exit_report.code,
6971 exit_signal = ?exit_report.signal,
6972 "protocol-none module exited cleanly without a stop request; handling it as a crash"
6973 );
6974 } else {
6975 warn!(
6976 module_id = %spec.module_id,
6977 exit_code = ?exit_report.code,
6978 exit_signal = ?exit_report.signal,
6979 "supervised module exited abnormally (crash)"
6980 );
6981 }
6982 let mut restart_schedule = None;
6983 let mut disposition = TerminalDisposition::Disabled;
6984 let mut disposition_detail = lock_snapshot(snapshot)
6988 .ok()
6989 .and_then(|mut state| state.spawn_failure.take());
6990 let now = Instant::now();
6991 if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
6992 clear_current_process_facts(state);
6993 state.last_exit = Some(exit_report.clone());
6994 if state.enabled {
6995 if let Some(schedule) = state.next_crash_restart(&policy, now) {
6996 state.state = ModuleState::Restarting;
6997 restart_schedule = Some(schedule);
6998 disposition = TerminalDisposition::Restarting;
6999 } else {
7000 state.state = ModuleState::Failed;
7001 disposition = TerminalDisposition::Failed;
7002 let budget = policy.budget_exhausted_detail();
7003 disposition_detail =
7004 Some(disposition_detail.take().map_or_else(
7005 || budget.clone(),
7006 |cause| format!("{cause}; {budget}"),
7007 ));
7008 }
7009 } else {
7010 state.state = ModuleState::Disabled;
7011 disposition = TerminalDisposition::Disabled;
7012 }
7013 }) {
7014 error!(module_id = %spec.module_id, error = %err, "failed to record crashed module exit");
7015 return NextAction::Stop {
7016 registration_released: false,
7017 };
7018 }
7019 if disposition == TerminalDisposition::Failed {
7020 error!(
7025 module_id = %spec.module_id,
7026 max_restarts = policy.max_restarts,
7027 window_secs = policy.window.as_secs(),
7028 "module stopped: {}",
7029 policy.budget_exhausted_detail()
7030 );
7031 }
7032 record_terminal_with_detail(
7033 &spec.module_id,
7034 terminal_ring,
7035 spawn_events,
7036 &exit_report,
7037 disposition,
7038 disposition_detail,
7039 );
7040
7041 if let Some(schedule) = restart_schedule {
7042 NextAction::Restart {
7043 schedule: Some(schedule),
7044 }
7045 } else {
7046 let registration_released = match wait_for_registration_release(
7047 registry,
7048 &spec.module_id,
7049 REGISTRY_RELEASE_TIMEOUT,
7050 )
7051 .await
7052 {
7053 Ok(()) => true,
7054 Err(err) => {
7055 warn!(module_id = %spec.module_id, error = %err, "registration still active after failed module");
7056 false
7057 }
7058 };
7059 NextAction::Stop {
7060 registration_released,
7061 }
7062 }
7063 }
7064 ExitKind::DeliberateSeverance => {
7065 warn!(
7066 module_id = %spec.module_id,
7067 exit_code = ?exit_report.code,
7068 exit_signal = ?exit_report.signal,
7069 "supervised module exited after deliberate connection severance"
7070 );
7071 let mut should_restart = false;
7072 let mut disposition = TerminalDisposition::Disabled;
7073 if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
7074 clear_current_process_facts(state);
7075 state.last_exit = Some(exit_report.clone());
7076 state.lifetime_restarts += 1;
7077 if state.enabled {
7078 state.state = ModuleState::Restarting;
7079 should_restart = true;
7080 disposition = TerminalDisposition::Restarting;
7081 } else {
7082 state.state = ModuleState::Disabled;
7083 }
7084 }) {
7085 error!(module_id = %spec.module_id, error = %err, "failed to record deliberately severed module exit");
7086 return NextAction::Stop {
7087 registration_released: false,
7088 };
7089 }
7090 record_terminal(
7091 &spec.module_id,
7092 terminal_ring,
7093 spawn_events,
7094 &exit_report,
7095 disposition,
7096 );
7097
7098 if should_restart {
7099 NextAction::Restart { schedule: None }
7100 } else {
7101 let registration_released = match wait_for_registration_release(
7102 registry,
7103 &spec.module_id,
7104 REGISTRY_RELEASE_TIMEOUT,
7105 )
7106 .await
7107 {
7108 Ok(()) => true,
7109 Err(err) => {
7110 warn!(module_id = %spec.module_id, error = %err, "registration still active after deliberately severed module exit");
7111 false
7112 }
7113 };
7114 NextAction::Stop {
7115 registration_released,
7116 }
7117 }
7118 }
7119 }
7120}
7121
7122async fn on_child_exit_during_daemon_shutdown(
7123 spec: &ModuleSpec,
7124 registry: &Registry,
7125 snapshot: &SharedSnapshot,
7126 terminal_ring: &Arc<Mutex<TerminalRing>>,
7127 spawn_events: &SpawnEventFeed,
7128 exit_report: ExitReport,
7129) -> NextAction {
7130 info!(
7131 module_id = %spec.module_id,
7132 exit_code = ?exit_report.code,
7133 exit_signal = ?exit_report.signal,
7134 exit_kind = ?exit_report.kind,
7135 "supervised module exited during daemon shutdown; not restarting it"
7136 );
7137 if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
7138 state.state = ModuleState::Stopped;
7139 clear_current_process_facts(state);
7140 state.last_exit = Some(exit_report.clone());
7141 }) {
7142 error!(module_id = %spec.module_id, error = %err, "failed to record module exit during daemon shutdown");
7143 }
7144 record_terminal(
7145 &spec.module_id,
7146 terminal_ring,
7147 spawn_events,
7148 &exit_report,
7149 TerminalDisposition::DaemonShutdown,
7150 );
7151 let registration_released =
7152 wait_for_registration_release(registry, &spec.module_id, REGISTRY_RELEASE_TIMEOUT)
7153 .await
7154 .is_ok();
7155 NextAction::Stop {
7156 registration_released,
7157 }
7158}
7159
7160fn record_wait_error_terminal(
7161 module_id: &str,
7162 terminal_ring: &Arc<Mutex<TerminalRing>>,
7163 spawn_events: &SpawnEventFeed,
7164) {
7165 record_terminal(
7166 module_id,
7167 terminal_ring,
7168 spawn_events,
7169 &wait_error_exit_report(),
7170 TerminalDisposition::Failed,
7171 );
7172}
7173
7174fn record_terminal(
7175 module_id: &str,
7176 terminal_ring: &Arc<Mutex<TerminalRing>>,
7177 spawn_events: &SpawnEventFeed,
7178 exit_report: &ExitReport,
7179 disposition: TerminalDisposition,
7180) {
7181 record_terminal_with_detail(
7182 module_id,
7183 terminal_ring,
7184 spawn_events,
7185 exit_report,
7186 disposition,
7187 None,
7188 );
7189}
7190
7191fn durable_terminal_history_of(
7195 terminal_ring: &Mutex<TerminalRing>,
7196 module_id: &str,
7197) -> subc_control::TerminalHistory {
7198 let read = terminal_ring
7199 .lock()
7200 .unwrap_or_else(|p| p.into_inner())
7201 .capture_durable_history();
7202 read.read(module_id)
7203}
7204
7205fn record_terminal_with_detail(
7206 module_id: &str,
7207 terminal_ring: &Arc<Mutex<TerminalRing>>,
7208 spawn_events: &SpawnEventFeed,
7209 exit_report: &ExitReport,
7210 disposition: TerminalDisposition,
7211 disposition_detail: Option<String>,
7212) {
7213 spawn_events.emit_exited(module_id, exit_report.code, exit_report.signal);
7214 let record = TerminalRecord {
7215 exit_code: exit_report.code,
7216 exit_signal: exit_report.signal,
7217 at_ms: exit_report.at_ms,
7218 disposition,
7219 exit_kind: exit_report.kind.into(),
7220 disposition_detail,
7221 };
7222 terminal_ring
7223 .lock()
7224 .unwrap_or_else(|poisoned| poisoned.into_inner())
7225 .record_exit(module_id, record);
7226}
7227
7228fn untrack_if_registration_released(
7229 process_liveness: &SupervisorProcessLiveness,
7230 registry: &Registry,
7231 module_id: &str,
7232 snapshot: &SharedSnapshot,
7233) {
7234 match registry.get_module(module_id) {
7235 Ok(None) => process_liveness.untrack_if_current(module_id, snapshot),
7236 Ok(Some(_)) => {}
7237 Err(err) => {
7238 warn!(module_id, error = %err, "could not determine whether supervisor liveness can be untracked");
7239 }
7240 }
7241}
7242
7243#[cfg(test)]
7257fn apply_wire_spawn_args(
7258 command: &mut Command,
7259 spec: &ModuleSpec,
7260 connection_file_path: Option<&std::path::Path>,
7261 handle: Option<&SupervisorHandle>,
7262) -> Result<Option<NonceHandoff>, SuperviseError> {
7263 apply_wire_spawn_args_for_role(
7264 command,
7265 spec,
7266 connection_file_path,
7267 handle,
7268 SpawnRole::Plain,
7269 )
7270}
7271
7272#[cfg(unix)]
7277type NonceHandoff = subc_os::LaunchNonceHandoff;
7278#[cfg(not(unix))]
7279type NonceHandoff = std::convert::Infallible;
7280
7281fn apply_wire_spawn_args_for_role(
7298 command: &mut Command,
7299 spec: &ModuleSpec,
7300 connection_file_path: Option<&std::path::Path>,
7301 handle: Option<&SupervisorHandle>,
7302 role: SpawnRole,
7303) -> Result<Option<NonceHandoff>, SuperviseError> {
7304 command.env(SUBC_MODULE_ID_ENV, &spec.module_id);
7305 command.env_remove(subc_os::LAUNCH_NONCE_FD_ENV);
7310 command.env_remove(SUBC_LAUNCH_NONCE_ENV);
7312 if spec.protocol == ModuleProtocol::None {
7313 return Ok(None);
7314 }
7315 if let Some(connection_file_path) = connection_file_path {
7316 command.arg(SUBC_ARG).arg(connection_file_path);
7317 }
7318
7319 let nonce = generate_launch_nonce()?;
7323 if let Some(handle) = handle {
7324 match role {
7325 SpawnRole::Plain => {
7326 handle.set_spawn_nonce(&spec.module_id, nonce.clone());
7327 if spec.reserved {
7328 handle.set_reserved_nonce(&spec.module_id, nonce.clone());
7329 }
7330 }
7331 SpawnRole::SwapCandidate => handle.open_swap(&spec.module_id, nonce.clone()),
7332 }
7333 }
7334 #[cfg(unix)]
7335 let handoff = {
7336 let handoff =
7337 subc_os::LaunchNonceHandoff::new(&nonce).map_err(|source| SuperviseError::Spawn {
7338 program: spec.program.clone(),
7339 source,
7340 cgroup_path: None,
7341 })?;
7342 command.env(subc_os::LAUNCH_NONCE_FD_ENV, handoff.fd_env_value());
7343 Some(handoff)
7344 };
7345 #[cfg(not(unix))]
7346 let handoff = None;
7347 #[cfg(not(unix))]
7350 command.env(SUBC_LAUNCH_NONCE_ENV, nonce);
7351 Ok(handoff)
7352}
7353
7354fn apply_child_env(command: &mut Command, spec: &ModuleSpec) {
7355 command.env_remove(CK_LOG_ENV);
7356 command.env_remove(SUBC_SPAWN_ROLE_ENV);
7363 for (key, value) in &spec.env {
7364 if matches!(
7368 key.as_str(),
7369 CAPTURE_MAX_FILE_MB_ENV | CAPTURE_KEEP_ENV | CAPTURE_MAX_AGE_DAYS_ENV
7370 ) || key == SUBC_SPAWN_ROLE_ENV
7371 {
7372 continue;
7373 }
7374 command.env(key, value);
7375 }
7376}
7377
7378#[derive(Debug, Clone, Copy, PartialEq, Eq)]
7381enum SpawnRole {
7382 Plain,
7383 SwapCandidate,
7384}
7385
7386fn apply_spawn_role(command: &mut Command, role: SpawnRole) {
7389 if role == SpawnRole::SwapCandidate {
7390 command.env(SUBC_SPAWN_ROLE_ENV, SPAWN_ROLE_SWAP_CANDIDATE);
7391 }
7392}
7393
7394fn spawn_child(
7395 spec: &ModuleSpec,
7396 connection_file_path: Option<&std::path::Path>,
7397 handle: Option<&SupervisorHandle>,
7398 ring: &Arc<Mutex<StderrRing>>,
7399 capture_logs_dir: Option<&std::path::Path>,
7400 roster: &ChildRoster,
7401 #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
7402) -> Result<SupervisedChild, SuperviseError> {
7403 spawn_child_in_slot(
7404 spec,
7405 connection_file_path,
7406 handle,
7407 ring,
7408 capture_logs_dir,
7409 roster,
7410 #[cfg(target_os = "linux")]
7411 cgroup_placement,
7412 SpawnRole::Plain,
7413 false,
7414 )
7415}
7416
7417#[allow(clippy::too_many_arguments)]
7430fn spawn_child_in_slot(
7431 spec: &ModuleSpec,
7432 connection_file_path: Option<&std::path::Path>,
7433 handle: Option<&SupervisorHandle>,
7434 ring: &Arc<Mutex<StderrRing>>,
7435 capture_logs_dir: Option<&std::path::Path>,
7436 roster: &ChildRoster,
7437 #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
7438 role: SpawnRole,
7439 alternate_slot: bool,
7440) -> Result<SupervisedChild, SuperviseError> {
7441 if roster.is_closed() {
7442 return Err(SuperviseError::Spawn {
7443 program: spec.program.clone(),
7444 source: io::Error::other("the daemon is shutting down; not starting a new process"),
7445 cgroup_path: None,
7446 });
7447 }
7448 #[cfg(target_os = "linux")]
7449 let cgroup_name = {
7450 if cgroup_placement.is_none() {
7455 swap::cgroup_name(&spec.module_id, alternate_slot)
7456 } else {
7457 let nonce = generate_launch_nonce()?;
7458 let suffix = u128::from_str_radix(&nonce[..32], 16).expect("hex launch nonce");
7459 let mut end = spec.module_id.len().min(64);
7462 while !spec.module_id.is_char_boundary(end) {
7463 end -= 1;
7464 }
7465 format!(
7466 "{}_{suffix}",
7467 swap::cgroup_name(&spec.module_id[..end], alternate_slot)
7468 )
7469 }
7470 };
7471 #[cfg(not(target_os = "linux"))]
7472 let _ = alternate_slot;
7473 #[cfg(target_os = "macos")]
7474 let (mut command, privacy_exec, exec_ack) = privacy_command(spec, roster)?;
7475 #[cfg(not(target_os = "macos"))]
7476 let mut command = Command::new(&spec.program);
7477 command.args(&spec.args);
7478 apply_child_env(&mut command, spec);
7508 apply_spawn_role(&mut command, role);
7509 let nonce_handoff =
7510 apply_wire_spawn_args_for_role(&mut command, spec, connection_file_path, handle, role)?;
7511
7512 #[cfg(target_os = "linux")]
7513 let cgroup_path = cgroup_placement
7514 .map(|placement| placement.module_path(&cgroup_name))
7515 .transpose()
7516 .map_err(|source| SuperviseError::Cgroup {
7517 module_id: spec.module_id.clone(),
7518 source,
7519 })?;
7520 #[cfg(not(target_os = "linux"))]
7521 let cgroup_path: Option<PathBuf> = None;
7522 #[cfg(target_os = "linux")]
7523 if let Some(path) = &cgroup_path {
7524 if let Err(error) = apply_cgroup_placement(&mut command, spec, path) {
7525 if let Some(placement) = cgroup_placement {
7526 remove_module_cgroup(placement, &cgroup_name);
7527 }
7528 return Err(error);
7529 }
7530 }
7531
7532 let output_sink = if let Some(logs_dir) = capture_logs_dir {
7533 let path = logs_dir.join(format!("{}.stderr.log", spec.module_id));
7534 match ChildOutputSink::open(&path, capture_retention(spec)) {
7535 Ok(sink) => sink,
7536 Err(error) => {
7537 warn!(
7538 module_id = %spec.module_id,
7539 path = %path.display(),
7540 error = %error,
7541 "could not open child output capture file; forwarding to stderr"
7542 );
7543 ChildOutputSink::Stderr
7544 }
7545 }
7546 } else {
7547 ChildOutputSink::Stderr
7548 };
7549
7550 command.stdout(Stdio::piped());
7551 command.stderr(Stdio::piped());
7552 command.kill_on_drop(true);
7553 #[cfg(unix)]
7570 command.process_group(0);
7571 command.stdin(Stdio::null());
7572 #[cfg(unix)]
7576 if let Some(handoff) = nonce_handoff {
7577 handoff.install_last(command.as_std_mut());
7578 }
7579 #[cfg(not(unix))]
7580 let _ = nonce_handoff;
7581
7582 #[cfg(windows)]
7587 subc_jobobject::suspend_on_create_async(&mut command);
7588 let mut child = match command.spawn() {
7589 Ok(child) => child,
7590 Err(source) => {
7591 #[cfg(target_os = "linux")]
7592 if let Some(placement) = cgroup_placement {
7593 remove_module_cgroup(placement, &cgroup_name);
7594 }
7595 return Err(SuperviseError::Spawn {
7596 program: spec.program.clone(),
7597 source,
7598 cgroup_path,
7599 });
7600 }
7601 };
7602 #[cfg(target_os = "macos")]
7607 drop(exec_ack);
7608
7609 #[cfg(windows)]
7611 let job = contain_spawned_child(&child, spec)?;
7612 let spawned_at_ms = unix_ms_now();
7613 let spawned_from = spec.program.clone();
7614 let spawned_file_identity = spawned_file_identity(&spawned_from);
7615 let pid = child.id().ok_or_else(|| SuperviseError::Spawn {
7616 program: spec.program.clone(),
7617 source: io::Error::other("spawned child exposed no live pid"),
7618 cgroup_path: cgroup_path.clone(),
7619 })?;
7620 let process_start_time = crate::provenance::process_start_time(pid);
7621 let process_identity = process_start_time.map(|start_time| ProcessIdentity { pid, start_time });
7622 #[cfg(all(test, target_os = "macos"))]
7623 privacy_exec_boundary_tests::before_image_sample(spec, pid);
7624 let recorded_image = observe_spawned_image(pid);
7630 #[cfg(target_os = "macos")]
7634 let recorded_image = if privacy_exec.is_some() {
7635 None
7636 } else {
7637 recorded_image
7638 };
7639 #[cfg(target_os = "linux")]
7640 let recorded_cgroup_name = cgroup_path.as_ref().map(|_| cgroup_name.clone());
7641 #[cfg(not(target_os = "linux"))]
7642 let recorded_cgroup_name = None;
7643 let roster_guard = roster.admit(
7644 spec.module_id.clone(),
7645 pid,
7646 spec.protocol,
7647 process_start_time,
7648 crate::child_roster::RecordedIdentity {
7649 start_time: recorded_image.map(|image| image.start_time),
7650 executable: recorded_image
7651 .and_then(|image| image.executable)
7652 .map(crate::live_children::ExecutableIdentity::from),
7653 cgroup_name: recorded_cgroup_name,
7654 #[cfg(target_os = "linux")]
7655 cgroup_placement: cgroup_placement.cloned(),
7656 },
7657 );
7658 if roster.is_closed() {
7667 #[cfg(target_os = "linux")]
7669 kill_module_cgroup(cgroup_placement, &cgroup_name);
7670 if let Err(error) = child.start_kill() {
7671 debug!(module_id = %spec.module_id, pid, %error, "kill of a process spawned during daemon shutdown failed; it may already have exited");
7672 }
7673 #[cfg(target_os = "linux")]
7674 if let Some(placement) = cgroup_placement {
7675 while matches!(child.try_wait(), Ok(None)) {
7679 std::thread::yield_now();
7680 }
7681 if matches!(
7682 subc_cgroup::kill_module(Some(placement), &cgroup_name),
7683 subc_cgroup::KillOutcome::Killed
7684 ) {
7685 if let Ok(path) = placement.module_path(&cgroup_name) {
7686 while std::fs::read_to_string(path.join("cgroup.events"))
7687 .ok()
7688 .is_some_and(|events| events.lines().any(|line| line == "populated 1"))
7689 {
7690 std::thread::yield_now();
7691 }
7692 }
7693 }
7694 remove_module_cgroup(placement, &cgroup_name);
7695 }
7696 drop(roster_guard);
7697 return Err(SuperviseError::Spawn {
7698 program: spec.program.clone(),
7699 source: io::Error::other(
7700 "the daemon began shutting down while this process was starting; ended it",
7701 ),
7702 cgroup_path,
7703 });
7704 }
7705
7706 let stdout_pump = match child.stdout.take() {
7707 Some(stdout) => Some(tokio::spawn(pump_stdout_to(stdout, output_sink.clone()))),
7708 None => {
7709 warn!(
7710 module_id = %spec.module_id,
7711 "spawned child exposed no stdout pipe; file capture will be incomplete"
7712 );
7713 None
7714 }
7715 };
7716 let stderr_pump = match child.stderr.take() {
7717 Some(stderr) => {
7718 let generation = ring
7719 .lock()
7720 .unwrap_or_else(|poisoned| poisoned.into_inner())
7721 .begin_process();
7722 Some(StderrPump {
7723 task: tokio::spawn(pump_stderr_to(
7724 stderr,
7725 Arc::clone(ring),
7726 generation,
7727 output_sink,
7728 )),
7729 generation,
7730 })
7731 }
7732 None => {
7733 ring.lock()
7737 .unwrap_or_else(|poisoned| poisoned.into_inner())
7738 .mark_not_captured("stderr pipe was not available on spawn");
7739 warn!(
7740 module_id = %spec.module_id,
7741 "spawned child exposed no stderr pipe; tail will be unavailable"
7742 );
7743 None
7744 }
7745 };
7746
7747 Ok(SupervisedChild {
7748 child,
7749 protocol: spec.protocol,
7750 #[cfg(target_os = "linux")]
7751 module_id: cgroup_name,
7752 #[cfg(target_os = "linux")]
7753 cgroup_placement: cgroup_placement.cloned(),
7754 #[cfg(windows)]
7755 job,
7756 stdout_pump,
7757 stderr_pump,
7758 stderr_ring: Arc::clone(ring),
7759 spawned_at_ms,
7760 spawned_from,
7761 spawned_file_identity,
7762 process_start_time,
7763 process_identity,
7764 pid,
7765 roster_guard: Some(roster_guard),
7766 #[cfg(target_os = "macos")]
7767 privacy_exec,
7768 #[cfg(target_os = "macos")]
7769 report_ready: Arc::new(OnceLock::new()),
7770 spawn_failure: None,
7771 })
7772}
7773
7774#[cfg(target_os = "linux")]
7775pub(crate) fn kill_module_cgroup(placement: Option<&subc_cgroup::Placement>, module_id: &str) {
7776 use subc_cgroup::KillOutcome;
7777 match subc_cgroup::kill_module(placement, module_id) {
7778 KillOutcome::Killed => {}
7779 KillOutcome::NotPlaced | KillOutcome::Unsupported => {
7780 debug!(
7781 module_id,
7782 "cgroup tree kill unavailable; using direct-child kill"
7783 );
7784 }
7785 KillOutcome::IoError { path, error } => {
7786 warn!(module_id, path = %path.display(), %error, "cgroup tree kill failed; using direct-child kill");
7787 }
7788 }
7789}
7790
7791#[cfg(windows)]
7805fn contain_spawned_child(
7806 child: &Child,
7807 spec: &ModuleSpec,
7808) -> Result<Option<subc_jobobject::JobObject>, SuperviseError> {
7809 let module_id = spec.module_id.as_str();
7810 let Some(pid) = child.id() else {
7811 warn!(
7814 module_id,
7815 "spawned child had already exited before containment; no job object attached"
7816 );
7817 return Ok(None);
7818 };
7819
7820 let job = match subc_jobobject::JobObject::new() {
7821 Ok(job) => job,
7822 Err(source) => {
7823 warn!(
7824 module_id,
7825 error = %source,
7826 "could not create a job object; this module's helper processes will not be \
7827 reaped on teardown"
7828 );
7829 resume_suspended_child(pid, spec)?;
7832 return Ok(None);
7833 }
7834 };
7835
7836 if let Err(source) = job.assign(child) {
7837 warn!(
7838 module_id,
7839 error = %source,
7840 "could not assign the child to its job object; this module's helper processes \
7841 will not be reaped on teardown"
7842 );
7843 resume_suspended_child(pid, spec)?;
7844 return Ok(None);
7845 }
7846
7847 resume_suspended_child(pid, spec)?;
7848 Ok(Some(job))
7849}
7850
7851#[cfg(windows)]
7856fn resume_suspended_child(pid: u32, spec: &ModuleSpec) -> Result<(), SuperviseError> {
7857 if let Err(source) = subc_jobobject::resume_main_thread(pid) {
7858 let _ = std::process::Command::new("taskkill.exe")
7862 .args(["/PID", &pid.to_string(), "/T", "/F"])
7863 .stdin(Stdio::null())
7864 .stdout(Stdio::null())
7865 .stderr(Stdio::null())
7866 .status();
7867 return Err(SuperviseError::Spawn {
7868 program: spec.program.clone(),
7869 source,
7870 cgroup_path: None,
7871 });
7872 }
7873 Ok(())
7874}
7875
7876#[cfg(target_os = "linux")]
7877fn remove_module_cgroup(placement: &subc_cgroup::Placement, module_id: &str) {
7878 match placement.remove_module(module_id) {
7879 Ok(()) => debug!(module_id, "removed module cgroup after process exit"),
7880 Err(error) => warn!(
7881 module_id,
7882 error = %error,
7883 "could not remove module cgroup after process exit; continuing teardown"
7884 ),
7885 }
7886}
7887
7888#[cfg(target_os = "linux")]
7889async fn cleanup_reaped_cgroup(placement: &subc_cgroup::Placement, module_id: &str) {
7890 if matches!(
7894 subc_cgroup::kill_module(Some(placement), module_id),
7895 subc_cgroup::KillOutcome::Killed
7896 ) {
7897 if let Ok(path) = placement.module_path(module_id) {
7898 while std::fs::read_to_string(path.join("cgroup.events"))
7899 .ok()
7900 .is_some_and(|events| events.lines().any(|line| line == "populated 1"))
7901 {
7902 sleep(Duration::from_millis(1)).await;
7903 }
7904 }
7905 }
7906 remove_module_cgroup(placement, module_id);
7907}
7908
7909#[cfg(target_os = "linux")]
7910fn apply_cgroup_placement(
7911 command: &mut Command,
7912 spec: &ModuleSpec,
7913 path: &std::path::Path,
7914) -> Result<(), SuperviseError> {
7915 subc_cgroup::apply(command, path).map_err(|source| SuperviseError::Cgroup {
7916 module_id: spec.module_id.clone(),
7917 source,
7918 })
7919}
7920
7921fn capture_retention(spec: &ModuleSpec) -> Retention {
7922 let defaults = Retention::default();
7923 let value = |name: &str| {
7924 spec.env
7925 .iter()
7926 .rev()
7927 .find_map(|(key, value)| (key == name).then_some(value.as_str()))
7928 };
7929 Retention {
7930 max_file_mb: value(CAPTURE_MAX_FILE_MB_ENV)
7931 .and_then(|value| value.parse().ok())
7932 .unwrap_or(defaults.max_file_mb),
7933 keep: value(CAPTURE_KEEP_ENV)
7934 .and_then(|value| value.parse().ok())
7935 .unwrap_or(defaults.keep),
7936 max_age_days: value(CAPTURE_MAX_AGE_DAYS_ENV)
7937 .and_then(|value| value.parse().ok())
7938 .unwrap_or(defaults.max_age_days),
7939 }
7940}
7941
7942fn generate_launch_nonce() -> Result<String, SuperviseError> {
7945 let mut bytes = [0u8; 32];
7946 getrandom::getrandom(&mut bytes).map_err(|source| SuperviseError::LaunchNonce {
7947 reason: source.to_string(),
7948 })?;
7949 let mut hex = String::with_capacity(64);
7950 for b in bytes {
7951 use std::fmt::Write;
7952 let _ = write!(hex, "{b:02x}");
7953 }
7954 Ok(hex)
7955}
7956
7957fn constant_time_eq(a: &[u8], b: &[u8]) -> bool {
7960 if a.len() != b.len() {
7961 return false;
7962 }
7963 let mut diff = 0u8;
7964 for (x, y) in a.iter().zip(b.iter()) {
7965 diff |= x ^ y;
7966 }
7967 diff == 0
7968}
7969
7970fn observe_spawned_image(pid: u32) -> Option<subc_os::Observation> {
7974 subc_os::Process::open(pid)
7975 .ok()
7976 .flatten()
7977 .and_then(|process| process.observe())
7978}
7979
7980fn spawn_and_mark_running(
7981 spec: &ModuleSpec,
7982 runtime: &SupervisorRuntimeConfig,
7983 snapshot: &SharedSnapshot,
7984) -> Result<SupervisedChild, SuperviseError> {
7985 let child = spawn_child(
7986 spec,
7987 runtime.connection_file_path.as_deref(),
7988 runtime.supervisor_handle.as_ref(),
7989 &runtime.stderr_ring,
7990 runtime.capture_logs_dir.as_deref(),
7991 &runtime.child_roster,
7992 #[cfg(target_os = "linux")]
7993 runtime.cgroup_placement.as_ref(),
7994 )?;
7995 set_running(snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
7996 Ok(child)
7997}
7998
7999enum RegistrationWaitOutcome {
8000 Registered,
8001 Exited(ExitReport),
8002 TimedOut,
8003}
8004
8005struct ReloadRegistrationFailure {
8006 exit_report: ExitReport,
8007 reason: String,
8008}
8009
8010#[derive(Debug, Clone, Copy, PartialEq, Eq)]
8011enum BusyGaugeObservation {
8012 Quiescent,
8013 Busy,
8014 Omitted,
8015}
8016
8017fn busy_gauge_observation(metrics: Option<&Value>, gauges: &[String]) -> BusyGaugeObservation {
8018 let Some(metrics) = metrics.and_then(Value::as_object) else {
8019 return BusyGaugeObservation::Omitted;
8020 };
8021 let mut sum = 0u128;
8022 for gauge in gauges {
8023 let Some(value) = metrics.get(gauge) else {
8024 return BusyGaugeObservation::Omitted;
8025 };
8026 let Some(value) = value.as_u64() else {
8027 return BusyGaugeObservation::Busy;
8028 };
8029 sum = sum.saturating_add(u128::from(value));
8030 }
8031 if sum == 0 {
8032 BusyGaugeObservation::Quiescent
8033 } else {
8034 BusyGaugeObservation::Busy
8035 }
8036}
8037
8038fn declared_busy_gauges(
8039 registry: &Registry,
8040 module_id: &str,
8041) -> Result<Vec<String>, SuperviseError> {
8042 busy_gauges_of(
8043 registry
8044 .get_module(module_id)
8045 .map_err(SuperviseError::Registry)?,
8046 )
8047}
8048
8049fn declared_busy_gauges_for_connection(
8053 registry: &Registry,
8054 connection_id: ConnectionId,
8055) -> Result<Vec<String>, SuperviseError> {
8056 busy_gauges_of(
8057 registry
8058 .get_module_by_connection(connection_id)
8059 .map_err(SuperviseError::Registry)?,
8060 )
8061}
8062
8063fn busy_gauges_of(
8064 registration: Option<crate::registry::ModuleRegistration>,
8065) -> Result<Vec<String>, SuperviseError> {
8066 let Some(registration) = registration else {
8067 return Ok(Vec::new());
8068 };
8069 let Some(self_signals) = registration.manifest.self_signals else {
8070 return Ok(Vec::new());
8071 };
8072
8073 let mut gauges = Vec::new();
8074 for declaration in self_signals {
8075 if declaration.kind != SelfSignalKind::Busy {
8076 continue;
8077 }
8078 match declaration.anchored_to {
8079 SignalAnchor::HealthGauges { gauges: declared } if !declared.is_empty() => {
8080 gauges.extend(declared)
8081 }
8082 _ => {
8083 gauges.push(String::new());
8086 }
8087 }
8088 }
8089 Ok(gauges)
8090}
8091
8092async fn wait_for_forwarding_quiescence(
8097 forwarding: &ForwardingTable,
8098 module_id: &str,
8099 runtime: &SupervisorRuntimeConfig,
8100 endpoint: crate::ModuleEndpointId,
8101 deadline: Instant,
8102 busy_gauges: &[String],
8103 scope: DrainScope,
8104) -> Result<bool, SuperviseError> {
8105 let mut gauges_quiescent = busy_gauges.is_empty();
8106 let mut next_probe_at = Instant::now();
8107 let mut omission_counted = false;
8108
8109 loop {
8110 let now = Instant::now();
8111 if !busy_gauges.is_empty() && now >= next_probe_at && now < deadline {
8112 let report = match scope {
8113 DrainScope::Active => probe_module_health(module_id, runtime, Some(deadline)).await,
8114 DrainScope::Endpoint(endpoint) => {
8115 probe_endpoint_health(endpoint, runtime, Some(deadline)).await
8116 }
8117 };
8118 gauges_quiescent = match report {
8119 Ok(report) => match busy_gauge_observation(report.metrics.as_ref(), busy_gauges) {
8120 BusyGaugeObservation::Quiescent => true,
8121 BusyGaugeObservation::Busy => false,
8122 BusyGaugeObservation::Omitted => {
8123 if !omission_counted {
8124 forwarding
8125 .counters()
8126 .increment_drains_with_undeclared_gauge();
8127 omission_counted = true;
8128 }
8129 false
8130 }
8131 },
8132 Err(err) => {
8133 warn!(
8134 module_id,
8135 error = %err,
8136 "drain health.check did not produce declared busy gauges; treating module as busy"
8137 );
8138 false
8139 }
8140 };
8141 next_probe_at = Instant::now() + runtime.health.cadence.max(REGISTRY_RELEASE_POLL);
8142 }
8143
8144 let in_flight = forwarding
8145 .endpoint_in_flight_count(endpoint)
8146 .map_err(SuperviseError::Forwarding)?;
8147 if in_flight == 0 && gauges_quiescent {
8148 return Ok(true);
8149 }
8150
8151 let now = Instant::now();
8152 if now >= deadline {
8153 return Ok(false);
8154 }
8155 let mut wait = deadline
8156 .saturating_duration_since(now)
8157 .min(REGISTRY_RELEASE_POLL);
8158 if !busy_gauges.is_empty() {
8159 wait = wait.min(next_probe_at.saturating_duration_since(now));
8160 }
8161 sleep(wait).await;
8162 }
8163}
8164
8165fn drained_after_quiescence_wait(wait_result: &Result<bool, SuperviseError>) -> bool {
8173 match wait_result {
8174 Ok(drained) => *drained,
8175 Err(_) => false,
8176 }
8177}
8178
8179fn send_route_goodbyes(forwarding: &ForwardingTable, released_routes: Vec<GoodbyeTarget>) {
8180 for released in released_routes {
8181 let frame = match Frame::build_with_version(
8182 released.negotiated_ver,
8183 FrameType::Goodbye,
8184 control_flags(),
8185 released.channel,
8186 released.epoch,
8187 0,
8188 Vec::new(),
8189 ) {
8190 Ok(frame) => frame,
8191 Err(err) => {
8192 warn!(
8193 route_channel = released.channel,
8194 error = %err,
8195 "failed to build supervisor drain route GOODBYE frame"
8196 );
8197 continue;
8198 }
8199 };
8200 if !released.close_on_delivery_failure() {
8201 crate::forwarding::send_module_route_goodbye(
8202 &forwarding.counters(),
8203 &released.sink,
8204 frame,
8205 released.module_id.as_deref(),
8206 "supervisor drain",
8207 );
8208 continue;
8209 }
8210 if let Err(err) = released.sink.try_send(frame) {
8211 warn!(
8212 target_connection_id = released.connection_id.get(),
8213 route_channel = released.channel,
8214 error = %err,
8215 "supervisor drain route GOODBYE was not delivered to client; closing target connection"
8216 );
8217 let _ = forwarding.escalate_client_delivery_failure(
8218 released.connection_id,
8219 released.channel,
8220 released.epoch,
8221 CloseReason::new(
8222 "route_goodbye_delivery_failed",
8223 format!(
8224 "failed to enqueue supervisor drain route GOODBYE for channel {}: {err}",
8225 released.channel
8226 ),
8227 ),
8228 crate::forwarding::UndeliveredFrame {
8229 module_id: released.module_id.as_deref(),
8230 sink: &released.sink,
8231 },
8232 );
8233 }
8234 }
8235}
8236
8237fn send_module_draining(
8238 module_id: &str,
8239 reason: RouteCloseReason,
8240 deadline_ms: u64,
8241 target: &ModuleDrainTarget,
8242) {
8243 let body = match serde_json::to_vec(&ModuleControlCommand::Draining {
8244 reason,
8245 deadline_ms,
8246 }) {
8247 Ok(body) => body,
8248 Err(err) => {
8249 warn!(
8250 module_id,
8251 error = %err,
8252 "failed to encode module draining command"
8253 );
8254 return;
8255 }
8256 };
8257 let frame = match Frame::build_with_version(
8258 target.negotiated_ver,
8259 FrameType::Push,
8260 control_flags(),
8261 0,
8262 0,
8263 0,
8264 body,
8265 ) {
8266 Ok(frame) => frame,
8267 Err(err) => {
8268 warn!(
8269 module_id,
8270 error = %err,
8271 "failed to build module draining command frame"
8272 );
8273 return;
8274 }
8275 };
8276 if let Err(err) = target.sink.try_send(frame) {
8277 warn!(
8278 module_id,
8279 target_connection_id = target.endpoint.connection_id.get(),
8280 error = %err,
8281 "module draining command was not delivered to peer"
8282 );
8283 }
8284}
8285
8286fn module_goodbye_frame(module_id: &str, negotiated_ver: u8) -> Option<Frame> {
8288 match Frame::build_with_version(
8289 negotiated_ver,
8290 FrameType::Goodbye,
8291 control_flags(),
8292 0,
8293 0,
8294 0,
8295 Vec::new(),
8296 ) {
8297 Ok(frame) => Some(frame),
8298 Err(err) => {
8299 warn!(
8300 module_id,
8301 error = %err,
8302 "failed to build module GOODBYE frame"
8303 );
8304 None
8305 }
8306 }
8307}
8308
8309#[cfg(unix)]
8323async fn send_module_goodbyes_for_daemon_shutdown(
8324 forwarding: &Arc<ForwardingTable>,
8325 reason: &CloseReason,
8326 wait_for_flush: bool,
8327) {
8328 const GOODBYE_BUDGET: Duration = Duration::from_millis(500);
8329 let targets = match forwarding.module_connections() {
8330 Ok(targets) => targets,
8331 Err(err) => {
8332 warn!(error = %err, "could not list module connections for shutdown GOODBYE");
8333 return;
8334 }
8335 };
8336 let deadline = Instant::now() + GOODBYE_BUDGET;
8337 let mut sends = tokio::task::JoinSet::new();
8338 for target in targets {
8339 let Some(frame) = module_goodbye_frame(&target.module_id, target.negotiated_ver) else {
8340 continue;
8341 };
8342 if !wait_for_flush {
8343 if let Err(err) = target.sink.try_send(frame) {
8344 debug!(
8345 module_id = %target.module_id,
8346 error = %err,
8347 "shutdown module GOODBYE was not queued"
8348 );
8349 }
8350 continue;
8351 }
8352 let forwarding = Arc::clone(forwarding);
8353 let reason = reason.clone();
8354 sends.spawn(async move {
8355 match timeout_at(deadline, target.sink.send_flushed(frame)).await {
8356 Ok(Ok(())) => {}
8357 Ok(Err(err)) => debug!(
8358 module_id = %target.module_id,
8359 error = %err,
8360 "module connection closed before its shutdown GOODBYE was written"
8361 ),
8362 Err(_) => warn!(
8363 module_id = %target.module_id,
8364 budget = ?GOODBYE_BUDGET,
8365 "shutdown module GOODBYE was not written within its budget; closing anyway"
8366 ),
8367 }
8368 forwarding.request_connection_close(target.endpoint.connection_id, reason);
8369 });
8370 }
8371 while sends.join_next().await.is_some() {}
8373}
8374
8375fn send_module_goodbye(module_id: &str, forwarding: &ForwardingTable, target: &ModuleDrainTarget) {
8376 let Some(frame) = module_goodbye_frame(module_id, target.negotiated_ver) else {
8377 return;
8378 };
8379 if let Err(err) = target.sink.try_send(frame) {
8380 warn!(
8381 module_id,
8382 target_connection_id = target.endpoint.connection_id.get(),
8383 error = %err,
8384 "supervisor drain module GOODBYE was not delivered to peer; closing module connection"
8385 );
8386 forwarding.request_connection_close(
8387 target.endpoint.connection_id,
8388 CloseReason::new(
8389 "module_goodbye_delivery_failed",
8390 format!("failed to enqueue supervisor drain module GOODBYE for module '{module_id}': {err}"),
8391 ),
8392 );
8393 }
8394}
8395
8396#[derive(Clone, Copy)]
8397struct ForwardingDrainContext<'a> {
8398 spec: &'a ModuleSpec,
8399 runtime: &'a SupervisorRuntimeConfig,
8400 registry: &'a Registry,
8401 scope: DrainScope,
8402}
8403
8404#[derive(Debug, Clone, Copy, PartialEq, Eq)]
8406enum DrainScope {
8407 Active,
8410 Endpoint(crate::ModuleEndpointId),
8415}
8416
8417#[derive(Debug, Clone, Copy, PartialEq, Eq)]
8425enum StopNotice {
8426 SentOverConnection,
8429 NoConnection,
8433 NotSent,
8437}
8438
8439async fn begin_forwarding_drain(
8440 spec: &ModuleSpec,
8441 runtime: &SupervisorRuntimeConfig,
8442 registry: &Registry,
8443 snapshot: &SharedSnapshot,
8444 enabled: Option<bool>,
8445 reason: RouteCloseReason,
8446) -> Result<StopNotice, SuperviseError> {
8447 let Some(forwarding) = runtime.forwarding.as_ref() else {
8448 return Err(SuperviseError::ReloadUnavailable {
8449 module_id: spec.module_id.clone(),
8450 reason: "supervisor was not configured with a forwarding table".to_string(),
8451 });
8452 };
8453
8454 begin_forwarding_drain_with(
8455 forwarding,
8456 ForwardingDrainContext {
8457 spec,
8458 runtime,
8459 registry,
8460 scope: DrainScope::Active,
8461 },
8462 snapshot,
8463 enabled,
8464 reason,
8465 runtime.drain_timeout,
8466 )
8467 .await
8468}
8469
8470async fn begin_forwarding_drain_if_configured(
8471 spec: &ModuleSpec,
8472 runtime: &SupervisorRuntimeConfig,
8473 registry: &Registry,
8474 snapshot: &SharedSnapshot,
8475 enabled: Option<bool>,
8476 reason: RouteCloseReason,
8477) -> Result<StopNotice, SuperviseError> {
8478 begin_forwarding_drain_with_timeout(
8479 spec,
8480 runtime,
8481 registry,
8482 snapshot,
8483 enabled,
8484 reason,
8485 runtime.drain_timeout,
8486 )
8487 .await
8488}
8489
8490async fn begin_forwarding_drain_with_timeout(
8494 spec: &ModuleSpec,
8495 runtime: &SupervisorRuntimeConfig,
8496 registry: &Registry,
8497 snapshot: &SharedSnapshot,
8498 enabled: Option<bool>,
8499 reason: RouteCloseReason,
8500 drain_timeout: Duration,
8501) -> Result<StopNotice, SuperviseError> {
8502 let Some(forwarding) = runtime.forwarding.as_ref() else {
8503 return Ok(StopNotice::NotSent);
8504 };
8505
8506 begin_forwarding_drain_with(
8507 forwarding,
8508 ForwardingDrainContext {
8509 spec,
8510 runtime,
8511 registry,
8512 scope: DrainScope::Active,
8513 },
8514 snapshot,
8515 enabled,
8516 reason,
8517 drain_timeout,
8518 )
8519 .await
8520}
8521
8522async fn begin_forwarding_drain_with(
8523 forwarding: &ForwardingTable,
8524 context: ForwardingDrainContext<'_>,
8525 snapshot: &SharedSnapshot,
8526 enabled: Option<bool>,
8527 reason: RouteCloseReason,
8528 drain_timeout: Duration,
8529) -> Result<StopNotice, SuperviseError> {
8530 let ForwardingDrainContext {
8531 spec,
8532 runtime,
8533 registry,
8534 scope,
8535 } = context;
8536 debug_assert_ne!(reason, RouteCloseReason::Crash);
8537 let terminal = matches!(reason, RouteCloseReason::Disable);
8538 let drain_started_at = Instant::now();
8539 let drain_deadline = drain_started_at + drain_timeout;
8540 let deadline_ms =
8541 unix_ms_now().saturating_add(u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX));
8542 let busy_gauges = match scope {
8543 DrainScope::Active => declared_busy_gauges(registry, &spec.module_id)?,
8544 DrainScope::Endpoint(endpoint) => {
8545 declared_busy_gauges_for_connection(registry, endpoint.connection_id)?
8546 }
8547 };
8548
8549 let gate_started = Instant::now();
8552 let drain_target = match scope {
8553 DrainScope::Active => forwarding.begin_module_drain(&spec.module_id, reason),
8554 DrainScope::Endpoint(endpoint) => forwarding.begin_endpoint_drain(endpoint, reason),
8555 }
8556 .map_err(SuperviseError::Forwarding)?;
8557 info!(
8562 module_id = %spec.module_id,
8563 ?reason,
8564 gate_ms = u64::try_from(gate_started.elapsed().as_millis()).unwrap_or(u64::MAX),
8565 connected = drain_target.is_some(),
8566 "module drain began; route admission closed"
8567 );
8568 if scope == DrainScope::Active {
8569 update_snapshot(snapshot, Some(&spec.module_id), |state| {
8570 state.state = ModuleState::Draining;
8571 state.draining_to_replace =
8572 matches!(reason, RouteCloseReason::Restart | RouteCloseReason::Reload);
8573 if let Some(enabled) = enabled {
8574 state.enabled = enabled;
8575 }
8576 })?;
8577 }
8578
8579 let Some(target) = drain_target.as_ref() else {
8580 return Ok(StopNotice::NoConnection);
8584 };
8585 {
8586 send_module_draining(&spec.module_id, reason, deadline_ms, target);
8587 let routes = forwarding
8588 .endpoint_routes(target.endpoint)
8589 .map_err(SuperviseError::Forwarding)?;
8590 let routes_notified = routes.len();
8591 crate::control::send_route_control_pushes(
8592 forwarding,
8593 routes.clone(),
8594 ClientControlPush::RouteClosing {
8595 module_id: spec.module_id.clone(),
8596 channels: Vec::new(),
8597 reason,
8598 },
8599 );
8600 send_route_goodbyes(forwarding, target.abandoned_bindings.clone());
8601
8602 let wait_result = wait_for_forwarding_quiescence(
8608 forwarding,
8609 &spec.module_id,
8610 runtime,
8611 target.endpoint,
8612 drain_deadline,
8613 &busy_gauges,
8614 scope,
8615 )
8616 .await;
8617 let drained = drained_after_quiescence_wait(&wait_result);
8618 if let Err(err) = &wait_result {
8619 error!(
8620 module_id = %spec.module_id,
8621 ?reason,
8622 error = %err,
8623 "forwarding quiescence wait failed after route.closing; forcing route.closed(drained: false) so the client is not left waiting on an unfulfilled promise"
8624 );
8625 } else if !drained {
8626 let holdouts = forwarding
8632 .endpoint_drain_holdouts(target.endpoint)
8633 .unwrap_or_default();
8634 warn!(
8635 module_id = %spec.module_id,
8636 waited = ?drain_timeout,
8637 ?reason,
8638 held_requests = holdouts.requests,
8639 held_routes = holdouts.routes,
8640 total_routes = holdouts.total_routes,
8641 top_connections = ?holdouts.top_connections,
8642 held = %holdouts
8645 .held
8646 .iter()
8647 .map(|(channel, corr)| format!("{channel}:{corr}"))
8648 .collect::<Vec<_>>()
8649 .join(","),
8650 "route drain timed out before request quiescence; forcing teardown"
8651 );
8652 }
8653 crate::control::send_route_control_pushes(
8654 forwarding,
8655 routes,
8656 ClientControlPush::RouteClosed {
8657 module_id: spec.module_id.clone(),
8658 channels: Vec::new(),
8659 reason,
8660 drained,
8661 abandoned: target.abandoned_bindings.len() as u32,
8662 excluded_subscriptions: target.excluded_subscriptions,
8663 terminal: Some(terminal),
8664 },
8665 );
8666 wait_result?;
8667
8668 let released_routes = match forwarding.release_module_endpoint_routes(target.endpoint) {
8674 Ok(routes) => routes,
8675 Err(err) => {
8676 warn!(
8677 module_id = %spec.module_id,
8678 ?reason,
8679 error = %err,
8680 "failed to release module endpoint routes after route.closed; module GOODBYE will still be sent"
8681 );
8682 send_module_goodbye(&spec.module_id, forwarding, target);
8683 return Err(SuperviseError::Forwarding(err));
8684 }
8685 };
8686 let route_goodbye_count = released_routes.len();
8687 send_route_goodbyes(forwarding, released_routes);
8688 send_module_goodbye(&spec.module_id, forwarding, target);
8689
8690 info!(
8696 module_id = %spec.module_id,
8697 ?reason,
8698 routes_notified,
8699 route_goodbyes = route_goodbye_count,
8700 abandoned_reservations = target.abandoned_bindings.len(),
8701 excluded_subscriptions = target.excluded_subscriptions,
8702 drained,
8703 "module drain complete; consumers notified via route.closing/route.closed pushes and per-route GOODBYE frames"
8704 );
8705 }
8706
8707 Ok(StopNotice::SentOverConnection)
8708}
8709
8710async fn wait_for_registration_after_reload(
8713 registry: &Registry,
8714 module_id: &str,
8715 snapshot: &SharedSnapshot,
8716 child: &mut SupervisedChild,
8717 wait: Duration,
8718) -> Result<RegistrationWaitOutcome, SuperviseError> {
8719 wait_for_slot_registration(
8720 registry,
8721 crate::registry::RegistrationSlot::Active(module_id),
8722 module_id,
8723 snapshot,
8724 child,
8725 wait,
8726 )
8727 .await
8728}
8729
8730async fn wait_for_slot_registration(
8738 registry: &Registry,
8739 slot: crate::registry::RegistrationSlot<'_>,
8740 module_id: &str,
8741 snapshot: &SharedSnapshot,
8742 child: &mut SupervisedChild,
8743 wait: Duration,
8744) -> Result<RegistrationWaitOutcome, SuperviseError> {
8745 let deadline = Instant::now() + wait;
8746 loop {
8747 if registry
8748 .registration(slot)
8749 .map_err(SuperviseError::Registry)?
8750 .is_some()
8751 {
8752 return Ok(RegistrationWaitOutcome::Registered);
8753 }
8754
8755 let now = Instant::now();
8756 if now >= deadline {
8757 return Ok(RegistrationWaitOutcome::TimedOut);
8758 }
8759 let remaining = deadline.saturating_duration_since(now);
8760 let poll = remaining.min(REGISTRY_RELEASE_POLL);
8761
8762 tokio::select! {
8763 wait_result = child.wait() => {
8764 let status = wait_result.map_err(|source| SuperviseError::Wait {
8765 module_id: module_id.to_string(),
8766 source,
8767 })?;
8768 return Ok(RegistrationWaitOutcome::Exited(classify_reaped_child_exit(
8769 snapshot,
8770 child,
8771 &status,
8772 )));
8773 }
8774 _ = sleep(poll) => {}
8775 }
8776 }
8777}
8778
8779fn registration_failure_exit_report(mut exit_report: ExitReport) -> ExitReport {
8780 if exit_report.kind != ExitKind::DeliberateSeverance {
8783 exit_report.kind = ExitKind::Crash;
8784 }
8785 exit_report
8786}
8787
8788async fn handle_reload_child_registration_failure(
8789 spec: &ModuleSpec,
8790 runtime: &SupervisorRuntimeConfig,
8791 registry: &Registry,
8792 process_liveness: &SupervisorProcessLiveness,
8793 snapshot: &SharedSnapshot,
8794 _child: &mut Option<SupervisedChild>,
8795 failure: ReloadRegistrationFailure,
8796) -> Result<(), SuperviseError> {
8797 let ReloadRegistrationFailure {
8798 exit_report,
8799 reason,
8800 } = failure;
8801 match on_child_exit(
8802 spec,
8803 runtime.restart_policy,
8804 registry,
8805 snapshot,
8806 &runtime.terminal_ring,
8807 &runtime.spawn_events,
8808 &runtime.child_roster,
8809 exit_report,
8810 )
8811 .await
8812 {
8813 NextAction::Stop {
8814 registration_released,
8815 } => {
8816 if registration_released {
8817 process_liveness.untrack_if_current(&spec.module_id, snapshot);
8818 }
8819 }
8820 NextAction::Restart { schedule } => {
8821 let delay = schedule.map_or(runtime.restart_policy.delay_for_restart(0), |schedule| {
8822 schedule.delay
8823 });
8824 if let Some(schedule) = schedule {
8825 log_crash_respawn(&spec.module_id, schedule);
8826 }
8827 schedule_respawn(
8828 runtime,
8829 snapshot,
8830 &spec.module_id,
8831 delay,
8832 RespawnKind::Spawn,
8833 )?;
8834 }
8835 }
8836 Err(SuperviseError::ReloadFailed {
8837 module_id: spec.module_id.clone(),
8838 reason,
8839 })
8840}
8841
8842async fn handle_reload_spawn_failure(
8843 spec: &ModuleSpec,
8844 runtime: &SupervisorRuntimeConfig,
8845 process_liveness: &SupervisorProcessLiveness,
8846 snapshot: &SharedSnapshot,
8847 _child: &mut Option<SupervisedChild>,
8848 reason: String,
8849) -> Result<(), SuperviseError> {
8850 let now = Instant::now();
8851 let mut schedule = None;
8852 update_snapshot(snapshot, Some(&spec.module_id), |state| {
8853 clear_current_process_facts(state);
8854 if state.enabled {
8855 schedule = state.next_crash_restart(&runtime.restart_policy, now);
8856 state.state = if schedule.is_some() {
8857 ModuleState::Restarting
8858 } else {
8859 ModuleState::Failed
8860 };
8861 } else {
8862 state.state = ModuleState::Disabled;
8863 }
8864 })?;
8865 if let Some(schedule) = schedule {
8866 schedule_respawn(
8867 runtime,
8868 snapshot,
8869 &spec.module_id,
8870 schedule.delay,
8871 RespawnKind::Spawn,
8872 )?;
8873 } else {
8874 process_liveness.untrack_if_current(&spec.module_id, snapshot);
8875 }
8876 Err(SuperviseError::ReloadFailed {
8877 module_id: spec.module_id.clone(),
8878 reason,
8879 })
8880}
8881
8882fn control_flags() -> Flags {
8883 Flags::new(false, Priority::Passive, false)
8884}
8885
8886#[allow(clippy::too_many_arguments)]
8887async fn drain_optional_child(
8888 module_id: &str,
8889 protocol: ModuleProtocol,
8890 stop_notice: StopNotice,
8891 registry: &Registry,
8892 forwarding: Option<&ForwardingTable>,
8893 snapshot: &SharedSnapshot,
8894 terminal_ring: &Arc<Mutex<TerminalRing>>,
8895 spawn_events: &SpawnEventFeed,
8896 child: &mut Option<SupervisedChild>,
8897 drain_timeout: Duration,
8898 final_state: ModuleState,
8899 enabled: Option<bool>,
8900) -> Result<(), SuperviseError> {
8901 if let Some(child) = child.take() {
8902 drain_child_to_state(
8903 module_id,
8904 protocol,
8905 stop_notice,
8906 registry,
8907 forwarding,
8908 snapshot,
8909 terminal_ring,
8910 spawn_events,
8911 child,
8912 drain_timeout,
8913 final_state,
8914 enabled,
8915 )
8916 .await
8917 } else {
8918 update_snapshot(snapshot, Some(module_id), |state| {
8919 state.state = final_state;
8920 if let Some(enabled) = enabled {
8921 state.enabled = enabled;
8922 }
8923 clear_current_process_facts(state);
8924 })?;
8925 release_dead_registration(registry, forwarding, snapshot, module_id).await
8926 }
8927}
8928
8929#[allow(clippy::too_many_arguments)]
8930async fn drain_child_to_state(
8931 module_id: &str,
8932 _protocol: ModuleProtocol,
8933 stop_notice: StopNotice,
8934 registry: &Registry,
8935 forwarding: Option<&ForwardingTable>,
8936 snapshot: &SharedSnapshot,
8937 terminal_ring: &Arc<Mutex<TerminalRing>>,
8938 spawn_events: &SpawnEventFeed,
8939 mut child: SupervisedChild,
8940 drain_timeout: Duration,
8941 final_state: ModuleState,
8942 enabled: Option<bool>,
8943) -> Result<(), SuperviseError> {
8944 let protocol = child.protocol;
8945 update_snapshot(snapshot, Some(module_id), |state| {
8946 state.state = ModuleState::Draining;
8947 state.draining_to_replace = final_state == ModuleState::Restarting;
8948 if let Some(enabled) = enabled {
8949 state.enabled = enabled;
8950 }
8951 })?;
8952
8953 if stop_notice != StopNotice::SentOverConnection {
8964 if protocol == ModuleProtocol::Subc && stop_notice == StopNotice::NoConnection {
8965 info!(
8966 module_id,
8967 pid = child.pid,
8968 budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
8969 "module has no connection yet; requesting stop by signal"
8970 );
8971 }
8972 request_graceful_stop(module_id, &child);
8973 }
8974
8975 let exit_report = match timeout(drain_timeout, child.wait()).await {
8976 Ok(Ok(status)) => classify_reaped_child_exit(snapshot, &child, &status),
8977 Ok(Err(source)) => {
8978 fail_snapshot(snapshot, Some(module_id), None);
8979 return Err(SuperviseError::Wait {
8980 module_id: module_id.to_string(),
8981 source,
8982 });
8983 }
8984 Err(_) => {
8985 warn!(
8998 module_id,
8999 pid = child.pid,
9000 budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
9001 reason = ?final_state,
9002 ?stop_notice,
9003 "drain budget expired before the module exited; killing it"
9004 );
9005 child.start_kill().map_err(|source| {
9006 fail_snapshot(snapshot, Some(module_id), None);
9007 SuperviseError::Kill {
9008 module_id: module_id.to_string(),
9009 source,
9010 }
9011 })?;
9012 let status = child.wait().await.map_err(|source| {
9013 fail_snapshot(snapshot, Some(module_id), None);
9014 SuperviseError::Wait {
9015 module_id: module_id.to_string(),
9016 source,
9017 }
9018 })?;
9019 classify_reaped_child_exit(snapshot, &child, &status)
9020 }
9021 };
9022
9023 update_snapshot(snapshot, Some(module_id), |state| {
9024 state.state = final_state;
9025 if let Some(enabled) = enabled {
9026 state.enabled = enabled;
9027 }
9028 clear_current_process_facts(state);
9029 state.last_exit = Some(exit_report.clone());
9030 if exit_report.kind == ExitKind::DeliberateSeverance {
9031 state.lifetime_restarts += 1;
9032 }
9033 })?;
9034 let detail = lock_snapshot(snapshot)?.drain_disposition_detail.take();
9035 record_terminal_with_detail(
9036 module_id,
9037 terminal_ring,
9038 spawn_events,
9039 &exit_report,
9040 terminal_disposition(final_state),
9041 detail,
9042 );
9043 child.drain_stderr(module_id).await;
9044
9045 release_dead_registration(registry, forwarding, snapshot, module_id).await
9046}
9047
9048#[cfg(unix)]
9068fn request_graceful_stop(module_id: &str, child: &SupervisedChild) {
9069 let Some(pid) = child
9070 .id()
9071 .and_then(|pid| i32::try_from(pid).ok())
9072 .and_then(rustix::process::Pid::from_raw)
9073 else {
9074 debug!(
9075 module_id,
9076 "no pid to signal for teardown; falling through to the drain wait"
9077 );
9078 return;
9079 };
9080 match rustix::process::kill_process(pid, rustix::process::Signal::TERM) {
9081 Ok(()) => debug!(
9082 module_id,
9083 "sent SIGTERM to a module nothing else asked to stop"
9084 ),
9085 Err(err) => debug!(
9086 module_id,
9087 error = %err,
9088 "SIGTERM to module failed; the drain wait and kill still apply"
9089 ),
9090 }
9091}
9092
9093#[cfg(not(unix))]
9101fn request_graceful_stop(module_id: &str, _child: &SupervisedChild) {
9102 debug!(
9103 module_id,
9104 "no graceful stop signal exists on this platform; teardown of a module nothing asked to stop waits, then kills"
9105 );
9106}
9107
9108fn terminal_disposition(final_state: ModuleState) -> TerminalDisposition {
9109 match final_state {
9110 ModuleState::Stopped => TerminalDisposition::Stopped,
9111 ModuleState::Disabled => TerminalDisposition::Disabled,
9112 ModuleState::Restarting => TerminalDisposition::Restarting,
9113 ModuleState::Failed => TerminalDisposition::Failed,
9114 ModuleState::Starting
9115 | ModuleState::Running
9116 | ModuleState::Unresponsive
9117 | ModuleState::Draining => {
9118 unreachable!("terminal exits only finish in terminal or restarting states")
9119 }
9120 }
9121}
9122
9123async fn release_dead_registration(
9131 registry: &Registry,
9132 forwarding: Option<&ForwardingTable>,
9133 snapshot: &SharedSnapshot,
9134 module_id: &str,
9135) -> Result<(), SuperviseError> {
9136 let result = async {
9137 let registration = registry
9138 .get_module(module_id)
9139 .map_err(SuperviseError::Registry)?;
9140 match wait_for_registration_release(registry, module_id, REGISTRY_RELEASE_TIMEOUT).await {
9141 Ok(()) => return Ok(()),
9142 Err(SuperviseError::RegistrationStillActive { .. }) => {}
9143 Err(err) => return Err(err),
9144 }
9145 let pid = lock_snapshot(snapshot)?.reaped_pid;
9146 if let (Some(registration), Some(forwarding), Some(pid)) = (registration, forwarding, pid) {
9147 warn!(
9148 module_id,
9149 pid,
9150 connection_id = registration.connection_id.get(),
9151 "reaped module registration outlived release grace; closing dead connection"
9152 );
9153 forwarding.request_connection_close(
9154 registration.connection_id,
9155 CloseReason::new(
9156 "supervised_process_reaped",
9157 format!("module '{module_id}' pid {pid} exited"),
9158 ),
9159 );
9160 wait_for_slot_registration_release(
9161 registry,
9162 crate::registry::RegistrationSlot::Connection(registration.connection_id),
9163 REGISTRY_RELEASE_TIMEOUT,
9164 )
9165 .await?;
9166 }
9167 wait_for_registration_release(registry, module_id, Duration::ZERO).await
9168 }
9169 .await;
9170 if let Err(err) = &result {
9171 fail_snapshot(snapshot, Some(module_id), None);
9172 error!(module_id, error = %err, "registration release failed after child exit; module is failed and start can retry");
9173 }
9174 result
9175}
9176
9177async fn wait_for_registration_release(
9180 registry: &Registry,
9181 module_id: &str,
9182 wait: Duration,
9183) -> Result<(), SuperviseError> {
9184 wait_for_slot_registration_release(
9185 registry,
9186 crate::registry::RegistrationSlot::Active(module_id),
9187 wait,
9188 )
9189 .await
9190}
9191
9192async fn wait_for_slot_registration_release(
9200 registry: &Registry,
9201 slot: crate::registry::RegistrationSlot<'_>,
9202 wait: Duration,
9203) -> Result<(), SuperviseError> {
9204 let deadline = Instant::now() + wait;
9205 let mut release_events = registration_release_events().subscribe();
9206 let still_active = |registration: &crate::registry::ModuleRegistration| {
9207 SuperviseError::RegistrationStillActive {
9208 module_id: registration.manifest.module_id.clone(),
9209 waited: wait,
9210 }
9211 };
9212 loop {
9213 let _observed_generation = *release_events.borrow_and_update();
9214 let Some(registration) = registry
9215 .registration(slot)
9216 .map_err(SuperviseError::Registry)?
9217 else {
9218 return Ok(());
9219 };
9220
9221 let now = Instant::now();
9222 if now >= deadline {
9223 return Err(still_active(®istration));
9224 }
9225
9226 let remaining = deadline.saturating_duration_since(now);
9227 match timeout(remaining, release_events.changed()).await {
9228 Ok(Ok(())) | Ok(Err(_)) => {}
9229 Err(_) => return Err(still_active(®istration)),
9230 }
9231 }
9232}
9233
9234#[cfg(test)]
9235mod slot_registration_wait_tests {
9236 use super::*;
9237 use crate::registry::{ConnectionId, RegistrationSlot};
9238 use subc_protocol::manifest::ModuleManifest;
9239
9240 #[tokio::test]
9241 async fn enable_release_failure_is_failed_and_a_second_enable_retries() {
9242 let registry = Arc::new(Registry::default());
9243 let supervisor = Supervisor::new_for_test(Arc::clone(®istry), RestartPolicy::default());
9244 let runtime = supervisor.runtime_config();
9245 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
9246 let spec = ModuleSpec {
9247 module_id: "enable-stale-registration".to_string(),
9248 program: PathBuf::from("/missing/enable-retry-test"),
9249 args: Vec::new(),
9250 env: Vec::new(),
9251 reserved: false,
9252 reserved_prefixes: Vec::new(),
9253 protocol: ModuleProtocol::Subc,
9254 overlap: Default::default(),
9255 };
9256 let connection = ConnectionId::new(90);
9257 registry
9258 .register_with_control_ops(
9259 ModuleManifest::builder(&spec.module_id, "0.1.0").build(),
9260 1,
9261 connection,
9262 Vec::new(),
9263 )
9264 .unwrap();
9265 let mut child = None;
9266 let err = set_child_enabled(
9267 &spec,
9268 &runtime,
9269 ®istry,
9270 &supervisor.process_liveness,
9271 &snapshot,
9272 &mut child,
9273 true,
9274 )
9275 .await
9276 .unwrap_err();
9277 assert!(matches!(
9278 err,
9279 SuperviseError::RegistrationStillActive { .. }
9280 ));
9281 assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
9282 assert!(child.is_none());
9283 registry.deregister_connection(connection).unwrap();
9284 let err = set_child_enabled(
9285 &spec,
9286 &runtime,
9287 ®istry,
9288 &supervisor.process_liveness,
9289 &snapshot,
9290 &mut child,
9291 true,
9292 )
9293 .await
9294 .unwrap_err();
9295 assert!(
9296 matches!(err, SuperviseError::Spawn { .. }),
9297 "second enable must attempt a spawn: {err}"
9298 );
9299 assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
9300 }
9301
9302 const INCUMBENT: u64 = 1;
9303 const CANDIDATE: u64 = 2;
9304
9305 fn swapped_registry() -> Arc<Registry> {
9306 let registry = Arc::new(Registry::default());
9307 let manifest = ModuleManifest::builder("m", "0.1.0").build();
9308 registry
9309 .register_with_control_ops(
9310 manifest.clone(),
9311 1,
9312 ConnectionId::new(INCUMBENT),
9313 Vec::new(),
9314 )
9315 .unwrap();
9316 registry
9317 .register_candidate_with_control_ops(
9318 manifest,
9319 1,
9320 ConnectionId::new(CANDIDATE),
9321 Vec::new(),
9322 )
9323 .unwrap();
9324 registry
9325 }
9326
9327 #[tokio::test]
9331 async fn incumbent_release_is_awaited_by_connection_not_by_module_id() {
9332 let registry = swapped_registry();
9333 registry.promote_candidate("m").unwrap().unwrap();
9334
9335 assert!(matches!(
9336 wait_for_registration_release(®istry, "m", Duration::from_millis(50)).await,
9337 Err(SuperviseError::RegistrationStillActive { .. })
9338 ));
9339
9340 assert!(matches!(
9342 wait_for_slot_registration_release(
9343 ®istry,
9344 RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
9345 Duration::from_millis(50),
9346 )
9347 .await,
9348 Err(SuperviseError::RegistrationStillActive { .. })
9349 ));
9350
9351 let releaser = Arc::clone(®istry);
9352 let release = tokio::spawn(async move {
9353 sleep(Duration::from_millis(20)).await;
9354 releaser
9355 .deregister_connection(ConnectionId::new(INCUMBENT))
9356 .unwrap();
9357 notify_registration_release();
9358 });
9359 wait_for_slot_registration_release(
9360 ®istry,
9361 RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
9362 Duration::from_secs(5),
9363 )
9364 .await
9365 .expect("the incumbent's own registration is released");
9366 release.await.unwrap();
9367 assert!(registry.get_module("m").unwrap().is_some());
9368 }
9369
9370 #[tokio::test]
9373 async fn candidate_slot_wait_ignores_the_incumbents_registration() {
9374 let registry = swapped_registry();
9375 assert!(matches!(
9376 wait_for_slot_registration_release(
9377 ®istry,
9378 RegistrationSlot::Candidate("m"),
9379 Duration::from_millis(50),
9380 )
9381 .await,
9382 Err(SuperviseError::RegistrationStillActive { .. })
9383 ));
9384 registry
9385 .deregister_connection(ConnectionId::new(CANDIDATE))
9386 .unwrap();
9387 wait_for_slot_registration_release(
9388 ®istry,
9389 RegistrationSlot::Candidate("m"),
9390 Duration::from_millis(50),
9391 )
9392 .await
9393 .expect("a candidate slot with no candidate is released");
9394 assert!(registry
9395 .registration(RegistrationSlot::Active("m"))
9396 .unwrap()
9397 .is_some());
9398 }
9399}
9400
9401fn classify_exit(status: &ExitStatus) -> ExitReport {
9402 ExitReport {
9403 kind: if status.success() {
9404 ExitKind::Clean
9405 } else {
9406 ExitKind::Crash
9407 },
9408 code: status.code(),
9409 signal: exit_signal(status),
9410 at_ms: unix_ms_now(),
9411 }
9412}
9413
9414fn wait_error_exit_report() -> ExitReport {
9420 ExitReport {
9421 kind: ExitKind::Crash,
9422 code: None,
9423 signal: None,
9424 at_ms: unix_ms_now(),
9425 }
9426}
9427
9428#[cfg(unix)]
9429fn exit_signal(status: &ExitStatus) -> Option<i32> {
9430 use std::os::unix::process::ExitStatusExt;
9431
9432 status.signal()
9433}
9434
9435#[cfg(not(unix))]
9436fn exit_signal(_status: &ExitStatus) -> Option<i32> {
9437 None
9438}
9439
9440fn reset_restart_count(snapshot: &SharedSnapshot, module_id: &str) -> Result<(), SuperviseError> {
9446 update_snapshot(snapshot, Some(module_id), |state| {
9447 state.clear_crash_restarts();
9448 })
9449}
9450
9451fn set_running(
9452 snapshot: &SharedSnapshot,
9453 child: &SupervisedChild,
9454 module_id: &str,
9455 spawn_events: &SpawnEventFeed,
9456) -> Result<(), SuperviseError> {
9457 let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
9458 module_id: Some(module_id.to_string()),
9459 })?;
9460 state.spawn_generation = spawn_events.emit_spawned(module_id, child.pid, child.spawned_at_ms);
9461 if std::mem::take(&mut state.coalesced_restart_pending) {
9462 let generation = state.spawn_generation;
9463 info!(module_id, "restart already satisfied by generation {generation}; coalesced pending request completed");
9464 }
9465 state.drain_disposition_detail = None;
9466 state.spawn_failure = None;
9467 state.in_alternate_slot = false;
9470 state.configuration_updated_since_spawn = false;
9471 state.spawned_protocol = Some(child.protocol);
9472 state.state = ModuleState::Running;
9473 state.enabled = true;
9474 state.process_alive = true;
9475 state.pid = child.id();
9476 #[cfg(target_os = "macos")]
9477 {
9478 state.report_ready = Some(Arc::clone(&child.report_ready));
9479 }
9480 state.spawned_at_ms = Some(child.spawned_at_ms);
9481 state.spawned_from = Some(child.spawned_from.clone());
9482 state.spawned_file_identity = child.spawned_file_identity;
9483 state.process_start_time = child.process_start_time;
9484 Ok(())
9485}
9486
9487fn clear_current_process_facts(state: &mut SupervisorSnapshot) {
9488 state.process_alive = false;
9489 state.spawned_protocol = None;
9490 state.pid = None;
9491 #[cfg(target_os = "macos")]
9492 {
9493 state.report_ready = None;
9494 }
9495 state.spawned_at_ms = None;
9496 state.spawned_from = None;
9497 state.spawned_file_identity = None;
9498 state.process_start_time = None;
9499 state.deliberate_severance = None;
9500}
9501
9502#[cfg(test)]
9503fn record_deliberate_severance(
9504 snapshot: &SharedSnapshot,
9505 identity: ProcessIdentity,
9506) -> Result<(), SuperviseError> {
9507 update_snapshot(snapshot, None, |state| {
9508 state.deliberate_severance = Some(identity);
9509 })
9510}
9511
9512fn apply_deliberate_severance_marker(
9513 snapshot: &SharedSnapshot,
9514 exited_identity: Option<ProcessIdentity>,
9515 mut exit_report: ExitReport,
9516) -> ExitReport {
9517 let marker = lock_snapshot(snapshot)
9518 .ok()
9519 .and_then(|mut state| state.deliberate_severance.take());
9520 if marker.is_some() && marker == exited_identity {
9521 exit_report.kind = ExitKind::DeliberateSeverance;
9522 }
9523 exit_report
9524}
9525
9526fn classify_reaped_child_exit(
9527 snapshot: &SharedSnapshot,
9528 child: &SupervisedChild,
9529 status: &ExitStatus,
9530) -> ExitReport {
9531 let _ = update_snapshot(snapshot, None, |state| {
9532 state.reaped_pid = Some(child.pid);
9533 state.spawn_failure = child.spawn_failure.clone();
9534 });
9535 apply_deliberate_severance_marker(snapshot, child.process_identity(), classify_exit(status))
9536}
9537
9538fn fail_snapshot(
9539 snapshot: &SharedSnapshot,
9540 module_id: Option<&str>,
9541 last_exit: Option<ExitReport>,
9542) {
9543 if let Err(err) = update_snapshot(snapshot, module_id, |state| {
9544 state.state = ModuleState::Failed;
9545 clear_current_process_facts(state);
9546 if let Some(last_exit) = last_exit {
9547 state.last_exit = Some(last_exit);
9548 }
9549 }) {
9550 error!(error = %err, "failed to mark supervisor state failed");
9551 }
9552}
9553
9554fn update_snapshot(
9555 snapshot: &SharedSnapshot,
9556 module_id: Option<&str>,
9557 update: impl FnOnce(&mut SupervisorSnapshot),
9558) -> Result<(), SuperviseError> {
9559 let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
9560 module_id: module_id.map(ToOwned::to_owned),
9561 })?;
9562 update(&mut state);
9563 Ok(())
9564}
9565
9566const SLOW_SNAPSHOT_LOCK_THRESHOLD: Duration = Duration::from_millis(250);
9567
9568fn lock_snapshot_for_control<'a>(
9569 snapshot: &'a SharedSnapshot,
9570 module_id: &str,
9571 caller: &'static str,
9572) -> Result<std::sync::MutexGuard<'a, SupervisorSnapshot>, SuperviseError> {
9573 let started_at = Instant::now();
9574 let guard = lock_snapshot(snapshot)?;
9575 let waited = started_at.elapsed();
9576 if waited >= SLOW_SNAPSHOT_LOCK_THRESHOLD {
9577 warn!(
9578 module_id = %module_id,
9579 waited_ms = waited.as_millis() as u64,
9580 caller = %caller,
9581 "slow snapshot lock"
9582 );
9583 }
9584 Ok(guard)
9585}
9586
9587fn lock_snapshot(
9588 snapshot: &SharedSnapshot,
9589) -> Result<std::sync::MutexGuard<'_, SupervisorSnapshot>, SuperviseError> {
9590 snapshot
9591 .lock()
9592 .map_err(|_| SuperviseError::StatePoisoned { module_id: None })
9593}
9594
9595#[cfg(test)]
9596mod terminal_history_tests {
9597 use std::{
9598 path::PathBuf,
9599 sync::Arc,
9600 time::{Duration, Instant},
9601 };
9602
9603 use tokio::time::sleep;
9604
9605 use super::{
9606 apply_deliberate_severance_marker, daemon_will_restart, drain_child_to_state,
9607 drained_after_quiescence_wait, handle_reload_spawn_failure, health_restart_child,
9608 lock_snapshot, on_child_exit, record_deliberate_severance, record_wait_error_terminal,
9609 reset_restart_count, spawn_and_mark_running, update_snapshot, wait_error_exit_report,
9610 ExitKind, ExitReport, ModuleProtocol, ModuleSpec, ModuleState, NextAction, ProcessIdentity,
9611 RestartPolicy, SpawnEventKind, StopNotice, SuperviseError, SupervisedModule, Supervisor,
9612 SupervisorHandle, SupervisorHealthStatus, SupervisorSnapshot,
9613 };
9614 use super::Instant as ClockInstant;
9619 use crate::{
9620 registry::Registry,
9621 terminal_ring::{TerminalRing, TerminalRingConfig},
9622 };
9623 use std::sync::Mutex;
9624 use subc_control::TerminalDisposition;
9625
9626 pub(super) fn fake_aft_stub_path() -> PathBuf {
9631 let mut path = std::env::current_exe().expect("current_exe available in tests");
9632 path.pop();
9633 path.pop();
9634 path.push(if cfg!(windows) {
9635 "fake-aft-stub.exe"
9636 } else {
9637 "fake-aft-stub"
9638 });
9639 assert!(
9640 path.exists(),
9641 "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
9642 [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
9643 path.display()
9644 );
9645 path
9646 }
9647
9648 #[test]
9649 fn reserved_never_spawned_refuses_every_hello() {
9650 let supervisor = SupervisorHandle::default();
9655 supervisor.apply_identity_configuration(&ModuleSpec {
9656 module_id: "never-spawned".to_string(),
9657 program: PathBuf::from("/usr/bin/false"),
9658 args: Vec::new(),
9659 env: Vec::new(),
9660 reserved: true,
9661 reserved_prefixes: Vec::new(),
9662 protocol: ModuleProtocol::Subc,
9663 overlap: Default::default(),
9664 });
9665 assert!(
9666 supervisor
9667 .reserved_hello_rejection("never-spawned", Some("any-forged-nonce"))
9668 .is_some(),
9669 "forged nonce must refuse on a reserved never-spawned id"
9670 );
9671 assert!(
9672 supervisor
9673 .reserved_hello_rejection("never-spawned", None)
9674 .is_some(),
9675 "absent nonce must refuse on a reserved never-spawned id"
9676 );
9677 supervisor.set_spawn_nonce("never-spawned", "minted".to_string());
9679 supervisor.apply_identity_configuration(&ModuleSpec {
9680 module_id: "never-spawned".to_string(),
9681 program: PathBuf::from("/usr/bin/false"),
9682 args: Vec::new(),
9683 env: Vec::new(),
9684 reserved: true,
9685 reserved_prefixes: Vec::new(),
9686 protocol: ModuleProtocol::Subc,
9687 overlap: Default::default(),
9688 });
9689 assert!(supervisor
9690 .reserved_hello_rejection("never-spawned", Some("minted"))
9691 .is_none());
9692 assert!(supervisor
9693 .reserved_hello_rejection("never-spawned", Some("forged"))
9694 .is_some());
9695 }
9696
9697 fn seed_crash_restarts(state: &mut SupervisorSnapshot, count: u32) {
9700 let now = ClockInstant::now();
9701 for _ in 0..count {
9702 state.crash_restarts.push_back(now);
9703 }
9704 }
9705
9706 fn age_oldest_crash_restart_out_of_window(state: &mut SupervisorSnapshot, window: Duration) {
9710 let aged = state
9711 .crash_restarts
9712 .front()
9713 .expect("a crash restart must be recorded before it can be aged")
9714 .checked_sub(window + Duration::from_secs(1))
9715 .expect("the test clock is far enough from its origin to age an instant");
9716 state.crash_restarts[0] = aged;
9717 }
9718
9719 fn snapshot_with_restarts(enabled: bool, count: u32) -> SupervisorSnapshot {
9720 let mut state = SupervisorSnapshot::new(ModuleState::Running, enabled);
9721 seed_crash_restarts(&mut state, count);
9722 state
9723 }
9724
9725 #[test]
9726 fn daemon_owned_recovery_predicate_uses_the_pre_increment_budget() {
9727 let policy = RestartPolicy::new(3, Duration::ZERO);
9728 let now = ClockInstant::now();
9729 assert!(daemon_will_restart(
9730 &mut snapshot_with_restarts(true, 2),
9731 &policy,
9732 now
9733 ));
9734 assert!(!daemon_will_restart(
9735 &mut snapshot_with_restarts(true, 3),
9736 &policy,
9737 now
9738 ));
9739 assert!(!daemon_will_restart(
9740 &mut snapshot_with_restarts(false, 0),
9741 &policy,
9742 now
9743 ));
9744 }
9745
9746 #[test]
9747 fn crash_restart_backoff_escalates_with_in_window_count() {
9748 let policy = RestartPolicy::new(4, Duration::from_millis(100))
9749 .with_max_backoff(Duration::from_secs(30));
9750 let now = ClockInstant::now();
9751 let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
9752 let schedules = (0..4)
9753 .map(|_| {
9754 state
9755 .next_crash_restart(&policy, now)
9756 .expect("the test policy allows four crash restarts")
9757 })
9758 .collect::<Vec<_>>();
9759
9760 assert_eq!(
9761 schedules
9762 .iter()
9763 .map(|schedule| schedule.restart_in_window)
9764 .collect::<Vec<_>>(),
9765 vec![0, 1, 2, 3]
9766 );
9767 assert_eq!(
9768 schedules
9769 .iter()
9770 .map(|schedule| schedule.delay)
9771 .collect::<Vec<_>>(),
9772 vec![
9773 Duration::from_millis(100),
9774 Duration::from_secs(1),
9775 Duration::from_secs(10),
9776 Duration::from_secs(30),
9777 ]
9778 );
9779 }
9780
9781 #[test]
9782 fn crash_restart_backoff_resets_after_ring_clear() {
9783 let policy = RestartPolicy::new(3, Duration::from_millis(100));
9784 let now = ClockInstant::now();
9785 let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
9786 assert_eq!(
9787 state.next_crash_restart(&policy, now).unwrap().delay,
9788 Duration::from_millis(100)
9789 );
9790 assert_eq!(
9791 state.next_crash_restart(&policy, now).unwrap().delay,
9792 Duration::from_secs(1)
9793 );
9794
9795 state.clear_crash_restarts();
9796 let schedule = state
9797 .next_crash_restart(&policy, now)
9798 .expect("a cleared ring must allow another restart");
9799 assert_eq!(schedule.restart_in_window, 0);
9800 assert_eq!(schedule.delay, Duration::from_millis(100));
9801 }
9802
9803 #[test]
9804 fn crash_restart_backoff_ignores_aged_restarts() {
9805 let policy = RestartPolicy::new(3, Duration::from_millis(100));
9806 let now = ClockInstant::now();
9807 let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
9808 state
9809 .next_crash_restart(&policy, now)
9810 .expect("the first restart is allowed");
9811 state
9812 .next_crash_restart(&policy, now)
9813 .expect("the second restart is allowed");
9814 state.crash_restarts[0] = now
9815 .checked_sub(policy.window + Duration::from_secs(1))
9816 .expect("the fake clock can age a restart past the window");
9817
9818 let schedule = state
9819 .next_crash_restart(&policy, now)
9820 .expect("an aged restart must release its slot");
9821 assert_eq!(schedule.restart_in_window, 1);
9822 assert_eq!(schedule.delay, Duration::from_secs(1));
9823 assert_eq!(state.crash_restarts.len(), 2);
9824 }
9825
9826 #[test]
9830 fn a_budget_spent_before_the_window_no_longer_refuses() {
9831 let policy = RestartPolicy::new(3, Duration::ZERO);
9832 let mut state = snapshot_with_restarts(true, 3);
9833 let now = ClockInstant::now();
9834 assert!(!daemon_will_restart(&mut state, &policy, now));
9835
9836 assert!(daemon_will_restart(
9837 &mut state,
9838 &policy,
9839 now + policy.window + Duration::from_secs(1)
9840 ));
9841 assert!(
9842 state.crash_restarts.is_empty(),
9843 "reading the budget must drop the instants that left the window"
9844 );
9845 }
9846
9847 fn module_with_recovery_snapshot(
9848 state: ModuleState,
9849 enabled: bool,
9850 restart_count: u32,
9851 ) -> SupervisedModule {
9852 let registry = Arc::new(Registry::default());
9853 let supervisor =
9854 Supervisor::new_for_test(Arc::clone(®istry), RestartPolicy::new(3, Duration::ZERO));
9855 let module = supervisor
9856 .spawn(ModuleSpec {
9857 module_id: "recovery-snapshot".to_string(),
9858 program: fake_aft_stub_path(),
9859 args: Vec::new(),
9860 env: Vec::new(),
9861 reserved: false,
9862 reserved_prefixes: Vec::new(),
9863 protocol: ModuleProtocol::Subc,
9864 overlap: Default::default(),
9865 })
9866 .unwrap();
9867 update_snapshot(
9868 &module.inner.snapshot,
9869 Some("recovery-snapshot"),
9870 |snapshot| {
9871 snapshot.state = state;
9872 snapshot.enabled = enabled;
9873 seed_crash_restarts(snapshot, restart_count);
9874 },
9875 )
9876 .unwrap();
9877 module
9878 }
9879
9880 #[cfg(target_os = "linux")]
9881 #[tokio::test]
9882 async fn no_cgroup_placement_does_not_block_fake_aft_stub_spawn() {
9883 let supervisor =
9884 Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
9885 .with_cgroup_placement(None);
9886 let result = supervisor.spawn(ModuleSpec {
9887 module_id: "no-cgroup-placement".to_string(),
9888 program: fake_aft_stub_path(),
9889 args: Vec::new(),
9890 env: Vec::new(),
9891 reserved: false,
9892 reserved_prefixes: Vec::new(),
9893 protocol: ModuleProtocol::Subc,
9894 overlap: Default::default(),
9895 });
9896
9897 assert!(
9898 result.is_ok(),
9899 "no delegation must not turn an otherwise valid spawn into a failure: {result:?}"
9900 );
9901 }
9902
9903 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9904 async fn undecided_snapshot_uses_shared_restart_predicate() {
9905 assert!(module_with_recovery_snapshot(ModuleState::Running, true, 2)
9906 .will_recover_after_connection_loss()
9907 .unwrap());
9908 assert!(
9909 !module_with_recovery_snapshot(ModuleState::Running, true, 3)
9910 .will_recover_after_connection_loss()
9911 .unwrap()
9912 );
9913 }
9914
9915 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9916 async fn restarting_snapshot_at_exhausted_budget_is_non_terminal() {
9917 assert!(
9918 module_with_recovery_snapshot(ModuleState::Restarting, true, 3)
9919 .will_recover_after_connection_loss()
9920 .unwrap()
9921 );
9922 }
9923
9924 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9925 async fn terminal_phase_snapshots_are_terminal_before_budget_exhaustion() {
9926 assert!(!module_with_recovery_snapshot(ModuleState::Failed, true, 0)
9927 .will_recover_after_connection_loss()
9928 .unwrap());
9929 assert!(
9930 !module_with_recovery_snapshot(ModuleState::Disabled, true, 0)
9931 .will_recover_after_connection_loss()
9932 .unwrap()
9933 );
9934 }
9935
9936 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9937 async fn warming_snapshot_is_limited_to_startup_phases() {
9938 for state in [
9939 ModuleState::Starting,
9940 ModuleState::Running,
9941 ModuleState::Restarting,
9942 ] {
9943 assert!(
9944 module_with_recovery_snapshot(state, true, 0)
9945 .is_warming()
9946 .unwrap(),
9947 "{state:?} should be warming"
9948 );
9949 }
9950 for state in [
9951 ModuleState::Unresponsive,
9952 ModuleState::Draining,
9953 ModuleState::Stopped,
9954 ModuleState::Failed,
9955 ModuleState::Disabled,
9956 ] {
9957 assert!(
9958 !module_with_recovery_snapshot(state, true, 0)
9959 .is_warming()
9960 .unwrap(),
9961 "{state:?} should not be warming"
9962 );
9963 }
9964 }
9965
9966 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9967 async fn terminal_history_survives_respawn_and_keeps_both_crashes_in_order() {
9968 let registry = Arc::new(Registry::default());
9969 let supervisor =
9970 Supervisor::new_for_test(Arc::clone(®istry), RestartPolicy::new(1, Duration::ZERO));
9971 let module = supervisor
9972 .spawn(ModuleSpec {
9973 module_id: "terminal-history".to_string(),
9974 program: fake_aft_stub_path(),
9975 args: Vec::new(),
9976 env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
9977 reserved: false,
9978 reserved_prefixes: Vec::new(),
9979 protocol: ModuleProtocol::Subc,
9980 overlap: Default::default(),
9981 })
9982 .unwrap();
9983
9984 let deadline = Instant::now() + Duration::from_secs(5);
9985 loop {
9986 let history = module.terminal_history();
9987 if history.entries.len() == 2 {
9988 assert_eq!(module.status().unwrap().state, ModuleState::Failed);
9989 assert_eq!(history.dropped, 0);
9990 assert_eq!(
9991 history
9992 .entries
9993 .iter()
9994 .map(|entry| entry.exit_code)
9995 .collect::<Vec<_>>(),
9996 vec![Some(23), Some(23)]
9997 );
9998 assert!(history.entries[0].at_ms <= history.entries[1].at_ms);
9999 return;
10000 }
10001 assert!(
10002 Instant::now() < deadline,
10003 "module did not retain two terminal exits: {history:?}"
10004 );
10005 sleep(Duration::from_millis(10)).await;
10006 }
10007 }
10008
10009 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10013 async fn disable_during_crash_backoff_cancels_pending_respawn() {
10014 let backoff = Duration::from_secs(2);
10015 let supervisor = Supervisor::new_for_test(
10016 Arc::new(Registry::default()),
10017 RestartPolicy::new(10, backoff),
10018 );
10019 let module = supervisor
10020 .spawn(ModuleSpec {
10021 module_id: "disable-during-backoff".to_string(),
10022 program: fake_aft_stub_path(),
10023 args: Vec::new(),
10024 env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
10025 reserved: false,
10026 reserved_prefixes: Vec::new(),
10027 protocol: ModuleProtocol::Subc,
10028 overlap: Default::default(),
10029 })
10030 .unwrap();
10031
10032 let deadline = Instant::now() + Duration::from_secs(5);
10034 loop {
10035 if module.status().unwrap().state == ModuleState::Restarting {
10036 break;
10037 }
10038 assert!(
10039 Instant::now() < deadline,
10040 "module never entered the crash backoff"
10041 );
10042 sleep(Duration::from_millis(10)).await;
10043 }
10044
10045 let started = Instant::now();
10046 module.set_enabled(false).await.unwrap();
10047 let waited = started.elapsed();
10048
10049 assert!(
10050 waited < backoff / 2,
10051 "disable waited {waited:?} behind the {backoff:?} crash backoff; the operator command must preempt the pending respawn"
10052 );
10053 assert_eq!(module.status().unwrap().state, ModuleState::Disabled);
10054
10055 sleep(backoff + Duration::from_millis(500)).await;
10057 let status = module.status().unwrap();
10058 assert_eq!(status.state, ModuleState::Disabled);
10059 assert_eq!(
10060 status.spawn_generation, 1,
10061 "module respawned after the operator disabled it"
10062 );
10063 }
10064
10065 #[cfg(unix)]
10068 fn protocol_none_sigterm_exits_clean_spec(
10069 module_id: &str,
10070 dir: &std::path::Path,
10071 ) -> (ModuleSpec, PathBuf, PathBuf) {
10072 let ready = dir.join("ready");
10073 let marker = dir.join("sigterm");
10074 let spec = ModuleSpec {
10075 module_id: module_id.to_string(),
10076 program: fake_aft_stub_path(),
10077 args: Vec::new(),
10078 env: vec![
10079 ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
10080 (
10081 "FAKE_AFT_SIGTERM_MARKER_PATH".to_string(),
10082 marker.display().to_string(),
10083 ),
10084 (
10085 "FAKE_AFT_NEVER_CONNECT_READY_PATH".to_string(),
10086 ready.display().to_string(),
10087 ),
10088 ],
10089 reserved: false,
10090 reserved_prefixes: Vec::new(),
10091 protocol: ModuleProtocol::None,
10092 overlap: Default::default(),
10093 };
10094 (spec, ready, marker)
10095 }
10096
10097 #[cfg(unix)]
10101 async fn wait_for_file(path: &std::path::Path) {
10102 let deadline = Instant::now() + Duration::from_secs(10);
10103 while !path.exists() {
10104 assert!(
10105 Instant::now() < deadline,
10106 "{} never appeared",
10107 path.display()
10108 );
10109 sleep(Duration::from_millis(10)).await;
10110 }
10111 }
10112
10113 #[cfg(unix)]
10117 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10118 async fn protocol_none_unrequested_clean_exit_restarts_as_a_crash() {
10119 let dir = subc_test_support::TestTempDir::new("none-unrequested-clean-exit");
10120 let (spec, ready, marker) =
10121 protocol_none_sigterm_exits_clean_spec("none-unrequested-clean-exit", dir.path());
10122 let supervisor = Supervisor::new_for_test(
10123 Arc::new(Registry::default()),
10124 RestartPolicy::new(3, Duration::ZERO),
10125 );
10126 let module = supervisor.spawn(spec).unwrap();
10127 wait_for_file(&ready).await;
10128 let first_pid = module
10129 .status()
10130 .unwrap()
10131 .pid
10132 .expect("a running module reports its pid");
10133
10134 rustix::process::kill_process(
10135 rustix::process::Pid::from_raw(i32::try_from(first_pid).unwrap()).unwrap(),
10136 rustix::process::Signal::TERM,
10137 )
10138 .unwrap();
10139
10140 let deadline = Instant::now() + Duration::from_secs(10);
10141 let respawned = loop {
10142 let status = module.status().unwrap();
10143 if status.state == ModuleState::Running
10144 && status.pid.is_some_and(|pid| pid != first_pid)
10145 {
10146 break status;
10147 }
10148 assert!(
10149 Instant::now() < deadline,
10150 "protocol-none module was not respawned after an unrequested clean exit: {status:?}"
10151 );
10152 sleep(Duration::from_millis(10)).await;
10153 };
10154 assert_eq!(respawned.spawn_generation, 2);
10155 assert!(
10156 marker.exists(),
10157 "the child must have exited through its SIGTERM handler (exit 0), or this proves nothing about clean exits"
10158 );
10159
10160 let history = module.terminal_history();
10161 assert_eq!(history.entries.len(), 1, "{history:?}");
10162 let entry = &history.entries[0];
10163 assert_eq!(entry.exit_code, Some(0));
10164 assert_eq!(entry.exit_kind, subc_control::TerminalExitKind::Clean);
10165 assert_eq!(entry.disposition, TerminalDisposition::Restarting);
10166
10167 module.stop().await.unwrap();
10168 }
10169
10170 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10174 async fn protocol_none_repeated_clean_exits_exhaust_the_restart_budget() {
10175 let supervisor = Supervisor::new_for_test(
10176 Arc::new(Registry::default()),
10177 RestartPolicy::new(1, Duration::ZERO),
10178 );
10179 let module = supervisor
10180 .spawn(ModuleSpec {
10181 module_id: "none-clean-exit-budget".to_string(),
10182 program: fake_aft_stub_path(),
10183 args: Vec::new(),
10184 env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
10185 reserved: false,
10186 reserved_prefixes: Vec::new(),
10187 protocol: ModuleProtocol::None,
10188 overlap: Default::default(),
10189 })
10190 .unwrap();
10191
10192 let deadline = Instant::now() + Duration::from_secs(10);
10193 loop {
10194 let status = module.status().unwrap();
10195 if status.state == ModuleState::Failed {
10196 break;
10197 }
10198 assert!(
10199 Instant::now() < deadline,
10200 "module never exhausted its budget: {status:?} {:?}",
10201 module.terminal_history()
10202 );
10203 sleep(Duration::from_millis(10)).await;
10204 }
10205 let history = module.terminal_history();
10206 assert_eq!(
10207 history
10208 .entries
10209 .iter()
10210 .map(|entry| (entry.exit_code, entry.disposition.clone()))
10211 .collect::<Vec<_>>(),
10212 vec![
10213 (Some(0), TerminalDisposition::Restarting),
10214 (Some(0), TerminalDisposition::Failed),
10215 ]
10216 );
10217 let detail = history.entries[1]
10218 .disposition_detail
10219 .as_deref()
10220 .expect("a budget failure names the budget");
10221 assert!(detail.contains("max_restarts=1"), "{detail}");
10222 assert_eq!(module.status().unwrap().spawn_generation, 2);
10223 }
10224
10225 #[cfg(unix)]
10228 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10229 async fn protocol_none_requested_stop_and_disable_do_not_respawn() {
10230 for disable in [false, true] {
10231 let label = if disable {
10232 "none-requested-disable"
10233 } else {
10234 "none-requested-stop"
10235 };
10236 let dir = subc_test_support::TestTempDir::new(label);
10237 let (spec, ready, marker) = protocol_none_sigterm_exits_clean_spec(label, dir.path());
10238 let supervisor = Supervisor::new_for_test(
10239 Arc::new(Registry::default()),
10240 RestartPolicy::new(3, Duration::ZERO),
10241 );
10242 let module = supervisor.spawn(spec).unwrap();
10243 wait_for_file(&ready).await;
10244
10245 if disable {
10246 module.set_enabled(false).await.unwrap();
10247 } else {
10248 module.stop().await.unwrap();
10249 }
10250 assert!(
10251 marker.exists(),
10252 "{label}: the child must have left through its SIGTERM handler with exit 0"
10253 );
10254
10255 sleep(Duration::from_millis(500)).await;
10258 let status = module.status().unwrap();
10259 let expected = if disable {
10260 ModuleState::Disabled
10261 } else {
10262 ModuleState::Stopped
10263 };
10264 assert_eq!(status.state, expected, "{label}");
10265 assert_eq!(
10266 status.spawn_generation, 1,
10267 "{label}: respawned after a requested stop"
10268 );
10269 let history = module.terminal_history();
10270 assert_eq!(history.entries.len(), 1, "{label}: {history:?}");
10271 assert_eq!(history.entries[0].exit_code, Some(0), "{label}");
10272 assert_ne!(
10273 history.entries[0].disposition,
10274 TerminalDisposition::Restarting,
10275 "{label}"
10276 );
10277 }
10278 }
10279
10280 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10283 async fn subc_wire_clean_exit_is_still_a_stop() {
10284 let supervisor = Supervisor::new_for_test(
10285 Arc::new(Registry::default()),
10286 RestartPolicy::new(3, Duration::ZERO),
10287 );
10288 let module = supervisor
10289 .spawn(ModuleSpec {
10290 module_id: "wire-clean-exit".to_string(),
10291 program: fake_aft_stub_path(),
10292 args: Vec::new(),
10293 env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
10294 reserved: false,
10295 reserved_prefixes: Vec::new(),
10296 protocol: ModuleProtocol::Subc,
10297 overlap: Default::default(),
10298 })
10299 .unwrap();
10300
10301 let deadline = Instant::now() + Duration::from_secs(10);
10302 while module.terminal_history().entries.is_empty() {
10303 assert!(Instant::now() < deadline, "module never exited");
10304 sleep(Duration::from_millis(10)).await;
10305 }
10306 sleep(Duration::from_millis(500)).await;
10308 let status = module.status().unwrap();
10309 assert_eq!(status.state, ModuleState::Stopped);
10310 assert_eq!(status.spawn_generation, 1);
10311 let history = module.terminal_history();
10312 assert_eq!(history.entries.len(), 1, "{history:?}");
10313 assert_eq!(history.entries[0].exit_code, Some(0));
10314 assert_eq!(history.entries[0].disposition, TerminalDisposition::Stopped);
10315 }
10316
10317 #[cfg(unix)]
10318 #[tokio::test]
10319 async fn reload_exit_keeps_roster_until_terminal_is_recorded_during_shutdown() {
10320 let dir = subc_test_support::TestTempDir::new("reload-roster-terminal-order");
10321 let record = dir.join("live-children.json");
10322 let supervisor = Supervisor::new_for_test(
10323 Arc::new(Registry::default()),
10324 RestartPolicy::new(0, Duration::ZERO),
10325 );
10326 let mut runtime = supervisor.runtime_config();
10327 runtime.child_roster.record_to(record.clone());
10328 let gate = Arc::new(super::ReloadExitRecordGate::default());
10329 runtime.test_reload_exit_record_gate = Some(Arc::clone(&gate));
10330 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10331 let spec = ModuleSpec {
10332 module_id: "reload-exit-roster".into(),
10333 program: fake_aft_stub_path(),
10334 args: Vec::new(),
10335 env: vec![("FAKE_AFT_EXIT_CODE".into(), "23".into())],
10336 reserved: false,
10337 reserved_prefixes: Vec::new(),
10338 protocol: ModuleProtocol::Subc,
10339 overlap: Default::default(),
10340 };
10341 let mut child = None;
10342 let reload = super::finish_reload_child(
10343 &spec,
10344 &runtime,
10345 &supervisor.registry,
10346 &supervisor.process_liveness,
10347 &snapshot,
10348 &mut child,
10349 );
10350 tokio::pin!(reload);
10351 tokio::select! {
10352 result = &mut reload => panic!("reload missed the before-record gate: {result:?}"),
10353 _ = gate.reached.notified() => {}
10354 }
10355 assert!(runtime
10356 .terminal_ring
10357 .lock()
10358 .unwrap()
10359 .snapshot()
10360 .entries
10361 .is_empty());
10362 assert_eq!(
10363 crate::live_children::read_record(&record).unwrap().len(),
10364 1,
10365 "shutdown must still wait for the reaped child until its terminal record exists"
10366 );
10367 runtime.child_roster.close();
10368 gate.resume.notify_one();
10369 assert!(reload.await.is_err());
10370 assert!(crate::live_children::read_record(&record)
10371 .unwrap()
10372 .is_empty());
10373 let history = runtime.terminal_ring.lock().unwrap().snapshot();
10374 assert_eq!(history.entries.len(), 1);
10375 assert_eq!(
10376 history.entries[0].disposition,
10377 TerminalDisposition::DaemonShutdown
10378 );
10379 }
10380
10381 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10385 async fn every_restart_increment_path_advances_lifetime_count() {
10386 let supervisor = Supervisor::new_for_test(
10387 Arc::new(Registry::default()),
10388 RestartPolicy::new(1, Duration::ZERO),
10389 );
10390 let runtime = supervisor.runtime_config();
10391 let spec = ModuleSpec {
10392 module_id: "lifetime-increment-path".to_string(),
10393 program: PathBuf::from("/unused/lifetime-increment-path"),
10394 args: Vec::new(),
10395 env: Vec::new(),
10396 reserved: false,
10397 reserved_prefixes: Vec::new(),
10398 protocol: ModuleProtocol::Subc,
10399 overlap: Default::default(),
10400 };
10401
10402 let crash_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10403 assert!(matches!(
10404 on_child_exit(
10405 &spec,
10406 runtime.restart_policy,
10407 &supervisor.registry,
10408 &crash_snapshot,
10409 &runtime.terminal_ring,
10410 &runtime.spawn_events,
10411 &runtime.child_roster,
10412 ExitReport {
10413 kind: ExitKind::Crash,
10414 code: Some(1),
10415 signal: None,
10416 at_ms: 1,
10417 },
10418 )
10419 .await,
10420 NextAction::Restart { schedule: _ }
10421 ));
10422 let (crash_restarts, crash_lifetime) = {
10423 let state = lock_snapshot(&crash_snapshot).unwrap();
10424 (state.crash_restarts.len(), state.lifetime_restarts)
10425 };
10426 assert_eq!(crash_restarts, 1);
10427 assert_eq!(crash_lifetime, 1);
10428
10429 let health_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10430 let mut health_child = None;
10431 assert!(matches!(
10432 health_restart_child(
10433 &spec,
10434 &runtime,
10435 &supervisor.registry,
10436 &supervisor.process_liveness,
10437 &health_snapshot,
10438 &mut health_child,
10439 SupervisorHealthStatus::Failing,
10440 None,
10441 2,
10442 )
10443 .await,
10444 Ok(())
10445 ));
10446 assert!(health_child.is_none());
10447 assert!(lock_snapshot(&health_snapshot).unwrap().respawn_pending);
10448 assert!(runtime.scheduled_respawn.lock().unwrap().take().is_some());
10449 let (health_restarts, health_lifetime) = {
10450 let state = lock_snapshot(&health_snapshot).unwrap();
10451 (state.crash_restarts.len(), state.lifetime_restarts)
10452 };
10453 assert_eq!(health_restarts, 1);
10454 assert_eq!(health_lifetime, 1);
10455
10456 let reload_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10457 let mut reload_child = None;
10458 assert!(matches!(
10459 handle_reload_spawn_failure(
10460 &spec,
10461 &runtime,
10462 &supervisor.process_liveness,
10463 &reload_snapshot,
10464 &mut reload_child,
10465 "forced reload spawn failure".to_string(),
10466 )
10467 .await,
10468 Err(SuperviseError::ReloadFailed { .. })
10469 ));
10470 let (reload_restarts, reload_lifetime) = {
10471 let state = lock_snapshot(&reload_snapshot).unwrap();
10472 (state.crash_restarts.len(), state.lifetime_restarts)
10473 };
10474 assert_eq!(reload_restarts, 1);
10475 assert_eq!(reload_lifetime, 1);
10476 }
10477
10478 #[tokio::test]
10479 async fn deliberately_severed_live_child_records_lifetime_without_spending_restart_budget() {
10480 let supervisor = Supervisor::new_for_test(
10481 Arc::new(Registry::default()),
10482 RestartPolicy::new(3, Duration::ZERO),
10483 );
10484 let runtime = supervisor.runtime_config();
10485 let spec = ModuleSpec {
10486 module_id: "deliberately-severed".to_string(),
10487 program: PathBuf::from("/unused/deliberately-severed"),
10488 args: Vec::new(),
10489 env: Vec::new(),
10490 reserved: false,
10491 reserved_prefixes: Vec::new(),
10492 protocol: ModuleProtocol::Subc,
10493 overlap: Default::default(),
10494 };
10495 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10496 let process = ProcessIdentity {
10497 pid: 41,
10498 start_time: 101,
10499 };
10500 record_deliberate_severance(&snapshot, process).unwrap();
10501 let exit_report = apply_deliberate_severance_marker(
10502 &snapshot,
10503 Some(process),
10504 ExitReport {
10505 kind: ExitKind::Crash,
10506 code: Some(1),
10507 signal: None,
10508 at_ms: 1,
10509 },
10510 );
10511 assert_eq!(exit_report.kind, ExitKind::DeliberateSeverance);
10512
10513 assert!(matches!(
10514 on_child_exit(
10515 &spec,
10516 runtime.restart_policy,
10517 &supervisor.registry,
10518 &snapshot,
10519 &runtime.terminal_ring,
10520 &runtime.spawn_events,
10521 &runtime.child_roster,
10522 exit_report,
10523 )
10524 .await,
10525 NextAction::Restart { schedule: _ }
10526 ));
10527 let state = lock_snapshot(&snapshot).unwrap();
10528 assert_eq!(state.lifetime_restarts, 1);
10529 assert_eq!(state.crash_restarts.len(), 0);
10530 }
10531
10532 #[tokio::test]
10533 async fn genuine_crash_spends_restart_budget_and_records_lifetime() {
10534 let supervisor = Supervisor::new_for_test(
10535 Arc::new(Registry::default()),
10536 RestartPolicy::new(3, Duration::ZERO),
10537 );
10538 let runtime = supervisor.runtime_config();
10539 let spec = ModuleSpec {
10540 module_id: "genuine-crash".to_string(),
10541 program: PathBuf::from("/unused/genuine-crash"),
10542 args: Vec::new(),
10543 env: Vec::new(),
10544 reserved: false,
10545 reserved_prefixes: Vec::new(),
10546 protocol: ModuleProtocol::Subc,
10547 overlap: Default::default(),
10548 };
10549 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10550
10551 assert!(matches!(
10552 on_child_exit(
10553 &spec,
10554 runtime.restart_policy,
10555 &supervisor.registry,
10556 &snapshot,
10557 &runtime.terminal_ring,
10558 &runtime.spawn_events,
10559 &runtime.child_roster,
10560 ExitReport {
10561 kind: ExitKind::Crash,
10562 code: Some(1),
10563 signal: None,
10564 at_ms: 1,
10565 },
10566 )
10567 .await,
10568 NextAction::Restart { schedule: _ }
10569 ));
10570 let state = lock_snapshot(&snapshot).unwrap();
10571 assert_eq!(state.lifetime_restarts, 1);
10572 assert_eq!(state.crash_restarts.len(), 1);
10573 }
10574
10575 fn crash_exit_report(at_ms: u64) -> ExitReport {
10576 ExitReport {
10577 kind: ExitKind::Crash,
10578 code: Some(1),
10579 signal: None,
10580 at_ms,
10581 }
10582 }
10583
10584 fn windowed_crash_spec(module_id: &str) -> ModuleSpec {
10585 ModuleSpec {
10586 module_id: module_id.to_string(),
10587 program: PathBuf::from("/unused").join(module_id),
10588 args: Vec::new(),
10589 env: Vec::new(),
10590 reserved: false,
10591 reserved_prefixes: Vec::new(),
10592 protocol: ModuleProtocol::Subc,
10593 overlap: Default::default(),
10594 }
10595 }
10596
10597 #[tokio::test]
10603 async fn three_crashes_inside_the_window_stop_the_module_and_name_the_window() {
10604 let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::ERROR);
10605 let supervisor = Supervisor::new_for_test(
10606 Arc::new(Registry::default()),
10607 RestartPolicy::new(2, Duration::ZERO),
10608 );
10609 let runtime = supervisor.runtime_config();
10610 let spec = windowed_crash_spec("crash-loop-in-window");
10611 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10612
10613 for attempt in 1..=2 {
10614 assert!(
10615 matches!(
10616 on_child_exit(
10617 &spec,
10618 runtime.restart_policy,
10619 &supervisor.registry,
10620 &snapshot,
10621 &runtime.terminal_ring,
10622 &runtime.spawn_events,
10623 &runtime.child_roster,
10624 crash_exit_report(attempt),
10625 )
10626 .await,
10627 NextAction::Restart { schedule: _ }
10628 ),
10629 "crash {attempt} is inside the budget and must respawn"
10630 );
10631 }
10632
10633 assert!(matches!(
10634 on_child_exit(
10635 &spec,
10636 runtime.restart_policy,
10637 &supervisor.registry,
10638 &snapshot,
10639 &runtime.terminal_ring,
10640 &runtime.spawn_events,
10641 &runtime.child_roster,
10642 crash_exit_report(3),
10643 )
10644 .await,
10645 NextAction::Stop { .. }
10646 ));
10647
10648 {
10649 let state = lock_snapshot(&snapshot).unwrap();
10650 assert_eq!(state.state, ModuleState::Failed);
10651 assert_eq!(state.crash_restarts.len(), 2);
10652 assert_eq!(state.lifetime_restarts, 2);
10653 }
10654
10655 let history = runtime
10656 .terminal_ring
10657 .lock()
10658 .expect("terminal ring is not poisoned")
10659 .snapshot();
10660 let last = history
10661 .entries
10662 .last()
10663 .expect("the refused crash is retained");
10664 assert_eq!(last.disposition, TerminalDisposition::Failed);
10665 assert_eq!(
10666 last.disposition_detail.as_deref(),
10667 Some("crash budget exhausted: max_restarts=2 within window_secs=600")
10668 );
10669
10670 let captured = crate::router::test_log::captured_logs(&logs);
10671 assert!(
10672 captured.contains("crash budget exhausted: max_restarts=2 within window_secs=600"),
10673 "the stop must be logged with its window: {captured}"
10674 );
10675 }
10676
10677 #[tokio::test]
10685 async fn a_crash_older_than_the_window_frees_its_slot_for_a_later_crash() {
10686 let supervisor = Supervisor::new_for_test(
10687 Arc::new(Registry::default()),
10688 RestartPolicy::new(2, Duration::ZERO),
10689 );
10690 let runtime = supervisor.runtime_config();
10691 let spec = windowed_crash_spec("crash-across-windows");
10692 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10693
10694 for attempt in 1..=2 {
10695 assert!(matches!(
10696 on_child_exit(
10697 &spec,
10698 runtime.restart_policy,
10699 &supervisor.registry,
10700 &snapshot,
10701 &runtime.terminal_ring,
10702 &runtime.spawn_events,
10703 &runtime.child_roster,
10704 crash_exit_report(attempt),
10705 )
10706 .await,
10707 NextAction::Restart { schedule: _ }
10708 ));
10709 }
10710
10711 update_snapshot(&snapshot, Some(&spec.module_id), |state| {
10714 age_oldest_crash_restart_out_of_window(state, runtime.restart_policy.window);
10715 })
10716 .unwrap();
10717
10718 assert!(
10719 matches!(
10720 on_child_exit(
10721 &spec,
10722 runtime.restart_policy,
10723 &supervisor.registry,
10724 &snapshot,
10725 &runtime.terminal_ring,
10726 &runtime.spawn_events,
10727 &runtime.child_roster,
10728 crash_exit_report(3),
10729 )
10730 .await,
10731 NextAction::Restart { schedule: _ }
10732 ),
10733 "a crash older than the window must not hold a budget slot"
10734 );
10735
10736 let state = lock_snapshot(&snapshot).unwrap();
10737 assert_eq!(state.state, ModuleState::Restarting);
10738 assert_eq!(
10739 state.crash_restarts.len(),
10740 2,
10741 "the aged instant is dropped and the new one takes its place"
10742 );
10743 assert_eq!(
10744 state.lifetime_restarts, 3,
10745 "the ledger counts every restart, including the ones the window forgot"
10746 );
10747 }
10748
10749 #[tokio::test]
10754 async fn an_operator_restart_clears_the_ring_and_leaves_the_ledger_alone() {
10755 let supervisor = Supervisor::new_for_test(
10756 Arc::new(Registry::default()),
10757 RestartPolicy::new(2, Duration::ZERO),
10758 );
10759 let runtime = supervisor.runtime_config();
10760 let spec = windowed_crash_spec("operator-cleared-budget");
10761 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10762
10763 for attempt in 1..=2 {
10764 assert!(matches!(
10765 on_child_exit(
10766 &spec,
10767 runtime.restart_policy,
10768 &supervisor.registry,
10769 &snapshot,
10770 &runtime.terminal_ring,
10771 &runtime.spawn_events,
10772 &runtime.child_roster,
10773 crash_exit_report(attempt),
10774 )
10775 .await,
10776 NextAction::Restart { schedule: _ }
10777 ));
10778 }
10779
10780 reset_restart_count(&snapshot, &spec.module_id).unwrap();
10781 {
10782 let state = lock_snapshot(&snapshot).unwrap();
10783 assert!(
10784 state.crash_restarts.is_empty(),
10785 "an operator restart returns the full budget"
10786 );
10787 assert_eq!(
10788 state.lifetime_restarts, 2,
10789 "clearing the budget must not unmake the crashes"
10790 );
10791 }
10792
10793 assert!(
10794 matches!(
10795 on_child_exit(
10796 &spec,
10797 runtime.restart_policy,
10798 &supervisor.registry,
10799 &snapshot,
10800 &runtime.terminal_ring,
10801 &runtime.spawn_events,
10802 &runtime.child_roster,
10803 crash_exit_report(3),
10804 )
10805 .await,
10806 NextAction::Restart { schedule: _ }
10807 ),
10808 "the cleared budget must be spendable again"
10809 );
10810 let state = lock_snapshot(&snapshot).unwrap();
10811 assert_eq!(state.crash_restarts.len(), 1);
10812 assert_eq!(state.lifetime_restarts, 3);
10813 }
10814
10815 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10816 async fn severance_marker_for_a_dead_child_does_not_label_its_successor() {
10817 let severed = ProcessIdentity {
10818 pid: 41,
10819 start_time: 101,
10820 };
10821 let successor = ProcessIdentity {
10822 pid: 41,
10823 start_time: 202,
10824 };
10825 let module = module_with_recovery_snapshot(ModuleState::Running, true, 0);
10826 update_snapshot(&module.inner.snapshot, Some("recovery-snapshot"), |state| {
10827 state.pid = Some(successor.pid);
10828 state.process_start_time = Some(successor.start_time);
10829 })
10830 .unwrap();
10831 assert!(!module.record_deliberate_severance(severed).unwrap());
10832
10833 let exit_report = apply_deliberate_severance_marker(
10834 &module.inner.snapshot,
10835 Some(successor),
10836 ExitReport {
10837 kind: ExitKind::Crash,
10838 code: Some(1),
10839 signal: None,
10840 at_ms: 1,
10841 },
10842 );
10843
10844 assert_eq!(exit_report.kind, ExitKind::Crash);
10845 }
10846
10847 #[tokio::test]
10848 async fn drain_reap_marks_deliberate_severance_and_records_lifetime_without_budget() {
10849 let registry = Registry::default();
10850 let supervisor = Supervisor::new_for_test(
10851 Arc::new(Registry::default()),
10852 RestartPolicy::new(3, Duration::ZERO),
10853 );
10854 let runtime = supervisor.runtime_config();
10855 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10856 let spec = ModuleSpec {
10857 module_id: "drain-deliberate-severance".to_string(),
10858 program: fake_aft_stub_path(),
10859 args: Vec::new(),
10860 env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
10861 reserved: false,
10862 reserved_prefixes: Vec::new(),
10863 protocol: ModuleProtocol::Subc,
10864 overlap: Default::default(),
10865 };
10866 let mut child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
10867 let process = ProcessIdentity {
10868 pid: 41,
10869 start_time: 101,
10870 };
10871 child.process_identity = Some(process);
10872 update_snapshot(&snapshot, Some(&spec.module_id), |state| {
10873 state.pid = Some(process.pid);
10874 state.process_start_time = Some(process.start_time);
10875 })
10876 .unwrap();
10877 record_deliberate_severance(&snapshot, process).unwrap();
10878
10879 drain_child_to_state(
10880 &spec.module_id,
10881 spec.protocol,
10882 StopNotice::SentOverConnection,
10885 ®istry,
10886 None,
10887 &snapshot,
10888 &runtime.terminal_ring,
10889 &runtime.spawn_events,
10890 child,
10891 Duration::from_secs(1),
10892 ModuleState::Stopped,
10893 Some(false),
10894 )
10895 .await
10896 .unwrap();
10897
10898 let state = lock_snapshot(&snapshot).unwrap();
10899 assert_eq!(
10900 state.last_exit.as_ref().map(|exit| exit.kind),
10901 Some(ExitKind::DeliberateSeverance)
10902 );
10903 assert_eq!(state.lifetime_restarts, 1);
10904 assert_eq!(state.crash_restarts.len(), 0);
10905 drop(state);
10906 let history = runtime.terminal_ring.lock().unwrap().snapshot();
10907 assert_eq!(
10908 history.entries[0].exit_kind,
10909 subc_control::TerminalExitKind::DeliberateSeverance
10910 );
10911 }
10912
10913 #[tokio::test]
10914 async fn ordinary_drain_reap_does_not_record_a_lifetime_restart() {
10915 let registry = Registry::default();
10916 let supervisor = Supervisor::new_for_test(
10917 Arc::new(Registry::default()),
10918 RestartPolicy::new(3, Duration::ZERO),
10919 );
10920 let runtime = supervisor.runtime_config();
10921 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10922 let spec = ModuleSpec {
10923 module_id: "ordinary-drain".to_string(),
10924 program: fake_aft_stub_path(),
10925 args: Vec::new(),
10926 env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
10927 reserved: false,
10928 reserved_prefixes: Vec::new(),
10929 protocol: ModuleProtocol::Subc,
10930 overlap: Default::default(),
10931 };
10932 let child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
10933
10934 drain_child_to_state(
10935 &spec.module_id,
10936 spec.protocol,
10937 StopNotice::SentOverConnection,
10940 ®istry,
10941 None,
10942 &snapshot,
10943 &runtime.terminal_ring,
10944 &runtime.spawn_events,
10945 child,
10946 Duration::from_secs(1),
10947 ModuleState::Stopped,
10948 Some(false),
10949 )
10950 .await
10951 .unwrap();
10952
10953 let state = lock_snapshot(&snapshot).unwrap();
10954 assert_eq!(
10955 state.last_exit.as_ref().map(|exit| exit.kind),
10956 Some(ExitKind::Crash)
10957 );
10958 assert_eq!(state.lifetime_restarts, 0);
10959 assert_eq!(state.crash_restarts.len(), 0);
10960 }
10961
10962 #[test]
10963 fn fatal_connection_teardown_cannot_arm_a_marker_for_a_surviving_process() {
10964 assert!(!include_str!("server.rs")
10970 .contains("router.record_deliberate_connection_severance(ctx.connection_id)"));
10971 }
10972
10973 #[test]
10980 fn drained_after_quiescence_wait_passes_ok_through_and_forces_false_on_err() {
10981 assert!(drained_after_quiescence_wait(&Ok(true)));
10982 assert!(!drained_after_quiescence_wait(&Ok(false)));
10983 assert!(!drained_after_quiescence_wait(&Err(
10984 SuperviseError::StatePoisoned { module_id: None }
10985 )));
10986 }
10987
10988 #[test]
10997 fn wait_error_exit_report_records_a_failed_terminal_with_no_code_or_signal() {
10998 let ring = Arc::new(Mutex::new(TerminalRing::new(
10999 TerminalRingConfig::default(),
11000 0,
11001 )));
11002 record_wait_error_terminal("wait-error", &ring, &super::SpawnEventFeed::default());
11003
11004 let snapshot = ring.lock().unwrap().snapshot();
11005 assert_eq!(snapshot.entries.len(), 1);
11006 let entry = &snapshot.entries[0];
11007 assert_eq!(entry.exit_code, None);
11008 assert_eq!(entry.exit_signal, None);
11009 assert_eq!(entry.disposition, TerminalDisposition::Failed);
11010 }
11011
11012 #[test]
11013 fn wait_error_exit_path_preserves_spawn_event_density() {
11014 let feed = super::SpawnEventFeed::default();
11015 feed.configure_incarnation("wait-error-density".to_string());
11016 feed.emit_spawned("wait-error", 41, 1);
11017 let ring = Arc::new(Mutex::new(TerminalRing::new(
11018 TerminalRingConfig::default(),
11019 0,
11020 )));
11021
11022 record_wait_error_terminal("wait-error", &ring, &feed);
11023 feed.emit_spawned("after-wait-error", 42, 2);
11024
11025 let state = feed.0.lock().unwrap();
11026 let sequences = state
11027 .events
11028 .iter()
11029 .map(|event| event.cursor.seq)
11030 .collect::<Vec<_>>();
11031 assert_eq!(sequences, vec![1, 2, 3]);
11032 assert_eq!(state.events[1].kind, SpawnEventKind::Exited);
11033 assert_eq!(state.events[1].exit_code, None);
11034 assert_eq!(state.events[1].exit_signal, None);
11035 }
11036
11037 #[test]
11041 fn wait_error_exit_report_is_classified_as_a_crash() {
11042 assert_eq!(wait_error_exit_report().kind, ExitKind::Crash);
11043 }
11044}
11045
11046#[cfg(test)]
11047mod health_evidence_tests {
11048 use super::{HealthProbeError, HealthProbeEvidence};
11049 use std::collections::HashSet;
11050
11051 #[test]
11059 fn only_a_dead_lane_is_proof_of_death() {
11060 assert!(HealthProbeError::lane_dead("gone").is_proof_of_death());
11061 assert!(!HealthProbeError::no_answer("timed out").is_proof_of_death());
11065 assert!(!HealthProbeError::bad_answer("garbage").is_proof_of_death());
11066 assert!(!HealthProbeError::misconfigured("no table").is_proof_of_death());
11067 }
11068
11069 #[test]
11075 fn every_evidence_class_has_a_distinct_label() {
11076 let labels = [
11077 HealthProbeError::lane_dead("").label(),
11078 HealthProbeError::no_answer("").label(),
11079 HealthProbeError::bad_answer("").label(),
11080 HealthProbeError::misconfigured("").label(),
11081 ];
11082 let unique: HashSet<_> = labels.iter().collect();
11083 assert_eq!(unique.len(), labels.len(), "labels collided: {labels:?}");
11084 }
11085
11086 #[test]
11092 fn classification_preserves_the_original_message() {
11093 let err = HealthProbeError::no_answer("module did not answer within 5s");
11094 assert_eq!(err.to_string(), "module did not answer within 5s");
11095 assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
11096 }
11097}
11098
11099#[cfg(test)]
11100mod health_tombstone_tests {
11101 use std::{path::PathBuf, sync::Arc, time::Duration};
11102
11103 use subc_protocol::{
11104 manifest::Concurrency,
11105 session::{HealthStatus, ModuleControlResponse},
11106 };
11107 use tokio::sync::mpsc;
11108
11109 use super::{
11110 probe_module_health, HealthAction, HealthConfig, HealthProbeEvidence, ModuleProtocol,
11111 ModuleSpec, RestartPolicy, Supervisor, SupervisorRuntimeConfig,
11112 };
11113 use crate::{
11114 control::ControlHandler,
11115 forwarding::{ForwardingTable, ModuleControlRpcCompletion, ModuleControlRpcOutcome},
11116 registry::{ConnectionId, Registry},
11117 router::FrameSink,
11118 };
11119
11120 struct ProbeHarness {
11121 spec: ModuleSpec,
11122 runtime: SupervisorRuntimeConfig,
11123 forwarding: Arc<ForwardingTable>,
11124 module_connection: ConnectionId,
11125 module_rx: mpsc::Receiver<crate::router::OutboundFrame>,
11126 handler: ControlHandler,
11127 module: super::SupervisedModule,
11128 }
11129
11130 fn probe_harness() -> ProbeHarness {
11131 let registry = Arc::new(Registry::default());
11132 let forwarding = Arc::new(ForwardingTable::default());
11133 let supervisor_handle = super::SupervisorHandle::new();
11134 let health = HealthConfig {
11135 http: None,
11136 cadence: Duration::from_secs(30),
11137 deadline: Duration::from_secs(5),
11138 failure_threshold: 3,
11139 on_degraded: HealthAction::Report,
11140 on_failing: HealthAction::Report,
11141 critical: false,
11142 };
11143 let supervisor = Supervisor::new_for_test(Arc::clone(®istry), RestartPolicy::default())
11144 .with_forwarding(Arc::clone(&forwarding))
11145 .with_handle(supervisor_handle.clone())
11146 .with_health_config(health);
11147 let spec = ModuleSpec {
11148 module_id: "late-health-module".to_string(),
11149 program: PathBuf::from("disabled-module"),
11150 args: Vec::new(),
11151 env: Vec::new(),
11152 reserved: false,
11153 reserved_prefixes: Vec::new(),
11154 protocol: ModuleProtocol::Subc,
11155 overlap: Default::default(),
11156 };
11157 let module = supervisor
11158 .supervise_configured(spec.clone(), false)
11159 .unwrap();
11160 let runtime = supervisor.runtime_config();
11161 let handler = ControlHandler::with_forwarding(registry, Arc::clone(&forwarding))
11162 .with_supervisor(supervisor_handle);
11163 let module_connection = ConnectionId::new(700);
11164 let (module_tx, module_rx) = mpsc::channel(8);
11165 forwarding
11166 .register_module_connection(
11167 module_connection,
11168 spec.module_id.clone(),
11169 subc_protocol::PROTOCOL_VERSION,
11170 Concurrency::ModuleManaged,
11171 FrameSink::new(module_tx),
11172 )
11173 .unwrap();
11174
11175 ProbeHarness {
11176 spec,
11177 runtime,
11178 forwarding,
11179 module_connection,
11180 module_rx,
11181 handler,
11182 module,
11183 }
11184 }
11185
11186 async fn finish_after(
11187 harness: &mut ProbeHarness,
11188 stall: Duration,
11189 ) -> ModuleControlRpcCompletion {
11190 assert!(stall > harness.runtime.health.deadline);
11191 let deadline = harness.runtime.health.deadline;
11192 let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
11193 let answer = async {
11194 let frame = harness.module_rx.recv().await.expect("health.check frame");
11195 tokio::time::advance(deadline).await;
11196 tokio::task::yield_now().await;
11197 tokio::time::advance(stall - deadline).await;
11198 harness
11199 .forwarding
11200 .complete_module_control_rpc(
11201 harness.module_connection,
11202 frame.header.corr,
11203 Some("health.check"),
11204 ModuleControlRpcOutcome::Response(ModuleControlResponse::HealthCheck {
11205 status: HealthStatus::Ok,
11206 detail: None,
11207 metrics: None,
11208 }),
11209 )
11210 .unwrap()
11211 };
11212 let (probe_result, completion) = tokio::join!(probe, answer);
11213 let err = probe_result.expect_err("probe must miss its deadline");
11214 assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
11215 completion
11216 }
11217
11218 async fn time_out_without_answer(harness: &mut ProbeHarness) {
11219 let deadline = harness.runtime.health.deadline;
11220 let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
11221 let exhaust_deadline = async {
11222 let _frame = harness.module_rx.recv().await.expect("health.check frame");
11223 tokio::time::advance(deadline).await;
11224 tokio::task::yield_now().await;
11225 };
11226 let (probe_result, ()) = tokio::join!(probe, exhaust_deadline);
11227 let err = probe_result.expect_err("probe must miss its deadline");
11228 assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
11229 }
11230
11231 async fn run_probe_cycle(harness: &mut ProbeHarness, answer: bool) {
11232 let registry = Arc::clone(&harness.module.inner.registry);
11233 let snapshot = Arc::clone(&harness.module.inner.snapshot);
11234 let process_liveness = super::SupervisorProcessLiveness::default();
11235 let mut child = None;
11236 let cycle = super::run_health_probe_cycle(
11237 &harness.spec,
11238 &harness.runtime,
11239 ®istry,
11240 &process_liveness,
11241 &snapshot,
11242 &mut child,
11243 );
11244 let peer = async {
11245 let frame = harness.module_rx.recv().await.expect("health.check frame");
11246 if answer {
11247 harness
11248 .forwarding
11249 .complete_module_control_rpc(
11250 harness.module_connection,
11251 frame.header.corr,
11252 Some("health.check"),
11253 ModuleControlRpcOutcome::Response(ModuleControlResponse::HealthCheck {
11254 status: HealthStatus::Ok,
11255 detail: None,
11256 metrics: Some(serde_json::json!({"ready": true})),
11257 }),
11258 )
11259 .unwrap();
11260 } else {
11261 tokio::time::advance(harness.runtime.health.deadline).await;
11262 tokio::task::yield_now().await;
11263 }
11264 };
11265 tokio::join!(cycle, peer);
11266 }
11267
11268 #[tokio::test(start_paused = true)]
11269 async fn unanswered_probe_is_unknown_until_threshold_and_ok_report_recovers() {
11270 let mut harness = probe_harness();
11271 let monitor = harness.module.inner.monitor.lock().unwrap().take().unwrap();
11275 monitor.abort();
11276 let _ = monitor.await;
11277 super::update_snapshot(&harness.module.inner.snapshot, None, |state| {
11278 state.enabled = true;
11279 state.state = super::ModuleState::Running;
11280 state.process_alive = true;
11281 })
11282 .unwrap();
11283
11284 run_probe_cycle(&mut harness, true).await;
11285 assert_eq!(
11286 harness.module.status().unwrap().health.status,
11287 super::SupervisorHealthStatus::Ok
11288 );
11289
11290 for failures in 1..harness.runtime.health.failure_threshold {
11291 run_probe_cycle(&mut harness, false).await;
11292 let status = harness.module.status().unwrap();
11293 assert_eq!(status.health.status, super::SupervisorHealthStatus::Unknown);
11294 assert_eq!(status.health.consecutive_failures, failures);
11295 assert!(status.health.last_probe_ms.is_some());
11296 assert!(status.health.detail.unwrap().starts_with("[no-answer]"));
11297 assert!(status.health.metrics.is_none());
11298 assert_eq!(status.state, super::ModuleState::Running);
11299 assert!(status.process_alive);
11300 assert_eq!(status.restart_count, 0);
11301 assert_eq!(status.lifetime_restarts, 0);
11302 assert!(status.health.last_action.is_none());
11303 assert!(harness.runtime.scheduled_respawn.lock().unwrap().is_none());
11304 }
11305
11306 run_probe_cycle(&mut harness, true).await;
11307 let recovered = harness.module.status().unwrap();
11308 assert_eq!(recovered.health.status, super::SupervisorHealthStatus::Ok);
11309 assert_eq!(recovered.health.consecutive_failures, 0);
11310 assert!(recovered.health.detail.is_none());
11311 assert_eq!(
11312 recovered.health.metrics,
11313 Some(serde_json::json!({"ready": true}))
11314 );
11315 assert_eq!(recovered.lifetime_restarts, 0);
11316
11317 for failures in 1..=harness.runtime.health.failure_threshold {
11318 run_probe_cycle(&mut harness, false).await;
11319 let status = harness.module.status().unwrap();
11320 assert_eq!(status.health.consecutive_failures, failures);
11321 if failures < harness.runtime.health.failure_threshold {
11322 assert_eq!(status.health.status, super::SupervisorHealthStatus::Unknown);
11323 assert_eq!(status.state, super::ModuleState::Running);
11324 assert_eq!(status.lifetime_restarts, 0);
11325 assert!(harness.runtime.scheduled_respawn.lock().unwrap().is_none());
11326 } else {
11327 assert_eq!(
11328 status.health.status,
11329 super::SupervisorHealthStatus::Unresponsive
11330 );
11331 assert_eq!(status.state, super::ModuleState::Restarting);
11332 assert_eq!(status.restart_count, 1);
11333 assert_eq!(status.lifetime_restarts, 1);
11334 assert_eq!(status.health.last_action.as_deref(), Some("restart"));
11335 assert!(harness.runtime.scheduled_respawn.lock().unwrap().is_some());
11336 }
11337 }
11338 }
11339
11340 #[tokio::test(start_paused = true)]
11341 async fn late_health_answers_record_start_anchored_latency_for_two_stalls() {
11342 let mut harness = probe_harness();
11343
11344 let first = finish_after(&mut harness, Duration::from_secs(8)).await;
11345 let first_latency = match &first {
11346 ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
11347 other => panic!("late answer was not retained: {other:?}"),
11348 };
11349 assert!(harness.handler.observe_module_control_completion(first));
11350
11351 let second = finish_after(&mut harness, Duration::from_secs(11)).await;
11352 let second_latency = match &second {
11353 ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
11354 other => panic!("late answer was not retained: {other:?}"),
11355 };
11356 assert!(harness.handler.observe_module_control_completion(second));
11357
11358 assert_eq!(first_latency, Duration::from_secs(8));
11359 assert_eq!(
11360 second_latency - first_latency,
11361 Duration::from_secs(3),
11362 "latency must grow linearly with the additional stall"
11363 );
11364 let health = harness.module.status().unwrap().health;
11365 assert_eq!(health.late_answer_count, 2);
11366 assert_eq!(health.last_late_answer_latency_ms, Some(11_000));
11367 }
11368
11369 #[tokio::test(start_paused = true)]
11377 async fn late_answer_clears_the_consecutive_failure_streak() {
11378 let mut harness = probe_harness();
11379
11380 time_out_without_answer(&mut harness).await;
11382 harness
11383 .module
11384 .record_health_probe_failure_for_test("[no-answer] test miss")
11385 .unwrap();
11386 assert_eq!(
11387 harness.module.status().unwrap().health.consecutive_failures,
11388 1,
11389 "precondition: the miss must be on the streak before the late answer"
11390 );
11391
11392 let late = finish_after(&mut harness, Duration::from_secs(9)).await;
11394 assert!(matches!(
11395 late,
11396 ModuleControlRpcCompletion::LateHealthAnswer { .. }
11397 ));
11398 assert!(harness.handler.observe_module_control_completion(late));
11399
11400 let health = harness.module.status().unwrap().health;
11401 assert_eq!(
11402 health.consecutive_failures, 0,
11403 "a late answer is an answer: the streak must reset"
11404 );
11405 assert_eq!(health.late_answer_count, 1);
11406 }
11407
11408 #[tokio::test(start_paused = true)]
11409 async fn repeated_serial_probe_cycles_keep_one_tombstone_per_endpoint() {
11410 let mut harness = probe_harness();
11411
11412 for _ in 0..20 {
11413 time_out_without_answer(&mut harness).await;
11414 assert_eq!(
11415 harness.forwarding.health_probe_tombstone_count().unwrap(),
11416 1
11417 );
11418 }
11419 }
11420}
11421
11422#[cfg(test)]
11423mod child_env_tests {
11424 use super::{
11425 apply_child_env, apply_spawn_role, apply_wire_spawn_args, ModuleProtocol, ModuleSpec,
11426 SpawnRole, SupervisorHandle, SPAWN_ROLE_SWAP_CANDIDATE, SUBC_ARG, SUBC_LAUNCH_NONCE_ENV,
11427 SUBC_MODULE_ID_ENV, SUBC_SPAWN_ROLE_ENV,
11428 };
11429 use std::{ffi::OsStr, path::PathBuf};
11430 use tokio::process::Command;
11431
11432 fn spec(env: Vec<(String, String)>) -> ModuleSpec {
11433 ModuleSpec {
11434 module_id: "env-plan".to_string(),
11435 program: PathBuf::from("/nonexistent"),
11436 args: Vec::new(),
11437 env,
11438 reserved: false,
11439 reserved_prefixes: Vec::new(),
11440 protocol: ModuleProtocol::Subc,
11441 overlap: Default::default(),
11442 }
11443 }
11444
11445 #[test]
11459 fn ambient_ck_log_is_removed_and_a_configured_one_survives() {
11460 let mut command = Command::new("/nonexistent");
11461 apply_child_env(&mut command, &spec(Vec::new()));
11462 let removed = command
11463 .as_std()
11464 .get_envs()
11465 .any(|(key, value)| key == OsStr::new("CK_LOG") && value.is_none());
11466 assert!(
11467 removed,
11468 "ambient CK_LOG must be explicitly removed for an unconfigured module"
11469 );
11470
11471 let mut configured = Command::new("/nonexistent");
11472 apply_child_env(
11473 &mut configured,
11474 &spec(vec![("CK_LOG".to_string(), "debug".to_string())]),
11475 );
11476 let effective = configured
11477 .as_std()
11478 .get_envs()
11479 .filter(|(key, _)| *key == OsStr::new("CK_LOG"))
11480 .last()
11481 .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
11482 assert_eq!(
11483 effective,
11484 Some(Some("debug".to_string())),
11485 "a module's configured CK_LOG must survive the ambient removal"
11486 );
11487 }
11488
11489 #[test]
11498 fn protocol_none_spawn_carries_no_subc_argument_and_no_nonce() {
11499 let connection_file = std::path::Path::new("/run/subc-connection.json");
11500 let handle = SupervisorHandle::new();
11501
11502 let mut none_spec = spec(Vec::new());
11503 none_spec.protocol = ModuleProtocol::None;
11504 let mut none = Command::new("/nonexistent");
11505 let none_handoff =
11506 apply_wire_spawn_args(&mut none, &none_spec, Some(connection_file), Some(&handle))
11507 .expect("protocol-none spawn args apply");
11508 assert!(
11509 none_handoff.is_none(),
11510 "protocol:none spawn must not receive a nonce descriptor"
11511 );
11512 assert!(
11513 !none.as_std().get_envs().any(|(key, value)| key
11514 == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV)
11515 && value.is_some()),
11516 "protocol:none spawn must not name a nonce descriptor"
11517 );
11518 let none_args: Vec<String> = none
11519 .as_std()
11520 .get_args()
11521 .map(|a| a.to_string_lossy().into_owned())
11522 .collect();
11523 assert!(
11524 !none_args.iter().any(|a| a == SUBC_ARG),
11525 "protocol:none argv must not carry --subc; got {none_args:?}"
11526 );
11527 let none_has_nonce = none
11528 .as_std()
11529 .get_envs()
11530 .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some());
11531 assert!(
11532 !none_has_nonce,
11533 "protocol:none spawn must not receive a launch nonce"
11534 );
11535 let none_has_module_id = none
11536 .as_std()
11537 .get_envs()
11538 .any(|(key, value)| key == OsStr::new(SUBC_MODULE_ID_ENV) && value.is_some());
11539 assert!(
11540 none_has_module_id,
11541 "SUBC_MODULE_ID is inert and stays on every path"
11542 );
11543 assert!(
11544 handle.spawn_nonce(&none_spec.module_id).is_none(),
11545 "no nonce record for a process that will never present one"
11546 );
11547
11548 let wire_spec = spec(Vec::new());
11550 let mut wire = Command::new("/nonexistent");
11551 let wire_handoff =
11552 apply_wire_spawn_args(&mut wire, &wire_spec, Some(connection_file), Some(&handle))
11553 .expect("subc-wire spawn args apply");
11554 let wire_fd_env = wire
11555 .as_std()
11556 .get_envs()
11557 .find(|(key, _)| *key == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV))
11558 .and_then(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
11559 #[cfg(unix)]
11560 assert_eq!(
11561 wire_fd_env,
11562 Some(wire_handoff.expect("a descriptor handoff").fd_env_value()),
11563 "a subc-wire spawn names the pipe it will receive at descriptor 3"
11564 );
11565 #[cfg(not(unix))]
11566 assert!(wire_handoff.is_none() && wire_fd_env.is_none());
11567 let wire_args: Vec<String> = wire
11568 .as_std()
11569 .get_args()
11570 .map(|a| a.to_string_lossy().into_owned())
11571 .collect();
11572 assert_eq!(
11573 wire_args,
11574 vec![
11575 SUBC_ARG.to_string(),
11576 connection_file.to_string_lossy().into_owned()
11577 ],
11578 "a subc-wire spawn still carries --subc <path>"
11579 );
11580 assert_eq!(
11581 wire.as_std()
11582 .get_envs()
11583 .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some()),
11584 !cfg!(unix),
11585 "only Windows supplies the environment nonce"
11586 );
11587 assert!(handle.spawn_nonce(&wire_spec.module_id).is_some());
11588 }
11589
11590 #[test]
11600 fn plain_spawn_removes_the_spawn_role_even_when_the_spec_sets_it() {
11601 let role = |command: &Command| {
11602 command
11603 .as_std()
11604 .get_envs()
11605 .filter(|(key, _)| *key == OsStr::new(SUBC_SPAWN_ROLE_ENV))
11606 .last()
11607 .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()))
11608 };
11609 let forged = spec(vec![(
11610 SUBC_SPAWN_ROLE_ENV.to_string(),
11611 SPAWN_ROLE_SWAP_CANDIDATE.to_string(),
11612 )]);
11613
11614 let mut plain = Command::new("/nonexistent");
11615 apply_child_env(&mut plain, &forged);
11616 apply_spawn_role(&mut plain, SpawnRole::Plain);
11617 assert_eq!(
11618 role(&plain),
11619 Some(None),
11620 "a plain spawn must remove SUBC_SPAWN_ROLE, whatever the spec says"
11621 );
11622
11623 let mut candidate = Command::new("/nonexistent");
11624 apply_child_env(&mut candidate, &spec(Vec::new()));
11625 apply_spawn_role(&mut candidate, SpawnRole::SwapCandidate);
11626 assert_eq!(
11627 role(&candidate),
11628 Some(Some(SPAWN_ROLE_SWAP_CANDIDATE.to_string()))
11629 );
11630 }
11631
11632 #[test]
11638 fn daemon_private_capture_keys_are_not_passed_to_the_child() {
11639 let mut command = Command::new("/nonexistent");
11640 apply_child_env(
11641 &mut command,
11642 &spec(vec![
11643 (super::CAPTURE_KEEP_ENV.to_string(), "5".to_string()),
11644 ("KEPT".to_string(), "yes".to_string()),
11645 ]),
11646 );
11647 let keys: Vec<String> = command
11648 .as_std()
11649 .get_envs()
11650 .filter(|(_, value)| value.is_some())
11651 .map(|(key, _)| key.to_string_lossy().into_owned())
11652 .collect();
11653 assert!(keys.contains(&"KEPT".to_string()), "got {keys:?}");
11654 assert!(
11655 !keys.contains(&super::CAPTURE_KEEP_ENV.to_string()),
11656 "daemon-private capture key leaked to the child: {keys:?}"
11657 );
11658 }
11659}
11660
11661#[cfg(test)]
11662mod jitter_tests {
11663 use super::jittered_health_delay;
11664 use std::{collections::HashSet, time::Duration};
11665
11666 const FLEET: [&str; 14] = [
11675 "aft",
11676 "alfonso-core",
11677 "magic-context",
11678 "broca",
11679 "thalamus",
11680 "quota",
11681 "engram",
11682 "plexus",
11683 "cerebellum",
11684 "astrocyte",
11685 "synapse",
11686 "subc-mcp",
11687 "cortexkit-credentials",
11688 "subc-federation",
11689 ];
11690
11691 #[test]
11699 fn probe_delays_disperse_across_the_fleet() {
11700 let cadence = Duration::from_secs(30);
11701 let delays: HashSet<Duration> = FLEET
11702 .iter()
11703 .map(|id| jittered_health_delay(id, 0, cadence))
11704 .collect();
11705 assert_eq!(
11706 delays.len(),
11707 FLEET.len(),
11708 "every supervised module must land on its own probe offset"
11709 );
11710 }
11711
11712 #[test]
11718 fn jitter_only_delays_and_stays_within_one_tenth_of_cadence() {
11719 let cadence = Duration::from_secs(30);
11720 let span = cadence / 10;
11721 for id in FLEET {
11722 for probe_index in 0..8 {
11723 let delay = jittered_health_delay(id, probe_index, cadence);
11724 assert!(
11725 delay >= cadence,
11726 "{id}#{probe_index}: jitter must not shorten the cadence"
11727 );
11728 assert!(
11729 delay < cadence + span,
11730 "{id}#{probe_index}: jitter must stay inside one tenth of the cadence"
11731 );
11732 }
11733 }
11734 }
11735
11736 #[test]
11742 fn a_module_offset_is_stable_across_restarts() {
11743 let cadence = Duration::from_secs(30);
11744 for id in FLEET {
11745 assert_eq!(
11746 jittered_health_delay(id, 0, cadence),
11747 jittered_health_delay(id, 0, cadence),
11748 "{id}: the same module and probe index must produce the same offset"
11749 );
11750 }
11751 }
11752
11753 #[test]
11755 fn zero_cadence_yields_zero_delay() {
11756 assert_eq!(
11757 jittered_health_delay("aft", 0, Duration::ZERO),
11758 Duration::ZERO
11759 );
11760 }
11761}
11762
11763#[cfg(all(test, target_os = "linux"))]
11764mod cgroup_placement_tests {
11765 use super::{
11766 apply_cgroup_placement, remove_module_cgroup, ModuleProtocol, ModuleSpec, SuperviseError,
11767 SupervisedChild,
11768 };
11769 use crate::stderr_tail::{StderrRing, StderrTailConfig};
11770 use std::{
11771 fs, io,
11772 path::{Path, PathBuf},
11773 sync::{Arc, Mutex},
11774 };
11775 use subc_test_support::TestTempDir;
11776 use tokio::process::Command;
11777
11778 #[tokio::test]
11779 async fn unique_spawn_cgroups_clean_up_across_restarts_and_shutdown() {
11780 use super::*;
11781 let dir = TestTempDir::new("unique-spawn-cgroups");
11782 let root = PathBuf::from(format!(
11783 "/sys/fs/cgroup/subc-unique-test-{}-{}",
11784 std::process::id(),
11785 unix_ms_now()
11786 ));
11787 if let Err(error) = fs::create_dir(&root) {
11788 assert!(
11789 std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
11790 "required cgroup test cannot execute: {error}"
11791 );
11792 eprintln!("SKIP unique_spawn_cgroups_clean_up_across_restarts_and_shutdown: {error}");
11793 return;
11794 }
11795 let placement = subc_cgroup::prepare_at(&root).unwrap().unwrap();
11796 let group_count = || {
11797 fs::read_dir(root.join("subc-modules"))
11798 .unwrap()
11799 .map(|entry| entry.unwrap().file_type().unwrap())
11800 .filter(|kind| kind.is_dir())
11801 .count()
11802 };
11803 let supervisor =
11804 Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
11805 .with_cgroup_placement(Some(placement.clone()));
11806 let runtime = supervisor.runtime_config();
11807 let mut spec = ModuleSpec {
11808 module_id: "unique-spawn".into(),
11809 program: PathBuf::from("/bin/sleep"),
11810 args: vec!["60".into()],
11811 env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
11812 .into_iter()
11813 .map(|key| (key.into(), dir.to_string_lossy().into_owned()))
11814 .collect(),
11815 reserved: false,
11816 reserved_prefixes: vec![],
11817 protocol: ModuleProtocol::None,
11818 overlap: Default::default(),
11819 };
11820 let spawn = |spec: &ModuleSpec| {
11821 spawn_child(
11822 spec,
11823 None,
11824 None,
11825 &runtime.stderr_ring,
11826 None,
11827 &runtime.child_roster,
11828 Some(&placement),
11829 )
11830 .unwrap()
11831 };
11832 let mut live = spawn(&spec);
11833 for _ in 0..3 {
11834 let next = spawn(&spec);
11836 assert_ne!(live.module_id, next.module_id);
11837 live.start_kill().unwrap();
11838 live.wait().await.unwrap();
11839 live = next;
11840 assert!(
11841 live.child.try_wait().unwrap().is_none(),
11842 "retiring the old slot must not kill the replacement"
11843 );
11844 assert_eq!(
11845 group_count(),
11846 1,
11847 "only the live spawn's cgroup should remain"
11848 );
11849 }
11850 supervisor.begin_daemon_shutdown();
11851 let reap = tokio::spawn(async move {
11852 live.wait().await.unwrap();
11853 });
11854 supervisor
11855 .end_children_for_daemon_shutdown(false, std::future::pending())
11856 .await;
11857 reap.await.unwrap();
11858 assert_eq!(group_count(), 0);
11859 spec.program = PathBuf::from("/bin/true");
11861 spec.args.clear();
11862 let fresh_roster = ChildRoster::default();
11863 let mut short = spawn_child(
11864 &spec,
11865 None,
11866 None,
11867 &runtime.stderr_ring,
11868 None,
11869 &fresh_roster,
11870 Some(&placement),
11871 )
11872 .unwrap();
11873 short.wait().await.unwrap();
11874 assert_eq!(group_count(), 0);
11875 spec.module_id = "_".repeat(255);
11876 let mut long_id = spawn_child(
11877 &spec,
11878 None,
11879 None,
11880 &runtime.stderr_ring,
11881 None,
11882 &fresh_roster,
11883 Some(&placement),
11884 )
11885 .unwrap();
11886 long_id.wait().await.unwrap();
11887 assert_eq!(
11888 group_count(),
11889 0,
11890 "valid long module IDs must not exceed cgroup NAME_MAX"
11891 );
11892 fs::remove_dir(root.join("subc-modules")).unwrap();
11893 fs::remove_dir(root).unwrap();
11894 eprintln!("EXECUTED unique_spawn_cgroups_clean_up_across_restarts_and_shutdown");
11895 }
11896
11897 #[test]
11898 fn failed_parent_cgroup_open_is_a_cgroup_supervision_error() {
11899 let path = Path::new("/definitely-missing-subc-cgroup");
11900 let mut command = Command::new("true");
11901 let error = apply_cgroup_placement(
11902 &mut command,
11903 &ModuleSpec {
11904 module_id: "broken-cgroup".to_string(),
11905 program: PathBuf::from("true"),
11906 args: Vec::new(),
11907 env: Vec::new(),
11908 reserved: false,
11909 reserved_prefixes: Vec::new(),
11910 protocol: ModuleProtocol::Subc,
11911 overlap: Default::default(),
11912 },
11913 path,
11914 )
11915 .expect_err("a parent cgroup open failure must reject the supervised spawn");
11916 let reason = error.to_string();
11917
11918 assert!(
11919 matches!(error, SuperviseError::Cgroup { .. }),
11920 "parent cgroup open must be reported as a cgroup supervision error: {reason}"
11921 );
11922 assert!(
11923 reason.contains("/definitely-missing-subc-cgroup/cgroup.procs"),
11924 "parent cgroup open failure must name cgroup.procs: {reason}"
11925 );
11926 }
11927
11928 #[tokio::test]
11929 async fn reaping_a_child_removes_its_empty_module_cgroup() {
11930 let root = TestTempDir::new("supervisor-reap-cgroup");
11931 fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
11932 let placement = subc_cgroup::prepare_at(&root)
11933 .expect("prepare scratch cgroup root")
11934 .expect("scratch root has a cgroup.procs marker");
11935 let module_id = "reaped-module";
11936 let module = placement
11937 .module_path(module_id)
11938 .expect("create scratch module cgroup");
11939 let child = Command::new("true")
11940 .env("XDG_DATA_HOME", root.path())
11941 .env("XDG_RUNTIME_DIR", root.path())
11942 .env("XDG_CONFIG_HOME", root.path())
11943 .spawn()
11944 .expect("spawn short-lived child");
11945 let pid = child.id().expect("spawned child has pid");
11946 let mut child = SupervisedChild {
11947 child,
11948 protocol: ModuleProtocol::Subc,
11949 module_id: module_id.to_string(),
11950 cgroup_placement: Some(placement),
11951 stdout_pump: None,
11952 stderr_pump: None,
11953 stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
11954 spawned_at_ms: 0,
11955 spawned_from: PathBuf::from("true"),
11956 spawned_file_identity: None,
11957 process_start_time: None,
11958 process_identity: None,
11959 pid,
11960 roster_guard: None,
11961 #[cfg(target_os = "macos")]
11962 privacy_exec: None,
11963 spawn_failure: None,
11964 };
11965
11966 child.wait().await.expect("reap short-lived child");
11967
11968 assert!(
11969 !module.exists(),
11970 "reaping the supervised child must remove its empty cgroup"
11971 );
11972 }
11973
11974 #[test]
11975 fn non_empty_cgroup_removal_is_reported_without_blocking_teardown() {
11976 let root = TestTempDir::new("supervisor-non-empty-cgroup");
11977 fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
11978 let placement = subc_cgroup::prepare_at(&root)
11979 .expect("prepare scratch cgroup root")
11980 .expect("scratch root has a cgroup.procs marker");
11981 let module = placement
11982 .module_path("surviving-module")
11983 .expect("create scratch module cgroup");
11984 fs::write(module.join("surviving-process"), b"still present")
11985 .expect("make scratch cgroup non-empty");
11986 let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::WARN);
11987
11988 remove_module_cgroup(&placement, "surviving-module");
11989
11990 let logs = crate::router::test_log::captured_logs(&logs);
11991 assert!(
11992 module.exists(),
11993 "failed removal must leave the cgroup intact"
11994 );
11995 assert!(
11996 logs.contains("could not remove module cgroup after process exit; continuing teardown")
11997 && logs.contains("surviving-module"),
11998 "best-effort removal must report the failure without returning it: {logs}"
11999 );
12000 }
12001
12002 #[test]
12003 fn cgroup_pre_exec_spawn_failure_names_the_cgroup_path() {
12004 let cgroup_path = PathBuf::from("/sys/fs/cgroup/subc-modules/broken-module");
12005 let reason = SuperviseError::Spawn {
12006 program: PathBuf::from("/bin/true"),
12007 source: io::Error::from_raw_os_error(13),
12008 cgroup_path: Some(cgroup_path.clone()),
12009 }
12010 .to_string();
12011
12012 assert!(
12013 reason.contains(&cgroup_path.display().to_string()),
12014 "a pre_exec spawn failure must name the cgroup path: {reason}"
12015 );
12016 }
12017}
12018
12019#[cfg(test)]
12020mod spawn_subscriber_lag_tests {
12021 use super::*;
12022
12023 #[tokio::test]
12028 async fn lagged_spawn_subscriber_receives_a_terminal_lagged_error_after_its_queued_frames() {
12029 let feed = SpawnEventFeed::default();
12030 feed.configure_incarnation("lag-incarnation".to_string());
12031 let (tx, mut rx) = mpsc::channel(1);
12034 feed.subscribe(ConnectionId::new(1), 7, 1, None, FrameSink::new(tx))
12035 .expect("subscribe");
12036 let emitted = SPAWN_SUBSCRIBER_BUFFER + 16;
12037 for index in 0..emitted {
12038 feed.emit_spawned(&format!("lag-module-{index}"), 1000, 0);
12039 tokio::task::yield_now().await;
12042 }
12043 assert_eq!(
12044 feed.subscriber_count(),
12045 0,
12046 "the lagged subscriber must be removed"
12047 );
12048
12049 let mut data = Vec::new();
12050 let mut last = None;
12051 loop {
12052 let next = tokio::time::timeout(Duration::from_secs(5), rx.recv())
12053 .await
12054 .expect("the forwarder must finish once the subscriber is dropped");
12055 let Some(outbound) = next else { break };
12056 let frame = outbound.frame;
12057 if frame.header.ty == FrameType::StreamData {
12058 assert!(last.is_none(), "no data may follow the terminal frame");
12059 let event: SpawnEvent = serde_json::from_slice(&frame.body).unwrap();
12060 data.push(event.cursor.seq);
12061 } else {
12062 assert!(last.is_none(), "exactly one terminal frame");
12063 last = Some(frame);
12064 }
12065 }
12066 assert!(!data.is_empty(), "queued frames drain before the terminal");
12067 for pair in data.windows(2) {
12068 assert_eq!(
12069 pair[1],
12070 pair[0] + 1,
12071 "queued frames arrive dense and in order"
12072 );
12073 }
12074 let terminal = last.expect("a lagged subscriber must receive a terminal frame");
12075 assert_eq!(terminal.header.ty, FrameType::Error);
12076 assert_eq!(terminal.header.corr, 7);
12077 let body: subc_protocol::ErrorBody = serde_json::from_slice(&terminal.body).unwrap();
12078 assert_eq!(body.code, SPAWN_SUBSCRIBER_LAGGED_CODE);
12079 let detail = body.detail.expect("lagged error carries detail");
12080 assert_eq!(
12081 detail["first_undelivered_cursor"]["seq"],
12082 data.last().unwrap() + 1,
12083 "the named cursor is the first event the subscriber did not receive"
12084 );
12085 assert_eq!(
12086 detail["first_undelivered_cursor"]["daemon_incarnation"],
12087 "lag-incarnation"
12088 );
12089 }
12090}
12091
12092#[cfg(test)]
12093mod terminal_history_read_concurrency_tests {
12094 use super::*;
12095 use crate::terminal_journal::read_pause;
12096 use std::sync::mpsc as std_mpsc;
12097 use subc_test_support::TestTempDir;
12098
12099 fn journaled_ring(
12100 journal: &Arc<crate::terminal_journal::TerminalJournal>,
12101 ) -> Arc<Mutex<TerminalRing>> {
12102 Arc::new(Mutex::new(
12103 TerminalRing::new(TerminalRingConfig::default(), 1)
12104 .with_journal(Some(Arc::clone(journal))),
12105 ))
12106 }
12107
12108 fn crash(at_ms: u64) -> ExitReport {
12109 ExitReport {
12110 kind: ExitKind::Crash,
12111 code: Some(1),
12112 signal: None,
12113 at_ms,
12114 }
12115 }
12116
12117 fn record_within(
12120 module_id: &'static str,
12121 ring: &Arc<Mutex<TerminalRing>>,
12122 at_ms: u64,
12123 bound: Duration,
12124 ) -> bool {
12125 let ring = Arc::clone(ring);
12126 let (done, done_rx) = std_mpsc::channel();
12127 std::thread::spawn(move || {
12128 record_terminal(
12129 module_id,
12130 &ring,
12131 &SpawnEventFeed::default(),
12132 &crash(at_ms),
12133 TerminalDisposition::Restarting,
12134 );
12135 let _ = done.send(());
12136 });
12137 done_rx.recv_timeout(bound).is_ok()
12138 }
12139
12140 #[test]
12145 fn exits_recorded_during_a_paused_history_read_are_not_blocked_or_half_merged() {
12146 let dir = TestTempDir::new("terminal-history-concurrent-read");
12147 let path = dir.join("terminals.jsonl");
12148 let journal = Arc::new(crate::terminal_journal::TerminalJournal::open(
12149 path.clone(),
12150 "daemon".into(),
12151 ));
12152 let reader_ring = journaled_ring(&journal);
12153 let other_ring = journaled_ring(&journal);
12154 assert!(record_within(
12155 "reader-module",
12156 &reader_ring,
12157 10,
12158 Duration::from_secs(5)
12159 ));
12160
12161 let (started, release) = read_pause::install(&path);
12162 let reading = {
12163 let ring = Arc::clone(&reader_ring);
12164 std::thread::spawn(move || durable_terminal_history_of(&ring, "reader-module"))
12165 };
12166 started
12167 .recv_timeout(Duration::from_secs(5))
12168 .expect("the history read reached its pause");
12169
12170 let bound = Duration::from_secs(1);
12171 assert!(
12172 record_within("other-module", &other_ring, 20, bound),
12173 "another module's exit waited on a history read (journal writer held)"
12174 );
12175 assert!(
12176 record_within("reader-module", &reader_ring, 30, bound),
12177 "the read module's own exit waited on its history read (ring held)"
12178 );
12179
12180 drop(release);
12181 let paused = reading.join().unwrap();
12182 assert_eq!(
12183 paused.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
12184 vec![10],
12185 "an exit recorded after the read began lands in neither half of it"
12186 );
12187 assert_eq!(paused.journal_skipped_lines, 0);
12188 assert_eq!(paused.journal_read_errors, 0);
12189
12190 let after = durable_terminal_history_of(&reader_ring, "reader-module");
12191 assert_eq!(
12192 after.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
12193 vec![10, 30],
12194 "the next read merges ring and journal with no duplicate"
12195 );
12196 assert_eq!(after.journal_skipped_lines, 0);
12197 }
12198}
12199
12200#[cfg(test)]
12205mod stderr_settle_tests {
12206 use std::{
12207 future::Future,
12208 io,
12209 pin::Pin,
12210 sync::{Arc, Mutex},
12211 task::{Context, Poll},
12212 time::Duration,
12213 };
12214
12215 use tokio::{
12216 io::{AsyncRead, ReadBuf},
12217 sync::oneshot,
12218 time::Instant,
12219 };
12220
12221 use super::{settle_stderr_pump, StderrPump};
12222 use crate::stderr_tail::{
12223 pump_stderr_to, untimed, CaptureState, OutputSink, StderrRing, StderrTailConfig, TailEntry,
12224 };
12225
12226 const BOUND: Duration = Duration::from_millis(250);
12227
12228 struct HeldReader {
12232 before: Option<Vec<u8>>,
12233 gate: Option<oneshot::Receiver<()>>,
12234 after: io::Cursor<Vec<u8>>,
12235 }
12236
12237 impl AsyncRead for HeldReader {
12238 fn poll_read(
12239 mut self: Pin<&mut Self>,
12240 cx: &mut Context<'_>,
12241 buf: &mut ReadBuf<'_>,
12242 ) -> Poll<io::Result<()>> {
12243 if let Some(bytes) = self.before.take() {
12244 buf.put_slice(&bytes);
12245 return Poll::Ready(Ok(()));
12246 }
12247 if let Some(gate) = self.gate.as_mut() {
12248 match Pin::new(gate).poll(cx) {
12249 Poll::Pending => return Poll::Pending,
12250 Poll::Ready(_) => self.gate = None,
12251 }
12252 }
12253 Pin::new(&mut self.after).poll_read(cx, buf)
12254 }
12255 }
12256
12257 struct DiscardSink;
12258
12259 impl OutputSink for DiscardSink {
12260 fn write_line(&mut self, _line: &[u8]) {}
12261 }
12262
12263 fn line(text: &str) -> TailEntry {
12264 TailEntry::Line {
12265 text: text.to_string(),
12266 truncated: false,
12267 at_ms: None,
12268 }
12269 }
12270
12271 fn lock(ring: &Arc<Mutex<StderrRing>>) -> std::sync::MutexGuard<'_, StderrRing> {
12272 ring.lock().unwrap()
12273 }
12274
12275 fn held_pump(
12279 ring: &Arc<Mutex<StderrRing>>,
12280 before: &str,
12281 after: &str,
12282 ) -> (StderrPump, oneshot::Sender<()>) {
12283 let generation = lock(ring).begin_process();
12284 let (release, gate) = oneshot::channel();
12285 let reader = HeldReader {
12286 before: Some(before.as_bytes().to_vec()),
12287 gate: Some(gate),
12288 after: io::Cursor::new(after.as_bytes().to_vec()),
12289 };
12290 let task = tokio::spawn(pump_stderr_to(
12291 reader,
12292 Arc::clone(ring),
12293 generation,
12294 DiscardSink,
12295 ));
12296 (StderrPump { task, generation }, release)
12297 }
12298
12299 async fn wait_until(ring: &Arc<Mutex<StderrRing>>, done: impl Fn(&StderrRing) -> bool) {
12300 for _ in 0..1000 {
12301 if done(&lock(ring)) {
12302 return;
12303 }
12304 tokio::time::sleep(Duration::from_millis(1)).await;
12305 }
12306 panic!(
12307 "ring never reached the expected state: {:?}",
12308 lock(ring).snapshot(None, None)
12309 );
12310 }
12311
12312 #[tokio::test(start_paused = true)]
12313 async fn a_crash_line_the_reader_had_not_reached_by_the_bound_is_kept_before_the_restart() {
12314 let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
12315 let (pump, release) = held_pump(&ring, "booting\n", "config error: missing storage\n");
12316
12317 settle_stderr_pump("crasher", &ring, pump, BOUND).await;
12318 let before_release = lock(&ring).snapshot(None, None);
12319 assert!(
12320 matches!(before_release.capture, CaptureState::Incomplete { .. }),
12321 "a reader that has not reached EOF cannot claim a whole tail: {before_release:?}"
12322 );
12323
12324 let next = lock(&ring).begin_process();
12327 lock(&ring).push_line_from(next, "next process booting");
12328 release.send(()).unwrap();
12329 wait_until(&ring, |ring| {
12330 ring.snapshot(None, None).capture == CaptureState::Captured
12331 })
12332 .await;
12333
12334 assert_eq!(
12335 untimed(lock(&ring).snapshot(None, None).entries),
12336 vec![
12337 line("booting"),
12338 line("config error: missing storage"),
12339 TailEntry::ProcessStart,
12340 line("next process booting"),
12341 ],
12342 "the crash's last line must survive a slow reader and stay in the crashed process's section"
12343 );
12344 }
12345
12346 #[tokio::test(start_paused = true)]
12347 async fn a_pipe_held_open_by_a_descendant_reads_incomplete_without_delaying_the_restart_past_the_bound(
12348 ) {
12349 let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
12350 let (pump, _held) = held_pump(&ring, "parent exiting\n", "");
12353
12354 let started = Instant::now();
12355 settle_stderr_pump("orphaning", &ring, pump, BOUND).await;
12356 assert_eq!(
12357 started.elapsed(),
12358 BOUND,
12359 "the restart must wait exactly the bound for a pipe that stays open, no longer"
12360 );
12361
12362 let next = lock(&ring).begin_process();
12363 lock(&ring).push_line_from(next, "next process booting");
12364 tokio::time::sleep(Duration::from_secs(60)).await;
12365
12366 let snapshot = lock(&ring).snapshot(None, None);
12367 match &snapshot.capture {
12368 CaptureState::Incomplete { reason } => assert!(
12369 reason.contains("had not reached EOF") && reason.contains("250ms"),
12370 "the reason must say what is missing and after how long: {reason}"
12371 ),
12372 other => panic!("expected Incomplete while the pipe is held open, got {other:?}"),
12373 }
12374 assert_eq!(
12375 untimed(snapshot.entries),
12376 vec![
12377 line("parent exiting"),
12378 TailEntry::ProcessStart,
12379 line("next process booting"),
12380 ]
12381 );
12382 }
12383
12384 #[tokio::test(start_paused = true)]
12385 async fn a_reader_that_reaches_eof_within_the_bound_leaves_the_tail_captured() {
12386 let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
12387 let (pump, release) = held_pump(&ring, "one\n", "two\n");
12388 release.send(()).unwrap();
12389
12390 settle_stderr_pump("clean", &ring, pump, BOUND).await;
12391
12392 let snapshot = lock(&ring).snapshot(None, None);
12393 assert_eq!(snapshot.capture, CaptureState::Captured);
12394 assert_eq!(untimed(snapshot.entries), vec![line("one"), line("two")]);
12395 }
12396}
12397
12398#[cfg(all(test, windows))]
12412mod job_containment_tests {
12413 use super::*;
12414 use std::{
12415 path::{Path, PathBuf},
12416 sync::{Arc, Mutex},
12417 time::{Duration, Instant},
12418 };
12419 use subc_test_support::TestTempDir;
12420
12421 fn stub_path() -> PathBuf {
12427 let mut path = std::env::current_exe().expect("current_exe available in tests");
12428 path.pop();
12429 path.pop();
12430 path.push("fake-aft-stub.exe");
12431 assert!(
12432 path.exists(),
12433 "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
12434 [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
12435 path.display()
12436 );
12437 path
12438 }
12439
12440 fn read_grandchild_pid(path: &Path) -> u32 {
12442 let deadline = Instant::now() + Duration::from_secs(10);
12443 loop {
12444 if let Ok(contents) = std::fs::read_to_string(path) {
12445 if let Ok(pid) = contents.trim().parse() {
12446 return pid;
12447 }
12448 }
12449 assert!(
12450 Instant::now() < deadline,
12451 "the stub never recorded a grandchild pid at {}",
12452 path.display()
12453 );
12454 std::thread::sleep(Duration::from_millis(10));
12455 }
12456 }
12457
12458 struct Fixture {
12461 _dir: TestTempDir,
12462 module_id: String,
12463 grandchild: u32,
12464 child: Option<SupervisedChild>,
12465 registry: Arc<Registry>,
12466 snapshot: Arc<Mutex<SupervisorSnapshot>>,
12467 terminal_ring: Arc<Mutex<TerminalRing>>,
12468 spawn_events: SpawnEventFeed,
12469 }
12470
12471 fn fixture(label: &str, module_id: &str) -> Fixture {
12472 let dir = TestTempDir::new(label);
12473 let pid_file = dir.join("grandchild.pid");
12474 let supervisor = Supervisor::new_for_test(
12475 Arc::new(Registry::default()),
12476 RestartPolicy::new(3, Duration::ZERO),
12477 );
12478 let runtime = supervisor.runtime_config();
12479 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
12480 let spec = ModuleSpec {
12481 module_id: module_id.to_string(),
12482 program: stub_path(),
12483 args: Vec::new(),
12487 env: vec![
12488 ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
12489 (
12490 "FAKE_AFT_GRANDCHILD_PID_FILE".to_string(),
12491 pid_file.display().to_string(),
12492 ),
12493 ],
12494 reserved: false,
12495 reserved_prefixes: Vec::new(),
12496 protocol: ModuleProtocol::Subc,
12497 overlap: Default::default(),
12498 };
12499 let child = spawn_and_mark_running(&spec, &runtime, &snapshot)
12500 .expect("spawn the supervised fixture");
12501 let grandchild = read_grandchild_pid(&pid_file);
12502 Fixture {
12503 _dir: dir,
12504 module_id: module_id.to_string(),
12505 grandchild,
12506 child: Some(child),
12507 registry: Arc::new(Registry::default()),
12508 snapshot,
12509 terminal_ring: Arc::clone(&runtime.terminal_ring),
12510 spawn_events: SpawnEventFeed::default(),
12511 }
12512 }
12513
12514 impl Fixture {
12515 async fn drain(&mut self) {
12517 let child = self
12518 .child
12519 .take()
12520 .expect("the fixture child is still present");
12521 drain_child_to_state(
12522 &self.module_id,
12523 ModuleProtocol::Subc,
12524 StopNotice::NotSent,
12527 &self.registry,
12528 None,
12529 &self.snapshot,
12530 &self.terminal_ring,
12531 &self.spawn_events,
12532 child,
12533 Duration::from_millis(500),
12534 ModuleState::Stopped,
12535 Some(false),
12536 )
12537 .await
12538 .expect("drain the supervised fixture");
12539 }
12540 }
12541
12542 #[tokio::test]
12548 async fn teardown_reaps_the_grandchild() {
12549 let mut fixture = fixture("teardown-grandchild", "tree-teardown");
12550 let grandchild = fixture.grandchild;
12551
12552 assert!(
12553 subc_jobobject::process_exists(grandchild),
12554 "grandchild {grandchild} must be alive before teardown, or this proves nothing"
12555 );
12556
12557 fixture.drain().await;
12558
12559 assert!(
12560 subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
12561 "grandchild {grandchild} outlived module teardown: the tree was not contained"
12562 );
12563 }
12564
12565 #[test]
12578 fn an_uncontained_grandchild_survives_a_direct_child_kill() {
12579 let dir = TestTempDir::new("teardown-uncontained");
12580 let pid_file = dir.join("grandchild.pid");
12581 let mut child = std::process::Command::new(stub_path())
12582 .env("FAKE_AFT_NEVER_CONNECT", "1")
12583 .env(
12584 "FAKE_AFT_GRANDCHILD_PID_FILE",
12585 pid_file.display().to_string(),
12586 )
12587 .stdin(std::process::Stdio::null())
12588 .stdout(std::process::Stdio::null())
12589 .stderr(std::process::Stdio::null())
12590 .spawn()
12591 .expect("spawn the uncontained fixture");
12592 let grandchild = read_grandchild_pid(&pid_file);
12593
12594 child.kill().expect("kill the direct child");
12596 let _ = child.wait();
12597
12598 assert!(
12599 subc_jobobject::process_exists(grandchild),
12600 "grandchild {grandchild} died with the direct child, so this control no longer \
12601 distinguishes contained from uncontained teardown and the regression test is \
12602 passing vacuously"
12603 );
12604
12605 kill_tree(grandchild);
12608 }
12609
12610 #[tokio::test]
12619 async fn dropping_containment_reaps_the_grandchild() {
12620 let mut fixture = fixture("drop-containment", "tree-drop");
12621 let grandchild = fixture.grandchild;
12622
12623 assert!(subc_jobobject::process_exists(grandchild));
12624
12625 fixture.child.as_mut().expect("child present").job = None;
12627
12628 assert!(
12629 subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
12630 "grandchild {grandchild} survived the containment handle closing, so a daemon \
12631 crash would leave the tree behind"
12632 );
12633 }
12634
12635 fn kill_tree(pid: u32) {
12637 let _ = std::process::Command::new("taskkill.exe")
12638 .args(["/PID", &pid.to_string(), "/T", "/F"])
12639 .stdin(std::process::Stdio::null())
12640 .stdout(std::process::Stdio::null())
12641 .stderr(std::process::Stdio::null())
12642 .status();
12643 assert!(
12644 subc_jobobject::wait_for_process_exit(pid, Duration::from_secs(10)),
12645 "could not clean up grandchild {pid}"
12646 );
12647 }
12648}
12649
12650#[cfg(test)]
12651mod privacy_trampoline_configuration_tests {
12652 #[tokio::test]
12653 #[cfg_attr(not(target_os = "macos"), ignore = "requires macOS privacy trampoline")]
12654 async fn macos_spawn_without_a_configured_trampoline_refuses_by_name() {
12655 #[cfg(target_os = "macos")]
12656 {
12657 let supervisor = super::Supervisor::new(
12658 std::sync::Arc::new(crate::Registry::default()),
12659 super::RestartPolicy::default(),
12660 );
12661 let error = supervisor.spawn(spec()).unwrap_err();
12662 assert!(
12663 error
12664 .to_string()
12665 .contains("no privacy trampoline configured"),
12666 "{error}"
12667 );
12668 }
12669 }
12670
12671 #[tokio::test]
12672 #[cfg_attr(not(target_os = "macos"), ignore = "requires macOS privacy trampoline")]
12673 async fn macos_wrong_trampoline_is_rejected_by_the_startup_probe() {
12674 #[cfg(target_os = "macos")]
12675 {
12676 let supervisor = super::Supervisor::new(
12677 std::sync::Arc::new(crate::Registry::default()),
12678 super::RestartPolicy::default(),
12679 )
12680 .with_privacy_trampoline(std::env::current_exe().unwrap());
12681 let error = supervisor.spawn(spec()).unwrap_err();
12682 assert!(
12683 error
12684 .to_string()
12685 .contains("binary does not implement the privacy trampoline protocol"),
12686 "{error}"
12687 );
12688 }
12689 }
12690
12691 #[cfg(target_os = "macos")]
12692 fn spec() -> super::ModuleSpec {
12693 super::ModuleSpec {
12694 module_id: "privacy-configuration".into(),
12695 program: "/bin/sleep".into(),
12696 args: vec!["30".into()],
12697 env: vec![],
12698 reserved: false,
12699 reserved_prefixes: vec![],
12700 protocol: subc_control::ModuleProtocol::None,
12701 overlap: super::ModuleOverlap::Exclusive,
12702 }
12703 }
12704}
12705
12706#[cfg(test)]
12707mod privacy_exec_boundary_tests {
12708 #[cfg(target_os = "macos")]
12709 use super::*;
12710 #[cfg(target_os = "macos")]
12711 use std::{
12712 io::{Read, Write},
12713 net::{TcpListener, TcpStream},
12714 };
12715
12716 #[cfg(target_os = "macos")]
12719 pub(super) fn before_image_sample(spec: &ModuleSpec, pid: u32) {
12720 if let Some((_, path)) = spec
12721 .env
12722 .iter()
12723 .find(|(key, _)| key == "SUBC_TEST_PRIVACY_SAMPLE_BARRIER")
12724 {
12725 let mut barrier = TcpStream::connect(path).unwrap();
12726 barrier
12727 .set_read_timeout(Some(Duration::from_secs(30)))
12728 .unwrap();
12729 barrier.write_all(&pid.to_ne_bytes()).unwrap();
12730 let mut release = [0];
12731 barrier.read_exact(&mut release).unwrap();
12732 assert_eq!(&release, b"X");
12733 }
12734 }
12735
12736 #[cfg(target_os = "macos")]
12737 fn spec(program: &str, args: &[&str]) -> ModuleSpec {
12738 ModuleSpec {
12739 module_id: "privacy-boundary".into(),
12740 program: program.into(),
12741 args: args.iter().map(|arg| (*arg).into()).collect(),
12742 env: vec![],
12743 reserved: false,
12744 reserved_prefixes: vec![],
12745 protocol: ModuleProtocol::None,
12746 overlap: ModuleOverlap::Exclusive,
12747 }
12748 }
12749
12750 #[cfg(target_os = "macos")]
12751 async fn accept(listener: TcpListener) -> TcpStream {
12752 let listener = tokio::net::TcpListener::from_std({
12754 listener.set_nonblocking(true).unwrap();
12755 listener
12756 })
12757 .unwrap();
12758 let (stream, _) = tokio::time::timeout(Duration::from_secs(30), listener.accept())
12759 .await
12760 .unwrap()
12761 .unwrap();
12762 let stream = stream.into_std().unwrap();
12763 stream.set_nonblocking(false).unwrap();
12764 stream
12765 .set_read_timeout(Some(Duration::from_secs(30)))
12766 .unwrap();
12767 stream
12768 }
12769
12770 #[tokio::test]
12771 #[cfg_attr(not(target_os = "macos"), ignore = "requires macOS privacy trampoline")]
12772 async fn macos_roster_withholds_a_nonnull_trampoline_image_until_exec_confirmation() {
12773 #[cfg(target_os = "macos")]
12774 {
12775 let root = subc_test_support::TestTempDir::new("privacy-roster-barrier");
12776 let exec_listener = TcpListener::bind("127.0.0.1:0").unwrap();
12779 let sample_listener = TcpListener::bind("127.0.0.1:0").unwrap();
12780 let record = root.join("live-children.json");
12781 let supervisor =
12782 Supervisor::new_for_test(Arc::new(Registry::default()), RestartPolicy::default())
12783 .with_live_children_record(&record);
12784 let runtime = supervisor.runtime_config();
12785 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
12786 let mut spec = spec("/bin/sleep", &["30"]);
12787 spec.env = vec![
12788 (
12789 "SUBC_TEST_PRIVACY_EXEC_BARRIER".into(),
12790 exec_listener.local_addr().unwrap().to_string(),
12791 ),
12792 (
12793 "SUBC_TEST_PRIVACY_SAMPLE_BARRIER".into(),
12794 sample_listener.local_addr().unwrap().to_string(),
12795 ),
12796 ];
12797 let spawn = tokio::task::spawn_blocking(move || {
12798 spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap()
12799 });
12800 let mut sample = accept(sample_listener).await;
12801 let mut pid = [0; 4];
12802 sample.read_exact(&mut pid).unwrap();
12803 let pid = u32::from_ne_bytes(pid);
12804 let mut exec = accept(exec_listener).await;
12805 let mut ready = [0];
12806 exec.read_exact(&mut ready).unwrap();
12807 assert_eq!(&ready, b"R");
12808 let trampoline = subc_os::file_identity(&test_privacy_trampoline()).unwrap();
12811 assert_eq!(
12812 observe_spawned_image(pid).unwrap().executable,
12813 Some(trampoline)
12814 );
12815 sample.write_all(b"X").unwrap();
12816 let mut child = spawn.await.unwrap();
12817 let early = crate::live_children::read_record(&record).unwrap();
12818 assert_eq!(early.len(), 1);
12819 assert_eq!(early[0].pid, pid);
12820 assert_eq!(
12821 early[0].executable, None,
12822 "unconfirmed trampoline image entered the roster"
12823 );
12824 assert!(child.report_ready.get().is_none());
12825 child.privacy_exec.as_mut().unwrap().deadline =
12828 tokio::time::Instant::now() + Duration::from_secs(30);
12829 exec.write_all(b"X").unwrap();
12830 child.confirm_privacy_exec().await;
12831 assert_eq!(child.spawn_failure, None);
12832 assert!(child.report_ready.get().is_some());
12833 let confirmed = crate::live_children::read_record(&record).unwrap();
12834 let module = subc_os::file_identity(std::path::Path::new("/bin/sleep")).unwrap();
12835 assert_ne!(module, trampoline);
12836 assert_eq!(confirmed[0].executable, Some(module.into()));
12837 child.start_kill().unwrap();
12838 child.wait().await.unwrap();
12839 child.release_roster();
12840 }
12841 }
12842
12843 #[tokio::test]
12844 #[cfg_attr(not(target_os = "macos"), ignore = "requires macOS privacy trampoline")]
12845 async fn macos_already_exited_121_in_try_wait_is_a_module_exit_not_a_trampoline_refusal() {
12846 #[cfg(target_os = "macos")]
12847 {
12848 let registry = Arc::new(Registry::default());
12849 let policy = RestartPolicy::new(0, Duration::ZERO);
12850 let supervisor = Supervisor::new_for_test(Arc::clone(®istry), policy);
12851 let runtime = supervisor.runtime_config();
12852 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
12853 let spec = spec("/bin/sh", &["-c", "exit 121"]);
12854 let mut child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
12858 let pid = child.pid;
12859 tokio::task::spawn_blocking(move || {
12860 subc_os::fork_exec_test::wait_for_child_exit_without_reaping(pid)
12861 })
12862 .await
12863 .unwrap()
12864 .unwrap();
12865 child.privacy_exec.as_mut().unwrap().deadline =
12866 tokio::time::Instant::now() + Duration::from_secs(30);
12867 let status = child.wait().await.unwrap();
12868 assert_eq!(status.code(), Some(121));
12869 assert!(child.privacy_exec.is_none());
12870 assert!(
12871 child.report_ready.get().is_none(),
12872 "an exited module must not publish a live pid"
12873 );
12874 let report = classify_reaped_child_exit(&snapshot, &child, &status);
12875 on_child_exit(
12876 &spec,
12877 policy,
12878 ®istry,
12879 &snapshot,
12880 &runtime.terminal_ring,
12881 &runtime.spawn_events,
12882 &runtime.child_roster,
12883 report,
12884 )
12885 .await;
12886 let state = lock_snapshot(&snapshot).unwrap();
12887 assert_eq!(state.state, ModuleState::Failed);
12888 assert_eq!(state.last_exit.as_ref().unwrap().code, Some(121));
12889 assert_eq!(state.reported_pid(), None);
12890 drop(state);
12891 let history = runtime.terminal_ring.lock().unwrap().snapshot();
12892 assert_eq!(history.entries.len(), 1);
12893 let terminal = &history.entries[0];
12894 assert_eq!(terminal.exit_code, Some(121));
12895 assert_eq!(terminal.exit_kind, subc_control::TerminalExitKind::Crash);
12896 assert_eq!(terminal.disposition, TerminalDisposition::Failed);
12897 assert_eq!(
12898 terminal.disposition_detail.as_deref(),
12899 Some(policy.budget_exhausted_detail().as_str()),
12900 "module exit 121 was classified as a trampoline refusal: {terminal:?}"
12901 );
12902 assert_eq!(child.spawn_failure, None);
12903 child.release_roster();
12904 }
12905 }
12906}
12907
12908#[cfg(all(test, unix))]
12912mod launch_nonce_descriptor_tests {
12913 use super::{ChildRoster, ModuleProtocol, ModuleSpec, SupervisorHandle};
12914 use crate::stderr_tail::{StderrRing, StderrTailConfig};
12915 use std::{
12916 path::PathBuf,
12917 sync::{Arc, Mutex},
12918 time::{Duration, Instant},
12919 };
12920 use subc_test_support::TestTempDir;
12921
12922 async fn probe(role: super::SpawnRole) {
12923 let scratch = TestTempDir::new("launch-nonce-descriptor");
12924 let fd_copy = scratch.join("from-descriptor");
12925 let env_copy = scratch.join("environment");
12926 let script = format!(
12927 "env > '{env}.tmp' && mv '{env}.tmp' '{env}'; cat <&3 > '{fd}.tmp' && mv '{fd}.tmp' '{fd}'; sleep 30",
12928 fd = fd_copy.display(), env = env_copy.display(),
12929 );
12930 let xdg = |name: &str| (name.to_string(), scratch.join(name).display().to_string());
12931 let spec = ModuleSpec {
12932 module_id: "nonce-descriptor-probe".to_string(),
12933 program: PathBuf::from("/bin/sh"),
12934 args: vec!["-c".to_string(), script],
12935 env: vec![
12936 xdg("XDG_DATA_HOME"),
12937 xdg("XDG_RUNTIME_DIR"),
12938 xdg("XDG_CONFIG_HOME"),
12939 ("SUBC_LAUNCH_NONCE".to_string(), "must-not-leak".to_string()),
12940 ],
12941 reserved: true,
12942 reserved_prefixes: Vec::new(),
12943 protocol: ModuleProtocol::Subc,
12944 overlap: Default::default(),
12945 };
12946 let handle = SupervisorHandle::new();
12947 let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
12948 let roster = ChildRoster::default();
12949 #[cfg(target_os = "macos")]
12950 {
12951 let path = super::test_privacy_trampoline();
12952 roster.set_privacy_trampoline(super::probe_privacy_trampoline(&path).map(|()| path));
12953 }
12954 let child = super::spawn_child_in_slot(
12955 &spec,
12956 None,
12957 Some(&handle),
12958 &ring,
12959 None,
12960 &roster,
12961 #[cfg(target_os = "linux")]
12962 None,
12963 role,
12964 matches!(role, super::SpawnRole::SwapCandidate),
12965 )
12966 .expect("spawn probe");
12967 let deadline = Instant::now() + Duration::from_secs(10);
12968 while !(fd_copy.exists() && env_copy.exists()) {
12969 assert!(Instant::now() < deadline, "probe never wrote its copies");
12970 tokio::time::sleep(Duration::from_millis(20)).await;
12971 }
12972 let nonce = std::fs::read_to_string(fd_copy).unwrap();
12973 assert!(!nonce.is_empty());
12974 let environment = std::fs::read_to_string(env_copy).unwrap();
12975 assert!(environment
12976 .lines()
12977 .any(|line| line.starts_with("SUBC_LAUNCH_NONCE_FD=3:")));
12978 let copy = environment
12979 .lines()
12980 .find_map(|line| line.strip_prefix("SUBC_LAUNCH_NONCE="));
12981 assert_eq!(
12982 copy, None,
12983 "Unix children must never receive the environment nonce"
12984 );
12985 if matches!(role, super::SpawnRole::Plain) {
12986 assert_eq!(
12987 handle.spawn_nonce(&spec.module_id).as_deref(),
12988 Some(nonce.as_str())
12989 );
12990 }
12991 drop(child);
12992 }
12993
12994 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
12995 async fn a_spawned_module_receives_its_nonce_only_on_descriptor_3() {
12996 probe(super::SpawnRole::Plain).await;
12997 }
12998
12999 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
13000 async fn swap_candidate_receives_its_nonce_only_on_descriptor_3() {
13001 probe(super::SpawnRole::SwapCandidate).await;
13002 }
13003}
13004
13005#[cfg(all(test, target_os = "linux"))]
13006mod cgroup_containment_tests {
13007 use super::*;
13008 use subc_test_support::TestTempDir;
13009
13010 fn running(pid: u32) -> bool {
13011 std::fs::read_to_string(format!("/proc/{pid}/stat"))
13013 .ok()
13014 .and_then(|stat| {
13015 stat.rsplit_once(") ")
13016 .map(|(_, rest)| rest.starts_with('Z'))
13017 })
13018 .is_some_and(|zombie| !zombie)
13019 }
13020
13021 #[tokio::test]
13022 async fn linux_teardown_reaps_the_grandchild() {
13023 teardown_tree("linux_teardown_reaps_the_grandchild", false).await;
13024 }
13025
13026 #[tokio::test]
13027 async fn linux_shutdown_straggler_reaps_the_grandchild() {
13028 teardown_tree("linux_shutdown_straggler_reaps_the_grandchild", true).await;
13029 }
13030
13031 async fn teardown_tree(test_name: &str, shutdown: bool) {
13032 let dir = TestTempDir::new(test_name);
13033 let root = PathBuf::from(format!(
13034 "/sys/fs/cgroup/subc-tree-test-{test_name}-{}-{}",
13035 std::process::id(),
13036 unix_ms_now()
13037 ));
13038 if let Err(error) = std::fs::create_dir(&root) {
13039 assert!(
13040 std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
13041 "required cgroup test cannot execute: {error}"
13042 );
13043 eprintln!(
13044 "SKIP {test_name}: no writable delegated cgroup at {}: {error}",
13045 root.display()
13046 );
13047 return;
13048 }
13049 let placement = subc_cgroup::prepare_at(&root)
13050 .expect("prepare isolated kernel cgroup")
13051 .expect("isolated cgroup is delegated");
13052 let module_id = "tree-teardown";
13053 let module = placement
13054 .module_path(module_id)
13055 .expect("create isolated module cgroup");
13056 if !module.join("cgroup.kill").exists() {
13057 std::fs::remove_dir(&module).unwrap();
13058 std::fs::remove_dir(root.join("subc-modules")).unwrap();
13059 std::fs::remove_dir(&root).unwrap();
13060 assert!(
13061 std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
13062 "required cgroup.kill interface unavailable"
13063 );
13064 eprintln!("SKIP {test_name}: cgroup.kill unavailable (kernel < 5.14)");
13065 return;
13066 }
13067 let supervisor = Supervisor::new_for_test(
13068 Arc::new(Registry::default()),
13069 RestartPolicy::new(3, Duration::ZERO),
13070 )
13071 .with_cgroup_placement(Some(placement));
13072 let mut runtime = supervisor.runtime_config();
13073 runtime.child_roster = runtime
13074 .child_roster
13075 .for_module(Arc::new(Mutex::new(Duration::from_millis(100))));
13076 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
13077 let pid_file = dir.join("grandchild.pid");
13078 let spec = ModuleSpec {
13079 module_id: module_id.to_string(),
13080 program: PathBuf::from("/bin/sh"),
13081 args: vec![
13082 "-c".into(),
13083 "trap '' TERM; sleep 600 & echo $! > \"$1\"; wait".into(),
13084 "fixture".into(),
13085 pid_file.display().to_string(),
13086 ],
13087 env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
13088 .into_iter()
13089 .map(|key| (key.to_string(), dir.display().to_string()))
13090 .collect(),
13091 reserved: false,
13092 reserved_prefixes: Vec::new(),
13093 protocol: ModuleProtocol::None,
13094 overlap: Default::default(),
13095 };
13096 let child =
13097 spawn_and_mark_running(&spec, &runtime, &snapshot).expect("spawn supervised tree");
13098 let deadline = tokio::time::Instant::now() + Duration::from_secs(10);
13099 let grandchild: u32 = loop {
13100 if let Ok(contents) = std::fs::read_to_string(&pid_file) {
13101 if let Ok(pid) = contents.trim().parse() {
13102 break pid;
13103 }
13104 }
13105 assert!(
13106 tokio::time::Instant::now() < deadline,
13107 "grandchild pid was not recorded"
13108 );
13109 tokio::time::sleep(Duration::from_millis(10)).await;
13110 };
13111 assert!(
13112 running(grandchild),
13113 "grandchild must be alive before teardown"
13114 );
13115 if shutdown {
13116 let mut child = child;
13117 crate::child_roster::end_children_for_daemon_shutdown(
13118 &runtime.child_roster,
13119 false,
13120 std::future::pending(),
13121 )
13122 .await;
13123 child.wait().await.expect("reap shutdown straggler");
13124 } else {
13125 drain_child_to_state(
13126 module_id,
13127 ModuleProtocol::None,
13128 StopNotice::NotSent,
13129 &Registry::default(),
13130 None,
13131 &snapshot,
13132 &runtime.terminal_ring,
13133 &SpawnEventFeed::default(),
13134 child,
13135 Duration::from_millis(100),
13136 ModuleState::Stopped,
13137 Some(false),
13138 )
13139 .await
13140 .expect("real supervisor teardown");
13141 }
13142 let deadline = tokio::time::Instant::now() + Duration::from_secs(5);
13143 while running(grandchild) && tokio::time::Instant::now() < deadline {
13144 tokio::time::sleep(Duration::from_millis(10)).await;
13145 }
13146 let survived = running(grandchild);
13147 if survived {
13149 let pid = rustix::process::Pid::from_raw(grandchild as i32).unwrap();
13150 let _ = rustix::process::kill_process(pid, rustix::process::Signal::KILL);
13151 tokio::time::sleep(Duration::from_millis(100)).await;
13152 }
13153 if module.exists() {
13154 std::fs::remove_dir(&module).expect("remove empty module cgroup");
13155 }
13156 std::fs::remove_dir(root.join("subc-modules")).unwrap();
13157 std::fs::remove_dir(&root).unwrap();
13158 assert!(
13159 !survived,
13160 "grandchild {grandchild} outlived module teardown"
13161 );
13162 eprintln!("EXECUTED {test_name}: grandchild {grandchild} killed in isolated cgroup");
13163 }
13164}