1use std::{
2 collections::{HashMap, VecDeque},
3 error::Error,
4 fmt, io,
5 path::PathBuf,
6 process::{ExitStatus, Stdio},
7 sync::{Arc, Mutex, OnceLock},
8 time::{Duration, SystemTime, UNIX_EPOCH},
9};
10
11use cortexkit_log::Retention;
12use serde_json::Value;
13use subc_control::{
14 ClientControlPush, LiveSpawn, ModuleProtocol, RouteCloseReason, SpawnCursor, SpawnEvent,
15 SpawnEventKind, SpawnSnapshot, SupervisorHealthStatus, TerminalDisposition, TerminalExitKind,
16};
17use subc_protocol::{
18 manifest::{SelfSignalKind, SignalAnchor},
19 session::{
20 HealthReport, HealthStatus, ModuleControlCommand, ModuleControlRequest,
21 MODULE_CONTROL_OP_HEALTH_CHECK,
22 },
23 Flags, FrameType, Priority, SUBC_LAUNCH_NONCE_ENV, SUBC_MODULE_ID_ENV,
24};
25use tokio::{
26 process::{Child, Command},
27 sync::{mpsc, oneshot, watch, Mutex as AsyncMutex},
28 task::JoinHandle,
29 time::{sleep, sleep_until, timeout, timeout_at, Instant},
30};
31use tracing::{debug, error, info, warn};
32
33use crate::{
34 child_roster::ChildRoster,
35 daemon_config::{
36 CAPTURE_KEEP_ENV, CAPTURE_MAX_AGE_DAYS_ENV, CAPTURE_MAX_FILE_MB_ENV, CK_LOG_ENV,
37 },
38 forwarding::{
39 CloseReason, ForwardingError, ForwardingTable, GoodbyeTarget, ModuleControlRpcOutcome,
40 ModuleDrainTarget, PendingModuleControlRpc,
41 },
42 provenance::{spawned_file_identity, ExecutableIdentityProbe, SpawnedFileIdentity},
43 registry::{ConnectionId, RegistryError},
44 stderr_tail::{
45 pump_stderr_to, pump_stdout_to, ChildOutputSink, StderrRing, StderrTailConfig,
46 StderrTailSnapshot,
47 },
48 terminal_ring::{TerminalHistorySnapshot, TerminalRecord, TerminalRing, TerminalRingConfig},
49 Frame, FrameSink, Registry,
50};
51
52#[path = "supervise_swap.rs"]
53mod swap;
54
55pub const SUBC_ARG: &str = "--subc";
61
62const DEFAULT_MAX_RESTARTS: u32 = 3;
63const DEFAULT_BACKOFF: Duration = Duration::from_millis(100);
64const DEFAULT_MAX_BACKOFF: Duration = Duration::from_secs(30);
65const DEFAULT_RESTART_WINDOW: Duration = Duration::from_secs(600);
69pub const DEFAULT_DRAIN_TIMEOUT: Duration = Duration::from_secs(30);
80const REGISTRY_RELEASE_TIMEOUT: Duration = Duration::from_secs(1);
81const REGISTRY_RELEASE_POLL: Duration = Duration::from_millis(10);
82const STDERR_PUMP_DRAIN_TIMEOUT: Duration = Duration::from_millis(250);
104pub const SPAWN_EVENT_RING_CAPACITY: usize = 4096;
106const SPAWN_SUBSCRIBER_BUFFER: usize = SPAWN_EVENT_RING_CAPACITY + 1;
107pub(crate) const SPAWN_SUBSCRIBER_LAGGED_CODE: &str = "spawn_subscriber_lagged";
112
113struct SupervisedChild {
114 child: Child,
115 #[cfg(target_os = "linux")]
118 module_id: String,
119 #[cfg(target_os = "linux")]
120 cgroup_placement: Option<subc_cgroup::Placement>,
121 #[cfg(windows)]
143 job: Option<subc_jobobject::JobObject>,
144 stdout_pump: Option<JoinHandle<()>>,
145 stderr_pump: Option<StderrPump>,
146 stderr_ring: Arc<Mutex<StderrRing>>,
147 spawned_at_ms: u64,
148 spawned_from: PathBuf,
149 spawned_file_identity: Option<SpawnedFileIdentity>,
150 process_start_time: Option<u64>,
151 process_identity: Option<ProcessIdentity>,
152 pid: u32,
153 roster_guard: Option<crate::child_roster::RosterGuard>,
156}
157
158impl SupervisedChild {
159 fn id(&self) -> Option<u32> {
160 Some(self.pid)
161 }
162
163 fn process_identity(&self) -> Option<ProcessIdentity> {
164 self.process_identity
165 }
166
167 async fn wait(&mut self) -> io::Result<ExitStatus> {
168 let result = self.child.wait().await;
176 #[cfg(target_os = "linux")]
177 if result.is_ok() {
178 if let Some(placement) = self.cgroup_placement.take() {
179 remove_module_cgroup(&placement, &self.module_id);
180 }
181 }
182 result
183 }
184
185 fn release_roster(&mut self) {
189 self.roster_guard = None;
190 }
191
192 fn start_kill(&mut self) -> io::Result<()> {
205 #[cfg(windows)]
206 if let Some(job) = &self.job {
207 if let Err(error) = job.terminate() {
208 debug!(
209 error = %error,
210 "job termination failed; the direct-child kill still owns the outcome"
211 );
212 }
213 }
214 self.child.start_kill()
215 }
216
217 async fn drain_stderr(&mut self, module_id: &str) {
218 if let Some(mut pump) = self.stdout_pump.take() {
219 match timeout(STDERR_PUMP_DRAIN_TIMEOUT, &mut pump).await {
220 Ok(Ok(())) => {}
221 Ok(Err(error)) => {
222 warn!(module_id, error = %error, "stdout pump ended unexpectedly");
223 }
224 Err(_) => {
225 pump.abort();
226 warn!(
227 module_id,
228 waited = ?STDERR_PUMP_DRAIN_TIMEOUT,
229 "stdout pump did not drain before restart; stopped it before the next process"
230 );
231 }
232 }
233 }
234
235 let Some(pump) = self.stderr_pump.take() else {
236 return;
237 };
238 settle_stderr_pump(
239 module_id,
240 &self.stderr_ring,
241 pump,
242 STDERR_PUMP_DRAIN_TIMEOUT,
243 )
244 .await;
245 }
246}
247
248struct StderrPump {
251 task: JoinHandle<()>,
252 generation: u64,
253}
254
255async fn settle_stderr_pump(
261 module_id: &str,
262 ring: &Arc<Mutex<StderrRing>>,
263 pump: StderrPump,
264 bound: Duration,
265) {
266 let lock = || ring.lock().unwrap_or_else(|poisoned| poisoned.into_inner());
267 let StderrPump {
268 mut task,
269 generation,
270 } = pump;
271 lock().retire_pump(generation);
272 match timeout(bound, &mut task).await {
273 Ok(Ok(())) => {}
274 Ok(Err(err)) => {
275 let mut ring = lock();
276 ring.mark_incomplete(format!("stderr pump ended unexpectedly: {err}"));
277 ring.finish_pump(generation);
278 warn!(module_id, error = %err, "stderr pump ended before clean EOF");
279 }
280 Err(_) => {
281 drop(task);
283 lock().mark_pump_late(
284 generation,
285 format!(
286 "stderr of the exited process had not reached EOF {bound:?} after it was \
287 retired (a descendant may still hold the pipe open); lines it still \
288 writes are kept in that process's section"
289 ),
290 );
291 warn!(
292 module_id,
293 waited = ?bound,
294 "stderr pipe of the exited process is still open; its reader keeps running without delaying the restart"
295 );
296 }
297 }
298}
299
300fn registration_release_events() -> &'static watch::Sender<u64> {
301 static EVENTS: OnceLock<watch::Sender<u64>> = OnceLock::new();
302 EVENTS.get_or_init(|| {
303 let (sender, _receiver) = watch::channel(0);
304 sender
305 })
306}
307
308pub(crate) fn notify_registration_release() {
309 let events = registration_release_events();
310 let next_generation = (*events.borrow()).wrapping_add(1);
311 events.send_replace(next_generation);
312}
313
314#[derive(Debug, Clone, PartialEq, Eq)]
316pub struct ModuleSpec {
317 pub module_id: String,
318 pub program: PathBuf,
319 pub args: Vec<String>,
320 pub env: Vec<(String, String)>,
321 pub reserved: bool,
326 pub reserved_prefixes: Vec<String>,
331 pub protocol: ModuleProtocol,
350 pub overlap: ModuleOverlap,
355}
356
357#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
364pub enum ModuleOverlap {
365 #[default]
367 Exclusive,
368 Safe,
380}
381
382impl ModuleOverlap {
383 pub fn as_str(self) -> &'static str {
384 match self {
385 Self::Exclusive => "exclusive",
386 Self::Safe => "safe",
387 }
388 }
389}
390
391pub const SUBC_SPAWN_ROLE_ENV: &str = "SUBC_SPAWN_ROLE";
401pub const SPAWN_ROLE_SWAP_CANDIDATE: &str = "swap_candidate";
403pub const DEFAULT_SWAP_READY_TIMEOUT: Duration = Duration::from_secs(100);
408
409#[derive(Debug, Clone, Copy, PartialEq, Eq)]
427pub struct RestartPolicy {
428 pub max_restarts: u32,
429 pub backoff: Duration,
432 pub max_backoff: Duration,
434 pub window: Duration,
438}
439
440impl RestartPolicy {
441 pub fn new(max_restarts: u32, backoff: Duration) -> Self {
445 Self {
446 max_restarts,
447 backoff,
448 max_backoff: DEFAULT_MAX_BACKOFF,
449 window: DEFAULT_RESTART_WINDOW,
450 }
451 }
452
453 pub fn with_max_backoff(mut self, max_backoff: Duration) -> Self {
454 self.max_backoff = max_backoff;
455 self
456 }
457
458 pub fn with_window(mut self, window: Duration) -> Self {
459 self.window = window;
460 self
461 }
462
463 fn delay_for_restart(&self, restart_in_window: u32) -> Duration {
468 if self.backoff.is_zero() || self.max_backoff.is_zero() {
469 return Duration::ZERO;
470 }
471
472 let mut delay = self.backoff;
473 for _ in 0..restart_in_window {
474 if delay >= self.max_backoff {
475 return self.max_backoff;
476 }
477 delay = delay
478 .checked_mul(10)
479 .unwrap_or(self.max_backoff)
480 .min(self.max_backoff);
481 }
482 delay.min(self.max_backoff)
483 }
484
485 fn budget_exhausted_detail(&self) -> String {
490 format!(
491 "crash budget exhausted: max_restarts={} within window_secs={}",
492 self.max_restarts,
493 self.window.as_secs()
494 )
495 }
496}
497
498impl Default for RestartPolicy {
499 fn default() -> Self {
500 Self {
501 max_restarts: DEFAULT_MAX_RESTARTS,
502 backoff: DEFAULT_BACKOFF,
503 max_backoff: DEFAULT_MAX_BACKOFF,
504 window: DEFAULT_RESTART_WINDOW,
505 }
506 }
507}
508
509#[derive(Debug, Clone, Copy, PartialEq, Eq)]
510struct CrashRestartSchedule {
511 restart_in_window: u32,
512 delay: Duration,
513}
514
515fn daemon_will_restart(
522 state: &mut SupervisorSnapshot,
523 policy: &RestartPolicy,
524 now: Instant,
525) -> bool {
526 state.enabled && state.crash_restarts_in_window(policy.window, now) < policy.max_restarts
527}
528
529const DEFAULT_HEALTH_CADENCE: Duration = Duration::from_secs(30);
530const DEFAULT_HEALTH_DEADLINE: Duration = Duration::from_secs(5);
531const DEFAULT_HEALTH_FAILURE_THRESHOLD: u32 = 3;
532const MAX_HEALTH_METRICS_BYTES: usize = 16 * 1024;
533
534#[derive(Debug, Clone, Copy, PartialEq, Eq)]
535pub enum HealthAction {
536 Report,
537 Restart,
538 Alert,
539}
540
541impl fmt::Display for HealthAction {
542 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
543 f.write_str(match self {
544 Self::Report => "report",
545 Self::Restart => "restart",
546 Self::Alert => "alert",
547 })
548 }
549}
550
551#[derive(Debug, Clone, Copy, PartialEq, Eq)]
552pub struct HealthConfig {
553 pub cadence: Duration,
554 pub deadline: Duration,
555 pub failure_threshold: u32,
556 pub on_degraded: HealthAction,
557 pub on_failing: HealthAction,
558 pub critical: bool,
559}
560
561impl Default for HealthConfig {
562 fn default() -> Self {
563 Self {
564 cadence: DEFAULT_HEALTH_CADENCE,
565 deadline: DEFAULT_HEALTH_DEADLINE,
566 failure_threshold: DEFAULT_HEALTH_FAILURE_THRESHOLD,
567 on_degraded: HealthAction::Report,
568 on_failing: HealthAction::Report,
569 critical: false,
570 }
571 }
572}
573
574#[derive(Debug, Clone, PartialEq)]
592pub struct ModuleHealthStatus {
593 pub status: SupervisorHealthStatus,
594 pub last_probe_ms: Option<u64>,
595 pub detail: Option<String>,
596 pub metrics: Option<Value>,
597 pub consecutive_failures: u32,
598 pub late_answer_count: u64,
601 pub last_late_answer_latency_ms: Option<u64>,
603 pub last_action: Option<String>,
604 pub last_action_ms: Option<u64>,
608}
609
610impl Default for ModuleHealthStatus {
611 fn default() -> Self {
612 Self {
613 status: SupervisorHealthStatus::Unknown,
614 last_probe_ms: None,
615 detail: None,
616 metrics: None,
617 consecutive_failures: 0,
618 late_answer_count: 0,
619 last_late_answer_latency_ms: None,
620 last_action: None,
621 last_action_ms: None,
622 }
623 }
624}
625
626#[derive(Debug, Clone, Copy, PartialEq, Eq)]
628pub enum ModuleState {
629 Starting,
630 Running,
631 Unresponsive,
632 Restarting,
633 Draining,
634 Stopped,
635 Failed,
636 Disabled,
637}
638
639impl fmt::Display for ModuleState {
640 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
641 f.write_str(match self {
642 Self::Starting => "starting",
643 Self::Running => "running",
644 Self::Unresponsive => "unresponsive",
645 Self::Restarting => "restarting",
646 Self::Draining => "draining",
647 Self::Stopped => "stopped",
648 Self::Failed => "failed",
649 Self::Disabled => "disabled",
650 })
651 }
652}
653
654#[derive(Debug, Clone, Copy, PartialEq, Eq)]
656pub enum ExitKind {
657 Clean,
658 Crash,
659 DeliberateSeverance,
660}
661
662impl From<ExitKind> for TerminalExitKind {
663 fn from(kind: ExitKind) -> Self {
664 match kind {
665 ExitKind::Clean => Self::Clean,
666 ExitKind::Crash => Self::Crash,
667 ExitKind::DeliberateSeverance => Self::DeliberateSeverance,
668 }
669 }
670}
671
672#[derive(Debug, Clone, Copy, PartialEq, Eq)]
675pub(crate) struct ProcessIdentity {
676 pub(crate) pid: u32,
677 pub(crate) start_time: u64,
678}
679
680#[derive(Debug, Clone, PartialEq, Eq)]
682pub struct ExitReport {
683 pub kind: ExitKind,
684 pub code: Option<i32>,
685 pub signal: Option<i32>,
686 pub at_ms: u64,
687}
688
689#[derive(Debug, Clone, PartialEq)]
692pub struct ModuleStatus {
693 pub module_id: String,
694 pub state: ModuleState,
695 pub enabled: bool,
696 pub process_alive: bool,
697 pub registration_active: bool,
698 pub protocol: ModuleProtocol,
701 pub live: bool,
712 pub restart_count: u32,
716 pub lifetime_restarts: u32,
720 pub spawn_generation: u64,
721 pub max_restarts: u32,
726 pub restart_window: Duration,
730 pub drain_timeout: Duration,
734 pub restart_backoff: Duration,
735 pub restart_max_backoff: Duration,
736 pub pid: Option<u32>,
737 pub spawned_at_ms: Option<u64>,
738 pub spawned_from: Option<PathBuf>,
739 pub process_start_time: Option<u64>,
740 pub last_exit: Option<ExitReport>,
741 pub health: ModuleHealthStatus,
742}
743
744#[derive(Debug, Clone, PartialEq)]
745struct SupervisorSnapshot {
746 state: ModuleState,
747 enabled: bool,
748 process_alive: bool,
749 crash_restarts: VecDeque<Instant>,
755 lifetime_restarts: u32,
756 spawn_generation: u64,
765 pid: Option<u32>,
766 spawned_at_ms: Option<u64>,
767 spawned_from: Option<PathBuf>,
768 spawned_file_identity: Option<SpawnedFileIdentity>,
769 process_start_time: Option<u64>,
770 deliberate_severance: Option<ProcessIdentity>,
771 last_exit: Option<ExitReport>,
772 health: ModuleHealthStatus,
773 in_alternate_slot: bool,
778 draining_to_replace: bool,
785 configuration_updated_since_spawn: bool,
791}
792
793impl SupervisorSnapshot {
794 fn starting() -> Self {
795 Self::new(ModuleState::Starting, true)
796 }
797
798 fn disabled() -> Self {
799 Self::new(ModuleState::Disabled, false)
800 }
801
802 fn failed() -> Self {
803 Self::new(ModuleState::Failed, true)
804 }
805
806 fn crash_restarts_in_window(&mut self, window: Duration, now: Instant) -> u32 {
810 while let Some(oldest) = self.crash_restarts.front() {
811 if now.duration_since(*oldest) > window {
812 self.crash_restarts.pop_front();
813 } else {
814 break;
815 }
816 }
817 u32::try_from(self.crash_restarts.len()).unwrap_or(u32::MAX)
818 }
819
820 fn record_crash_restart(&mut self, policy: &RestartPolicy, now: Instant) {
826 self.crash_restarts.push_back(now);
827 while self.crash_restarts.len() > policy.max_restarts as usize {
828 self.crash_restarts.pop_front();
829 }
830 self.lifetime_restarts += 1;
831 }
832
833 fn next_crash_restart(
837 &mut self,
838 policy: &RestartPolicy,
839 now: Instant,
840 ) -> Option<CrashRestartSchedule> {
841 let restart_in_window = self.crash_restarts_in_window(policy.window, now);
842 if restart_in_window >= policy.max_restarts {
843 return None;
844 }
845 self.record_crash_restart(policy, now);
846 Some(CrashRestartSchedule {
847 restart_in_window,
848 delay: policy.delay_for_restart(restart_in_window),
849 })
850 }
851
852 fn clear_crash_restarts(&mut self) {
857 self.crash_restarts.clear();
858 }
859
860 fn new(state: ModuleState, enabled: bool) -> Self {
861 Self {
862 state,
863 enabled,
864 process_alive: false,
865 crash_restarts: VecDeque::new(),
866 lifetime_restarts: 0,
867 spawn_generation: 0,
868 pid: None,
869 spawned_at_ms: None,
870 spawned_from: None,
871 spawned_file_identity: None,
872 process_start_time: None,
873 deliberate_severance: None,
874 last_exit: None,
875 health: ModuleHealthStatus::default(),
876 in_alternate_slot: false,
877 draining_to_replace: false,
878 configuration_updated_since_spawn: false,
879 }
880 }
881}
882
883type SharedSnapshot = Arc<Mutex<SupervisorSnapshot>>;
884
885type SpawnSubscriberKey = (ConnectionId, u64);
886
887#[derive(Debug)]
888struct SpawnSubscriber {
889 version: u8,
890 frames: mpsc::Sender<Frame>,
891 lagged: Option<oneshot::Sender<SpawnCursor>>,
895}
896
897#[derive(Debug)]
898struct SpawnEventState {
899 daemon_incarnation: String,
900 seq: u64,
901 capacity: usize,
902 live: HashMap<String, LiveSpawn>,
903 generations: HashMap<String, u64>,
904 events: VecDeque<SpawnEvent>,
905 subscribers: HashMap<SpawnSubscriberKey, SpawnSubscriber>,
906}
907
908impl Default for SpawnEventState {
909 fn default() -> Self {
910 Self {
911 daemon_incarnation: "unconfigured".to_string(),
912 seq: 0,
913 capacity: SPAWN_EVENT_RING_CAPACITY,
914 live: HashMap::new(),
915 generations: HashMap::new(),
916 events: VecDeque::new(),
917 subscribers: HashMap::new(),
918 }
919 }
920}
921
922#[derive(Debug, Clone, Default)]
923struct SpawnEventFeed(Arc<Mutex<SpawnEventState>>);
924
925#[derive(Debug, Clone, PartialEq, Eq)]
926pub(crate) enum SpawnSubscribeRefusal {
927 ForeignIncarnation { current: String },
928 TooOld { oldest: SpawnCursor },
929 Frame(String),
930}
931
932impl SpawnEventFeed {
933 fn configure_incarnation(&self, daemon_incarnation: String) {
934 let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
935 state.daemon_incarnation = daemon_incarnation;
936 state.seq = 0;
937 state.live.clear();
938 state.generations.clear();
939 state.events.clear();
940 state.subscribers.clear();
941 }
942
943 fn cursor(state: &SpawnEventState) -> SpawnCursor {
944 SpawnCursor {
945 daemon_incarnation: state.daemon_incarnation.clone(),
946 seq: state.seq,
947 }
948 }
949
950 fn snapshot(&self) -> SpawnSnapshot {
951 let state = self.0.lock().unwrap_or_else(|p| p.into_inner());
952 let mut live = state.live.values().cloned().collect::<Vec<_>>();
953 live.sort_by(|left, right| left.module_id.cmp(&right.module_id));
954 SpawnSnapshot {
955 cursor: Self::cursor(&state),
956 ring_bound: state.capacity as u64,
957 live,
958 }
959 }
960
961 fn emit_spawned(&self, module_id: &str, pid: u32, spawned_at_ms: u64) -> u64 {
962 let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
963 let generation = state
964 .generations
965 .get(module_id)
966 .copied()
967 .unwrap_or(0)
968 .checked_add(1)
969 .expect("spawn generation exhausted");
970 state.generations.insert(module_id.to_string(), generation);
971 let live = LiveSpawn {
972 module_id: module_id.to_string(),
973 spawn_generation: generation,
974 pid,
975 spawned_at_ms,
976 };
977 state.live.insert(module_id.to_string(), live);
978 Self::emit_locked(
979 &mut state,
980 SpawnEventKind::Spawned,
981 module_id.to_string(),
982 generation,
983 pid,
984 None,
985 None,
986 );
987 generation
988 }
989
990 fn emit_exited(&self, module_id: &str, exit_code: Option<i32>, exit_signal: Option<i32>) {
991 let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
992 let Some(live) = state.live.remove(module_id) else {
993 warn!(
994 module_id,
995 "terminal record had no live spawn event identity"
996 );
997 return;
998 };
999 Self::emit_locked(
1000 &mut state,
1001 SpawnEventKind::Exited,
1002 module_id.to_string(),
1003 live.spawn_generation,
1004 live.pid,
1005 exit_code,
1006 exit_signal,
1007 );
1008 }
1009
1010 fn emit_superseded_exited(
1017 &self,
1018 module_id: &str,
1019 spawn_generation: u64,
1020 pid: u32,
1021 exit_code: Option<i32>,
1022 exit_signal: Option<i32>,
1023 ) {
1024 let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1025 if state
1026 .live
1027 .get(module_id)
1028 .is_some_and(|live| live.spawn_generation == spawn_generation)
1029 {
1030 state.live.remove(module_id);
1031 }
1032 Self::emit_locked(
1033 &mut state,
1034 SpawnEventKind::Exited,
1035 module_id.to_string(),
1036 spawn_generation,
1037 pid,
1038 exit_code,
1039 exit_signal,
1040 );
1041 }
1042
1043 #[allow(clippy::too_many_arguments)]
1044 fn emit_locked(
1045 state: &mut SpawnEventState,
1046 kind: SpawnEventKind,
1047 module_id: String,
1048 spawn_generation: u64,
1049 pid: u32,
1050 exit_code: Option<i32>,
1051 exit_signal: Option<i32>,
1052 ) {
1053 state.seq = state
1054 .seq
1055 .checked_add(1)
1056 .expect("spawn event sequence exhausted");
1057 let event = SpawnEvent {
1058 cursor: Self::cursor(state),
1059 kind,
1060 module_id,
1061 spawn_generation,
1062 pid,
1063 exit_code,
1064 exit_signal,
1065 };
1066 state.events.push_back(event.clone());
1067 while state.events.len() > state.capacity {
1068 state.events.pop_front();
1069 }
1070 let body = match serde_json::to_vec(&event) {
1071 Ok(body) => body,
1072 Err(error) => {
1073 error!(%error, "failed to serialize supervisor spawn event");
1074 return;
1075 }
1076 };
1077 state.subscribers.retain(|(connection_id, corr), subscriber| {
1078 let frame = Frame::build_with_version(
1079 subscriber.version,
1080 FrameType::StreamData,
1081 control_flags(),
1082 0,
1083 0,
1084 *corr,
1085 body.clone(),
1086 );
1087 match frame {
1088 Ok(frame) => {
1089 if subscriber.frames.try_send(frame).is_ok() {
1090 true
1091 } else {
1092 warn!(connection_id = connection_id.get(), corr, "dropping lagged supervisor spawn subscriber");
1093 if let Some(lagged) = subscriber.lagged.take() {
1094 let _ = lagged.send(event.cursor.clone());
1095 }
1096 false
1097 }
1098 }
1099 Err(error) => {
1100 warn!(connection_id = connection_id.get(), corr, %error, "dropping supervisor spawn subscriber after frame build failure");
1101 false
1102 }
1103 }
1104 });
1105 }
1106
1107 fn subscribe(
1108 &self,
1109 connection_id: ConnectionId,
1110 corr: u64,
1111 version: u8,
1112 since: Option<SpawnCursor>,
1113 sink: FrameSink,
1114 ) -> Result<(), SpawnSubscribeRefusal> {
1115 let (frames, mut receiver) = mpsc::channel(SPAWN_SUBSCRIBER_BUFFER);
1116 let (lagged, mut lagged_rx) = oneshot::channel::<SpawnCursor>();
1117 {
1118 let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1119 let replay = if let Some(since) = since {
1120 if since.daemon_incarnation != state.daemon_incarnation {
1121 return Err(SpawnSubscribeRefusal::ForeignIncarnation {
1122 current: state.daemon_incarnation.clone(),
1123 });
1124 }
1125 if let Some(oldest) = state.events.front().map(|event| event.cursor.clone()) {
1126 if since.seq < oldest.seq.saturating_sub(1) {
1127 return Err(SpawnSubscribeRefusal::TooOld { oldest });
1128 }
1129 }
1130 state
1131 .events
1132 .iter()
1133 .filter(|event| event.cursor.seq > since.seq)
1134 .cloned()
1135 .collect::<Vec<_>>()
1136 } else {
1137 Vec::new()
1138 };
1139 for event in replay {
1140 let body = serde_json::to_vec(&event)
1141 .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1142 let frame = Frame::build_with_version(
1143 version,
1144 FrameType::StreamData,
1145 control_flags(),
1146 0,
1147 0,
1148 corr,
1149 body,
1150 )
1151 .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1152 frames
1153 .try_send(frame)
1154 .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1155 }
1156 state.subscribers.insert(
1157 (connection_id, corr),
1158 SpawnSubscriber {
1159 version,
1160 frames,
1161 lagged: Some(lagged),
1162 },
1163 );
1164 }
1165 tokio::spawn(async move {
1176 while let Some(frame) = receiver.recv().await {
1177 if sink.send(frame).await.is_err() {
1178 return;
1179 }
1180 }
1181 let Ok(first_undelivered) = lagged_rx.try_recv() else {
1182 return;
1183 };
1184 match spawn_subscriber_lagged_frame(version, corr, first_undelivered) {
1185 Ok(frame) => {
1186 let _ = sink.send(frame).await;
1187 }
1188 Err(error) => {
1189 error!(%error, corr, "failed to build lagged spawn subscriber terminal frame");
1190 }
1191 }
1192 });
1193 Ok(())
1194 }
1195
1196 fn cancel(&self, connection_id: ConnectionId, corr: u64) -> bool {
1197 let Some(subscriber) = self
1198 .0
1199 .lock()
1200 .unwrap_or_else(|p| p.into_inner())
1201 .subscribers
1202 .remove(&(connection_id, corr))
1203 else {
1204 return false;
1205 };
1206 if let Ok(frame) = Frame::build_with_version(
1207 subscriber.version,
1208 FrameType::StreamEnd,
1209 control_flags(),
1210 0,
1211 0,
1212 corr,
1213 Vec::new(),
1214 ) {
1215 tokio::spawn(async move {
1216 let _ = subscriber.frames.send(frame).await;
1217 });
1218 }
1219 true
1220 }
1221
1222 fn remove_connection(&self, connection_id: ConnectionId) {
1223 self.0
1224 .lock()
1225 .unwrap_or_else(|p| p.into_inner())
1226 .subscribers
1227 .retain(|(subscriber_connection, _), _| *subscriber_connection != connection_id);
1228 }
1229
1230 #[cfg(any(test, feature = "test-support"))]
1231 fn set_capacity(&self, capacity: usize) {
1232 self.0.lock().unwrap_or_else(|p| p.into_inner()).capacity = capacity;
1233 }
1234
1235 #[cfg(any(test, feature = "test-support"))]
1236 fn subscriber_count(&self) -> usize {
1237 self.0
1238 .lock()
1239 .unwrap_or_else(|p| p.into_inner())
1240 .subscribers
1241 .len()
1242 }
1243}
1244
1245fn spawn_subscriber_lagged_frame(
1248 version: u8,
1249 corr: u64,
1250 first_undelivered: SpawnCursor,
1251) -> Result<Frame, String> {
1252 let body = serde_json::to_vec(&subc_protocol::ErrorBody {
1253 code: SPAWN_SUBSCRIBER_LAGGED_CODE.to_string(),
1254 message: "spawn subscriber fell behind and was dropped; resubscribe from the last cursor received"
1255 .to_string(),
1256 detail: Some(serde_json::json!({
1257 "first_undelivered_cursor": first_undelivered
1258 })),
1259 })
1260 .map_err(|error| error.to_string())?;
1261 Frame::build_with_version(version, FrameType::Error, control_flags(), 0, 0, corr, body)
1262 .map_err(|error| error.to_string())
1263}
1264
1265pub trait ModuleProcessLiveness: Send + Sync {
1266 fn process_live(&self, module_id: &str) -> Option<bool>;
1267
1268 fn process_replacing(&self, _module_id: &str) -> bool {
1274 false
1275 }
1276}
1277
1278#[derive(Debug, Clone, Default)]
1280pub struct SupervisorProcessLiveness {
1281 snapshots: Arc<Mutex<HashMap<String, SharedSnapshot>>>,
1282}
1283
1284impl SupervisorProcessLiveness {
1285 pub fn new() -> Self {
1286 Self::default()
1287 }
1288
1289 fn track(&self, module_id: String, snapshot: SharedSnapshot) {
1290 let mut snapshots = self
1291 .snapshots
1292 .lock()
1293 .unwrap_or_else(|poisoned| poisoned.into_inner());
1294 snapshots.insert(module_id, snapshot);
1295 }
1296
1297 fn untrack_if_current(&self, module_id: &str, snapshot: &SharedSnapshot) {
1298 let mut snapshots = self
1299 .snapshots
1300 .lock()
1301 .unwrap_or_else(|poisoned| poisoned.into_inner());
1302 let is_current = snapshots
1303 .get(module_id)
1304 .map(|tracked| Arc::ptr_eq(tracked, snapshot))
1305 .unwrap_or(false);
1306 if is_current {
1307 snapshots.remove(module_id);
1308 }
1309 }
1310}
1311
1312impl ModuleProcessLiveness for SupervisorProcessLiveness {
1313 fn process_live(&self, module_id: &str) -> Option<bool> {
1314 let snapshot = {
1315 let snapshots = self
1316 .snapshots
1317 .lock()
1318 .unwrap_or_else(|poisoned| poisoned.into_inner());
1319 snapshots.get(module_id).cloned()
1320 }?;
1321 let snapshot = snapshot
1322 .lock()
1323 .unwrap_or_else(|poisoned| poisoned.into_inner());
1324 Some(snapshot.state == ModuleState::Running && snapshot.process_alive)
1325 }
1326
1327 fn process_replacing(&self, module_id: &str) -> bool {
1328 let Some(snapshot) = self
1329 .snapshots
1330 .lock()
1331 .unwrap_or_else(|poisoned| poisoned.into_inner())
1332 .get(module_id)
1333 .cloned()
1334 else {
1335 return false;
1336 };
1337 let snapshot = snapshot
1338 .lock()
1339 .unwrap_or_else(|poisoned| poisoned.into_inner());
1340 snapshot.enabled
1341 && match snapshot.state {
1342 ModuleState::Restarting => true,
1343 ModuleState::Draining => snapshot.draining_to_replace,
1344 ModuleState::Starting
1345 | ModuleState::Running
1346 | ModuleState::Unresponsive
1347 | ModuleState::Stopped
1348 | ModuleState::Failed
1349 | ModuleState::Disabled => false,
1350 }
1351 }
1352}
1353
1354#[derive(Debug, Clone)]
1355struct SupervisorRuntimeConfig {
1356 restart_policy: RestartPolicy,
1357 drain_timeout: Duration,
1360 effective_drain_timeout: Arc<Mutex<Duration>>,
1363 default_drain_timeout: Duration,
1366 health: HealthConfig,
1367 connection_file_path: Option<PathBuf>,
1368 capture_logs_dir: Option<PathBuf>,
1369 forwarding: Option<Arc<ForwardingTable>>,
1370 supervisor_handle: Option<SupervisorHandle>,
1373 stderr_ring: Arc<Mutex<StderrRing>>,
1380 terminal_ring: Arc<Mutex<TerminalRing>>,
1381 spawn_events: SpawnEventFeed,
1382 child_roster: ChildRoster,
1383 #[cfg(target_os = "linux")]
1384 cgroup_placement: Option<subc_cgroup::Placement>,
1385 #[cfg(test)]
1386 test_seed_stale_facts_before_enable_spawn: bool,
1387}
1388
1389#[derive(Debug, Clone, PartialEq, Eq)]
1390struct SupervisedConfiguration {
1391 spec: ModuleSpec,
1392 health: HealthConfig,
1393}
1394
1395#[derive(Debug, Clone, Default)]
1401pub struct SupervisorHandle {
1402 modules: Arc<Mutex<HashMap<String, SupervisedModule>>>,
1403 spawn_events: SpawnEventFeed,
1404 reserved_nonces: Arc<Mutex<HashMap<String, Option<String>>>>,
1415 removal_tombstones: Arc<Mutex<HashMap<String, u64>>>,
1421 spawn_nonces: Arc<Mutex<HashMap<String, String>>>,
1425 reserved_prefix_owners: Arc<Mutex<HashMap<String, String>>>,
1433 swaps: Arc<Mutex<HashMap<String, OpenSwap>>>,
1439 promotion_observer: PromotionObserverSlot,
1441 operation_lock: Arc<AsyncMutex<()>>,
1445}
1446
1447pub(crate) trait SwapPromotionObserver: Send + Sync {
1456 fn swap_promoted(&self, registration: &crate::registry::ModuleRegistration);
1457}
1458
1459#[derive(Clone, Default)]
1463struct PromotionObserverSlot(Arc<Mutex<Option<std::sync::Weak<dyn SwapPromotionObserver>>>>);
1464
1465impl fmt::Debug for PromotionObserverSlot {
1466 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1467 f.write_str("PromotionObserverSlot")
1468 }
1469}
1470
1471#[derive(Debug, Clone)]
1473struct OpenSwap {
1474 candidate_nonce: String,
1477 incumbent_nonce: Option<String>,
1482 candidate_admitted: bool,
1486}
1487
1488#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1491pub(crate) enum SwapHelloAdmission {
1492 NotSwapping,
1495 Candidate,
1497 Refused,
1500}
1501
1502#[derive(Debug, Clone, PartialEq, Eq)]
1503pub(crate) enum ReservedHelloRejection {
1504 Exact {
1505 module_id: String,
1506 },
1507 Prefix {
1508 prefix: String,
1509 owner_module_id: String,
1510 },
1511}
1512
1513impl SupervisorHandle {
1514 pub fn new() -> Self {
1515 Self::default()
1516 }
1517
1518 pub(crate) fn spawn_snapshot(&self) -> SpawnSnapshot {
1519 self.spawn_events.snapshot()
1520 }
1521
1522 pub(crate) fn subscribe_spawns(
1523 &self,
1524 connection_id: ConnectionId,
1525 corr: u64,
1526 version: u8,
1527 since: Option<SpawnCursor>,
1528 sink: FrameSink,
1529 ) -> Result<(), SpawnSubscribeRefusal> {
1530 self.spawn_events
1531 .subscribe(connection_id, corr, version, since, sink)
1532 }
1533
1534 pub(crate) fn cancel_spawn_subscription(&self, connection_id: ConnectionId, corr: u64) -> bool {
1535 self.spawn_events.cancel(connection_id, corr)
1536 }
1537
1538 pub(crate) fn remove_spawn_subscribers(&self, connection_id: ConnectionId) {
1539 self.spawn_events.remove_connection(connection_id);
1540 }
1541
1542 #[cfg(any(test, feature = "test-support"))]
1543 pub fn set_spawn_event_capacity_for_test(&self, capacity: usize) {
1544 assert!(capacity > 0, "spawn event capacity must be non-zero");
1545 self.spawn_events.set_capacity(capacity);
1546 }
1547
1548 #[cfg(any(test, feature = "test-support"))]
1549 pub fn spawn_subscriber_count_for_test(&self) -> usize {
1550 self.spawn_events.subscriber_count()
1551 }
1552
1553 pub fn set_spawn_nonce(&self, module_id: &str, nonce: String) {
1556 self.spawn_nonces
1557 .lock()
1558 .unwrap_or_else(|poisoned| poisoned.into_inner())
1559 .insert(module_id.to_string(), nonce);
1560 }
1561
1562 pub fn set_reserved_nonce(&self, module_id: &str, nonce: String) {
1565 self.reserved_nonces
1566 .lock()
1567 .unwrap_or_else(|poisoned| poisoned.into_inner())
1568 .insert(module_id.to_string(), Some(nonce));
1569 }
1570
1571 pub fn set_reserved_prefixes(&self, owner_module_id: &str, prefixes: &[String]) {
1573 let mut owners = self
1574 .reserved_prefix_owners
1575 .lock()
1576 .unwrap_or_else(|poisoned| poisoned.into_inner());
1577 owners.retain(|_, owner| owner != owner_module_id);
1578 for prefix in prefixes {
1579 owners.insert(prefix.clone(), owner_module_id.to_string());
1580 }
1581 }
1582
1583 #[cfg(test)]
1585 pub(crate) fn spawn_nonce(&self, module_id: &str) -> Option<String> {
1586 self.spawn_nonces
1587 .lock()
1588 .unwrap_or_else(|poisoned| poisoned.into_inner())
1589 .get(module_id)
1590 .cloned()
1591 }
1592
1593 fn apply_identity_configuration(&self, spec: &ModuleSpec) {
1594 self.set_reserved_prefixes(&spec.module_id, &spec.reserved_prefixes);
1595 let spawn_nonce = self
1596 .spawn_nonces
1597 .lock()
1598 .unwrap_or_else(|poisoned| poisoned.into_inner())
1599 .get(&spec.module_id)
1600 .cloned();
1601 let mut reserved_nonces = self
1602 .reserved_nonces
1603 .lock()
1604 .unwrap_or_else(|poisoned| poisoned.into_inner());
1605 if spec.reserved {
1606 reserved_nonces.insert(spec.module_id.clone(), spawn_nonce);
1611 }
1612 drop(reserved_nonces);
1613 self.removal_tombstones
1617 .lock()
1618 .unwrap_or_else(|poisoned| poisoned.into_inner())
1619 .remove(&spec.module_id);
1620 }
1621
1622 pub fn reserved_hello_authorized(&self, module_id: &str, presented: Option<&str>) -> bool {
1627 self.reserved_hello_rejection(module_id, presented)
1628 .is_none()
1629 }
1630
1631 pub(crate) fn reserved_hello_rejection(
1632 &self,
1633 module_id: &str,
1634 presented: Option<&str>,
1635 ) -> Option<ReservedHelloRejection> {
1636 let nonces = self
1637 .reserved_nonces
1638 .lock()
1639 .unwrap_or_else(|poisoned| poisoned.into_inner());
1640 if let Some(expected) = nonces.get(module_id) {
1641 let authorized = match expected {
1645 Some(expected) => {
1646 presented.is_some_and(|p| constant_time_eq(expected.as_bytes(), p.as_bytes()))
1647 }
1648 None => false,
1649 };
1650 if authorized {
1651 return None;
1652 }
1653 return Some(ReservedHelloRejection::Exact {
1654 module_id: module_id.to_string(),
1655 });
1656 }
1657 drop(nonces);
1658
1659 let matched_prefix = self
1660 .reserved_prefix_owners
1661 .lock()
1662 .unwrap_or_else(|poisoned| poisoned.into_inner())
1663 .iter()
1664 .filter(|(prefix, _)| module_id.starts_with(prefix.as_str()))
1665 .max_by_key(|(prefix, _)| prefix.len())
1666 .map(|(prefix, owner)| (prefix.clone(), owner.clone()));
1667 let (prefix, owner_module_id) = matched_prefix?;
1668
1669 let authorized = presented.is_some_and(|presented| {
1670 self.spawn_nonces
1671 .lock()
1672 .unwrap_or_else(|poisoned| poisoned.into_inner())
1673 .get(&owner_module_id)
1674 .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()))
1675 || self.swap_nonce_matches(&owner_module_id, presented)
1678 });
1679 if authorized {
1680 None
1681 } else {
1682 Some(ReservedHelloRejection::Prefix {
1683 prefix,
1684 owner_module_id,
1685 })
1686 }
1687 }
1688
1689 pub fn spawned_consumer_authorized(&self, module_id: &str, presented: &str) -> bool {
1694 if presented.is_empty() {
1695 return false;
1696 }
1697 let nonces = self
1698 .spawn_nonces
1699 .lock()
1700 .unwrap_or_else(|poisoned| poisoned.into_inner());
1701 let current = nonces
1702 .get(module_id)
1703 .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()));
1704 drop(nonces);
1705 current || self.swap_nonce_matches(module_id, presented)
1710 }
1711
1712 fn swap_nonce_matches(&self, module_id: &str, presented: &str) -> bool {
1714 let swaps = self
1715 .swaps
1716 .lock()
1717 .unwrap_or_else(|poisoned| poisoned.into_inner());
1718 swaps.get(module_id).is_some_and(|swap| {
1719 constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes())
1720 || swap.incumbent_nonce.as_deref().is_some_and(|incumbent| {
1721 constant_time_eq(incumbent.as_bytes(), presented.as_bytes())
1722 })
1723 })
1724 }
1725
1726 pub(crate) fn open_swap(&self, module_id: &str, candidate_nonce: String) {
1729 let incumbent_nonce = self
1730 .spawn_nonces
1731 .lock()
1732 .unwrap_or_else(|poisoned| poisoned.into_inner())
1733 .get(module_id)
1734 .cloned();
1735 self.swaps
1736 .lock()
1737 .unwrap_or_else(|poisoned| poisoned.into_inner())
1738 .insert(
1739 module_id.to_string(),
1740 OpenSwap {
1741 candidate_nonce,
1742 incumbent_nonce,
1743 candidate_admitted: false,
1744 },
1745 );
1746 }
1747
1748 pub(crate) fn close_swap(&self, module_id: &str) {
1751 self.swaps
1752 .lock()
1753 .unwrap_or_else(|poisoned| poisoned.into_inner())
1754 .remove(module_id);
1755 }
1756
1757 pub(crate) fn set_swap_promotion_observer(
1760 &self,
1761 observer: std::sync::Weak<dyn SwapPromotionObserver>,
1762 ) {
1763 *self
1764 .promotion_observer
1765 .0
1766 .lock()
1767 .unwrap_or_else(|poisoned| poisoned.into_inner()) = Some(observer);
1768 }
1769
1770 fn notify_swap_promoted(&self, registration: &crate::registry::ModuleRegistration) {
1773 let observer = self
1774 .promotion_observer
1775 .0
1776 .lock()
1777 .unwrap_or_else(|poisoned| poisoned.into_inner())
1778 .as_ref()
1779 .and_then(std::sync::Weak::upgrade);
1780 if let Some(observer) = observer {
1781 observer.swap_promoted(registration);
1782 }
1783 }
1784
1785 pub(crate) fn swap_open(&self, module_id: &str) -> bool {
1787 self.swaps
1788 .lock()
1789 .unwrap_or_else(|poisoned| poisoned.into_inner())
1790 .contains_key(module_id)
1791 }
1792
1793 fn promote_swap_nonce(&self, module_id: &str, reserved: bool) {
1798 let candidate_nonce = self
1799 .swaps
1800 .lock()
1801 .unwrap_or_else(|poisoned| poisoned.into_inner())
1802 .get(module_id)
1803 .map(|swap| swap.candidate_nonce.clone());
1804 let Some(nonce) = candidate_nonce else {
1805 return;
1806 };
1807 self.set_spawn_nonce(module_id, nonce.clone());
1808 if reserved {
1809 self.set_reserved_nonce(module_id, nonce);
1810 }
1811 }
1812
1813 pub(crate) fn swap_hello_admission(
1828 &self,
1829 module_id: &str,
1830 presented: Option<&str>,
1831 ) -> SwapHelloAdmission {
1832 let swaps = self
1833 .swaps
1834 .lock()
1835 .unwrap_or_else(|poisoned| poisoned.into_inner());
1836 let Some(swap) = swaps.get(module_id) else {
1837 return SwapHelloAdmission::NotSwapping;
1838 };
1839 let Some(presented) = presented else {
1840 return SwapHelloAdmission::Refused;
1841 };
1842 if constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes()) {
1843 return if swap.candidate_admitted {
1844 SwapHelloAdmission::Refused
1845 } else {
1846 SwapHelloAdmission::Candidate
1847 };
1848 }
1849 if swap
1850 .incumbent_nonce
1851 .as_deref()
1852 .is_some_and(|incumbent| constant_time_eq(incumbent.as_bytes(), presented.as_bytes()))
1853 {
1854 return SwapHelloAdmission::NotSwapping;
1855 }
1856 SwapHelloAdmission::Refused
1857 }
1858
1859 pub(crate) fn mark_swap_candidate_admitted(&self, module_id: &str) {
1862 if let Some(swap) = self
1863 .swaps
1864 .lock()
1865 .unwrap_or_else(|poisoned| poisoned.into_inner())
1866 .get_mut(module_id)
1867 {
1868 swap.candidate_admitted = true;
1869 }
1870 }
1871
1872 pub fn spawn_launch_nonce_for(&self, module_id: &str) -> Option<String> {
1874 self.spawn_nonces
1875 .lock()
1876 .unwrap_or_else(|poisoned| poisoned.into_inner())
1877 .get(module_id)
1878 .cloned()
1879 }
1880
1881 pub fn reserved_launch_nonce_for(&self, module_id: &str) -> Option<String> {
1883 self.reserved_nonces
1884 .lock()
1885 .unwrap_or_else(|poisoned| poisoned.into_inner())
1886 .get(module_id)
1887 .cloned()
1888 .flatten()
1889 }
1890
1891 pub fn insert(&self, module: SupervisedModule) -> Option<SupervisedModule> {
1892 let mut modules = self
1893 .modules
1894 .lock()
1895 .unwrap_or_else(|poisoned| poisoned.into_inner());
1896 modules.insert(module.module_id().to_string(), module)
1897 }
1898
1899 pub fn get(&self, module_id: &str) -> Option<SupervisedModule> {
1900 let modules = self
1901 .modules
1902 .lock()
1903 .unwrap_or_else(|poisoned| poisoned.into_inner());
1904 modules.get(module_id).cloned()
1905 }
1906
1907 pub(crate) fn record_late_health_answer(
1908 &self,
1909 module_id: &str,
1910 latency_ms: u64,
1911 ) -> Result<bool, SuperviseError> {
1912 let Some(module) = self.get(module_id) else {
1913 return Ok(false);
1914 };
1915 update_snapshot(&module.inner.snapshot, Some(module_id), |state| {
1916 state.health.late_answer_count = state.health.late_answer_count.saturating_add(1);
1917 state.health.last_late_answer_latency_ms = Some(latency_ms);
1918 state.health.consecutive_failures = 0;
1926 })?;
1927 Ok(true)
1928 }
1929
1930 pub fn record_deliberate_severance(&self, module_id: &str) -> Result<bool, SuperviseError> {
1936 let Some(module) = self.get(module_id) else {
1937 return Ok(false);
1938 };
1939 let status = module.status()?;
1940 let Some((pid, start_time)) = status.pid.zip(status.process_start_time) else {
1941 return Ok(false);
1942 };
1943 module.record_deliberate_severance(ProcessIdentity { pid, start_time })
1944 }
1945
1946 pub fn list(&self) -> Vec<SupervisedModule> {
1947 let modules = self
1948 .modules
1949 .lock()
1950 .unwrap_or_else(|poisoned| poisoned.into_inner());
1951 let mut modules = modules.values().cloned().collect::<Vec<_>>();
1952 modules.sort_by(|left, right| left.module_id().cmp(right.module_id()));
1953 modules
1954 }
1955
1956 pub(crate) fn retire(&self, module_id: &str) -> Option<SupervisedModule> {
1957 self.spawn_nonces
1958 .lock()
1959 .unwrap_or_else(|poisoned| poisoned.into_inner())
1960 .remove(module_id);
1961 self.close_swap(module_id);
1962 let mut reserved_nonces = self
1963 .reserved_nonces
1964 .lock()
1965 .unwrap_or_else(|poisoned| poisoned.into_inner());
1966 if reserved_nonces.contains_key(module_id) {
1967 reserved_nonces.insert(module_id.to_string(), None);
1970 }
1971 drop(reserved_nonces);
1972 self.reserved_prefix_owners
1973 .lock()
1974 .unwrap_or_else(|poisoned| poisoned.into_inner())
1975 .retain(|_, owner| owner != module_id);
1976 self.modules
1977 .lock()
1978 .unwrap_or_else(|poisoned| poisoned.into_inner())
1979 .remove(module_id)
1980 }
1981
1982 pub(crate) fn record_rescan_removal(&self, module_id: &str) {
1985 self.removal_tombstones
1986 .lock()
1987 .unwrap_or_else(|poisoned| poisoned.into_inner())
1988 .insert(module_id.to_string(), unix_ms_now());
1989 }
1990
1991 pub(crate) fn removal_tombstone_age_ms(&self, module_id: &str) -> Option<u64> {
1993 self.removal_tombstones
1994 .lock()
1995 .unwrap_or_else(|poisoned| poisoned.into_inner())
1996 .get(module_id)
1997 .copied()
1998 .map(|removed_at_ms| unix_ms_now().saturating_sub(removed_at_ms))
1999 }
2000
2001 pub(crate) fn release_retained_reserved_gate(&self, module_id: &str) -> bool {
2006 if self.get(module_id).is_some() {
2007 return false;
2008 }
2009 let mut reserved_nonces = self
2010 .reserved_nonces
2011 .lock()
2012 .unwrap_or_else(|poisoned| poisoned.into_inner());
2013 if !matches!(reserved_nonces.get(module_id), Some(None)) {
2014 return false;
2015 }
2016 reserved_nonces.remove(module_id);
2017 true
2018 }
2019
2020 pub(crate) fn operation_lock(&self) -> Arc<AsyncMutex<()>> {
2021 Arc::clone(&self.operation_lock)
2022 }
2023}
2024
2025#[derive(Debug, Clone)]
2027pub struct Supervisor {
2028 registry: Arc<Registry>,
2029 restart_policy: RestartPolicy,
2030 drain_timeout: Duration,
2031 connection_file_path: Option<PathBuf>,
2032 capture_logs_dir: Option<PathBuf>,
2033 forwarding: Option<Arc<ForwardingTable>>,
2034 process_liveness: Arc<SupervisorProcessLiveness>,
2035 supervisor_handle: Option<SupervisorHandle>,
2036 health: HealthConfig,
2037 daemon_start_clock: crate::clock::StartClock,
2038 terminal_journal: Option<Arc<crate::terminal_journal::TerminalJournal>>,
2039 spawn_events: SpawnEventFeed,
2040 provenance_probe: ExecutableIdentityProbe,
2041 child_roster: ChildRoster,
2044 #[cfg(target_os = "linux")]
2045 cgroup_placement: Option<subc_cgroup::Placement>,
2046}
2047
2048impl Supervisor {
2049 #[cfg(unix)]
2060 pub(crate) fn begin_daemon_shutdown(&self) {
2061 self.child_roster.close();
2062 if let Some(journal) = &self.terminal_journal {
2063 journal.stamp_shutdown();
2064 }
2065 }
2066
2067 #[cfg(unix)]
2071 pub(crate) async fn drain_for_daemon_shutdown(&self) -> Result<(), SuperviseError> {
2072 const NOTICE_BUDGET: Duration = Duration::from_millis(500);
2073 const DRAIN_BUDGET: Duration = Duration::from_secs(2);
2074 let Some(forwarding) = &self.forwarding else {
2075 return Ok(());
2076 };
2077 let module_ids = forwarding
2078 .begin_daemon_drain()
2079 .map_err(SuperviseError::Forwarding)?;
2080 let deadline_ms =
2081 unix_ms_now().saturating_add((NOTICE_BUDGET + DRAIN_BUDGET).as_millis() as u64);
2082 let mut notices = tokio::task::JoinSet::new();
2083 let mut drains = Vec::new();
2084 for module_id in module_ids {
2085 let Some(target) = forwarding
2086 .begin_module_drain(&module_id, RouteCloseReason::Restart)
2087 .map_err(SuperviseError::Forwarding)?
2088 else {
2089 continue;
2090 };
2091 let routes = forwarding
2092 .endpoint_routes(target.endpoint)
2093 .map_err(SuperviseError::Forwarding)?;
2094 let command = serde_json::to_vec(&ModuleControlCommand::Draining {
2100 reason: RouteCloseReason::Restart,
2101 deadline_ms,
2102 })
2103 .expect("module draining serializes");
2104 let closing = serde_json::to_vec(&ClientControlPush::RouteClosing {
2105 module_id: module_id.clone(),
2106 reason: RouteCloseReason::Restart,
2107 })
2108 .expect("route closing serializes");
2109 let mut recipients = vec![(target.sink.clone(), target.negotiated_ver, command)];
2110 let mut seen = std::collections::HashSet::new();
2111 for route in routes {
2112 let client = route.goodbye_target;
2113 if seen.insert(client.connection_id) {
2114 recipients.push((client.sink, client.negotiated_ver, closing.clone()));
2115 }
2116 }
2117 for (sink, version, body) in recipients {
2118 notices.spawn(async move {
2119 let frame = Frame::build_with_version(
2120 version,
2121 FrameType::Push,
2122 control_flags(),
2123 0,
2124 0,
2125 0,
2126 body,
2127 )
2128 .expect("bounded lifecycle notice frame builds");
2129 sink.send_flushed(frame).await
2130 });
2131 }
2132 let gauges = declared_busy_gauges(&self.registry, &module_id)?;
2133 drains.push((module_id, target.endpoint, gauges));
2134 }
2135 let notice_deadline = Instant::now() + NOTICE_BUDGET;
2138 while let Ok(Some(result)) = timeout_at(notice_deadline, notices.join_next()).await {
2139 if !matches!(result, Ok(Ok(()))) {
2140 warn!(?result, "daemon shutdown notice delivery failed");
2141 }
2142 }
2143 notices.abort_all();
2144 let deadline = Instant::now() + DRAIN_BUDGET;
2145 let mut waits = tokio::task::JoinSet::new();
2146 for (module_id, endpoint, gauges) in drains {
2147 let forwarding = Arc::clone(forwarding);
2148 let mut runtime = self.runtime_config();
2149 runtime.health.cadence = Duration::from_millis(100);
2150 waits.spawn(async move {
2151 wait_for_forwarding_quiescence(
2152 &forwarding,
2153 &module_id,
2154 &runtime,
2155 endpoint,
2156 deadline,
2157 &gauges,
2158 DrainScope::Active,
2159 )
2160 .await
2161 });
2162 }
2163 while let Ok(Some(result)) = timeout_at(deadline, waits.join_next()).await {
2164 if !matches!(result, Ok(Ok(true))) {
2165 warn!(?result, "daemon shutdown drain did not reach quiescence");
2166 }
2167 }
2168 Ok(())
2169 }
2170
2171 #[cfg(unix)]
2183 pub(crate) async fn end_children_for_daemon_shutdown(
2184 &self,
2185 already_escalated: bool,
2186 escalate: impl std::future::Future<Output = ()>,
2187 ) {
2188 tokio::pin!(escalate);
2189 let mut escalated = already_escalated;
2190 if let Some(forwarding) = &self.forwarding {
2191 let reason = CloseReason::new(
2192 "daemon_shutdown",
2193 "the daemon is exiting after its shutdown notice and drain",
2194 );
2195 if escalated {
2196 send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, false).await;
2199 } else {
2200 tokio::select! {
2201 biased;
2202 _ = escalate.as_mut() => {
2203 info!("second SIGTERM: abandoning module GOODBYE delivery");
2204 escalated = true;
2205 }
2206 _ = send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, true) => {}
2207 }
2208 }
2209 let closed = forwarding.close_all_connections(&reason);
2210 debug!(closed, "closed established connections for daemon shutdown");
2211 }
2212 let escalated_here = escalated && !already_escalated;
2216 let remaining_escalate = async move {
2217 if escalated_here {
2218 std::future::pending::<()>().await;
2219 } else {
2220 escalate.await;
2221 }
2222 };
2223 crate::child_roster::end_children_for_daemon_shutdown(
2224 &self.child_roster,
2225 escalated,
2226 remaining_escalate,
2227 )
2228 .await;
2229 }
2230
2231 pub fn new(registry: Arc<Registry>, restart_policy: RestartPolicy) -> Self {
2232 Self {
2233 registry,
2234 restart_policy,
2235 drain_timeout: DEFAULT_DRAIN_TIMEOUT,
2236 connection_file_path: None,
2237 capture_logs_dir: None,
2238 forwarding: None,
2239 process_liveness: Arc::new(SupervisorProcessLiveness::default()),
2240 supervisor_handle: None,
2241 health: HealthConfig::default(),
2242 daemon_start_clock: crate::clock::StartClock::capture(),
2243 terminal_journal: None,
2244 spawn_events: SpawnEventFeed::default(),
2245 provenance_probe: ExecutableIdentityProbe::default(),
2246 child_roster: ChildRoster::default(),
2247 #[cfg(target_os = "linux")]
2248 cgroup_placement: None,
2249 }
2250 }
2251
2252 pub fn with_drain_timeout(mut self, drain_timeout: Duration) -> Self {
2253 self.drain_timeout = drain_timeout;
2254 self
2255 }
2256
2257 pub fn with_process_liveness(
2258 mut self,
2259 process_liveness: Arc<SupervisorProcessLiveness>,
2260 ) -> Self {
2261 self.process_liveness = process_liveness;
2262 self
2263 }
2264
2265 pub fn with_connection_file_path(mut self, connection_file_path: impl Into<PathBuf>) -> Self {
2266 self.connection_file_path = Some(connection_file_path.into());
2267 self
2268 }
2269
2270 pub fn with_capture_logs_dir(mut self, logs_dir: impl Into<PathBuf>) -> Self {
2272 self.capture_logs_dir = Some(logs_dir.into());
2273 self
2274 }
2275
2276 pub fn with_daemon_incarnation(self, daemon_incarnation: String) -> Self {
2279 self.spawn_events.configure_incarnation(daemon_incarnation);
2283 self
2284 }
2285
2286 pub fn with_terminal_journal(self, path: PathBuf, daemon_incarnation: String) -> Self {
2289 let mut this = self.with_daemon_incarnation(daemon_incarnation.clone());
2290 this.terminal_journal = Some(Arc::new(crate::terminal_journal::TerminalJournal::open(
2291 path,
2292 daemon_incarnation,
2293 )));
2294 this
2295 }
2296
2297 pub fn with_forwarding(mut self, forwarding: Arc<ForwardingTable>) -> Self {
2298 self.forwarding = Some(forwarding);
2299 self
2300 }
2301
2302 pub fn with_handle(mut self, supervisor_handle: SupervisorHandle) -> Self {
2303 self.spawn_events = supervisor_handle.spawn_events.clone();
2304 self.supervisor_handle = Some(supervisor_handle);
2305 self
2306 }
2307
2308 pub fn with_health_config(mut self, health: HealthConfig) -> Self {
2309 self.health = health;
2310 self
2311 }
2312
2313 pub fn with_live_children_record(self, path: impl Into<PathBuf>) -> Self {
2317 self.child_roster.record_to(path.into());
2318 self
2319 }
2320
2321 #[cfg(target_os = "linux")]
2322 pub fn with_cgroup_placement(
2323 mut self,
2324 cgroup_placement: Option<subc_cgroup::Placement>,
2325 ) -> Self {
2326 self.cgroup_placement = cgroup_placement;
2327 self
2328 }
2329
2330 pub fn spawn(&self, spec: ModuleSpec) -> Result<SupervisedModule, SuperviseError> {
2336 validate_spec(&spec)?;
2337
2338 let runtime = self.runtime_config();
2339 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2340 let child = spawn_child(
2341 &spec,
2342 runtime.connection_file_path.as_deref(),
2343 self.supervisor_handle.as_ref(),
2344 &runtime.stderr_ring,
2345 runtime.capture_logs_dir.as_deref(),
2346 &runtime.child_roster,
2347 #[cfg(target_os = "linux")]
2348 runtime.cgroup_placement.as_ref(),
2349 )?;
2350 set_running(&snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
2351 self.process_liveness
2352 .track(spec.module_id.clone(), Arc::clone(&snapshot));
2353
2354 Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2355 }
2356
2357 pub fn supervise_configured(
2363 &self,
2364 spec: ModuleSpec,
2365 enabled: bool,
2366 ) -> Result<SupervisedModule, SuperviseError> {
2367 validate_spec(&spec)?;
2368
2369 let runtime = self.runtime_config();
2370 if !enabled {
2371 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
2372 return Ok(self.supervised_module(spec, runtime, snapshot, None));
2373 }
2374
2375 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2376 match spawn_child(
2377 &spec,
2378 runtime.connection_file_path.as_deref(),
2379 self.supervisor_handle.as_ref(),
2380 &runtime.stderr_ring,
2381 runtime.capture_logs_dir.as_deref(),
2382 &runtime.child_roster,
2383 #[cfg(target_os = "linux")]
2384 runtime.cgroup_placement.as_ref(),
2385 ) {
2386 Ok(child) => {
2387 set_running(&snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
2388 self.process_liveness
2389 .track(spec.module_id.clone(), Arc::clone(&snapshot));
2390 Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2391 }
2392 Err(err) => {
2393 error!(
2394 module_id = %spec.module_id,
2395 program = %spec.program.display(),
2396 error = %err,
2397 "configured module failed to spawn; marking failed and continuing"
2398 );
2399 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
2400 Ok(self.supervised_module(spec, runtime, snapshot, None))
2401 }
2402 }
2403 }
2404
2405 pub fn supervise_configured_with_health(
2411 &self,
2412 spec: ModuleSpec,
2413 enabled: bool,
2414 health: HealthConfig,
2415 drain_timeout_ms: Option<u64>,
2416 restart_policy: RestartPolicy,
2417 ) -> Result<SupervisedModule, SuperviseError> {
2418 validate_spec(&spec)?;
2419
2420 let mut runtime = self.runtime_config();
2421 runtime.health = health;
2422 runtime.restart_policy = restart_policy;
2423 if let Some(ms) = drain_timeout_ms {
2424 runtime.drain_timeout = Duration::from_millis(ms);
2425 *runtime
2426 .effective_drain_timeout
2427 .lock()
2428 .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
2429 }
2430 if !enabled {
2431 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
2432 return Ok(self.supervised_module(spec, runtime, snapshot, None));
2433 }
2434
2435 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2436 match spawn_child(
2437 &spec,
2438 runtime.connection_file_path.as_deref(),
2439 self.supervisor_handle.as_ref(),
2440 &runtime.stderr_ring,
2441 runtime.capture_logs_dir.as_deref(),
2442 &runtime.child_roster,
2443 #[cfg(target_os = "linux")]
2444 runtime.cgroup_placement.as_ref(),
2445 ) {
2446 Ok(child) => {
2447 set_running(&snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
2448 self.process_liveness
2449 .track(spec.module_id.clone(), Arc::clone(&snapshot));
2450 Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2451 }
2452 Err(err) => {
2453 if health.critical {
2454 error!(
2455 module_id = %spec.module_id,
2456 program = %spec.program.display(),
2457 error = %err,
2458 "critical configured module failed to spawn; marking failed and alerting"
2459 );
2460 } else {
2461 error!(
2462 module_id = %spec.module_id,
2463 program = %spec.program.display(),
2464 error = %err,
2465 "configured module failed to spawn; marking failed and continuing"
2466 );
2467 }
2468 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
2469 Ok(self.supervised_module(spec, runtime, snapshot, None))
2470 }
2471 }
2472 }
2473
2474 fn runtime_config(&self) -> SupervisorRuntimeConfig {
2475 let effective_drain_timeout = Arc::new(Mutex::new(self.drain_timeout));
2476 SupervisorRuntimeConfig {
2477 restart_policy: self.restart_policy,
2478 drain_timeout: self.drain_timeout,
2479 child_roster: self
2482 .child_roster
2483 .for_module(Arc::clone(&effective_drain_timeout)),
2484 effective_drain_timeout,
2485 default_drain_timeout: self.drain_timeout,
2486 health: self.health,
2487 connection_file_path: self.connection_file_path.clone(),
2488 capture_logs_dir: self.capture_logs_dir.clone(),
2489 forwarding: self.forwarding.clone(),
2490 supervisor_handle: self.supervisor_handle.clone(),
2491 stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
2492 terminal_ring: Arc::new(Mutex::new(
2493 TerminalRing::new(
2494 TerminalRingConfig::default(),
2495 self.daemon_start_clock.started_at_ms(),
2496 )
2497 .with_start_clock(self.daemon_start_clock)
2498 .with_journal(self.terminal_journal.clone())
2499 .with_daemon_shutdown(self.child_roster.shutdown_flag()),
2500 )),
2501 spawn_events: self.spawn_events.clone(),
2502 #[cfg(target_os = "linux")]
2503 cgroup_placement: self.cgroup_placement.clone(),
2504 #[cfg(test)]
2505 test_seed_stale_facts_before_enable_spawn: false,
2506 }
2507 }
2508
2509 fn supervised_module(
2510 &self,
2511 spec: ModuleSpec,
2512 runtime: SupervisorRuntimeConfig,
2513 snapshot: SharedSnapshot,
2514 child: Option<SupervisedChild>,
2515 ) -> SupervisedModule {
2516 let configuration = Arc::new(Mutex::new(SupervisedConfiguration {
2517 spec: spec.clone(),
2518 health: runtime.health,
2519 }));
2520 let stderr_ring = Arc::clone(&runtime.stderr_ring);
2521 let terminal_ring = Arc::clone(&runtime.terminal_ring);
2522 let restart_policy = runtime.restart_policy;
2526 let effective_drain_timeout = Arc::clone(&runtime.effective_drain_timeout);
2527 let (tx, rx) = mpsc::channel(4);
2528 let monitor = tokio::spawn(supervise_loop(
2529 spec.clone(),
2530 runtime,
2531 Arc::clone(&self.registry),
2532 Arc::clone(&self.process_liveness),
2533 Arc::clone(&snapshot),
2534 child,
2535 rx,
2536 ));
2537
2538 let module_id = spec.module_id.clone();
2539 let module = SupervisedModule {
2540 inner: Arc::new(SupervisedModuleInner {
2541 module_id: module_id.clone(),
2542 registry: Arc::clone(&self.registry),
2543 snapshot,
2544 configuration,
2545 stderr_ring,
2546 terminal_ring,
2547 commands: tx,
2548 monitor: Mutex::new(Some(monitor)),
2549 restart_policy,
2550 effective_drain_timeout,
2551 provenance_probe: self.provenance_probe.clone(),
2552 }),
2553 };
2554 if let Some(supervisor_handle) = &self.supervisor_handle {
2555 supervisor_handle.apply_identity_configuration(&spec);
2556 supervisor_handle.insert(module.clone());
2557 }
2558 module
2559 }
2560}
2561
2562impl Default for Supervisor {
2563 fn default() -> Self {
2564 Self::new(Arc::new(Registry::default()), RestartPolicy::default())
2565 }
2566}
2567
2568#[derive(Clone)]
2570pub struct SupervisedModule {
2571 inner: Arc<SupervisedModuleInner>,
2572}
2573
2574struct SupervisedModuleInner {
2575 module_id: String,
2576 registry: Arc<Registry>,
2577 snapshot: SharedSnapshot,
2578 configuration: Arc<Mutex<SupervisedConfiguration>>,
2579 stderr_ring: Arc<Mutex<StderrRing>>,
2580 terminal_ring: Arc<Mutex<TerminalRing>>,
2581 commands: mpsc::Sender<SupervisorCommand>,
2582 monitor: Mutex<Option<JoinHandle<()>>>,
2583 restart_policy: RestartPolicy,
2587 effective_drain_timeout: Arc<Mutex<Duration>>,
2588 provenance_probe: ExecutableIdentityProbe,
2589}
2590
2591impl fmt::Debug for SupervisedModule {
2592 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
2593 f.debug_struct("SupervisedModule")
2594 .field("module_id", &self.inner.module_id)
2595 .field("status", &self.status())
2596 .finish_non_exhaustive()
2597 }
2598}
2599
2600impl SupervisedModule {
2601 pub fn module_id(&self) -> &str {
2602 &self.inner.module_id
2603 }
2604
2605 #[cfg(test)]
2609 pub(crate) fn record_health_probe_failure_for_test(
2610 &self,
2611 detail: &str,
2612 ) -> Result<(), SuperviseError> {
2613 update_snapshot(&self.inner.snapshot, Some(&self.inner.module_id), |state| {
2614 state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
2615 state.health.detail = Some(detail.to_string());
2616 })
2617 }
2618
2619 pub fn state(&self) -> Result<ModuleState, SuperviseError> {
2620 Ok(lock_snapshot(&self.inner.snapshot)?.state)
2621 }
2622
2623 pub fn stderr_tail(
2630 &self,
2631 max_lines: Option<usize>,
2632 max_bytes: Option<usize>,
2633 ) -> StderrTailSnapshot {
2634 self.inner
2635 .stderr_ring
2636 .lock()
2637 .unwrap_or_else(|poisoned| poisoned.into_inner())
2638 .snapshot(max_lines, max_bytes)
2639 }
2640
2641 pub fn terminal_history(&self) -> TerminalHistorySnapshot {
2646 self.inner
2647 .terminal_ring
2648 .lock()
2649 .unwrap_or_else(|poisoned| poisoned.into_inner())
2650 .snapshot()
2651 }
2652
2653 pub fn durable_terminal_history(&self) -> subc_control::TerminalHistory {
2658 durable_terminal_history_of(&self.inner.terminal_ring, &self.inner.module_id)
2659 }
2660
2661 pub(crate) async fn read_durable_terminal_history(
2666 &self,
2667 ) -> Result<subc_control::TerminalHistory, tokio::task::JoinError> {
2668 let terminal_ring = Arc::clone(&self.inner.terminal_ring);
2669 let module_id = self.inner.module_id.clone();
2670 tokio::task::spawn_blocking(move || durable_terminal_history_of(&terminal_ring, &module_id))
2671 .await
2672 }
2673
2674 pub fn status(&self) -> Result<ModuleStatus, SuperviseError> {
2675 self.status_with_snapshot_lock(&self.inner.snapshot, None)
2676 }
2677
2678 pub(crate) fn record_deliberate_severance(
2679 &self,
2680 identity: ProcessIdentity,
2681 ) -> Result<bool, SuperviseError> {
2682 let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
2683 if snapshot.pid != Some(identity.pid)
2684 || snapshot.process_start_time != Some(identity.start_time)
2685 {
2686 return Ok(false);
2687 }
2688 snapshot.deliberate_severance = Some(identity);
2689 Ok(true)
2690 }
2691
2692 pub(crate) fn status_for_control(
2697 &self,
2698 caller: &'static str,
2699 ) -> Result<ModuleStatus, SuperviseError> {
2700 self.status_with_snapshot_lock(&self.inner.snapshot, Some(caller))
2701 }
2702
2703 fn status_with_snapshot_lock(
2704 &self,
2705 snapshot: &SharedSnapshot,
2706 caller: Option<&'static str>,
2707 ) -> Result<ModuleStatus, SuperviseError> {
2708 let mut guard = match caller {
2709 Some(caller) => lock_snapshot_for_control(snapshot, &self.inner.module_id, caller)?,
2710 None => lock_snapshot(snapshot)?,
2711 };
2712 let restart_count =
2715 guard.crash_restarts_in_window(self.inner.restart_policy.window, Instant::now());
2716 let snapshot = guard.clone();
2717 drop(guard);
2718 let drain_timeout = *self.inner.effective_drain_timeout.lock().map_err(|_| {
2719 SuperviseError::StatePoisoned {
2720 module_id: Some(self.inner.module_id.clone()),
2721 }
2722 })?;
2723 let registration_active = self
2724 .inner
2725 .registry
2726 .get_module(&self.inner.module_id)
2727 .map_err(SuperviseError::Registry)?
2728 .is_some();
2729 let protocol = self.declared_protocol()?;
2730 let running_process =
2731 snapshot.enabled && snapshot.state == ModuleState::Running && snapshot.process_alive;
2732 let live = match protocol {
2738 ModuleProtocol::Subc => running_process && registration_active,
2739 ModuleProtocol::None => running_process,
2740 };
2741
2742 Ok(ModuleStatus {
2743 module_id: self.inner.module_id.clone(),
2744 state: snapshot.state,
2745 enabled: snapshot.enabled,
2746 process_alive: snapshot.process_alive,
2747 registration_active,
2748 protocol,
2749 live,
2750 restart_count,
2751 lifetime_restarts: snapshot.lifetime_restarts,
2752 spawn_generation: snapshot.spawn_generation,
2753 max_restarts: self.inner.restart_policy.max_restarts,
2754 restart_window: self.inner.restart_policy.window,
2755 drain_timeout,
2756 restart_backoff: self.inner.restart_policy.backoff,
2757 restart_max_backoff: self.inner.restart_policy.max_backoff,
2758 pid: snapshot.pid,
2759 spawned_at_ms: snapshot.spawned_at_ms,
2760 spawned_from: snapshot.spawned_from,
2761 process_start_time: snapshot.process_start_time,
2762 last_exit: snapshot.last_exit,
2763 health: snapshot.health,
2764 })
2765 }
2766
2767 #[cfg(test)]
2768 pub(crate) fn hold_snapshot_for_test(
2769 &self,
2770 acquired: std::sync::mpsc::Sender<()>,
2771 hold: Duration,
2772 ) -> std::thread::JoinHandle<()> {
2773 let snapshot = Arc::clone(&self.inner.snapshot);
2774 std::thread::spawn(move || {
2775 let _guard = snapshot.lock().expect("test snapshot lock is not poisoned");
2776 acquired
2777 .send(())
2778 .expect("test receiver waits for snapshot lock");
2779 std::thread::sleep(hold);
2780 })
2781 }
2782
2783 pub(crate) async fn running_image_agreement(&self) -> subc_control::RunningImageAgreement {
2784 let snapshot = match lock_snapshot(&self.inner.snapshot) {
2785 Ok(snapshot) => snapshot.clone(),
2786 Err(_) => {
2787 return subc_control::RunningImageAgreement::Unavailable {
2788 reason: subc_control::RunningImageUnavailableReason::NotRunning,
2789 };
2790 }
2791 };
2792 self.inner
2793 .provenance_probe
2794 .observe(
2795 snapshot.pid,
2796 snapshot.spawned_from.as_deref(),
2797 snapshot.spawned_file_identity,
2798 snapshot.process_start_time,
2799 )
2800 .await
2801 }
2802
2803 pub(crate) fn child_resource_usage(&self) -> subc_control::ChildResourceUsage {
2806 let (pid, start_time) = match lock_snapshot(&self.inner.snapshot) {
2807 Ok(snapshot) => (snapshot.pid, snapshot.process_start_time),
2808 Err(_) => {
2809 return subc_control::ChildResourceUsage::Unavailable {
2810 reason: subc_control::ChildResourceUnavailableReason::Unreadable,
2811 }
2812 }
2813 };
2814 crate::child_resources::read(pid, start_time)
2815 }
2816
2817 pub(crate) fn will_recover_after_connection_loss(&self) -> Result<bool, SuperviseError> {
2818 let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
2819 Ok(match snapshot.state {
2820 ModuleState::Restarting => true,
2821 ModuleState::Failed | ModuleState::Disabled => false,
2822 _ => daemon_will_restart(&mut snapshot, &self.inner.restart_policy, Instant::now()),
2823 })
2824 }
2825
2826 #[cfg(test)]
2827 pub(crate) fn is_warming(&self) -> Result<bool, SuperviseError> {
2828 self.is_warming_with_snapshot_lock(None)
2829 }
2830
2831 pub(crate) fn is_warming_for_control(
2832 &self,
2833 caller: &'static str,
2834 ) -> Result<bool, SuperviseError> {
2835 self.is_warming_with_snapshot_lock(Some(caller))
2836 }
2837
2838 fn is_warming_with_snapshot_lock(
2839 &self,
2840 caller: Option<&'static str>,
2841 ) -> Result<bool, SuperviseError> {
2842 let snapshot = match caller {
2843 Some(caller) => {
2844 lock_snapshot_for_control(&self.inner.snapshot, &self.inner.module_id, caller)?
2845 }
2846 None => lock_snapshot(&self.inner.snapshot)?,
2847 }
2848 .clone();
2849 Ok(matches!(
2850 snapshot.state,
2851 ModuleState::Starting | ModuleState::Running | ModuleState::Restarting
2852 ))
2853 }
2854
2855 pub async fn drain(&self) -> Result<(), SuperviseError> {
2857 self.stop().await
2858 }
2859
2860 pub(crate) async fn retire(&self) -> Result<(), SuperviseError> {
2861 match self.state()? {
2862 ModuleState::Stopped | ModuleState::Failed => return Ok(()),
2863 ModuleState::Starting
2864 | ModuleState::Running
2865 | ModuleState::Unresponsive
2866 | ModuleState::Restarting
2867 | ModuleState::Draining
2868 | ModuleState::Disabled => {}
2869 }
2870
2871 let (reply_tx, reply_rx) = oneshot::channel();
2872 self.inner
2873 .commands
2874 .send(SupervisorCommand::Retire { reply: reply_tx })
2875 .await
2876 .map_err(|_| SuperviseError::CommandClosed {
2877 module_id: self.inner.module_id.clone(),
2878 })?;
2879 reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
2880 module_id: self.inner.module_id.clone(),
2881 })?
2882 }
2883
2884 pub async fn stop(&self) -> Result<(), SuperviseError> {
2885 match self.state()? {
2886 ModuleState::Stopped | ModuleState::Failed => return Ok(()),
2887 ModuleState::Starting
2888 | ModuleState::Running
2889 | ModuleState::Unresponsive
2890 | ModuleState::Restarting
2891 | ModuleState::Draining
2892 | ModuleState::Disabled => {}
2893 }
2894
2895 let (reply_tx, reply_rx) = oneshot::channel();
2896 self.inner
2897 .commands
2898 .send(SupervisorCommand::Drain { reply: reply_tx })
2899 .await
2900 .map_err(|_| SuperviseError::CommandClosed {
2901 module_id: self.inner.module_id.clone(),
2902 })?;
2903 reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
2904 module_id: self.inner.module_id.clone(),
2905 })?
2906 }
2907
2908 pub async fn restart(&self, drain_timeout_ms: Option<u64>) -> Result<(), SuperviseError> {
2909 let received_at_generation = lock_snapshot(&self.inner.snapshot)?.spawn_generation;
2910 let (reply_tx, reply_rx) = oneshot::channel();
2911 self.inner
2912 .commands
2913 .send(SupervisorCommand::Restart {
2914 drain_timeout_ms,
2915 received_at_generation,
2916 queued_at: Instant::now(),
2917 reply: reply_tx,
2918 })
2919 .await
2920 .map_err(|_| SuperviseError::CommandClosed {
2921 module_id: self.inner.module_id.clone(),
2922 })?;
2923 reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
2924 module_id: self.inner.module_id.clone(),
2925 })?
2926 }
2927
2928 pub async fn swap(&self, ready_timeout: Option<Duration>) -> Result<(), SuperviseError> {
2933 let (reply_tx, reply_rx) = oneshot::channel();
2934 self.inner
2935 .commands
2936 .send(SupervisorCommand::Swap {
2937 ready_timeout,
2938 reply: reply_tx,
2939 })
2940 .await
2941 .map_err(|_| SuperviseError::CommandClosed {
2942 module_id: self.inner.module_id.clone(),
2943 })?;
2944 reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
2945 module_id: self.inner.module_id.clone(),
2946 })?
2947 }
2948
2949 pub async fn reload(&self) -> Result<(), SuperviseError> {
2950 let (reply_tx, reply_rx) = oneshot::channel();
2951 self.inner
2952 .commands
2953 .send(SupervisorCommand::Reload { reply: reply_tx })
2954 .await
2955 .map_err(|_| SuperviseError::CommandClosed {
2956 module_id: self.inner.module_id.clone(),
2957 })?;
2958 reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
2959 module_id: self.inner.module_id.clone(),
2960 })?
2961 }
2962
2963 pub async fn set_enabled(&self, enabled: bool) -> Result<bool, SuperviseError> {
2964 let (reply_tx, reply_rx) = oneshot::channel();
2965 self.inner
2966 .commands
2967 .send(SupervisorCommand::SetEnabled {
2968 enabled,
2969 reply: reply_tx,
2970 })
2971 .await
2972 .map_err(|_| SuperviseError::CommandClosed {
2973 module_id: self.inner.module_id.clone(),
2974 })?;
2975 reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
2976 module_id: self.inner.module_id.clone(),
2977 })?
2978 }
2979
2980 pub(crate) fn declared_protocol(&self) -> Result<ModuleProtocol, SuperviseError> {
2985 Ok(self
2986 .inner
2987 .configuration
2988 .lock()
2989 .map_err(|_| SuperviseError::StatePoisoned {
2990 module_id: Some(self.inner.module_id.clone()),
2991 })?
2992 .spec
2993 .protocol)
2994 }
2995
2996 pub(crate) fn configuration(&self) -> Result<(ModuleSpec, HealthConfig), SuperviseError> {
2997 let configuration =
2998 self.inner
2999 .configuration
3000 .lock()
3001 .map_err(|_| SuperviseError::StatePoisoned {
3002 module_id: Some(self.inner.module_id.clone()),
3003 })?;
3004 Ok((configuration.spec.clone(), configuration.health))
3005 }
3006
3007 #[cfg(any(test, feature = "test-support"))]
3011 pub async fn update_spec_for_test(&self, spec: ModuleSpec) -> Result<(), SuperviseError> {
3012 let (_, health) = self.configuration()?;
3013 let drain_timeout_ms = u64::try_from(
3014 self.inner
3015 .effective_drain_timeout
3016 .lock()
3017 .unwrap_or_else(|poisoned| poisoned.into_inner())
3018 .as_millis(),
3019 )
3020 .ok();
3021 self.update_configuration(spec, health, drain_timeout_ms)
3022 .await
3023 }
3024
3025 pub(crate) async fn update_configuration(
3026 &self,
3027 spec: ModuleSpec,
3028 health: HealthConfig,
3029 drain_timeout_ms: Option<u64>,
3030 ) -> Result<(), SuperviseError> {
3031 if spec.module_id != self.inner.module_id {
3032 return Err(SuperviseError::InvalidSpec {
3033 reason: "a supervised module's module_id cannot be changed".to_string(),
3034 });
3035 }
3036 validate_spec(&spec)?;
3037 let (reply_tx, reply_rx) = oneshot::channel();
3038 self.inner
3039 .commands
3040 .send(SupervisorCommand::UpdateConfiguration {
3041 spec: spec.clone(),
3042 health,
3043 drain_timeout_ms,
3044 reply: reply_tx,
3045 })
3046 .await
3047 .map_err(|_| SuperviseError::CommandClosed {
3048 module_id: self.inner.module_id.clone(),
3049 })?;
3050 reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3051 module_id: self.inner.module_id.clone(),
3052 })?;
3053 let mut configuration =
3054 self.inner
3055 .configuration
3056 .lock()
3057 .map_err(|_| SuperviseError::StatePoisoned {
3058 module_id: Some(self.inner.module_id.clone()),
3059 })?;
3060 configuration.spec = spec;
3061 configuration.health = health;
3062 Ok(())
3063 }
3064}
3065
3066impl Drop for SupervisedModuleInner {
3067 fn drop(&mut self) {
3068 let Ok(mut monitor) = self.monitor.lock() else {
3069 return;
3070 };
3071 if let Some(monitor) = monitor.as_ref().filter(|monitor| !monitor.is_finished()) {
3072 let _ = update_snapshot(&self.snapshot, Some(&self.module_id), |state| {
3073 state.state = ModuleState::Stopped;
3074 clear_current_process_facts(state);
3075 });
3076 monitor.abort();
3077 }
3078 let _ = monitor.take();
3079 }
3080}
3081
3082#[derive(Debug)]
3083enum SupervisorCommand {
3084 Drain {
3085 reply: oneshot::Sender<Result<(), SuperviseError>>,
3086 },
3087 Retire {
3088 reply: oneshot::Sender<Result<(), SuperviseError>>,
3089 },
3090 Restart {
3091 drain_timeout_ms: Option<u64>,
3096 received_at_generation: u64,
3100 queued_at: Instant,
3103 reply: oneshot::Sender<Result<(), SuperviseError>>,
3104 },
3105 Reload {
3106 reply: oneshot::Sender<Result<(), SuperviseError>>,
3107 },
3108 SetEnabled {
3109 enabled: bool,
3110 reply: oneshot::Sender<Result<bool, SuperviseError>>,
3111 },
3112 UpdateConfiguration {
3113 spec: ModuleSpec,
3114 health: HealthConfig,
3115 drain_timeout_ms: Option<u64>,
3118 reply: oneshot::Sender<()>,
3119 },
3120 Swap {
3121 ready_timeout: Option<Duration>,
3124 reply: oneshot::Sender<Result<(), SuperviseError>>,
3126 },
3127}
3128
3129#[derive(Debug)]
3130pub enum SuperviseError {
3131 InvalidSpec {
3132 reason: String,
3133 },
3134 Spawn {
3135 program: PathBuf,
3136 source: io::Error,
3137 cgroup_path: Option<PathBuf>,
3138 },
3139 Cgroup {
3140 module_id: String,
3141 source: io::Error,
3142 },
3143 LaunchNonce {
3146 reason: String,
3147 },
3148 Wait {
3149 module_id: String,
3150 source: io::Error,
3151 },
3152 Kill {
3153 module_id: String,
3154 source: io::Error,
3155 },
3156 Forwarding(ForwardingError),
3157 Registry(RegistryError),
3158 ReloadUnavailable {
3159 module_id: String,
3160 reason: String,
3161 },
3162 Disabled {
3167 module_id: String,
3168 },
3169 ReloadFailed {
3170 module_id: String,
3171 reason: String,
3172 },
3173 RegistrationStillActive {
3174 module_id: String,
3175 waited: Duration,
3176 },
3177 StatePoisoned {
3178 module_id: Option<String>,
3179 },
3180 CommandClosed {
3181 module_id: String,
3182 },
3183 SwapInProgress {
3187 module_id: String,
3188 },
3189 SwapRefused {
3191 module_id: String,
3192 reason: SwapRefusal,
3193 },
3194 SwapFailed {
3198 module_id: String,
3199 arm: SwapFailureArm,
3200 detail: String,
3201 candidate_exit: Option<ExitReport>,
3204 },
3205}
3206
3207#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3209pub enum SwapRefusal {
3210 OverlapExclusive,
3212 NotRegistered,
3215 ProtocolNone,
3218 NotConfigured,
3221 AlreadySwapping,
3223}
3224
3225impl SwapRefusal {
3226 pub fn as_str(self) -> &'static str {
3227 match self {
3228 Self::OverlapExclusive => "overlap_exclusive",
3229 Self::NotRegistered => "not_registered",
3230 Self::ProtocolNone => "protocol_none",
3231 Self::NotConfigured => "not_configured",
3232 Self::AlreadySwapping => "already_swapping",
3233 }
3234 }
3235}
3236
3237#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3240pub enum SwapFailureArm {
3241 SpawnFailed,
3243 NeverRegistered,
3245 NeverReady,
3247 CandidateExited,
3249 CandidateUnhealthy,
3251 Interrupted,
3255 CutoverLost,
3260}
3261
3262impl SwapFailureArm {
3263 pub fn as_str(self) -> &'static str {
3264 match self {
3265 Self::SpawnFailed => "spawn_failed",
3266 Self::NeverRegistered => "never_registered",
3267 Self::NeverReady => "never_ready",
3268 Self::CandidateExited => "candidate_exited",
3269 Self::CandidateUnhealthy => "candidate_unhealthy",
3270 Self::Interrupted => "interrupted",
3271 Self::CutoverLost => "cutover_lost",
3272 }
3273 }
3274}
3275
3276impl fmt::Display for SuperviseError {
3277 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3278 match self {
3279 Self::InvalidSpec { reason } => write!(f, "invalid module spec: {reason}"),
3280 Self::Spawn {
3281 program,
3282 source,
3283 cgroup_path: Some(cgroup_path),
3284 } => write!(
3285 f,
3286 "failed to place module in cgroup '{}' while spawning '{}': {source}",
3287 cgroup_path.display(),
3288 program.display()
3289 ),
3290 Self::Spawn {
3291 program,
3292 source,
3293 cgroup_path: None,
3294 } => write!(
3295 f,
3296 "failed to spawn module '{}': {source}",
3297 program.display()
3298 ),
3299 Self::Cgroup { module_id, source } => {
3300 write!(
3301 f,
3302 "failed to prepare cgroup for module '{module_id}': {source}"
3303 )
3304 }
3305 Self::LaunchNonce { reason } => {
3306 write!(
3307 f,
3308 "failed to generate reserved-module launch nonce: {reason}"
3309 )
3310 }
3311 Self::Wait { module_id, source } => {
3312 write!(f, "failed to wait for module '{module_id}': {source}")
3313 }
3314 Self::Kill { module_id, source } => {
3315 write!(f, "failed to kill module '{module_id}': {source}")
3316 }
3317 Self::Forwarding(err) => write!(f, "forwarding error: {err}"),
3318 Self::Registry(err) => write!(f, "registry error: {err}"),
3319 Self::ReloadUnavailable { module_id, reason } => {
3320 write!(f, "reload unavailable for module '{module_id}': {reason}")
3321 }
3322 Self::Disabled { module_id } => {
3323 write!(
3324 f,
3325 "module '{module_id}' is disabled; enable it before restart or reload"
3326 )
3327 }
3328 Self::ReloadFailed { module_id, reason } => {
3329 write!(f, "reload failed for module '{module_id}': {reason}")
3330 }
3331 Self::RegistrationStillActive { module_id, waited } => write!(
3332 f,
3333 "module '{module_id}' registration remained active after waiting {waited:?}"
3334 ),
3335 Self::StatePoisoned { module_id } => match module_id {
3336 Some(module_id) => {
3337 write!(f, "supervisor state for module '{module_id}' was poisoned")
3338 }
3339 None => write!(f, "supervisor state was poisoned"),
3340 },
3341 Self::CommandClosed { module_id } => {
3342 write!(
3343 f,
3344 "supervisor command channel for module '{module_id}' is closed"
3345 )
3346 }
3347 Self::SwapInProgress { module_id } => write!(
3348 f,
3349 "module '{module_id}' is being swapped; retry once the swap has cut over or failed, or stop the module to abort the swap"
3350 ),
3351 Self::SwapRefused { module_id, reason } => match reason {
3352 SwapRefusal::OverlapExclusive => write!(
3353 f,
3354 "module '{module_id}' is declared overlap: \"exclusive\" (the default): two processes of it must not run at once, so it cannot be swapped; use a plain restart, or declare overlap: \"safe\" in its config if it really tolerates a second process"
3355 ),
3356 SwapRefusal::NotRegistered => write!(
3357 f,
3358 "module '{module_id}' is not registered, so there is no serving process to keep while a replacement warms; use a plain restart"
3359 ),
3360 SwapRefusal::ProtocolNone => write!(
3361 f,
3362 "module '{module_id}' is protocol: \"none\" and never registers, so a swap could never see its replacement become ready; use a plain restart"
3363 ),
3364 SwapRefusal::NotConfigured => write!(
3365 f,
3366 "module '{module_id}' cannot be swapped: the supervisor was built without the forwarding table or shared handle a swap needs"
3367 ),
3368 SwapRefusal::AlreadySwapping => {
3369 write!(f, "module '{module_id}' is already being swapped")
3370 }
3371 },
3372 Self::SwapFailed {
3373 module_id,
3374 arm,
3375 detail,
3376 ..
3377 } => write!(
3378 f,
3379 "swap of module '{module_id}' failed ({}): {detail}; the running process was left serving",
3380 arm.as_str()
3381 ),
3382 }
3383 }
3384}
3385
3386impl Error for SuperviseError {
3387 fn source(&self) -> Option<&(dyn Error + 'static)> {
3388 match self {
3389 Self::Spawn { source, .. }
3390 | Self::Cgroup { source, .. }
3391 | Self::Wait { source, .. }
3392 | Self::Kill { source, .. } => Some(source),
3393 Self::Forwarding(err) => Some(err),
3394 Self::Registry(err) => Some(err),
3395 Self::LaunchNonce { .. }
3396 | Self::InvalidSpec { .. }
3397 | Self::ReloadUnavailable { .. }
3398 | Self::Disabled { .. }
3399 | Self::ReloadFailed { .. }
3400 | Self::RegistrationStillActive { .. }
3401 | Self::StatePoisoned { .. }
3402 | Self::CommandClosed { .. }
3403 | Self::SwapInProgress { .. }
3404 | Self::SwapRefused { .. }
3405 | Self::SwapFailed { .. } => None,
3406 }
3407 }
3408}
3409
3410pub(crate) fn validate_spec(spec: &ModuleSpec) -> Result<(), SuperviseError> {
3411 if spec.module_id.trim().is_empty() {
3412 return Err(SuperviseError::InvalidSpec {
3413 reason: "module_id must not be empty".to_string(),
3414 });
3415 }
3416
3417 Ok(())
3418}
3419
3420#[derive(Debug, Default)]
3421struct HealthProbeRuntime {
3422 registered_connection: Option<crate::ConnectionId>,
3423 advertised: bool,
3424 next_probe_at: Option<Instant>,
3425 probe_index: u64,
3426}
3427
3428impl HealthProbeRuntime {
3429 fn refresh_registration(
3430 &mut self,
3431 spec: &ModuleSpec,
3432 runtime: &SupervisorRuntimeConfig,
3433 registry: &Registry,
3434 snapshot: &SharedSnapshot,
3435 ) {
3436 if spec.protocol == ModuleProtocol::None {
3448 self.registered_connection = None;
3449 self.advertised = false;
3450 self.next_probe_at = None;
3451 return;
3452 }
3453
3454 let registration = match registry.get_module(&spec.module_id) {
3455 Ok(registration) => registration,
3456 Err(err) => {
3457 warn!(module_id = %spec.module_id, error = %err, "health prober could not read registry");
3458 self.advertised = false;
3459 self.next_probe_at = None;
3460 return;
3461 }
3462 };
3463
3464 let Some(registration) = registration else {
3465 self.registered_connection = None;
3466 self.advertised = false;
3467 self.next_probe_at = None;
3468 return;
3469 };
3470
3471 let advertised = registration
3472 .control_ops
3473 .iter()
3474 .any(|op| op == MODULE_CONTROL_OP_HEALTH_CHECK);
3475 if !advertised {
3476 self.registered_connection = Some(registration.connection_id);
3477 self.advertised = false;
3478 self.next_probe_at = None;
3479 let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3480 state.health.status = SupervisorHealthStatus::Unknown;
3481 state.health.consecutive_failures = 0;
3482 state.health.last_probe_ms = None;
3483 state.health.detail = None;
3484 state.health.metrics = None;
3485 });
3486 return;
3487 }
3488
3489 let reregistered = self.registered_connection != Some(registration.connection_id);
3490 self.registered_connection = Some(registration.connection_id);
3491 self.advertised = true;
3492 if reregistered || self.next_probe_at.is_none() {
3493 self.probe_index = 0;
3494 self.next_probe_at = Some(
3495 Instant::now() + jittered_health_delay(&spec.module_id, 0, runtime.health.cadence),
3496 );
3497 let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3498 state.health.status = SupervisorHealthStatus::Unknown;
3499 state.health.consecutive_failures = 0;
3500 state.health.detail = None;
3501 state.health.metrics = None;
3502 });
3503 }
3504 }
3505
3506 fn wake_after(&self) -> Duration {
3507 if !self.advertised {
3508 return REGISTRY_RELEASE_POLL;
3509 }
3510 self.next_probe_at
3511 .map(|next| next.saturating_duration_since(Instant::now()))
3512 .unwrap_or(REGISTRY_RELEASE_POLL)
3513 }
3514
3515 fn due(&self) -> bool {
3516 self.advertised
3517 && self
3518 .next_probe_at
3519 .is_some_and(|next| Instant::now() >= next)
3520 }
3521
3522 fn schedule_next(&mut self, spec: &ModuleSpec, cadence: Duration) {
3523 self.probe_index = self.probe_index.wrapping_add(1);
3524 self.next_probe_at = Some(
3525 Instant::now() + jittered_health_delay(&spec.module_id, self.probe_index, cadence),
3526 );
3527 }
3528}
3529
3530#[derive(Debug)]
3565enum HealthProbeEvidence {
3566 LaneDead,
3568 NoAnswer,
3570 BadAnswer,
3572 Misconfigured,
3574}
3575
3576#[derive(Debug)]
3577struct HealthProbeError {
3578 evidence: HealthProbeEvidence,
3579 message: String,
3580}
3581
3582impl HealthProbeError {
3583 fn lane_dead(message: impl Into<String>) -> Self {
3584 Self::with(HealthProbeEvidence::LaneDead, message)
3585 }
3586
3587 fn no_answer(message: impl Into<String>) -> Self {
3588 Self::with(HealthProbeEvidence::NoAnswer, message)
3589 }
3590
3591 fn bad_answer(message: impl Into<String>) -> Self {
3592 Self::with(HealthProbeEvidence::BadAnswer, message)
3593 }
3594
3595 fn misconfigured(message: impl Into<String>) -> Self {
3596 Self::with(HealthProbeEvidence::Misconfigured, message)
3597 }
3598
3599 fn with(evidence: HealthProbeEvidence, message: impl Into<String>) -> Self {
3600 Self {
3601 evidence,
3602 message: message.into(),
3603 }
3604 }
3605
3606 #[allow(dead_code)]
3620 fn is_proof_of_death(&self) -> bool {
3621 matches!(self.evidence, HealthProbeEvidence::LaneDead)
3622 }
3623
3624 fn label(&self) -> &'static str {
3632 match self.evidence {
3633 HealthProbeEvidence::LaneDead => "lane-dead",
3634 HealthProbeEvidence::NoAnswer => "no-answer",
3635 HealthProbeEvidence::BadAnswer => "bad-answer",
3636 HealthProbeEvidence::Misconfigured => "daemon-misconfigured",
3637 }
3638 }
3639}
3640
3641impl fmt::Display for HealthProbeError {
3642 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3643 f.write_str(&self.message)
3644 }
3645}
3646
3647async fn run_health_probe_cycle(
3648 spec: &ModuleSpec,
3649 runtime: &SupervisorRuntimeConfig,
3650 registry: &Registry,
3651 process_liveness: &SupervisorProcessLiveness,
3652 snapshot: &SharedSnapshot,
3653 child: &mut Option<SupervisedChild>,
3654) {
3655 let now_ms = unix_ms_now();
3656 match probe_module_health(&spec.module_id, runtime, None).await {
3657 Ok(report) => {
3658 handle_health_report(
3659 spec,
3660 runtime,
3661 registry,
3662 process_liveness,
3663 snapshot,
3664 child,
3665 report,
3666 now_ms,
3667 )
3668 .await;
3669 }
3670 Err(err) => {
3671 handle_health_probe_failure(
3672 spec,
3673 runtime,
3674 registry,
3675 process_liveness,
3676 snapshot,
3677 child,
3678 err,
3679 now_ms,
3680 )
3681 .await;
3682 }
3683 }
3684}
3685
3686async fn probe_module_health(
3687 module_id: &str,
3688 runtime: &SupervisorRuntimeConfig,
3689 drain_deadline: Option<Instant>,
3690) -> Result<HealthReport, HealthProbeError> {
3691 let Some(forwarding) = runtime.forwarding.as_ref() else {
3692 return Err(HealthProbeError::misconfigured(
3693 "supervisor was not configured with a forwarding table",
3694 ));
3695 };
3696 let probe_started_at = Instant::now();
3697 let mut deadline = probe_started_at + runtime.health.deadline;
3698 if let Some(drain_deadline) = drain_deadline {
3699 deadline = deadline.min(drain_deadline);
3700 }
3701 let pending = if drain_deadline.is_some() {
3702 forwarding.begin_drain_health_probe_rpc_for(
3703 module_id,
3704 MODULE_CONTROL_OP_HEALTH_CHECK,
3705 probe_started_at,
3706 deadline,
3707 )
3708 } else {
3709 forwarding.begin_health_probe_rpc_for(
3710 module_id,
3711 MODULE_CONTROL_OP_HEALTH_CHECK,
3712 probe_started_at,
3713 deadline,
3714 )
3715 }
3716 .map_err(|err| {
3717 HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
3720 })?;
3721 await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
3722}
3723
3724async fn probe_endpoint_health(
3731 endpoint: crate::ModuleEndpointId,
3732 runtime: &SupervisorRuntimeConfig,
3733 deadline_cap: Option<Instant>,
3734) -> Result<HealthReport, HealthProbeError> {
3735 let Some(forwarding) = runtime.forwarding.as_ref() else {
3736 return Err(HealthProbeError::misconfigured(
3737 "supervisor was not configured with a forwarding table",
3738 ));
3739 };
3740 let probe_started_at = Instant::now();
3741 let mut deadline = probe_started_at + runtime.health.deadline;
3742 if let Some(cap) = deadline_cap {
3743 deadline = deadline.min(cap);
3744 }
3745 let pending = forwarding
3746 .begin_endpoint_health_probe_rpc_for(
3747 endpoint,
3748 MODULE_CONTROL_OP_HEALTH_CHECK,
3749 probe_started_at,
3750 deadline,
3751 )
3752 .map_err(|err| {
3753 HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
3754 })?;
3755 await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
3756}
3757
3758async fn await_health_probe(
3760 forwarding: &ForwardingTable,
3761 pending: PendingModuleControlRpc,
3762 deadline: Instant,
3763 probe_budget: Duration,
3764) -> Result<HealthReport, HealthProbeError> {
3765 let PendingModuleControlRpc {
3766 endpoint,
3767 module_sink,
3768 negotiated_ver,
3769 corr,
3770 receiver,
3771 } = pending;
3772 let body = serde_json::to_vec(&ModuleControlRequest::HealthCheck {}).map_err(|err| {
3773 HealthProbeError::misconfigured(format!("failed to encode health.check: {err}"))
3774 })?;
3775 let frame = Frame::build_with_version(
3776 negotiated_ver,
3777 FrameType::Request,
3778 control_flags(),
3779 0,
3780 0,
3781 corr,
3782 body,
3783 )
3784 .map_err(|err| {
3785 HealthProbeError::misconfigured(format!("failed to build health.check frame: {err}"))
3786 })?;
3787
3788 match timeout_at(deadline, module_sink.send(frame)).await {
3794 Ok(Ok(())) => {}
3795 Ok(Err(err)) => {
3796 let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
3797 return Err(HealthProbeError::lane_dead(format!(
3800 "failed to send health.check: {err}"
3801 )));
3802 }
3803 Err(_elapsed) => {
3804 let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
3805 return Err(HealthProbeError::no_answer(
3809 "health.check send timed out before enqueue (module egress full)",
3810 ));
3811 }
3812 }
3813
3814 match timeout_at(deadline, receiver).await {
3815 Ok(Ok(ModuleControlRpcOutcome::Response(response))) => {
3819 response.health_report().ok_or_else(|| {
3820 HealthProbeError::bad_answer("health.check RPC returned a non-health response")
3821 })
3822 }
3823 Ok(Ok(ModuleControlRpcOutcome::Rejected(body))) => Err(HealthProbeError::bad_answer(
3824 format!("health.check rejected: {}", body.message),
3825 )),
3826 Ok(Ok(ModuleControlRpcOutcome::ModuleGone(message))) => {
3827 Err(HealthProbeError::lane_dead(message))
3828 }
3829 Ok(Ok(ModuleControlRpcOutcome::MalformedResponse(message))) => {
3830 Err(HealthProbeError::bad_answer(message))
3831 }
3832 Ok(Ok(ModuleControlRpcOutcome::UnexpectedOp { expected, actual })) => {
3833 Err(HealthProbeError::bad_answer(format!(
3834 "expected module-control op '{expected}', got '{actual}'"
3835 )))
3836 }
3837 Ok(Ok(ModuleControlRpcOutcome::DeadlineElapsed)) => Err(HealthProbeError::bad_answer(
3841 "module answered health.check after its daemon deadline",
3842 )),
3843 Ok(Err(_)) => Err(HealthProbeError::misconfigured(
3844 "health.check waiter was canceled before the module responded",
3845 )),
3846 Err(_) => {
3847 let _ = forwarding.tombstone_health_probe_rpc(endpoint, corr);
3848 Err(HealthProbeError::no_answer(format!(
3849 "module did not answer health.check within {probe_budget:?}"
3850 )))
3851 }
3852 }
3853}
3854
3855#[allow(clippy::too_many_arguments)]
3856async fn handle_health_report(
3857 spec: &ModuleSpec,
3858 runtime: &SupervisorRuntimeConfig,
3859 registry: &Registry,
3860 process_liveness: &SupervisorProcessLiveness,
3861 snapshot: &SharedSnapshot,
3862 child: &mut Option<SupervisedChild>,
3863 report: HealthReport,
3864 now_ms: u64,
3865) {
3866 let status = supervisor_health_status(report.status);
3867 let detail = report.detail.clone();
3868 let metrics = truncate_health_metrics(report.metrics);
3869 let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3870 state.health.status = status;
3871 state.health.last_probe_ms = Some(now_ms);
3872 state.health.detail = detail.clone();
3873 state.health.metrics = metrics.clone();
3874 state.health.consecutive_failures = 0;
3875 });
3876
3877 let action = match report.status {
3878 HealthStatus::Ok => return,
3879 HealthStatus::Degraded => runtime.health.on_degraded,
3880 HealthStatus::Failing => runtime.health.on_failing,
3881 };
3882 apply_l3_health_action(
3883 spec,
3884 runtime,
3885 registry,
3886 process_liveness,
3887 snapshot,
3888 child,
3889 status,
3890 detail.as_deref(),
3891 action,
3892 now_ms,
3893 )
3894 .await;
3895}
3896
3897#[allow(clippy::too_many_arguments)]
3898async fn handle_health_probe_failure(
3899 spec: &ModuleSpec,
3900 runtime: &SupervisorRuntimeConfig,
3901 registry: &Registry,
3902 process_liveness: &SupervisorProcessLiveness,
3903 snapshot: &SharedSnapshot,
3904 child: &mut Option<SupervisedChild>,
3905 err: HealthProbeError,
3906 now_ms: u64,
3907) {
3908 let threshold = runtime.health.failure_threshold.max(1);
3909 let mut failures = 0;
3910 let detail = format!("[{}] {err}", err.label());
3915 let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3916 state.health.last_probe_ms = Some(now_ms);
3917 state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
3918 state.health.detail = Some(detail.clone());
3919 state.health.metrics = None;
3920 failures = state.health.consecutive_failures;
3921 });
3922
3923 if failures < threshold {
3924 warn!(
3925 module_id = %spec.module_id,
3926 consecutive_failures = failures,
3927 threshold,
3928 evidence = err.label(),
3929 detail = %detail,
3930 "health.check probe failed"
3931 );
3932 return;
3933 }
3934
3935 let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3936 state.state = ModuleState::Unresponsive;
3937 state.health.status = SupervisorHealthStatus::Unresponsive;
3938 });
3939 if runtime.health.critical {
3943 error!(
3944 module_id = %spec.module_id,
3945 status = "unresponsive",
3946 evidence = err.label(),
3947 detail = %detail,
3948 "critical module health alert"
3949 );
3950 } else {
3951 warn!(
3952 module_id = %spec.module_id,
3953 status = "unresponsive",
3954 evidence = err.label(),
3955 detail = %detail,
3956 "module health threshold breached"
3957 );
3958 }
3959 if let Err(err) = health_restart_child(
3960 spec,
3961 runtime,
3962 registry,
3963 process_liveness,
3964 snapshot,
3965 child,
3966 SupervisorHealthStatus::Unresponsive,
3967 Some(&detail),
3968 now_ms,
3969 )
3970 .await
3971 {
3972 error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
3973 }
3974}
3975
3976#[allow(clippy::too_many_arguments)]
3977async fn apply_l3_health_action(
3978 spec: &ModuleSpec,
3979 runtime: &SupervisorRuntimeConfig,
3980 registry: &Registry,
3981 process_liveness: &SupervisorProcessLiveness,
3982 snapshot: &SharedSnapshot,
3983 child: &mut Option<SupervisedChild>,
3984 status: SupervisorHealthStatus,
3985 detail: Option<&str>,
3986 action: HealthAction,
3987 now_ms: u64,
3988) {
3989 record_health_action(snapshot, &spec.module_id, action.to_string(), now_ms);
3990 match action {
3991 HealthAction::Report => {
3992 info!(
3993 module_id = %spec.module_id,
3994 status = ?status,
3995 detail,
3996 "module reported non-ok health"
3997 );
3998 }
3999 HealthAction::Alert => {
4000 error!(
4001 module_id = %spec.module_id,
4002 status = ?status,
4003 detail,
4004 "module health alert"
4005 );
4006 }
4007 HealthAction::Restart => {
4008 if let Err(err) = health_restart_child(
4009 spec,
4010 runtime,
4011 registry,
4012 process_liveness,
4013 snapshot,
4014 child,
4015 status,
4016 detail,
4017 now_ms,
4018 )
4019 .await
4020 {
4021 error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
4022 }
4023 }
4024 }
4025}
4026
4027#[allow(clippy::too_many_arguments)]
4028async fn health_restart_child(
4029 spec: &ModuleSpec,
4030 runtime: &SupervisorRuntimeConfig,
4031 registry: &Registry,
4032 process_liveness: &SupervisorProcessLiveness,
4033 snapshot: &SharedSnapshot,
4034 child: &mut Option<SupervisedChild>,
4035 status: SupervisorHealthStatus,
4036 detail: Option<&str>,
4037 now_ms: u64,
4038) -> Result<(), SuperviseError> {
4039 let (enabled, schedule) = {
4040 let mut state = lock_snapshot(snapshot)?;
4041 let enabled = state.enabled;
4042 let schedule = if enabled {
4043 state.next_crash_restart(&runtime.restart_policy, Instant::now())
4044 } else {
4045 None
4046 };
4047 (enabled, schedule)
4048 };
4049
4050 if !enabled {
4051 return Err(SuperviseError::Disabled {
4052 module_id: spec.module_id.clone(),
4053 });
4054 }
4055
4056 if schedule.is_none() {
4057 record_health_action(snapshot, &spec.module_id, "disabled".to_string(), now_ms);
4058 error!(
4059 module_id = %spec.module_id,
4060 status = ?status,
4061 detail,
4062 max_restarts = runtime.restart_policy.max_restarts,
4063 window_secs = runtime.restart_policy.window.as_secs(),
4064 "health restart budget exhausted; disabling module"
4065 );
4066 let stop_notice = begin_forwarding_drain_if_configured(
4067 spec,
4068 runtime,
4069 registry,
4070 snapshot,
4071 Some(false),
4072 RouteCloseReason::Disable,
4073 )
4074 .await?;
4075 drain_optional_child(
4076 &spec.module_id,
4077 spec.protocol,
4078 stop_notice,
4079 registry,
4080 snapshot,
4081 &runtime.terminal_ring,
4082 &runtime.spawn_events,
4083 child,
4084 runtime.drain_timeout,
4085 ModuleState::Disabled,
4086 Some(false),
4087 )
4088 .await?;
4089 process_liveness.untrack_if_current(&spec.module_id, snapshot);
4090 return Ok(());
4091 }
4092
4093 let schedule = schedule.expect("a health restart must have a crash-restart schedule");
4094 let mut restart_count = 0;
4095 update_snapshot(snapshot, Some(&spec.module_id), |state| {
4096 restart_count = state.crash_restarts.len();
4097 state.state = ModuleState::Unresponsive;
4098 state.health.status = status;
4099 state.health.last_action = Some(HealthAction::Restart.to_string());
4100 state.health.last_action_ms = Some(now_ms);
4101 })?;
4102 warn!(
4103 module_id = %spec.module_id,
4104 status = ?status,
4105 detail,
4106 restart_count,
4107 restart_in_window = schedule.restart_in_window,
4108 delay_ms = schedule.delay.as_millis() as u64,
4109 "health-triggered module restart"
4110 );
4111
4112 let stop_notice = begin_forwarding_drain_if_configured(
4113 spec,
4114 runtime,
4115 registry,
4116 snapshot,
4117 Some(true),
4118 RouteCloseReason::Restart,
4119 )
4120 .await?;
4121 drain_optional_child(
4122 &spec.module_id,
4123 spec.protocol,
4124 stop_notice,
4125 registry,
4126 snapshot,
4127 &runtime.terminal_ring,
4128 &runtime.spawn_events,
4129 child,
4130 runtime.drain_timeout,
4131 ModuleState::Restarting,
4132 Some(true),
4133 )
4134 .await?;
4135 sleep(schedule.delay).await;
4136 if !respawn_still_pending(snapshot) {
4140 process_liveness.untrack_if_current(&spec.module_id, snapshot);
4141 return Ok(());
4142 }
4143 process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
4144 match spawn_and_mark_running(spec, runtime, snapshot) {
4145 Ok(next_child) => {
4146 *child = Some(next_child);
4147 Ok(())
4148 }
4149 Err(err) => {
4150 fail_snapshot(snapshot, Some(&spec.module_id), None);
4151 process_liveness.untrack_if_current(&spec.module_id, snapshot);
4152 *child = None;
4153 Err(err)
4154 }
4155 }
4156}
4157
4158fn record_health_action(snapshot: &SharedSnapshot, module_id: &str, action: String, now_ms: u64) {
4159 let _ = update_snapshot(snapshot, Some(module_id), |state| {
4160 state.health.last_action = Some(action);
4161 state.health.last_action_ms = Some(now_ms);
4162 });
4163}
4164
4165fn supervisor_health_status(status: HealthStatus) -> SupervisorHealthStatus {
4166 match status {
4167 HealthStatus::Ok => SupervisorHealthStatus::Ok,
4168 HealthStatus::Degraded => SupervisorHealthStatus::Degraded,
4169 HealthStatus::Failing => SupervisorHealthStatus::Failing,
4170 }
4171}
4172
4173fn truncate_health_metrics(metrics: Option<Value>) -> Option<Value> {
4185 let metrics = metrics?;
4186 match serde_json::to_vec(&metrics) {
4187 Ok(encoded) if encoded.len() > MAX_HEALTH_METRICS_BYTES => Some(serde_json::json!({
4188 "truncated": true,
4189 "original_bytes": encoded.len(),
4190 })),
4191 Ok(_) | Err(_) => Some(metrics),
4192 }
4193}
4194
4195fn jittered_health_delay(module_id: &str, probe_index: u64, cadence: Duration) -> Duration {
4201 if cadence.is_zero() {
4202 return Duration::ZERO;
4203 }
4204 let cadence_ms = cadence.as_millis() as u64;
4205 if cadence_ms == 0 {
4221 return cadence;
4222 }
4223 let jitter_span = (cadence_ms / 10).max(1);
4238 let hash = module_id.as_bytes().iter().fold(
4239 probe_index.wrapping_mul(0x9E37_79B9_7F4A_7C15),
4240 |acc, byte| {
4241 acc.wrapping_mul(1099511628211)
4242 .wrapping_add(u64::from(*byte))
4243 },
4244 );
4245 cadence + Duration::from_millis(hash % jitter_span)
4246}
4247
4248#[cfg(test)]
4249mod tests {
4250 use super::*;
4251
4252 #[test]
4253 fn readding_a_module_clears_its_rescan_removal_tombstone() {
4254 let handle = SupervisorHandle::new();
4255 let module_id = "readded-tombstone";
4256 handle.record_rescan_removal(module_id);
4257 assert!(handle.removal_tombstone_age_ms(module_id).is_some());
4258
4259 handle.apply_identity_configuration(&ModuleSpec {
4260 module_id: module_id.to_string(),
4261 program: PathBuf::from("/test/module"),
4262 args: Vec::new(),
4263 env: Vec::new(),
4264 reserved: false,
4265 reserved_prefixes: Vec::new(),
4266 protocol: ModuleProtocol::Subc,
4267 overlap: Default::default(),
4268 });
4269
4270 assert!(
4271 handle.removal_tombstone_age_ms(module_id).is_none(),
4272 "a re-added module must not retain a stale removal tombstone"
4273 );
4274 }
4275
4276 fn stale_process_snapshot(state: ModuleState, enabled: bool) -> SharedSnapshot {
4277 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(state, enabled)));
4278 update_snapshot(&snapshot, Some("stale-process-facts"), |snapshot| {
4279 snapshot.process_alive = true;
4280 snapshot.pid = Some(41);
4281 snapshot.spawned_at_ms = Some(42);
4282 snapshot.spawned_from = Some(PathBuf::from("/spawned/module"));
4283 snapshot.spawned_file_identity = Some(SpawnedFileIdentity {
4284 device: 43,
4285 inode: 44,
4286 });
4287 })
4288 .unwrap();
4289 snapshot
4290 }
4291
4292 fn assert_snapshot_process_facts_cleared(snapshot: &SharedSnapshot) {
4293 let snapshot = lock_snapshot(snapshot).unwrap();
4294 assert!(!snapshot.process_alive);
4295 assert_eq!(snapshot.pid, None);
4296 assert_eq!(snapshot.spawned_at_ms, None);
4297 assert_eq!(snapshot.spawned_from, None);
4298 assert_eq!(snapshot.spawned_file_identity, None);
4299 }
4300
4301 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
4302 async fn failed_enable_spawn_clears_preexisting_current_process_facts() {
4303 let supervisor = Supervisor::default();
4304 let mut runtime = supervisor.runtime_config();
4305 runtime.test_seed_stale_facts_before_enable_spawn = true;
4306 let snapshot = stale_process_snapshot(ModuleState::Disabled, false);
4307 let mut child = None;
4308 let spec = ModuleSpec {
4309 module_id: "failed-enable-clears-facts".to_string(),
4310 program: PathBuf::from("/definitely/missing/failed-enable-module"),
4311 args: Vec::new(),
4312 env: Vec::new(),
4313 reserved: false,
4314 reserved_prefixes: Vec::new(),
4315 protocol: ModuleProtocol::Subc,
4316 overlap: Default::default(),
4317 };
4318
4319 let result = set_child_enabled(
4320 &spec,
4321 &runtime,
4322 &supervisor.registry,
4323 &supervisor.process_liveness,
4324 &snapshot,
4325 &mut child,
4326 true,
4327 )
4328 .await;
4329
4330 assert!(matches!(result, Err(SuperviseError::Spawn { .. })));
4331 assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
4332 assert_snapshot_process_facts_cleared(&snapshot);
4333 }
4334
4335 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
4336 async fn failed_reload_spawn_clears_current_process_facts() {
4337 let supervisor = Supervisor::default();
4338 let mut runtime = supervisor.runtime_config();
4339 runtime.restart_policy = RestartPolicy::new(0, Duration::ZERO);
4340 let snapshot = stale_process_snapshot(ModuleState::Running, true);
4341 let mut child = None;
4342 let spec = ModuleSpec {
4343 module_id: "failed-reload-clears-facts".to_string(),
4344 program: PathBuf::from("/unused/failed-reload-module"),
4345 args: Vec::new(),
4346 env: Vec::new(),
4347 reserved: false,
4348 reserved_prefixes: Vec::new(),
4349 protocol: ModuleProtocol::Subc,
4350 overlap: Default::default(),
4351 };
4352
4353 let result = handle_reload_spawn_failure(
4354 &spec,
4355 &runtime,
4356 &supervisor.process_liveness,
4357 &snapshot,
4358 &mut child,
4359 "forced reload spawn failure".to_string(),
4360 )
4361 .await;
4362
4363 assert!(matches!(result, Err(SuperviseError::ReloadFailed { .. })));
4364 assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
4365 assert_snapshot_process_facts_cleared(&snapshot);
4366 }
4367
4368 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
4369 async fn dropping_a_module_with_an_active_monitor_clears_current_process_facts() {
4370 let supervisor = Supervisor::default();
4371 let snapshot = stale_process_snapshot(ModuleState::Running, true);
4372 let module = supervisor.supervised_module(
4373 ModuleSpec {
4374 module_id: "drop-clears-facts".to_string(),
4375 program: PathBuf::from("/unused/drop-module"),
4376 args: Vec::new(),
4377 env: Vec::new(),
4378 reserved: false,
4379 reserved_prefixes: Vec::new(),
4380 protocol: ModuleProtocol::Subc,
4381 overlap: Default::default(),
4382 },
4383 supervisor.runtime_config(),
4384 Arc::clone(&snapshot),
4385 None,
4386 );
4387 assert!(!module
4388 .inner
4389 .monitor
4390 .lock()
4391 .unwrap()
4392 .as_ref()
4393 .unwrap()
4394 .is_finished());
4395
4396 drop(module);
4397
4398 assert_eq!(
4399 lock_snapshot(&snapshot).unwrap().state,
4400 ModuleState::Stopped
4401 );
4402 assert_snapshot_process_facts_cleared(&snapshot);
4403 }
4404
4405 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
4406 async fn configuration_update_does_not_replace_captured_running_process_facts() {
4407 let supervisor = Supervisor::default();
4408 let snapshot = stale_process_snapshot(ModuleState::Running, true);
4409 let initial = ModuleSpec {
4410 module_id: "rescan-preserves-spawn-facts".to_string(),
4411 program: PathBuf::from("/spawned/module"),
4412 args: Vec::new(),
4413 env: Vec::new(),
4414 reserved: false,
4415 reserved_prefixes: Vec::new(),
4416 protocol: ModuleProtocol::Subc,
4417 overlap: Default::default(),
4418 };
4419 let module = supervisor.supervised_module(
4420 initial.clone(),
4421 supervisor.runtime_config(),
4422 snapshot,
4423 None,
4424 );
4425 let before = module.status().unwrap();
4426 let mut replacement = initial;
4427 replacement.program = PathBuf::from("/rescanned/replacement-module");
4428
4429 module
4430 .update_configuration(replacement, HealthConfig::default(), None)
4431 .await
4432 .unwrap();
4433
4434 let after = module.status().unwrap();
4435 assert_eq!(after.pid, before.pid);
4436 assert_eq!(after.spawned_at_ms, before.spawned_at_ms);
4437 assert_eq!(after.spawned_from, before.spawned_from);
4438 drop(module);
4439 }
4440}
4441
4442fn unix_ms_now() -> u64 {
4443 SystemTime::now()
4444 .duration_since(UNIX_EPOCH)
4445 .map(|duration| duration.as_millis().min(u128::from(u64::MAX)) as u64)
4446 .unwrap_or(0)
4447}
4448
4449async fn supervise_loop(
4450 mut spec: ModuleSpec,
4451 mut runtime: SupervisorRuntimeConfig,
4452 registry: Arc<Registry>,
4453 process_liveness: Arc<SupervisorProcessLiveness>,
4454 snapshot: SharedSnapshot,
4455 mut child: Option<SupervisedChild>,
4456 mut commands: mpsc::Receiver<SupervisorCommand>,
4457) {
4458 let mut health_probe = HealthProbeRuntime::default();
4459 let mut pending_respawn: Option<Instant> = None;
4463 let mut requeued: VecDeque<SupervisorCommand> = VecDeque::new();
4466 loop {
4467 if let Some(command) = requeued.pop_front() {
4468 if !handle_supervisor_command(
4469 command,
4470 &mut spec,
4471 &mut runtime,
4472 ®istry,
4473 &process_liveness,
4474 &snapshot,
4475 &mut child,
4476 &mut commands,
4477 &mut requeued,
4478 )
4479 .await
4480 {
4481 return;
4482 }
4483 if child.is_some() || !respawn_still_pending(&snapshot) {
4484 pending_respawn = None;
4485 }
4486 continue;
4487 }
4488 if child.is_some() {
4489 health_probe.refresh_registration(&spec, &runtime, ®istry, &snapshot);
4490 let probe_sleep = sleep(health_probe.wake_after());
4491 tokio::pin!(probe_sleep);
4492 let active_child = child.as_mut().expect("child checked above");
4493 tokio::select! {
4494 wait_result = active_child.wait() => {
4495 let exit_report = match wait_result {
4504 Ok(status) => classify_reaped_child_exit(&snapshot, active_child, &status),
4505 Err(err) => {
4506 active_child.drain_stderr(&spec.module_id).await;
4507 fail_snapshot(&snapshot, Some(&spec.module_id), None);
4508 record_wait_error_terminal(
4514 &spec.module_id,
4515 &runtime.terminal_ring,
4516 &runtime.spawn_events,
4517 );
4518 untrack_if_registration_released(
4519 &process_liveness,
4520 ®istry,
4521 &spec.module_id,
4522 &snapshot,
4523 );
4524 error!(module_id = %spec.module_id, error = %err, "failed to wait for supervised module");
4525 child = None;
4526 continue;
4527 }
4528 };
4529 active_child.drain_stderr(&spec.module_id).await;
4530
4531 let next = on_child_exit(
4532 &spec,
4533 runtime.restart_policy,
4534 ®istry,
4535 &snapshot,
4536 &runtime.terminal_ring,
4537 &runtime.spawn_events,
4538 &runtime.child_roster,
4539 exit_report,
4540 ).await;
4541 active_child.release_roster();
4544 match next {
4545 NextAction::Stop { registration_released } => {
4546 if registration_released {
4547 process_liveness.untrack_if_current(&spec.module_id, &snapshot);
4548 }
4549 child = None;
4550 }
4551 NextAction::Restart { schedule } => {
4552 let delay = schedule.map_or(
4553 runtime.restart_policy.delay_for_restart(0),
4554 |schedule| schedule.delay,
4555 );
4556 if let Some(schedule) = schedule {
4557 log_crash_respawn(&spec.module_id, schedule);
4558 }
4559 child = None;
4567 pending_respawn = Some(Instant::now() + delay);
4568 }
4569 }
4570 }
4571 command = commands.recv() => {
4572 let Some(command) = command else {
4573 return;
4574 };
4575 if !handle_supervisor_command(
4576 command,
4577 &mut spec,
4578 &mut runtime,
4579 ®istry,
4580 &process_liveness,
4581 &snapshot,
4582 &mut child,
4583 &mut commands,
4584 &mut requeued,
4585 ).await {
4586 return;
4587 }
4588 }
4589 _ = &mut probe_sleep => {
4590 if health_probe.due() {
4591 run_health_probe_cycle(
4592 &spec,
4593 &runtime,
4594 ®istry,
4595 &process_liveness,
4596 &snapshot,
4597 &mut child,
4598 ).await;
4599 if child.is_some() {
4600 health_probe.schedule_next(&spec, runtime.health.cadence);
4601 }
4602 }
4603 }
4604 }
4605 } else if let Some(deadline) = pending_respawn {
4606 tokio::select! {
4607 _ = sleep_until(deadline) => {
4608 pending_respawn = None;
4609 if !respawn_still_pending(&snapshot) {
4613 continue;
4614 }
4615 if runtime.child_roster.is_closed() {
4620 let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
4621 state.state = ModuleState::Stopped;
4622 });
4623 debug!(module_id = %spec.module_id, "crash respawn cancelled by daemon shutdown");
4624 continue;
4625 }
4626 if let Err(err) = wait_for_registration_release(
4627 ®istry,
4628 &spec.module_id,
4629 REGISTRY_RELEASE_TIMEOUT,
4630 ).await {
4631 fail_snapshot(&snapshot, Some(&spec.module_id), None);
4632 error!(module_id = %spec.module_id, error = %err, "registration did not release before restart");
4633 continue;
4634 }
4635
4636 match spawn_and_mark_running(&spec, &runtime, &snapshot) {
4637 Ok(next_child) => {
4638 child = Some(next_child);
4639 debug!(module_id = %spec.module_id, "supervised module restarted after crash");
4640 }
4641 Err(err) => {
4642 fail_snapshot(&snapshot, Some(&spec.module_id), None);
4643 process_liveness.untrack_if_current(&spec.module_id, &snapshot);
4644 error!(module_id = %spec.module_id, error = %err, "failed to restart supervised module");
4645 }
4646 }
4647 }
4648 command = commands.recv() => {
4649 let Some(command) = command else {
4650 return;
4651 };
4652 if !handle_supervisor_command(
4653 command,
4654 &mut spec,
4655 &mut runtime,
4656 ®istry,
4657 &process_liveness,
4658 &snapshot,
4659 &mut child,
4660 &mut commands,
4661 &mut requeued,
4662 ).await {
4663 return;
4664 }
4665 if child.is_some() || !respawn_still_pending(&snapshot) {
4670 pending_respawn = None;
4671 }
4672 }
4673 }
4674 } else {
4675 let Some(command) = commands.recv().await else {
4676 return;
4677 };
4678 if !handle_supervisor_command(
4679 command,
4680 &mut spec,
4681 &mut runtime,
4682 ®istry,
4683 &process_liveness,
4684 &snapshot,
4685 &mut child,
4686 &mut commands,
4687 &mut requeued,
4688 )
4689 .await
4690 {
4691 return;
4692 }
4693 }
4694 }
4695}
4696
4697fn log_crash_respawn(module_id: &str, schedule: CrashRestartSchedule) {
4698 info!(
4699 module_id,
4700 restart_in_window = schedule.restart_in_window,
4701 delay_ms = schedule.delay.as_millis() as u64,
4702 "respawning after crash"
4703 );
4704}
4705
4706fn respawn_still_pending(snapshot: &SharedSnapshot) -> bool {
4712 matches!(
4713 lock_snapshot(snapshot),
4714 Ok(state) if state.enabled && state.state == ModuleState::Restarting
4715 )
4716}
4717
4718enum NextAction {
4719 Stop {
4720 registration_released: bool,
4721 },
4722 Restart {
4723 schedule: Option<CrashRestartSchedule>,
4724 },
4725}
4726
4727#[allow(clippy::too_many_arguments)]
4728async fn handle_supervisor_command(
4729 command: SupervisorCommand,
4730 spec: &mut ModuleSpec,
4731 runtime: &mut SupervisorRuntimeConfig,
4732 registry: &Registry,
4733 process_liveness: &SupervisorProcessLiveness,
4734 snapshot: &SharedSnapshot,
4735 child: &mut Option<SupervisedChild>,
4736 commands: &mut mpsc::Receiver<SupervisorCommand>,
4737 requeued: &mut VecDeque<SupervisorCommand>,
4738) -> bool {
4739 match command {
4740 SupervisorCommand::Drain { reply } => {
4741 let result = drain_optional_child(
4744 &spec.module_id,
4745 spec.protocol,
4746 StopNotice::NotSent,
4747 registry,
4748 snapshot,
4749 &runtime.terminal_ring,
4750 &runtime.spawn_events,
4751 child,
4752 runtime.drain_timeout,
4753 ModuleState::Stopped,
4754 None,
4755 )
4756 .await;
4757 let registration_released = result.is_ok();
4758 let _ = reply.send(result);
4759 if registration_released {
4760 process_liveness.untrack_if_current(&spec.module_id, snapshot);
4761 }
4762 false
4763 }
4764 SupervisorCommand::Retire { reply } => {
4765 let result = async {
4766 let stop_notice = begin_forwarding_drain_if_configured(
4767 spec,
4768 runtime,
4769 registry,
4770 snapshot,
4771 None,
4772 RouteCloseReason::Disable,
4773 )
4774 .await?;
4775 drain_optional_child(
4776 &spec.module_id,
4777 spec.protocol,
4778 stop_notice,
4779 registry,
4780 snapshot,
4781 &runtime.terminal_ring,
4782 &runtime.spawn_events,
4783 child,
4784 runtime.drain_timeout,
4785 ModuleState::Stopped,
4786 None,
4787 )
4788 .await
4789 }
4790 .await;
4791 let registration_released = result.is_ok();
4792 let _ = reply.send(result);
4793 if registration_released {
4794 process_liveness.untrack_if_current(&spec.module_id, snapshot);
4795 }
4796 false
4797 }
4798 SupervisorCommand::Restart {
4799 drain_timeout_ms,
4800 received_at_generation,
4801 queued_at,
4802 reply,
4803 } => {
4804 info!(
4808 module_id = %spec.module_id,
4809 queued_ms = u64::try_from(queued_at.elapsed().as_millis()).unwrap_or(u64::MAX),
4810 "restart command dequeued"
4811 );
4812 let validation = match lock_snapshot(snapshot) {
4824 Ok(state) if !state.enabled => Err(SuperviseError::Disabled {
4825 module_id: spec.module_id.clone(),
4826 }),
4827 Ok(_) => Ok(()),
4828 Err(err) => Err(err),
4829 };
4830 let initiated = validation.is_ok();
4831 let _ = reply.send(validation);
4832 let satisfied_by_generation = if initiated && child.is_some() {
4843 lock_snapshot(snapshot).ok().and_then(|state| {
4844 (state.spawn_generation > received_at_generation
4845 && !state.configuration_updated_since_spawn)
4846 .then_some(state.spawn_generation)
4847 })
4848 } else {
4849 None
4850 };
4851 if let Some(generation) = satisfied_by_generation {
4852 info!(
4853 module_id = %spec.module_id,
4854 received_at_generation,
4855 "restart already satisfied by generation {generation}; not restarting again"
4856 );
4857 } else if initiated {
4858 let drain_timeout = drain_timeout_ms
4861 .map(Duration::from_millis)
4862 .unwrap_or(runtime.drain_timeout);
4863 if let Err(err) = restart_child(
4864 spec,
4865 runtime,
4866 registry,
4867 process_liveness,
4868 snapshot,
4869 child,
4870 drain_timeout,
4871 )
4872 .await
4873 {
4874 warn!(
4875 module_id = %spec.module_id,
4876 error = %err,
4877 "operator restart failed after initiation ack; module state carries the outcome"
4878 );
4879 let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4880 state.state = ModuleState::Failed;
4881 clear_current_process_facts(state);
4882 });
4883 }
4884 }
4885 true
4886 }
4887 SupervisorCommand::Reload { reply } => {
4888 let result =
4889 reload_child(spec, runtime, registry, process_liveness, snapshot, child).await;
4890 let _ = reply.send(result);
4891 true
4892 }
4893 SupervisorCommand::SetEnabled { enabled, reply } => {
4894 let result = set_child_enabled(
4895 spec,
4896 runtime,
4897 registry,
4898 process_liveness,
4899 snapshot,
4900 child,
4901 enabled,
4902 )
4903 .await;
4904 let _ = reply.send(result);
4905 true
4906 }
4907 SupervisorCommand::UpdateConfiguration {
4908 spec: next_spec,
4909 health,
4910 drain_timeout_ms,
4911 reply,
4912 } => {
4913 if let Some(handle) = &runtime.supervisor_handle {
4914 handle.apply_identity_configuration(&next_spec);
4915 }
4916 *spec = next_spec;
4917 let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
4918 state.configuration_updated_since_spawn = true;
4919 });
4920 runtime.health = health;
4921 runtime.drain_timeout = drain_timeout_ms
4922 .map(Duration::from_millis)
4923 .unwrap_or(runtime.default_drain_timeout);
4924 *runtime
4925 .effective_drain_timeout
4926 .lock()
4927 .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
4928 let _ = reply.send(());
4929 true
4930 }
4931 SupervisorCommand::Swap {
4932 ready_timeout,
4933 reply,
4934 } => {
4935 let end = swap::run_swap(
4936 spec,
4937 runtime,
4938 registry,
4939 process_liveness,
4940 snapshot,
4941 child,
4942 commands,
4943 ready_timeout.unwrap_or(DEFAULT_SWAP_READY_TIMEOUT),
4944 reply,
4945 )
4946 .await;
4947 requeued.extend(end.requeue);
4948 true
4949 }
4950 }
4951}
4952
4953async fn restart_child(
4954 spec: &ModuleSpec,
4955 runtime: &SupervisorRuntimeConfig,
4956 registry: &Registry,
4957 process_liveness: &SupervisorProcessLiveness,
4958 snapshot: &SharedSnapshot,
4959 child: &mut Option<SupervisedChild>,
4960 drain_timeout: Duration,
4961) -> Result<(), SuperviseError> {
4962 if !lock_snapshot(snapshot)?.enabled {
4964 return Err(SuperviseError::Disabled {
4965 module_id: spec.module_id.clone(),
4966 });
4967 }
4968 let stop_notice = begin_forwarding_drain_with_timeout(
4969 spec,
4970 runtime,
4971 registry,
4972 snapshot,
4973 None,
4974 RouteCloseReason::Restart,
4975 drain_timeout,
4976 )
4977 .await?;
4978
4979 if child.is_some() {
4980 drain_optional_child(
4981 &spec.module_id,
4982 spec.protocol,
4983 stop_notice,
4984 registry,
4985 snapshot,
4986 &runtime.terminal_ring,
4987 &runtime.spawn_events,
4988 child,
4989 drain_timeout,
4990 ModuleState::Restarting,
4991 Some(true),
4992 )
4993 .await?;
4994 } else {
4995 update_snapshot(snapshot, Some(&spec.module_id), |state| {
4996 state.enabled = true;
4997 state.state = ModuleState::Restarting;
4998 clear_current_process_facts(state);
4999 })?;
5000 wait_for_registration_release(registry, &spec.module_id, REGISTRY_RELEASE_TIMEOUT).await?;
5001 }
5002
5003 reset_restart_count(snapshot, &spec.module_id)?;
5004 sleep(runtime.restart_policy.backoff).await;
5005 if !respawn_still_pending(snapshot) {
5008 process_liveness.untrack_if_current(&spec.module_id, snapshot);
5009 return Ok(());
5010 }
5011 process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
5012 match spawn_and_mark_running(spec, runtime, snapshot) {
5018 Ok(next_child) => {
5019 *child = Some(next_child);
5020 debug!(module_id = %spec.module_id, "supervised module restarted by operator request");
5021 Ok(())
5022 }
5023 Err(err) => {
5024 fail_snapshot(snapshot, Some(&spec.module_id), None);
5025 process_liveness.untrack_if_current(&spec.module_id, snapshot);
5026 *child = None;
5027 Err(err)
5028 }
5029 }
5030}
5031
5032async fn reload_child(
5033 spec: &ModuleSpec,
5034 runtime: &SupervisorRuntimeConfig,
5035 registry: &Registry,
5036 process_liveness: &SupervisorProcessLiveness,
5037 snapshot: &SharedSnapshot,
5038 child: &mut Option<SupervisedChild>,
5039) -> Result<(), SuperviseError> {
5040 if !lock_snapshot(snapshot)?.enabled {
5042 return Err(SuperviseError::Disabled {
5043 module_id: spec.module_id.clone(),
5044 });
5045 }
5046 let stop_notice = begin_forwarding_drain(
5047 spec,
5048 runtime,
5049 registry,
5050 snapshot,
5051 Some(true),
5052 RouteCloseReason::Reload,
5053 )
5054 .await?;
5055
5056 if child.is_some() {
5057 drain_optional_child(
5058 &spec.module_id,
5059 spec.protocol,
5060 stop_notice,
5061 registry,
5062 snapshot,
5063 &runtime.terminal_ring,
5064 &runtime.spawn_events,
5065 child,
5066 runtime.drain_timeout,
5067 ModuleState::Restarting,
5068 Some(true),
5069 )
5070 .await?;
5071 } else {
5072 update_snapshot(snapshot, Some(&spec.module_id), |state| {
5073 state.enabled = true;
5074 state.state = ModuleState::Restarting;
5075 clear_current_process_facts(state);
5076 })?;
5077 wait_for_registration_release(registry, &spec.module_id, REGISTRY_RELEASE_TIMEOUT).await?;
5078 }
5079
5080 reset_restart_count(snapshot, &spec.module_id)?;
5081 sleep(runtime.restart_policy.backoff).await;
5082 if !respawn_still_pending(snapshot) {
5085 process_liveness.untrack_if_current(&spec.module_id, snapshot);
5086 return Ok(());
5087 }
5088 process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
5089 let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
5090 Ok(next_child) => next_child,
5091 Err(err) => {
5092 return handle_reload_spawn_failure(
5093 spec,
5094 runtime,
5095 process_liveness,
5096 snapshot,
5097 child,
5098 format!("new child failed to spawn: {err}"),
5099 )
5100 .await;
5101 }
5102 };
5103 *child = Some(next_child);
5104
5105 let wait_outcome = {
5106 let active_child = child.as_mut().expect("new reload child was just stored");
5107 wait_for_registration_after_reload(
5108 registry,
5109 &spec.module_id,
5110 snapshot,
5111 active_child,
5112 REGISTRY_RELEASE_TIMEOUT,
5113 )
5114 .await?
5115 };
5116
5117 match wait_outcome {
5118 RegistrationWaitOutcome::Registered => {
5119 debug!(module_id = %spec.module_id, "supervised module reloaded and registered");
5120 Ok(())
5121 }
5122 RegistrationWaitOutcome::Exited(exit_report) => {
5123 if let Some(active_child) = child.as_mut() {
5124 active_child.drain_stderr(&spec.module_id).await;
5125 }
5126 *child = None;
5127 handle_reload_child_registration_failure(
5128 spec,
5129 runtime,
5130 registry,
5131 process_liveness,
5132 snapshot,
5133 child,
5134 ReloadRegistrationFailure {
5135 exit_report: registration_failure_exit_report(exit_report),
5136 reason: "new child exited before registering".to_string(),
5137 },
5138 )
5139 .await
5140 }
5141 RegistrationWaitOutcome::TimedOut => {
5142 let mut timed_out_child = child
5143 .take()
5144 .expect("timed-out reload child is still running");
5145 timed_out_child
5146 .start_kill()
5147 .map_err(|source| SuperviseError::Kill {
5148 module_id: spec.module_id.clone(),
5149 source,
5150 })?;
5151 let status = timed_out_child
5152 .wait()
5153 .await
5154 .map_err(|source| SuperviseError::Wait {
5155 module_id: spec.module_id.clone(),
5156 source,
5157 })?;
5158 timed_out_child.drain_stderr(&spec.module_id).await;
5159 handle_reload_child_registration_failure(
5160 spec,
5161 runtime,
5162 registry,
5163 process_liveness,
5164 snapshot,
5165 child,
5166 ReloadRegistrationFailure {
5167 exit_report: registration_failure_exit_report(classify_reaped_child_exit(
5168 snapshot,
5169 &timed_out_child,
5170 &status,
5171 )),
5172 reason: format!(
5173 "new child did not register within {:?}",
5174 REGISTRY_RELEASE_TIMEOUT
5175 ),
5176 },
5177 )
5178 .await
5179 }
5180 }
5181}
5182
5183async fn set_child_enabled(
5184 spec: &ModuleSpec,
5185 runtime: &SupervisorRuntimeConfig,
5186 registry: &Registry,
5187 process_liveness: &SupervisorProcessLiveness,
5188 snapshot: &SharedSnapshot,
5189 child: &mut Option<SupervisedChild>,
5190 enabled: bool,
5191) -> Result<bool, SuperviseError> {
5192 let (current_enabled, current_state) = {
5193 let state = lock_snapshot(snapshot)?;
5194 (state.enabled, state.state)
5195 };
5196 let revive_terminal = enabled
5204 && current_enabled
5205 && child.is_none()
5206 && matches!(current_state, ModuleState::Failed | ModuleState::Stopped);
5207 if current_enabled == enabled && !revive_terminal {
5208 return Ok(false);
5209 }
5210
5211 if enabled {
5212 update_snapshot(snapshot, Some(&spec.module_id), |state| {
5213 state.enabled = true;
5214 state.state = ModuleState::Starting;
5215 clear_current_process_facts(state);
5216 })?;
5217 #[cfg(test)]
5218 if runtime.test_seed_stale_facts_before_enable_spawn {
5219 update_snapshot(snapshot, Some(&spec.module_id), |state| {
5220 state.process_alive = true;
5221 state.pid = Some(41);
5222 state.spawned_at_ms = Some(42);
5223 state.spawned_from = Some(PathBuf::from("/spawned/module"));
5224 state.spawned_file_identity = Some(SpawnedFileIdentity {
5225 device: 43,
5226 inode: 44,
5227 });
5228 })?;
5229 }
5230 wait_for_registration_release(registry, &spec.module_id, REGISTRY_RELEASE_TIMEOUT).await?;
5231 reset_restart_count(snapshot, &spec.module_id)?;
5232 process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
5233 let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
5234 Ok(next_child) => next_child,
5235 Err(err) => {
5236 if let Err(state_err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5237 state.state = ModuleState::Failed;
5238 clear_current_process_facts(state);
5239 }) {
5240 error!(module_id = %spec.module_id, error = %state_err, "failed to record enable spawn failure");
5241 }
5242 process_liveness.untrack_if_current(&spec.module_id, snapshot);
5243 return Err(err);
5244 }
5245 };
5246 *child = Some(next_child);
5247 debug!(module_id = %spec.module_id, "supervised module enabled");
5248 Ok(true)
5249 } else {
5250 let stop_notice = begin_forwarding_drain_if_configured(
5251 spec,
5252 runtime,
5253 registry,
5254 snapshot,
5255 Some(false),
5256 RouteCloseReason::Disable,
5257 )
5258 .await?;
5259 drain_optional_child(
5260 &spec.module_id,
5261 spec.protocol,
5262 stop_notice,
5263 registry,
5264 snapshot,
5265 &runtime.terminal_ring,
5266 &runtime.spawn_events,
5267 child,
5268 runtime.drain_timeout,
5269 ModuleState::Disabled,
5270 Some(false),
5271 )
5272 .await?;
5273 debug!(module_id = %spec.module_id, "supervised module disabled");
5274 Ok(true)
5275 }
5276}
5277
5278#[allow(clippy::too_many_arguments)]
5279async fn on_child_exit(
5280 spec: &ModuleSpec,
5281 policy: RestartPolicy,
5282 registry: &Registry,
5283 snapshot: &SharedSnapshot,
5284 terminal_ring: &Arc<Mutex<TerminalRing>>,
5285 spawn_events: &SpawnEventFeed,
5286 roster: &ChildRoster,
5287 exit_report: ExitReport,
5288) -> NextAction {
5289 if roster.is_closed() {
5295 return on_child_exit_during_daemon_shutdown(
5296 spec,
5297 registry,
5298 snapshot,
5299 terminal_ring,
5300 spawn_events,
5301 exit_report,
5302 )
5303 .await;
5304 }
5305 let unrequested_clean_exit_of_protocol_none =
5320 exit_report.kind == ExitKind::Clean && spec.protocol == ModuleProtocol::None;
5321 match exit_report.kind {
5322 ExitKind::Clean if !unrequested_clean_exit_of_protocol_none => {
5323 info!(
5324 module_id = %spec.module_id,
5325 exit_code = ?exit_report.code,
5326 exit_signal = ?exit_report.signal,
5327 "supervised module exited cleanly"
5328 );
5329 if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5330 state.state = ModuleState::Stopped;
5331 clear_current_process_facts(state);
5332 state.last_exit = Some(exit_report.clone());
5333 }) {
5334 error!(module_id = %spec.module_id, error = %err, "failed to record clean module exit");
5335 }
5336 record_terminal(
5337 &spec.module_id,
5338 terminal_ring,
5339 spawn_events,
5340 &exit_report,
5341 TerminalDisposition::Stopped,
5342 );
5343 let registration_released = match wait_for_registration_release(
5344 registry,
5345 &spec.module_id,
5346 REGISTRY_RELEASE_TIMEOUT,
5347 )
5348 .await
5349 {
5350 Ok(()) => true,
5351 Err(err) => {
5352 warn!(module_id = %spec.module_id, error = %err, "registration still active after clean exit");
5353 false
5354 }
5355 };
5356 NextAction::Stop {
5357 registration_released,
5358 }
5359 }
5360 ExitKind::Clean | ExitKind::Crash => {
5361 if unrequested_clean_exit_of_protocol_none {
5362 warn!(
5363 module_id = %spec.module_id,
5364 exit_code = ?exit_report.code,
5365 exit_signal = ?exit_report.signal,
5366 "protocol-none module exited cleanly without a stop request; handling it as a crash"
5367 );
5368 } else {
5369 warn!(
5370 module_id = %spec.module_id,
5371 exit_code = ?exit_report.code,
5372 exit_signal = ?exit_report.signal,
5373 "supervised module exited abnormally (crash)"
5374 );
5375 }
5376 let mut restart_schedule = None;
5377 let mut disposition = TerminalDisposition::Disabled;
5378 let mut disposition_detail = None;
5382 let now = Instant::now();
5383 if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5384 clear_current_process_facts(state);
5385 state.last_exit = Some(exit_report.clone());
5386 if state.enabled {
5387 if let Some(schedule) = state.next_crash_restart(&policy, now) {
5388 state.state = ModuleState::Restarting;
5389 restart_schedule = Some(schedule);
5390 disposition = TerminalDisposition::Restarting;
5391 } else {
5392 state.state = ModuleState::Failed;
5393 disposition = TerminalDisposition::Failed;
5394 disposition_detail = Some(policy.budget_exhausted_detail());
5395 }
5396 } else {
5397 state.state = ModuleState::Disabled;
5398 disposition = TerminalDisposition::Disabled;
5399 }
5400 }) {
5401 error!(module_id = %spec.module_id, error = %err, "failed to record crashed module exit");
5402 return NextAction::Stop {
5403 registration_released: false,
5404 };
5405 }
5406 if disposition_detail.is_some() {
5407 error!(
5412 module_id = %spec.module_id,
5413 max_restarts = policy.max_restarts,
5414 window_secs = policy.window.as_secs(),
5415 "module stopped: {}",
5416 policy.budget_exhausted_detail()
5417 );
5418 }
5419 record_terminal_with_detail(
5420 &spec.module_id,
5421 terminal_ring,
5422 spawn_events,
5423 &exit_report,
5424 disposition,
5425 disposition_detail,
5426 );
5427
5428 if let Some(schedule) = restart_schedule {
5429 NextAction::Restart {
5430 schedule: Some(schedule),
5431 }
5432 } else {
5433 let registration_released = match wait_for_registration_release(
5434 registry,
5435 &spec.module_id,
5436 REGISTRY_RELEASE_TIMEOUT,
5437 )
5438 .await
5439 {
5440 Ok(()) => true,
5441 Err(err) => {
5442 warn!(module_id = %spec.module_id, error = %err, "registration still active after failed module");
5443 false
5444 }
5445 };
5446 NextAction::Stop {
5447 registration_released,
5448 }
5449 }
5450 }
5451 ExitKind::DeliberateSeverance => {
5452 warn!(
5453 module_id = %spec.module_id,
5454 exit_code = ?exit_report.code,
5455 exit_signal = ?exit_report.signal,
5456 "supervised module exited after deliberate connection severance"
5457 );
5458 let mut should_restart = false;
5459 let mut disposition = TerminalDisposition::Disabled;
5460 if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5461 clear_current_process_facts(state);
5462 state.last_exit = Some(exit_report.clone());
5463 state.lifetime_restarts += 1;
5464 if state.enabled {
5465 state.state = ModuleState::Restarting;
5466 should_restart = true;
5467 disposition = TerminalDisposition::Restarting;
5468 } else {
5469 state.state = ModuleState::Disabled;
5470 }
5471 }) {
5472 error!(module_id = %spec.module_id, error = %err, "failed to record deliberately severed module exit");
5473 return NextAction::Stop {
5474 registration_released: false,
5475 };
5476 }
5477 record_terminal(
5478 &spec.module_id,
5479 terminal_ring,
5480 spawn_events,
5481 &exit_report,
5482 disposition,
5483 );
5484
5485 if should_restart {
5486 NextAction::Restart { schedule: None }
5487 } else {
5488 let registration_released = match wait_for_registration_release(
5489 registry,
5490 &spec.module_id,
5491 REGISTRY_RELEASE_TIMEOUT,
5492 )
5493 .await
5494 {
5495 Ok(()) => true,
5496 Err(err) => {
5497 warn!(module_id = %spec.module_id, error = %err, "registration still active after deliberately severed module exit");
5498 false
5499 }
5500 };
5501 NextAction::Stop {
5502 registration_released,
5503 }
5504 }
5505 }
5506 }
5507}
5508
5509async fn on_child_exit_during_daemon_shutdown(
5510 spec: &ModuleSpec,
5511 registry: &Registry,
5512 snapshot: &SharedSnapshot,
5513 terminal_ring: &Arc<Mutex<TerminalRing>>,
5514 spawn_events: &SpawnEventFeed,
5515 exit_report: ExitReport,
5516) -> NextAction {
5517 info!(
5518 module_id = %spec.module_id,
5519 exit_code = ?exit_report.code,
5520 exit_signal = ?exit_report.signal,
5521 exit_kind = ?exit_report.kind,
5522 "supervised module exited during daemon shutdown; not restarting it"
5523 );
5524 if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5525 state.state = ModuleState::Stopped;
5526 clear_current_process_facts(state);
5527 state.last_exit = Some(exit_report.clone());
5528 }) {
5529 error!(module_id = %spec.module_id, error = %err, "failed to record module exit during daemon shutdown");
5530 }
5531 record_terminal(
5532 &spec.module_id,
5533 terminal_ring,
5534 spawn_events,
5535 &exit_report,
5536 TerminalDisposition::DaemonShutdown,
5537 );
5538 let registration_released =
5539 wait_for_registration_release(registry, &spec.module_id, REGISTRY_RELEASE_TIMEOUT)
5540 .await
5541 .is_ok();
5542 NextAction::Stop {
5543 registration_released,
5544 }
5545}
5546
5547fn record_wait_error_terminal(
5548 module_id: &str,
5549 terminal_ring: &Arc<Mutex<TerminalRing>>,
5550 spawn_events: &SpawnEventFeed,
5551) {
5552 record_terminal(
5553 module_id,
5554 terminal_ring,
5555 spawn_events,
5556 &wait_error_exit_report(),
5557 TerminalDisposition::Failed,
5558 );
5559}
5560
5561fn record_terminal(
5562 module_id: &str,
5563 terminal_ring: &Arc<Mutex<TerminalRing>>,
5564 spawn_events: &SpawnEventFeed,
5565 exit_report: &ExitReport,
5566 disposition: TerminalDisposition,
5567) {
5568 record_terminal_with_detail(
5569 module_id,
5570 terminal_ring,
5571 spawn_events,
5572 exit_report,
5573 disposition,
5574 None,
5575 );
5576}
5577
5578fn durable_terminal_history_of(
5582 terminal_ring: &Mutex<TerminalRing>,
5583 module_id: &str,
5584) -> subc_control::TerminalHistory {
5585 let read = terminal_ring
5586 .lock()
5587 .unwrap_or_else(|p| p.into_inner())
5588 .capture_durable_history();
5589 read.read(module_id)
5590}
5591
5592fn record_terminal_with_detail(
5593 module_id: &str,
5594 terminal_ring: &Arc<Mutex<TerminalRing>>,
5595 spawn_events: &SpawnEventFeed,
5596 exit_report: &ExitReport,
5597 disposition: TerminalDisposition,
5598 disposition_detail: Option<String>,
5599) {
5600 spawn_events.emit_exited(module_id, exit_report.code, exit_report.signal);
5601 let record = TerminalRecord {
5602 exit_code: exit_report.code,
5603 exit_signal: exit_report.signal,
5604 at_ms: exit_report.at_ms,
5605 disposition,
5606 exit_kind: exit_report.kind.into(),
5607 disposition_detail,
5608 };
5609 terminal_ring
5610 .lock()
5611 .unwrap_or_else(|poisoned| poisoned.into_inner())
5612 .record_exit(module_id, record);
5613}
5614
5615fn untrack_if_registration_released(
5616 process_liveness: &SupervisorProcessLiveness,
5617 registry: &Registry,
5618 module_id: &str,
5619 snapshot: &SharedSnapshot,
5620) {
5621 match registry.get_module(module_id) {
5622 Ok(None) => process_liveness.untrack_if_current(module_id, snapshot),
5623 Ok(Some(_)) => {}
5624 Err(err) => {
5625 warn!(module_id, error = %err, "could not determine whether supervisor liveness can be untracked");
5626 }
5627 }
5628}
5629
5630#[cfg(test)]
5644fn apply_wire_spawn_args(
5645 command: &mut Command,
5646 spec: &ModuleSpec,
5647 connection_file_path: Option<&std::path::Path>,
5648 handle: Option<&SupervisorHandle>,
5649) -> Result<Option<NonceHandoff>, SuperviseError> {
5650 apply_wire_spawn_args_for_role(
5651 command,
5652 spec,
5653 connection_file_path,
5654 handle,
5655 SpawnRole::Plain,
5656 )
5657}
5658
5659#[cfg(unix)]
5664type NonceHandoff = subc_os::LaunchNonceHandoff;
5665#[cfg(not(unix))]
5666type NonceHandoff = std::convert::Infallible;
5667
5668fn apply_wire_spawn_args_for_role(
5687 command: &mut Command,
5688 spec: &ModuleSpec,
5689 connection_file_path: Option<&std::path::Path>,
5690 handle: Option<&SupervisorHandle>,
5691 role: SpawnRole,
5692) -> Result<Option<NonceHandoff>, SuperviseError> {
5693 command.env(SUBC_MODULE_ID_ENV, &spec.module_id);
5694 command.env_remove(subc_os::LAUNCH_NONCE_FD_ENV);
5699 if spec.protocol == ModuleProtocol::None {
5700 return Ok(None);
5701 }
5702 if let Some(connection_file_path) = connection_file_path {
5703 command.arg(SUBC_ARG).arg(connection_file_path);
5704 }
5705
5706 let nonce = generate_launch_nonce()?;
5710 if let Some(handle) = handle {
5711 match role {
5712 SpawnRole::Plain => {
5713 handle.set_spawn_nonce(&spec.module_id, nonce.clone());
5714 if spec.reserved {
5715 handle.set_reserved_nonce(&spec.module_id, nonce.clone());
5716 }
5717 }
5718 SpawnRole::SwapCandidate => handle.open_swap(&spec.module_id, nonce.clone()),
5719 }
5720 }
5721 #[cfg(unix)]
5722 let handoff = {
5723 let handoff =
5724 subc_os::LaunchNonceHandoff::new(&nonce).map_err(|source| SuperviseError::Spawn {
5725 program: spec.program.clone(),
5726 source,
5727 cgroup_path: None,
5728 })?;
5729 command.env(subc_os::LAUNCH_NONCE_FD_ENV, handoff.fd_env_value());
5730 Some(handoff)
5731 };
5732 #[cfg(not(unix))]
5733 let handoff = None;
5734 command.env(SUBC_LAUNCH_NONCE_ENV, nonce);
5735 Ok(handoff)
5736}
5737
5738fn apply_child_env(command: &mut Command, spec: &ModuleSpec) {
5739 command.env_remove(CK_LOG_ENV);
5740 command.env_remove(SUBC_SPAWN_ROLE_ENV);
5747 for (key, value) in &spec.env {
5748 if matches!(
5752 key.as_str(),
5753 CAPTURE_MAX_FILE_MB_ENV | CAPTURE_KEEP_ENV | CAPTURE_MAX_AGE_DAYS_ENV
5754 ) || key == SUBC_SPAWN_ROLE_ENV
5755 {
5756 continue;
5757 }
5758 command.env(key, value);
5759 }
5760}
5761
5762#[derive(Debug, Clone, Copy, PartialEq, Eq)]
5765enum SpawnRole {
5766 Plain,
5767 SwapCandidate,
5768}
5769
5770fn apply_spawn_role(command: &mut Command, role: SpawnRole) {
5773 if role == SpawnRole::SwapCandidate {
5774 command.env(SUBC_SPAWN_ROLE_ENV, SPAWN_ROLE_SWAP_CANDIDATE);
5775 }
5776}
5777
5778fn spawn_child(
5779 spec: &ModuleSpec,
5780 connection_file_path: Option<&std::path::Path>,
5781 handle: Option<&SupervisorHandle>,
5782 ring: &Arc<Mutex<StderrRing>>,
5783 capture_logs_dir: Option<&std::path::Path>,
5784 roster: &ChildRoster,
5785 #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
5786) -> Result<SupervisedChild, SuperviseError> {
5787 spawn_child_in_slot(
5788 spec,
5789 connection_file_path,
5790 handle,
5791 ring,
5792 capture_logs_dir,
5793 roster,
5794 #[cfg(target_os = "linux")]
5795 cgroup_placement,
5796 SpawnRole::Plain,
5797 false,
5798 )
5799}
5800
5801#[allow(clippy::too_many_arguments)]
5814fn spawn_child_in_slot(
5815 spec: &ModuleSpec,
5816 connection_file_path: Option<&std::path::Path>,
5817 handle: Option<&SupervisorHandle>,
5818 ring: &Arc<Mutex<StderrRing>>,
5819 capture_logs_dir: Option<&std::path::Path>,
5820 roster: &ChildRoster,
5821 #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
5822 role: SpawnRole,
5823 alternate_slot: bool,
5824) -> Result<SupervisedChild, SuperviseError> {
5825 if roster.is_closed() {
5826 return Err(SuperviseError::Spawn {
5827 program: spec.program.clone(),
5828 source: io::Error::other("the daemon is shutting down; not starting a new process"),
5829 cgroup_path: None,
5830 });
5831 }
5832 #[cfg(target_os = "linux")]
5833 let cgroup_name = swap::cgroup_name(&spec.module_id, alternate_slot);
5834 #[cfg(not(target_os = "linux"))]
5835 let _ = alternate_slot;
5836 let mut command = Command::new(&spec.program);
5837 command.args(&spec.args);
5838 apply_child_env(&mut command, spec);
5868 apply_spawn_role(&mut command, role);
5869 let nonce_handoff =
5870 apply_wire_spawn_args_for_role(&mut command, spec, connection_file_path, handle, role)?;
5871
5872 #[cfg(target_os = "linux")]
5873 let cgroup_path = cgroup_placement
5874 .map(|placement| placement.module_path(&cgroup_name))
5875 .transpose()
5876 .map_err(|source| SuperviseError::Cgroup {
5877 module_id: spec.module_id.clone(),
5878 source,
5879 })?;
5880 #[cfg(not(target_os = "linux"))]
5881 let cgroup_path: Option<PathBuf> = None;
5882 #[cfg(target_os = "linux")]
5883 if let Some(path) = &cgroup_path {
5884 if let Err(error) = apply_cgroup_placement(&mut command, spec, path) {
5885 if let Some(placement) = cgroup_placement {
5886 remove_module_cgroup(placement, &cgroup_name);
5887 }
5888 return Err(error);
5889 }
5890 }
5891
5892 let output_sink = if let Some(logs_dir) = capture_logs_dir {
5893 let path = logs_dir.join(format!("{}.stderr.log", spec.module_id));
5894 match ChildOutputSink::open(&path, capture_retention(spec)) {
5895 Ok(sink) => sink,
5896 Err(error) => {
5897 warn!(
5898 module_id = %spec.module_id,
5899 path = %path.display(),
5900 error = %error,
5901 "could not open child output capture file; forwarding to stderr"
5902 );
5903 ChildOutputSink::Stderr
5904 }
5905 }
5906 } else {
5907 ChildOutputSink::Stderr
5908 };
5909
5910 command.stdout(Stdio::piped());
5911 command.stderr(Stdio::piped());
5912 command.kill_on_drop(true);
5913 #[cfg(unix)]
5930 command.process_group(0);
5931 command.stdin(Stdio::null());
5932 #[cfg(unix)]
5936 if let Some(handoff) = nonce_handoff {
5937 handoff.install_last(command.as_std_mut());
5938 }
5939 #[cfg(not(unix))]
5940 let _ = nonce_handoff;
5941
5942 #[cfg(windows)]
5947 subc_jobobject::suspend_on_create_async(&mut command);
5948 let mut child = match command.spawn() {
5949 Ok(child) => child,
5950 Err(source) => {
5951 #[cfg(target_os = "linux")]
5952 if let Some(placement) = cgroup_placement {
5953 remove_module_cgroup(placement, &cgroup_name);
5954 }
5955 return Err(SuperviseError::Spawn {
5956 program: spec.program.clone(),
5957 source,
5958 cgroup_path,
5959 });
5960 }
5961 };
5962
5963 #[cfg(windows)]
5965 let job = contain_spawned_child(&child, spec)?;
5966 let spawned_at_ms = unix_ms_now();
5967 let spawned_from = spec.program.clone();
5968 let spawned_file_identity = spawned_file_identity(&spawned_from);
5969 let pid = child.id().ok_or_else(|| SuperviseError::Spawn {
5970 program: spec.program.clone(),
5971 source: io::Error::other("spawned child exposed no live pid"),
5972 cgroup_path: cgroup_path.clone(),
5973 })?;
5974 let process_start_time = crate::provenance::process_start_time(pid);
5975 let process_identity = process_start_time.map(|start_time| ProcessIdentity { pid, start_time });
5976 #[cfg(target_os = "linux")]
5980 let recorded_cgroup_name = cgroup_path.as_ref().map(|_| cgroup_name.clone());
5981 #[cfg(not(target_os = "linux"))]
5982 let recorded_cgroup_name = None;
5983 let roster_guard = roster.admit(
5984 spec.module_id.clone(),
5985 pid,
5986 spec.protocol,
5987 process_start_time,
5988 crate::child_roster::RecordedIdentity {
5989 start_time: subc_os::start_time(pid),
5990 executable: spawned_file_identity.map(|identity| {
5991 crate::live_children::ExecutableIdentity {
5992 device: identity.device,
5993 inode: identity.inode,
5994 }
5995 }),
5996 cgroup_name: recorded_cgroup_name,
5997 },
5998 );
5999 if roster.is_closed() {
6008 if let Err(error) = child.start_kill() {
6009 debug!(module_id = %spec.module_id, pid, %error, "kill of a process spawned during daemon shutdown failed; it may already have exited");
6010 }
6011 drop(roster_guard);
6012 return Err(SuperviseError::Spawn {
6013 program: spec.program.clone(),
6014 source: io::Error::other(
6015 "the daemon began shutting down while this process was starting; ended it",
6016 ),
6017 cgroup_path,
6018 });
6019 }
6020
6021 let stdout_pump = match child.stdout.take() {
6022 Some(stdout) => Some(tokio::spawn(pump_stdout_to(stdout, output_sink.clone()))),
6023 None => {
6024 warn!(
6025 module_id = %spec.module_id,
6026 "spawned child exposed no stdout pipe; file capture will be incomplete"
6027 );
6028 None
6029 }
6030 };
6031 let stderr_pump = match child.stderr.take() {
6032 Some(stderr) => {
6033 let generation = ring
6034 .lock()
6035 .unwrap_or_else(|poisoned| poisoned.into_inner())
6036 .begin_process();
6037 Some(StderrPump {
6038 task: tokio::spawn(pump_stderr_to(
6039 stderr,
6040 Arc::clone(ring),
6041 generation,
6042 output_sink,
6043 )),
6044 generation,
6045 })
6046 }
6047 None => {
6048 ring.lock()
6052 .unwrap_or_else(|poisoned| poisoned.into_inner())
6053 .mark_not_captured("stderr pipe was not available on spawn");
6054 warn!(
6055 module_id = %spec.module_id,
6056 "spawned child exposed no stderr pipe; tail will be unavailable"
6057 );
6058 None
6059 }
6060 };
6061
6062 Ok(SupervisedChild {
6063 child,
6064 #[cfg(target_os = "linux")]
6065 module_id: cgroup_name,
6066 #[cfg(target_os = "linux")]
6067 cgroup_placement: cgroup_placement.cloned(),
6068 #[cfg(windows)]
6069 job,
6070 stdout_pump,
6071 stderr_pump,
6072 stderr_ring: Arc::clone(ring),
6073 spawned_at_ms,
6074 spawned_from,
6075 spawned_file_identity,
6076 process_start_time,
6077 process_identity,
6078 pid,
6079 roster_guard: Some(roster_guard),
6080 })
6081}
6082
6083#[cfg(windows)]
6097fn contain_spawned_child(
6098 child: &Child,
6099 spec: &ModuleSpec,
6100) -> Result<Option<subc_jobobject::JobObject>, SuperviseError> {
6101 let module_id = spec.module_id.as_str();
6102 let Some(pid) = child.id() else {
6103 warn!(
6106 module_id,
6107 "spawned child had already exited before containment; no job object attached"
6108 );
6109 return Ok(None);
6110 };
6111
6112 let job = match subc_jobobject::JobObject::new() {
6113 Ok(job) => job,
6114 Err(source) => {
6115 warn!(
6116 module_id,
6117 error = %source,
6118 "could not create a job object; this module's helper processes will not be \
6119 reaped on teardown"
6120 );
6121 resume_suspended_child(pid, spec)?;
6124 return Ok(None);
6125 }
6126 };
6127
6128 if let Err(source) = job.assign(child) {
6129 warn!(
6130 module_id,
6131 error = %source,
6132 "could not assign the child to its job object; this module's helper processes \
6133 will not be reaped on teardown"
6134 );
6135 resume_suspended_child(pid, spec)?;
6136 return Ok(None);
6137 }
6138
6139 resume_suspended_child(pid, spec)?;
6140 Ok(Some(job))
6141}
6142
6143#[cfg(windows)]
6148fn resume_suspended_child(pid: u32, spec: &ModuleSpec) -> Result<(), SuperviseError> {
6149 if let Err(source) = subc_jobobject::resume_main_thread(pid) {
6150 let _ = std::process::Command::new("taskkill.exe")
6154 .args(["/PID", &pid.to_string(), "/T", "/F"])
6155 .stdin(Stdio::null())
6156 .stdout(Stdio::null())
6157 .stderr(Stdio::null())
6158 .status();
6159 return Err(SuperviseError::Spawn {
6160 program: spec.program.clone(),
6161 source,
6162 cgroup_path: None,
6163 });
6164 }
6165 Ok(())
6166}
6167
6168#[cfg(target_os = "linux")]
6169fn remove_module_cgroup(placement: &subc_cgroup::Placement, module_id: &str) {
6170 match placement.remove_module(module_id) {
6171 Ok(()) => debug!(module_id, "removed module cgroup after process exit"),
6172 Err(error) => warn!(
6173 module_id,
6174 error = %error,
6175 "could not remove module cgroup after process exit; continuing teardown"
6176 ),
6177 }
6178}
6179
6180#[cfg(target_os = "linux")]
6181fn apply_cgroup_placement(
6182 command: &mut Command,
6183 spec: &ModuleSpec,
6184 path: &std::path::Path,
6185) -> Result<(), SuperviseError> {
6186 subc_cgroup::apply(command, path).map_err(|source| SuperviseError::Cgroup {
6187 module_id: spec.module_id.clone(),
6188 source,
6189 })
6190}
6191
6192fn capture_retention(spec: &ModuleSpec) -> Retention {
6193 let defaults = Retention::default();
6194 let value = |name: &str| {
6195 spec.env
6196 .iter()
6197 .rev()
6198 .find_map(|(key, value)| (key == name).then_some(value.as_str()))
6199 };
6200 Retention {
6201 max_file_mb: value(CAPTURE_MAX_FILE_MB_ENV)
6202 .and_then(|value| value.parse().ok())
6203 .unwrap_or(defaults.max_file_mb),
6204 keep: value(CAPTURE_KEEP_ENV)
6205 .and_then(|value| value.parse().ok())
6206 .unwrap_or(defaults.keep),
6207 max_age_days: value(CAPTURE_MAX_AGE_DAYS_ENV)
6208 .and_then(|value| value.parse().ok())
6209 .unwrap_or(defaults.max_age_days),
6210 }
6211}
6212
6213fn generate_launch_nonce() -> Result<String, SuperviseError> {
6216 let mut bytes = [0u8; 32];
6217 getrandom::getrandom(&mut bytes).map_err(|source| SuperviseError::LaunchNonce {
6218 reason: source.to_string(),
6219 })?;
6220 let mut hex = String::with_capacity(64);
6221 for b in bytes {
6222 use std::fmt::Write;
6223 let _ = write!(hex, "{b:02x}");
6224 }
6225 Ok(hex)
6226}
6227
6228fn constant_time_eq(a: &[u8], b: &[u8]) -> bool {
6231 if a.len() != b.len() {
6232 return false;
6233 }
6234 let mut diff = 0u8;
6235 for (x, y) in a.iter().zip(b.iter()) {
6236 diff |= x ^ y;
6237 }
6238 diff == 0
6239}
6240
6241fn spawn_and_mark_running(
6242 spec: &ModuleSpec,
6243 runtime: &SupervisorRuntimeConfig,
6244 snapshot: &SharedSnapshot,
6245) -> Result<SupervisedChild, SuperviseError> {
6246 let child = spawn_child(
6247 spec,
6248 runtime.connection_file_path.as_deref(),
6249 runtime.supervisor_handle.as_ref(),
6250 &runtime.stderr_ring,
6251 runtime.capture_logs_dir.as_deref(),
6252 &runtime.child_roster,
6253 #[cfg(target_os = "linux")]
6254 runtime.cgroup_placement.as_ref(),
6255 )?;
6256 set_running(snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
6257 Ok(child)
6258}
6259
6260enum RegistrationWaitOutcome {
6261 Registered,
6262 Exited(ExitReport),
6263 TimedOut,
6264}
6265
6266struct ReloadRegistrationFailure {
6267 exit_report: ExitReport,
6268 reason: String,
6269}
6270
6271#[derive(Debug, Clone, Copy, PartialEq, Eq)]
6272enum BusyGaugeObservation {
6273 Quiescent,
6274 Busy,
6275 Omitted,
6276}
6277
6278fn busy_gauge_observation(metrics: Option<&Value>, gauges: &[String]) -> BusyGaugeObservation {
6279 let Some(metrics) = metrics.and_then(Value::as_object) else {
6280 return BusyGaugeObservation::Omitted;
6281 };
6282 let mut sum = 0u128;
6283 for gauge in gauges {
6284 let Some(value) = metrics.get(gauge) else {
6285 return BusyGaugeObservation::Omitted;
6286 };
6287 let Some(value) = value.as_u64() else {
6288 return BusyGaugeObservation::Busy;
6289 };
6290 sum = sum.saturating_add(u128::from(value));
6291 }
6292 if sum == 0 {
6293 BusyGaugeObservation::Quiescent
6294 } else {
6295 BusyGaugeObservation::Busy
6296 }
6297}
6298
6299fn declared_busy_gauges(
6300 registry: &Registry,
6301 module_id: &str,
6302) -> Result<Vec<String>, SuperviseError> {
6303 busy_gauges_of(
6304 registry
6305 .get_module(module_id)
6306 .map_err(SuperviseError::Registry)?,
6307 )
6308}
6309
6310fn declared_busy_gauges_for_connection(
6314 registry: &Registry,
6315 connection_id: ConnectionId,
6316) -> Result<Vec<String>, SuperviseError> {
6317 busy_gauges_of(
6318 registry
6319 .get_module_by_connection(connection_id)
6320 .map_err(SuperviseError::Registry)?,
6321 )
6322}
6323
6324fn busy_gauges_of(
6325 registration: Option<crate::registry::ModuleRegistration>,
6326) -> Result<Vec<String>, SuperviseError> {
6327 let Some(registration) = registration else {
6328 return Ok(Vec::new());
6329 };
6330 let Some(self_signals) = registration.manifest.self_signals else {
6331 return Ok(Vec::new());
6332 };
6333
6334 let mut gauges = Vec::new();
6335 for declaration in self_signals {
6336 if declaration.kind != SelfSignalKind::Busy {
6337 continue;
6338 }
6339 match declaration.anchored_to {
6340 SignalAnchor::HealthGauges { gauges: declared } if !declared.is_empty() => {
6341 gauges.extend(declared)
6342 }
6343 _ => {
6344 gauges.push(String::new());
6347 }
6348 }
6349 }
6350 Ok(gauges)
6351}
6352
6353async fn wait_for_forwarding_quiescence(
6358 forwarding: &ForwardingTable,
6359 module_id: &str,
6360 runtime: &SupervisorRuntimeConfig,
6361 endpoint: crate::ModuleEndpointId,
6362 deadline: Instant,
6363 busy_gauges: &[String],
6364 scope: DrainScope,
6365) -> Result<bool, SuperviseError> {
6366 let mut gauges_quiescent = busy_gauges.is_empty();
6367 let mut next_probe_at = Instant::now();
6368 let mut omission_counted = false;
6369
6370 loop {
6371 let now = Instant::now();
6372 if !busy_gauges.is_empty() && now >= next_probe_at && now < deadline {
6373 let report = match scope {
6374 DrainScope::Active => probe_module_health(module_id, runtime, Some(deadline)).await,
6375 DrainScope::Endpoint(endpoint) => {
6376 probe_endpoint_health(endpoint, runtime, Some(deadline)).await
6377 }
6378 };
6379 gauges_quiescent = match report {
6380 Ok(report) => match busy_gauge_observation(report.metrics.as_ref(), busy_gauges) {
6381 BusyGaugeObservation::Quiescent => true,
6382 BusyGaugeObservation::Busy => false,
6383 BusyGaugeObservation::Omitted => {
6384 if !omission_counted {
6385 forwarding
6386 .counters()
6387 .increment_drains_with_undeclared_gauge();
6388 omission_counted = true;
6389 }
6390 false
6391 }
6392 },
6393 Err(err) => {
6394 warn!(
6395 module_id,
6396 error = %err,
6397 "drain health.check did not produce declared busy gauges; treating module as busy"
6398 );
6399 false
6400 }
6401 };
6402 next_probe_at = Instant::now() + runtime.health.cadence.max(REGISTRY_RELEASE_POLL);
6403 }
6404
6405 let in_flight = forwarding
6406 .endpoint_in_flight_count(endpoint)
6407 .map_err(SuperviseError::Forwarding)?;
6408 if in_flight == 0 && gauges_quiescent {
6409 return Ok(true);
6410 }
6411
6412 let now = Instant::now();
6413 if now >= deadline {
6414 return Ok(false);
6415 }
6416 let mut wait = deadline
6417 .saturating_duration_since(now)
6418 .min(REGISTRY_RELEASE_POLL);
6419 if !busy_gauges.is_empty() {
6420 wait = wait.min(next_probe_at.saturating_duration_since(now));
6421 }
6422 sleep(wait).await;
6423 }
6424}
6425
6426fn drained_after_quiescence_wait(wait_result: &Result<bool, SuperviseError>) -> bool {
6434 match wait_result {
6435 Ok(drained) => *drained,
6436 Err(_) => false,
6437 }
6438}
6439
6440fn send_route_goodbyes(forwarding: &ForwardingTable, released_routes: Vec<GoodbyeTarget>) {
6441 for released in released_routes {
6442 let frame = match Frame::build_with_version(
6443 released.negotiated_ver,
6444 FrameType::Goodbye,
6445 control_flags(),
6446 released.channel,
6447 released.epoch,
6448 0,
6449 Vec::new(),
6450 ) {
6451 Ok(frame) => frame,
6452 Err(err) => {
6453 warn!(
6454 route_channel = released.channel,
6455 error = %err,
6456 "failed to build supervisor drain route GOODBYE frame"
6457 );
6458 continue;
6459 }
6460 };
6461 if !released.close_on_delivery_failure() {
6462 crate::forwarding::send_module_route_goodbye(
6463 &forwarding.counters(),
6464 &released.sink,
6465 frame,
6466 released.module_id.as_deref(),
6467 "supervisor drain",
6468 );
6469 continue;
6470 }
6471 if let Err(err) = released.sink.try_send(frame) {
6472 warn!(
6473 target_connection_id = released.connection_id.get(),
6474 route_channel = released.channel,
6475 error = %err,
6476 "supervisor drain route GOODBYE was not delivered to client; closing target connection"
6477 );
6478 let _ = forwarding.escalate_client_delivery_failure(
6479 released.connection_id,
6480 released.channel,
6481 released.epoch,
6482 CloseReason::new(
6483 "route_goodbye_delivery_failed",
6484 format!(
6485 "failed to enqueue supervisor drain route GOODBYE for channel {}: {err}",
6486 released.channel
6487 ),
6488 ),
6489 crate::forwarding::UndeliveredFrame {
6490 module_id: released.module_id.as_deref(),
6491 sink: &released.sink,
6492 },
6493 );
6494 }
6495 }
6496}
6497
6498fn send_module_draining(
6499 module_id: &str,
6500 reason: RouteCloseReason,
6501 deadline_ms: u64,
6502 target: &ModuleDrainTarget,
6503) {
6504 let body = match serde_json::to_vec(&ModuleControlCommand::Draining {
6505 reason,
6506 deadline_ms,
6507 }) {
6508 Ok(body) => body,
6509 Err(err) => {
6510 warn!(
6511 module_id,
6512 error = %err,
6513 "failed to encode module draining command"
6514 );
6515 return;
6516 }
6517 };
6518 let frame = match Frame::build_with_version(
6519 target.negotiated_ver,
6520 FrameType::Push,
6521 control_flags(),
6522 0,
6523 0,
6524 0,
6525 body,
6526 ) {
6527 Ok(frame) => frame,
6528 Err(err) => {
6529 warn!(
6530 module_id,
6531 error = %err,
6532 "failed to build module draining command frame"
6533 );
6534 return;
6535 }
6536 };
6537 if let Err(err) = target.sink.try_send(frame) {
6538 warn!(
6539 module_id,
6540 target_connection_id = target.endpoint.connection_id.get(),
6541 error = %err,
6542 "module draining command was not delivered to peer"
6543 );
6544 }
6545}
6546
6547fn module_goodbye_frame(module_id: &str, negotiated_ver: u8) -> Option<Frame> {
6549 match Frame::build_with_version(
6550 negotiated_ver,
6551 FrameType::Goodbye,
6552 control_flags(),
6553 0,
6554 0,
6555 0,
6556 Vec::new(),
6557 ) {
6558 Ok(frame) => Some(frame),
6559 Err(err) => {
6560 warn!(
6561 module_id,
6562 error = %err,
6563 "failed to build module GOODBYE frame"
6564 );
6565 None
6566 }
6567 }
6568}
6569
6570#[cfg(unix)]
6584async fn send_module_goodbyes_for_daemon_shutdown(
6585 forwarding: &Arc<ForwardingTable>,
6586 reason: &CloseReason,
6587 wait_for_flush: bool,
6588) {
6589 const GOODBYE_BUDGET: Duration = Duration::from_millis(500);
6590 let targets = match forwarding.module_connections() {
6591 Ok(targets) => targets,
6592 Err(err) => {
6593 warn!(error = %err, "could not list module connections for shutdown GOODBYE");
6594 return;
6595 }
6596 };
6597 let deadline = Instant::now() + GOODBYE_BUDGET;
6598 let mut sends = tokio::task::JoinSet::new();
6599 for target in targets {
6600 let Some(frame) = module_goodbye_frame(&target.module_id, target.negotiated_ver) else {
6601 continue;
6602 };
6603 if !wait_for_flush {
6604 if let Err(err) = target.sink.try_send(frame) {
6605 debug!(
6606 module_id = %target.module_id,
6607 error = %err,
6608 "shutdown module GOODBYE was not queued"
6609 );
6610 }
6611 continue;
6612 }
6613 let forwarding = Arc::clone(forwarding);
6614 let reason = reason.clone();
6615 sends.spawn(async move {
6616 match timeout_at(deadline, target.sink.send_flushed(frame)).await {
6617 Ok(Ok(())) => {}
6618 Ok(Err(err)) => debug!(
6619 module_id = %target.module_id,
6620 error = %err,
6621 "module connection closed before its shutdown GOODBYE was written"
6622 ),
6623 Err(_) => warn!(
6624 module_id = %target.module_id,
6625 budget = ?GOODBYE_BUDGET,
6626 "shutdown module GOODBYE was not written within its budget; closing anyway"
6627 ),
6628 }
6629 forwarding.request_connection_close(target.endpoint.connection_id, reason);
6630 });
6631 }
6632 while sends.join_next().await.is_some() {}
6634}
6635
6636fn send_module_goodbye(module_id: &str, forwarding: &ForwardingTable, target: &ModuleDrainTarget) {
6637 let Some(frame) = module_goodbye_frame(module_id, target.negotiated_ver) else {
6638 return;
6639 };
6640 if let Err(err) = target.sink.try_send(frame) {
6641 warn!(
6642 module_id,
6643 target_connection_id = target.endpoint.connection_id.get(),
6644 error = %err,
6645 "supervisor drain module GOODBYE was not delivered to peer; closing module connection"
6646 );
6647 forwarding.request_connection_close(
6648 target.endpoint.connection_id,
6649 CloseReason::new(
6650 "module_goodbye_delivery_failed",
6651 format!("failed to enqueue supervisor drain module GOODBYE for module '{module_id}': {err}"),
6652 ),
6653 );
6654 }
6655}
6656
6657#[derive(Clone, Copy)]
6658struct ForwardingDrainContext<'a> {
6659 spec: &'a ModuleSpec,
6660 runtime: &'a SupervisorRuntimeConfig,
6661 registry: &'a Registry,
6662 scope: DrainScope,
6663}
6664
6665#[derive(Debug, Clone, Copy, PartialEq, Eq)]
6667enum DrainScope {
6668 Active,
6671 Endpoint(crate::ModuleEndpointId),
6676}
6677
6678#[derive(Debug, Clone, Copy, PartialEq, Eq)]
6686enum StopNotice {
6687 SentOverConnection,
6690 NoConnection,
6694 NotSent,
6698}
6699
6700async fn begin_forwarding_drain(
6701 spec: &ModuleSpec,
6702 runtime: &SupervisorRuntimeConfig,
6703 registry: &Registry,
6704 snapshot: &SharedSnapshot,
6705 enabled: Option<bool>,
6706 reason: RouteCloseReason,
6707) -> Result<StopNotice, SuperviseError> {
6708 let Some(forwarding) = runtime.forwarding.as_ref() else {
6709 return Err(SuperviseError::ReloadUnavailable {
6710 module_id: spec.module_id.clone(),
6711 reason: "supervisor was not configured with a forwarding table".to_string(),
6712 });
6713 };
6714
6715 begin_forwarding_drain_with(
6716 forwarding,
6717 ForwardingDrainContext {
6718 spec,
6719 runtime,
6720 registry,
6721 scope: DrainScope::Active,
6722 },
6723 snapshot,
6724 enabled,
6725 reason,
6726 runtime.drain_timeout,
6727 )
6728 .await
6729}
6730
6731async fn begin_forwarding_drain_if_configured(
6732 spec: &ModuleSpec,
6733 runtime: &SupervisorRuntimeConfig,
6734 registry: &Registry,
6735 snapshot: &SharedSnapshot,
6736 enabled: Option<bool>,
6737 reason: RouteCloseReason,
6738) -> Result<StopNotice, SuperviseError> {
6739 begin_forwarding_drain_with_timeout(
6740 spec,
6741 runtime,
6742 registry,
6743 snapshot,
6744 enabled,
6745 reason,
6746 runtime.drain_timeout,
6747 )
6748 .await
6749}
6750
6751async fn begin_forwarding_drain_with_timeout(
6755 spec: &ModuleSpec,
6756 runtime: &SupervisorRuntimeConfig,
6757 registry: &Registry,
6758 snapshot: &SharedSnapshot,
6759 enabled: Option<bool>,
6760 reason: RouteCloseReason,
6761 drain_timeout: Duration,
6762) -> Result<StopNotice, SuperviseError> {
6763 let Some(forwarding) = runtime.forwarding.as_ref() else {
6764 return Ok(StopNotice::NotSent);
6765 };
6766
6767 begin_forwarding_drain_with(
6768 forwarding,
6769 ForwardingDrainContext {
6770 spec,
6771 runtime,
6772 registry,
6773 scope: DrainScope::Active,
6774 },
6775 snapshot,
6776 enabled,
6777 reason,
6778 drain_timeout,
6779 )
6780 .await
6781}
6782
6783async fn begin_forwarding_drain_with(
6784 forwarding: &ForwardingTable,
6785 context: ForwardingDrainContext<'_>,
6786 snapshot: &SharedSnapshot,
6787 enabled: Option<bool>,
6788 reason: RouteCloseReason,
6789 drain_timeout: Duration,
6790) -> Result<StopNotice, SuperviseError> {
6791 let ForwardingDrainContext {
6792 spec,
6793 runtime,
6794 registry,
6795 scope,
6796 } = context;
6797 debug_assert_ne!(reason, RouteCloseReason::Crash);
6798 let terminal = matches!(reason, RouteCloseReason::Disable);
6799 let drain_started_at = Instant::now();
6800 let drain_deadline = drain_started_at + drain_timeout;
6801 let deadline_ms =
6802 unix_ms_now().saturating_add(u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX));
6803 let busy_gauges = match scope {
6804 DrainScope::Active => declared_busy_gauges(registry, &spec.module_id)?,
6805 DrainScope::Endpoint(endpoint) => {
6806 declared_busy_gauges_for_connection(registry, endpoint.connection_id)?
6807 }
6808 };
6809
6810 let gate_started = Instant::now();
6813 let drain_target = match scope {
6814 DrainScope::Active => forwarding.begin_module_drain(&spec.module_id, reason),
6815 DrainScope::Endpoint(endpoint) => forwarding.begin_endpoint_drain(endpoint, reason),
6816 }
6817 .map_err(SuperviseError::Forwarding)?;
6818 info!(
6823 module_id = %spec.module_id,
6824 ?reason,
6825 gate_ms = u64::try_from(gate_started.elapsed().as_millis()).unwrap_or(u64::MAX),
6826 connected = drain_target.is_some(),
6827 "module drain began; route admission closed"
6828 );
6829 if scope == DrainScope::Active {
6830 update_snapshot(snapshot, Some(&spec.module_id), |state| {
6831 state.state = ModuleState::Draining;
6832 state.draining_to_replace =
6833 matches!(reason, RouteCloseReason::Restart | RouteCloseReason::Reload);
6834 if let Some(enabled) = enabled {
6835 state.enabled = enabled;
6836 }
6837 })?;
6838 }
6839
6840 let Some(target) = drain_target.as_ref() else {
6841 return Ok(StopNotice::NoConnection);
6845 };
6846 {
6847 send_module_draining(&spec.module_id, reason, deadline_ms, target);
6848 let routes = forwarding
6849 .endpoint_routes(target.endpoint)
6850 .map_err(SuperviseError::Forwarding)?;
6851 let routes_notified = routes.len();
6852 crate::control::send_route_control_pushes(
6853 forwarding,
6854 routes.clone(),
6855 ClientControlPush::RouteClosing {
6856 module_id: spec.module_id.clone(),
6857 reason,
6858 },
6859 );
6860 send_route_goodbyes(forwarding, target.abandoned_bindings.clone());
6861
6862 let wait_result = wait_for_forwarding_quiescence(
6868 forwarding,
6869 &spec.module_id,
6870 runtime,
6871 target.endpoint,
6872 drain_deadline,
6873 &busy_gauges,
6874 scope,
6875 )
6876 .await;
6877 let drained = drained_after_quiescence_wait(&wait_result);
6878 if let Err(err) = &wait_result {
6879 error!(
6880 module_id = %spec.module_id,
6881 ?reason,
6882 error = %err,
6883 "forwarding quiescence wait failed after route.closing; forcing route.closed(drained: false) so the client is not left waiting on an unfulfilled promise"
6884 );
6885 } else if !drained {
6886 let holdouts = forwarding
6892 .endpoint_drain_holdouts(target.endpoint)
6893 .unwrap_or_default();
6894 warn!(
6895 module_id = %spec.module_id,
6896 waited = ?drain_timeout,
6897 ?reason,
6898 held_requests = holdouts.requests,
6899 held_routes = holdouts.routes,
6900 total_routes = holdouts.total_routes,
6901 top_connections = ?holdouts.top_connections,
6902 held = %holdouts
6905 .held
6906 .iter()
6907 .map(|(channel, corr)| format!("{channel}:{corr}"))
6908 .collect::<Vec<_>>()
6909 .join(","),
6910 "route drain timed out before request quiescence; forcing teardown"
6911 );
6912 }
6913 crate::control::send_route_control_pushes(
6914 forwarding,
6915 routes,
6916 ClientControlPush::RouteClosed {
6917 module_id: spec.module_id.clone(),
6918 reason,
6919 drained,
6920 abandoned: target.abandoned_bindings.len() as u32,
6921 excluded_subscriptions: target.excluded_subscriptions,
6922 terminal: Some(terminal),
6923 },
6924 );
6925 wait_result?;
6926
6927 let released_routes = match forwarding.release_module_endpoint_routes(target.endpoint) {
6933 Ok(routes) => routes,
6934 Err(err) => {
6935 warn!(
6936 module_id = %spec.module_id,
6937 ?reason,
6938 error = %err,
6939 "failed to release module endpoint routes after route.closed; module GOODBYE will still be sent"
6940 );
6941 send_module_goodbye(&spec.module_id, forwarding, target);
6942 return Err(SuperviseError::Forwarding(err));
6943 }
6944 };
6945 let route_goodbye_count = released_routes.len();
6946 send_route_goodbyes(forwarding, released_routes);
6947 send_module_goodbye(&spec.module_id, forwarding, target);
6948
6949 info!(
6955 module_id = %spec.module_id,
6956 ?reason,
6957 routes_notified,
6958 route_goodbyes = route_goodbye_count,
6959 abandoned_reservations = target.abandoned_bindings.len(),
6960 excluded_subscriptions = target.excluded_subscriptions,
6961 drained,
6962 "module drain complete; consumers notified via route.closing/route.closed pushes and per-route GOODBYE frames"
6963 );
6964 }
6965
6966 Ok(StopNotice::SentOverConnection)
6967}
6968
6969async fn wait_for_registration_after_reload(
6972 registry: &Registry,
6973 module_id: &str,
6974 snapshot: &SharedSnapshot,
6975 child: &mut SupervisedChild,
6976 wait: Duration,
6977) -> Result<RegistrationWaitOutcome, SuperviseError> {
6978 wait_for_slot_registration(
6979 registry,
6980 crate::registry::RegistrationSlot::Active(module_id),
6981 module_id,
6982 snapshot,
6983 child,
6984 wait,
6985 )
6986 .await
6987}
6988
6989async fn wait_for_slot_registration(
6997 registry: &Registry,
6998 slot: crate::registry::RegistrationSlot<'_>,
6999 module_id: &str,
7000 snapshot: &SharedSnapshot,
7001 child: &mut SupervisedChild,
7002 wait: Duration,
7003) -> Result<RegistrationWaitOutcome, SuperviseError> {
7004 let deadline = Instant::now() + wait;
7005 loop {
7006 if registry
7007 .registration(slot)
7008 .map_err(SuperviseError::Registry)?
7009 .is_some()
7010 {
7011 return Ok(RegistrationWaitOutcome::Registered);
7012 }
7013
7014 let now = Instant::now();
7015 if now >= deadline {
7016 return Ok(RegistrationWaitOutcome::TimedOut);
7017 }
7018 let remaining = deadline.saturating_duration_since(now);
7019 let poll = remaining.min(REGISTRY_RELEASE_POLL);
7020
7021 tokio::select! {
7022 wait_result = child.wait() => {
7023 let status = wait_result.map_err(|source| SuperviseError::Wait {
7024 module_id: module_id.to_string(),
7025 source,
7026 })?;
7027 return Ok(RegistrationWaitOutcome::Exited(classify_reaped_child_exit(
7028 snapshot,
7029 child,
7030 &status,
7031 )));
7032 }
7033 _ = sleep(poll) => {}
7034 }
7035 }
7036}
7037
7038fn registration_failure_exit_report(mut exit_report: ExitReport) -> ExitReport {
7039 if exit_report.kind != ExitKind::DeliberateSeverance {
7042 exit_report.kind = ExitKind::Crash;
7043 }
7044 exit_report
7045}
7046
7047async fn handle_reload_child_registration_failure(
7048 spec: &ModuleSpec,
7049 runtime: &SupervisorRuntimeConfig,
7050 registry: &Registry,
7051 process_liveness: &SupervisorProcessLiveness,
7052 snapshot: &SharedSnapshot,
7053 child: &mut Option<SupervisedChild>,
7054 failure: ReloadRegistrationFailure,
7055) -> Result<(), SuperviseError> {
7056 let ReloadRegistrationFailure {
7057 exit_report,
7058 reason,
7059 } = failure;
7060 match on_child_exit(
7061 spec,
7062 runtime.restart_policy,
7063 registry,
7064 snapshot,
7065 &runtime.terminal_ring,
7066 &runtime.spawn_events,
7067 &runtime.child_roster,
7068 exit_report,
7069 )
7070 .await
7071 {
7072 NextAction::Stop {
7073 registration_released,
7074 } => {
7075 if registration_released {
7076 process_liveness.untrack_if_current(&spec.module_id, snapshot);
7077 }
7078 }
7079 NextAction::Restart { schedule } => {
7080 let delay = schedule.map_or(runtime.restart_policy.delay_for_restart(0), |schedule| {
7081 schedule.delay
7082 });
7083 if let Some(schedule) = schedule {
7084 log_crash_respawn(&spec.module_id, schedule);
7085 }
7086 sleep(delay).await;
7087 if respawn_still_pending(snapshot) {
7091 if let Err(err) = wait_for_registration_release(
7092 registry,
7093 &spec.module_id,
7094 REGISTRY_RELEASE_TIMEOUT,
7095 )
7096 .await
7097 {
7098 fail_snapshot(snapshot, Some(&spec.module_id), None);
7099 process_liveness.untrack_if_current(&spec.module_id, snapshot);
7100 return Err(SuperviseError::ReloadFailed {
7101 module_id: spec.module_id.clone(),
7102 reason: format!(
7103 "{reason}; registration did not release before policy retry: {err}"
7104 ),
7105 });
7106 }
7107 process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
7108 match spawn_and_mark_running(spec, runtime, snapshot) {
7109 Ok(next_child) => {
7110 *child = Some(next_child);
7111 }
7112 Err(err) => {
7113 fail_snapshot(snapshot, Some(&spec.module_id), None);
7114 process_liveness.untrack_if_current(&spec.module_id, snapshot);
7115 return Err(SuperviseError::ReloadFailed {
7116 module_id: spec.module_id.clone(),
7117 reason: format!("{reason}; policy retry spawn failed: {err}"),
7118 });
7119 }
7120 }
7121 }
7122 }
7123 }
7124
7125 Err(SuperviseError::ReloadFailed {
7126 module_id: spec.module_id.clone(),
7127 reason,
7128 })
7129}
7130
7131async fn handle_reload_spawn_failure(
7132 spec: &ModuleSpec,
7133 runtime: &SupervisorRuntimeConfig,
7134 process_liveness: &SupervisorProcessLiveness,
7135 snapshot: &SharedSnapshot,
7136 child: &mut Option<SupervisedChild>,
7137 reason: String,
7138) -> Result<(), SuperviseError> {
7139 let mut should_retry = false;
7140 let now = Instant::now();
7141 update_snapshot(snapshot, Some(&spec.module_id), |state| {
7142 clear_current_process_facts(state);
7143 if daemon_will_restart(state, &runtime.restart_policy, now) {
7144 state.record_crash_restart(&runtime.restart_policy, now);
7145 state.state = ModuleState::Restarting;
7146 should_retry = true;
7147 } else if state.enabled {
7148 state.state = ModuleState::Failed;
7149 } else {
7150 state.state = ModuleState::Disabled;
7151 }
7152 })?;
7153
7154 if should_retry {
7155 sleep(runtime.restart_policy.backoff).await;
7156 if respawn_still_pending(snapshot) {
7160 process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
7161 match spawn_and_mark_running(spec, runtime, snapshot) {
7162 Ok(next_child) => {
7163 *child = Some(next_child);
7164 }
7165 Err(err) => {
7166 fail_snapshot(snapshot, Some(&spec.module_id), None);
7167 process_liveness.untrack_if_current(&spec.module_id, snapshot);
7168 return Err(SuperviseError::ReloadFailed {
7169 module_id: spec.module_id.clone(),
7170 reason: format!("{reason}; policy retry spawn failed: {err}"),
7171 });
7172 }
7173 }
7174 }
7175 } else {
7176 process_liveness.untrack_if_current(&spec.module_id, snapshot);
7177 }
7178
7179 Err(SuperviseError::ReloadFailed {
7180 module_id: spec.module_id.clone(),
7181 reason,
7182 })
7183}
7184
7185fn control_flags() -> Flags {
7186 Flags::new(false, Priority::Passive, false)
7187}
7188
7189#[allow(clippy::too_many_arguments)]
7190async fn drain_optional_child(
7191 module_id: &str,
7192 protocol: ModuleProtocol,
7193 stop_notice: StopNotice,
7194 registry: &Registry,
7195 snapshot: &SharedSnapshot,
7196 terminal_ring: &Arc<Mutex<TerminalRing>>,
7197 spawn_events: &SpawnEventFeed,
7198 child: &mut Option<SupervisedChild>,
7199 drain_timeout: Duration,
7200 final_state: ModuleState,
7201 enabled: Option<bool>,
7202) -> Result<(), SuperviseError> {
7203 if let Some(child) = child.take() {
7204 drain_child_to_state(
7205 module_id,
7206 protocol,
7207 stop_notice,
7208 registry,
7209 snapshot,
7210 terminal_ring,
7211 spawn_events,
7212 child,
7213 drain_timeout,
7214 final_state,
7215 enabled,
7216 )
7217 .await
7218 } else {
7219 update_snapshot(snapshot, Some(module_id), |state| {
7220 state.state = final_state;
7221 if let Some(enabled) = enabled {
7222 state.enabled = enabled;
7223 }
7224 clear_current_process_facts(state);
7225 })?;
7226 wait_for_registration_release(registry, module_id, REGISTRY_RELEASE_TIMEOUT).await
7227 }
7228}
7229
7230#[allow(clippy::too_many_arguments)]
7231async fn drain_child_to_state(
7232 module_id: &str,
7233 protocol: ModuleProtocol,
7234 stop_notice: StopNotice,
7235 registry: &Registry,
7236 snapshot: &SharedSnapshot,
7237 terminal_ring: &Arc<Mutex<TerminalRing>>,
7238 spawn_events: &SpawnEventFeed,
7239 mut child: SupervisedChild,
7240 drain_timeout: Duration,
7241 final_state: ModuleState,
7242 enabled: Option<bool>,
7243) -> Result<(), SuperviseError> {
7244 update_snapshot(snapshot, Some(module_id), |state| {
7245 state.state = ModuleState::Draining;
7246 state.draining_to_replace = final_state == ModuleState::Restarting;
7247 if let Some(enabled) = enabled {
7248 state.enabled = enabled;
7249 }
7250 })?;
7251
7252 if stop_notice != StopNotice::SentOverConnection {
7263 if protocol == ModuleProtocol::Subc && stop_notice == StopNotice::NoConnection {
7264 info!(
7265 module_id,
7266 pid = child.pid,
7267 budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
7268 "module has no connection yet; requesting stop by signal"
7269 );
7270 }
7271 request_graceful_stop(module_id, &child);
7272 }
7273
7274 let exit_report = match timeout(drain_timeout, child.wait()).await {
7275 Ok(Ok(status)) => classify_reaped_child_exit(snapshot, &child, &status),
7276 Ok(Err(source)) => {
7277 fail_snapshot(snapshot, Some(module_id), None);
7278 return Err(SuperviseError::Wait {
7279 module_id: module_id.to_string(),
7280 source,
7281 });
7282 }
7283 Err(_) => {
7284 warn!(
7297 module_id,
7298 pid = child.pid,
7299 budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
7300 reason = ?final_state,
7301 ?stop_notice,
7302 "drain budget expired before the module exited; killing it"
7303 );
7304 child.start_kill().map_err(|source| {
7305 fail_snapshot(snapshot, Some(module_id), None);
7306 SuperviseError::Kill {
7307 module_id: module_id.to_string(),
7308 source,
7309 }
7310 })?;
7311 let status = child.wait().await.map_err(|source| {
7312 fail_snapshot(snapshot, Some(module_id), None);
7313 SuperviseError::Wait {
7314 module_id: module_id.to_string(),
7315 source,
7316 }
7317 })?;
7318 classify_reaped_child_exit(snapshot, &child, &status)
7319 }
7320 };
7321
7322 update_snapshot(snapshot, Some(module_id), |state| {
7323 state.state = final_state;
7324 if let Some(enabled) = enabled {
7325 state.enabled = enabled;
7326 }
7327 clear_current_process_facts(state);
7328 state.last_exit = Some(exit_report.clone());
7329 if exit_report.kind == ExitKind::DeliberateSeverance {
7330 state.lifetime_restarts += 1;
7331 }
7332 })?;
7333 record_terminal(
7334 module_id,
7335 terminal_ring,
7336 spawn_events,
7337 &exit_report,
7338 terminal_disposition(final_state),
7339 );
7340 child.drain_stderr(module_id).await;
7341
7342 wait_for_registration_release(registry, module_id, REGISTRY_RELEASE_TIMEOUT).await
7343}
7344
7345#[cfg(unix)]
7365fn request_graceful_stop(module_id: &str, child: &SupervisedChild) {
7366 let Some(pid) = child
7367 .id()
7368 .and_then(|pid| i32::try_from(pid).ok())
7369 .and_then(rustix::process::Pid::from_raw)
7370 else {
7371 debug!(
7372 module_id,
7373 "no pid to signal for teardown; falling through to the drain wait"
7374 );
7375 return;
7376 };
7377 match rustix::process::kill_process(pid, rustix::process::Signal::TERM) {
7378 Ok(()) => debug!(
7379 module_id,
7380 "sent SIGTERM to a module nothing else asked to stop"
7381 ),
7382 Err(err) => debug!(
7383 module_id,
7384 error = %err,
7385 "SIGTERM to module failed; the drain wait and kill still apply"
7386 ),
7387 }
7388}
7389
7390#[cfg(not(unix))]
7398fn request_graceful_stop(module_id: &str, _child: &SupervisedChild) {
7399 debug!(
7400 module_id,
7401 "no graceful stop signal exists on this platform; teardown of a module nothing asked to stop waits, then kills"
7402 );
7403}
7404
7405fn terminal_disposition(final_state: ModuleState) -> TerminalDisposition {
7406 match final_state {
7407 ModuleState::Stopped => TerminalDisposition::Stopped,
7408 ModuleState::Disabled => TerminalDisposition::Disabled,
7409 ModuleState::Restarting => TerminalDisposition::Restarting,
7410 ModuleState::Failed => TerminalDisposition::Failed,
7411 ModuleState::Starting
7412 | ModuleState::Running
7413 | ModuleState::Unresponsive
7414 | ModuleState::Draining => {
7415 unreachable!("terminal exits only finish in terminal or restarting states")
7416 }
7417 }
7418}
7419
7420async fn wait_for_registration_release(
7423 registry: &Registry,
7424 module_id: &str,
7425 wait: Duration,
7426) -> Result<(), SuperviseError> {
7427 wait_for_slot_registration_release(
7428 registry,
7429 crate::registry::RegistrationSlot::Active(module_id),
7430 wait,
7431 )
7432 .await
7433}
7434
7435async fn wait_for_slot_registration_release(
7443 registry: &Registry,
7444 slot: crate::registry::RegistrationSlot<'_>,
7445 wait: Duration,
7446) -> Result<(), SuperviseError> {
7447 let deadline = Instant::now() + wait;
7448 let mut release_events = registration_release_events().subscribe();
7449 let still_active = |registration: &crate::registry::ModuleRegistration| {
7450 SuperviseError::RegistrationStillActive {
7451 module_id: registration.manifest.module_id.clone(),
7452 waited: wait,
7453 }
7454 };
7455 loop {
7456 let _observed_generation = *release_events.borrow_and_update();
7457 let Some(registration) = registry
7458 .registration(slot)
7459 .map_err(SuperviseError::Registry)?
7460 else {
7461 return Ok(());
7462 };
7463
7464 let now = Instant::now();
7465 if now >= deadline {
7466 return Err(still_active(®istration));
7467 }
7468
7469 let remaining = deadline.saturating_duration_since(now);
7470 match timeout(remaining, release_events.changed()).await {
7471 Ok(Ok(())) | Ok(Err(_)) => {}
7472 Err(_) => return Err(still_active(®istration)),
7473 }
7474 }
7475}
7476
7477#[cfg(test)]
7478mod slot_registration_wait_tests {
7479 use super::*;
7480 use crate::registry::{ConnectionId, RegistrationSlot};
7481 use subc_protocol::manifest::ModuleManifest;
7482
7483 const INCUMBENT: u64 = 1;
7484 const CANDIDATE: u64 = 2;
7485
7486 fn swapped_registry() -> Arc<Registry> {
7487 let registry = Arc::new(Registry::default());
7488 let manifest = ModuleManifest::builder("m", "0.1.0").build();
7489 registry
7490 .register_with_control_ops(
7491 manifest.clone(),
7492 1,
7493 ConnectionId::new(INCUMBENT),
7494 Vec::new(),
7495 )
7496 .unwrap();
7497 registry
7498 .register_candidate_with_control_ops(
7499 manifest,
7500 1,
7501 ConnectionId::new(CANDIDATE),
7502 Vec::new(),
7503 )
7504 .unwrap();
7505 registry
7506 }
7507
7508 #[tokio::test]
7512 async fn incumbent_release_is_awaited_by_connection_not_by_module_id() {
7513 let registry = swapped_registry();
7514 registry.promote_candidate("m").unwrap().unwrap();
7515
7516 assert!(matches!(
7517 wait_for_registration_release(®istry, "m", Duration::from_millis(50)).await,
7518 Err(SuperviseError::RegistrationStillActive { .. })
7519 ));
7520
7521 assert!(matches!(
7523 wait_for_slot_registration_release(
7524 ®istry,
7525 RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
7526 Duration::from_millis(50),
7527 )
7528 .await,
7529 Err(SuperviseError::RegistrationStillActive { .. })
7530 ));
7531
7532 let releaser = Arc::clone(®istry);
7533 let release = tokio::spawn(async move {
7534 sleep(Duration::from_millis(20)).await;
7535 releaser
7536 .deregister_connection(ConnectionId::new(INCUMBENT))
7537 .unwrap();
7538 notify_registration_release();
7539 });
7540 wait_for_slot_registration_release(
7541 ®istry,
7542 RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
7543 Duration::from_secs(5),
7544 )
7545 .await
7546 .expect("the incumbent's own registration is released");
7547 release.await.unwrap();
7548 assert!(registry.get_module("m").unwrap().is_some());
7549 }
7550
7551 #[tokio::test]
7554 async fn candidate_slot_wait_ignores_the_incumbents_registration() {
7555 let registry = swapped_registry();
7556 assert!(matches!(
7557 wait_for_slot_registration_release(
7558 ®istry,
7559 RegistrationSlot::Candidate("m"),
7560 Duration::from_millis(50),
7561 )
7562 .await,
7563 Err(SuperviseError::RegistrationStillActive { .. })
7564 ));
7565 registry
7566 .deregister_connection(ConnectionId::new(CANDIDATE))
7567 .unwrap();
7568 wait_for_slot_registration_release(
7569 ®istry,
7570 RegistrationSlot::Candidate("m"),
7571 Duration::from_millis(50),
7572 )
7573 .await
7574 .expect("a candidate slot with no candidate is released");
7575 assert!(registry
7576 .registration(RegistrationSlot::Active("m"))
7577 .unwrap()
7578 .is_some());
7579 }
7580}
7581
7582fn classify_exit(status: &ExitStatus) -> ExitReport {
7583 ExitReport {
7584 kind: if status.success() {
7585 ExitKind::Clean
7586 } else {
7587 ExitKind::Crash
7588 },
7589 code: status.code(),
7590 signal: exit_signal(status),
7591 at_ms: unix_ms_now(),
7592 }
7593}
7594
7595fn wait_error_exit_report() -> ExitReport {
7601 ExitReport {
7602 kind: ExitKind::Crash,
7603 code: None,
7604 signal: None,
7605 at_ms: unix_ms_now(),
7606 }
7607}
7608
7609#[cfg(unix)]
7610fn exit_signal(status: &ExitStatus) -> Option<i32> {
7611 use std::os::unix::process::ExitStatusExt;
7612
7613 status.signal()
7614}
7615
7616#[cfg(not(unix))]
7617fn exit_signal(_status: &ExitStatus) -> Option<i32> {
7618 None
7619}
7620
7621fn reset_restart_count(snapshot: &SharedSnapshot, module_id: &str) -> Result<(), SuperviseError> {
7627 update_snapshot(snapshot, Some(module_id), |state| {
7628 state.clear_crash_restarts();
7629 })
7630}
7631
7632fn set_running(
7633 snapshot: &SharedSnapshot,
7634 child: &SupervisedChild,
7635 module_id: &str,
7636 spawn_events: &SpawnEventFeed,
7637) -> Result<(), SuperviseError> {
7638 let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
7639 module_id: Some(module_id.to_string()),
7640 })?;
7641 state.spawn_generation = spawn_events.emit_spawned(module_id, child.pid, child.spawned_at_ms);
7642 state.in_alternate_slot = false;
7645 state.configuration_updated_since_spawn = false;
7646 state.state = ModuleState::Running;
7647 state.enabled = true;
7648 state.process_alive = true;
7649 state.pid = child.id();
7650 state.spawned_at_ms = Some(child.spawned_at_ms);
7651 state.spawned_from = Some(child.spawned_from.clone());
7652 state.spawned_file_identity = child.spawned_file_identity;
7653 state.process_start_time = child.process_start_time;
7654 Ok(())
7655}
7656
7657fn clear_current_process_facts(state: &mut SupervisorSnapshot) {
7658 state.process_alive = false;
7659 state.pid = None;
7660 state.spawned_at_ms = None;
7661 state.spawned_from = None;
7662 state.spawned_file_identity = None;
7663 state.process_start_time = None;
7664 state.deliberate_severance = None;
7665}
7666
7667#[cfg(test)]
7668fn record_deliberate_severance(
7669 snapshot: &SharedSnapshot,
7670 identity: ProcessIdentity,
7671) -> Result<(), SuperviseError> {
7672 update_snapshot(snapshot, None, |state| {
7673 state.deliberate_severance = Some(identity);
7674 })
7675}
7676
7677fn apply_deliberate_severance_marker(
7678 snapshot: &SharedSnapshot,
7679 exited_identity: Option<ProcessIdentity>,
7680 mut exit_report: ExitReport,
7681) -> ExitReport {
7682 let marker = lock_snapshot(snapshot)
7683 .ok()
7684 .and_then(|mut state| state.deliberate_severance.take());
7685 if marker.is_some() && marker == exited_identity {
7686 exit_report.kind = ExitKind::DeliberateSeverance;
7687 }
7688 exit_report
7689}
7690
7691fn classify_reaped_child_exit(
7692 snapshot: &SharedSnapshot,
7693 child: &SupervisedChild,
7694 status: &ExitStatus,
7695) -> ExitReport {
7696 apply_deliberate_severance_marker(snapshot, child.process_identity(), classify_exit(status))
7697}
7698
7699fn fail_snapshot(
7700 snapshot: &SharedSnapshot,
7701 module_id: Option<&str>,
7702 last_exit: Option<ExitReport>,
7703) {
7704 if let Err(err) = update_snapshot(snapshot, module_id, |state| {
7705 state.state = ModuleState::Failed;
7706 clear_current_process_facts(state);
7707 if let Some(last_exit) = last_exit {
7708 state.last_exit = Some(last_exit);
7709 }
7710 }) {
7711 error!(error = %err, "failed to mark supervisor state failed");
7712 }
7713}
7714
7715fn update_snapshot(
7716 snapshot: &SharedSnapshot,
7717 module_id: Option<&str>,
7718 update: impl FnOnce(&mut SupervisorSnapshot),
7719) -> Result<(), SuperviseError> {
7720 let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
7721 module_id: module_id.map(ToOwned::to_owned),
7722 })?;
7723 update(&mut state);
7724 Ok(())
7725}
7726
7727const SLOW_SNAPSHOT_LOCK_THRESHOLD: Duration = Duration::from_millis(250);
7728
7729fn lock_snapshot_for_control<'a>(
7730 snapshot: &'a SharedSnapshot,
7731 module_id: &str,
7732 caller: &'static str,
7733) -> Result<std::sync::MutexGuard<'a, SupervisorSnapshot>, SuperviseError> {
7734 let started_at = Instant::now();
7735 let guard = lock_snapshot(snapshot)?;
7736 let waited = started_at.elapsed();
7737 if waited >= SLOW_SNAPSHOT_LOCK_THRESHOLD {
7738 warn!(
7739 module_id = %module_id,
7740 waited_ms = waited.as_millis() as u64,
7741 caller = %caller,
7742 "slow snapshot lock"
7743 );
7744 }
7745 Ok(guard)
7746}
7747
7748fn lock_snapshot(
7749 snapshot: &SharedSnapshot,
7750) -> Result<std::sync::MutexGuard<'_, SupervisorSnapshot>, SuperviseError> {
7751 snapshot
7752 .lock()
7753 .map_err(|_| SuperviseError::StatePoisoned { module_id: None })
7754}
7755
7756#[cfg(test)]
7757mod terminal_history_tests {
7758 use std::{
7759 path::PathBuf,
7760 sync::Arc,
7761 time::{Duration, Instant},
7762 };
7763
7764 use tokio::time::sleep;
7765
7766 use super::{
7767 apply_deliberate_severance_marker, daemon_will_restart, drain_child_to_state,
7768 drained_after_quiescence_wait, handle_reload_spawn_failure, health_restart_child,
7769 lock_snapshot, on_child_exit, record_deliberate_severance, record_wait_error_terminal,
7770 reset_restart_count, spawn_and_mark_running, update_snapshot, wait_error_exit_report,
7771 ExitKind, ExitReport, ModuleProtocol, ModuleSpec, ModuleState, NextAction, ProcessIdentity,
7772 RestartPolicy, SpawnEventKind, StopNotice, SuperviseError, SupervisedModule, Supervisor,
7773 SupervisorHandle, SupervisorHealthStatus, SupervisorSnapshot,
7774 };
7775 use super::Instant as ClockInstant;
7780 use crate::{
7781 registry::Registry,
7782 terminal_ring::{TerminalRing, TerminalRingConfig},
7783 };
7784 use std::sync::Mutex;
7785 use subc_control::TerminalDisposition;
7786
7787 fn fake_aft_stub_path() -> PathBuf {
7792 let mut path = std::env::current_exe().expect("current_exe available in tests");
7793 path.pop();
7794 path.pop();
7795 path.push(if cfg!(windows) {
7796 "fake-aft-stub.exe"
7797 } else {
7798 "fake-aft-stub"
7799 });
7800 assert!(
7801 path.exists(),
7802 "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
7803 [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
7804 path.display()
7805 );
7806 path
7807 }
7808
7809 #[test]
7810 fn reserved_never_spawned_refuses_every_hello() {
7811 let supervisor = SupervisorHandle::default();
7816 supervisor.apply_identity_configuration(&ModuleSpec {
7817 module_id: "never-spawned".to_string(),
7818 program: PathBuf::from("/usr/bin/false"),
7819 args: Vec::new(),
7820 env: Vec::new(),
7821 reserved: true,
7822 reserved_prefixes: Vec::new(),
7823 protocol: ModuleProtocol::Subc,
7824 overlap: Default::default(),
7825 });
7826 assert!(
7827 supervisor
7828 .reserved_hello_rejection("never-spawned", Some("any-forged-nonce"))
7829 .is_some(),
7830 "forged nonce must refuse on a reserved never-spawned id"
7831 );
7832 assert!(
7833 supervisor
7834 .reserved_hello_rejection("never-spawned", None)
7835 .is_some(),
7836 "absent nonce must refuse on a reserved never-spawned id"
7837 );
7838 supervisor.set_spawn_nonce("never-spawned", "minted".to_string());
7840 supervisor.apply_identity_configuration(&ModuleSpec {
7841 module_id: "never-spawned".to_string(),
7842 program: PathBuf::from("/usr/bin/false"),
7843 args: Vec::new(),
7844 env: Vec::new(),
7845 reserved: true,
7846 reserved_prefixes: Vec::new(),
7847 protocol: ModuleProtocol::Subc,
7848 overlap: Default::default(),
7849 });
7850 assert!(supervisor
7851 .reserved_hello_rejection("never-spawned", Some("minted"))
7852 .is_none());
7853 assert!(supervisor
7854 .reserved_hello_rejection("never-spawned", Some("forged"))
7855 .is_some());
7856 }
7857
7858 fn seed_crash_restarts(state: &mut SupervisorSnapshot, count: u32) {
7861 let now = ClockInstant::now();
7862 for _ in 0..count {
7863 state.crash_restarts.push_back(now);
7864 }
7865 }
7866
7867 fn age_oldest_crash_restart_out_of_window(state: &mut SupervisorSnapshot, window: Duration) {
7871 let aged = state
7872 .crash_restarts
7873 .front()
7874 .expect("a crash restart must be recorded before it can be aged")
7875 .checked_sub(window + Duration::from_secs(1))
7876 .expect("the test clock is far enough from its origin to age an instant");
7877 state.crash_restarts[0] = aged;
7878 }
7879
7880 fn snapshot_with_restarts(enabled: bool, count: u32) -> SupervisorSnapshot {
7881 let mut state = SupervisorSnapshot::new(ModuleState::Running, enabled);
7882 seed_crash_restarts(&mut state, count);
7883 state
7884 }
7885
7886 #[test]
7887 fn daemon_owned_recovery_predicate_uses_the_pre_increment_budget() {
7888 let policy = RestartPolicy::new(3, Duration::ZERO);
7889 let now = ClockInstant::now();
7890 assert!(daemon_will_restart(
7891 &mut snapshot_with_restarts(true, 2),
7892 &policy,
7893 now
7894 ));
7895 assert!(!daemon_will_restart(
7896 &mut snapshot_with_restarts(true, 3),
7897 &policy,
7898 now
7899 ));
7900 assert!(!daemon_will_restart(
7901 &mut snapshot_with_restarts(false, 0),
7902 &policy,
7903 now
7904 ));
7905 }
7906
7907 #[test]
7908 fn crash_restart_backoff_escalates_with_in_window_count() {
7909 let policy = RestartPolicy::new(4, Duration::from_millis(100))
7910 .with_max_backoff(Duration::from_secs(30));
7911 let now = ClockInstant::now();
7912 let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
7913 let schedules = (0..4)
7914 .map(|_| {
7915 state
7916 .next_crash_restart(&policy, now)
7917 .expect("the test policy allows four crash restarts")
7918 })
7919 .collect::<Vec<_>>();
7920
7921 assert_eq!(
7922 schedules
7923 .iter()
7924 .map(|schedule| schedule.restart_in_window)
7925 .collect::<Vec<_>>(),
7926 vec![0, 1, 2, 3]
7927 );
7928 assert_eq!(
7929 schedules
7930 .iter()
7931 .map(|schedule| schedule.delay)
7932 .collect::<Vec<_>>(),
7933 vec![
7934 Duration::from_millis(100),
7935 Duration::from_secs(1),
7936 Duration::from_secs(10),
7937 Duration::from_secs(30),
7938 ]
7939 );
7940 }
7941
7942 #[test]
7943 fn crash_restart_backoff_resets_after_ring_clear() {
7944 let policy = RestartPolicy::new(3, Duration::from_millis(100));
7945 let now = ClockInstant::now();
7946 let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
7947 assert_eq!(
7948 state.next_crash_restart(&policy, now).unwrap().delay,
7949 Duration::from_millis(100)
7950 );
7951 assert_eq!(
7952 state.next_crash_restart(&policy, now).unwrap().delay,
7953 Duration::from_secs(1)
7954 );
7955
7956 state.clear_crash_restarts();
7957 let schedule = state
7958 .next_crash_restart(&policy, now)
7959 .expect("a cleared ring must allow another restart");
7960 assert_eq!(schedule.restart_in_window, 0);
7961 assert_eq!(schedule.delay, Duration::from_millis(100));
7962 }
7963
7964 #[test]
7965 fn crash_restart_backoff_ignores_aged_restarts() {
7966 let policy = RestartPolicy::new(3, Duration::from_millis(100));
7967 let now = ClockInstant::now();
7968 let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
7969 state
7970 .next_crash_restart(&policy, now)
7971 .expect("the first restart is allowed");
7972 state
7973 .next_crash_restart(&policy, now)
7974 .expect("the second restart is allowed");
7975 state.crash_restarts[0] = now
7976 .checked_sub(policy.window + Duration::from_secs(1))
7977 .expect("the fake clock can age a restart past the window");
7978
7979 let schedule = state
7980 .next_crash_restart(&policy, now)
7981 .expect("an aged restart must release its slot");
7982 assert_eq!(schedule.restart_in_window, 1);
7983 assert_eq!(schedule.delay, Duration::from_secs(1));
7984 assert_eq!(state.crash_restarts.len(), 2);
7985 }
7986
7987 #[test]
7991 fn a_budget_spent_before_the_window_no_longer_refuses() {
7992 let policy = RestartPolicy::new(3, Duration::ZERO);
7993 let mut state = snapshot_with_restarts(true, 3);
7994 let now = ClockInstant::now();
7995 assert!(!daemon_will_restart(&mut state, &policy, now));
7996
7997 assert!(daemon_will_restart(
7998 &mut state,
7999 &policy,
8000 now + policy.window + Duration::from_secs(1)
8001 ));
8002 assert!(
8003 state.crash_restarts.is_empty(),
8004 "reading the budget must drop the instants that left the window"
8005 );
8006 }
8007
8008 fn module_with_recovery_snapshot(
8009 state: ModuleState,
8010 enabled: bool,
8011 restart_count: u32,
8012 ) -> SupervisedModule {
8013 let registry = Arc::new(Registry::default());
8014 let supervisor =
8015 Supervisor::new(Arc::clone(®istry), RestartPolicy::new(3, Duration::ZERO));
8016 let module = supervisor
8017 .spawn(ModuleSpec {
8018 module_id: "recovery-snapshot".to_string(),
8019 program: fake_aft_stub_path(),
8020 args: Vec::new(),
8021 env: Vec::new(),
8022 reserved: false,
8023 reserved_prefixes: Vec::new(),
8024 protocol: ModuleProtocol::Subc,
8025 overlap: Default::default(),
8026 })
8027 .unwrap();
8028 update_snapshot(
8029 &module.inner.snapshot,
8030 Some("recovery-snapshot"),
8031 |snapshot| {
8032 snapshot.state = state;
8033 snapshot.enabled = enabled;
8034 seed_crash_restarts(snapshot, restart_count);
8035 },
8036 )
8037 .unwrap();
8038 module
8039 }
8040
8041 #[cfg(target_os = "linux")]
8042 #[tokio::test]
8043 async fn no_cgroup_placement_does_not_block_fake_aft_stub_spawn() {
8044 let supervisor = Supervisor::new(Arc::new(Registry::default()), RestartPolicy::default())
8045 .with_cgroup_placement(None);
8046 let result = supervisor.spawn(ModuleSpec {
8047 module_id: "no-cgroup-placement".to_string(),
8048 program: fake_aft_stub_path(),
8049 args: Vec::new(),
8050 env: Vec::new(),
8051 reserved: false,
8052 reserved_prefixes: Vec::new(),
8053 protocol: ModuleProtocol::Subc,
8054 overlap: Default::default(),
8055 });
8056
8057 assert!(
8058 result.is_ok(),
8059 "no delegation must not turn an otherwise valid spawn into a failure: {result:?}"
8060 );
8061 }
8062
8063 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8064 async fn undecided_snapshot_uses_shared_restart_predicate() {
8065 assert!(module_with_recovery_snapshot(ModuleState::Running, true, 2)
8066 .will_recover_after_connection_loss()
8067 .unwrap());
8068 assert!(
8069 !module_with_recovery_snapshot(ModuleState::Running, true, 3)
8070 .will_recover_after_connection_loss()
8071 .unwrap()
8072 );
8073 }
8074
8075 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8076 async fn restarting_snapshot_at_exhausted_budget_is_non_terminal() {
8077 assert!(
8078 module_with_recovery_snapshot(ModuleState::Restarting, true, 3)
8079 .will_recover_after_connection_loss()
8080 .unwrap()
8081 );
8082 }
8083
8084 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8085 async fn terminal_phase_snapshots_are_terminal_before_budget_exhaustion() {
8086 assert!(!module_with_recovery_snapshot(ModuleState::Failed, true, 0)
8087 .will_recover_after_connection_loss()
8088 .unwrap());
8089 assert!(
8090 !module_with_recovery_snapshot(ModuleState::Disabled, true, 0)
8091 .will_recover_after_connection_loss()
8092 .unwrap()
8093 );
8094 }
8095
8096 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8097 async fn warming_snapshot_is_limited_to_startup_phases() {
8098 for state in [
8099 ModuleState::Starting,
8100 ModuleState::Running,
8101 ModuleState::Restarting,
8102 ] {
8103 assert!(
8104 module_with_recovery_snapshot(state, true, 0)
8105 .is_warming()
8106 .unwrap(),
8107 "{state:?} should be warming"
8108 );
8109 }
8110 for state in [
8111 ModuleState::Unresponsive,
8112 ModuleState::Draining,
8113 ModuleState::Stopped,
8114 ModuleState::Failed,
8115 ModuleState::Disabled,
8116 ] {
8117 assert!(
8118 !module_with_recovery_snapshot(state, true, 0)
8119 .is_warming()
8120 .unwrap(),
8121 "{state:?} should not be warming"
8122 );
8123 }
8124 }
8125
8126 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8127 async fn terminal_history_survives_respawn_and_keeps_both_crashes_in_order() {
8128 let registry = Arc::new(Registry::default());
8129 let supervisor =
8130 Supervisor::new(Arc::clone(®istry), RestartPolicy::new(1, Duration::ZERO));
8131 let module = supervisor
8132 .spawn(ModuleSpec {
8133 module_id: "terminal-history".to_string(),
8134 program: fake_aft_stub_path(),
8135 args: Vec::new(),
8136 env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
8137 reserved: false,
8138 reserved_prefixes: Vec::new(),
8139 protocol: ModuleProtocol::Subc,
8140 overlap: Default::default(),
8141 })
8142 .unwrap();
8143
8144 let deadline = Instant::now() + Duration::from_secs(5);
8145 loop {
8146 let history = module.terminal_history();
8147 if history.entries.len() == 2 {
8148 assert_eq!(module.status().unwrap().state, ModuleState::Failed);
8149 assert_eq!(history.dropped, 0);
8150 assert_eq!(
8151 history
8152 .entries
8153 .iter()
8154 .map(|entry| entry.exit_code)
8155 .collect::<Vec<_>>(),
8156 vec![Some(23), Some(23)]
8157 );
8158 assert!(history.entries[0].at_ms <= history.entries[1].at_ms);
8159 return;
8160 }
8161 assert!(
8162 Instant::now() < deadline,
8163 "module did not retain two terminal exits: {history:?}"
8164 );
8165 sleep(Duration::from_millis(10)).await;
8166 }
8167 }
8168
8169 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8173 async fn disable_during_crash_backoff_cancels_pending_respawn() {
8174 let backoff = Duration::from_secs(2);
8175 let supervisor = Supervisor::new(
8176 Arc::new(Registry::default()),
8177 RestartPolicy::new(10, backoff),
8178 );
8179 let module = supervisor
8180 .spawn(ModuleSpec {
8181 module_id: "disable-during-backoff".to_string(),
8182 program: fake_aft_stub_path(),
8183 args: Vec::new(),
8184 env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
8185 reserved: false,
8186 reserved_prefixes: Vec::new(),
8187 protocol: ModuleProtocol::Subc,
8188 overlap: Default::default(),
8189 })
8190 .unwrap();
8191
8192 let deadline = Instant::now() + Duration::from_secs(5);
8194 loop {
8195 if module.status().unwrap().state == ModuleState::Restarting {
8196 break;
8197 }
8198 assert!(
8199 Instant::now() < deadline,
8200 "module never entered the crash backoff"
8201 );
8202 sleep(Duration::from_millis(10)).await;
8203 }
8204
8205 let started = Instant::now();
8206 module.set_enabled(false).await.unwrap();
8207 let waited = started.elapsed();
8208
8209 assert!(
8210 waited < backoff / 2,
8211 "disable waited {waited:?} behind the {backoff:?} crash backoff; the operator command must preempt the pending respawn"
8212 );
8213 assert_eq!(module.status().unwrap().state, ModuleState::Disabled);
8214
8215 sleep(backoff + Duration::from_millis(500)).await;
8217 let status = module.status().unwrap();
8218 assert_eq!(status.state, ModuleState::Disabled);
8219 assert_eq!(
8220 status.spawn_generation, 1,
8221 "module respawned after the operator disabled it"
8222 );
8223 }
8224
8225 #[cfg(unix)]
8228 fn protocol_none_sigterm_exits_clean_spec(
8229 module_id: &str,
8230 dir: &std::path::Path,
8231 ) -> (ModuleSpec, PathBuf, PathBuf) {
8232 let ready = dir.join("ready");
8233 let marker = dir.join("sigterm");
8234 let spec = ModuleSpec {
8235 module_id: module_id.to_string(),
8236 program: fake_aft_stub_path(),
8237 args: Vec::new(),
8238 env: vec![
8239 ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
8240 (
8241 "FAKE_AFT_SIGTERM_MARKER_PATH".to_string(),
8242 marker.display().to_string(),
8243 ),
8244 (
8245 "FAKE_AFT_NEVER_CONNECT_READY_PATH".to_string(),
8246 ready.display().to_string(),
8247 ),
8248 ],
8249 reserved: false,
8250 reserved_prefixes: Vec::new(),
8251 protocol: ModuleProtocol::None,
8252 overlap: Default::default(),
8253 };
8254 (spec, ready, marker)
8255 }
8256
8257 #[cfg(unix)]
8261 async fn wait_for_file(path: &std::path::Path) {
8262 let deadline = Instant::now() + Duration::from_secs(10);
8263 while !path.exists() {
8264 assert!(
8265 Instant::now() < deadline,
8266 "{} never appeared",
8267 path.display()
8268 );
8269 sleep(Duration::from_millis(10)).await;
8270 }
8271 }
8272
8273 #[cfg(unix)]
8277 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8278 async fn protocol_none_unrequested_clean_exit_restarts_as_a_crash() {
8279 let dir = subc_test_support::TestTempDir::new("none-unrequested-clean-exit");
8280 let (spec, ready, marker) =
8281 protocol_none_sigterm_exits_clean_spec("none-unrequested-clean-exit", dir.path());
8282 let supervisor = Supervisor::new(
8283 Arc::new(Registry::default()),
8284 RestartPolicy::new(3, Duration::ZERO),
8285 );
8286 let module = supervisor.spawn(spec).unwrap();
8287 wait_for_file(&ready).await;
8288 let first_pid = module
8289 .status()
8290 .unwrap()
8291 .pid
8292 .expect("a running module reports its pid");
8293
8294 rustix::process::kill_process(
8295 rustix::process::Pid::from_raw(i32::try_from(first_pid).unwrap()).unwrap(),
8296 rustix::process::Signal::TERM,
8297 )
8298 .unwrap();
8299
8300 let deadline = Instant::now() + Duration::from_secs(10);
8301 let respawned = loop {
8302 let status = module.status().unwrap();
8303 if status.state == ModuleState::Running
8304 && status.pid.is_some_and(|pid| pid != first_pid)
8305 {
8306 break status;
8307 }
8308 assert!(
8309 Instant::now() < deadline,
8310 "protocol-none module was not respawned after an unrequested clean exit: {status:?}"
8311 );
8312 sleep(Duration::from_millis(10)).await;
8313 };
8314 assert_eq!(respawned.spawn_generation, 2);
8315 assert!(
8316 marker.exists(),
8317 "the child must have exited through its SIGTERM handler (exit 0), or this proves nothing about clean exits"
8318 );
8319
8320 let history = module.terminal_history();
8321 assert_eq!(history.entries.len(), 1, "{history:?}");
8322 let entry = &history.entries[0];
8323 assert_eq!(entry.exit_code, Some(0));
8324 assert_eq!(entry.exit_kind, subc_control::TerminalExitKind::Clean);
8325 assert_eq!(entry.disposition, TerminalDisposition::Restarting);
8326
8327 module.stop().await.unwrap();
8328 }
8329
8330 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8334 async fn protocol_none_repeated_clean_exits_exhaust_the_restart_budget() {
8335 let supervisor = Supervisor::new(
8336 Arc::new(Registry::default()),
8337 RestartPolicy::new(1, Duration::ZERO),
8338 );
8339 let module = supervisor
8340 .spawn(ModuleSpec {
8341 module_id: "none-clean-exit-budget".to_string(),
8342 program: fake_aft_stub_path(),
8343 args: Vec::new(),
8344 env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
8345 reserved: false,
8346 reserved_prefixes: Vec::new(),
8347 protocol: ModuleProtocol::None,
8348 overlap: Default::default(),
8349 })
8350 .unwrap();
8351
8352 let deadline = Instant::now() + Duration::from_secs(10);
8353 loop {
8354 let status = module.status().unwrap();
8355 if status.state == ModuleState::Failed {
8356 break;
8357 }
8358 assert!(
8359 Instant::now() < deadline,
8360 "module never exhausted its budget: {status:?} {:?}",
8361 module.terminal_history()
8362 );
8363 sleep(Duration::from_millis(10)).await;
8364 }
8365 let history = module.terminal_history();
8366 assert_eq!(
8367 history
8368 .entries
8369 .iter()
8370 .map(|entry| (entry.exit_code, entry.disposition.clone()))
8371 .collect::<Vec<_>>(),
8372 vec![
8373 (Some(0), TerminalDisposition::Restarting),
8374 (Some(0), TerminalDisposition::Failed),
8375 ]
8376 );
8377 let detail = history.entries[1]
8378 .disposition_detail
8379 .as_deref()
8380 .expect("a budget failure names the budget");
8381 assert!(detail.contains("max_restarts=1"), "{detail}");
8382 assert_eq!(module.status().unwrap().spawn_generation, 2);
8383 }
8384
8385 #[cfg(unix)]
8388 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8389 async fn protocol_none_requested_stop_and_disable_do_not_respawn() {
8390 for disable in [false, true] {
8391 let label = if disable {
8392 "none-requested-disable"
8393 } else {
8394 "none-requested-stop"
8395 };
8396 let dir = subc_test_support::TestTempDir::new(label);
8397 let (spec, ready, marker) = protocol_none_sigterm_exits_clean_spec(label, dir.path());
8398 let supervisor = Supervisor::new(
8399 Arc::new(Registry::default()),
8400 RestartPolicy::new(3, Duration::ZERO),
8401 );
8402 let module = supervisor.spawn(spec).unwrap();
8403 wait_for_file(&ready).await;
8404
8405 if disable {
8406 module.set_enabled(false).await.unwrap();
8407 } else {
8408 module.stop().await.unwrap();
8409 }
8410 assert!(
8411 marker.exists(),
8412 "{label}: the child must have left through its SIGTERM handler with exit 0"
8413 );
8414
8415 sleep(Duration::from_millis(500)).await;
8418 let status = module.status().unwrap();
8419 let expected = if disable {
8420 ModuleState::Disabled
8421 } else {
8422 ModuleState::Stopped
8423 };
8424 assert_eq!(status.state, expected, "{label}");
8425 assert_eq!(
8426 status.spawn_generation, 1,
8427 "{label}: respawned after a requested stop"
8428 );
8429 let history = module.terminal_history();
8430 assert_eq!(history.entries.len(), 1, "{label}: {history:?}");
8431 assert_eq!(history.entries[0].exit_code, Some(0), "{label}");
8432 assert_ne!(
8433 history.entries[0].disposition,
8434 TerminalDisposition::Restarting,
8435 "{label}"
8436 );
8437 }
8438 }
8439
8440 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8443 async fn subc_wire_clean_exit_is_still_a_stop() {
8444 let supervisor = Supervisor::new(
8445 Arc::new(Registry::default()),
8446 RestartPolicy::new(3, Duration::ZERO),
8447 );
8448 let module = supervisor
8449 .spawn(ModuleSpec {
8450 module_id: "wire-clean-exit".to_string(),
8451 program: fake_aft_stub_path(),
8452 args: Vec::new(),
8453 env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
8454 reserved: false,
8455 reserved_prefixes: Vec::new(),
8456 protocol: ModuleProtocol::Subc,
8457 overlap: Default::default(),
8458 })
8459 .unwrap();
8460
8461 let deadline = Instant::now() + Duration::from_secs(10);
8462 while module.terminal_history().entries.is_empty() {
8463 assert!(Instant::now() < deadline, "module never exited");
8464 sleep(Duration::from_millis(10)).await;
8465 }
8466 sleep(Duration::from_millis(500)).await;
8468 let status = module.status().unwrap();
8469 assert_eq!(status.state, ModuleState::Stopped);
8470 assert_eq!(status.spawn_generation, 1);
8471 let history = module.terminal_history();
8472 assert_eq!(history.entries.len(), 1, "{history:?}");
8473 assert_eq!(history.entries[0].exit_code, Some(0));
8474 assert_eq!(history.entries[0].disposition, TerminalDisposition::Stopped);
8475 }
8476
8477 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8481 async fn every_restart_increment_path_advances_lifetime_count() {
8482 let supervisor = Supervisor::new(
8483 Arc::new(Registry::default()),
8484 RestartPolicy::new(1, Duration::ZERO),
8485 );
8486 let runtime = supervisor.runtime_config();
8487 let spec = ModuleSpec {
8488 module_id: "lifetime-increment-path".to_string(),
8489 program: PathBuf::from("/unused/lifetime-increment-path"),
8490 args: Vec::new(),
8491 env: Vec::new(),
8492 reserved: false,
8493 reserved_prefixes: Vec::new(),
8494 protocol: ModuleProtocol::Subc,
8495 overlap: Default::default(),
8496 };
8497
8498 let crash_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8499 assert!(matches!(
8500 on_child_exit(
8501 &spec,
8502 runtime.restart_policy,
8503 &supervisor.registry,
8504 &crash_snapshot,
8505 &runtime.terminal_ring,
8506 &runtime.spawn_events,
8507 &runtime.child_roster,
8508 ExitReport {
8509 kind: ExitKind::Crash,
8510 code: Some(1),
8511 signal: None,
8512 at_ms: 1,
8513 },
8514 )
8515 .await,
8516 NextAction::Restart { schedule: _ }
8517 ));
8518 let (crash_restarts, crash_lifetime) = {
8519 let state = lock_snapshot(&crash_snapshot).unwrap();
8520 (state.crash_restarts.len(), state.lifetime_restarts)
8521 };
8522 assert_eq!(crash_restarts, 1);
8523 assert_eq!(crash_lifetime, 1);
8524
8525 let health_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8526 let mut health_child = None;
8527 assert!(matches!(
8528 health_restart_child(
8529 &spec,
8530 &runtime,
8531 &supervisor.registry,
8532 &supervisor.process_liveness,
8533 &health_snapshot,
8534 &mut health_child,
8535 SupervisorHealthStatus::Failing,
8536 None,
8537 2,
8538 )
8539 .await,
8540 Err(SuperviseError::Spawn { .. })
8541 ));
8542 let (health_restarts, health_lifetime) = {
8543 let state = lock_snapshot(&health_snapshot).unwrap();
8544 (state.crash_restarts.len(), state.lifetime_restarts)
8545 };
8546 assert_eq!(health_restarts, 1);
8547 assert_eq!(health_lifetime, 1);
8548
8549 let reload_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8550 let mut reload_child = None;
8551 assert!(matches!(
8552 handle_reload_spawn_failure(
8553 &spec,
8554 &runtime,
8555 &supervisor.process_liveness,
8556 &reload_snapshot,
8557 &mut reload_child,
8558 "forced reload spawn failure".to_string(),
8559 )
8560 .await,
8561 Err(SuperviseError::ReloadFailed { .. })
8562 ));
8563 let (reload_restarts, reload_lifetime) = {
8564 let state = lock_snapshot(&reload_snapshot).unwrap();
8565 (state.crash_restarts.len(), state.lifetime_restarts)
8566 };
8567 assert_eq!(reload_restarts, 1);
8568 assert_eq!(reload_lifetime, 1);
8569 }
8570
8571 #[tokio::test]
8572 async fn deliberately_severed_live_child_records_lifetime_without_spending_restart_budget() {
8573 let supervisor = Supervisor::new(
8574 Arc::new(Registry::default()),
8575 RestartPolicy::new(3, Duration::ZERO),
8576 );
8577 let runtime = supervisor.runtime_config();
8578 let spec = ModuleSpec {
8579 module_id: "deliberately-severed".to_string(),
8580 program: PathBuf::from("/unused/deliberately-severed"),
8581 args: Vec::new(),
8582 env: Vec::new(),
8583 reserved: false,
8584 reserved_prefixes: Vec::new(),
8585 protocol: ModuleProtocol::Subc,
8586 overlap: Default::default(),
8587 };
8588 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8589 let process = ProcessIdentity {
8590 pid: 41,
8591 start_time: 101,
8592 };
8593 record_deliberate_severance(&snapshot, process).unwrap();
8594 let exit_report = apply_deliberate_severance_marker(
8595 &snapshot,
8596 Some(process),
8597 ExitReport {
8598 kind: ExitKind::Crash,
8599 code: Some(1),
8600 signal: None,
8601 at_ms: 1,
8602 },
8603 );
8604 assert_eq!(exit_report.kind, ExitKind::DeliberateSeverance);
8605
8606 assert!(matches!(
8607 on_child_exit(
8608 &spec,
8609 runtime.restart_policy,
8610 &supervisor.registry,
8611 &snapshot,
8612 &runtime.terminal_ring,
8613 &runtime.spawn_events,
8614 &runtime.child_roster,
8615 exit_report,
8616 )
8617 .await,
8618 NextAction::Restart { schedule: _ }
8619 ));
8620 let state = lock_snapshot(&snapshot).unwrap();
8621 assert_eq!(state.lifetime_restarts, 1);
8622 assert_eq!(state.crash_restarts.len(), 0);
8623 }
8624
8625 #[tokio::test]
8626 async fn genuine_crash_spends_restart_budget_and_records_lifetime() {
8627 let supervisor = Supervisor::new(
8628 Arc::new(Registry::default()),
8629 RestartPolicy::new(3, Duration::ZERO),
8630 );
8631 let runtime = supervisor.runtime_config();
8632 let spec = ModuleSpec {
8633 module_id: "genuine-crash".to_string(),
8634 program: PathBuf::from("/unused/genuine-crash"),
8635 args: Vec::new(),
8636 env: Vec::new(),
8637 reserved: false,
8638 reserved_prefixes: Vec::new(),
8639 protocol: ModuleProtocol::Subc,
8640 overlap: Default::default(),
8641 };
8642 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8643
8644 assert!(matches!(
8645 on_child_exit(
8646 &spec,
8647 runtime.restart_policy,
8648 &supervisor.registry,
8649 &snapshot,
8650 &runtime.terminal_ring,
8651 &runtime.spawn_events,
8652 &runtime.child_roster,
8653 ExitReport {
8654 kind: ExitKind::Crash,
8655 code: Some(1),
8656 signal: None,
8657 at_ms: 1,
8658 },
8659 )
8660 .await,
8661 NextAction::Restart { schedule: _ }
8662 ));
8663 let state = lock_snapshot(&snapshot).unwrap();
8664 assert_eq!(state.lifetime_restarts, 1);
8665 assert_eq!(state.crash_restarts.len(), 1);
8666 }
8667
8668 fn crash_exit_report(at_ms: u64) -> ExitReport {
8669 ExitReport {
8670 kind: ExitKind::Crash,
8671 code: Some(1),
8672 signal: None,
8673 at_ms,
8674 }
8675 }
8676
8677 fn windowed_crash_spec(module_id: &str) -> ModuleSpec {
8678 ModuleSpec {
8679 module_id: module_id.to_string(),
8680 program: PathBuf::from("/unused").join(module_id),
8681 args: Vec::new(),
8682 env: Vec::new(),
8683 reserved: false,
8684 reserved_prefixes: Vec::new(),
8685 protocol: ModuleProtocol::Subc,
8686 overlap: Default::default(),
8687 }
8688 }
8689
8690 #[tokio::test]
8696 async fn three_crashes_inside_the_window_stop_the_module_and_name_the_window() {
8697 let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::ERROR);
8698 let supervisor = Supervisor::new(
8699 Arc::new(Registry::default()),
8700 RestartPolicy::new(2, Duration::ZERO),
8701 );
8702 let runtime = supervisor.runtime_config();
8703 let spec = windowed_crash_spec("crash-loop-in-window");
8704 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8705
8706 for attempt in 1..=2 {
8707 assert!(
8708 matches!(
8709 on_child_exit(
8710 &spec,
8711 runtime.restart_policy,
8712 &supervisor.registry,
8713 &snapshot,
8714 &runtime.terminal_ring,
8715 &runtime.spawn_events,
8716 &runtime.child_roster,
8717 crash_exit_report(attempt),
8718 )
8719 .await,
8720 NextAction::Restart { schedule: _ }
8721 ),
8722 "crash {attempt} is inside the budget and must respawn"
8723 );
8724 }
8725
8726 assert!(matches!(
8727 on_child_exit(
8728 &spec,
8729 runtime.restart_policy,
8730 &supervisor.registry,
8731 &snapshot,
8732 &runtime.terminal_ring,
8733 &runtime.spawn_events,
8734 &runtime.child_roster,
8735 crash_exit_report(3),
8736 )
8737 .await,
8738 NextAction::Stop { .. }
8739 ));
8740
8741 {
8742 let state = lock_snapshot(&snapshot).unwrap();
8743 assert_eq!(state.state, ModuleState::Failed);
8744 assert_eq!(state.crash_restarts.len(), 2);
8745 assert_eq!(state.lifetime_restarts, 2);
8746 }
8747
8748 let history = runtime
8749 .terminal_ring
8750 .lock()
8751 .expect("terminal ring is not poisoned")
8752 .snapshot();
8753 let last = history
8754 .entries
8755 .last()
8756 .expect("the refused crash is retained");
8757 assert_eq!(last.disposition, TerminalDisposition::Failed);
8758 assert_eq!(
8759 last.disposition_detail.as_deref(),
8760 Some("crash budget exhausted: max_restarts=2 within window_secs=600")
8761 );
8762
8763 let captured = crate::router::test_log::captured_logs(&logs);
8764 assert!(
8765 captured.contains("crash budget exhausted: max_restarts=2 within window_secs=600"),
8766 "the stop must be logged with its window: {captured}"
8767 );
8768 }
8769
8770 #[tokio::test]
8778 async fn a_crash_older_than_the_window_frees_its_slot_for_a_later_crash() {
8779 let supervisor = Supervisor::new(
8780 Arc::new(Registry::default()),
8781 RestartPolicy::new(2, Duration::ZERO),
8782 );
8783 let runtime = supervisor.runtime_config();
8784 let spec = windowed_crash_spec("crash-across-windows");
8785 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8786
8787 for attempt in 1..=2 {
8788 assert!(matches!(
8789 on_child_exit(
8790 &spec,
8791 runtime.restart_policy,
8792 &supervisor.registry,
8793 &snapshot,
8794 &runtime.terminal_ring,
8795 &runtime.spawn_events,
8796 &runtime.child_roster,
8797 crash_exit_report(attempt),
8798 )
8799 .await,
8800 NextAction::Restart { schedule: _ }
8801 ));
8802 }
8803
8804 update_snapshot(&snapshot, Some(&spec.module_id), |state| {
8807 age_oldest_crash_restart_out_of_window(state, runtime.restart_policy.window);
8808 })
8809 .unwrap();
8810
8811 assert!(
8812 matches!(
8813 on_child_exit(
8814 &spec,
8815 runtime.restart_policy,
8816 &supervisor.registry,
8817 &snapshot,
8818 &runtime.terminal_ring,
8819 &runtime.spawn_events,
8820 &runtime.child_roster,
8821 crash_exit_report(3),
8822 )
8823 .await,
8824 NextAction::Restart { schedule: _ }
8825 ),
8826 "a crash older than the window must not hold a budget slot"
8827 );
8828
8829 let state = lock_snapshot(&snapshot).unwrap();
8830 assert_eq!(state.state, ModuleState::Restarting);
8831 assert_eq!(
8832 state.crash_restarts.len(),
8833 2,
8834 "the aged instant is dropped and the new one takes its place"
8835 );
8836 assert_eq!(
8837 state.lifetime_restarts, 3,
8838 "the ledger counts every restart, including the ones the window forgot"
8839 );
8840 }
8841
8842 #[tokio::test]
8847 async fn an_operator_restart_clears_the_ring_and_leaves_the_ledger_alone() {
8848 let supervisor = Supervisor::new(
8849 Arc::new(Registry::default()),
8850 RestartPolicy::new(2, Duration::ZERO),
8851 );
8852 let runtime = supervisor.runtime_config();
8853 let spec = windowed_crash_spec("operator-cleared-budget");
8854 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8855
8856 for attempt in 1..=2 {
8857 assert!(matches!(
8858 on_child_exit(
8859 &spec,
8860 runtime.restart_policy,
8861 &supervisor.registry,
8862 &snapshot,
8863 &runtime.terminal_ring,
8864 &runtime.spawn_events,
8865 &runtime.child_roster,
8866 crash_exit_report(attempt),
8867 )
8868 .await,
8869 NextAction::Restart { schedule: _ }
8870 ));
8871 }
8872
8873 reset_restart_count(&snapshot, &spec.module_id).unwrap();
8874 {
8875 let state = lock_snapshot(&snapshot).unwrap();
8876 assert!(
8877 state.crash_restarts.is_empty(),
8878 "an operator restart returns the full budget"
8879 );
8880 assert_eq!(
8881 state.lifetime_restarts, 2,
8882 "clearing the budget must not unmake the crashes"
8883 );
8884 }
8885
8886 assert!(
8887 matches!(
8888 on_child_exit(
8889 &spec,
8890 runtime.restart_policy,
8891 &supervisor.registry,
8892 &snapshot,
8893 &runtime.terminal_ring,
8894 &runtime.spawn_events,
8895 &runtime.child_roster,
8896 crash_exit_report(3),
8897 )
8898 .await,
8899 NextAction::Restart { schedule: _ }
8900 ),
8901 "the cleared budget must be spendable again"
8902 );
8903 let state = lock_snapshot(&snapshot).unwrap();
8904 assert_eq!(state.crash_restarts.len(), 1);
8905 assert_eq!(state.lifetime_restarts, 3);
8906 }
8907
8908 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8909 async fn severance_marker_for_a_dead_child_does_not_label_its_successor() {
8910 let severed = ProcessIdentity {
8911 pid: 41,
8912 start_time: 101,
8913 };
8914 let successor = ProcessIdentity {
8915 pid: 41,
8916 start_time: 202,
8917 };
8918 let module = module_with_recovery_snapshot(ModuleState::Running, true, 0);
8919 update_snapshot(&module.inner.snapshot, Some("recovery-snapshot"), |state| {
8920 state.pid = Some(successor.pid);
8921 state.process_start_time = Some(successor.start_time);
8922 })
8923 .unwrap();
8924 assert!(!module.record_deliberate_severance(severed).unwrap());
8925
8926 let exit_report = apply_deliberate_severance_marker(
8927 &module.inner.snapshot,
8928 Some(successor),
8929 ExitReport {
8930 kind: ExitKind::Crash,
8931 code: Some(1),
8932 signal: None,
8933 at_ms: 1,
8934 },
8935 );
8936
8937 assert_eq!(exit_report.kind, ExitKind::Crash);
8938 }
8939
8940 #[tokio::test]
8941 async fn drain_reap_marks_deliberate_severance_and_records_lifetime_without_budget() {
8942 let registry = Registry::default();
8943 let supervisor = Supervisor::new(
8944 Arc::new(Registry::default()),
8945 RestartPolicy::new(3, Duration::ZERO),
8946 );
8947 let runtime = supervisor.runtime_config();
8948 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8949 let spec = ModuleSpec {
8950 module_id: "drain-deliberate-severance".to_string(),
8951 program: fake_aft_stub_path(),
8952 args: Vec::new(),
8953 env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
8954 reserved: false,
8955 reserved_prefixes: Vec::new(),
8956 protocol: ModuleProtocol::Subc,
8957 overlap: Default::default(),
8958 };
8959 let mut child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
8960 let process = ProcessIdentity {
8961 pid: 41,
8962 start_time: 101,
8963 };
8964 child.process_identity = Some(process);
8965 update_snapshot(&snapshot, Some(&spec.module_id), |state| {
8966 state.pid = Some(process.pid);
8967 state.process_start_time = Some(process.start_time);
8968 })
8969 .unwrap();
8970 record_deliberate_severance(&snapshot, process).unwrap();
8971
8972 drain_child_to_state(
8973 &spec.module_id,
8974 spec.protocol,
8975 StopNotice::SentOverConnection,
8978 ®istry,
8979 &snapshot,
8980 &runtime.terminal_ring,
8981 &runtime.spawn_events,
8982 child,
8983 Duration::from_secs(1),
8984 ModuleState::Stopped,
8985 Some(false),
8986 )
8987 .await
8988 .unwrap();
8989
8990 let state = lock_snapshot(&snapshot).unwrap();
8991 assert_eq!(
8992 state.last_exit.as_ref().map(|exit| exit.kind),
8993 Some(ExitKind::DeliberateSeverance)
8994 );
8995 assert_eq!(state.lifetime_restarts, 1);
8996 assert_eq!(state.crash_restarts.len(), 0);
8997 drop(state);
8998 let history = runtime.terminal_ring.lock().unwrap().snapshot();
8999 assert_eq!(
9000 history.entries[0].exit_kind,
9001 subc_control::TerminalExitKind::DeliberateSeverance
9002 );
9003 }
9004
9005 #[tokio::test]
9006 async fn ordinary_drain_reap_does_not_record_a_lifetime_restart() {
9007 let registry = Registry::default();
9008 let supervisor = Supervisor::new(
9009 Arc::new(Registry::default()),
9010 RestartPolicy::new(3, Duration::ZERO),
9011 );
9012 let runtime = supervisor.runtime_config();
9013 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
9014 let spec = ModuleSpec {
9015 module_id: "ordinary-drain".to_string(),
9016 program: fake_aft_stub_path(),
9017 args: Vec::new(),
9018 env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
9019 reserved: false,
9020 reserved_prefixes: Vec::new(),
9021 protocol: ModuleProtocol::Subc,
9022 overlap: Default::default(),
9023 };
9024 let child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
9025
9026 drain_child_to_state(
9027 &spec.module_id,
9028 spec.protocol,
9029 StopNotice::SentOverConnection,
9032 ®istry,
9033 &snapshot,
9034 &runtime.terminal_ring,
9035 &runtime.spawn_events,
9036 child,
9037 Duration::from_secs(1),
9038 ModuleState::Stopped,
9039 Some(false),
9040 )
9041 .await
9042 .unwrap();
9043
9044 let state = lock_snapshot(&snapshot).unwrap();
9045 assert_eq!(
9046 state.last_exit.as_ref().map(|exit| exit.kind),
9047 Some(ExitKind::Crash)
9048 );
9049 assert_eq!(state.lifetime_restarts, 0);
9050 assert_eq!(state.crash_restarts.len(), 0);
9051 }
9052
9053 #[test]
9054 fn fatal_connection_teardown_cannot_arm_a_marker_for_a_surviving_process() {
9055 assert!(!include_str!("server.rs")
9061 .contains("router.record_deliberate_connection_severance(ctx.connection_id)"));
9062 }
9063
9064 #[test]
9071 fn drained_after_quiescence_wait_passes_ok_through_and_forces_false_on_err() {
9072 assert!(drained_after_quiescence_wait(&Ok(true)));
9073 assert!(!drained_after_quiescence_wait(&Ok(false)));
9074 assert!(!drained_after_quiescence_wait(&Err(
9075 SuperviseError::StatePoisoned { module_id: None }
9076 )));
9077 }
9078
9079 #[test]
9088 fn wait_error_exit_report_records_a_failed_terminal_with_no_code_or_signal() {
9089 let ring = Arc::new(Mutex::new(TerminalRing::new(
9090 TerminalRingConfig::default(),
9091 0,
9092 )));
9093 record_wait_error_terminal("wait-error", &ring, &super::SpawnEventFeed::default());
9094
9095 let snapshot = ring.lock().unwrap().snapshot();
9096 assert_eq!(snapshot.entries.len(), 1);
9097 let entry = &snapshot.entries[0];
9098 assert_eq!(entry.exit_code, None);
9099 assert_eq!(entry.exit_signal, None);
9100 assert_eq!(entry.disposition, TerminalDisposition::Failed);
9101 }
9102
9103 #[test]
9104 fn wait_error_exit_path_preserves_spawn_event_density() {
9105 let feed = super::SpawnEventFeed::default();
9106 feed.configure_incarnation("wait-error-density".to_string());
9107 feed.emit_spawned("wait-error", 41, 1);
9108 let ring = Arc::new(Mutex::new(TerminalRing::new(
9109 TerminalRingConfig::default(),
9110 0,
9111 )));
9112
9113 record_wait_error_terminal("wait-error", &ring, &feed);
9114 feed.emit_spawned("after-wait-error", 42, 2);
9115
9116 let state = feed.0.lock().unwrap();
9117 let sequences = state
9118 .events
9119 .iter()
9120 .map(|event| event.cursor.seq)
9121 .collect::<Vec<_>>();
9122 assert_eq!(sequences, vec![1, 2, 3]);
9123 assert_eq!(state.events[1].kind, SpawnEventKind::Exited);
9124 assert_eq!(state.events[1].exit_code, None);
9125 assert_eq!(state.events[1].exit_signal, None);
9126 }
9127
9128 #[test]
9132 fn wait_error_exit_report_is_classified_as_a_crash() {
9133 assert_eq!(wait_error_exit_report().kind, ExitKind::Crash);
9134 }
9135}
9136
9137#[cfg(test)]
9138mod health_evidence_tests {
9139 use super::{HealthProbeError, HealthProbeEvidence};
9140 use std::collections::HashSet;
9141
9142 #[test]
9150 fn only_a_dead_lane_is_proof_of_death() {
9151 assert!(HealthProbeError::lane_dead("gone").is_proof_of_death());
9152 assert!(!HealthProbeError::no_answer("timed out").is_proof_of_death());
9156 assert!(!HealthProbeError::bad_answer("garbage").is_proof_of_death());
9157 assert!(!HealthProbeError::misconfigured("no table").is_proof_of_death());
9158 }
9159
9160 #[test]
9166 fn every_evidence_class_has_a_distinct_label() {
9167 let labels = [
9168 HealthProbeError::lane_dead("").label(),
9169 HealthProbeError::no_answer("").label(),
9170 HealthProbeError::bad_answer("").label(),
9171 HealthProbeError::misconfigured("").label(),
9172 ];
9173 let unique: HashSet<_> = labels.iter().collect();
9174 assert_eq!(unique.len(), labels.len(), "labels collided: {labels:?}");
9175 }
9176
9177 #[test]
9183 fn classification_preserves_the_original_message() {
9184 let err = HealthProbeError::no_answer("module did not answer within 5s");
9185 assert_eq!(err.to_string(), "module did not answer within 5s");
9186 assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
9187 }
9188}
9189
9190#[cfg(test)]
9191mod health_tombstone_tests {
9192 use std::{path::PathBuf, sync::Arc, time::Duration};
9193
9194 use subc_protocol::{
9195 manifest::Concurrency,
9196 session::{HealthStatus, ModuleControlResponse},
9197 };
9198 use tokio::sync::mpsc;
9199
9200 use super::{
9201 probe_module_health, HealthAction, HealthConfig, HealthProbeEvidence, ModuleProtocol,
9202 ModuleSpec, RestartPolicy, Supervisor, SupervisorRuntimeConfig,
9203 };
9204 use crate::{
9205 control::ControlHandler,
9206 forwarding::{ForwardingTable, ModuleControlRpcCompletion, ModuleControlRpcOutcome},
9207 registry::{ConnectionId, Registry},
9208 router::FrameSink,
9209 };
9210
9211 struct ProbeHarness {
9212 spec: ModuleSpec,
9213 runtime: SupervisorRuntimeConfig,
9214 forwarding: Arc<ForwardingTable>,
9215 module_connection: ConnectionId,
9216 module_rx: mpsc::Receiver<crate::router::OutboundFrame>,
9217 handler: ControlHandler,
9218 module: super::SupervisedModule,
9219 }
9220
9221 fn probe_harness() -> ProbeHarness {
9222 let registry = Arc::new(Registry::default());
9223 let forwarding = Arc::new(ForwardingTable::default());
9224 let supervisor_handle = super::SupervisorHandle::new();
9225 let health = HealthConfig {
9226 cadence: Duration::from_secs(30),
9227 deadline: Duration::from_secs(5),
9228 failure_threshold: 3,
9229 on_degraded: HealthAction::Report,
9230 on_failing: HealthAction::Report,
9231 critical: false,
9232 };
9233 let supervisor = Supervisor::new(Arc::clone(®istry), RestartPolicy::default())
9234 .with_forwarding(Arc::clone(&forwarding))
9235 .with_handle(supervisor_handle.clone())
9236 .with_health_config(health);
9237 let spec = ModuleSpec {
9238 module_id: "late-health-module".to_string(),
9239 program: PathBuf::from("disabled-module"),
9240 args: Vec::new(),
9241 env: Vec::new(),
9242 reserved: false,
9243 reserved_prefixes: Vec::new(),
9244 protocol: ModuleProtocol::Subc,
9245 overlap: Default::default(),
9246 };
9247 let module = supervisor
9248 .supervise_configured(spec.clone(), false)
9249 .unwrap();
9250 let runtime = supervisor.runtime_config();
9251 let handler = ControlHandler::with_forwarding(registry, Arc::clone(&forwarding))
9252 .with_supervisor(supervisor_handle);
9253 let module_connection = ConnectionId::new(700);
9254 let (module_tx, module_rx) = mpsc::channel(8);
9255 forwarding
9256 .register_module_connection(
9257 module_connection,
9258 spec.module_id.clone(),
9259 subc_protocol::PROTOCOL_VERSION,
9260 Concurrency::ModuleManaged,
9261 FrameSink::new(module_tx),
9262 )
9263 .unwrap();
9264
9265 ProbeHarness {
9266 spec,
9267 runtime,
9268 forwarding,
9269 module_connection,
9270 module_rx,
9271 handler,
9272 module,
9273 }
9274 }
9275
9276 async fn finish_after(
9277 harness: &mut ProbeHarness,
9278 stall: Duration,
9279 ) -> ModuleControlRpcCompletion {
9280 assert!(stall > harness.runtime.health.deadline);
9281 let deadline = harness.runtime.health.deadline;
9282 let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
9283 let answer = async {
9284 let frame = harness.module_rx.recv().await.expect("health.check frame");
9285 tokio::time::advance(deadline).await;
9286 tokio::task::yield_now().await;
9287 tokio::time::advance(stall - deadline).await;
9288 harness
9289 .forwarding
9290 .complete_module_control_rpc(
9291 harness.module_connection,
9292 frame.header.corr,
9293 Some("health.check"),
9294 ModuleControlRpcOutcome::Response(ModuleControlResponse::HealthCheck {
9295 status: HealthStatus::Ok,
9296 detail: None,
9297 metrics: None,
9298 }),
9299 )
9300 .unwrap()
9301 };
9302 let (probe_result, completion) = tokio::join!(probe, answer);
9303 let err = probe_result.expect_err("probe must miss its deadline");
9304 assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
9305 completion
9306 }
9307
9308 async fn time_out_without_answer(harness: &mut ProbeHarness) {
9309 let deadline = harness.runtime.health.deadline;
9310 let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
9311 let exhaust_deadline = async {
9312 let _frame = harness.module_rx.recv().await.expect("health.check frame");
9313 tokio::time::advance(deadline).await;
9314 tokio::task::yield_now().await;
9315 };
9316 let (probe_result, ()) = tokio::join!(probe, exhaust_deadline);
9317 let err = probe_result.expect_err("probe must miss its deadline");
9318 assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
9319 }
9320
9321 #[tokio::test(start_paused = true)]
9322 async fn late_health_answers_record_start_anchored_latency_for_two_stalls() {
9323 let mut harness = probe_harness();
9324
9325 let first = finish_after(&mut harness, Duration::from_secs(8)).await;
9326 let first_latency = match &first {
9327 ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
9328 other => panic!("late answer was not retained: {other:?}"),
9329 };
9330 assert!(harness.handler.observe_module_control_completion(first));
9331
9332 let second = finish_after(&mut harness, Duration::from_secs(11)).await;
9333 let second_latency = match &second {
9334 ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
9335 other => panic!("late answer was not retained: {other:?}"),
9336 };
9337 assert!(harness.handler.observe_module_control_completion(second));
9338
9339 assert_eq!(first_latency, Duration::from_secs(8));
9340 assert_eq!(
9341 second_latency - first_latency,
9342 Duration::from_secs(3),
9343 "latency must grow linearly with the additional stall"
9344 );
9345 let health = harness.module.status().unwrap().health;
9346 assert_eq!(health.late_answer_count, 2);
9347 assert_eq!(health.last_late_answer_latency_ms, Some(11_000));
9348 }
9349
9350 #[tokio::test(start_paused = true)]
9358 async fn late_answer_clears_the_consecutive_failure_streak() {
9359 let mut harness = probe_harness();
9360
9361 time_out_without_answer(&mut harness).await;
9363 harness
9364 .module
9365 .record_health_probe_failure_for_test("[no-answer] test miss")
9366 .unwrap();
9367 assert_eq!(
9368 harness.module.status().unwrap().health.consecutive_failures,
9369 1,
9370 "precondition: the miss must be on the streak before the late answer"
9371 );
9372
9373 let late = finish_after(&mut harness, Duration::from_secs(9)).await;
9375 assert!(matches!(
9376 late,
9377 ModuleControlRpcCompletion::LateHealthAnswer { .. }
9378 ));
9379 assert!(harness.handler.observe_module_control_completion(late));
9380
9381 let health = harness.module.status().unwrap().health;
9382 assert_eq!(
9383 health.consecutive_failures, 0,
9384 "a late answer is an answer: the streak must reset"
9385 );
9386 assert_eq!(health.late_answer_count, 1);
9387 }
9388
9389 #[tokio::test(start_paused = true)]
9390 async fn repeated_serial_probe_cycles_keep_one_tombstone_per_endpoint() {
9391 let mut harness = probe_harness();
9392
9393 for _ in 0..20 {
9394 time_out_without_answer(&mut harness).await;
9395 assert_eq!(
9396 harness.forwarding.health_probe_tombstone_count().unwrap(),
9397 1
9398 );
9399 }
9400 }
9401}
9402
9403#[cfg(test)]
9404mod child_env_tests {
9405 use super::{
9406 apply_child_env, apply_spawn_role, apply_wire_spawn_args, ModuleProtocol, ModuleSpec,
9407 SpawnRole, SupervisorHandle, SPAWN_ROLE_SWAP_CANDIDATE, SUBC_ARG, SUBC_LAUNCH_NONCE_ENV,
9408 SUBC_MODULE_ID_ENV, SUBC_SPAWN_ROLE_ENV,
9409 };
9410 use std::{ffi::OsStr, path::PathBuf};
9411 use tokio::process::Command;
9412
9413 fn spec(env: Vec<(String, String)>) -> ModuleSpec {
9414 ModuleSpec {
9415 module_id: "env-plan".to_string(),
9416 program: PathBuf::from("/nonexistent"),
9417 args: Vec::new(),
9418 env,
9419 reserved: false,
9420 reserved_prefixes: Vec::new(),
9421 protocol: ModuleProtocol::Subc,
9422 overlap: Default::default(),
9423 }
9424 }
9425
9426 #[test]
9440 fn ambient_ck_log_is_removed_and_a_configured_one_survives() {
9441 let mut command = Command::new("/nonexistent");
9442 apply_child_env(&mut command, &spec(Vec::new()));
9443 let removed = command
9444 .as_std()
9445 .get_envs()
9446 .any(|(key, value)| key == OsStr::new("CK_LOG") && value.is_none());
9447 assert!(
9448 removed,
9449 "ambient CK_LOG must be explicitly removed for an unconfigured module"
9450 );
9451
9452 let mut configured = Command::new("/nonexistent");
9453 apply_child_env(
9454 &mut configured,
9455 &spec(vec![("CK_LOG".to_string(), "debug".to_string())]),
9456 );
9457 let effective = configured
9458 .as_std()
9459 .get_envs()
9460 .filter(|(key, _)| *key == OsStr::new("CK_LOG"))
9461 .last()
9462 .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
9463 assert_eq!(
9464 effective,
9465 Some(Some("debug".to_string())),
9466 "a module's configured CK_LOG must survive the ambient removal"
9467 );
9468 }
9469
9470 #[test]
9479 fn protocol_none_spawn_carries_no_subc_argument_and_no_nonce() {
9480 let connection_file = std::path::Path::new("/run/subc-connection.json");
9481 let handle = SupervisorHandle::new();
9482
9483 let mut none_spec = spec(Vec::new());
9484 none_spec.protocol = ModuleProtocol::None;
9485 let mut none = Command::new("/nonexistent");
9486 let none_handoff =
9487 apply_wire_spawn_args(&mut none, &none_spec, Some(connection_file), Some(&handle))
9488 .expect("protocol-none spawn args apply");
9489 assert!(
9490 none_handoff.is_none(),
9491 "protocol:none spawn must not receive a nonce descriptor"
9492 );
9493 assert!(
9494 !none.as_std().get_envs().any(|(key, value)| key
9495 == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV)
9496 && value.is_some()),
9497 "protocol:none spawn must not name a nonce descriptor"
9498 );
9499 let none_args: Vec<String> = none
9500 .as_std()
9501 .get_args()
9502 .map(|a| a.to_string_lossy().into_owned())
9503 .collect();
9504 assert!(
9505 !none_args.iter().any(|a| a == SUBC_ARG),
9506 "protocol:none argv must not carry --subc; got {none_args:?}"
9507 );
9508 let none_has_nonce = none
9509 .as_std()
9510 .get_envs()
9511 .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some());
9512 assert!(
9513 !none_has_nonce,
9514 "protocol:none spawn must not receive a launch nonce"
9515 );
9516 let none_has_module_id = none
9517 .as_std()
9518 .get_envs()
9519 .any(|(key, value)| key == OsStr::new(SUBC_MODULE_ID_ENV) && value.is_some());
9520 assert!(
9521 none_has_module_id,
9522 "SUBC_MODULE_ID is inert and stays on every path"
9523 );
9524 assert!(
9525 handle.spawn_nonce(&none_spec.module_id).is_none(),
9526 "no nonce record for a process that will never present one"
9527 );
9528
9529 let wire_spec = spec(Vec::new());
9531 let mut wire = Command::new("/nonexistent");
9532 let wire_handoff =
9533 apply_wire_spawn_args(&mut wire, &wire_spec, Some(connection_file), Some(&handle))
9534 .expect("subc-wire spawn args apply");
9535 let wire_fd_env = wire
9536 .as_std()
9537 .get_envs()
9538 .find(|(key, _)| *key == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV))
9539 .and_then(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
9540 #[cfg(unix)]
9541 assert_eq!(
9542 wire_fd_env,
9543 Some(wire_handoff.expect("a descriptor handoff").fd_env_value()),
9544 "a subc-wire spawn names the pipe it will receive at descriptor 3"
9545 );
9546 #[cfg(not(unix))]
9547 assert!(wire_handoff.is_none() && wire_fd_env.is_none());
9548 let wire_args: Vec<String> = wire
9549 .as_std()
9550 .get_args()
9551 .map(|a| a.to_string_lossy().into_owned())
9552 .collect();
9553 assert_eq!(
9554 wire_args,
9555 vec![
9556 SUBC_ARG.to_string(),
9557 connection_file.to_string_lossy().into_owned()
9558 ],
9559 "a subc-wire spawn still carries --subc <path>"
9560 );
9561 assert!(wire
9562 .as_std()
9563 .get_envs()
9564 .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some()));
9565 assert!(handle.spawn_nonce(&wire_spec.module_id).is_some());
9566 }
9567
9568 #[test]
9578 fn plain_spawn_removes_the_spawn_role_even_when_the_spec_sets_it() {
9579 let role = |command: &Command| {
9580 command
9581 .as_std()
9582 .get_envs()
9583 .filter(|(key, _)| *key == OsStr::new(SUBC_SPAWN_ROLE_ENV))
9584 .last()
9585 .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()))
9586 };
9587 let forged = spec(vec![(
9588 SUBC_SPAWN_ROLE_ENV.to_string(),
9589 SPAWN_ROLE_SWAP_CANDIDATE.to_string(),
9590 )]);
9591
9592 let mut plain = Command::new("/nonexistent");
9593 apply_child_env(&mut plain, &forged);
9594 apply_spawn_role(&mut plain, SpawnRole::Plain);
9595 assert_eq!(
9596 role(&plain),
9597 Some(None),
9598 "a plain spawn must remove SUBC_SPAWN_ROLE, whatever the spec says"
9599 );
9600
9601 let mut candidate = Command::new("/nonexistent");
9602 apply_child_env(&mut candidate, &spec(Vec::new()));
9603 apply_spawn_role(&mut candidate, SpawnRole::SwapCandidate);
9604 assert_eq!(
9605 role(&candidate),
9606 Some(Some(SPAWN_ROLE_SWAP_CANDIDATE.to_string()))
9607 );
9608 }
9609
9610 #[test]
9616 fn daemon_private_capture_keys_are_not_passed_to_the_child() {
9617 let mut command = Command::new("/nonexistent");
9618 apply_child_env(
9619 &mut command,
9620 &spec(vec![
9621 (super::CAPTURE_KEEP_ENV.to_string(), "5".to_string()),
9622 ("KEPT".to_string(), "yes".to_string()),
9623 ]),
9624 );
9625 let keys: Vec<String> = command
9626 .as_std()
9627 .get_envs()
9628 .filter(|(_, value)| value.is_some())
9629 .map(|(key, _)| key.to_string_lossy().into_owned())
9630 .collect();
9631 assert!(keys.contains(&"KEPT".to_string()), "got {keys:?}");
9632 assert!(
9633 !keys.contains(&super::CAPTURE_KEEP_ENV.to_string()),
9634 "daemon-private capture key leaked to the child: {keys:?}"
9635 );
9636 }
9637}
9638
9639#[cfg(test)]
9640mod jitter_tests {
9641 use super::jittered_health_delay;
9642 use std::{collections::HashSet, time::Duration};
9643
9644 const FLEET: [&str; 14] = [
9653 "aft",
9654 "alfonso-core",
9655 "magic-context",
9656 "broca",
9657 "thalamus",
9658 "quota",
9659 "engram",
9660 "plexus",
9661 "cerebellum",
9662 "astrocyte",
9663 "synapse",
9664 "subc-mcp",
9665 "cortexkit-credentials",
9666 "subc-federation",
9667 ];
9668
9669 #[test]
9677 fn probe_delays_disperse_across_the_fleet() {
9678 let cadence = Duration::from_secs(30);
9679 let delays: HashSet<Duration> = FLEET
9680 .iter()
9681 .map(|id| jittered_health_delay(id, 0, cadence))
9682 .collect();
9683 assert_eq!(
9684 delays.len(),
9685 FLEET.len(),
9686 "every supervised module must land on its own probe offset"
9687 );
9688 }
9689
9690 #[test]
9696 fn jitter_only_delays_and_stays_within_one_tenth_of_cadence() {
9697 let cadence = Duration::from_secs(30);
9698 let span = cadence / 10;
9699 for id in FLEET {
9700 for probe_index in 0..8 {
9701 let delay = jittered_health_delay(id, probe_index, cadence);
9702 assert!(
9703 delay >= cadence,
9704 "{id}#{probe_index}: jitter must not shorten the cadence"
9705 );
9706 assert!(
9707 delay < cadence + span,
9708 "{id}#{probe_index}: jitter must stay inside one tenth of the cadence"
9709 );
9710 }
9711 }
9712 }
9713
9714 #[test]
9720 fn a_module_offset_is_stable_across_restarts() {
9721 let cadence = Duration::from_secs(30);
9722 for id in FLEET {
9723 assert_eq!(
9724 jittered_health_delay(id, 0, cadence),
9725 jittered_health_delay(id, 0, cadence),
9726 "{id}: the same module and probe index must produce the same offset"
9727 );
9728 }
9729 }
9730
9731 #[test]
9733 fn zero_cadence_yields_zero_delay() {
9734 assert_eq!(
9735 jittered_health_delay("aft", 0, Duration::ZERO),
9736 Duration::ZERO
9737 );
9738 }
9739}
9740
9741#[cfg(all(test, target_os = "linux"))]
9742mod cgroup_placement_tests {
9743 use super::{
9744 apply_cgroup_placement, remove_module_cgroup, ModuleProtocol, ModuleSpec, SuperviseError,
9745 SupervisedChild,
9746 };
9747 use crate::stderr_tail::{StderrRing, StderrTailConfig};
9748 use std::{
9749 fs, io,
9750 path::{Path, PathBuf},
9751 sync::{Arc, Mutex},
9752 };
9753 use subc_test_support::TestTempDir;
9754 use tokio::process::Command;
9755
9756 #[test]
9757 fn failed_parent_cgroup_open_is_a_cgroup_supervision_error() {
9758 let path = Path::new("/definitely-missing-subc-cgroup");
9759 let mut command = Command::new("true");
9760 let error = apply_cgroup_placement(
9761 &mut command,
9762 &ModuleSpec {
9763 module_id: "broken-cgroup".to_string(),
9764 program: PathBuf::from("true"),
9765 args: Vec::new(),
9766 env: Vec::new(),
9767 reserved: false,
9768 reserved_prefixes: Vec::new(),
9769 protocol: ModuleProtocol::Subc,
9770 overlap: Default::default(),
9771 },
9772 path,
9773 )
9774 .expect_err("a parent cgroup open failure must reject the supervised spawn");
9775 let reason = error.to_string();
9776
9777 assert!(
9778 matches!(error, SuperviseError::Cgroup { .. }),
9779 "parent cgroup open must be reported as a cgroup supervision error: {reason}"
9780 );
9781 assert!(
9782 reason.contains("/definitely-missing-subc-cgroup/cgroup.procs"),
9783 "parent cgroup open failure must name cgroup.procs: {reason}"
9784 );
9785 }
9786
9787 #[tokio::test]
9788 async fn reaping_a_child_removes_its_empty_module_cgroup() {
9789 let root = TestTempDir::new("supervisor-reap-cgroup");
9790 fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
9791 let placement = subc_cgroup::prepare_at(&root)
9792 .expect("prepare scratch cgroup root")
9793 .expect("scratch root has a cgroup.procs marker");
9794 let module_id = "reaped-module";
9795 let module = placement
9796 .module_path(module_id)
9797 .expect("create scratch module cgroup");
9798 let child = Command::new("true")
9799 .spawn()
9800 .expect("spawn short-lived child");
9801 let pid = child.id().expect("spawned child has pid");
9802 let mut child = SupervisedChild {
9803 child,
9804 module_id: module_id.to_string(),
9805 cgroup_placement: Some(placement),
9806 stdout_pump: None,
9807 stderr_pump: None,
9808 stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
9809 spawned_at_ms: 0,
9810 spawned_from: PathBuf::from("true"),
9811 spawned_file_identity: None,
9812 process_start_time: None,
9813 process_identity: None,
9814 pid,
9815 roster_guard: None,
9816 };
9817
9818 child.wait().await.expect("reap short-lived child");
9819
9820 assert!(
9821 !module.exists(),
9822 "reaping the supervised child must remove its empty cgroup"
9823 );
9824 }
9825
9826 #[test]
9827 fn non_empty_cgroup_removal_is_reported_without_blocking_teardown() {
9828 let root = TestTempDir::new("supervisor-non-empty-cgroup");
9829 fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
9830 let placement = subc_cgroup::prepare_at(&root)
9831 .expect("prepare scratch cgroup root")
9832 .expect("scratch root has a cgroup.procs marker");
9833 let module = placement
9834 .module_path("surviving-module")
9835 .expect("create scratch module cgroup");
9836 fs::write(module.join("surviving-process"), b"still present")
9837 .expect("make scratch cgroup non-empty");
9838 let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::WARN);
9839
9840 remove_module_cgroup(&placement, "surviving-module");
9841
9842 let logs = crate::router::test_log::captured_logs(&logs);
9843 assert!(
9844 module.exists(),
9845 "failed removal must leave the cgroup intact"
9846 );
9847 assert!(
9848 logs.contains("could not remove module cgroup after process exit; continuing teardown")
9849 && logs.contains("surviving-module"),
9850 "best-effort removal must report the failure without returning it: {logs}"
9851 );
9852 }
9853
9854 #[test]
9855 fn cgroup_pre_exec_spawn_failure_names_the_cgroup_path() {
9856 let cgroup_path = PathBuf::from("/sys/fs/cgroup/subc-modules/broken-module");
9857 let reason = SuperviseError::Spawn {
9858 program: PathBuf::from("/bin/true"),
9859 source: io::Error::from_raw_os_error(13),
9860 cgroup_path: Some(cgroup_path.clone()),
9861 }
9862 .to_string();
9863
9864 assert!(
9865 reason.contains(&cgroup_path.display().to_string()),
9866 "a pre_exec spawn failure must name the cgroup path: {reason}"
9867 );
9868 }
9869}
9870
9871#[cfg(test)]
9872mod spawn_subscriber_lag_tests {
9873 use super::*;
9874
9875 #[tokio::test]
9880 async fn lagged_spawn_subscriber_receives_a_terminal_lagged_error_after_its_queued_frames() {
9881 let feed = SpawnEventFeed::default();
9882 feed.configure_incarnation("lag-incarnation".to_string());
9883 let (tx, mut rx) = mpsc::channel(1);
9886 feed.subscribe(ConnectionId::new(1), 7, 1, None, FrameSink::new(tx))
9887 .expect("subscribe");
9888 let emitted = SPAWN_SUBSCRIBER_BUFFER + 16;
9889 for index in 0..emitted {
9890 feed.emit_spawned(&format!("lag-module-{index}"), 1000, 0);
9891 tokio::task::yield_now().await;
9894 }
9895 assert_eq!(
9896 feed.subscriber_count(),
9897 0,
9898 "the lagged subscriber must be removed"
9899 );
9900
9901 let mut data = Vec::new();
9902 let mut last = None;
9903 loop {
9904 let next = tokio::time::timeout(Duration::from_secs(5), rx.recv())
9905 .await
9906 .expect("the forwarder must finish once the subscriber is dropped");
9907 let Some(outbound) = next else { break };
9908 let frame = outbound.frame;
9909 if frame.header.ty == FrameType::StreamData {
9910 assert!(last.is_none(), "no data may follow the terminal frame");
9911 let event: SpawnEvent = serde_json::from_slice(&frame.body).unwrap();
9912 data.push(event.cursor.seq);
9913 } else {
9914 assert!(last.is_none(), "exactly one terminal frame");
9915 last = Some(frame);
9916 }
9917 }
9918 assert!(!data.is_empty(), "queued frames drain before the terminal");
9919 for pair in data.windows(2) {
9920 assert_eq!(
9921 pair[1],
9922 pair[0] + 1,
9923 "queued frames arrive dense and in order"
9924 );
9925 }
9926 let terminal = last.expect("a lagged subscriber must receive a terminal frame");
9927 assert_eq!(terminal.header.ty, FrameType::Error);
9928 assert_eq!(terminal.header.corr, 7);
9929 let body: subc_protocol::ErrorBody = serde_json::from_slice(&terminal.body).unwrap();
9930 assert_eq!(body.code, SPAWN_SUBSCRIBER_LAGGED_CODE);
9931 let detail = body.detail.expect("lagged error carries detail");
9932 assert_eq!(
9933 detail["first_undelivered_cursor"]["seq"],
9934 data.last().unwrap() + 1,
9935 "the named cursor is the first event the subscriber did not receive"
9936 );
9937 assert_eq!(
9938 detail["first_undelivered_cursor"]["daemon_incarnation"],
9939 "lag-incarnation"
9940 );
9941 }
9942}
9943
9944#[cfg(test)]
9945mod terminal_history_read_concurrency_tests {
9946 use super::*;
9947 use crate::terminal_journal::read_pause;
9948 use std::sync::mpsc as std_mpsc;
9949 use subc_test_support::TestTempDir;
9950
9951 fn journaled_ring(
9952 journal: &Arc<crate::terminal_journal::TerminalJournal>,
9953 ) -> Arc<Mutex<TerminalRing>> {
9954 Arc::new(Mutex::new(
9955 TerminalRing::new(TerminalRingConfig::default(), 1)
9956 .with_journal(Some(Arc::clone(journal))),
9957 ))
9958 }
9959
9960 fn crash(at_ms: u64) -> ExitReport {
9961 ExitReport {
9962 kind: ExitKind::Crash,
9963 code: Some(1),
9964 signal: None,
9965 at_ms,
9966 }
9967 }
9968
9969 fn record_within(
9972 module_id: &'static str,
9973 ring: &Arc<Mutex<TerminalRing>>,
9974 at_ms: u64,
9975 bound: Duration,
9976 ) -> bool {
9977 let ring = Arc::clone(ring);
9978 let (done, done_rx) = std_mpsc::channel();
9979 std::thread::spawn(move || {
9980 record_terminal(
9981 module_id,
9982 &ring,
9983 &SpawnEventFeed::default(),
9984 &crash(at_ms),
9985 TerminalDisposition::Restarting,
9986 );
9987 let _ = done.send(());
9988 });
9989 done_rx.recv_timeout(bound).is_ok()
9990 }
9991
9992 #[test]
9997 fn exits_recorded_during_a_paused_history_read_are_not_blocked_or_half_merged() {
9998 let dir = TestTempDir::new("terminal-history-concurrent-read");
9999 let path = dir.join("terminals.jsonl");
10000 let journal = Arc::new(crate::terminal_journal::TerminalJournal::open(
10001 path.clone(),
10002 "daemon".into(),
10003 ));
10004 let reader_ring = journaled_ring(&journal);
10005 let other_ring = journaled_ring(&journal);
10006 assert!(record_within(
10007 "reader-module",
10008 &reader_ring,
10009 10,
10010 Duration::from_secs(5)
10011 ));
10012
10013 let (started, release) = read_pause::install(&path);
10014 let reading = {
10015 let ring = Arc::clone(&reader_ring);
10016 std::thread::spawn(move || durable_terminal_history_of(&ring, "reader-module"))
10017 };
10018 started
10019 .recv_timeout(Duration::from_secs(5))
10020 .expect("the history read reached its pause");
10021
10022 let bound = Duration::from_secs(1);
10023 assert!(
10024 record_within("other-module", &other_ring, 20, bound),
10025 "another module's exit waited on a history read (journal writer held)"
10026 );
10027 assert!(
10028 record_within("reader-module", &reader_ring, 30, bound),
10029 "the read module's own exit waited on its history read (ring held)"
10030 );
10031
10032 drop(release);
10033 let paused = reading.join().unwrap();
10034 assert_eq!(
10035 paused.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
10036 vec![10],
10037 "an exit recorded after the read began lands in neither half of it"
10038 );
10039 assert_eq!(paused.journal_skipped_lines, 0);
10040 assert_eq!(paused.journal_read_errors, 0);
10041
10042 let after = durable_terminal_history_of(&reader_ring, "reader-module");
10043 assert_eq!(
10044 after.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
10045 vec![10, 30],
10046 "the next read merges ring and journal with no duplicate"
10047 );
10048 assert_eq!(after.journal_skipped_lines, 0);
10049 }
10050}
10051
10052#[cfg(test)]
10057mod stderr_settle_tests {
10058 use std::{
10059 future::Future,
10060 io,
10061 pin::Pin,
10062 sync::{Arc, Mutex},
10063 task::{Context, Poll},
10064 time::Duration,
10065 };
10066
10067 use tokio::{
10068 io::{AsyncRead, ReadBuf},
10069 sync::oneshot,
10070 time::Instant,
10071 };
10072
10073 use super::{settle_stderr_pump, StderrPump};
10074 use crate::stderr_tail::{
10075 pump_stderr_to, untimed, CaptureState, OutputSink, StderrRing, StderrTailConfig, TailEntry,
10076 };
10077
10078 const BOUND: Duration = Duration::from_millis(250);
10079
10080 struct HeldReader {
10084 before: Option<Vec<u8>>,
10085 gate: Option<oneshot::Receiver<()>>,
10086 after: io::Cursor<Vec<u8>>,
10087 }
10088
10089 impl AsyncRead for HeldReader {
10090 fn poll_read(
10091 mut self: Pin<&mut Self>,
10092 cx: &mut Context<'_>,
10093 buf: &mut ReadBuf<'_>,
10094 ) -> Poll<io::Result<()>> {
10095 if let Some(bytes) = self.before.take() {
10096 buf.put_slice(&bytes);
10097 return Poll::Ready(Ok(()));
10098 }
10099 if let Some(gate) = self.gate.as_mut() {
10100 match Pin::new(gate).poll(cx) {
10101 Poll::Pending => return Poll::Pending,
10102 Poll::Ready(_) => self.gate = None,
10103 }
10104 }
10105 Pin::new(&mut self.after).poll_read(cx, buf)
10106 }
10107 }
10108
10109 struct DiscardSink;
10110
10111 impl OutputSink for DiscardSink {
10112 fn write_line(&mut self, _line: &[u8]) {}
10113 }
10114
10115 fn line(text: &str) -> TailEntry {
10116 TailEntry::Line {
10117 text: text.to_string(),
10118 truncated: false,
10119 at_ms: None,
10120 }
10121 }
10122
10123 fn lock(ring: &Arc<Mutex<StderrRing>>) -> std::sync::MutexGuard<'_, StderrRing> {
10124 ring.lock().unwrap()
10125 }
10126
10127 fn held_pump(
10131 ring: &Arc<Mutex<StderrRing>>,
10132 before: &str,
10133 after: &str,
10134 ) -> (StderrPump, oneshot::Sender<()>) {
10135 let generation = lock(ring).begin_process();
10136 let (release, gate) = oneshot::channel();
10137 let reader = HeldReader {
10138 before: Some(before.as_bytes().to_vec()),
10139 gate: Some(gate),
10140 after: io::Cursor::new(after.as_bytes().to_vec()),
10141 };
10142 let task = tokio::spawn(pump_stderr_to(
10143 reader,
10144 Arc::clone(ring),
10145 generation,
10146 DiscardSink,
10147 ));
10148 (StderrPump { task, generation }, release)
10149 }
10150
10151 async fn wait_until(ring: &Arc<Mutex<StderrRing>>, done: impl Fn(&StderrRing) -> bool) {
10152 for _ in 0..1000 {
10153 if done(&lock(ring)) {
10154 return;
10155 }
10156 tokio::time::sleep(Duration::from_millis(1)).await;
10157 }
10158 panic!(
10159 "ring never reached the expected state: {:?}",
10160 lock(ring).snapshot(None, None)
10161 );
10162 }
10163
10164 #[tokio::test(start_paused = true)]
10165 async fn a_crash_line_the_reader_had_not_reached_by_the_bound_is_kept_before_the_restart() {
10166 let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
10167 let (pump, release) = held_pump(&ring, "booting\n", "config error: missing storage\n");
10168
10169 settle_stderr_pump("crasher", &ring, pump, BOUND).await;
10170 let before_release = lock(&ring).snapshot(None, None);
10171 assert!(
10172 matches!(before_release.capture, CaptureState::Incomplete { .. }),
10173 "a reader that has not reached EOF cannot claim a whole tail: {before_release:?}"
10174 );
10175
10176 let next = lock(&ring).begin_process();
10179 lock(&ring).push_line_from(next, "next process booting");
10180 release.send(()).unwrap();
10181 wait_until(&ring, |ring| {
10182 ring.snapshot(None, None).capture == CaptureState::Captured
10183 })
10184 .await;
10185
10186 assert_eq!(
10187 untimed(lock(&ring).snapshot(None, None).entries),
10188 vec![
10189 line("booting"),
10190 line("config error: missing storage"),
10191 TailEntry::ProcessStart,
10192 line("next process booting"),
10193 ],
10194 "the crash's last line must survive a slow reader and stay in the crashed process's section"
10195 );
10196 }
10197
10198 #[tokio::test(start_paused = true)]
10199 async fn a_pipe_held_open_by_a_descendant_reads_incomplete_without_delaying_the_restart_past_the_bound(
10200 ) {
10201 let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
10202 let (pump, _held) = held_pump(&ring, "parent exiting\n", "");
10205
10206 let started = Instant::now();
10207 settle_stderr_pump("orphaning", &ring, pump, BOUND).await;
10208 assert_eq!(
10209 started.elapsed(),
10210 BOUND,
10211 "the restart must wait exactly the bound for a pipe that stays open, no longer"
10212 );
10213
10214 let next = lock(&ring).begin_process();
10215 lock(&ring).push_line_from(next, "next process booting");
10216 tokio::time::sleep(Duration::from_secs(60)).await;
10217
10218 let snapshot = lock(&ring).snapshot(None, None);
10219 match &snapshot.capture {
10220 CaptureState::Incomplete { reason } => assert!(
10221 reason.contains("had not reached EOF") && reason.contains("250ms"),
10222 "the reason must say what is missing and after how long: {reason}"
10223 ),
10224 other => panic!("expected Incomplete while the pipe is held open, got {other:?}"),
10225 }
10226 assert_eq!(
10227 untimed(snapshot.entries),
10228 vec![
10229 line("parent exiting"),
10230 TailEntry::ProcessStart,
10231 line("next process booting"),
10232 ]
10233 );
10234 }
10235
10236 #[tokio::test(start_paused = true)]
10237 async fn a_reader_that_reaches_eof_within_the_bound_leaves_the_tail_captured() {
10238 let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
10239 let (pump, release) = held_pump(&ring, "one\n", "two\n");
10240 release.send(()).unwrap();
10241
10242 settle_stderr_pump("clean", &ring, pump, BOUND).await;
10243
10244 let snapshot = lock(&ring).snapshot(None, None);
10245 assert_eq!(snapshot.capture, CaptureState::Captured);
10246 assert_eq!(untimed(snapshot.entries), vec![line("one"), line("two")]);
10247 }
10248}
10249
10250#[cfg(all(test, windows))]
10264mod job_containment_tests {
10265 use super::*;
10266 use std::{
10267 path::{Path, PathBuf},
10268 sync::{Arc, Mutex},
10269 time::{Duration, Instant},
10270 };
10271 use subc_test_support::TestTempDir;
10272
10273 fn stub_path() -> PathBuf {
10279 let mut path = std::env::current_exe().expect("current_exe available in tests");
10280 path.pop();
10281 path.pop();
10282 path.push("fake-aft-stub.exe");
10283 assert!(
10284 path.exists(),
10285 "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
10286 [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
10287 path.display()
10288 );
10289 path
10290 }
10291
10292 fn read_grandchild_pid(path: &Path) -> u32 {
10294 let deadline = Instant::now() + Duration::from_secs(10);
10295 loop {
10296 if let Ok(contents) = std::fs::read_to_string(path) {
10297 if let Ok(pid) = contents.trim().parse() {
10298 return pid;
10299 }
10300 }
10301 assert!(
10302 Instant::now() < deadline,
10303 "the stub never recorded a grandchild pid at {}",
10304 path.display()
10305 );
10306 std::thread::sleep(Duration::from_millis(10));
10307 }
10308 }
10309
10310 struct Fixture {
10313 _dir: TestTempDir,
10314 module_id: String,
10315 grandchild: u32,
10316 child: Option<SupervisedChild>,
10317 registry: Arc<Registry>,
10318 snapshot: Arc<Mutex<SupervisorSnapshot>>,
10319 terminal_ring: Arc<Mutex<TerminalRing>>,
10320 spawn_events: SpawnEventFeed,
10321 }
10322
10323 fn fixture(label: &str, module_id: &str) -> Fixture {
10324 let dir = TestTempDir::new(label);
10325 let pid_file = dir.join("grandchild.pid");
10326 let supervisor = Supervisor::new(
10327 Arc::new(Registry::default()),
10328 RestartPolicy::new(3, Duration::ZERO),
10329 );
10330 let runtime = supervisor.runtime_config();
10331 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10332 let spec = ModuleSpec {
10333 module_id: module_id.to_string(),
10334 program: stub_path(),
10335 args: Vec::new(),
10339 env: vec![
10340 ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
10341 (
10342 "FAKE_AFT_GRANDCHILD_PID_FILE".to_string(),
10343 pid_file.display().to_string(),
10344 ),
10345 ],
10346 reserved: false,
10347 reserved_prefixes: Vec::new(),
10348 protocol: ModuleProtocol::Subc,
10349 overlap: Default::default(),
10350 };
10351 let child = spawn_and_mark_running(&spec, &runtime, &snapshot)
10352 .expect("spawn the supervised fixture");
10353 let grandchild = read_grandchild_pid(&pid_file);
10354 Fixture {
10355 _dir: dir,
10356 module_id: module_id.to_string(),
10357 grandchild,
10358 child: Some(child),
10359 registry: Arc::new(Registry::default()),
10360 snapshot,
10361 terminal_ring: Arc::clone(&runtime.terminal_ring),
10362 spawn_events: SpawnEventFeed::default(),
10363 }
10364 }
10365
10366 impl Fixture {
10367 async fn drain(&mut self) {
10369 let child = self
10370 .child
10371 .take()
10372 .expect("the fixture child is still present");
10373 drain_child_to_state(
10374 &self.module_id,
10375 ModuleProtocol::Subc,
10376 StopNotice::NotSent,
10379 &self.registry,
10380 &self.snapshot,
10381 &self.terminal_ring,
10382 &self.spawn_events,
10383 child,
10384 Duration::from_millis(500),
10385 ModuleState::Stopped,
10386 Some(false),
10387 )
10388 .await
10389 .expect("drain the supervised fixture");
10390 }
10391 }
10392
10393 #[tokio::test]
10399 async fn teardown_reaps_the_grandchild() {
10400 let mut fixture = fixture("teardown-grandchild", "tree-teardown");
10401 let grandchild = fixture.grandchild;
10402
10403 assert!(
10404 subc_jobobject::process_exists(grandchild),
10405 "grandchild {grandchild} must be alive before teardown, or this proves nothing"
10406 );
10407
10408 fixture.drain().await;
10409
10410 assert!(
10411 subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
10412 "grandchild {grandchild} outlived module teardown: the tree was not contained"
10413 );
10414 }
10415
10416 #[test]
10429 fn an_uncontained_grandchild_survives_a_direct_child_kill() {
10430 let dir = TestTempDir::new("teardown-uncontained");
10431 let pid_file = dir.join("grandchild.pid");
10432 let mut child = std::process::Command::new(stub_path())
10433 .env("FAKE_AFT_NEVER_CONNECT", "1")
10434 .env(
10435 "FAKE_AFT_GRANDCHILD_PID_FILE",
10436 pid_file.display().to_string(),
10437 )
10438 .stdin(std::process::Stdio::null())
10439 .stdout(std::process::Stdio::null())
10440 .stderr(std::process::Stdio::null())
10441 .spawn()
10442 .expect("spawn the uncontained fixture");
10443 let grandchild = read_grandchild_pid(&pid_file);
10444
10445 child.kill().expect("kill the direct child");
10447 let _ = child.wait();
10448
10449 assert!(
10450 subc_jobobject::process_exists(grandchild),
10451 "grandchild {grandchild} died with the direct child, so this control no longer \
10452 distinguishes contained from uncontained teardown and the regression test is \
10453 passing vacuously"
10454 );
10455
10456 kill_tree(grandchild);
10459 }
10460
10461 #[tokio::test]
10470 async fn dropping_containment_reaps_the_grandchild() {
10471 let mut fixture = fixture("drop-containment", "tree-drop");
10472 let grandchild = fixture.grandchild;
10473
10474 assert!(subc_jobobject::process_exists(grandchild));
10475
10476 fixture.child.as_mut().expect("child present").job = None;
10478
10479 assert!(
10480 subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
10481 "grandchild {grandchild} survived the containment handle closing, so a daemon \
10482 crash would leave the tree behind"
10483 );
10484 }
10485
10486 fn kill_tree(pid: u32) {
10488 let _ = std::process::Command::new("taskkill.exe")
10489 .args(["/PID", &pid.to_string(), "/T", "/F"])
10490 .stdin(std::process::Stdio::null())
10491 .stdout(std::process::Stdio::null())
10492 .stderr(std::process::Stdio::null())
10493 .status();
10494 assert!(
10495 subc_jobobject::wait_for_process_exit(pid, Duration::from_secs(10)),
10496 "could not clean up grandchild {pid}"
10497 );
10498 }
10499}
10500
10501#[cfg(all(test, unix))]
10506mod launch_nonce_descriptor_tests {
10507 use super::{spawn_child, ChildRoster, ModuleProtocol, ModuleSpec, SupervisorHandle};
10508 use crate::stderr_tail::{StderrRing, StderrTailConfig};
10509 use std::{
10510 path::PathBuf,
10511 sync::{Arc, Mutex},
10512 time::{Duration, Instant},
10513 };
10514 use subc_test_support::TestTempDir;
10515
10516 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
10517 async fn a_spawned_module_receives_its_nonce_on_descriptor_3_and_in_the_environment() {
10518 let scratch = TestTempDir::new("launch-nonce-descriptor");
10519 let fd_copy = scratch.join("from-descriptor");
10520 let env_copy = scratch.join("from-environment");
10521 let script = format!(
10522 "cat <&3 > '{fd}.tmp' && mv '{fd}.tmp' '{fd}'; \
10523 printf %s \"$SUBC_LAUNCH_NONCE\" > '{env}.tmp' && mv '{env}.tmp' '{env}'; \
10524 sleep 30",
10525 fd = fd_copy.display(),
10526 env = env_copy.display(),
10527 );
10528 let xdg = |name: &str| (name.to_string(), scratch.join(name).display().to_string());
10529 let spec = ModuleSpec {
10530 module_id: "nonce-descriptor-probe".to_string(),
10531 program: PathBuf::from("/bin/sh"),
10532 args: vec!["-c".to_string(), script],
10533 env: vec![
10534 xdg("XDG_DATA_HOME"),
10535 xdg("XDG_RUNTIME_DIR"),
10536 xdg("XDG_CONFIG_HOME"),
10537 ],
10538 reserved: true,
10539 reserved_prefixes: Vec::new(),
10540 protocol: ModuleProtocol::Subc,
10541 overlap: Default::default(),
10542 };
10543 let handle = SupervisorHandle::new();
10544 let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
10545 let roster = ChildRoster::default();
10546 let child = spawn_child(
10547 &spec,
10548 None,
10549 Some(&handle),
10550 &ring,
10551 None,
10552 &roster,
10553 #[cfg(target_os = "linux")]
10554 None,
10555 )
10556 .expect("spawn the probe");
10557 let expected = handle
10558 .spawn_nonce("nonce-descriptor-probe")
10559 .expect("a recorded spawn nonce");
10560
10561 let deadline = Instant::now() + Duration::from_secs(10);
10562 while !(fd_copy.exists() && env_copy.exists()) {
10563 assert!(
10564 Instant::now() < deadline,
10565 "the probe never wrote its copies"
10566 );
10567 tokio::time::sleep(Duration::from_millis(20)).await;
10568 }
10569 assert_eq!(
10570 std::fs::read_to_string(&fd_copy).unwrap(),
10571 expected,
10572 "descriptor 3 must hold the spawn nonce"
10573 );
10574 assert_eq!(
10575 std::fs::read_to_string(&env_copy).unwrap(),
10576 expected,
10577 "the environment copy is still set during the rollout"
10578 );
10579 drop(child);
10580 }
10581}