1use std::{
2 collections::{HashMap, VecDeque},
3 error::Error,
4 fmt, io,
5 path::PathBuf,
6 process::{ExitStatus, Stdio},
7 sync::{Arc, Mutex, OnceLock},
8 time::{Duration, SystemTime, UNIX_EPOCH},
9};
10
11use cortexkit_log::Retention;
12use serde_json::Value;
13use subc_control::{
14 ClientControlPush, LiveSpawn, ModuleProtocol, RouteCloseReason, SpawnCursor, SpawnEvent,
15 SpawnEventKind, SpawnSnapshot, SupervisorHealthStatus, TerminalDisposition, TerminalExitKind,
16};
17use subc_protocol::{
18 manifest::{SelfSignalKind, SignalAnchor},
19 session::{
20 HealthReport, HealthStatus, ModuleControlCommand, ModuleControlRequest,
21 MODULE_CONTROL_OP_HEALTH_CHECK,
22 },
23 Flags, FrameType, Priority, SUBC_LAUNCH_NONCE_ENV, SUBC_MODULE_ID_ENV,
24};
25use tokio::{
26 process::{Child, Command},
27 sync::{mpsc, oneshot, watch, Mutex as AsyncMutex},
28 task::JoinHandle,
29 time::{sleep, sleep_until, timeout, timeout_at, Instant},
30};
31use tracing::{debug, error, info, warn};
32
33use crate::{
34 child_roster::ChildRoster,
35 daemon_config::{
36 CAPTURE_KEEP_ENV, CAPTURE_MAX_AGE_DAYS_ENV, CAPTURE_MAX_FILE_MB_ENV, CK_LOG_ENV,
37 },
38 forwarding::{
39 CloseReason, ForwardingError, ForwardingTable, GoodbyeTarget, ModuleControlRpcOutcome,
40 ModuleDrainTarget, PendingModuleControlRpc,
41 },
42 provenance::{spawned_file_identity, ExecutableIdentityProbe, SpawnedFileIdentity},
43 registry::{ConnectionId, RegistryError},
44 stderr_tail::{
45 pump_stderr_to, pump_stdout_to, ChildOutputSink, StderrRing, StderrTailConfig,
46 StderrTailSnapshot,
47 },
48 terminal_ring::{TerminalHistorySnapshot, TerminalRecord, TerminalRing, TerminalRingConfig},
49 Frame, FrameSink, Registry,
50};
51
52#[path = "supervise_swap.rs"]
53mod swap;
54
55pub const SUBC_ARG: &str = "--subc";
61
62const DEFAULT_MAX_RESTARTS: u32 = 3;
63const DEFAULT_BACKOFF: Duration = Duration::from_millis(100);
64const DEFAULT_MAX_BACKOFF: Duration = Duration::from_secs(30);
65const DEFAULT_RESTART_WINDOW: Duration = Duration::from_secs(600);
69pub const DEFAULT_DRAIN_TIMEOUT: Duration = Duration::from_secs(30);
80const REGISTRY_RELEASE_TIMEOUT: Duration = Duration::from_secs(1);
81const REGISTRY_RELEASE_POLL: Duration = Duration::from_millis(10);
82const STDERR_PUMP_DRAIN_TIMEOUT: Duration = Duration::from_millis(250);
104pub const SPAWN_EVENT_RING_CAPACITY: usize = 4096;
106const SPAWN_SUBSCRIBER_BUFFER: usize = SPAWN_EVENT_RING_CAPACITY + 1;
107pub(crate) const SPAWN_SUBSCRIBER_LAGGED_CODE: &str = "spawn_subscriber_lagged";
112
113struct SupervisedChild {
114 child: Child,
115 #[cfg(target_os = "linux")]
118 module_id: String,
119 #[cfg(target_os = "linux")]
120 cgroup_placement: Option<subc_cgroup::Placement>,
121 #[cfg(windows)]
143 job: Option<subc_jobobject::JobObject>,
144 stdout_pump: Option<JoinHandle<()>>,
145 stderr_pump: Option<StderrPump>,
146 stderr_ring: Arc<Mutex<StderrRing>>,
147 spawned_at_ms: u64,
148 spawned_from: PathBuf,
149 spawned_file_identity: Option<SpawnedFileIdentity>,
150 process_start_time: Option<u64>,
151 process_identity: Option<ProcessIdentity>,
152 pid: u32,
153 roster_guard: Option<crate::child_roster::RosterGuard>,
156}
157
158impl SupervisedChild {
159 fn id(&self) -> Option<u32> {
160 Some(self.pid)
161 }
162
163 fn process_identity(&self) -> Option<ProcessIdentity> {
164 self.process_identity
165 }
166
167 async fn wait(&mut self) -> io::Result<ExitStatus> {
168 let result = self.child.wait().await;
176 #[cfg(target_os = "linux")]
177 if result.is_ok() {
178 if let Some(placement) = self.cgroup_placement.take() {
179 remove_module_cgroup(&placement, &self.module_id);
180 }
181 }
182 result
183 }
184
185 fn release_roster(&mut self) {
189 self.roster_guard = None;
190 }
191
192 fn start_kill(&mut self) -> io::Result<()> {
205 #[cfg(windows)]
206 if let Some(job) = &self.job {
207 if let Err(error) = job.terminate() {
208 debug!(
209 error = %error,
210 "job termination failed; the direct-child kill still owns the outcome"
211 );
212 }
213 }
214 #[cfg(target_os = "linux")]
215 kill_module_cgroup(self.cgroup_placement.as_ref(), &self.module_id);
216 self.child.start_kill()
217 }
218
219 async fn drain_stderr(&mut self, module_id: &str) {
220 if let Some(mut pump) = self.stdout_pump.take() {
221 match timeout(STDERR_PUMP_DRAIN_TIMEOUT, &mut pump).await {
222 Ok(Ok(())) => {}
223 Ok(Err(error)) => {
224 warn!(module_id, error = %error, "stdout pump ended unexpectedly");
225 }
226 Err(_) => {
227 pump.abort();
228 warn!(
229 module_id,
230 waited = ?STDERR_PUMP_DRAIN_TIMEOUT,
231 "stdout pump did not drain before restart; stopped it before the next process"
232 );
233 }
234 }
235 }
236
237 let Some(pump) = self.stderr_pump.take() else {
238 return;
239 };
240 settle_stderr_pump(
241 module_id,
242 &self.stderr_ring,
243 pump,
244 STDERR_PUMP_DRAIN_TIMEOUT,
245 )
246 .await;
247 }
248}
249
250struct StderrPump {
253 task: JoinHandle<()>,
254 generation: u64,
255}
256
257async fn settle_stderr_pump(
263 module_id: &str,
264 ring: &Arc<Mutex<StderrRing>>,
265 pump: StderrPump,
266 bound: Duration,
267) {
268 let lock = || ring.lock().unwrap_or_else(|poisoned| poisoned.into_inner());
269 let StderrPump {
270 mut task,
271 generation,
272 } = pump;
273 lock().retire_pump(generation);
274 match timeout(bound, &mut task).await {
275 Ok(Ok(())) => {}
276 Ok(Err(err)) => {
277 let mut ring = lock();
278 ring.mark_incomplete(format!("stderr pump ended unexpectedly: {err}"));
279 ring.finish_pump(generation);
280 warn!(module_id, error = %err, "stderr pump ended before clean EOF");
281 }
282 Err(_) => {
283 drop(task);
285 lock().mark_pump_late(
286 generation,
287 format!(
288 "stderr of the exited process had not reached EOF {bound:?} after it was \
289 retired (a descendant may still hold the pipe open); lines it still \
290 writes are kept in that process's section"
291 ),
292 );
293 warn!(
294 module_id,
295 waited = ?bound,
296 "stderr pipe of the exited process is still open; its reader keeps running without delaying the restart"
297 );
298 }
299 }
300}
301
302fn registration_release_events() -> &'static watch::Sender<u64> {
303 static EVENTS: OnceLock<watch::Sender<u64>> = OnceLock::new();
304 EVENTS.get_or_init(|| {
305 let (sender, _receiver) = watch::channel(0);
306 sender
307 })
308}
309
310pub(crate) fn notify_registration_release() {
311 let events = registration_release_events();
312 let next_generation = (*events.borrow()).wrapping_add(1);
313 events.send_replace(next_generation);
314}
315
316#[derive(Debug, Clone, PartialEq, Eq)]
318pub struct ModuleSpec {
319 pub module_id: String,
320 pub program: PathBuf,
321 pub args: Vec<String>,
322 pub env: Vec<(String, String)>,
323 pub reserved: bool,
328 pub reserved_prefixes: Vec<String>,
333 pub protocol: ModuleProtocol,
352 pub overlap: ModuleOverlap,
357}
358
359#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
366pub enum ModuleOverlap {
367 #[default]
369 Exclusive,
370 Safe,
382}
383
384impl ModuleOverlap {
385 pub fn as_str(self) -> &'static str {
386 match self {
387 Self::Exclusive => "exclusive",
388 Self::Safe => "safe",
389 }
390 }
391}
392
393pub const SUBC_SPAWN_ROLE_ENV: &str = "SUBC_SPAWN_ROLE";
403pub const SPAWN_ROLE_SWAP_CANDIDATE: &str = "swap_candidate";
405pub const DEFAULT_SWAP_READY_TIMEOUT: Duration = Duration::from_secs(100);
410
411#[derive(Debug, Clone, Copy, PartialEq, Eq)]
429pub struct RestartPolicy {
430 pub max_restarts: u32,
431 pub backoff: Duration,
434 pub max_backoff: Duration,
436 pub window: Duration,
440}
441
442impl RestartPolicy {
443 pub fn new(max_restarts: u32, backoff: Duration) -> Self {
447 Self {
448 max_restarts,
449 backoff,
450 max_backoff: DEFAULT_MAX_BACKOFF,
451 window: DEFAULT_RESTART_WINDOW,
452 }
453 }
454
455 pub fn with_max_backoff(mut self, max_backoff: Duration) -> Self {
456 self.max_backoff = max_backoff;
457 self
458 }
459
460 pub fn with_window(mut self, window: Duration) -> Self {
461 self.window = window;
462 self
463 }
464
465 fn delay_for_restart(&self, restart_in_window: u32) -> Duration {
470 if self.backoff.is_zero() || self.max_backoff.is_zero() {
471 return Duration::ZERO;
472 }
473
474 let mut delay = self.backoff;
475 for _ in 0..restart_in_window {
476 if delay >= self.max_backoff {
477 return self.max_backoff;
478 }
479 delay = delay
480 .checked_mul(10)
481 .unwrap_or(self.max_backoff)
482 .min(self.max_backoff);
483 }
484 delay.min(self.max_backoff)
485 }
486
487 fn budget_exhausted_detail(&self) -> String {
492 format!(
493 "crash budget exhausted: max_restarts={} within window_secs={}",
494 self.max_restarts,
495 self.window.as_secs()
496 )
497 }
498}
499
500impl Default for RestartPolicy {
501 fn default() -> Self {
502 Self {
503 max_restarts: DEFAULT_MAX_RESTARTS,
504 backoff: DEFAULT_BACKOFF,
505 max_backoff: DEFAULT_MAX_BACKOFF,
506 window: DEFAULT_RESTART_WINDOW,
507 }
508 }
509}
510
511#[derive(Debug, Clone, Copy, PartialEq, Eq)]
512struct CrashRestartSchedule {
513 restart_in_window: u32,
514 delay: Duration,
515}
516
517fn daemon_will_restart(
524 state: &mut SupervisorSnapshot,
525 policy: &RestartPolicy,
526 now: Instant,
527) -> bool {
528 state.enabled && state.crash_restarts_in_window(policy.window, now) < policy.max_restarts
529}
530
531const DEFAULT_HEALTH_CADENCE: Duration = Duration::from_secs(30);
532const DEFAULT_HEALTH_DEADLINE: Duration = Duration::from_secs(5);
533const DEFAULT_HEALTH_FAILURE_THRESHOLD: u32 = 3;
534const MAX_HEALTH_METRICS_BYTES: usize = 16 * 1024;
535
536#[derive(Debug, Clone, Copy, PartialEq, Eq)]
537pub enum HealthAction {
538 Report,
539 Restart,
540 Alert,
541}
542
543impl fmt::Display for HealthAction {
544 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
545 f.write_str(match self {
546 Self::Report => "report",
547 Self::Restart => "restart",
548 Self::Alert => "alert",
549 })
550 }
551}
552
553#[derive(Debug, Clone, Copy, PartialEq, Eq)]
554pub struct HealthConfig {
555 pub cadence: Duration,
556 pub deadline: Duration,
557 pub failure_threshold: u32,
558 pub on_degraded: HealthAction,
559 pub on_failing: HealthAction,
560 pub critical: bool,
561}
562
563impl Default for HealthConfig {
564 fn default() -> Self {
565 Self {
566 cadence: DEFAULT_HEALTH_CADENCE,
567 deadline: DEFAULT_HEALTH_DEADLINE,
568 failure_threshold: DEFAULT_HEALTH_FAILURE_THRESHOLD,
569 on_degraded: HealthAction::Report,
570 on_failing: HealthAction::Report,
571 critical: false,
572 }
573 }
574}
575
576#[derive(Debug, Clone, PartialEq)]
594pub struct ModuleHealthStatus {
595 pub status: SupervisorHealthStatus,
596 pub last_probe_ms: Option<u64>,
597 pub detail: Option<String>,
598 pub metrics: Option<Value>,
599 pub consecutive_failures: u32,
600 pub late_answer_count: u64,
603 pub last_late_answer_latency_ms: Option<u64>,
605 pub last_action: Option<String>,
606 pub last_action_ms: Option<u64>,
610}
611
612impl Default for ModuleHealthStatus {
613 fn default() -> Self {
614 Self {
615 status: SupervisorHealthStatus::Unknown,
616 last_probe_ms: None,
617 detail: None,
618 metrics: None,
619 consecutive_failures: 0,
620 late_answer_count: 0,
621 last_late_answer_latency_ms: None,
622 last_action: None,
623 last_action_ms: None,
624 }
625 }
626}
627
628#[derive(Debug, Clone, Copy, PartialEq, Eq)]
630pub enum ModuleState {
631 Starting,
632 Running,
633 Unresponsive,
634 Restarting,
635 Draining,
636 Stopped,
637 Failed,
638 Disabled,
639}
640
641impl fmt::Display for ModuleState {
642 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
643 f.write_str(match self {
644 Self::Starting => "starting",
645 Self::Running => "running",
646 Self::Unresponsive => "unresponsive",
647 Self::Restarting => "restarting",
648 Self::Draining => "draining",
649 Self::Stopped => "stopped",
650 Self::Failed => "failed",
651 Self::Disabled => "disabled",
652 })
653 }
654}
655
656#[derive(Debug, Clone, Copy, PartialEq, Eq)]
658pub enum ExitKind {
659 Clean,
660 Crash,
661 DeliberateSeverance,
662}
663
664impl From<ExitKind> for TerminalExitKind {
665 fn from(kind: ExitKind) -> Self {
666 match kind {
667 ExitKind::Clean => Self::Clean,
668 ExitKind::Crash => Self::Crash,
669 ExitKind::DeliberateSeverance => Self::DeliberateSeverance,
670 }
671 }
672}
673
674#[derive(Debug, Clone, Copy, PartialEq, Eq)]
677pub(crate) struct ProcessIdentity {
678 pub(crate) pid: u32,
679 pub(crate) start_time: u64,
680}
681
682#[derive(Debug, Clone, PartialEq, Eq)]
684pub struct ExitReport {
685 pub kind: ExitKind,
686 pub code: Option<i32>,
687 pub signal: Option<i32>,
688 pub at_ms: u64,
689}
690
691#[derive(Debug, Clone, PartialEq)]
694pub struct ModuleStatus {
695 pub module_id: String,
696 pub state: ModuleState,
697 pub enabled: bool,
698 pub process_alive: bool,
699 pub registration_active: bool,
700 pub protocol: ModuleProtocol,
703 pub live: bool,
714 pub restart_count: u32,
718 pub lifetime_restarts: u32,
722 pub spawn_generation: u64,
723 pub max_restarts: u32,
728 pub restart_window: Duration,
732 pub drain_timeout: Duration,
736 pub restart_backoff: Duration,
737 pub restart_max_backoff: Duration,
738 pub pid: Option<u32>,
739 pub spawned_at_ms: Option<u64>,
740 pub spawned_from: Option<PathBuf>,
741 pub process_start_time: Option<u64>,
742 pub last_exit: Option<ExitReport>,
743 pub health: ModuleHealthStatus,
744}
745
746#[derive(Debug, Clone, PartialEq)]
747struct SupervisorSnapshot {
748 state: ModuleState,
749 enabled: bool,
750 process_alive: bool,
751 crash_restarts: VecDeque<Instant>,
757 lifetime_restarts: u32,
758 spawn_generation: u64,
767 pid: Option<u32>,
768 reaped_pid: Option<u32>,
770 respawn_pending: bool,
772 coalesced_restart_pending: bool,
774 spawned_at_ms: Option<u64>,
775 spawned_from: Option<PathBuf>,
776 spawned_file_identity: Option<SpawnedFileIdentity>,
777 process_start_time: Option<u64>,
778 deliberate_severance: Option<ProcessIdentity>,
779 last_exit: Option<ExitReport>,
780 drain_disposition_detail: Option<String>,
782 health: ModuleHealthStatus,
783 in_alternate_slot: bool,
788 draining_to_replace: bool,
795 configuration_updated_since_spawn: bool,
801}
802
803impl SupervisorSnapshot {
804 fn starting() -> Self {
805 Self::new(ModuleState::Starting, true)
806 }
807
808 fn disabled() -> Self {
809 Self::new(ModuleState::Disabled, false)
810 }
811
812 fn failed() -> Self {
813 Self::new(ModuleState::Failed, true)
814 }
815
816 fn crash_restarts_in_window(&mut self, window: Duration, now: Instant) -> u32 {
820 while let Some(oldest) = self.crash_restarts.front() {
821 if now.duration_since(*oldest) > window {
822 self.crash_restarts.pop_front();
823 } else {
824 break;
825 }
826 }
827 u32::try_from(self.crash_restarts.len()).unwrap_or(u32::MAX)
828 }
829
830 fn record_crash_restart(&mut self, policy: &RestartPolicy, now: Instant) {
836 self.crash_restarts.push_back(now);
837 while self.crash_restarts.len() > policy.max_restarts as usize {
838 self.crash_restarts.pop_front();
839 }
840 self.lifetime_restarts += 1;
841 }
842
843 fn next_crash_restart(
847 &mut self,
848 policy: &RestartPolicy,
849 now: Instant,
850 ) -> Option<CrashRestartSchedule> {
851 let restart_in_window = self.crash_restarts_in_window(policy.window, now);
852 if restart_in_window >= policy.max_restarts {
853 return None;
854 }
855 self.record_crash_restart(policy, now);
856 Some(CrashRestartSchedule {
857 restart_in_window,
858 delay: policy.delay_for_restart(restart_in_window),
859 })
860 }
861
862 fn clear_crash_restarts(&mut self) {
867 self.crash_restarts.clear();
868 }
869
870 fn new(state: ModuleState, enabled: bool) -> Self {
871 Self {
872 state,
873 enabled,
874 process_alive: false,
875 crash_restarts: VecDeque::new(),
876 lifetime_restarts: 0,
877 spawn_generation: 0,
878 pid: None,
879 reaped_pid: None,
880 respawn_pending: false,
881 coalesced_restart_pending: false,
882 spawned_at_ms: None,
883 spawned_from: None,
884 spawned_file_identity: None,
885 process_start_time: None,
886 deliberate_severance: None,
887 last_exit: None,
888 drain_disposition_detail: None,
889 health: ModuleHealthStatus::default(),
890 in_alternate_slot: false,
891 draining_to_replace: false,
892 configuration_updated_since_spawn: false,
893 }
894 }
895}
896
897type SharedSnapshot = Arc<Mutex<SupervisorSnapshot>>;
898
899type SpawnSubscriberKey = (ConnectionId, u64);
900
901#[derive(Debug)]
902struct SpawnSubscriber {
903 version: u8,
904 frames: mpsc::Sender<Frame>,
905 lagged: Option<oneshot::Sender<SpawnCursor>>,
909}
910
911#[derive(Debug)]
912struct SpawnEventState {
913 daemon_incarnation: String,
914 seq: u64,
915 capacity: usize,
916 live: HashMap<String, LiveSpawn>,
917 generations: HashMap<String, u64>,
918 events: VecDeque<SpawnEvent>,
919 subscribers: HashMap<SpawnSubscriberKey, SpawnSubscriber>,
920}
921
922impl Default for SpawnEventState {
923 fn default() -> Self {
924 Self {
925 daemon_incarnation: "unconfigured".to_string(),
926 seq: 0,
927 capacity: SPAWN_EVENT_RING_CAPACITY,
928 live: HashMap::new(),
929 generations: HashMap::new(),
930 events: VecDeque::new(),
931 subscribers: HashMap::new(),
932 }
933 }
934}
935
936#[derive(Debug, Clone, Default)]
937struct SpawnEventFeed(Arc<Mutex<SpawnEventState>>);
938
939#[derive(Debug, Clone, PartialEq, Eq)]
940pub(crate) enum SpawnSubscribeRefusal {
941 ForeignIncarnation { current: String },
942 TooOld { oldest: SpawnCursor },
943 Frame(String),
944}
945
946impl SpawnEventFeed {
947 fn configure_incarnation(&self, daemon_incarnation: String) {
948 let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
949 state.daemon_incarnation = daemon_incarnation;
950 state.seq = 0;
951 state.live.clear();
952 state.generations.clear();
953 state.events.clear();
954 state.subscribers.clear();
955 }
956
957 fn cursor(state: &SpawnEventState) -> SpawnCursor {
958 SpawnCursor {
959 daemon_incarnation: state.daemon_incarnation.clone(),
960 seq: state.seq,
961 }
962 }
963
964 fn snapshot(&self) -> SpawnSnapshot {
965 let state = self.0.lock().unwrap_or_else(|p| p.into_inner());
966 let mut live = state.live.values().cloned().collect::<Vec<_>>();
967 live.sort_by(|left, right| left.module_id.cmp(&right.module_id));
968 SpawnSnapshot {
969 cursor: Self::cursor(&state),
970 ring_bound: state.capacity as u64,
971 live,
972 }
973 }
974
975 fn emit_spawned(&self, module_id: &str, pid: u32, spawned_at_ms: u64) -> u64 {
976 let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
977 let generation = state
978 .generations
979 .get(module_id)
980 .copied()
981 .unwrap_or(0)
982 .checked_add(1)
983 .expect("spawn generation exhausted");
984 state.generations.insert(module_id.to_string(), generation);
985 let live = LiveSpawn {
986 module_id: module_id.to_string(),
987 spawn_generation: generation,
988 pid,
989 spawned_at_ms,
990 };
991 state.live.insert(module_id.to_string(), live);
992 Self::emit_locked(
993 &mut state,
994 SpawnEventKind::Spawned,
995 module_id.to_string(),
996 generation,
997 pid,
998 None,
999 None,
1000 );
1001 generation
1002 }
1003
1004 fn emit_exited(&self, module_id: &str, exit_code: Option<i32>, exit_signal: Option<i32>) {
1005 let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1006 let Some(live) = state.live.remove(module_id) else {
1007 warn!(
1008 module_id,
1009 "terminal record had no live spawn event identity"
1010 );
1011 return;
1012 };
1013 Self::emit_locked(
1014 &mut state,
1015 SpawnEventKind::Exited,
1016 module_id.to_string(),
1017 live.spawn_generation,
1018 live.pid,
1019 exit_code,
1020 exit_signal,
1021 );
1022 }
1023
1024 fn emit_superseded_exited(
1031 &self,
1032 module_id: &str,
1033 spawn_generation: u64,
1034 pid: u32,
1035 exit_code: Option<i32>,
1036 exit_signal: Option<i32>,
1037 ) {
1038 let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1039 if state
1040 .live
1041 .get(module_id)
1042 .is_some_and(|live| live.spawn_generation == spawn_generation)
1043 {
1044 state.live.remove(module_id);
1045 }
1046 Self::emit_locked(
1047 &mut state,
1048 SpawnEventKind::Exited,
1049 module_id.to_string(),
1050 spawn_generation,
1051 pid,
1052 exit_code,
1053 exit_signal,
1054 );
1055 }
1056
1057 #[allow(clippy::too_many_arguments)]
1058 fn emit_locked(
1059 state: &mut SpawnEventState,
1060 kind: SpawnEventKind,
1061 module_id: String,
1062 spawn_generation: u64,
1063 pid: u32,
1064 exit_code: Option<i32>,
1065 exit_signal: Option<i32>,
1066 ) {
1067 state.seq = state
1068 .seq
1069 .checked_add(1)
1070 .expect("spawn event sequence exhausted");
1071 let event = SpawnEvent {
1072 cursor: Self::cursor(state),
1073 kind,
1074 module_id,
1075 spawn_generation,
1076 pid,
1077 exit_code,
1078 exit_signal,
1079 };
1080 state.events.push_back(event.clone());
1081 while state.events.len() > state.capacity {
1082 state.events.pop_front();
1083 }
1084 let body = match serde_json::to_vec(&event) {
1085 Ok(body) => body,
1086 Err(error) => {
1087 error!(%error, "failed to serialize supervisor spawn event");
1088 return;
1089 }
1090 };
1091 state.subscribers.retain(|(connection_id, corr), subscriber| {
1092 let frame = Frame::build_with_version(
1093 subscriber.version,
1094 FrameType::StreamData,
1095 control_flags(),
1096 0,
1097 0,
1098 *corr,
1099 body.clone(),
1100 );
1101 match frame {
1102 Ok(frame) => {
1103 if subscriber.frames.try_send(frame).is_ok() {
1104 true
1105 } else {
1106 warn!(connection_id = connection_id.get(), corr, "dropping lagged supervisor spawn subscriber");
1107 if let Some(lagged) = subscriber.lagged.take() {
1108 let _ = lagged.send(event.cursor.clone());
1109 }
1110 false
1111 }
1112 }
1113 Err(error) => {
1114 warn!(connection_id = connection_id.get(), corr, %error, "dropping supervisor spawn subscriber after frame build failure");
1115 false
1116 }
1117 }
1118 });
1119 }
1120
1121 fn subscribe(
1122 &self,
1123 connection_id: ConnectionId,
1124 corr: u64,
1125 version: u8,
1126 since: Option<SpawnCursor>,
1127 sink: FrameSink,
1128 ) -> Result<(), SpawnSubscribeRefusal> {
1129 let (frames, mut receiver) = mpsc::channel(SPAWN_SUBSCRIBER_BUFFER);
1130 let (lagged, mut lagged_rx) = oneshot::channel::<SpawnCursor>();
1131 {
1132 let mut state = self.0.lock().unwrap_or_else(|p| p.into_inner());
1133 let replay = if let Some(since) = since {
1134 if since.daemon_incarnation != state.daemon_incarnation {
1135 return Err(SpawnSubscribeRefusal::ForeignIncarnation {
1136 current: state.daemon_incarnation.clone(),
1137 });
1138 }
1139 if let Some(oldest) = state.events.front().map(|event| event.cursor.clone()) {
1140 if since.seq < oldest.seq.saturating_sub(1) {
1141 return Err(SpawnSubscribeRefusal::TooOld { oldest });
1142 }
1143 }
1144 state
1145 .events
1146 .iter()
1147 .filter(|event| event.cursor.seq > since.seq)
1148 .cloned()
1149 .collect::<Vec<_>>()
1150 } else {
1151 Vec::new()
1152 };
1153 for event in replay {
1154 let body = serde_json::to_vec(&event)
1155 .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1156 let frame = Frame::build_with_version(
1157 version,
1158 FrameType::StreamData,
1159 control_flags(),
1160 0,
1161 0,
1162 corr,
1163 body,
1164 )
1165 .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1166 frames
1167 .try_send(frame)
1168 .map_err(|error| SpawnSubscribeRefusal::Frame(error.to_string()))?;
1169 }
1170 state.subscribers.insert(
1171 (connection_id, corr),
1172 SpawnSubscriber {
1173 version,
1174 frames,
1175 lagged: Some(lagged),
1176 },
1177 );
1178 }
1179 tokio::spawn(async move {
1190 while let Some(frame) = receiver.recv().await {
1191 if sink.send(frame).await.is_err() {
1192 return;
1193 }
1194 }
1195 let Ok(first_undelivered) = lagged_rx.try_recv() else {
1196 return;
1197 };
1198 match spawn_subscriber_lagged_frame(version, corr, first_undelivered) {
1199 Ok(frame) => {
1200 let _ = sink.send(frame).await;
1201 }
1202 Err(error) => {
1203 error!(%error, corr, "failed to build lagged spawn subscriber terminal frame");
1204 }
1205 }
1206 });
1207 Ok(())
1208 }
1209
1210 fn cancel(&self, connection_id: ConnectionId, corr: u64) -> bool {
1211 let Some(subscriber) = self
1212 .0
1213 .lock()
1214 .unwrap_or_else(|p| p.into_inner())
1215 .subscribers
1216 .remove(&(connection_id, corr))
1217 else {
1218 return false;
1219 };
1220 if let Ok(frame) = Frame::build_with_version(
1221 subscriber.version,
1222 FrameType::StreamEnd,
1223 control_flags(),
1224 0,
1225 0,
1226 corr,
1227 Vec::new(),
1228 ) {
1229 tokio::spawn(async move {
1230 let _ = subscriber.frames.send(frame).await;
1231 });
1232 }
1233 true
1234 }
1235
1236 fn remove_connection(&self, connection_id: ConnectionId) {
1237 self.0
1238 .lock()
1239 .unwrap_or_else(|p| p.into_inner())
1240 .subscribers
1241 .retain(|(subscriber_connection, _), _| *subscriber_connection != connection_id);
1242 }
1243
1244 #[cfg(any(test, feature = "test-support"))]
1245 fn set_capacity(&self, capacity: usize) {
1246 self.0.lock().unwrap_or_else(|p| p.into_inner()).capacity = capacity;
1247 }
1248
1249 #[cfg(any(test, feature = "test-support"))]
1250 fn subscriber_count(&self) -> usize {
1251 self.0
1252 .lock()
1253 .unwrap_or_else(|p| p.into_inner())
1254 .subscribers
1255 .len()
1256 }
1257}
1258
1259fn spawn_subscriber_lagged_frame(
1262 version: u8,
1263 corr: u64,
1264 first_undelivered: SpawnCursor,
1265) -> Result<Frame, String> {
1266 let body = serde_json::to_vec(&subc_protocol::ErrorBody {
1267 code: SPAWN_SUBSCRIBER_LAGGED_CODE.to_string(),
1268 message: "spawn subscriber fell behind and was dropped; resubscribe from the last cursor received"
1269 .to_string(),
1270 detail: Some(serde_json::json!({
1271 "first_undelivered_cursor": first_undelivered
1272 })),
1273 })
1274 .map_err(|error| error.to_string())?;
1275 Frame::build_with_version(version, FrameType::Error, control_flags(), 0, 0, corr, body)
1276 .map_err(|error| error.to_string())
1277}
1278
1279pub trait ModuleProcessLiveness: Send + Sync {
1280 fn process_live(&self, module_id: &str) -> Option<bool>;
1281
1282 fn process_replacing(&self, _module_id: &str) -> bool {
1288 false
1289 }
1290}
1291
1292#[derive(Debug, Clone, Default)]
1294pub struct SupervisorProcessLiveness {
1295 snapshots: Arc<Mutex<HashMap<String, SharedSnapshot>>>,
1296}
1297
1298impl SupervisorProcessLiveness {
1299 pub fn new() -> Self {
1300 Self::default()
1301 }
1302
1303 fn track(&self, module_id: String, snapshot: SharedSnapshot) {
1304 let mut snapshots = self
1305 .snapshots
1306 .lock()
1307 .unwrap_or_else(|poisoned| poisoned.into_inner());
1308 snapshots.insert(module_id, snapshot);
1309 }
1310
1311 fn untrack_if_current(&self, module_id: &str, snapshot: &SharedSnapshot) {
1312 let mut snapshots = self
1313 .snapshots
1314 .lock()
1315 .unwrap_or_else(|poisoned| poisoned.into_inner());
1316 let is_current = snapshots
1317 .get(module_id)
1318 .map(|tracked| Arc::ptr_eq(tracked, snapshot))
1319 .unwrap_or(false);
1320 if is_current {
1321 snapshots.remove(module_id);
1322 }
1323 }
1324}
1325
1326impl ModuleProcessLiveness for SupervisorProcessLiveness {
1327 fn process_live(&self, module_id: &str) -> Option<bool> {
1328 let snapshot = {
1329 let snapshots = self
1330 .snapshots
1331 .lock()
1332 .unwrap_or_else(|poisoned| poisoned.into_inner());
1333 snapshots.get(module_id).cloned()
1334 }?;
1335 let snapshot = snapshot
1336 .lock()
1337 .unwrap_or_else(|poisoned| poisoned.into_inner());
1338 Some(snapshot.state == ModuleState::Running && snapshot.process_alive)
1339 }
1340
1341 fn process_replacing(&self, module_id: &str) -> bool {
1342 let Some(snapshot) = self
1343 .snapshots
1344 .lock()
1345 .unwrap_or_else(|poisoned| poisoned.into_inner())
1346 .get(module_id)
1347 .cloned()
1348 else {
1349 return false;
1350 };
1351 let snapshot = snapshot
1352 .lock()
1353 .unwrap_or_else(|poisoned| poisoned.into_inner());
1354 snapshot.enabled
1355 && match snapshot.state {
1356 ModuleState::Restarting => true,
1357 ModuleState::Draining => snapshot.draining_to_replace,
1358 ModuleState::Starting
1359 | ModuleState::Running
1360 | ModuleState::Unresponsive
1361 | ModuleState::Stopped
1362 | ModuleState::Failed
1363 | ModuleState::Disabled => false,
1364 }
1365 }
1366}
1367
1368#[cfg(test)]
1369#[derive(Debug, Default)]
1370struct ReloadExitRecordGate {
1371 reached: tokio::sync::Notify,
1372 resume: tokio::sync::Notify,
1373}
1374
1375#[derive(Debug, Clone, Copy)]
1376enum RespawnKind {
1377 Spawn,
1378 Reload,
1379}
1380
1381#[derive(Debug, Clone, Copy)]
1382struct PendingRespawn {
1383 deadline: Instant,
1384 kind: RespawnKind,
1385}
1386
1387type ReloadReply = oneshot::Sender<Result<(), SuperviseError>>;
1388
1389#[derive(Debug, Clone)]
1390struct SupervisorRuntimeConfig {
1391 scheduled_respawn: Arc<Mutex<Option<PendingRespawn>>>,
1393 deferred_reload_reply: Arc<Mutex<Option<ReloadReply>>>,
1395 restart_policy: RestartPolicy,
1396 drain_timeout: Duration,
1399 effective_drain_timeout: Arc<Mutex<Duration>>,
1402 default_drain_timeout: Duration,
1405 health: HealthConfig,
1406 connection_file_path: Option<PathBuf>,
1407 capture_logs_dir: Option<PathBuf>,
1408 forwarding: Option<Arc<ForwardingTable>>,
1409 supervisor_handle: Option<SupervisorHandle>,
1412 stderr_ring: Arc<Mutex<StderrRing>>,
1419 terminal_ring: Arc<Mutex<TerminalRing>>,
1420 spawn_events: SpawnEventFeed,
1421 child_roster: ChildRoster,
1422 #[cfg(target_os = "linux")]
1423 cgroup_placement: Option<subc_cgroup::Placement>,
1424 #[cfg(test)]
1425 test_seed_stale_facts_before_enable_spawn: bool,
1426 #[cfg(test)]
1427 test_reload_exit_record_gate: Option<Arc<ReloadExitRecordGate>>,
1428}
1429
1430#[derive(Debug, Clone, PartialEq, Eq)]
1431struct SupervisedConfiguration {
1432 spec: ModuleSpec,
1433 health: HealthConfig,
1434}
1435
1436#[derive(Debug, Clone, Default)]
1442pub struct SupervisorHandle {
1443 modules: Arc<Mutex<HashMap<String, SupervisedModule>>>,
1444 spawn_events: SpawnEventFeed,
1445 reserved_nonces: Arc<Mutex<HashMap<String, Option<String>>>>,
1456 removal_tombstones: Arc<Mutex<HashMap<String, u64>>>,
1462 spawn_nonces: Arc<Mutex<HashMap<String, String>>>,
1466 reserved_prefix_owners: Arc<Mutex<HashMap<String, String>>>,
1474 swaps: Arc<Mutex<HashMap<String, OpenSwap>>>,
1480 promotion_observer: PromotionObserverSlot,
1482 operation_lock: Arc<AsyncMutex<()>>,
1486}
1487
1488pub(crate) trait SwapPromotionObserver: Send + Sync {
1497 fn swap_promoted(&self, registration: &crate::registry::ModuleRegistration);
1498}
1499
1500#[derive(Clone, Default)]
1504struct PromotionObserverSlot(Arc<Mutex<Option<std::sync::Weak<dyn SwapPromotionObserver>>>>);
1505
1506impl fmt::Debug for PromotionObserverSlot {
1507 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
1508 f.write_str("PromotionObserverSlot")
1509 }
1510}
1511
1512#[derive(Debug, Clone)]
1514struct OpenSwap {
1515 candidate_nonce: String,
1518 incumbent_nonce: Option<String>,
1523 candidate_admitted: bool,
1527}
1528
1529#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1532pub(crate) enum SwapHelloAdmission {
1533 NotSwapping,
1536 Candidate,
1538 Refused,
1541}
1542
1543#[derive(Debug, Clone, PartialEq, Eq)]
1544pub(crate) enum ReservedHelloRejection {
1545 Exact {
1546 module_id: String,
1547 },
1548 Prefix {
1549 prefix: String,
1550 owner_module_id: String,
1551 },
1552}
1553
1554impl SupervisorHandle {
1555 pub fn new() -> Self {
1556 Self::default()
1557 }
1558
1559 pub(crate) fn spawn_snapshot(&self) -> SpawnSnapshot {
1560 self.spawn_events.snapshot()
1561 }
1562
1563 pub(crate) fn subscribe_spawns(
1564 &self,
1565 connection_id: ConnectionId,
1566 corr: u64,
1567 version: u8,
1568 since: Option<SpawnCursor>,
1569 sink: FrameSink,
1570 ) -> Result<(), SpawnSubscribeRefusal> {
1571 self.spawn_events
1572 .subscribe(connection_id, corr, version, since, sink)
1573 }
1574
1575 pub(crate) fn cancel_spawn_subscription(&self, connection_id: ConnectionId, corr: u64) -> bool {
1576 self.spawn_events.cancel(connection_id, corr)
1577 }
1578
1579 pub(crate) fn remove_spawn_subscribers(&self, connection_id: ConnectionId) {
1580 self.spawn_events.remove_connection(connection_id);
1581 }
1582
1583 #[cfg(any(test, feature = "test-support"))]
1584 pub fn set_spawn_event_capacity_for_test(&self, capacity: usize) {
1585 assert!(capacity > 0, "spawn event capacity must be non-zero");
1586 self.spawn_events.set_capacity(capacity);
1587 }
1588
1589 #[cfg(any(test, feature = "test-support"))]
1590 pub fn spawn_subscriber_count_for_test(&self) -> usize {
1591 self.spawn_events.subscriber_count()
1592 }
1593
1594 pub fn set_spawn_nonce(&self, module_id: &str, nonce: String) {
1597 self.spawn_nonces
1598 .lock()
1599 .unwrap_or_else(|poisoned| poisoned.into_inner())
1600 .insert(module_id.to_string(), nonce);
1601 }
1602
1603 pub fn set_reserved_nonce(&self, module_id: &str, nonce: String) {
1606 self.reserved_nonces
1607 .lock()
1608 .unwrap_or_else(|poisoned| poisoned.into_inner())
1609 .insert(module_id.to_string(), Some(nonce));
1610 }
1611
1612 pub fn set_reserved_prefixes(&self, owner_module_id: &str, prefixes: &[String]) {
1614 let mut owners = self
1615 .reserved_prefix_owners
1616 .lock()
1617 .unwrap_or_else(|poisoned| poisoned.into_inner());
1618 owners.retain(|_, owner| owner != owner_module_id);
1619 for prefix in prefixes {
1620 owners.insert(prefix.clone(), owner_module_id.to_string());
1621 }
1622 }
1623
1624 #[cfg(test)]
1626 pub(crate) fn spawn_nonce(&self, module_id: &str) -> Option<String> {
1627 self.spawn_nonces
1628 .lock()
1629 .unwrap_or_else(|poisoned| poisoned.into_inner())
1630 .get(module_id)
1631 .cloned()
1632 }
1633
1634 fn apply_identity_configuration(&self, spec: &ModuleSpec) {
1635 self.set_reserved_prefixes(&spec.module_id, &spec.reserved_prefixes);
1636 let spawn_nonce = self
1637 .spawn_nonces
1638 .lock()
1639 .unwrap_or_else(|poisoned| poisoned.into_inner())
1640 .get(&spec.module_id)
1641 .cloned();
1642 let mut reserved_nonces = self
1643 .reserved_nonces
1644 .lock()
1645 .unwrap_or_else(|poisoned| poisoned.into_inner());
1646 if spec.reserved {
1647 reserved_nonces.insert(spec.module_id.clone(), spawn_nonce);
1652 }
1653 drop(reserved_nonces);
1654 self.removal_tombstones
1658 .lock()
1659 .unwrap_or_else(|poisoned| poisoned.into_inner())
1660 .remove(&spec.module_id);
1661 }
1662
1663 pub fn reserved_hello_authorized(&self, module_id: &str, presented: Option<&str>) -> bool {
1668 self.reserved_hello_rejection(module_id, presented)
1669 .is_none()
1670 }
1671
1672 pub(crate) fn reserved_hello_rejection(
1673 &self,
1674 module_id: &str,
1675 presented: Option<&str>,
1676 ) -> Option<ReservedHelloRejection> {
1677 let nonces = self
1678 .reserved_nonces
1679 .lock()
1680 .unwrap_or_else(|poisoned| poisoned.into_inner());
1681 if let Some(expected) = nonces.get(module_id) {
1682 let authorized = match expected {
1686 Some(expected) => {
1687 presented.is_some_and(|p| constant_time_eq(expected.as_bytes(), p.as_bytes()))
1688 }
1689 None => false,
1690 };
1691 if authorized {
1692 return None;
1693 }
1694 return Some(ReservedHelloRejection::Exact {
1695 module_id: module_id.to_string(),
1696 });
1697 }
1698 drop(nonces);
1699
1700 let matched_prefix = self
1701 .reserved_prefix_owners
1702 .lock()
1703 .unwrap_or_else(|poisoned| poisoned.into_inner())
1704 .iter()
1705 .filter(|(prefix, _)| module_id.starts_with(prefix.as_str()))
1706 .max_by_key(|(prefix, _)| prefix.len())
1707 .map(|(prefix, owner)| (prefix.clone(), owner.clone()));
1708 let (prefix, owner_module_id) = matched_prefix?;
1709
1710 let authorized = presented.is_some_and(|presented| {
1711 self.spawn_nonces
1712 .lock()
1713 .unwrap_or_else(|poisoned| poisoned.into_inner())
1714 .get(&owner_module_id)
1715 .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()))
1716 || self.swap_nonce_matches(&owner_module_id, presented)
1719 });
1720 if authorized {
1721 None
1722 } else {
1723 Some(ReservedHelloRejection::Prefix {
1724 prefix,
1725 owner_module_id,
1726 })
1727 }
1728 }
1729
1730 pub fn spawned_consumer_authorized(&self, module_id: &str, presented: &str) -> bool {
1735 if presented.is_empty() {
1736 return false;
1737 }
1738 let nonces = self
1739 .spawn_nonces
1740 .lock()
1741 .unwrap_or_else(|poisoned| poisoned.into_inner());
1742 let current = nonces
1743 .get(module_id)
1744 .is_some_and(|expected| constant_time_eq(expected.as_bytes(), presented.as_bytes()));
1745 drop(nonces);
1746 current || self.swap_nonce_matches(module_id, presented)
1751 }
1752
1753 fn swap_nonce_matches(&self, module_id: &str, presented: &str) -> bool {
1755 let swaps = self
1756 .swaps
1757 .lock()
1758 .unwrap_or_else(|poisoned| poisoned.into_inner());
1759 swaps.get(module_id).is_some_and(|swap| {
1760 constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes())
1761 || swap.incumbent_nonce.as_deref().is_some_and(|incumbent| {
1762 constant_time_eq(incumbent.as_bytes(), presented.as_bytes())
1763 })
1764 })
1765 }
1766
1767 pub(crate) fn open_swap(&self, module_id: &str, candidate_nonce: String) {
1770 let incumbent_nonce = self
1771 .spawn_nonces
1772 .lock()
1773 .unwrap_or_else(|poisoned| poisoned.into_inner())
1774 .get(module_id)
1775 .cloned();
1776 self.swaps
1777 .lock()
1778 .unwrap_or_else(|poisoned| poisoned.into_inner())
1779 .insert(
1780 module_id.to_string(),
1781 OpenSwap {
1782 candidate_nonce,
1783 incumbent_nonce,
1784 candidate_admitted: false,
1785 },
1786 );
1787 }
1788
1789 pub(crate) fn close_swap(&self, module_id: &str) {
1792 self.swaps
1793 .lock()
1794 .unwrap_or_else(|poisoned| poisoned.into_inner())
1795 .remove(module_id);
1796 }
1797
1798 pub(crate) fn set_swap_promotion_observer(
1801 &self,
1802 observer: std::sync::Weak<dyn SwapPromotionObserver>,
1803 ) {
1804 *self
1805 .promotion_observer
1806 .0
1807 .lock()
1808 .unwrap_or_else(|poisoned| poisoned.into_inner()) = Some(observer);
1809 }
1810
1811 fn notify_swap_promoted(&self, registration: &crate::registry::ModuleRegistration) {
1814 let observer = self
1815 .promotion_observer
1816 .0
1817 .lock()
1818 .unwrap_or_else(|poisoned| poisoned.into_inner())
1819 .as_ref()
1820 .and_then(std::sync::Weak::upgrade);
1821 if let Some(observer) = observer {
1822 observer.swap_promoted(registration);
1823 }
1824 }
1825
1826 pub(crate) fn swap_open(&self, module_id: &str) -> bool {
1828 self.swaps
1829 .lock()
1830 .unwrap_or_else(|poisoned| poisoned.into_inner())
1831 .contains_key(module_id)
1832 }
1833
1834 fn promote_swap_nonce(&self, module_id: &str, reserved: bool) {
1839 let candidate_nonce = self
1840 .swaps
1841 .lock()
1842 .unwrap_or_else(|poisoned| poisoned.into_inner())
1843 .get(module_id)
1844 .map(|swap| swap.candidate_nonce.clone());
1845 let Some(nonce) = candidate_nonce else {
1846 return;
1847 };
1848 self.set_spawn_nonce(module_id, nonce.clone());
1849 if reserved {
1850 self.set_reserved_nonce(module_id, nonce);
1851 }
1852 }
1853
1854 pub(crate) fn swap_hello_admission(
1869 &self,
1870 module_id: &str,
1871 presented: Option<&str>,
1872 ) -> SwapHelloAdmission {
1873 let swaps = self
1874 .swaps
1875 .lock()
1876 .unwrap_or_else(|poisoned| poisoned.into_inner());
1877 let Some(swap) = swaps.get(module_id) else {
1878 return SwapHelloAdmission::NotSwapping;
1879 };
1880 let Some(presented) = presented else {
1881 return SwapHelloAdmission::Refused;
1882 };
1883 if constant_time_eq(swap.candidate_nonce.as_bytes(), presented.as_bytes()) {
1884 return if swap.candidate_admitted {
1885 SwapHelloAdmission::Refused
1886 } else {
1887 SwapHelloAdmission::Candidate
1888 };
1889 }
1890 if swap
1891 .incumbent_nonce
1892 .as_deref()
1893 .is_some_and(|incumbent| constant_time_eq(incumbent.as_bytes(), presented.as_bytes()))
1894 {
1895 return SwapHelloAdmission::NotSwapping;
1896 }
1897 SwapHelloAdmission::Refused
1898 }
1899
1900 pub(crate) fn mark_swap_candidate_admitted(&self, module_id: &str) {
1903 if let Some(swap) = self
1904 .swaps
1905 .lock()
1906 .unwrap_or_else(|poisoned| poisoned.into_inner())
1907 .get_mut(module_id)
1908 {
1909 swap.candidate_admitted = true;
1910 }
1911 }
1912
1913 pub fn spawn_launch_nonce_for(&self, module_id: &str) -> Option<String> {
1915 self.spawn_nonces
1916 .lock()
1917 .unwrap_or_else(|poisoned| poisoned.into_inner())
1918 .get(module_id)
1919 .cloned()
1920 }
1921
1922 pub fn reserved_launch_nonce_for(&self, module_id: &str) -> Option<String> {
1924 self.reserved_nonces
1925 .lock()
1926 .unwrap_or_else(|poisoned| poisoned.into_inner())
1927 .get(module_id)
1928 .cloned()
1929 .flatten()
1930 }
1931
1932 pub fn insert(&self, module: SupervisedModule) -> Option<SupervisedModule> {
1933 let mut modules = self
1934 .modules
1935 .lock()
1936 .unwrap_or_else(|poisoned| poisoned.into_inner());
1937 modules.insert(module.module_id().to_string(), module)
1938 }
1939
1940 pub fn get(&self, module_id: &str) -> Option<SupervisedModule> {
1941 let modules = self
1942 .modules
1943 .lock()
1944 .unwrap_or_else(|poisoned| poisoned.into_inner());
1945 modules.get(module_id).cloned()
1946 }
1947
1948 pub(crate) fn record_late_health_answer(
1949 &self,
1950 module_id: &str,
1951 latency_ms: u64,
1952 ) -> Result<bool, SuperviseError> {
1953 let Some(module) = self.get(module_id) else {
1954 return Ok(false);
1955 };
1956 update_snapshot(&module.inner.snapshot, Some(module_id), |state| {
1957 state.health.late_answer_count = state.health.late_answer_count.saturating_add(1);
1958 state.health.last_late_answer_latency_ms = Some(latency_ms);
1959 state.health.consecutive_failures = 0;
1967 })?;
1968 Ok(true)
1969 }
1970
1971 pub fn record_deliberate_severance(&self, module_id: &str) -> Result<bool, SuperviseError> {
1977 let Some(module) = self.get(module_id) else {
1978 return Ok(false);
1979 };
1980 let status = module.status()?;
1981 let Some((pid, start_time)) = status.pid.zip(status.process_start_time) else {
1982 return Ok(false);
1983 };
1984 module.record_deliberate_severance(ProcessIdentity { pid, start_time })
1985 }
1986
1987 pub fn list(&self) -> Vec<SupervisedModule> {
1988 let modules = self
1989 .modules
1990 .lock()
1991 .unwrap_or_else(|poisoned| poisoned.into_inner());
1992 let mut modules = modules.values().cloned().collect::<Vec<_>>();
1993 modules.sort_by(|left, right| left.module_id().cmp(right.module_id()));
1994 modules
1995 }
1996
1997 pub(crate) fn retire(&self, module_id: &str) -> Option<SupervisedModule> {
1998 self.spawn_nonces
1999 .lock()
2000 .unwrap_or_else(|poisoned| poisoned.into_inner())
2001 .remove(module_id);
2002 self.close_swap(module_id);
2003 let mut reserved_nonces = self
2004 .reserved_nonces
2005 .lock()
2006 .unwrap_or_else(|poisoned| poisoned.into_inner());
2007 if reserved_nonces.contains_key(module_id) {
2008 reserved_nonces.insert(module_id.to_string(), None);
2011 }
2012 drop(reserved_nonces);
2013 self.reserved_prefix_owners
2014 .lock()
2015 .unwrap_or_else(|poisoned| poisoned.into_inner())
2016 .retain(|_, owner| owner != module_id);
2017 self.modules
2018 .lock()
2019 .unwrap_or_else(|poisoned| poisoned.into_inner())
2020 .remove(module_id)
2021 }
2022
2023 pub(crate) fn record_rescan_removal(&self, module_id: &str) {
2026 self.removal_tombstones
2027 .lock()
2028 .unwrap_or_else(|poisoned| poisoned.into_inner())
2029 .insert(module_id.to_string(), unix_ms_now());
2030 }
2031
2032 pub(crate) fn removal_tombstone_age_ms(&self, module_id: &str) -> Option<u64> {
2034 self.removal_tombstones
2035 .lock()
2036 .unwrap_or_else(|poisoned| poisoned.into_inner())
2037 .get(module_id)
2038 .copied()
2039 .map(|removed_at_ms| unix_ms_now().saturating_sub(removed_at_ms))
2040 }
2041
2042 pub(crate) fn release_retained_reserved_gate(&self, module_id: &str) -> bool {
2047 if self.get(module_id).is_some() {
2048 return false;
2049 }
2050 let mut reserved_nonces = self
2051 .reserved_nonces
2052 .lock()
2053 .unwrap_or_else(|poisoned| poisoned.into_inner());
2054 if !matches!(reserved_nonces.get(module_id), Some(None)) {
2055 return false;
2056 }
2057 reserved_nonces.remove(module_id);
2058 true
2059 }
2060
2061 pub(crate) fn operation_lock(&self) -> Arc<AsyncMutex<()>> {
2062 Arc::clone(&self.operation_lock)
2063 }
2064}
2065
2066#[derive(Debug, Clone)]
2068pub struct Supervisor {
2069 registry: Arc<Registry>,
2070 restart_policy: RestartPolicy,
2071 drain_timeout: Duration,
2072 connection_file_path: Option<PathBuf>,
2073 capture_logs_dir: Option<PathBuf>,
2074 forwarding: Option<Arc<ForwardingTable>>,
2075 process_liveness: Arc<SupervisorProcessLiveness>,
2076 supervisor_handle: Option<SupervisorHandle>,
2077 health: HealthConfig,
2078 daemon_start_clock: crate::clock::StartClock,
2079 terminal_journal: Option<Arc<crate::terminal_journal::TerminalJournal>>,
2080 spawn_events: SpawnEventFeed,
2081 provenance_probe: ExecutableIdentityProbe,
2082 child_roster: ChildRoster,
2085 #[cfg(target_os = "linux")]
2086 cgroup_placement: Option<subc_cgroup::Placement>,
2087}
2088
2089impl Supervisor {
2090 #[cfg(unix)]
2101 pub(crate) fn begin_daemon_shutdown(&self) {
2102 self.child_roster.close();
2103 if let Some(journal) = &self.terminal_journal {
2104 journal.stamp_shutdown();
2105 }
2106 }
2107
2108 #[cfg(unix)]
2112 pub(crate) async fn drain_for_daemon_shutdown(&self) -> Result<(), SuperviseError> {
2113 const NOTICE_BUDGET: Duration = Duration::from_millis(500);
2114 const DRAIN_BUDGET: Duration = Duration::from_secs(2);
2115 let Some(forwarding) = &self.forwarding else {
2116 return Ok(());
2117 };
2118 let module_ids = forwarding
2119 .begin_daemon_drain()
2120 .map_err(SuperviseError::Forwarding)?;
2121 let deadline_ms =
2122 unix_ms_now().saturating_add((NOTICE_BUDGET + DRAIN_BUDGET).as_millis() as u64);
2123 let mut notices = tokio::task::JoinSet::new();
2124 let mut drains = Vec::new();
2125 for module_id in module_ids {
2126 let Some(target) = forwarding
2127 .begin_module_drain(&module_id, RouteCloseReason::Restart)
2128 .map_err(SuperviseError::Forwarding)?
2129 else {
2130 continue;
2131 };
2132 let routes = forwarding
2133 .endpoint_routes(target.endpoint)
2134 .map_err(SuperviseError::Forwarding)?;
2135 let command = serde_json::to_vec(&ModuleControlCommand::Draining {
2141 reason: RouteCloseReason::Restart,
2142 deadline_ms,
2143 })
2144 .expect("module draining serializes");
2145 let mut recipients = vec![(target.sink.clone(), target.negotiated_ver, command)];
2146 let mut clients: Vec<(crate::forwarding::GoodbyeTarget, Vec<u16>)> = Vec::new();
2147 for route in routes {
2148 let client = route.goodbye_target;
2149 if let Some((_, channels)) = clients
2150 .iter_mut()
2151 .find(|(existing, _)| existing.connection_id == client.connection_id)
2152 {
2153 channels.push(client.channel);
2154 } else {
2155 let channel = client.channel;
2156 clients.push((client, vec![channel]));
2157 }
2158 }
2159 for (client, mut channels) in clients {
2160 channels.sort_unstable();
2161 channels.dedup();
2162 let closing = serde_json::to_vec(&ClientControlPush::RouteClosing {
2163 module_id: module_id.clone(),
2164 channels,
2165 reason: RouteCloseReason::Restart,
2166 })
2167 .expect("route closing serializes");
2168 recipients.push((client.sink, client.negotiated_ver, closing));
2169 }
2170 for (sink, version, body) in recipients {
2171 notices.spawn(async move {
2172 let frame = Frame::build_with_version(
2173 version,
2174 FrameType::Push,
2175 control_flags(),
2176 0,
2177 0,
2178 0,
2179 body,
2180 )
2181 .expect("bounded lifecycle notice frame builds");
2182 sink.send_flushed(frame).await
2183 });
2184 }
2185 let gauges = declared_busy_gauges(&self.registry, &module_id)?;
2186 drains.push((module_id, target.endpoint, gauges));
2187 }
2188 let notice_deadline = Instant::now() + NOTICE_BUDGET;
2191 while let Ok(Some(result)) = timeout_at(notice_deadline, notices.join_next()).await {
2192 if !matches!(result, Ok(Ok(()))) {
2193 warn!(?result, "daemon shutdown notice delivery failed");
2194 }
2195 }
2196 notices.abort_all();
2197 let deadline = Instant::now() + DRAIN_BUDGET;
2198 let mut waits = tokio::task::JoinSet::new();
2199 for (module_id, endpoint, gauges) in drains {
2200 let forwarding = Arc::clone(forwarding);
2201 let mut runtime = self.runtime_config();
2202 runtime.health.cadence = Duration::from_millis(100);
2203 waits.spawn(async move {
2204 wait_for_forwarding_quiescence(
2205 &forwarding,
2206 &module_id,
2207 &runtime,
2208 endpoint,
2209 deadline,
2210 &gauges,
2211 DrainScope::Active,
2212 )
2213 .await
2214 });
2215 }
2216 while let Ok(Some(result)) = timeout_at(deadline, waits.join_next()).await {
2217 if !matches!(result, Ok(Ok(true))) {
2218 warn!(?result, "daemon shutdown drain did not reach quiescence");
2219 }
2220 }
2221 Ok(())
2222 }
2223
2224 #[cfg(unix)]
2236 pub(crate) async fn end_children_for_daemon_shutdown(
2237 &self,
2238 already_escalated: bool,
2239 escalate: impl std::future::Future<Output = ()>,
2240 ) {
2241 tokio::pin!(escalate);
2242 let mut escalated = already_escalated;
2243 if let Some(forwarding) = &self.forwarding {
2244 let reason = CloseReason::new(
2245 "daemon_shutdown",
2246 "the daemon is exiting after its shutdown notice and drain",
2247 );
2248 if escalated {
2249 send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, false).await;
2252 } else {
2253 tokio::select! {
2254 biased;
2255 _ = escalate.as_mut() => {
2256 info!("second SIGTERM: abandoning module GOODBYE delivery");
2257 escalated = true;
2258 }
2259 _ = send_module_goodbyes_for_daemon_shutdown(forwarding, &reason, true) => {}
2260 }
2261 }
2262 let closed = forwarding.close_all_connections(&reason);
2263 debug!(closed, "closed established connections for daemon shutdown");
2264 }
2265 let escalated_here = escalated && !already_escalated;
2269 let remaining_escalate = async move {
2270 if escalated_here {
2271 std::future::pending::<()>().await;
2272 } else {
2273 escalate.await;
2274 }
2275 };
2276 crate::child_roster::end_children_for_daemon_shutdown(
2277 &self.child_roster,
2278 escalated,
2279 remaining_escalate,
2280 )
2281 .await;
2282 }
2283
2284 pub fn new(registry: Arc<Registry>, restart_policy: RestartPolicy) -> Self {
2285 Self {
2286 registry,
2287 restart_policy,
2288 drain_timeout: DEFAULT_DRAIN_TIMEOUT,
2289 connection_file_path: None,
2290 capture_logs_dir: None,
2291 forwarding: None,
2292 process_liveness: Arc::new(SupervisorProcessLiveness::default()),
2293 supervisor_handle: None,
2294 health: HealthConfig::default(),
2295 daemon_start_clock: crate::clock::StartClock::capture(),
2296 terminal_journal: None,
2297 spawn_events: SpawnEventFeed::default(),
2298 provenance_probe: ExecutableIdentityProbe::default(),
2299 child_roster: ChildRoster::default(),
2300 #[cfg(target_os = "linux")]
2301 cgroup_placement: None,
2302 }
2303 }
2304
2305 pub fn with_drain_timeout(mut self, drain_timeout: Duration) -> Self {
2306 self.drain_timeout = drain_timeout;
2307 self
2308 }
2309
2310 pub fn with_process_liveness(
2311 mut self,
2312 process_liveness: Arc<SupervisorProcessLiveness>,
2313 ) -> Self {
2314 self.process_liveness = process_liveness;
2315 self
2316 }
2317
2318 pub fn with_connection_file_path(mut self, connection_file_path: impl Into<PathBuf>) -> Self {
2319 self.connection_file_path = Some(connection_file_path.into());
2320 self
2321 }
2322
2323 pub fn with_capture_logs_dir(mut self, logs_dir: impl Into<PathBuf>) -> Self {
2325 self.capture_logs_dir = Some(logs_dir.into());
2326 self
2327 }
2328
2329 pub fn with_daemon_incarnation(self, daemon_incarnation: String) -> Self {
2332 self.spawn_events.configure_incarnation(daemon_incarnation);
2336 self
2337 }
2338
2339 pub fn with_terminal_journal(self, path: PathBuf, daemon_incarnation: String) -> Self {
2342 let mut this = self.with_daemon_incarnation(daemon_incarnation.clone());
2343 this.terminal_journal = Some(Arc::new(crate::terminal_journal::TerminalJournal::open(
2344 path,
2345 daemon_incarnation,
2346 )));
2347 this
2348 }
2349
2350 pub fn with_forwarding(mut self, forwarding: Arc<ForwardingTable>) -> Self {
2351 self.forwarding = Some(forwarding);
2352 self
2353 }
2354
2355 pub fn with_handle(mut self, supervisor_handle: SupervisorHandle) -> Self {
2356 self.spawn_events = supervisor_handle.spawn_events.clone();
2357 self.supervisor_handle = Some(supervisor_handle);
2358 self
2359 }
2360
2361 pub fn with_health_config(mut self, health: HealthConfig) -> Self {
2362 self.health = health;
2363 self
2364 }
2365
2366 pub fn with_live_children_record(self, path: impl Into<PathBuf>) -> Self {
2370 self.child_roster.record_to(path.into());
2371 self
2372 }
2373
2374 #[cfg(target_os = "linux")]
2375 pub fn with_cgroup_placement(
2376 mut self,
2377 cgroup_placement: Option<subc_cgroup::Placement>,
2378 ) -> Self {
2379 self.cgroup_placement = cgroup_placement;
2380 self
2381 }
2382
2383 pub fn spawn(&self, spec: ModuleSpec) -> Result<SupervisedModule, SuperviseError> {
2389 validate_spec(&spec)?;
2390
2391 let runtime = self.runtime_config();
2392 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2393 let child = spawn_child(
2394 &spec,
2395 runtime.connection_file_path.as_deref(),
2396 self.supervisor_handle.as_ref(),
2397 &runtime.stderr_ring,
2398 runtime.capture_logs_dir.as_deref(),
2399 &runtime.child_roster,
2400 #[cfg(target_os = "linux")]
2401 runtime.cgroup_placement.as_ref(),
2402 )?;
2403 set_running(&snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
2404 self.process_liveness
2405 .track(spec.module_id.clone(), Arc::clone(&snapshot));
2406
2407 Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2408 }
2409
2410 pub fn supervise_configured(
2416 &self,
2417 spec: ModuleSpec,
2418 enabled: bool,
2419 ) -> Result<SupervisedModule, SuperviseError> {
2420 validate_spec(&spec)?;
2421
2422 let runtime = self.runtime_config();
2423 if !enabled {
2424 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
2425 return Ok(self.supervised_module(spec, runtime, snapshot, None));
2426 }
2427
2428 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2429 match spawn_child(
2430 &spec,
2431 runtime.connection_file_path.as_deref(),
2432 self.supervisor_handle.as_ref(),
2433 &runtime.stderr_ring,
2434 runtime.capture_logs_dir.as_deref(),
2435 &runtime.child_roster,
2436 #[cfg(target_os = "linux")]
2437 runtime.cgroup_placement.as_ref(),
2438 ) {
2439 Ok(child) => {
2440 set_running(&snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
2441 self.process_liveness
2442 .track(spec.module_id.clone(), Arc::clone(&snapshot));
2443 Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2444 }
2445 Err(err) => {
2446 error!(
2447 module_id = %spec.module_id,
2448 program = %spec.program.display(),
2449 error = %err,
2450 "configured module failed to spawn; marking failed and continuing"
2451 );
2452 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
2453 Ok(self.supervised_module(spec, runtime, snapshot, None))
2454 }
2455 }
2456 }
2457
2458 pub fn supervise_configured_with_health(
2464 &self,
2465 spec: ModuleSpec,
2466 enabled: bool,
2467 health: HealthConfig,
2468 drain_timeout_ms: Option<u64>,
2469 restart_policy: RestartPolicy,
2470 ) -> Result<SupervisedModule, SuperviseError> {
2471 validate_spec(&spec)?;
2472
2473 let mut runtime = self.runtime_config();
2474 runtime.health = health;
2475 runtime.restart_policy = restart_policy;
2476 if let Some(ms) = drain_timeout_ms {
2477 runtime.drain_timeout = Duration::from_millis(ms);
2478 *runtime
2479 .effective_drain_timeout
2480 .lock()
2481 .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
2482 }
2483 if !enabled {
2484 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
2485 return Ok(self.supervised_module(spec, runtime, snapshot, None));
2486 }
2487
2488 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
2489 match spawn_child(
2490 &spec,
2491 runtime.connection_file_path.as_deref(),
2492 self.supervisor_handle.as_ref(),
2493 &runtime.stderr_ring,
2494 runtime.capture_logs_dir.as_deref(),
2495 &runtime.child_roster,
2496 #[cfg(target_os = "linux")]
2497 runtime.cgroup_placement.as_ref(),
2498 ) {
2499 Ok(child) => {
2500 set_running(&snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
2501 self.process_liveness
2502 .track(spec.module_id.clone(), Arc::clone(&snapshot));
2503 Ok(self.supervised_module(spec, runtime, snapshot, Some(child)))
2504 }
2505 Err(err) => {
2506 if health.critical {
2507 error!(
2508 module_id = %spec.module_id,
2509 program = %spec.program.display(),
2510 error = %err,
2511 "critical configured module failed to spawn; marking failed and alerting"
2512 );
2513 } else {
2514 error!(
2515 module_id = %spec.module_id,
2516 program = %spec.program.display(),
2517 error = %err,
2518 "configured module failed to spawn; marking failed and continuing"
2519 );
2520 }
2521 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::failed()));
2522 Ok(self.supervised_module(spec, runtime, snapshot, None))
2523 }
2524 }
2525 }
2526
2527 fn runtime_config(&self) -> SupervisorRuntimeConfig {
2528 let effective_drain_timeout = Arc::new(Mutex::new(self.drain_timeout));
2529 SupervisorRuntimeConfig {
2530 scheduled_respawn: Arc::default(),
2531 deferred_reload_reply: Arc::default(),
2532 restart_policy: self.restart_policy,
2533 drain_timeout: self.drain_timeout,
2534 child_roster: self
2537 .child_roster
2538 .for_module(Arc::clone(&effective_drain_timeout)),
2539 effective_drain_timeout,
2540 default_drain_timeout: self.drain_timeout,
2541 health: self.health,
2542 connection_file_path: self.connection_file_path.clone(),
2543 capture_logs_dir: self.capture_logs_dir.clone(),
2544 forwarding: self.forwarding.clone(),
2545 supervisor_handle: self.supervisor_handle.clone(),
2546 stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
2547 terminal_ring: Arc::new(Mutex::new(
2548 TerminalRing::new(
2549 TerminalRingConfig::default(),
2550 self.daemon_start_clock.started_at_ms(),
2551 )
2552 .with_start_clock(self.daemon_start_clock)
2553 .with_journal(self.terminal_journal.clone())
2554 .with_daemon_shutdown(self.child_roster.shutdown_flag()),
2555 )),
2556 spawn_events: self.spawn_events.clone(),
2557 #[cfg(target_os = "linux")]
2558 cgroup_placement: self.cgroup_placement.clone(),
2559 #[cfg(test)]
2560 test_seed_stale_facts_before_enable_spawn: false,
2561 #[cfg(test)]
2562 test_reload_exit_record_gate: None,
2563 }
2564 }
2565
2566 fn supervised_module(
2567 &self,
2568 spec: ModuleSpec,
2569 runtime: SupervisorRuntimeConfig,
2570 snapshot: SharedSnapshot,
2571 child: Option<SupervisedChild>,
2572 ) -> SupervisedModule {
2573 let configuration = Arc::new(Mutex::new(SupervisedConfiguration {
2574 spec: spec.clone(),
2575 health: runtime.health,
2576 }));
2577 let stderr_ring = Arc::clone(&runtime.stderr_ring);
2578 let terminal_ring = Arc::clone(&runtime.terminal_ring);
2579 let restart_policy = runtime.restart_policy;
2583 let effective_drain_timeout = Arc::clone(&runtime.effective_drain_timeout);
2584 let (tx, rx) = mpsc::channel(4);
2585 let monitor = tokio::spawn(supervise_loop(
2586 spec.clone(),
2587 runtime,
2588 Arc::clone(&self.registry),
2589 Arc::clone(&self.process_liveness),
2590 Arc::clone(&snapshot),
2591 child,
2592 rx,
2593 ));
2594
2595 let module_id = spec.module_id.clone();
2596 let module = SupervisedModule {
2597 inner: Arc::new(SupervisedModuleInner {
2598 module_id: module_id.clone(),
2599 registry: Arc::clone(&self.registry),
2600 snapshot,
2601 configuration,
2602 stderr_ring,
2603 terminal_ring,
2604 commands: tx,
2605 monitor: Mutex::new(Some(monitor)),
2606 restart_policy,
2607 effective_drain_timeout,
2608 provenance_probe: self.provenance_probe.clone(),
2609 }),
2610 };
2611 if let Some(supervisor_handle) = &self.supervisor_handle {
2612 supervisor_handle.apply_identity_configuration(&spec);
2613 supervisor_handle.insert(module.clone());
2614 }
2615 module
2616 }
2617}
2618
2619impl Default for Supervisor {
2620 fn default() -> Self {
2621 Self::new(Arc::new(Registry::default()), RestartPolicy::default())
2622 }
2623}
2624
2625#[derive(Clone)]
2627pub struct SupervisedModule {
2628 inner: Arc<SupervisedModuleInner>,
2629}
2630
2631struct SupervisedModuleInner {
2632 module_id: String,
2633 registry: Arc<Registry>,
2634 snapshot: SharedSnapshot,
2635 configuration: Arc<Mutex<SupervisedConfiguration>>,
2636 stderr_ring: Arc<Mutex<StderrRing>>,
2637 terminal_ring: Arc<Mutex<TerminalRing>>,
2638 commands: mpsc::Sender<SupervisorCommand>,
2639 monitor: Mutex<Option<JoinHandle<()>>>,
2640 restart_policy: RestartPolicy,
2644 effective_drain_timeout: Arc<Mutex<Duration>>,
2645 provenance_probe: ExecutableIdentityProbe,
2646}
2647
2648impl fmt::Debug for SupervisedModule {
2649 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
2650 f.debug_struct("SupervisedModule")
2651 .field("module_id", &self.inner.module_id)
2652 .field("status", &self.status())
2653 .finish_non_exhaustive()
2654 }
2655}
2656
2657impl SupervisedModule {
2658 pub fn module_id(&self) -> &str {
2659 &self.inner.module_id
2660 }
2661
2662 #[cfg(test)]
2666 pub(crate) fn record_health_probe_failure_for_test(
2667 &self,
2668 detail: &str,
2669 ) -> Result<(), SuperviseError> {
2670 update_snapshot(&self.inner.snapshot, Some(&self.inner.module_id), |state| {
2671 state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
2672 state.health.detail = Some(detail.to_string());
2673 })
2674 }
2675
2676 pub fn state(&self) -> Result<ModuleState, SuperviseError> {
2677 Ok(lock_snapshot(&self.inner.snapshot)?.state)
2678 }
2679
2680 pub fn stderr_tail(
2687 &self,
2688 max_lines: Option<usize>,
2689 max_bytes: Option<usize>,
2690 ) -> StderrTailSnapshot {
2691 self.inner
2692 .stderr_ring
2693 .lock()
2694 .unwrap_or_else(|poisoned| poisoned.into_inner())
2695 .snapshot(max_lines, max_bytes)
2696 }
2697
2698 pub fn terminal_history(&self) -> TerminalHistorySnapshot {
2703 self.inner
2704 .terminal_ring
2705 .lock()
2706 .unwrap_or_else(|poisoned| poisoned.into_inner())
2707 .snapshot()
2708 }
2709
2710 pub fn durable_terminal_history(&self) -> subc_control::TerminalHistory {
2715 durable_terminal_history_of(&self.inner.terminal_ring, &self.inner.module_id)
2716 }
2717
2718 pub(crate) async fn read_durable_terminal_history(
2723 &self,
2724 ) -> Result<subc_control::TerminalHistory, tokio::task::JoinError> {
2725 let terminal_ring = Arc::clone(&self.inner.terminal_ring);
2726 let module_id = self.inner.module_id.clone();
2727 tokio::task::spawn_blocking(move || durable_terminal_history_of(&terminal_ring, &module_id))
2728 .await
2729 }
2730
2731 pub fn status(&self) -> Result<ModuleStatus, SuperviseError> {
2732 self.status_with_snapshot_lock(&self.inner.snapshot, None)
2733 }
2734
2735 pub(crate) fn record_deliberate_severance(
2736 &self,
2737 identity: ProcessIdentity,
2738 ) -> Result<bool, SuperviseError> {
2739 let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
2740 if snapshot.pid != Some(identity.pid)
2741 || snapshot.process_start_time != Some(identity.start_time)
2742 {
2743 return Ok(false);
2744 }
2745 snapshot.deliberate_severance = Some(identity);
2746 Ok(true)
2747 }
2748
2749 pub(crate) fn status_for_control(
2754 &self,
2755 caller: &'static str,
2756 ) -> Result<ModuleStatus, SuperviseError> {
2757 self.status_with_snapshot_lock(&self.inner.snapshot, Some(caller))
2758 }
2759
2760 fn status_with_snapshot_lock(
2761 &self,
2762 snapshot: &SharedSnapshot,
2763 caller: Option<&'static str>,
2764 ) -> Result<ModuleStatus, SuperviseError> {
2765 let mut guard = match caller {
2766 Some(caller) => lock_snapshot_for_control(snapshot, &self.inner.module_id, caller)?,
2767 None => lock_snapshot(snapshot)?,
2768 };
2769 let restart_count =
2772 guard.crash_restarts_in_window(self.inner.restart_policy.window, Instant::now());
2773 let snapshot = guard.clone();
2774 drop(guard);
2775 let drain_timeout = *self.inner.effective_drain_timeout.lock().map_err(|_| {
2776 SuperviseError::StatePoisoned {
2777 module_id: Some(self.inner.module_id.clone()),
2778 }
2779 })?;
2780 let registration_active = self
2781 .inner
2782 .registry
2783 .get_module(&self.inner.module_id)
2784 .map_err(SuperviseError::Registry)?
2785 .is_some();
2786 let protocol = self.declared_protocol()?;
2787 let running_process =
2788 snapshot.enabled && snapshot.state == ModuleState::Running && snapshot.process_alive;
2789 let live = match protocol {
2795 ModuleProtocol::Subc => running_process && registration_active,
2796 ModuleProtocol::None => running_process,
2797 };
2798
2799 Ok(ModuleStatus {
2800 module_id: self.inner.module_id.clone(),
2801 state: snapshot.state,
2802 enabled: snapshot.enabled,
2803 process_alive: snapshot.process_alive,
2804 registration_active,
2805 protocol,
2806 live,
2807 restart_count,
2808 lifetime_restarts: snapshot.lifetime_restarts,
2809 spawn_generation: snapshot.spawn_generation,
2810 max_restarts: self.inner.restart_policy.max_restarts,
2811 restart_window: self.inner.restart_policy.window,
2812 drain_timeout,
2813 restart_backoff: self.inner.restart_policy.backoff,
2814 restart_max_backoff: self.inner.restart_policy.max_backoff,
2815 pid: snapshot.pid,
2816 spawned_at_ms: snapshot.spawned_at_ms,
2817 spawned_from: snapshot.spawned_from,
2818 process_start_time: snapshot.process_start_time,
2819 last_exit: snapshot.last_exit,
2820 health: snapshot.health,
2821 })
2822 }
2823
2824 #[cfg(test)]
2825 pub(crate) fn hold_snapshot_for_test(
2826 &self,
2827 acquired: std::sync::mpsc::Sender<()>,
2828 hold: Duration,
2829 ) -> std::thread::JoinHandle<()> {
2830 let snapshot = Arc::clone(&self.inner.snapshot);
2831 std::thread::spawn(move || {
2832 let _guard = snapshot.lock().expect("test snapshot lock is not poisoned");
2833 acquired
2834 .send(())
2835 .expect("test receiver waits for snapshot lock");
2836 std::thread::sleep(hold);
2837 })
2838 }
2839
2840 pub(crate) async fn running_image_agreement(&self) -> subc_control::RunningImageAgreement {
2841 let snapshot = match lock_snapshot(&self.inner.snapshot) {
2842 Ok(snapshot) => snapshot.clone(),
2843 Err(_) => {
2844 return subc_control::RunningImageAgreement::Unavailable {
2845 reason: subc_control::RunningImageUnavailableReason::NotRunning,
2846 };
2847 }
2848 };
2849 self.inner
2850 .provenance_probe
2851 .observe(
2852 snapshot.pid,
2853 snapshot.spawned_from.as_deref(),
2854 snapshot.spawned_file_identity,
2855 snapshot.process_start_time,
2856 )
2857 .await
2858 }
2859
2860 pub(crate) fn child_resource_usage(&self) -> subc_control::ChildResourceUsage {
2863 let (pid, start_time) = match lock_snapshot(&self.inner.snapshot) {
2864 Ok(snapshot) => (snapshot.pid, snapshot.process_start_time),
2865 Err(_) => {
2866 return subc_control::ChildResourceUsage::Unavailable {
2867 reason: subc_control::ChildResourceUnavailableReason::Unreadable,
2868 }
2869 }
2870 };
2871 crate::child_resources::read(pid, start_time)
2872 }
2873
2874 pub(crate) fn will_recover_after_connection_loss(&self) -> Result<bool, SuperviseError> {
2875 let mut snapshot = lock_snapshot(&self.inner.snapshot)?;
2876 Ok(match snapshot.state {
2877 ModuleState::Restarting => true,
2878 ModuleState::Failed | ModuleState::Disabled => false,
2879 _ => daemon_will_restart(&mut snapshot, &self.inner.restart_policy, Instant::now()),
2880 })
2881 }
2882
2883 #[cfg(test)]
2884 pub(crate) fn is_warming(&self) -> Result<bool, SuperviseError> {
2885 self.is_warming_with_snapshot_lock(None)
2886 }
2887
2888 pub(crate) fn is_warming_for_control(
2889 &self,
2890 caller: &'static str,
2891 ) -> Result<bool, SuperviseError> {
2892 self.is_warming_with_snapshot_lock(Some(caller))
2893 }
2894
2895 fn is_warming_with_snapshot_lock(
2896 &self,
2897 caller: Option<&'static str>,
2898 ) -> Result<bool, SuperviseError> {
2899 let snapshot = match caller {
2900 Some(caller) => {
2901 lock_snapshot_for_control(&self.inner.snapshot, &self.inner.module_id, caller)?
2902 }
2903 None => lock_snapshot(&self.inner.snapshot)?,
2904 }
2905 .clone();
2906 Ok(matches!(
2907 snapshot.state,
2908 ModuleState::Starting | ModuleState::Running | ModuleState::Restarting
2909 ))
2910 }
2911
2912 pub async fn drain(&self) -> Result<(), SuperviseError> {
2914 self.stop().await
2915 }
2916
2917 pub(crate) async fn retire(&self) -> Result<(), SuperviseError> {
2918 match self.state()? {
2919 ModuleState::Stopped | ModuleState::Failed => return Ok(()),
2920 ModuleState::Starting
2921 | ModuleState::Running
2922 | ModuleState::Unresponsive
2923 | ModuleState::Restarting
2924 | ModuleState::Draining
2925 | ModuleState::Disabled => {}
2926 }
2927
2928 let (reply_tx, reply_rx) = oneshot::channel();
2929 self.inner
2930 .commands
2931 .send(SupervisorCommand::Retire { reply: reply_tx })
2932 .await
2933 .map_err(|_| SuperviseError::CommandClosed {
2934 module_id: self.inner.module_id.clone(),
2935 })?;
2936 reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
2937 module_id: self.inner.module_id.clone(),
2938 })?
2939 }
2940
2941 pub async fn stop(&self) -> Result<(), SuperviseError> {
2942 match self.state()? {
2943 ModuleState::Stopped | ModuleState::Failed => return Ok(()),
2944 ModuleState::Starting
2945 | ModuleState::Running
2946 | ModuleState::Unresponsive
2947 | ModuleState::Restarting
2948 | ModuleState::Draining
2949 | ModuleState::Disabled => {}
2950 }
2951
2952 let (reply_tx, reply_rx) = oneshot::channel();
2953 self.inner
2954 .commands
2955 .send(SupervisorCommand::Drain { reply: reply_tx })
2956 .await
2957 .map_err(|_| SuperviseError::CommandClosed {
2958 module_id: self.inner.module_id.clone(),
2959 })?;
2960 reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
2961 module_id: self.inner.module_id.clone(),
2962 })?
2963 }
2964
2965 pub async fn restart(&self, drain_timeout_ms: Option<u64>) -> Result<(), SuperviseError> {
2966 let received_at_generation = lock_snapshot(&self.inner.snapshot)?.spawn_generation;
2967 let (reply_tx, reply_rx) = oneshot::channel();
2968 self.inner
2969 .commands
2970 .send(SupervisorCommand::Restart {
2971 drain_timeout_ms,
2972 received_at_generation,
2973 queued_at: Instant::now(),
2974 reply: reply_tx,
2975 })
2976 .await
2977 .map_err(|_| SuperviseError::CommandClosed {
2978 module_id: self.inner.module_id.clone(),
2979 })?;
2980 reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
2981 module_id: self.inner.module_id.clone(),
2982 })?
2983 }
2984
2985 pub async fn swap(&self, ready_timeout: Option<Duration>) -> Result<(), SuperviseError> {
2990 let (reply_tx, reply_rx) = oneshot::channel();
2991 self.inner
2992 .commands
2993 .send(SupervisorCommand::Swap {
2994 ready_timeout,
2995 reply: reply_tx,
2996 })
2997 .await
2998 .map_err(|_| SuperviseError::CommandClosed {
2999 module_id: self.inner.module_id.clone(),
3000 })?;
3001 reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3002 module_id: self.inner.module_id.clone(),
3003 })?
3004 }
3005
3006 pub async fn reload(&self) -> Result<(), SuperviseError> {
3007 let (reply_tx, reply_rx) = oneshot::channel();
3008 self.inner
3009 .commands
3010 .send(SupervisorCommand::Reload { reply: reply_tx })
3011 .await
3012 .map_err(|_| SuperviseError::CommandClosed {
3013 module_id: self.inner.module_id.clone(),
3014 })?;
3015 reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3016 module_id: self.inner.module_id.clone(),
3017 })?
3018 }
3019
3020 pub async fn set_enabled(&self, enabled: bool) -> Result<bool, SuperviseError> {
3021 let (reply_tx, reply_rx) = oneshot::channel();
3022 self.inner
3023 .commands
3024 .send(SupervisorCommand::SetEnabled {
3025 enabled,
3026 reply: reply_tx,
3027 })
3028 .await
3029 .map_err(|_| SuperviseError::CommandClosed {
3030 module_id: self.inner.module_id.clone(),
3031 })?;
3032 reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3033 module_id: self.inner.module_id.clone(),
3034 })?
3035 }
3036
3037 pub(crate) fn declared_protocol(&self) -> Result<ModuleProtocol, SuperviseError> {
3042 Ok(self
3043 .inner
3044 .configuration
3045 .lock()
3046 .map_err(|_| SuperviseError::StatePoisoned {
3047 module_id: Some(self.inner.module_id.clone()),
3048 })?
3049 .spec
3050 .protocol)
3051 }
3052
3053 pub(crate) fn configuration(&self) -> Result<(ModuleSpec, HealthConfig), SuperviseError> {
3054 let configuration =
3055 self.inner
3056 .configuration
3057 .lock()
3058 .map_err(|_| SuperviseError::StatePoisoned {
3059 module_id: Some(self.inner.module_id.clone()),
3060 })?;
3061 Ok((configuration.spec.clone(), configuration.health))
3062 }
3063
3064 #[cfg(any(test, feature = "test-support"))]
3068 pub async fn update_spec_for_test(&self, spec: ModuleSpec) -> Result<(), SuperviseError> {
3069 let (_, health) = self.configuration()?;
3070 let drain_timeout_ms = u64::try_from(
3071 self.inner
3072 .effective_drain_timeout
3073 .lock()
3074 .unwrap_or_else(|poisoned| poisoned.into_inner())
3075 .as_millis(),
3076 )
3077 .ok();
3078 self.update_configuration(spec, health, drain_timeout_ms)
3079 .await
3080 }
3081
3082 pub(crate) async fn update_configuration(
3083 &self,
3084 spec: ModuleSpec,
3085 health: HealthConfig,
3086 drain_timeout_ms: Option<u64>,
3087 ) -> Result<(), SuperviseError> {
3088 if spec.module_id != self.inner.module_id {
3089 return Err(SuperviseError::InvalidSpec {
3090 reason: "a supervised module's module_id cannot be changed".to_string(),
3091 });
3092 }
3093 validate_spec(&spec)?;
3094 let (reply_tx, reply_rx) = oneshot::channel();
3095 self.inner
3096 .commands
3097 .send(SupervisorCommand::UpdateConfiguration {
3098 spec: spec.clone(),
3099 health,
3100 drain_timeout_ms,
3101 reply: reply_tx,
3102 })
3103 .await
3104 .map_err(|_| SuperviseError::CommandClosed {
3105 module_id: self.inner.module_id.clone(),
3106 })?;
3107 reply_rx.await.map_err(|_| SuperviseError::CommandClosed {
3108 module_id: self.inner.module_id.clone(),
3109 })?;
3110 let mut configuration =
3111 self.inner
3112 .configuration
3113 .lock()
3114 .map_err(|_| SuperviseError::StatePoisoned {
3115 module_id: Some(self.inner.module_id.clone()),
3116 })?;
3117 configuration.spec = spec;
3118 configuration.health = health;
3119 Ok(())
3120 }
3121}
3122
3123impl Drop for SupervisedModuleInner {
3124 fn drop(&mut self) {
3125 let Ok(mut monitor) = self.monitor.lock() else {
3126 return;
3127 };
3128 if let Some(monitor) = monitor.as_ref().filter(|monitor| !monitor.is_finished()) {
3129 let _ = update_snapshot(&self.snapshot, Some(&self.module_id), |state| {
3130 state.state = ModuleState::Stopped;
3131 clear_current_process_facts(state);
3132 });
3133 monitor.abort();
3134 }
3135 let _ = monitor.take();
3136 }
3137}
3138
3139#[derive(Debug)]
3140enum SupervisorCommand {
3141 Drain {
3142 reply: oneshot::Sender<Result<(), SuperviseError>>,
3143 },
3144 Retire {
3145 reply: oneshot::Sender<Result<(), SuperviseError>>,
3146 },
3147 Restart {
3148 drain_timeout_ms: Option<u64>,
3153 received_at_generation: u64,
3157 queued_at: Instant,
3160 reply: oneshot::Sender<Result<(), SuperviseError>>,
3161 },
3162 Reload {
3163 reply: oneshot::Sender<Result<(), SuperviseError>>,
3164 },
3165 SetEnabled {
3166 enabled: bool,
3167 reply: oneshot::Sender<Result<bool, SuperviseError>>,
3168 },
3169 UpdateConfiguration {
3170 spec: ModuleSpec,
3171 health: HealthConfig,
3172 drain_timeout_ms: Option<u64>,
3175 reply: oneshot::Sender<()>,
3176 },
3177 Swap {
3178 ready_timeout: Option<Duration>,
3181 reply: oneshot::Sender<Result<(), SuperviseError>>,
3183 },
3184}
3185
3186#[derive(Debug)]
3187pub enum SuperviseError {
3188 InvalidSpec {
3189 reason: String,
3190 },
3191 Spawn {
3192 program: PathBuf,
3193 source: io::Error,
3194 cgroup_path: Option<PathBuf>,
3195 },
3196 Cgroup {
3197 module_id: String,
3198 source: io::Error,
3199 },
3200 LaunchNonce {
3203 reason: String,
3204 },
3205 Wait {
3206 module_id: String,
3207 source: io::Error,
3208 },
3209 Kill {
3210 module_id: String,
3211 source: io::Error,
3212 },
3213 Forwarding(ForwardingError),
3214 Registry(RegistryError),
3215 ReloadUnavailable {
3216 module_id: String,
3217 reason: String,
3218 },
3219 Disabled {
3224 module_id: String,
3225 },
3226 ReloadFailed {
3227 module_id: String,
3228 reason: String,
3229 },
3230 RegistrationStillActive {
3231 module_id: String,
3232 waited: Duration,
3233 },
3234 StatePoisoned {
3235 module_id: Option<String>,
3236 },
3237 CommandClosed {
3238 module_id: String,
3239 },
3240 SwapInProgress {
3244 module_id: String,
3245 },
3246 SwapRefused {
3248 module_id: String,
3249 reason: SwapRefusal,
3250 },
3251 SwapFailed {
3255 module_id: String,
3256 arm: SwapFailureArm,
3257 detail: String,
3258 candidate_exit: Option<ExitReport>,
3261 },
3262}
3263
3264#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3266pub enum SwapRefusal {
3267 OverlapExclusive,
3269 NotRegistered,
3272 ProtocolNone,
3275 NotConfigured,
3278 AlreadySwapping,
3280}
3281
3282impl SwapRefusal {
3283 pub fn as_str(self) -> &'static str {
3284 match self {
3285 Self::OverlapExclusive => "overlap_exclusive",
3286 Self::NotRegistered => "not_registered",
3287 Self::ProtocolNone => "protocol_none",
3288 Self::NotConfigured => "not_configured",
3289 Self::AlreadySwapping => "already_swapping",
3290 }
3291 }
3292}
3293
3294#[derive(Debug, Clone, Copy, PartialEq, Eq)]
3297pub enum SwapFailureArm {
3298 SpawnFailed,
3300 NeverRegistered,
3302 NeverReady,
3304 CandidateExited,
3306 CandidateUnhealthy,
3308 Interrupted,
3312 CutoverLost,
3317}
3318
3319impl SwapFailureArm {
3320 pub fn as_str(self) -> &'static str {
3321 match self {
3322 Self::SpawnFailed => "spawn_failed",
3323 Self::NeverRegistered => "never_registered",
3324 Self::NeverReady => "never_ready",
3325 Self::CandidateExited => "candidate_exited",
3326 Self::CandidateUnhealthy => "candidate_unhealthy",
3327 Self::Interrupted => "interrupted",
3328 Self::CutoverLost => "cutover_lost",
3329 }
3330 }
3331}
3332
3333impl fmt::Display for SuperviseError {
3334 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3335 match self {
3336 Self::InvalidSpec { reason } => write!(f, "invalid module spec: {reason}"),
3337 Self::Spawn {
3338 program,
3339 source,
3340 cgroup_path: Some(cgroup_path),
3341 } => write!(
3342 f,
3343 "failed to place module in cgroup '{}' while spawning '{}': {source}",
3344 cgroup_path.display(),
3345 program.display()
3346 ),
3347 Self::Spawn {
3348 program,
3349 source,
3350 cgroup_path: None,
3351 } => write!(
3352 f,
3353 "failed to spawn module '{}': {source}",
3354 program.display()
3355 ),
3356 Self::Cgroup { module_id, source } => {
3357 write!(
3358 f,
3359 "failed to prepare cgroup for module '{module_id}': {source}"
3360 )
3361 }
3362 Self::LaunchNonce { reason } => {
3363 write!(
3364 f,
3365 "failed to generate reserved-module launch nonce: {reason}"
3366 )
3367 }
3368 Self::Wait { module_id, source } => {
3369 write!(f, "failed to wait for module '{module_id}': {source}")
3370 }
3371 Self::Kill { module_id, source } => {
3372 write!(f, "failed to kill module '{module_id}': {source}")
3373 }
3374 Self::Forwarding(err) => write!(f, "forwarding error: {err}"),
3375 Self::Registry(err) => write!(f, "registry error: {err}"),
3376 Self::ReloadUnavailable { module_id, reason } => {
3377 write!(f, "reload unavailable for module '{module_id}': {reason}")
3378 }
3379 Self::Disabled { module_id } => {
3380 write!(
3381 f,
3382 "module '{module_id}' is disabled; enable it before restart or reload"
3383 )
3384 }
3385 Self::ReloadFailed { module_id, reason } => {
3386 write!(f, "reload failed for module '{module_id}': {reason}")
3387 }
3388 Self::RegistrationStillActive { module_id, waited } => write!(
3389 f,
3390 "module '{module_id}' registration remained active after waiting {waited:?}"
3391 ),
3392 Self::StatePoisoned { module_id } => match module_id {
3393 Some(module_id) => {
3394 write!(f, "supervisor state for module '{module_id}' was poisoned")
3395 }
3396 None => write!(f, "supervisor state was poisoned"),
3397 },
3398 Self::CommandClosed { module_id } => {
3399 write!(
3400 f,
3401 "supervisor command channel for module '{module_id}' is closed"
3402 )
3403 }
3404 Self::SwapInProgress { module_id } => write!(
3405 f,
3406 "module '{module_id}' is being swapped; retry once the swap has cut over or failed, or stop the module to abort the swap"
3407 ),
3408 Self::SwapRefused { module_id, reason } => match reason {
3409 SwapRefusal::OverlapExclusive => write!(
3410 f,
3411 "module '{module_id}' is declared overlap: \"exclusive\" (the default): two processes of it must not run at once, so it cannot be swapped; use a plain restart, or declare overlap: \"safe\" in its config if it really tolerates a second process"
3412 ),
3413 SwapRefusal::NotRegistered => write!(
3414 f,
3415 "module '{module_id}' is not registered, so there is no serving process to keep while a replacement warms; use a plain restart"
3416 ),
3417 SwapRefusal::ProtocolNone => write!(
3418 f,
3419 "module '{module_id}' is protocol: \"none\" and never registers, so a swap could never see its replacement become ready; use a plain restart"
3420 ),
3421 SwapRefusal::NotConfigured => write!(
3422 f,
3423 "module '{module_id}' cannot be swapped: the supervisor was built without the forwarding table or shared handle a swap needs"
3424 ),
3425 SwapRefusal::AlreadySwapping => {
3426 write!(f, "module '{module_id}' is already being swapped")
3427 }
3428 },
3429 Self::SwapFailed {
3430 module_id,
3431 arm,
3432 detail,
3433 ..
3434 } => write!(
3435 f,
3436 "swap of module '{module_id}' failed ({}): {detail}; the running process was left serving",
3437 arm.as_str()
3438 ),
3439 }
3440 }
3441}
3442
3443impl Error for SuperviseError {
3444 fn source(&self) -> Option<&(dyn Error + 'static)> {
3445 match self {
3446 Self::Spawn { source, .. }
3447 | Self::Cgroup { source, .. }
3448 | Self::Wait { source, .. }
3449 | Self::Kill { source, .. } => Some(source),
3450 Self::Forwarding(err) => Some(err),
3451 Self::Registry(err) => Some(err),
3452 Self::LaunchNonce { .. }
3453 | Self::InvalidSpec { .. }
3454 | Self::ReloadUnavailable { .. }
3455 | Self::Disabled { .. }
3456 | Self::ReloadFailed { .. }
3457 | Self::RegistrationStillActive { .. }
3458 | Self::StatePoisoned { .. }
3459 | Self::CommandClosed { .. }
3460 | Self::SwapInProgress { .. }
3461 | Self::SwapRefused { .. }
3462 | Self::SwapFailed { .. } => None,
3463 }
3464 }
3465}
3466
3467pub(crate) fn validate_spec(spec: &ModuleSpec) -> Result<(), SuperviseError> {
3468 if spec.module_id.trim().is_empty() {
3469 return Err(SuperviseError::InvalidSpec {
3470 reason: "module_id must not be empty".to_string(),
3471 });
3472 }
3473
3474 Ok(())
3475}
3476
3477#[derive(Debug, Default)]
3478struct HealthProbeRuntime {
3479 registered_connection: Option<crate::ConnectionId>,
3480 advertised: bool,
3481 next_probe_at: Option<Instant>,
3482 probe_index: u64,
3483}
3484
3485impl HealthProbeRuntime {
3486 fn refresh_registration(
3487 &mut self,
3488 spec: &ModuleSpec,
3489 runtime: &SupervisorRuntimeConfig,
3490 registry: &Registry,
3491 snapshot: &SharedSnapshot,
3492 ) {
3493 if spec.protocol == ModuleProtocol::None {
3505 self.registered_connection = None;
3506 self.advertised = false;
3507 self.next_probe_at = None;
3508 return;
3509 }
3510
3511 let registration = match registry.get_module(&spec.module_id) {
3512 Ok(registration) => registration,
3513 Err(err) => {
3514 warn!(module_id = %spec.module_id, error = %err, "health prober could not read registry");
3515 self.advertised = false;
3516 self.next_probe_at = None;
3517 return;
3518 }
3519 };
3520
3521 let Some(registration) = registration else {
3522 self.registered_connection = None;
3523 self.advertised = false;
3524 self.next_probe_at = None;
3525 return;
3526 };
3527
3528 let advertised = registration
3529 .control_ops
3530 .iter()
3531 .any(|op| op == MODULE_CONTROL_OP_HEALTH_CHECK);
3532 if !advertised {
3533 self.registered_connection = Some(registration.connection_id);
3534 self.advertised = false;
3535 self.next_probe_at = None;
3536 let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3537 state.health.status = SupervisorHealthStatus::Unknown;
3538 state.health.consecutive_failures = 0;
3539 state.health.last_probe_ms = None;
3540 state.health.detail = None;
3541 state.health.metrics = None;
3542 });
3543 return;
3544 }
3545
3546 let reregistered = self.registered_connection != Some(registration.connection_id);
3547 self.registered_connection = Some(registration.connection_id);
3548 self.advertised = true;
3549 if reregistered || self.next_probe_at.is_none() {
3550 self.probe_index = 0;
3551 self.next_probe_at = Some(
3552 Instant::now() + jittered_health_delay(&spec.module_id, 0, runtime.health.cadence),
3553 );
3554 let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3555 state.health.status = SupervisorHealthStatus::Unknown;
3556 state.health.consecutive_failures = 0;
3557 state.health.detail = None;
3558 state.health.metrics = None;
3559 });
3560 }
3561 }
3562
3563 fn wake_after(&self) -> Duration {
3564 if !self.advertised {
3565 return REGISTRY_RELEASE_POLL;
3566 }
3567 self.next_probe_at
3568 .map(|next| next.saturating_duration_since(Instant::now()))
3569 .unwrap_or(REGISTRY_RELEASE_POLL)
3570 }
3571
3572 fn due(&self) -> bool {
3573 self.advertised
3574 && self
3575 .next_probe_at
3576 .is_some_and(|next| Instant::now() >= next)
3577 }
3578
3579 fn schedule_next(&mut self, spec: &ModuleSpec, cadence: Duration) {
3580 self.probe_index = self.probe_index.wrapping_add(1);
3581 self.next_probe_at = Some(
3582 Instant::now() + jittered_health_delay(&spec.module_id, self.probe_index, cadence),
3583 );
3584 }
3585}
3586
3587#[derive(Debug)]
3622enum HealthProbeEvidence {
3623 LaneDead,
3625 NoAnswer,
3627 BadAnswer,
3629 Misconfigured,
3631}
3632
3633#[derive(Debug)]
3634struct HealthProbeError {
3635 evidence: HealthProbeEvidence,
3636 message: String,
3637}
3638
3639impl HealthProbeError {
3640 fn lane_dead(message: impl Into<String>) -> Self {
3641 Self::with(HealthProbeEvidence::LaneDead, message)
3642 }
3643
3644 fn no_answer(message: impl Into<String>) -> Self {
3645 Self::with(HealthProbeEvidence::NoAnswer, message)
3646 }
3647
3648 fn bad_answer(message: impl Into<String>) -> Self {
3649 Self::with(HealthProbeEvidence::BadAnswer, message)
3650 }
3651
3652 fn misconfigured(message: impl Into<String>) -> Self {
3653 Self::with(HealthProbeEvidence::Misconfigured, message)
3654 }
3655
3656 fn with(evidence: HealthProbeEvidence, message: impl Into<String>) -> Self {
3657 Self {
3658 evidence,
3659 message: message.into(),
3660 }
3661 }
3662
3663 #[allow(dead_code)]
3677 fn is_proof_of_death(&self) -> bool {
3678 matches!(self.evidence, HealthProbeEvidence::LaneDead)
3679 }
3680
3681 fn label(&self) -> &'static str {
3689 match self.evidence {
3690 HealthProbeEvidence::LaneDead => "lane-dead",
3691 HealthProbeEvidence::NoAnswer => "no-answer",
3692 HealthProbeEvidence::BadAnswer => "bad-answer",
3693 HealthProbeEvidence::Misconfigured => "daemon-misconfigured",
3694 }
3695 }
3696}
3697
3698impl fmt::Display for HealthProbeError {
3699 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
3700 f.write_str(&self.message)
3701 }
3702}
3703
3704async fn run_health_probe_cycle(
3705 spec: &ModuleSpec,
3706 runtime: &SupervisorRuntimeConfig,
3707 registry: &Registry,
3708 process_liveness: &SupervisorProcessLiveness,
3709 snapshot: &SharedSnapshot,
3710 child: &mut Option<SupervisedChild>,
3711) {
3712 let now_ms = unix_ms_now();
3713 match probe_module_health(&spec.module_id, runtime, None).await {
3714 Ok(report) => {
3715 handle_health_report(
3716 spec,
3717 runtime,
3718 registry,
3719 process_liveness,
3720 snapshot,
3721 child,
3722 report,
3723 now_ms,
3724 )
3725 .await;
3726 }
3727 Err(err) => {
3728 handle_health_probe_failure(
3729 spec,
3730 runtime,
3731 registry,
3732 process_liveness,
3733 snapshot,
3734 child,
3735 err,
3736 now_ms,
3737 )
3738 .await;
3739 }
3740 }
3741}
3742
3743async fn probe_module_health(
3744 module_id: &str,
3745 runtime: &SupervisorRuntimeConfig,
3746 drain_deadline: Option<Instant>,
3747) -> Result<HealthReport, HealthProbeError> {
3748 let Some(forwarding) = runtime.forwarding.as_ref() else {
3749 return Err(HealthProbeError::misconfigured(
3750 "supervisor was not configured with a forwarding table",
3751 ));
3752 };
3753 let probe_started_at = Instant::now();
3754 let mut deadline = probe_started_at + runtime.health.deadline;
3755 if let Some(drain_deadline) = drain_deadline {
3756 deadline = deadline.min(drain_deadline);
3757 }
3758 let pending = if drain_deadline.is_some() {
3759 forwarding.begin_drain_health_probe_rpc_for(
3760 module_id,
3761 MODULE_CONTROL_OP_HEALTH_CHECK,
3762 probe_started_at,
3763 deadline,
3764 )
3765 } else {
3766 forwarding.begin_health_probe_rpc_for(
3767 module_id,
3768 MODULE_CONTROL_OP_HEALTH_CHECK,
3769 probe_started_at,
3770 deadline,
3771 )
3772 }
3773 .map_err(|err| {
3774 HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
3777 })?;
3778 await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
3779}
3780
3781async fn probe_endpoint_health(
3788 endpoint: crate::ModuleEndpointId,
3789 runtime: &SupervisorRuntimeConfig,
3790 deadline_cap: Option<Instant>,
3791) -> Result<HealthReport, HealthProbeError> {
3792 let Some(forwarding) = runtime.forwarding.as_ref() else {
3793 return Err(HealthProbeError::misconfigured(
3794 "supervisor was not configured with a forwarding table",
3795 ));
3796 };
3797 let probe_started_at = Instant::now();
3798 let mut deadline = probe_started_at + runtime.health.deadline;
3799 if let Some(cap) = deadline_cap {
3800 deadline = deadline.min(cap);
3801 }
3802 let pending = forwarding
3803 .begin_endpoint_health_probe_rpc_for(
3804 endpoint,
3805 MODULE_CONTROL_OP_HEALTH_CHECK,
3806 probe_started_at,
3807 deadline,
3808 )
3809 .map_err(|err| {
3810 HealthProbeError::lane_dead(format!("failed to begin health.check RPC: {err}"))
3811 })?;
3812 await_health_probe(forwarding, pending, deadline, runtime.health.deadline).await
3813}
3814
3815async fn await_health_probe(
3817 forwarding: &ForwardingTable,
3818 pending: PendingModuleControlRpc,
3819 deadline: Instant,
3820 probe_budget: Duration,
3821) -> Result<HealthReport, HealthProbeError> {
3822 let PendingModuleControlRpc {
3823 endpoint,
3824 module_sink,
3825 negotiated_ver,
3826 corr,
3827 receiver,
3828 } = pending;
3829 let body = serde_json::to_vec(&ModuleControlRequest::HealthCheck {}).map_err(|err| {
3830 HealthProbeError::misconfigured(format!("failed to encode health.check: {err}"))
3831 })?;
3832 let frame = Frame::build_with_version(
3833 negotiated_ver,
3834 FrameType::Request,
3835 control_flags(),
3836 0,
3837 0,
3838 corr,
3839 body,
3840 )
3841 .map_err(|err| {
3842 HealthProbeError::misconfigured(format!("failed to build health.check frame: {err}"))
3843 })?;
3844
3845 match timeout_at(deadline, module_sink.send(frame)).await {
3851 Ok(Ok(())) => {}
3852 Ok(Err(err)) => {
3853 let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
3854 return Err(HealthProbeError::lane_dead(format!(
3857 "failed to send health.check: {err}"
3858 )));
3859 }
3860 Err(_elapsed) => {
3861 let _ = forwarding.cancel_module_control_rpc(endpoint, corr);
3862 return Err(HealthProbeError::no_answer(
3866 "health.check send timed out before enqueue (module egress full)",
3867 ));
3868 }
3869 }
3870
3871 match timeout_at(deadline, receiver).await {
3872 Ok(Ok(ModuleControlRpcOutcome::Response(response))) => {
3876 response.health_report().ok_or_else(|| {
3877 HealthProbeError::bad_answer("health.check RPC returned a non-health response")
3878 })
3879 }
3880 Ok(Ok(ModuleControlRpcOutcome::Rejected(body))) => Err(HealthProbeError::bad_answer(
3881 format!("health.check rejected: {}", body.message),
3882 )),
3883 Ok(Ok(ModuleControlRpcOutcome::ModuleGone(message))) => {
3884 Err(HealthProbeError::lane_dead(message))
3885 }
3886 Ok(Ok(ModuleControlRpcOutcome::MalformedResponse(message))) => {
3887 Err(HealthProbeError::bad_answer(message))
3888 }
3889 Ok(Ok(ModuleControlRpcOutcome::UnexpectedOp { expected, actual })) => {
3890 Err(HealthProbeError::bad_answer(format!(
3891 "expected module-control op '{expected}', got '{actual}'"
3892 )))
3893 }
3894 Ok(Ok(ModuleControlRpcOutcome::DeadlineElapsed)) => Err(HealthProbeError::bad_answer(
3898 "module answered health.check after its daemon deadline",
3899 )),
3900 Ok(Err(_)) => Err(HealthProbeError::misconfigured(
3901 "health.check waiter was canceled before the module responded",
3902 )),
3903 Err(_) => {
3904 let _ = forwarding.tombstone_health_probe_rpc(endpoint, corr);
3905 Err(HealthProbeError::no_answer(format!(
3906 "module did not answer health.check within {probe_budget:?}"
3907 )))
3908 }
3909 }
3910}
3911
3912#[allow(clippy::too_many_arguments)]
3913async fn handle_health_report(
3914 spec: &ModuleSpec,
3915 runtime: &SupervisorRuntimeConfig,
3916 registry: &Registry,
3917 process_liveness: &SupervisorProcessLiveness,
3918 snapshot: &SharedSnapshot,
3919 child: &mut Option<SupervisedChild>,
3920 report: HealthReport,
3921 now_ms: u64,
3922) {
3923 let status = supervisor_health_status(report.status);
3924 let detail = report.detail.clone();
3925 let metrics = truncate_health_metrics(report.metrics);
3926 let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3927 state.health.status = status;
3928 state.health.last_probe_ms = Some(now_ms);
3929 state.health.detail = detail.clone();
3930 state.health.metrics = metrics.clone();
3931 state.health.consecutive_failures = 0;
3932 });
3933
3934 let action = match report.status {
3935 HealthStatus::Ok => return,
3936 HealthStatus::Degraded => runtime.health.on_degraded,
3937 HealthStatus::Failing => runtime.health.on_failing,
3938 };
3939 apply_l3_health_action(
3940 spec,
3941 runtime,
3942 registry,
3943 process_liveness,
3944 snapshot,
3945 child,
3946 status,
3947 detail.as_deref(),
3948 action,
3949 now_ms,
3950 )
3951 .await;
3952}
3953
3954#[allow(clippy::too_many_arguments)]
3955async fn handle_health_probe_failure(
3956 spec: &ModuleSpec,
3957 runtime: &SupervisorRuntimeConfig,
3958 registry: &Registry,
3959 process_liveness: &SupervisorProcessLiveness,
3960 snapshot: &SharedSnapshot,
3961 child: &mut Option<SupervisedChild>,
3962 err: HealthProbeError,
3963 now_ms: u64,
3964) {
3965 let threshold = runtime.health.failure_threshold.max(1);
3966 let mut failures = 0;
3967 let detail = format!("[{}] {err}", err.label());
3972 let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3973 state.health.last_probe_ms = Some(now_ms);
3974 state.health.consecutive_failures = state.health.consecutive_failures.saturating_add(1);
3975 state.health.detail = Some(detail.clone());
3976 state.health.metrics = None;
3977 failures = state.health.consecutive_failures;
3978 });
3979
3980 if failures < threshold {
3981 warn!(
3982 module_id = %spec.module_id,
3983 consecutive_failures = failures,
3984 threshold,
3985 evidence = err.label(),
3986 detail = %detail,
3987 "health.check probe failed"
3988 );
3989 return;
3990 }
3991
3992 let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
3993 state.state = ModuleState::Unresponsive;
3994 state.health.status = SupervisorHealthStatus::Unresponsive;
3995 });
3996 if runtime.health.critical {
4000 error!(
4001 module_id = %spec.module_id,
4002 status = "unresponsive",
4003 evidence = err.label(),
4004 detail = %detail,
4005 "critical module health alert"
4006 );
4007 } else {
4008 warn!(
4009 module_id = %spec.module_id,
4010 status = "unresponsive",
4011 evidence = err.label(),
4012 detail = %detail,
4013 "module health threshold breached"
4014 );
4015 }
4016 if let Err(err) = health_restart_child(
4017 spec,
4018 runtime,
4019 registry,
4020 process_liveness,
4021 snapshot,
4022 child,
4023 SupervisorHealthStatus::Unresponsive,
4024 Some(&detail),
4025 now_ms,
4026 )
4027 .await
4028 {
4029 error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
4030 }
4031}
4032
4033#[allow(clippy::too_many_arguments)]
4034async fn apply_l3_health_action(
4035 spec: &ModuleSpec,
4036 runtime: &SupervisorRuntimeConfig,
4037 registry: &Registry,
4038 process_liveness: &SupervisorProcessLiveness,
4039 snapshot: &SharedSnapshot,
4040 child: &mut Option<SupervisedChild>,
4041 status: SupervisorHealthStatus,
4042 detail: Option<&str>,
4043 action: HealthAction,
4044 now_ms: u64,
4045) {
4046 record_health_action(snapshot, &spec.module_id, action.to_string(), now_ms);
4047 match action {
4048 HealthAction::Report => {
4049 info!(
4050 module_id = %spec.module_id,
4051 status = ?status,
4052 detail,
4053 "module reported non-ok health"
4054 );
4055 }
4056 HealthAction::Alert => {
4057 error!(
4058 module_id = %spec.module_id,
4059 status = ?status,
4060 detail,
4061 "module health alert"
4062 );
4063 }
4064 HealthAction::Restart => {
4065 if let Err(err) = health_restart_child(
4066 spec,
4067 runtime,
4068 registry,
4069 process_liveness,
4070 snapshot,
4071 child,
4072 status,
4073 detail,
4074 now_ms,
4075 )
4076 .await
4077 {
4078 error!(module_id = %spec.module_id, error = %err, "health-triggered restart failed");
4079 }
4080 }
4081 }
4082}
4083
4084#[allow(clippy::too_many_arguments)]
4085async fn health_restart_child(
4086 spec: &ModuleSpec,
4087 runtime: &SupervisorRuntimeConfig,
4088 registry: &Registry,
4089 process_liveness: &SupervisorProcessLiveness,
4090 snapshot: &SharedSnapshot,
4091 child: &mut Option<SupervisedChild>,
4092 status: SupervisorHealthStatus,
4093 detail: Option<&str>,
4094 now_ms: u64,
4095) -> Result<(), SuperviseError> {
4096 let (enabled, schedule) = {
4097 let mut state = lock_snapshot(snapshot)?;
4098 let enabled = state.enabled;
4099 let schedule = if enabled {
4100 state.next_crash_restart(&runtime.restart_policy, Instant::now())
4101 } else {
4102 None
4103 };
4104 (enabled, schedule)
4105 };
4106
4107 if !enabled {
4108 return Err(SuperviseError::Disabled {
4109 module_id: spec.module_id.clone(),
4110 });
4111 }
4112
4113 if schedule.is_none() {
4114 record_health_action(snapshot, &spec.module_id, "failed".to_string(), now_ms);
4115 error!(
4116 module_id = %spec.module_id,
4117 status = ?status,
4118 detail,
4119 max_restarts = runtime.restart_policy.max_restarts,
4120 window_secs = runtime.restart_policy.window.as_secs(),
4121 reason = %runtime.restart_policy.budget_exhausted_detail(),
4122 "health restart budget exhausted; marking module failed"
4123 );
4124 let stop_notice = begin_forwarding_drain_if_configured(
4125 spec,
4126 runtime,
4127 registry,
4128 snapshot,
4129 Some(true),
4130 RouteCloseReason::Disable,
4131 )
4132 .await?;
4133 update_snapshot(snapshot, Some(&spec.module_id), |state| {
4134 state.drain_disposition_detail = Some(runtime.restart_policy.budget_exhausted_detail());
4135 })?;
4136 drain_optional_child(
4137 &spec.module_id,
4138 spec.protocol,
4139 stop_notice,
4140 registry,
4141 runtime.forwarding.as_deref(),
4142 snapshot,
4143 &runtime.terminal_ring,
4144 &runtime.spawn_events,
4145 child,
4146 runtime.drain_timeout,
4147 ModuleState::Failed,
4148 Some(true),
4149 )
4150 .await?;
4151 process_liveness.untrack_if_current(&spec.module_id, snapshot);
4152 return Ok(());
4153 }
4154
4155 let schedule = schedule.expect("a health restart must have a crash-restart schedule");
4156 let mut restart_count = 0;
4157 update_snapshot(snapshot, Some(&spec.module_id), |state| {
4158 restart_count = state.crash_restarts.len();
4159 state.state = ModuleState::Unresponsive;
4160 state.health.status = status;
4161 state.health.last_action = Some(HealthAction::Restart.to_string());
4162 state.health.last_action_ms = Some(now_ms);
4163 })?;
4164 warn!(
4165 module_id = %spec.module_id,
4166 status = ?status,
4167 detail,
4168 restart_count,
4169 restart_in_window = schedule.restart_in_window,
4170 delay_ms = schedule.delay.as_millis() as u64,
4171 "health-triggered module restart"
4172 );
4173
4174 let stop_notice = begin_forwarding_drain_if_configured(
4175 spec,
4176 runtime,
4177 registry,
4178 snapshot,
4179 Some(true),
4180 RouteCloseReason::Restart,
4181 )
4182 .await?;
4183 drain_optional_child(
4184 &spec.module_id,
4185 spec.protocol,
4186 stop_notice,
4187 registry,
4188 runtime.forwarding.as_deref(),
4189 snapshot,
4190 &runtime.terminal_ring,
4191 &runtime.spawn_events,
4192 child,
4193 runtime.drain_timeout,
4194 ModuleState::Restarting,
4195 Some(true),
4196 )
4197 .await?;
4198 schedule_respawn(
4199 runtime,
4200 snapshot,
4201 &spec.module_id,
4202 schedule.delay,
4203 RespawnKind::Spawn,
4204 )
4205}
4206
4207fn cancel_deferred_reload(runtime: &SupervisorRuntimeConfig, module_id: &str, reason: &str) {
4208 if let Some(reply) = runtime
4209 .deferred_reload_reply
4210 .lock()
4211 .unwrap_or_else(|p| p.into_inner())
4212 .take()
4213 {
4214 let _ = reply.send(Err(SuperviseError::ReloadFailed {
4215 module_id: module_id.to_string(),
4216 reason: reason.to_string(),
4217 }));
4218 }
4219}
4220
4221fn schedule_respawn(
4222 runtime: &SupervisorRuntimeConfig,
4223 snapshot: &SharedSnapshot,
4224 module_id: &str,
4225 delay: Duration,
4226 kind: RespawnKind,
4227) -> Result<(), SuperviseError> {
4228 cancel_deferred_reload(runtime, module_id, "respawn superseded by another restart");
4229 update_snapshot(snapshot, Some(module_id), |state| {
4230 state.respawn_pending = true
4231 })?;
4232 *runtime
4233 .scheduled_respawn
4234 .lock()
4235 .unwrap_or_else(|p| p.into_inner()) = Some(PendingRespawn {
4236 deadline: Instant::now() + delay,
4237 kind,
4238 });
4239 Ok(())
4240}
4241
4242fn record_health_action(snapshot: &SharedSnapshot, module_id: &str, action: String, now_ms: u64) {
4243 let _ = update_snapshot(snapshot, Some(module_id), |state| {
4244 state.health.last_action = Some(action);
4245 state.health.last_action_ms = Some(now_ms);
4246 });
4247}
4248
4249fn supervisor_health_status(status: HealthStatus) -> SupervisorHealthStatus {
4250 match status {
4251 HealthStatus::Ok => SupervisorHealthStatus::Ok,
4252 HealthStatus::Degraded => SupervisorHealthStatus::Degraded,
4253 HealthStatus::Failing => SupervisorHealthStatus::Failing,
4254 }
4255}
4256
4257fn truncate_health_metrics(metrics: Option<Value>) -> Option<Value> {
4269 let metrics = metrics?;
4270 match serde_json::to_vec(&metrics) {
4271 Ok(encoded) if encoded.len() > MAX_HEALTH_METRICS_BYTES => Some(serde_json::json!({
4272 "truncated": true,
4273 "original_bytes": encoded.len(),
4274 })),
4275 Ok(_) | Err(_) => Some(metrics),
4276 }
4277}
4278
4279fn jittered_health_delay(module_id: &str, probe_index: u64, cadence: Duration) -> Duration {
4285 if cadence.is_zero() {
4286 return Duration::ZERO;
4287 }
4288 let cadence_ms = cadence.as_millis() as u64;
4289 if cadence_ms == 0 {
4305 return cadence;
4306 }
4307 let jitter_span = (cadence_ms / 10).max(1);
4322 let hash = module_id.as_bytes().iter().fold(
4323 probe_index.wrapping_mul(0x9E37_79B9_7F4A_7C15),
4324 |acc, byte| {
4325 acc.wrapping_mul(1099511628211)
4326 .wrapping_add(u64::from(*byte))
4327 },
4328 );
4329 cadence + Duration::from_millis(hash % jitter_span)
4330}
4331
4332#[cfg(test)]
4333mod tests {
4334 use super::*;
4335
4336 #[test]
4337 fn readding_a_module_clears_its_rescan_removal_tombstone() {
4338 let handle = SupervisorHandle::new();
4339 let module_id = "readded-tombstone";
4340 handle.record_rescan_removal(module_id);
4341 assert!(handle.removal_tombstone_age_ms(module_id).is_some());
4342
4343 handle.apply_identity_configuration(&ModuleSpec {
4344 module_id: module_id.to_string(),
4345 program: PathBuf::from("/test/module"),
4346 args: Vec::new(),
4347 env: Vec::new(),
4348 reserved: false,
4349 reserved_prefixes: Vec::new(),
4350 protocol: ModuleProtocol::Subc,
4351 overlap: Default::default(),
4352 });
4353
4354 assert!(
4355 handle.removal_tombstone_age_ms(module_id).is_none(),
4356 "a re-added module must not retain a stale removal tombstone"
4357 );
4358 }
4359
4360 fn stale_process_snapshot(state: ModuleState, enabled: bool) -> SharedSnapshot {
4361 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(state, enabled)));
4362 update_snapshot(&snapshot, Some("stale-process-facts"), |snapshot| {
4363 snapshot.process_alive = true;
4364 snapshot.pid = Some(41);
4365 snapshot.spawned_at_ms = Some(42);
4366 snapshot.spawned_from = Some(PathBuf::from("/spawned/module"));
4367 snapshot.spawned_file_identity = Some(SpawnedFileIdentity {
4368 device: 43,
4369 inode: 44,
4370 });
4371 })
4372 .unwrap();
4373 snapshot
4374 }
4375
4376 fn assert_snapshot_process_facts_cleared(snapshot: &SharedSnapshot) {
4377 let snapshot = lock_snapshot(snapshot).unwrap();
4378 assert!(!snapshot.process_alive);
4379 assert_eq!(snapshot.pid, None);
4380 assert_eq!(snapshot.spawned_at_ms, None);
4381 assert_eq!(snapshot.spawned_from, None);
4382 assert_eq!(snapshot.spawned_file_identity, None);
4383 }
4384
4385 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
4386 async fn failed_enable_spawn_clears_preexisting_current_process_facts() {
4387 let supervisor = Supervisor::default();
4388 let mut runtime = supervisor.runtime_config();
4389 runtime.test_seed_stale_facts_before_enable_spawn = true;
4390 let snapshot = stale_process_snapshot(ModuleState::Disabled, false);
4391 let mut child = None;
4392 let spec = ModuleSpec {
4393 module_id: "failed-enable-clears-facts".to_string(),
4394 program: PathBuf::from("/definitely/missing/failed-enable-module"),
4395 args: Vec::new(),
4396 env: Vec::new(),
4397 reserved: false,
4398 reserved_prefixes: Vec::new(),
4399 protocol: ModuleProtocol::Subc,
4400 overlap: Default::default(),
4401 };
4402
4403 let result = set_child_enabled(
4404 &spec,
4405 &runtime,
4406 &supervisor.registry,
4407 &supervisor.process_liveness,
4408 &snapshot,
4409 &mut child,
4410 true,
4411 )
4412 .await;
4413
4414 assert!(matches!(result, Err(SuperviseError::Spawn { .. })));
4415 assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
4416 assert_snapshot_process_facts_cleared(&snapshot);
4417 }
4418
4419 #[tokio::test]
4420 async fn start_revives_stranded_restarting_but_not_pending_backoff() {
4421 let supervisor = Supervisor::default();
4422 let runtime = supervisor.runtime_config();
4423 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::new(
4424 ModuleState::Restarting,
4425 true,
4426 )));
4427 let spec = ModuleSpec {
4428 module_id: "start-stranded-restarting".to_string(),
4429 program: super::terminal_history_tests::fake_aft_stub_path(),
4430 args: Vec::new(),
4431 env: vec![("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string())],
4432 reserved: false,
4433 reserved_prefixes: Vec::new(),
4434 protocol: ModuleProtocol::None,
4435 overlap: Default::default(),
4436 };
4437 let mut child = None;
4438 lock_snapshot(&snapshot).unwrap().respawn_pending = true;
4439 assert!(!super::set_child_enabled(
4440 &spec,
4441 &runtime,
4442 &Registry::default(),
4443 &supervisor.process_liveness,
4444 &snapshot,
4445 &mut child,
4446 true
4447 )
4448 .await
4449 .unwrap());
4450 assert!(child.is_none());
4451 lock_snapshot(&snapshot).unwrap().respawn_pending = false;
4452 assert!(super::set_child_enabled(
4453 &spec,
4454 &runtime,
4455 &Registry::default(),
4456 &supervisor.process_liveness,
4457 &snapshot,
4458 &mut child,
4459 true
4460 )
4461 .await
4462 .unwrap());
4463 assert_eq!(
4464 lock_snapshot(&snapshot).unwrap().state,
4465 ModuleState::Running
4466 );
4467 let mut child = child.unwrap();
4468 child.start_kill().unwrap();
4469 child.wait().await.unwrap();
4470 }
4471
4472 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
4473 async fn failed_reload_spawn_clears_current_process_facts() {
4474 let supervisor = Supervisor::default();
4475 let mut runtime = supervisor.runtime_config();
4476 runtime.restart_policy = RestartPolicy::new(0, Duration::ZERO);
4477 let snapshot = stale_process_snapshot(ModuleState::Running, true);
4478 let mut child = None;
4479 let spec = ModuleSpec {
4480 module_id: "failed-reload-clears-facts".to_string(),
4481 program: PathBuf::from("/unused/failed-reload-module"),
4482 args: Vec::new(),
4483 env: Vec::new(),
4484 reserved: false,
4485 reserved_prefixes: Vec::new(),
4486 protocol: ModuleProtocol::Subc,
4487 overlap: Default::default(),
4488 };
4489
4490 let result = handle_reload_spawn_failure(
4491 &spec,
4492 &runtime,
4493 &supervisor.process_liveness,
4494 &snapshot,
4495 &mut child,
4496 "forced reload spawn failure".to_string(),
4497 )
4498 .await;
4499
4500 assert!(matches!(result, Err(SuperviseError::ReloadFailed { .. })));
4501 assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
4502 assert_snapshot_process_facts_cleared(&snapshot);
4503 }
4504
4505 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
4506 async fn dropping_a_module_with_an_active_monitor_clears_current_process_facts() {
4507 let supervisor = Supervisor::default();
4508 let snapshot = stale_process_snapshot(ModuleState::Running, true);
4509 let module = supervisor.supervised_module(
4510 ModuleSpec {
4511 module_id: "drop-clears-facts".to_string(),
4512 program: PathBuf::from("/unused/drop-module"),
4513 args: Vec::new(),
4514 env: Vec::new(),
4515 reserved: false,
4516 reserved_prefixes: Vec::new(),
4517 protocol: ModuleProtocol::Subc,
4518 overlap: Default::default(),
4519 },
4520 supervisor.runtime_config(),
4521 Arc::clone(&snapshot),
4522 None,
4523 );
4524 assert!(!module
4525 .inner
4526 .monitor
4527 .lock()
4528 .unwrap()
4529 .as_ref()
4530 .unwrap()
4531 .is_finished());
4532
4533 drop(module);
4534
4535 assert_eq!(
4536 lock_snapshot(&snapshot).unwrap().state,
4537 ModuleState::Stopped
4538 );
4539 assert_snapshot_process_facts_cleared(&snapshot);
4540 }
4541
4542 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
4543 async fn configuration_update_does_not_replace_captured_running_process_facts() {
4544 let supervisor = Supervisor::default();
4545 let snapshot = stale_process_snapshot(ModuleState::Running, true);
4546 let initial = ModuleSpec {
4547 module_id: "rescan-preserves-spawn-facts".to_string(),
4548 program: PathBuf::from("/spawned/module"),
4549 args: Vec::new(),
4550 env: Vec::new(),
4551 reserved: false,
4552 reserved_prefixes: Vec::new(),
4553 protocol: ModuleProtocol::Subc,
4554 overlap: Default::default(),
4555 };
4556 let module = supervisor.supervised_module(
4557 initial.clone(),
4558 supervisor.runtime_config(),
4559 snapshot,
4560 None,
4561 );
4562 let before = module.status().unwrap();
4563 let mut replacement = initial;
4564 replacement.program = PathBuf::from("/rescanned/replacement-module");
4565
4566 module
4567 .update_configuration(replacement, HealthConfig::default(), None)
4568 .await
4569 .unwrap();
4570
4571 let after = module.status().unwrap();
4572 assert_eq!(after.pid, before.pid);
4573 assert_eq!(after.spawned_at_ms, before.spawned_at_ms);
4574 assert_eq!(after.spawned_from, before.spawned_from);
4575 drop(module);
4576 }
4577}
4578
4579fn unix_ms_now() -> u64 {
4580 SystemTime::now()
4581 .duration_since(UNIX_EPOCH)
4582 .map(|duration| duration.as_millis().min(u128::from(u64::MAX)) as u64)
4583 .unwrap_or(0)
4584}
4585
4586async fn supervise_loop(
4587 mut spec: ModuleSpec,
4588 mut runtime: SupervisorRuntimeConfig,
4589 registry: Arc<Registry>,
4590 process_liveness: Arc<SupervisorProcessLiveness>,
4591 snapshot: SharedSnapshot,
4592 mut child: Option<SupervisedChild>,
4593 mut commands: mpsc::Receiver<SupervisorCommand>,
4594) {
4595 let mut health_probe = HealthProbeRuntime::default();
4596 let mut pending_respawn: Option<PendingRespawn> = None;
4600 let mut requeued: VecDeque<SupervisorCommand> = VecDeque::new();
4603 loop {
4604 if let Some(scheduled) = runtime
4605 .scheduled_respawn
4606 .lock()
4607 .unwrap_or_else(|p| p.into_inner())
4608 .take()
4609 {
4610 pending_respawn = Some(scheduled);
4611 }
4612 if pending_respawn.is_some() && (child.is_some() || !respawn_still_pending(&snapshot)) {
4613 pending_respawn = None;
4614 cancel_deferred_reload(
4615 &runtime,
4616 &spec.module_id,
4617 "respawn cancelled by a supervisor command",
4618 );
4619 }
4620 if child.is_none() && pending_respawn.is_none() {
4621 cancel_deferred_reload(
4622 &runtime,
4623 &spec.module_id,
4624 "respawn cancelled before a replacement was spawned",
4625 );
4626 let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
4627 state.respawn_pending = false;
4628 state.coalesced_restart_pending = false;
4629 if matches!(
4630 state.state,
4631 ModuleState::Restarting
4632 | ModuleState::Starting
4633 | ModuleState::Draining
4634 | ModuleState::Unresponsive
4635 ) {
4636 error!(module_id = %spec.module_id, state = ?state.state,
4637 "supervision operation ended without a child or pending respawn; marking failed so start can retry");
4638 state.state = ModuleState::Failed;
4639 clear_current_process_facts(state);
4640 }
4641 });
4642 }
4643 if let Some(command) = requeued.pop_front() {
4644 if !handle_supervisor_command(
4645 command,
4646 &mut spec,
4647 &mut runtime,
4648 ®istry,
4649 &process_liveness,
4650 &snapshot,
4651 &mut child,
4652 &mut commands,
4653 &mut requeued,
4654 )
4655 .await
4656 {
4657 return;
4658 }
4659 if child.is_some() || !respawn_still_pending(&snapshot) {
4660 pending_respawn = None;
4661 let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
4662 state.respawn_pending = false
4663 });
4664 }
4665 continue;
4666 }
4667 if child.is_some() {
4668 health_probe.refresh_registration(&spec, &runtime, ®istry, &snapshot);
4669 let probe_sleep = sleep(health_probe.wake_after());
4670 tokio::pin!(probe_sleep);
4671 let active_child = child.as_mut().expect("child checked above");
4672 tokio::select! {
4673 wait_result = active_child.wait() => {
4674 let exit_report = match wait_result {
4683 Ok(status) => classify_reaped_child_exit(&snapshot, active_child, &status),
4684 Err(err) => {
4685 active_child.drain_stderr(&spec.module_id).await;
4686 fail_snapshot(&snapshot, Some(&spec.module_id), None);
4687 record_wait_error_terminal(
4693 &spec.module_id,
4694 &runtime.terminal_ring,
4695 &runtime.spawn_events,
4696 );
4697 untrack_if_registration_released(
4698 &process_liveness,
4699 ®istry,
4700 &spec.module_id,
4701 &snapshot,
4702 );
4703 error!(module_id = %spec.module_id, error = %err, "failed to wait for supervised module");
4704 child = None;
4705 continue;
4706 }
4707 };
4708 active_child.drain_stderr(&spec.module_id).await;
4709
4710 let next = on_child_exit(
4711 &spec,
4712 runtime.restart_policy,
4713 ®istry,
4714 &snapshot,
4715 &runtime.terminal_ring,
4716 &runtime.spawn_events,
4717 &runtime.child_roster,
4718 exit_report,
4719 ).await;
4720 active_child.release_roster();
4723 match next {
4724 NextAction::Stop { registration_released } => {
4725 if registration_released {
4726 process_liveness.untrack_if_current(&spec.module_id, &snapshot);
4727 }
4728 child = None;
4729 }
4730 NextAction::Restart { schedule } => {
4731 let delay = schedule.map_or(
4732 runtime.restart_policy.delay_for_restart(0),
4733 |schedule| schedule.delay,
4734 );
4735 if let Some(schedule) = schedule {
4736 log_crash_respawn(&spec.module_id, schedule);
4737 }
4738 child = None;
4746 pending_respawn = Some(PendingRespawn { deadline: Instant::now() + delay, kind: RespawnKind::Spawn });
4747 let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = true);
4748 }
4749 }
4750 }
4751 command = commands.recv() => {
4752 let Some(command) = command else {
4753 return;
4754 };
4755 if !handle_supervisor_command(
4756 command,
4757 &mut spec,
4758 &mut runtime,
4759 ®istry,
4760 &process_liveness,
4761 &snapshot,
4762 &mut child,
4763 &mut commands,
4764 &mut requeued,
4765 ).await {
4766 return;
4767 }
4768 }
4769 _ = &mut probe_sleep => {
4770 if health_probe.due() {
4771 run_health_probe_cycle(
4772 &spec,
4773 &runtime,
4774 ®istry,
4775 &process_liveness,
4776 &snapshot,
4777 &mut child,
4778 ).await;
4779 if child.is_some() {
4780 health_probe.schedule_next(&spec, runtime.health.cadence);
4781 }
4782 }
4783 }
4784 }
4785 } else if let Some(pending) = pending_respawn {
4786 tokio::select! {
4787 _ = sleep_until(pending.deadline) => {
4788 pending_respawn = None;
4789 let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = false);
4790 if !respawn_still_pending(&snapshot) {
4794 continue;
4795 }
4796 if runtime.child_roster.is_closed() {
4801 let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| {
4802 state.state = ModuleState::Stopped;
4803 });
4804 debug!(module_id = %spec.module_id, "crash respawn cancelled by daemon shutdown");
4805 continue;
4806 }
4807 if let Err(err) = release_dead_registration(
4808 ®istry,
4809 runtime.forwarding.as_deref(),
4810 &snapshot,
4811 &spec.module_id,
4812 ).await {
4813 fail_snapshot(&snapshot, Some(&spec.module_id), None);
4814 error!(module_id = %spec.module_id, error = %err, "registration did not release before restart");
4815 continue;
4816 }
4817
4818 if matches!(pending.kind, RespawnKind::Reload) {
4819 let reply = runtime.deferred_reload_reply.lock().unwrap_or_else(|p| p.into_inner()).take();
4820 let result = finish_reload_child(&spec, &runtime, ®istry, &process_liveness, &snapshot, &mut child).await;
4821 if let Some(reply) = reply { let _ = reply.send(result); }
4822 continue;
4823 }
4824 process_liveness.track(spec.module_id.clone(), Arc::clone(&snapshot));
4825 match spawn_and_mark_running(&spec, &runtime, &snapshot) {
4826 Ok(next_child) => {
4827 child = Some(next_child);
4828 debug!(module_id = %spec.module_id, "supervised module restarted after crash");
4829 }
4830 Err(err) => {
4831 fail_snapshot(&snapshot, Some(&spec.module_id), None);
4832 process_liveness.untrack_if_current(&spec.module_id, &snapshot);
4833 error!(module_id = %spec.module_id, error = %err, "failed to restart supervised module");
4834 }
4835 }
4836 }
4837 command = commands.recv() => {
4838 let Some(command) = command else {
4839 return;
4840 };
4841 if !handle_supervisor_command(
4842 command,
4843 &mut spec,
4844 &mut runtime,
4845 ®istry,
4846 &process_liveness,
4847 &snapshot,
4848 &mut child,
4849 &mut commands,
4850 &mut requeued,
4851 ).await {
4852 return;
4853 }
4854 if child.is_some() || !respawn_still_pending(&snapshot) {
4859 pending_respawn = None;
4860 let _ = update_snapshot(&snapshot, Some(&spec.module_id), |state| state.respawn_pending = false);
4861 }
4862 }
4863 }
4864 } else {
4865 let Some(command) = commands.recv().await else {
4866 return;
4867 };
4868 if !handle_supervisor_command(
4869 command,
4870 &mut spec,
4871 &mut runtime,
4872 ®istry,
4873 &process_liveness,
4874 &snapshot,
4875 &mut child,
4876 &mut commands,
4877 &mut requeued,
4878 )
4879 .await
4880 {
4881 return;
4882 }
4883 }
4884 }
4885}
4886
4887fn log_crash_respawn(module_id: &str, schedule: CrashRestartSchedule) {
4888 info!(
4889 module_id,
4890 restart_in_window = schedule.restart_in_window,
4891 delay_ms = schedule.delay.as_millis() as u64,
4892 "respawning after crash"
4893 );
4894}
4895
4896fn respawn_still_pending(snapshot: &SharedSnapshot) -> bool {
4902 matches!(
4903 lock_snapshot(snapshot),
4904 Ok(state) if state.enabled && state.state == ModuleState::Restarting
4905 )
4906}
4907
4908enum NextAction {
4909 Stop {
4910 registration_released: bool,
4911 },
4912 Restart {
4913 schedule: Option<CrashRestartSchedule>,
4914 },
4915}
4916
4917#[allow(clippy::too_many_arguments)]
4918async fn handle_supervisor_command(
4919 command: SupervisorCommand,
4920 spec: &mut ModuleSpec,
4921 runtime: &mut SupervisorRuntimeConfig,
4922 registry: &Registry,
4923 process_liveness: &SupervisorProcessLiveness,
4924 snapshot: &SharedSnapshot,
4925 child: &mut Option<SupervisedChild>,
4926 commands: &mut mpsc::Receiver<SupervisorCommand>,
4927 requeued: &mut VecDeque<SupervisorCommand>,
4928) -> bool {
4929 match command {
4930 SupervisorCommand::Drain { reply } => {
4931 let result = drain_optional_child(
4934 &spec.module_id,
4935 spec.protocol,
4936 StopNotice::NotSent,
4937 registry,
4938 runtime.forwarding.as_deref(),
4939 snapshot,
4940 &runtime.terminal_ring,
4941 &runtime.spawn_events,
4942 child,
4943 runtime.drain_timeout,
4944 ModuleState::Stopped,
4945 None,
4946 )
4947 .await;
4948 let registration_released = result.is_ok();
4949 let _ = reply.send(result);
4950 if registration_released {
4951 process_liveness.untrack_if_current(&spec.module_id, snapshot);
4952 }
4953 false
4954 }
4955 SupervisorCommand::Retire { reply } => {
4956 let result = async {
4957 let stop_notice = begin_forwarding_drain_if_configured(
4958 spec,
4959 runtime,
4960 registry,
4961 snapshot,
4962 None,
4963 RouteCloseReason::Disable,
4964 )
4965 .await?;
4966 drain_optional_child(
4967 &spec.module_id,
4968 spec.protocol,
4969 stop_notice,
4970 registry,
4971 runtime.forwarding.as_deref(),
4972 snapshot,
4973 &runtime.terminal_ring,
4974 &runtime.spawn_events,
4975 child,
4976 runtime.drain_timeout,
4977 ModuleState::Stopped,
4978 None,
4979 )
4980 .await
4981 }
4982 .await;
4983 let registration_released = result.is_ok();
4984 let _ = reply.send(result);
4985 if registration_released {
4986 process_liveness.untrack_if_current(&spec.module_id, snapshot);
4987 }
4988 false
4989 }
4990 SupervisorCommand::Restart {
4991 drain_timeout_ms,
4992 received_at_generation,
4993 queued_at,
4994 reply,
4995 } => {
4996 info!(
5000 module_id = %spec.module_id,
5001 queued_ms = u64::try_from(queued_at.elapsed().as_millis()).unwrap_or(u64::MAX),
5002 "restart command dequeued"
5003 );
5004 let validation = match lock_snapshot(snapshot) {
5016 Ok(state) if !state.enabled => Err(SuperviseError::Disabled {
5017 module_id: spec.module_id.clone(),
5018 }),
5019 Ok(_) => Ok(()),
5020 Err(err) => Err(err),
5021 };
5022 let initiated = validation.is_ok();
5023 let _ = reply.send(validation);
5024 let satisfied_by_generation = if initiated && child.is_some() {
5035 lock_snapshot(snapshot).ok().and_then(|state| {
5036 (state.spawn_generation > received_at_generation
5037 && !state.configuration_updated_since_spawn)
5038 .then_some(state.spawn_generation)
5039 })
5040 } else {
5041 None
5042 };
5043 let satisfied_by_pending = initiated
5044 && child.is_none()
5045 && lock_snapshot(snapshot).ok().is_some_and(|mut state| {
5046 let pending = state.respawn_pending && !state.configuration_updated_since_spawn;
5047 if pending {
5048 state.coalesced_restart_pending = true;
5049 }
5050 pending
5051 });
5052 if satisfied_by_pending {
5053 debug!(module_id = %spec.module_id, "restart coalesced into the pending replacement");
5054 } else if let Some(generation) = satisfied_by_generation {
5055 info!(
5056 module_id = %spec.module_id,
5057 received_at_generation,
5058 "restart already satisfied by generation {generation}; not restarting again"
5059 );
5060 } else if initiated {
5061 let drain_timeout = drain_timeout_ms
5064 .map(Duration::from_millis)
5065 .unwrap_or(runtime.drain_timeout);
5066 if let Err(err) = restart_child(
5067 spec,
5068 runtime,
5069 registry,
5070 process_liveness,
5071 snapshot,
5072 child,
5073 drain_timeout,
5074 )
5075 .await
5076 {
5077 warn!(
5078 module_id = %spec.module_id,
5079 error = %err,
5080 "operator restart failed after initiation ack; module state carries the outcome"
5081 );
5082 let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5083 state.state = ModuleState::Failed;
5084 clear_current_process_facts(state);
5085 });
5086 }
5087 }
5088 true
5089 }
5090 SupervisorCommand::Reload { reply } => {
5091 let result =
5092 reload_child(spec, runtime, registry, process_liveness, snapshot, child).await;
5093 if result.is_ok()
5094 && runtime
5095 .scheduled_respawn
5096 .lock()
5097 .unwrap_or_else(|p| p.into_inner())
5098 .as_ref()
5099 .is_some_and(|pending| matches!(pending.kind, RespawnKind::Reload))
5100 {
5101 *runtime
5102 .deferred_reload_reply
5103 .lock()
5104 .unwrap_or_else(|p| p.into_inner()) = Some(reply);
5105 } else {
5106 let _ = reply.send(result);
5107 }
5108 true
5109 }
5110 SupervisorCommand::SetEnabled { enabled, reply } => {
5111 let result = set_child_enabled(
5112 spec,
5113 runtime,
5114 registry,
5115 process_liveness,
5116 snapshot,
5117 child,
5118 enabled,
5119 )
5120 .await;
5121 let _ = reply.send(result);
5122 true
5123 }
5124 SupervisorCommand::UpdateConfiguration {
5125 spec: next_spec,
5126 health,
5127 drain_timeout_ms,
5128 reply,
5129 } => {
5130 if let Some(handle) = &runtime.supervisor_handle {
5131 handle.apply_identity_configuration(&next_spec);
5132 }
5133 *spec = next_spec;
5134 let _ = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5135 state.configuration_updated_since_spawn = true;
5136 });
5137 runtime.health = health;
5138 runtime.drain_timeout = drain_timeout_ms
5139 .map(Duration::from_millis)
5140 .unwrap_or(runtime.default_drain_timeout);
5141 *runtime
5142 .effective_drain_timeout
5143 .lock()
5144 .unwrap_or_else(|poisoned| poisoned.into_inner()) = runtime.drain_timeout;
5145 let _ = reply.send(());
5146 true
5147 }
5148 SupervisorCommand::Swap {
5149 ready_timeout,
5150 reply,
5151 } => {
5152 let end = swap::run_swap(
5153 spec,
5154 runtime,
5155 registry,
5156 process_liveness,
5157 snapshot,
5158 child,
5159 commands,
5160 ready_timeout.unwrap_or(DEFAULT_SWAP_READY_TIMEOUT),
5161 reply,
5162 )
5163 .await;
5164 requeued.extend(end.requeue);
5165 true
5166 }
5167 }
5168}
5169
5170async fn restart_child(
5171 spec: &ModuleSpec,
5172 runtime: &SupervisorRuntimeConfig,
5173 registry: &Registry,
5174 process_liveness: &SupervisorProcessLiveness,
5175 snapshot: &SharedSnapshot,
5176 child: &mut Option<SupervisedChild>,
5177 drain_timeout: Duration,
5178) -> Result<(), SuperviseError> {
5179 if !lock_snapshot(snapshot)?.enabled {
5181 return Err(SuperviseError::Disabled {
5182 module_id: spec.module_id.clone(),
5183 });
5184 }
5185 let stop_notice = begin_forwarding_drain_with_timeout(
5186 spec,
5187 runtime,
5188 registry,
5189 snapshot,
5190 None,
5191 RouteCloseReason::Restart,
5192 drain_timeout,
5193 )
5194 .await?;
5195
5196 if child.is_some() {
5197 drain_optional_child(
5198 &spec.module_id,
5199 spec.protocol,
5200 stop_notice,
5201 registry,
5202 runtime.forwarding.as_deref(),
5203 snapshot,
5204 &runtime.terminal_ring,
5205 &runtime.spawn_events,
5206 child,
5207 drain_timeout,
5208 ModuleState::Restarting,
5209 Some(true),
5210 )
5211 .await?;
5212 } else {
5213 update_snapshot(snapshot, Some(&spec.module_id), |state| {
5214 state.enabled = true;
5215 state.state = ModuleState::Restarting;
5216 clear_current_process_facts(state);
5217 })?;
5218 release_dead_registration(
5219 registry,
5220 runtime.forwarding.as_deref(),
5221 snapshot,
5222 &spec.module_id,
5223 )
5224 .await?;
5225 }
5226
5227 reset_restart_count(snapshot, &spec.module_id)?;
5228 process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
5229 schedule_respawn(
5230 runtime,
5231 snapshot,
5232 &spec.module_id,
5233 runtime.restart_policy.backoff,
5234 RespawnKind::Spawn,
5235 )
5236}
5237
5238async fn reload_child(
5239 spec: &ModuleSpec,
5240 runtime: &SupervisorRuntimeConfig,
5241 registry: &Registry,
5242 process_liveness: &SupervisorProcessLiveness,
5243 snapshot: &SharedSnapshot,
5244 child: &mut Option<SupervisedChild>,
5245) -> Result<(), SuperviseError> {
5246 if !lock_snapshot(snapshot)?.enabled {
5248 return Err(SuperviseError::Disabled {
5249 module_id: spec.module_id.clone(),
5250 });
5251 }
5252 let stop_notice = begin_forwarding_drain(
5253 spec,
5254 runtime,
5255 registry,
5256 snapshot,
5257 Some(true),
5258 RouteCloseReason::Reload,
5259 )
5260 .await?;
5261
5262 if child.is_some() {
5263 drain_optional_child(
5264 &spec.module_id,
5265 spec.protocol,
5266 stop_notice,
5267 registry,
5268 runtime.forwarding.as_deref(),
5269 snapshot,
5270 &runtime.terminal_ring,
5271 &runtime.spawn_events,
5272 child,
5273 runtime.drain_timeout,
5274 ModuleState::Restarting,
5275 Some(true),
5276 )
5277 .await?;
5278 } else {
5279 update_snapshot(snapshot, Some(&spec.module_id), |state| {
5280 state.enabled = true;
5281 state.state = ModuleState::Restarting;
5282 clear_current_process_facts(state);
5283 })?;
5284 release_dead_registration(
5285 registry,
5286 runtime.forwarding.as_deref(),
5287 snapshot,
5288 &spec.module_id,
5289 )
5290 .await?;
5291 }
5292
5293 reset_restart_count(snapshot, &spec.module_id)?;
5294 process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
5295 schedule_respawn(
5296 runtime,
5297 snapshot,
5298 &spec.module_id,
5299 runtime.restart_policy.backoff,
5300 RespawnKind::Reload,
5301 )
5302}
5303
5304async fn finish_reload_child(
5305 spec: &ModuleSpec,
5306 runtime: &SupervisorRuntimeConfig,
5307 registry: &Registry,
5308 process_liveness: &SupervisorProcessLiveness,
5309 snapshot: &SharedSnapshot,
5310 child: &mut Option<SupervisedChild>,
5311) -> Result<(), SuperviseError> {
5312 process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
5313 let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
5314 Ok(next_child) => next_child,
5315 Err(err) => {
5316 return handle_reload_spawn_failure(
5317 spec,
5318 runtime,
5319 process_liveness,
5320 snapshot,
5321 child,
5322 format!("new child failed to spawn: {err}"),
5323 )
5324 .await;
5325 }
5326 };
5327 *child = Some(next_child);
5328
5329 let wait_outcome = {
5330 let active_child = child.as_mut().expect("new reload child was just stored");
5331 wait_for_registration_after_reload(
5332 registry,
5333 &spec.module_id,
5334 snapshot,
5335 active_child,
5336 REGISTRY_RELEASE_TIMEOUT,
5337 )
5338 .await?
5339 };
5340
5341 match wait_outcome {
5342 RegistrationWaitOutcome::Registered => {
5343 debug!(module_id = %spec.module_id, "supervised module reloaded and registered");
5344 Ok(())
5345 }
5346 RegistrationWaitOutcome::Exited(exit_report) => {
5347 if let Some(active_child) = child.as_mut() {
5348 active_child.drain_stderr(&spec.module_id).await;
5349 }
5350 let mut exited_child = child.take().expect("exited reload child is still stored");
5353 #[cfg(test)]
5354 if let Some(gate) = &runtime.test_reload_exit_record_gate {
5355 gate.reached.notify_one();
5356 gate.resume.notified().await;
5357 }
5358 let result = handle_reload_child_registration_failure(
5359 spec,
5360 runtime,
5361 registry,
5362 process_liveness,
5363 snapshot,
5364 child,
5365 ReloadRegistrationFailure {
5366 exit_report: registration_failure_exit_report(exit_report),
5367 reason: "new child exited before registering".to_string(),
5368 },
5369 )
5370 .await;
5371 exited_child.release_roster();
5372 result
5373 }
5374 RegistrationWaitOutcome::TimedOut => {
5375 let mut timed_out_child = child
5376 .take()
5377 .expect("timed-out reload child is still running");
5378 timed_out_child
5379 .start_kill()
5380 .map_err(|source| SuperviseError::Kill {
5381 module_id: spec.module_id.clone(),
5382 source,
5383 })?;
5384 let status = timed_out_child
5385 .wait()
5386 .await
5387 .map_err(|source| SuperviseError::Wait {
5388 module_id: spec.module_id.clone(),
5389 source,
5390 })?;
5391 timed_out_child.drain_stderr(&spec.module_id).await;
5392 handle_reload_child_registration_failure(
5393 spec,
5394 runtime,
5395 registry,
5396 process_liveness,
5397 snapshot,
5398 child,
5399 ReloadRegistrationFailure {
5400 exit_report: registration_failure_exit_report(classify_reaped_child_exit(
5401 snapshot,
5402 &timed_out_child,
5403 &status,
5404 )),
5405 reason: format!(
5406 "new child did not register within {:?}",
5407 REGISTRY_RELEASE_TIMEOUT
5408 ),
5409 },
5410 )
5411 .await
5412 }
5413 }
5414}
5415
5416async fn set_child_enabled(
5417 spec: &ModuleSpec,
5418 runtime: &SupervisorRuntimeConfig,
5419 registry: &Registry,
5420 process_liveness: &SupervisorProcessLiveness,
5421 snapshot: &SharedSnapshot,
5422 child: &mut Option<SupervisedChild>,
5423 enabled: bool,
5424) -> Result<bool, SuperviseError> {
5425 let (current_enabled, current_state, respawn_pending) = {
5426 let state = lock_snapshot(snapshot)?;
5427 (state.enabled, state.state, state.respawn_pending)
5428 };
5429 let revive_terminal = enabled
5437 && current_enabled
5438 && child.is_none()
5439 && (matches!(current_state, ModuleState::Failed | ModuleState::Stopped)
5440 || (current_state == ModuleState::Restarting && !respawn_pending));
5441 if current_enabled == enabled && !revive_terminal {
5442 return Ok(false);
5443 }
5444
5445 if enabled {
5446 update_snapshot(snapshot, Some(&spec.module_id), |state| {
5447 state.enabled = true;
5448 state.state = ModuleState::Starting;
5449 clear_current_process_facts(state);
5450 })?;
5451 #[cfg(test)]
5452 if runtime.test_seed_stale_facts_before_enable_spawn {
5453 update_snapshot(snapshot, Some(&spec.module_id), |state| {
5454 state.process_alive = true;
5455 state.pid = Some(41);
5456 state.spawned_at_ms = Some(42);
5457 state.spawned_from = Some(PathBuf::from("/spawned/module"));
5458 state.spawned_file_identity = Some(SpawnedFileIdentity {
5459 device: 43,
5460 inode: 44,
5461 });
5462 })?;
5463 }
5464 release_dead_registration(
5465 registry,
5466 runtime.forwarding.as_deref(),
5467 snapshot,
5468 &spec.module_id,
5469 )
5470 .await?;
5471 reset_restart_count(snapshot, &spec.module_id)?;
5472 process_liveness.track(spec.module_id.clone(), Arc::clone(snapshot));
5473 let next_child = match spawn_and_mark_running(spec, runtime, snapshot) {
5474 Ok(next_child) => next_child,
5475 Err(err) => {
5476 if let Err(state_err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5477 state.state = ModuleState::Failed;
5478 clear_current_process_facts(state);
5479 }) {
5480 error!(module_id = %spec.module_id, error = %state_err, "failed to record enable spawn failure");
5481 }
5482 process_liveness.untrack_if_current(&spec.module_id, snapshot);
5483 return Err(err);
5484 }
5485 };
5486 *child = Some(next_child);
5487 debug!(module_id = %spec.module_id, "supervised module enabled");
5488 Ok(true)
5489 } else {
5490 let stop_notice = begin_forwarding_drain_if_configured(
5491 spec,
5492 runtime,
5493 registry,
5494 snapshot,
5495 Some(false),
5496 RouteCloseReason::Disable,
5497 )
5498 .await?;
5499 drain_optional_child(
5500 &spec.module_id,
5501 spec.protocol,
5502 stop_notice,
5503 registry,
5504 runtime.forwarding.as_deref(),
5505 snapshot,
5506 &runtime.terminal_ring,
5507 &runtime.spawn_events,
5508 child,
5509 runtime.drain_timeout,
5510 ModuleState::Disabled,
5511 Some(false),
5512 )
5513 .await?;
5514 debug!(module_id = %spec.module_id, "supervised module disabled");
5515 Ok(true)
5516 }
5517}
5518
5519#[allow(clippy::too_many_arguments)]
5520async fn on_child_exit(
5521 spec: &ModuleSpec,
5522 policy: RestartPolicy,
5523 registry: &Registry,
5524 snapshot: &SharedSnapshot,
5525 terminal_ring: &Arc<Mutex<TerminalRing>>,
5526 spawn_events: &SpawnEventFeed,
5527 roster: &ChildRoster,
5528 exit_report: ExitReport,
5529) -> NextAction {
5530 if roster.is_closed() {
5536 return on_child_exit_during_daemon_shutdown(
5537 spec,
5538 registry,
5539 snapshot,
5540 terminal_ring,
5541 spawn_events,
5542 exit_report,
5543 )
5544 .await;
5545 }
5546 let unrequested_clean_exit_of_protocol_none =
5561 exit_report.kind == ExitKind::Clean && spec.protocol == ModuleProtocol::None;
5562 match exit_report.kind {
5563 ExitKind::Clean if !unrequested_clean_exit_of_protocol_none => {
5564 info!(
5565 module_id = %spec.module_id,
5566 exit_code = ?exit_report.code,
5567 exit_signal = ?exit_report.signal,
5568 "supervised module exited cleanly"
5569 );
5570 if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5571 state.state = ModuleState::Stopped;
5572 clear_current_process_facts(state);
5573 state.last_exit = Some(exit_report.clone());
5574 }) {
5575 error!(module_id = %spec.module_id, error = %err, "failed to record clean module exit");
5576 }
5577 record_terminal(
5578 &spec.module_id,
5579 terminal_ring,
5580 spawn_events,
5581 &exit_report,
5582 TerminalDisposition::Stopped,
5583 );
5584 let registration_released = match wait_for_registration_release(
5585 registry,
5586 &spec.module_id,
5587 REGISTRY_RELEASE_TIMEOUT,
5588 )
5589 .await
5590 {
5591 Ok(()) => true,
5592 Err(err) => {
5593 warn!(module_id = %spec.module_id, error = %err, "registration still active after clean exit");
5594 false
5595 }
5596 };
5597 NextAction::Stop {
5598 registration_released,
5599 }
5600 }
5601 ExitKind::Clean | ExitKind::Crash => {
5602 if unrequested_clean_exit_of_protocol_none {
5603 warn!(
5604 module_id = %spec.module_id,
5605 exit_code = ?exit_report.code,
5606 exit_signal = ?exit_report.signal,
5607 "protocol-none module exited cleanly without a stop request; handling it as a crash"
5608 );
5609 } else {
5610 warn!(
5611 module_id = %spec.module_id,
5612 exit_code = ?exit_report.code,
5613 exit_signal = ?exit_report.signal,
5614 "supervised module exited abnormally (crash)"
5615 );
5616 }
5617 let mut restart_schedule = None;
5618 let mut disposition = TerminalDisposition::Disabled;
5619 let mut disposition_detail = None;
5623 let now = Instant::now();
5624 if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5625 clear_current_process_facts(state);
5626 state.last_exit = Some(exit_report.clone());
5627 if state.enabled {
5628 if let Some(schedule) = state.next_crash_restart(&policy, now) {
5629 state.state = ModuleState::Restarting;
5630 restart_schedule = Some(schedule);
5631 disposition = TerminalDisposition::Restarting;
5632 } else {
5633 state.state = ModuleState::Failed;
5634 disposition = TerminalDisposition::Failed;
5635 disposition_detail = Some(policy.budget_exhausted_detail());
5636 }
5637 } else {
5638 state.state = ModuleState::Disabled;
5639 disposition = TerminalDisposition::Disabled;
5640 }
5641 }) {
5642 error!(module_id = %spec.module_id, error = %err, "failed to record crashed module exit");
5643 return NextAction::Stop {
5644 registration_released: false,
5645 };
5646 }
5647 if disposition_detail.is_some() {
5648 error!(
5653 module_id = %spec.module_id,
5654 max_restarts = policy.max_restarts,
5655 window_secs = policy.window.as_secs(),
5656 "module stopped: {}",
5657 policy.budget_exhausted_detail()
5658 );
5659 }
5660 record_terminal_with_detail(
5661 &spec.module_id,
5662 terminal_ring,
5663 spawn_events,
5664 &exit_report,
5665 disposition,
5666 disposition_detail,
5667 );
5668
5669 if let Some(schedule) = restart_schedule {
5670 NextAction::Restart {
5671 schedule: Some(schedule),
5672 }
5673 } else {
5674 let registration_released = match wait_for_registration_release(
5675 registry,
5676 &spec.module_id,
5677 REGISTRY_RELEASE_TIMEOUT,
5678 )
5679 .await
5680 {
5681 Ok(()) => true,
5682 Err(err) => {
5683 warn!(module_id = %spec.module_id, error = %err, "registration still active after failed module");
5684 false
5685 }
5686 };
5687 NextAction::Stop {
5688 registration_released,
5689 }
5690 }
5691 }
5692 ExitKind::DeliberateSeverance => {
5693 warn!(
5694 module_id = %spec.module_id,
5695 exit_code = ?exit_report.code,
5696 exit_signal = ?exit_report.signal,
5697 "supervised module exited after deliberate connection severance"
5698 );
5699 let mut should_restart = false;
5700 let mut disposition = TerminalDisposition::Disabled;
5701 if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5702 clear_current_process_facts(state);
5703 state.last_exit = Some(exit_report.clone());
5704 state.lifetime_restarts += 1;
5705 if state.enabled {
5706 state.state = ModuleState::Restarting;
5707 should_restart = true;
5708 disposition = TerminalDisposition::Restarting;
5709 } else {
5710 state.state = ModuleState::Disabled;
5711 }
5712 }) {
5713 error!(module_id = %spec.module_id, error = %err, "failed to record deliberately severed module exit");
5714 return NextAction::Stop {
5715 registration_released: false,
5716 };
5717 }
5718 record_terminal(
5719 &spec.module_id,
5720 terminal_ring,
5721 spawn_events,
5722 &exit_report,
5723 disposition,
5724 );
5725
5726 if should_restart {
5727 NextAction::Restart { schedule: None }
5728 } else {
5729 let registration_released = match wait_for_registration_release(
5730 registry,
5731 &spec.module_id,
5732 REGISTRY_RELEASE_TIMEOUT,
5733 )
5734 .await
5735 {
5736 Ok(()) => true,
5737 Err(err) => {
5738 warn!(module_id = %spec.module_id, error = %err, "registration still active after deliberately severed module exit");
5739 false
5740 }
5741 };
5742 NextAction::Stop {
5743 registration_released,
5744 }
5745 }
5746 }
5747 }
5748}
5749
5750async fn on_child_exit_during_daemon_shutdown(
5751 spec: &ModuleSpec,
5752 registry: &Registry,
5753 snapshot: &SharedSnapshot,
5754 terminal_ring: &Arc<Mutex<TerminalRing>>,
5755 spawn_events: &SpawnEventFeed,
5756 exit_report: ExitReport,
5757) -> NextAction {
5758 info!(
5759 module_id = %spec.module_id,
5760 exit_code = ?exit_report.code,
5761 exit_signal = ?exit_report.signal,
5762 exit_kind = ?exit_report.kind,
5763 "supervised module exited during daemon shutdown; not restarting it"
5764 );
5765 if let Err(err) = update_snapshot(snapshot, Some(&spec.module_id), |state| {
5766 state.state = ModuleState::Stopped;
5767 clear_current_process_facts(state);
5768 state.last_exit = Some(exit_report.clone());
5769 }) {
5770 error!(module_id = %spec.module_id, error = %err, "failed to record module exit during daemon shutdown");
5771 }
5772 record_terminal(
5773 &spec.module_id,
5774 terminal_ring,
5775 spawn_events,
5776 &exit_report,
5777 TerminalDisposition::DaemonShutdown,
5778 );
5779 let registration_released =
5780 wait_for_registration_release(registry, &spec.module_id, REGISTRY_RELEASE_TIMEOUT)
5781 .await
5782 .is_ok();
5783 NextAction::Stop {
5784 registration_released,
5785 }
5786}
5787
5788fn record_wait_error_terminal(
5789 module_id: &str,
5790 terminal_ring: &Arc<Mutex<TerminalRing>>,
5791 spawn_events: &SpawnEventFeed,
5792) {
5793 record_terminal(
5794 module_id,
5795 terminal_ring,
5796 spawn_events,
5797 &wait_error_exit_report(),
5798 TerminalDisposition::Failed,
5799 );
5800}
5801
5802fn record_terminal(
5803 module_id: &str,
5804 terminal_ring: &Arc<Mutex<TerminalRing>>,
5805 spawn_events: &SpawnEventFeed,
5806 exit_report: &ExitReport,
5807 disposition: TerminalDisposition,
5808) {
5809 record_terminal_with_detail(
5810 module_id,
5811 terminal_ring,
5812 spawn_events,
5813 exit_report,
5814 disposition,
5815 None,
5816 );
5817}
5818
5819fn durable_terminal_history_of(
5823 terminal_ring: &Mutex<TerminalRing>,
5824 module_id: &str,
5825) -> subc_control::TerminalHistory {
5826 let read = terminal_ring
5827 .lock()
5828 .unwrap_or_else(|p| p.into_inner())
5829 .capture_durable_history();
5830 read.read(module_id)
5831}
5832
5833fn record_terminal_with_detail(
5834 module_id: &str,
5835 terminal_ring: &Arc<Mutex<TerminalRing>>,
5836 spawn_events: &SpawnEventFeed,
5837 exit_report: &ExitReport,
5838 disposition: TerminalDisposition,
5839 disposition_detail: Option<String>,
5840) {
5841 spawn_events.emit_exited(module_id, exit_report.code, exit_report.signal);
5842 let record = TerminalRecord {
5843 exit_code: exit_report.code,
5844 exit_signal: exit_report.signal,
5845 at_ms: exit_report.at_ms,
5846 disposition,
5847 exit_kind: exit_report.kind.into(),
5848 disposition_detail,
5849 };
5850 terminal_ring
5851 .lock()
5852 .unwrap_or_else(|poisoned| poisoned.into_inner())
5853 .record_exit(module_id, record);
5854}
5855
5856fn untrack_if_registration_released(
5857 process_liveness: &SupervisorProcessLiveness,
5858 registry: &Registry,
5859 module_id: &str,
5860 snapshot: &SharedSnapshot,
5861) {
5862 match registry.get_module(module_id) {
5863 Ok(None) => process_liveness.untrack_if_current(module_id, snapshot),
5864 Ok(Some(_)) => {}
5865 Err(err) => {
5866 warn!(module_id, error = %err, "could not determine whether supervisor liveness can be untracked");
5867 }
5868 }
5869}
5870
5871#[cfg(test)]
5885fn apply_wire_spawn_args(
5886 command: &mut Command,
5887 spec: &ModuleSpec,
5888 connection_file_path: Option<&std::path::Path>,
5889 handle: Option<&SupervisorHandle>,
5890) -> Result<Option<NonceHandoff>, SuperviseError> {
5891 apply_wire_spawn_args_for_role(
5892 command,
5893 spec,
5894 connection_file_path,
5895 handle,
5896 SpawnRole::Plain,
5897 )
5898}
5899
5900#[cfg(unix)]
5905type NonceHandoff = subc_os::LaunchNonceHandoff;
5906#[cfg(not(unix))]
5907type NonceHandoff = std::convert::Infallible;
5908
5909fn apply_wire_spawn_args_for_role(
5926 command: &mut Command,
5927 spec: &ModuleSpec,
5928 connection_file_path: Option<&std::path::Path>,
5929 handle: Option<&SupervisorHandle>,
5930 role: SpawnRole,
5931) -> Result<Option<NonceHandoff>, SuperviseError> {
5932 command.env(SUBC_MODULE_ID_ENV, &spec.module_id);
5933 command.env_remove(subc_os::LAUNCH_NONCE_FD_ENV);
5938 command.env_remove(SUBC_LAUNCH_NONCE_ENV);
5940 if spec.protocol == ModuleProtocol::None {
5941 return Ok(None);
5942 }
5943 if let Some(connection_file_path) = connection_file_path {
5944 command.arg(SUBC_ARG).arg(connection_file_path);
5945 }
5946
5947 let nonce = generate_launch_nonce()?;
5951 if let Some(handle) = handle {
5952 match role {
5953 SpawnRole::Plain => {
5954 handle.set_spawn_nonce(&spec.module_id, nonce.clone());
5955 if spec.reserved {
5956 handle.set_reserved_nonce(&spec.module_id, nonce.clone());
5957 }
5958 }
5959 SpawnRole::SwapCandidate => handle.open_swap(&spec.module_id, nonce.clone()),
5960 }
5961 }
5962 #[cfg(unix)]
5963 let handoff = {
5964 let handoff =
5965 subc_os::LaunchNonceHandoff::new(&nonce).map_err(|source| SuperviseError::Spawn {
5966 program: spec.program.clone(),
5967 source,
5968 cgroup_path: None,
5969 })?;
5970 command.env(subc_os::LAUNCH_NONCE_FD_ENV, handoff.fd_env_value());
5971 Some(handoff)
5972 };
5973 #[cfg(not(unix))]
5974 let handoff = None;
5975 #[cfg(not(unix))]
5978 command.env(SUBC_LAUNCH_NONCE_ENV, nonce);
5979 Ok(handoff)
5980}
5981
5982fn apply_child_env(command: &mut Command, spec: &ModuleSpec) {
5983 command.env_remove(CK_LOG_ENV);
5984 command.env_remove(SUBC_SPAWN_ROLE_ENV);
5991 for (key, value) in &spec.env {
5992 if matches!(
5996 key.as_str(),
5997 CAPTURE_MAX_FILE_MB_ENV | CAPTURE_KEEP_ENV | CAPTURE_MAX_AGE_DAYS_ENV
5998 ) || key == SUBC_SPAWN_ROLE_ENV
5999 {
6000 continue;
6001 }
6002 command.env(key, value);
6003 }
6004}
6005
6006#[derive(Debug, Clone, Copy, PartialEq, Eq)]
6009enum SpawnRole {
6010 Plain,
6011 SwapCandidate,
6012}
6013
6014fn apply_spawn_role(command: &mut Command, role: SpawnRole) {
6017 if role == SpawnRole::SwapCandidate {
6018 command.env(SUBC_SPAWN_ROLE_ENV, SPAWN_ROLE_SWAP_CANDIDATE);
6019 }
6020}
6021
6022fn spawn_child(
6023 spec: &ModuleSpec,
6024 connection_file_path: Option<&std::path::Path>,
6025 handle: Option<&SupervisorHandle>,
6026 ring: &Arc<Mutex<StderrRing>>,
6027 capture_logs_dir: Option<&std::path::Path>,
6028 roster: &ChildRoster,
6029 #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
6030) -> Result<SupervisedChild, SuperviseError> {
6031 spawn_child_in_slot(
6032 spec,
6033 connection_file_path,
6034 handle,
6035 ring,
6036 capture_logs_dir,
6037 roster,
6038 #[cfg(target_os = "linux")]
6039 cgroup_placement,
6040 SpawnRole::Plain,
6041 false,
6042 )
6043}
6044
6045#[allow(clippy::too_many_arguments)]
6058fn spawn_child_in_slot(
6059 spec: &ModuleSpec,
6060 connection_file_path: Option<&std::path::Path>,
6061 handle: Option<&SupervisorHandle>,
6062 ring: &Arc<Mutex<StderrRing>>,
6063 capture_logs_dir: Option<&std::path::Path>,
6064 roster: &ChildRoster,
6065 #[cfg(target_os = "linux")] cgroup_placement: Option<&subc_cgroup::Placement>,
6066 role: SpawnRole,
6067 alternate_slot: bool,
6068) -> Result<SupervisedChild, SuperviseError> {
6069 if roster.is_closed() {
6070 return Err(SuperviseError::Spawn {
6071 program: spec.program.clone(),
6072 source: io::Error::other("the daemon is shutting down; not starting a new process"),
6073 cgroup_path: None,
6074 });
6075 }
6076 #[cfg(target_os = "linux")]
6077 let cgroup_name = swap::cgroup_name(&spec.module_id, alternate_slot);
6078 #[cfg(not(target_os = "linux"))]
6079 let _ = alternate_slot;
6080 let mut command = Command::new(&spec.program);
6081 command.args(&spec.args);
6082 apply_child_env(&mut command, spec);
6112 apply_spawn_role(&mut command, role);
6113 let nonce_handoff =
6114 apply_wire_spawn_args_for_role(&mut command, spec, connection_file_path, handle, role)?;
6115
6116 #[cfg(target_os = "linux")]
6117 let cgroup_path = cgroup_placement
6118 .map(|placement| placement.module_path(&cgroup_name))
6119 .transpose()
6120 .map_err(|source| SuperviseError::Cgroup {
6121 module_id: spec.module_id.clone(),
6122 source,
6123 })?;
6124 #[cfg(not(target_os = "linux"))]
6125 let cgroup_path: Option<PathBuf> = None;
6126 #[cfg(target_os = "linux")]
6127 if let Some(path) = &cgroup_path {
6128 if let Err(error) = apply_cgroup_placement(&mut command, spec, path) {
6129 if let Some(placement) = cgroup_placement {
6130 remove_module_cgroup(placement, &cgroup_name);
6131 }
6132 return Err(error);
6133 }
6134 }
6135
6136 let output_sink = if let Some(logs_dir) = capture_logs_dir {
6137 let path = logs_dir.join(format!("{}.stderr.log", spec.module_id));
6138 match ChildOutputSink::open(&path, capture_retention(spec)) {
6139 Ok(sink) => sink,
6140 Err(error) => {
6141 warn!(
6142 module_id = %spec.module_id,
6143 path = %path.display(),
6144 error = %error,
6145 "could not open child output capture file; forwarding to stderr"
6146 );
6147 ChildOutputSink::Stderr
6148 }
6149 }
6150 } else {
6151 ChildOutputSink::Stderr
6152 };
6153
6154 command.stdout(Stdio::piped());
6155 command.stderr(Stdio::piped());
6156 command.kill_on_drop(true);
6157 #[cfg(unix)]
6174 command.process_group(0);
6175 command.stdin(Stdio::null());
6176 #[cfg(unix)]
6180 if let Some(handoff) = nonce_handoff {
6181 handoff.install_last(command.as_std_mut());
6182 }
6183 #[cfg(not(unix))]
6184 let _ = nonce_handoff;
6185
6186 #[cfg(windows)]
6191 subc_jobobject::suspend_on_create_async(&mut command);
6192 let mut child = match command.spawn() {
6193 Ok(child) => child,
6194 Err(source) => {
6195 #[cfg(target_os = "linux")]
6196 if let Some(placement) = cgroup_placement {
6197 remove_module_cgroup(placement, &cgroup_name);
6198 }
6199 return Err(SuperviseError::Spawn {
6200 program: spec.program.clone(),
6201 source,
6202 cgroup_path,
6203 });
6204 }
6205 };
6206
6207 #[cfg(windows)]
6209 let job = contain_spawned_child(&child, spec)?;
6210 let spawned_at_ms = unix_ms_now();
6211 let spawned_from = spec.program.clone();
6212 let spawned_file_identity = spawned_file_identity(&spawned_from);
6213 let pid = child.id().ok_or_else(|| SuperviseError::Spawn {
6214 program: spec.program.clone(),
6215 source: io::Error::other("spawned child exposed no live pid"),
6216 cgroup_path: cgroup_path.clone(),
6217 })?;
6218 let process_start_time = crate::provenance::process_start_time(pid);
6219 let process_identity = process_start_time.map(|start_time| ProcessIdentity { pid, start_time });
6220 #[cfg(target_os = "linux")]
6224 let recorded_cgroup_name = cgroup_path.as_ref().map(|_| cgroup_name.clone());
6225 #[cfg(not(target_os = "linux"))]
6226 let recorded_cgroup_name = None;
6227 let roster_guard = roster.admit(
6228 spec.module_id.clone(),
6229 pid,
6230 spec.protocol,
6231 process_start_time,
6232 crate::child_roster::RecordedIdentity {
6233 start_time: subc_os::start_time(pid),
6234 executable: spawned_file_identity.map(|identity| {
6235 crate::live_children::ExecutableIdentity {
6236 device: identity.device,
6237 inode: identity.inode,
6238 }
6239 }),
6240 cgroup_name: recorded_cgroup_name,
6241 #[cfg(target_os = "linux")]
6242 cgroup_placement: cgroup_placement.cloned(),
6243 },
6244 );
6245 if roster.is_closed() {
6254 #[cfg(target_os = "linux")]
6256 kill_module_cgroup(cgroup_placement, &cgroup_name);
6257 if let Err(error) = child.start_kill() {
6258 debug!(module_id = %spec.module_id, pid, %error, "kill of a process spawned during daemon shutdown failed; it may already have exited");
6259 }
6260 drop(roster_guard);
6261 return Err(SuperviseError::Spawn {
6262 program: spec.program.clone(),
6263 source: io::Error::other(
6264 "the daemon began shutting down while this process was starting; ended it",
6265 ),
6266 cgroup_path,
6267 });
6268 }
6269
6270 let stdout_pump = match child.stdout.take() {
6271 Some(stdout) => Some(tokio::spawn(pump_stdout_to(stdout, output_sink.clone()))),
6272 None => {
6273 warn!(
6274 module_id = %spec.module_id,
6275 "spawned child exposed no stdout pipe; file capture will be incomplete"
6276 );
6277 None
6278 }
6279 };
6280 let stderr_pump = match child.stderr.take() {
6281 Some(stderr) => {
6282 let generation = ring
6283 .lock()
6284 .unwrap_or_else(|poisoned| poisoned.into_inner())
6285 .begin_process();
6286 Some(StderrPump {
6287 task: tokio::spawn(pump_stderr_to(
6288 stderr,
6289 Arc::clone(ring),
6290 generation,
6291 output_sink,
6292 )),
6293 generation,
6294 })
6295 }
6296 None => {
6297 ring.lock()
6301 .unwrap_or_else(|poisoned| poisoned.into_inner())
6302 .mark_not_captured("stderr pipe was not available on spawn");
6303 warn!(
6304 module_id = %spec.module_id,
6305 "spawned child exposed no stderr pipe; tail will be unavailable"
6306 );
6307 None
6308 }
6309 };
6310
6311 Ok(SupervisedChild {
6312 child,
6313 #[cfg(target_os = "linux")]
6314 module_id: cgroup_name,
6315 #[cfg(target_os = "linux")]
6316 cgroup_placement: cgroup_placement.cloned(),
6317 #[cfg(windows)]
6318 job,
6319 stdout_pump,
6320 stderr_pump,
6321 stderr_ring: Arc::clone(ring),
6322 spawned_at_ms,
6323 spawned_from,
6324 spawned_file_identity,
6325 process_start_time,
6326 process_identity,
6327 pid,
6328 roster_guard: Some(roster_guard),
6329 })
6330}
6331
6332#[cfg(target_os = "linux")]
6333pub(crate) fn kill_module_cgroup(placement: Option<&subc_cgroup::Placement>, module_id: &str) {
6334 use subc_cgroup::KillOutcome;
6335 match subc_cgroup::kill_module(placement, module_id) {
6336 KillOutcome::Killed => {}
6337 KillOutcome::NotPlaced | KillOutcome::Unsupported => {
6338 debug!(
6339 module_id,
6340 "cgroup tree kill unavailable; using direct-child kill"
6341 );
6342 }
6343 KillOutcome::IoError { path, error } => {
6344 warn!(module_id, path = %path.display(), %error, "cgroup tree kill failed; using direct-child kill");
6345 }
6346 }
6347}
6348
6349#[cfg(windows)]
6363fn contain_spawned_child(
6364 child: &Child,
6365 spec: &ModuleSpec,
6366) -> Result<Option<subc_jobobject::JobObject>, SuperviseError> {
6367 let module_id = spec.module_id.as_str();
6368 let Some(pid) = child.id() else {
6369 warn!(
6372 module_id,
6373 "spawned child had already exited before containment; no job object attached"
6374 );
6375 return Ok(None);
6376 };
6377
6378 let job = match subc_jobobject::JobObject::new() {
6379 Ok(job) => job,
6380 Err(source) => {
6381 warn!(
6382 module_id,
6383 error = %source,
6384 "could not create a job object; this module's helper processes will not be \
6385 reaped on teardown"
6386 );
6387 resume_suspended_child(pid, spec)?;
6390 return Ok(None);
6391 }
6392 };
6393
6394 if let Err(source) = job.assign(child) {
6395 warn!(
6396 module_id,
6397 error = %source,
6398 "could not assign the child to its job object; this module's helper processes \
6399 will not be reaped on teardown"
6400 );
6401 resume_suspended_child(pid, spec)?;
6402 return Ok(None);
6403 }
6404
6405 resume_suspended_child(pid, spec)?;
6406 Ok(Some(job))
6407}
6408
6409#[cfg(windows)]
6414fn resume_suspended_child(pid: u32, spec: &ModuleSpec) -> Result<(), SuperviseError> {
6415 if let Err(source) = subc_jobobject::resume_main_thread(pid) {
6416 let _ = std::process::Command::new("taskkill.exe")
6420 .args(["/PID", &pid.to_string(), "/T", "/F"])
6421 .stdin(Stdio::null())
6422 .stdout(Stdio::null())
6423 .stderr(Stdio::null())
6424 .status();
6425 return Err(SuperviseError::Spawn {
6426 program: spec.program.clone(),
6427 source,
6428 cgroup_path: None,
6429 });
6430 }
6431 Ok(())
6432}
6433
6434#[cfg(target_os = "linux")]
6435fn remove_module_cgroup(placement: &subc_cgroup::Placement, module_id: &str) {
6436 match placement.remove_module(module_id) {
6437 Ok(()) => debug!(module_id, "removed module cgroup after process exit"),
6438 Err(error) => warn!(
6439 module_id,
6440 error = %error,
6441 "could not remove module cgroup after process exit; continuing teardown"
6442 ),
6443 }
6444}
6445
6446#[cfg(target_os = "linux")]
6447fn apply_cgroup_placement(
6448 command: &mut Command,
6449 spec: &ModuleSpec,
6450 path: &std::path::Path,
6451) -> Result<(), SuperviseError> {
6452 subc_cgroup::apply(command, path).map_err(|source| SuperviseError::Cgroup {
6453 module_id: spec.module_id.clone(),
6454 source,
6455 })
6456}
6457
6458fn capture_retention(spec: &ModuleSpec) -> Retention {
6459 let defaults = Retention::default();
6460 let value = |name: &str| {
6461 spec.env
6462 .iter()
6463 .rev()
6464 .find_map(|(key, value)| (key == name).then_some(value.as_str()))
6465 };
6466 Retention {
6467 max_file_mb: value(CAPTURE_MAX_FILE_MB_ENV)
6468 .and_then(|value| value.parse().ok())
6469 .unwrap_or(defaults.max_file_mb),
6470 keep: value(CAPTURE_KEEP_ENV)
6471 .and_then(|value| value.parse().ok())
6472 .unwrap_or(defaults.keep),
6473 max_age_days: value(CAPTURE_MAX_AGE_DAYS_ENV)
6474 .and_then(|value| value.parse().ok())
6475 .unwrap_or(defaults.max_age_days),
6476 }
6477}
6478
6479fn generate_launch_nonce() -> Result<String, SuperviseError> {
6482 let mut bytes = [0u8; 32];
6483 getrandom::getrandom(&mut bytes).map_err(|source| SuperviseError::LaunchNonce {
6484 reason: source.to_string(),
6485 })?;
6486 let mut hex = String::with_capacity(64);
6487 for b in bytes {
6488 use std::fmt::Write;
6489 let _ = write!(hex, "{b:02x}");
6490 }
6491 Ok(hex)
6492}
6493
6494fn constant_time_eq(a: &[u8], b: &[u8]) -> bool {
6497 if a.len() != b.len() {
6498 return false;
6499 }
6500 let mut diff = 0u8;
6501 for (x, y) in a.iter().zip(b.iter()) {
6502 diff |= x ^ y;
6503 }
6504 diff == 0
6505}
6506
6507fn spawn_and_mark_running(
6508 spec: &ModuleSpec,
6509 runtime: &SupervisorRuntimeConfig,
6510 snapshot: &SharedSnapshot,
6511) -> Result<SupervisedChild, SuperviseError> {
6512 let child = spawn_child(
6513 spec,
6514 runtime.connection_file_path.as_deref(),
6515 runtime.supervisor_handle.as_ref(),
6516 &runtime.stderr_ring,
6517 runtime.capture_logs_dir.as_deref(),
6518 &runtime.child_roster,
6519 #[cfg(target_os = "linux")]
6520 runtime.cgroup_placement.as_ref(),
6521 )?;
6522 set_running(snapshot, &child, &spec.module_id, &runtime.spawn_events)?;
6523 Ok(child)
6524}
6525
6526enum RegistrationWaitOutcome {
6527 Registered,
6528 Exited(ExitReport),
6529 TimedOut,
6530}
6531
6532struct ReloadRegistrationFailure {
6533 exit_report: ExitReport,
6534 reason: String,
6535}
6536
6537#[derive(Debug, Clone, Copy, PartialEq, Eq)]
6538enum BusyGaugeObservation {
6539 Quiescent,
6540 Busy,
6541 Omitted,
6542}
6543
6544fn busy_gauge_observation(metrics: Option<&Value>, gauges: &[String]) -> BusyGaugeObservation {
6545 let Some(metrics) = metrics.and_then(Value::as_object) else {
6546 return BusyGaugeObservation::Omitted;
6547 };
6548 let mut sum = 0u128;
6549 for gauge in gauges {
6550 let Some(value) = metrics.get(gauge) else {
6551 return BusyGaugeObservation::Omitted;
6552 };
6553 let Some(value) = value.as_u64() else {
6554 return BusyGaugeObservation::Busy;
6555 };
6556 sum = sum.saturating_add(u128::from(value));
6557 }
6558 if sum == 0 {
6559 BusyGaugeObservation::Quiescent
6560 } else {
6561 BusyGaugeObservation::Busy
6562 }
6563}
6564
6565fn declared_busy_gauges(
6566 registry: &Registry,
6567 module_id: &str,
6568) -> Result<Vec<String>, SuperviseError> {
6569 busy_gauges_of(
6570 registry
6571 .get_module(module_id)
6572 .map_err(SuperviseError::Registry)?,
6573 )
6574}
6575
6576fn declared_busy_gauges_for_connection(
6580 registry: &Registry,
6581 connection_id: ConnectionId,
6582) -> Result<Vec<String>, SuperviseError> {
6583 busy_gauges_of(
6584 registry
6585 .get_module_by_connection(connection_id)
6586 .map_err(SuperviseError::Registry)?,
6587 )
6588}
6589
6590fn busy_gauges_of(
6591 registration: Option<crate::registry::ModuleRegistration>,
6592) -> Result<Vec<String>, SuperviseError> {
6593 let Some(registration) = registration else {
6594 return Ok(Vec::new());
6595 };
6596 let Some(self_signals) = registration.manifest.self_signals else {
6597 return Ok(Vec::new());
6598 };
6599
6600 let mut gauges = Vec::new();
6601 for declaration in self_signals {
6602 if declaration.kind != SelfSignalKind::Busy {
6603 continue;
6604 }
6605 match declaration.anchored_to {
6606 SignalAnchor::HealthGauges { gauges: declared } if !declared.is_empty() => {
6607 gauges.extend(declared)
6608 }
6609 _ => {
6610 gauges.push(String::new());
6613 }
6614 }
6615 }
6616 Ok(gauges)
6617}
6618
6619async fn wait_for_forwarding_quiescence(
6624 forwarding: &ForwardingTable,
6625 module_id: &str,
6626 runtime: &SupervisorRuntimeConfig,
6627 endpoint: crate::ModuleEndpointId,
6628 deadline: Instant,
6629 busy_gauges: &[String],
6630 scope: DrainScope,
6631) -> Result<bool, SuperviseError> {
6632 let mut gauges_quiescent = busy_gauges.is_empty();
6633 let mut next_probe_at = Instant::now();
6634 let mut omission_counted = false;
6635
6636 loop {
6637 let now = Instant::now();
6638 if !busy_gauges.is_empty() && now >= next_probe_at && now < deadline {
6639 let report = match scope {
6640 DrainScope::Active => probe_module_health(module_id, runtime, Some(deadline)).await,
6641 DrainScope::Endpoint(endpoint) => {
6642 probe_endpoint_health(endpoint, runtime, Some(deadline)).await
6643 }
6644 };
6645 gauges_quiescent = match report {
6646 Ok(report) => match busy_gauge_observation(report.metrics.as_ref(), busy_gauges) {
6647 BusyGaugeObservation::Quiescent => true,
6648 BusyGaugeObservation::Busy => false,
6649 BusyGaugeObservation::Omitted => {
6650 if !omission_counted {
6651 forwarding
6652 .counters()
6653 .increment_drains_with_undeclared_gauge();
6654 omission_counted = true;
6655 }
6656 false
6657 }
6658 },
6659 Err(err) => {
6660 warn!(
6661 module_id,
6662 error = %err,
6663 "drain health.check did not produce declared busy gauges; treating module as busy"
6664 );
6665 false
6666 }
6667 };
6668 next_probe_at = Instant::now() + runtime.health.cadence.max(REGISTRY_RELEASE_POLL);
6669 }
6670
6671 let in_flight = forwarding
6672 .endpoint_in_flight_count(endpoint)
6673 .map_err(SuperviseError::Forwarding)?;
6674 if in_flight == 0 && gauges_quiescent {
6675 return Ok(true);
6676 }
6677
6678 let now = Instant::now();
6679 if now >= deadline {
6680 return Ok(false);
6681 }
6682 let mut wait = deadline
6683 .saturating_duration_since(now)
6684 .min(REGISTRY_RELEASE_POLL);
6685 if !busy_gauges.is_empty() {
6686 wait = wait.min(next_probe_at.saturating_duration_since(now));
6687 }
6688 sleep(wait).await;
6689 }
6690}
6691
6692fn drained_after_quiescence_wait(wait_result: &Result<bool, SuperviseError>) -> bool {
6700 match wait_result {
6701 Ok(drained) => *drained,
6702 Err(_) => false,
6703 }
6704}
6705
6706fn send_route_goodbyes(forwarding: &ForwardingTable, released_routes: Vec<GoodbyeTarget>) {
6707 for released in released_routes {
6708 let frame = match Frame::build_with_version(
6709 released.negotiated_ver,
6710 FrameType::Goodbye,
6711 control_flags(),
6712 released.channel,
6713 released.epoch,
6714 0,
6715 Vec::new(),
6716 ) {
6717 Ok(frame) => frame,
6718 Err(err) => {
6719 warn!(
6720 route_channel = released.channel,
6721 error = %err,
6722 "failed to build supervisor drain route GOODBYE frame"
6723 );
6724 continue;
6725 }
6726 };
6727 if !released.close_on_delivery_failure() {
6728 crate::forwarding::send_module_route_goodbye(
6729 &forwarding.counters(),
6730 &released.sink,
6731 frame,
6732 released.module_id.as_deref(),
6733 "supervisor drain",
6734 );
6735 continue;
6736 }
6737 if let Err(err) = released.sink.try_send(frame) {
6738 warn!(
6739 target_connection_id = released.connection_id.get(),
6740 route_channel = released.channel,
6741 error = %err,
6742 "supervisor drain route GOODBYE was not delivered to client; closing target connection"
6743 );
6744 let _ = forwarding.escalate_client_delivery_failure(
6745 released.connection_id,
6746 released.channel,
6747 released.epoch,
6748 CloseReason::new(
6749 "route_goodbye_delivery_failed",
6750 format!(
6751 "failed to enqueue supervisor drain route GOODBYE for channel {}: {err}",
6752 released.channel
6753 ),
6754 ),
6755 crate::forwarding::UndeliveredFrame {
6756 module_id: released.module_id.as_deref(),
6757 sink: &released.sink,
6758 },
6759 );
6760 }
6761 }
6762}
6763
6764fn send_module_draining(
6765 module_id: &str,
6766 reason: RouteCloseReason,
6767 deadline_ms: u64,
6768 target: &ModuleDrainTarget,
6769) {
6770 let body = match serde_json::to_vec(&ModuleControlCommand::Draining {
6771 reason,
6772 deadline_ms,
6773 }) {
6774 Ok(body) => body,
6775 Err(err) => {
6776 warn!(
6777 module_id,
6778 error = %err,
6779 "failed to encode module draining command"
6780 );
6781 return;
6782 }
6783 };
6784 let frame = match Frame::build_with_version(
6785 target.negotiated_ver,
6786 FrameType::Push,
6787 control_flags(),
6788 0,
6789 0,
6790 0,
6791 body,
6792 ) {
6793 Ok(frame) => frame,
6794 Err(err) => {
6795 warn!(
6796 module_id,
6797 error = %err,
6798 "failed to build module draining command frame"
6799 );
6800 return;
6801 }
6802 };
6803 if let Err(err) = target.sink.try_send(frame) {
6804 warn!(
6805 module_id,
6806 target_connection_id = target.endpoint.connection_id.get(),
6807 error = %err,
6808 "module draining command was not delivered to peer"
6809 );
6810 }
6811}
6812
6813fn module_goodbye_frame(module_id: &str, negotiated_ver: u8) -> Option<Frame> {
6815 match Frame::build_with_version(
6816 negotiated_ver,
6817 FrameType::Goodbye,
6818 control_flags(),
6819 0,
6820 0,
6821 0,
6822 Vec::new(),
6823 ) {
6824 Ok(frame) => Some(frame),
6825 Err(err) => {
6826 warn!(
6827 module_id,
6828 error = %err,
6829 "failed to build module GOODBYE frame"
6830 );
6831 None
6832 }
6833 }
6834}
6835
6836#[cfg(unix)]
6850async fn send_module_goodbyes_for_daemon_shutdown(
6851 forwarding: &Arc<ForwardingTable>,
6852 reason: &CloseReason,
6853 wait_for_flush: bool,
6854) {
6855 const GOODBYE_BUDGET: Duration = Duration::from_millis(500);
6856 let targets = match forwarding.module_connections() {
6857 Ok(targets) => targets,
6858 Err(err) => {
6859 warn!(error = %err, "could not list module connections for shutdown GOODBYE");
6860 return;
6861 }
6862 };
6863 let deadline = Instant::now() + GOODBYE_BUDGET;
6864 let mut sends = tokio::task::JoinSet::new();
6865 for target in targets {
6866 let Some(frame) = module_goodbye_frame(&target.module_id, target.negotiated_ver) else {
6867 continue;
6868 };
6869 if !wait_for_flush {
6870 if let Err(err) = target.sink.try_send(frame) {
6871 debug!(
6872 module_id = %target.module_id,
6873 error = %err,
6874 "shutdown module GOODBYE was not queued"
6875 );
6876 }
6877 continue;
6878 }
6879 let forwarding = Arc::clone(forwarding);
6880 let reason = reason.clone();
6881 sends.spawn(async move {
6882 match timeout_at(deadline, target.sink.send_flushed(frame)).await {
6883 Ok(Ok(())) => {}
6884 Ok(Err(err)) => debug!(
6885 module_id = %target.module_id,
6886 error = %err,
6887 "module connection closed before its shutdown GOODBYE was written"
6888 ),
6889 Err(_) => warn!(
6890 module_id = %target.module_id,
6891 budget = ?GOODBYE_BUDGET,
6892 "shutdown module GOODBYE was not written within its budget; closing anyway"
6893 ),
6894 }
6895 forwarding.request_connection_close(target.endpoint.connection_id, reason);
6896 });
6897 }
6898 while sends.join_next().await.is_some() {}
6900}
6901
6902fn send_module_goodbye(module_id: &str, forwarding: &ForwardingTable, target: &ModuleDrainTarget) {
6903 let Some(frame) = module_goodbye_frame(module_id, target.negotiated_ver) else {
6904 return;
6905 };
6906 if let Err(err) = target.sink.try_send(frame) {
6907 warn!(
6908 module_id,
6909 target_connection_id = target.endpoint.connection_id.get(),
6910 error = %err,
6911 "supervisor drain module GOODBYE was not delivered to peer; closing module connection"
6912 );
6913 forwarding.request_connection_close(
6914 target.endpoint.connection_id,
6915 CloseReason::new(
6916 "module_goodbye_delivery_failed",
6917 format!("failed to enqueue supervisor drain module GOODBYE for module '{module_id}': {err}"),
6918 ),
6919 );
6920 }
6921}
6922
6923#[derive(Clone, Copy)]
6924struct ForwardingDrainContext<'a> {
6925 spec: &'a ModuleSpec,
6926 runtime: &'a SupervisorRuntimeConfig,
6927 registry: &'a Registry,
6928 scope: DrainScope,
6929}
6930
6931#[derive(Debug, Clone, Copy, PartialEq, Eq)]
6933enum DrainScope {
6934 Active,
6937 Endpoint(crate::ModuleEndpointId),
6942}
6943
6944#[derive(Debug, Clone, Copy, PartialEq, Eq)]
6952enum StopNotice {
6953 SentOverConnection,
6956 NoConnection,
6960 NotSent,
6964}
6965
6966async fn begin_forwarding_drain(
6967 spec: &ModuleSpec,
6968 runtime: &SupervisorRuntimeConfig,
6969 registry: &Registry,
6970 snapshot: &SharedSnapshot,
6971 enabled: Option<bool>,
6972 reason: RouteCloseReason,
6973) -> Result<StopNotice, SuperviseError> {
6974 let Some(forwarding) = runtime.forwarding.as_ref() else {
6975 return Err(SuperviseError::ReloadUnavailable {
6976 module_id: spec.module_id.clone(),
6977 reason: "supervisor was not configured with a forwarding table".to_string(),
6978 });
6979 };
6980
6981 begin_forwarding_drain_with(
6982 forwarding,
6983 ForwardingDrainContext {
6984 spec,
6985 runtime,
6986 registry,
6987 scope: DrainScope::Active,
6988 },
6989 snapshot,
6990 enabled,
6991 reason,
6992 runtime.drain_timeout,
6993 )
6994 .await
6995}
6996
6997async fn begin_forwarding_drain_if_configured(
6998 spec: &ModuleSpec,
6999 runtime: &SupervisorRuntimeConfig,
7000 registry: &Registry,
7001 snapshot: &SharedSnapshot,
7002 enabled: Option<bool>,
7003 reason: RouteCloseReason,
7004) -> Result<StopNotice, SuperviseError> {
7005 begin_forwarding_drain_with_timeout(
7006 spec,
7007 runtime,
7008 registry,
7009 snapshot,
7010 enabled,
7011 reason,
7012 runtime.drain_timeout,
7013 )
7014 .await
7015}
7016
7017async fn begin_forwarding_drain_with_timeout(
7021 spec: &ModuleSpec,
7022 runtime: &SupervisorRuntimeConfig,
7023 registry: &Registry,
7024 snapshot: &SharedSnapshot,
7025 enabled: Option<bool>,
7026 reason: RouteCloseReason,
7027 drain_timeout: Duration,
7028) -> Result<StopNotice, SuperviseError> {
7029 let Some(forwarding) = runtime.forwarding.as_ref() else {
7030 return Ok(StopNotice::NotSent);
7031 };
7032
7033 begin_forwarding_drain_with(
7034 forwarding,
7035 ForwardingDrainContext {
7036 spec,
7037 runtime,
7038 registry,
7039 scope: DrainScope::Active,
7040 },
7041 snapshot,
7042 enabled,
7043 reason,
7044 drain_timeout,
7045 )
7046 .await
7047}
7048
7049async fn begin_forwarding_drain_with(
7050 forwarding: &ForwardingTable,
7051 context: ForwardingDrainContext<'_>,
7052 snapshot: &SharedSnapshot,
7053 enabled: Option<bool>,
7054 reason: RouteCloseReason,
7055 drain_timeout: Duration,
7056) -> Result<StopNotice, SuperviseError> {
7057 let ForwardingDrainContext {
7058 spec,
7059 runtime,
7060 registry,
7061 scope,
7062 } = context;
7063 debug_assert_ne!(reason, RouteCloseReason::Crash);
7064 let terminal = matches!(reason, RouteCloseReason::Disable);
7065 let drain_started_at = Instant::now();
7066 let drain_deadline = drain_started_at + drain_timeout;
7067 let deadline_ms =
7068 unix_ms_now().saturating_add(u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX));
7069 let busy_gauges = match scope {
7070 DrainScope::Active => declared_busy_gauges(registry, &spec.module_id)?,
7071 DrainScope::Endpoint(endpoint) => {
7072 declared_busy_gauges_for_connection(registry, endpoint.connection_id)?
7073 }
7074 };
7075
7076 let gate_started = Instant::now();
7079 let drain_target = match scope {
7080 DrainScope::Active => forwarding.begin_module_drain(&spec.module_id, reason),
7081 DrainScope::Endpoint(endpoint) => forwarding.begin_endpoint_drain(endpoint, reason),
7082 }
7083 .map_err(SuperviseError::Forwarding)?;
7084 info!(
7089 module_id = %spec.module_id,
7090 ?reason,
7091 gate_ms = u64::try_from(gate_started.elapsed().as_millis()).unwrap_or(u64::MAX),
7092 connected = drain_target.is_some(),
7093 "module drain began; route admission closed"
7094 );
7095 if scope == DrainScope::Active {
7096 update_snapshot(snapshot, Some(&spec.module_id), |state| {
7097 state.state = ModuleState::Draining;
7098 state.draining_to_replace =
7099 matches!(reason, RouteCloseReason::Restart | RouteCloseReason::Reload);
7100 if let Some(enabled) = enabled {
7101 state.enabled = enabled;
7102 }
7103 })?;
7104 }
7105
7106 let Some(target) = drain_target.as_ref() else {
7107 return Ok(StopNotice::NoConnection);
7111 };
7112 {
7113 send_module_draining(&spec.module_id, reason, deadline_ms, target);
7114 let routes = forwarding
7115 .endpoint_routes(target.endpoint)
7116 .map_err(SuperviseError::Forwarding)?;
7117 let routes_notified = routes.len();
7118 crate::control::send_route_control_pushes(
7119 forwarding,
7120 routes.clone(),
7121 ClientControlPush::RouteClosing {
7122 module_id: spec.module_id.clone(),
7123 channels: Vec::new(),
7124 reason,
7125 },
7126 );
7127 send_route_goodbyes(forwarding, target.abandoned_bindings.clone());
7128
7129 let wait_result = wait_for_forwarding_quiescence(
7135 forwarding,
7136 &spec.module_id,
7137 runtime,
7138 target.endpoint,
7139 drain_deadline,
7140 &busy_gauges,
7141 scope,
7142 )
7143 .await;
7144 let drained = drained_after_quiescence_wait(&wait_result);
7145 if let Err(err) = &wait_result {
7146 error!(
7147 module_id = %spec.module_id,
7148 ?reason,
7149 error = %err,
7150 "forwarding quiescence wait failed after route.closing; forcing route.closed(drained: false) so the client is not left waiting on an unfulfilled promise"
7151 );
7152 } else if !drained {
7153 let holdouts = forwarding
7159 .endpoint_drain_holdouts(target.endpoint)
7160 .unwrap_or_default();
7161 warn!(
7162 module_id = %spec.module_id,
7163 waited = ?drain_timeout,
7164 ?reason,
7165 held_requests = holdouts.requests,
7166 held_routes = holdouts.routes,
7167 total_routes = holdouts.total_routes,
7168 top_connections = ?holdouts.top_connections,
7169 held = %holdouts
7172 .held
7173 .iter()
7174 .map(|(channel, corr)| format!("{channel}:{corr}"))
7175 .collect::<Vec<_>>()
7176 .join(","),
7177 "route drain timed out before request quiescence; forcing teardown"
7178 );
7179 }
7180 crate::control::send_route_control_pushes(
7181 forwarding,
7182 routes,
7183 ClientControlPush::RouteClosed {
7184 module_id: spec.module_id.clone(),
7185 channels: Vec::new(),
7186 reason,
7187 drained,
7188 abandoned: target.abandoned_bindings.len() as u32,
7189 excluded_subscriptions: target.excluded_subscriptions,
7190 terminal: Some(terminal),
7191 },
7192 );
7193 wait_result?;
7194
7195 let released_routes = match forwarding.release_module_endpoint_routes(target.endpoint) {
7201 Ok(routes) => routes,
7202 Err(err) => {
7203 warn!(
7204 module_id = %spec.module_id,
7205 ?reason,
7206 error = %err,
7207 "failed to release module endpoint routes after route.closed; module GOODBYE will still be sent"
7208 );
7209 send_module_goodbye(&spec.module_id, forwarding, target);
7210 return Err(SuperviseError::Forwarding(err));
7211 }
7212 };
7213 let route_goodbye_count = released_routes.len();
7214 send_route_goodbyes(forwarding, released_routes);
7215 send_module_goodbye(&spec.module_id, forwarding, target);
7216
7217 info!(
7223 module_id = %spec.module_id,
7224 ?reason,
7225 routes_notified,
7226 route_goodbyes = route_goodbye_count,
7227 abandoned_reservations = target.abandoned_bindings.len(),
7228 excluded_subscriptions = target.excluded_subscriptions,
7229 drained,
7230 "module drain complete; consumers notified via route.closing/route.closed pushes and per-route GOODBYE frames"
7231 );
7232 }
7233
7234 Ok(StopNotice::SentOverConnection)
7235}
7236
7237async fn wait_for_registration_after_reload(
7240 registry: &Registry,
7241 module_id: &str,
7242 snapshot: &SharedSnapshot,
7243 child: &mut SupervisedChild,
7244 wait: Duration,
7245) -> Result<RegistrationWaitOutcome, SuperviseError> {
7246 wait_for_slot_registration(
7247 registry,
7248 crate::registry::RegistrationSlot::Active(module_id),
7249 module_id,
7250 snapshot,
7251 child,
7252 wait,
7253 )
7254 .await
7255}
7256
7257async fn wait_for_slot_registration(
7265 registry: &Registry,
7266 slot: crate::registry::RegistrationSlot<'_>,
7267 module_id: &str,
7268 snapshot: &SharedSnapshot,
7269 child: &mut SupervisedChild,
7270 wait: Duration,
7271) -> Result<RegistrationWaitOutcome, SuperviseError> {
7272 let deadline = Instant::now() + wait;
7273 loop {
7274 if registry
7275 .registration(slot)
7276 .map_err(SuperviseError::Registry)?
7277 .is_some()
7278 {
7279 return Ok(RegistrationWaitOutcome::Registered);
7280 }
7281
7282 let now = Instant::now();
7283 if now >= deadline {
7284 return Ok(RegistrationWaitOutcome::TimedOut);
7285 }
7286 let remaining = deadline.saturating_duration_since(now);
7287 let poll = remaining.min(REGISTRY_RELEASE_POLL);
7288
7289 tokio::select! {
7290 wait_result = child.wait() => {
7291 let status = wait_result.map_err(|source| SuperviseError::Wait {
7292 module_id: module_id.to_string(),
7293 source,
7294 })?;
7295 return Ok(RegistrationWaitOutcome::Exited(classify_reaped_child_exit(
7296 snapshot,
7297 child,
7298 &status,
7299 )));
7300 }
7301 _ = sleep(poll) => {}
7302 }
7303 }
7304}
7305
7306fn registration_failure_exit_report(mut exit_report: ExitReport) -> ExitReport {
7307 if exit_report.kind != ExitKind::DeliberateSeverance {
7310 exit_report.kind = ExitKind::Crash;
7311 }
7312 exit_report
7313}
7314
7315async fn handle_reload_child_registration_failure(
7316 spec: &ModuleSpec,
7317 runtime: &SupervisorRuntimeConfig,
7318 registry: &Registry,
7319 process_liveness: &SupervisorProcessLiveness,
7320 snapshot: &SharedSnapshot,
7321 _child: &mut Option<SupervisedChild>,
7322 failure: ReloadRegistrationFailure,
7323) -> Result<(), SuperviseError> {
7324 let ReloadRegistrationFailure {
7325 exit_report,
7326 reason,
7327 } = failure;
7328 match on_child_exit(
7329 spec,
7330 runtime.restart_policy,
7331 registry,
7332 snapshot,
7333 &runtime.terminal_ring,
7334 &runtime.spawn_events,
7335 &runtime.child_roster,
7336 exit_report,
7337 )
7338 .await
7339 {
7340 NextAction::Stop {
7341 registration_released,
7342 } => {
7343 if registration_released {
7344 process_liveness.untrack_if_current(&spec.module_id, snapshot);
7345 }
7346 }
7347 NextAction::Restart { schedule } => {
7348 let delay = schedule.map_or(runtime.restart_policy.delay_for_restart(0), |schedule| {
7349 schedule.delay
7350 });
7351 if let Some(schedule) = schedule {
7352 log_crash_respawn(&spec.module_id, schedule);
7353 }
7354 schedule_respawn(
7355 runtime,
7356 snapshot,
7357 &spec.module_id,
7358 delay,
7359 RespawnKind::Spawn,
7360 )?;
7361 }
7362 }
7363 Err(SuperviseError::ReloadFailed {
7364 module_id: spec.module_id.clone(),
7365 reason,
7366 })
7367}
7368
7369async fn handle_reload_spawn_failure(
7370 spec: &ModuleSpec,
7371 runtime: &SupervisorRuntimeConfig,
7372 process_liveness: &SupervisorProcessLiveness,
7373 snapshot: &SharedSnapshot,
7374 _child: &mut Option<SupervisedChild>,
7375 reason: String,
7376) -> Result<(), SuperviseError> {
7377 let now = Instant::now();
7378 let mut schedule = None;
7379 update_snapshot(snapshot, Some(&spec.module_id), |state| {
7380 clear_current_process_facts(state);
7381 if state.enabled {
7382 schedule = state.next_crash_restart(&runtime.restart_policy, now);
7383 state.state = if schedule.is_some() {
7384 ModuleState::Restarting
7385 } else {
7386 ModuleState::Failed
7387 };
7388 } else {
7389 state.state = ModuleState::Disabled;
7390 }
7391 })?;
7392 if let Some(schedule) = schedule {
7393 schedule_respawn(
7394 runtime,
7395 snapshot,
7396 &spec.module_id,
7397 schedule.delay,
7398 RespawnKind::Spawn,
7399 )?;
7400 } else {
7401 process_liveness.untrack_if_current(&spec.module_id, snapshot);
7402 }
7403 Err(SuperviseError::ReloadFailed {
7404 module_id: spec.module_id.clone(),
7405 reason,
7406 })
7407}
7408
7409fn control_flags() -> Flags {
7410 Flags::new(false, Priority::Passive, false)
7411}
7412
7413#[allow(clippy::too_many_arguments)]
7414async fn drain_optional_child(
7415 module_id: &str,
7416 protocol: ModuleProtocol,
7417 stop_notice: StopNotice,
7418 registry: &Registry,
7419 forwarding: Option<&ForwardingTable>,
7420 snapshot: &SharedSnapshot,
7421 terminal_ring: &Arc<Mutex<TerminalRing>>,
7422 spawn_events: &SpawnEventFeed,
7423 child: &mut Option<SupervisedChild>,
7424 drain_timeout: Duration,
7425 final_state: ModuleState,
7426 enabled: Option<bool>,
7427) -> Result<(), SuperviseError> {
7428 if let Some(child) = child.take() {
7429 drain_child_to_state(
7430 module_id,
7431 protocol,
7432 stop_notice,
7433 registry,
7434 forwarding,
7435 snapshot,
7436 terminal_ring,
7437 spawn_events,
7438 child,
7439 drain_timeout,
7440 final_state,
7441 enabled,
7442 )
7443 .await
7444 } else {
7445 update_snapshot(snapshot, Some(module_id), |state| {
7446 state.state = final_state;
7447 if let Some(enabled) = enabled {
7448 state.enabled = enabled;
7449 }
7450 clear_current_process_facts(state);
7451 })?;
7452 release_dead_registration(registry, forwarding, snapshot, module_id).await
7453 }
7454}
7455
7456#[allow(clippy::too_many_arguments)]
7457async fn drain_child_to_state(
7458 module_id: &str,
7459 protocol: ModuleProtocol,
7460 stop_notice: StopNotice,
7461 registry: &Registry,
7462 forwarding: Option<&ForwardingTable>,
7463 snapshot: &SharedSnapshot,
7464 terminal_ring: &Arc<Mutex<TerminalRing>>,
7465 spawn_events: &SpawnEventFeed,
7466 mut child: SupervisedChild,
7467 drain_timeout: Duration,
7468 final_state: ModuleState,
7469 enabled: Option<bool>,
7470) -> Result<(), SuperviseError> {
7471 update_snapshot(snapshot, Some(module_id), |state| {
7472 state.state = ModuleState::Draining;
7473 state.draining_to_replace = final_state == ModuleState::Restarting;
7474 if let Some(enabled) = enabled {
7475 state.enabled = enabled;
7476 }
7477 })?;
7478
7479 if stop_notice != StopNotice::SentOverConnection {
7490 if protocol == ModuleProtocol::Subc && stop_notice == StopNotice::NoConnection {
7491 info!(
7492 module_id,
7493 pid = child.pid,
7494 budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
7495 "module has no connection yet; requesting stop by signal"
7496 );
7497 }
7498 request_graceful_stop(module_id, &child);
7499 }
7500
7501 let exit_report = match timeout(drain_timeout, child.wait()).await {
7502 Ok(Ok(status)) => classify_reaped_child_exit(snapshot, &child, &status),
7503 Ok(Err(source)) => {
7504 fail_snapshot(snapshot, Some(module_id), None);
7505 return Err(SuperviseError::Wait {
7506 module_id: module_id.to_string(),
7507 source,
7508 });
7509 }
7510 Err(_) => {
7511 warn!(
7524 module_id,
7525 pid = child.pid,
7526 budget_ms = u64::try_from(drain_timeout.as_millis()).unwrap_or(u64::MAX),
7527 reason = ?final_state,
7528 ?stop_notice,
7529 "drain budget expired before the module exited; killing it"
7530 );
7531 child.start_kill().map_err(|source| {
7532 fail_snapshot(snapshot, Some(module_id), None);
7533 SuperviseError::Kill {
7534 module_id: module_id.to_string(),
7535 source,
7536 }
7537 })?;
7538 let status = child.wait().await.map_err(|source| {
7539 fail_snapshot(snapshot, Some(module_id), None);
7540 SuperviseError::Wait {
7541 module_id: module_id.to_string(),
7542 source,
7543 }
7544 })?;
7545 classify_reaped_child_exit(snapshot, &child, &status)
7546 }
7547 };
7548
7549 update_snapshot(snapshot, Some(module_id), |state| {
7550 state.state = final_state;
7551 if let Some(enabled) = enabled {
7552 state.enabled = enabled;
7553 }
7554 clear_current_process_facts(state);
7555 state.last_exit = Some(exit_report.clone());
7556 if exit_report.kind == ExitKind::DeliberateSeverance {
7557 state.lifetime_restarts += 1;
7558 }
7559 })?;
7560 let detail = lock_snapshot(snapshot)?.drain_disposition_detail.take();
7561 record_terminal_with_detail(
7562 module_id,
7563 terminal_ring,
7564 spawn_events,
7565 &exit_report,
7566 terminal_disposition(final_state),
7567 detail,
7568 );
7569 child.drain_stderr(module_id).await;
7570
7571 release_dead_registration(registry, forwarding, snapshot, module_id).await
7572}
7573
7574#[cfg(unix)]
7594fn request_graceful_stop(module_id: &str, child: &SupervisedChild) {
7595 let Some(pid) = child
7596 .id()
7597 .and_then(|pid| i32::try_from(pid).ok())
7598 .and_then(rustix::process::Pid::from_raw)
7599 else {
7600 debug!(
7601 module_id,
7602 "no pid to signal for teardown; falling through to the drain wait"
7603 );
7604 return;
7605 };
7606 match rustix::process::kill_process(pid, rustix::process::Signal::TERM) {
7607 Ok(()) => debug!(
7608 module_id,
7609 "sent SIGTERM to a module nothing else asked to stop"
7610 ),
7611 Err(err) => debug!(
7612 module_id,
7613 error = %err,
7614 "SIGTERM to module failed; the drain wait and kill still apply"
7615 ),
7616 }
7617}
7618
7619#[cfg(not(unix))]
7627fn request_graceful_stop(module_id: &str, _child: &SupervisedChild) {
7628 debug!(
7629 module_id,
7630 "no graceful stop signal exists on this platform; teardown of a module nothing asked to stop waits, then kills"
7631 );
7632}
7633
7634fn terminal_disposition(final_state: ModuleState) -> TerminalDisposition {
7635 match final_state {
7636 ModuleState::Stopped => TerminalDisposition::Stopped,
7637 ModuleState::Disabled => TerminalDisposition::Disabled,
7638 ModuleState::Restarting => TerminalDisposition::Restarting,
7639 ModuleState::Failed => TerminalDisposition::Failed,
7640 ModuleState::Starting
7641 | ModuleState::Running
7642 | ModuleState::Unresponsive
7643 | ModuleState::Draining => {
7644 unreachable!("terminal exits only finish in terminal or restarting states")
7645 }
7646 }
7647}
7648
7649async fn release_dead_registration(
7657 registry: &Registry,
7658 forwarding: Option<&ForwardingTable>,
7659 snapshot: &SharedSnapshot,
7660 module_id: &str,
7661) -> Result<(), SuperviseError> {
7662 let result = async {
7663 let registration = registry
7664 .get_module(module_id)
7665 .map_err(SuperviseError::Registry)?;
7666 match wait_for_registration_release(registry, module_id, REGISTRY_RELEASE_TIMEOUT).await {
7667 Ok(()) => return Ok(()),
7668 Err(SuperviseError::RegistrationStillActive { .. }) => {}
7669 Err(err) => return Err(err),
7670 }
7671 let pid = lock_snapshot(snapshot)?.reaped_pid;
7672 if let (Some(registration), Some(forwarding), Some(pid)) = (registration, forwarding, pid) {
7673 warn!(
7674 module_id,
7675 pid,
7676 connection_id = registration.connection_id.get(),
7677 "reaped module registration outlived release grace; closing dead connection"
7678 );
7679 forwarding.request_connection_close(
7680 registration.connection_id,
7681 CloseReason::new(
7682 "supervised_process_reaped",
7683 format!("module '{module_id}' pid {pid} exited"),
7684 ),
7685 );
7686 wait_for_slot_registration_release(
7687 registry,
7688 crate::registry::RegistrationSlot::Connection(registration.connection_id),
7689 REGISTRY_RELEASE_TIMEOUT,
7690 )
7691 .await?;
7692 }
7693 wait_for_registration_release(registry, module_id, Duration::ZERO).await
7694 }
7695 .await;
7696 if let Err(err) = &result {
7697 fail_snapshot(snapshot, Some(module_id), None);
7698 error!(module_id, error = %err, "registration release failed after child exit; module is failed and start can retry");
7699 }
7700 result
7701}
7702
7703async fn wait_for_registration_release(
7706 registry: &Registry,
7707 module_id: &str,
7708 wait: Duration,
7709) -> Result<(), SuperviseError> {
7710 wait_for_slot_registration_release(
7711 registry,
7712 crate::registry::RegistrationSlot::Active(module_id),
7713 wait,
7714 )
7715 .await
7716}
7717
7718async fn wait_for_slot_registration_release(
7726 registry: &Registry,
7727 slot: crate::registry::RegistrationSlot<'_>,
7728 wait: Duration,
7729) -> Result<(), SuperviseError> {
7730 let deadline = Instant::now() + wait;
7731 let mut release_events = registration_release_events().subscribe();
7732 let still_active = |registration: &crate::registry::ModuleRegistration| {
7733 SuperviseError::RegistrationStillActive {
7734 module_id: registration.manifest.module_id.clone(),
7735 waited: wait,
7736 }
7737 };
7738 loop {
7739 let _observed_generation = *release_events.borrow_and_update();
7740 let Some(registration) = registry
7741 .registration(slot)
7742 .map_err(SuperviseError::Registry)?
7743 else {
7744 return Ok(());
7745 };
7746
7747 let now = Instant::now();
7748 if now >= deadline {
7749 return Err(still_active(®istration));
7750 }
7751
7752 let remaining = deadline.saturating_duration_since(now);
7753 match timeout(remaining, release_events.changed()).await {
7754 Ok(Ok(())) | Ok(Err(_)) => {}
7755 Err(_) => return Err(still_active(®istration)),
7756 }
7757 }
7758}
7759
7760#[cfg(test)]
7761mod slot_registration_wait_tests {
7762 use super::*;
7763 use crate::registry::{ConnectionId, RegistrationSlot};
7764 use subc_protocol::manifest::ModuleManifest;
7765
7766 #[tokio::test]
7767 async fn enable_release_failure_is_failed_and_a_second_enable_retries() {
7768 let registry = Arc::new(Registry::default());
7769 let supervisor = Supervisor::new(Arc::clone(®istry), RestartPolicy::default());
7770 let runtime = supervisor.runtime_config();
7771 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::disabled()));
7772 let spec = ModuleSpec {
7773 module_id: "enable-stale-registration".to_string(),
7774 program: PathBuf::from("/missing/enable-retry-test"),
7775 args: Vec::new(),
7776 env: Vec::new(),
7777 reserved: false,
7778 reserved_prefixes: Vec::new(),
7779 protocol: ModuleProtocol::Subc,
7780 overlap: Default::default(),
7781 };
7782 let connection = ConnectionId::new(90);
7783 registry
7784 .register_with_control_ops(
7785 ModuleManifest::builder(&spec.module_id, "0.1.0").build(),
7786 1,
7787 connection,
7788 Vec::new(),
7789 )
7790 .unwrap();
7791 let mut child = None;
7792 let err = set_child_enabled(
7793 &spec,
7794 &runtime,
7795 ®istry,
7796 &supervisor.process_liveness,
7797 &snapshot,
7798 &mut child,
7799 true,
7800 )
7801 .await
7802 .unwrap_err();
7803 assert!(matches!(
7804 err,
7805 SuperviseError::RegistrationStillActive { .. }
7806 ));
7807 assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
7808 assert!(child.is_none());
7809 registry.deregister_connection(connection).unwrap();
7810 let err = set_child_enabled(
7811 &spec,
7812 &runtime,
7813 ®istry,
7814 &supervisor.process_liveness,
7815 &snapshot,
7816 &mut child,
7817 true,
7818 )
7819 .await
7820 .unwrap_err();
7821 assert!(
7822 matches!(err, SuperviseError::Spawn { .. }),
7823 "second enable must attempt a spawn: {err}"
7824 );
7825 assert_eq!(lock_snapshot(&snapshot).unwrap().state, ModuleState::Failed);
7826 }
7827
7828 const INCUMBENT: u64 = 1;
7829 const CANDIDATE: u64 = 2;
7830
7831 fn swapped_registry() -> Arc<Registry> {
7832 let registry = Arc::new(Registry::default());
7833 let manifest = ModuleManifest::builder("m", "0.1.0").build();
7834 registry
7835 .register_with_control_ops(
7836 manifest.clone(),
7837 1,
7838 ConnectionId::new(INCUMBENT),
7839 Vec::new(),
7840 )
7841 .unwrap();
7842 registry
7843 .register_candidate_with_control_ops(
7844 manifest,
7845 1,
7846 ConnectionId::new(CANDIDATE),
7847 Vec::new(),
7848 )
7849 .unwrap();
7850 registry
7851 }
7852
7853 #[tokio::test]
7857 async fn incumbent_release_is_awaited_by_connection_not_by_module_id() {
7858 let registry = swapped_registry();
7859 registry.promote_candidate("m").unwrap().unwrap();
7860
7861 assert!(matches!(
7862 wait_for_registration_release(®istry, "m", Duration::from_millis(50)).await,
7863 Err(SuperviseError::RegistrationStillActive { .. })
7864 ));
7865
7866 assert!(matches!(
7868 wait_for_slot_registration_release(
7869 ®istry,
7870 RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
7871 Duration::from_millis(50),
7872 )
7873 .await,
7874 Err(SuperviseError::RegistrationStillActive { .. })
7875 ));
7876
7877 let releaser = Arc::clone(®istry);
7878 let release = tokio::spawn(async move {
7879 sleep(Duration::from_millis(20)).await;
7880 releaser
7881 .deregister_connection(ConnectionId::new(INCUMBENT))
7882 .unwrap();
7883 notify_registration_release();
7884 });
7885 wait_for_slot_registration_release(
7886 ®istry,
7887 RegistrationSlot::Connection(ConnectionId::new(INCUMBENT)),
7888 Duration::from_secs(5),
7889 )
7890 .await
7891 .expect("the incumbent's own registration is released");
7892 release.await.unwrap();
7893 assert!(registry.get_module("m").unwrap().is_some());
7894 }
7895
7896 #[tokio::test]
7899 async fn candidate_slot_wait_ignores_the_incumbents_registration() {
7900 let registry = swapped_registry();
7901 assert!(matches!(
7902 wait_for_slot_registration_release(
7903 ®istry,
7904 RegistrationSlot::Candidate("m"),
7905 Duration::from_millis(50),
7906 )
7907 .await,
7908 Err(SuperviseError::RegistrationStillActive { .. })
7909 ));
7910 registry
7911 .deregister_connection(ConnectionId::new(CANDIDATE))
7912 .unwrap();
7913 wait_for_slot_registration_release(
7914 ®istry,
7915 RegistrationSlot::Candidate("m"),
7916 Duration::from_millis(50),
7917 )
7918 .await
7919 .expect("a candidate slot with no candidate is released");
7920 assert!(registry
7921 .registration(RegistrationSlot::Active("m"))
7922 .unwrap()
7923 .is_some());
7924 }
7925}
7926
7927fn classify_exit(status: &ExitStatus) -> ExitReport {
7928 ExitReport {
7929 kind: if status.success() {
7930 ExitKind::Clean
7931 } else {
7932 ExitKind::Crash
7933 },
7934 code: status.code(),
7935 signal: exit_signal(status),
7936 at_ms: unix_ms_now(),
7937 }
7938}
7939
7940fn wait_error_exit_report() -> ExitReport {
7946 ExitReport {
7947 kind: ExitKind::Crash,
7948 code: None,
7949 signal: None,
7950 at_ms: unix_ms_now(),
7951 }
7952}
7953
7954#[cfg(unix)]
7955fn exit_signal(status: &ExitStatus) -> Option<i32> {
7956 use std::os::unix::process::ExitStatusExt;
7957
7958 status.signal()
7959}
7960
7961#[cfg(not(unix))]
7962fn exit_signal(_status: &ExitStatus) -> Option<i32> {
7963 None
7964}
7965
7966fn reset_restart_count(snapshot: &SharedSnapshot, module_id: &str) -> Result<(), SuperviseError> {
7972 update_snapshot(snapshot, Some(module_id), |state| {
7973 state.clear_crash_restarts();
7974 })
7975}
7976
7977fn set_running(
7978 snapshot: &SharedSnapshot,
7979 child: &SupervisedChild,
7980 module_id: &str,
7981 spawn_events: &SpawnEventFeed,
7982) -> Result<(), SuperviseError> {
7983 let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
7984 module_id: Some(module_id.to_string()),
7985 })?;
7986 state.spawn_generation = spawn_events.emit_spawned(module_id, child.pid, child.spawned_at_ms);
7987 if std::mem::take(&mut state.coalesced_restart_pending) {
7988 let generation = state.spawn_generation;
7989 info!(module_id, "restart already satisfied by generation {generation}; coalesced pending request completed");
7990 }
7991 state.drain_disposition_detail = None;
7992 state.in_alternate_slot = false;
7995 state.configuration_updated_since_spawn = false;
7996 state.state = ModuleState::Running;
7997 state.enabled = true;
7998 state.process_alive = true;
7999 state.pid = child.id();
8000 state.spawned_at_ms = Some(child.spawned_at_ms);
8001 state.spawned_from = Some(child.spawned_from.clone());
8002 state.spawned_file_identity = child.spawned_file_identity;
8003 state.process_start_time = child.process_start_time;
8004 Ok(())
8005}
8006
8007fn clear_current_process_facts(state: &mut SupervisorSnapshot) {
8008 state.process_alive = false;
8009 state.pid = None;
8010 state.spawned_at_ms = None;
8011 state.spawned_from = None;
8012 state.spawned_file_identity = None;
8013 state.process_start_time = None;
8014 state.deliberate_severance = None;
8015}
8016
8017#[cfg(test)]
8018fn record_deliberate_severance(
8019 snapshot: &SharedSnapshot,
8020 identity: ProcessIdentity,
8021) -> Result<(), SuperviseError> {
8022 update_snapshot(snapshot, None, |state| {
8023 state.deliberate_severance = Some(identity);
8024 })
8025}
8026
8027fn apply_deliberate_severance_marker(
8028 snapshot: &SharedSnapshot,
8029 exited_identity: Option<ProcessIdentity>,
8030 mut exit_report: ExitReport,
8031) -> ExitReport {
8032 let marker = lock_snapshot(snapshot)
8033 .ok()
8034 .and_then(|mut state| state.deliberate_severance.take());
8035 if marker.is_some() && marker == exited_identity {
8036 exit_report.kind = ExitKind::DeliberateSeverance;
8037 }
8038 exit_report
8039}
8040
8041fn classify_reaped_child_exit(
8042 snapshot: &SharedSnapshot,
8043 child: &SupervisedChild,
8044 status: &ExitStatus,
8045) -> ExitReport {
8046 let _ = update_snapshot(snapshot, None, |state| state.reaped_pid = Some(child.pid));
8047 apply_deliberate_severance_marker(snapshot, child.process_identity(), classify_exit(status))
8048}
8049
8050fn fail_snapshot(
8051 snapshot: &SharedSnapshot,
8052 module_id: Option<&str>,
8053 last_exit: Option<ExitReport>,
8054) {
8055 if let Err(err) = update_snapshot(snapshot, module_id, |state| {
8056 state.state = ModuleState::Failed;
8057 clear_current_process_facts(state);
8058 if let Some(last_exit) = last_exit {
8059 state.last_exit = Some(last_exit);
8060 }
8061 }) {
8062 error!(error = %err, "failed to mark supervisor state failed");
8063 }
8064}
8065
8066fn update_snapshot(
8067 snapshot: &SharedSnapshot,
8068 module_id: Option<&str>,
8069 update: impl FnOnce(&mut SupervisorSnapshot),
8070) -> Result<(), SuperviseError> {
8071 let mut state = snapshot.lock().map_err(|_| SuperviseError::StatePoisoned {
8072 module_id: module_id.map(ToOwned::to_owned),
8073 })?;
8074 update(&mut state);
8075 Ok(())
8076}
8077
8078const SLOW_SNAPSHOT_LOCK_THRESHOLD: Duration = Duration::from_millis(250);
8079
8080fn lock_snapshot_for_control<'a>(
8081 snapshot: &'a SharedSnapshot,
8082 module_id: &str,
8083 caller: &'static str,
8084) -> Result<std::sync::MutexGuard<'a, SupervisorSnapshot>, SuperviseError> {
8085 let started_at = Instant::now();
8086 let guard = lock_snapshot(snapshot)?;
8087 let waited = started_at.elapsed();
8088 if waited >= SLOW_SNAPSHOT_LOCK_THRESHOLD {
8089 warn!(
8090 module_id = %module_id,
8091 waited_ms = waited.as_millis() as u64,
8092 caller = %caller,
8093 "slow snapshot lock"
8094 );
8095 }
8096 Ok(guard)
8097}
8098
8099fn lock_snapshot(
8100 snapshot: &SharedSnapshot,
8101) -> Result<std::sync::MutexGuard<'_, SupervisorSnapshot>, SuperviseError> {
8102 snapshot
8103 .lock()
8104 .map_err(|_| SuperviseError::StatePoisoned { module_id: None })
8105}
8106
8107#[cfg(test)]
8108mod terminal_history_tests {
8109 use std::{
8110 path::PathBuf,
8111 sync::Arc,
8112 time::{Duration, Instant},
8113 };
8114
8115 use tokio::time::sleep;
8116
8117 use super::{
8118 apply_deliberate_severance_marker, daemon_will_restart, drain_child_to_state,
8119 drained_after_quiescence_wait, handle_reload_spawn_failure, health_restart_child,
8120 lock_snapshot, on_child_exit, record_deliberate_severance, record_wait_error_terminal,
8121 reset_restart_count, spawn_and_mark_running, update_snapshot, wait_error_exit_report,
8122 ExitKind, ExitReport, ModuleProtocol, ModuleSpec, ModuleState, NextAction, ProcessIdentity,
8123 RestartPolicy, SpawnEventKind, StopNotice, SuperviseError, SupervisedModule, Supervisor,
8124 SupervisorHandle, SupervisorHealthStatus, SupervisorSnapshot,
8125 };
8126 use super::Instant as ClockInstant;
8131 use crate::{
8132 registry::Registry,
8133 terminal_ring::{TerminalRing, TerminalRingConfig},
8134 };
8135 use std::sync::Mutex;
8136 use subc_control::TerminalDisposition;
8137
8138 pub(super) fn fake_aft_stub_path() -> PathBuf {
8143 let mut path = std::env::current_exe().expect("current_exe available in tests");
8144 path.pop();
8145 path.pop();
8146 path.push(if cfg!(windows) {
8147 "fake-aft-stub.exe"
8148 } else {
8149 "fake-aft-stub"
8150 });
8151 assert!(
8152 path.exists(),
8153 "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
8154 [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
8155 path.display()
8156 );
8157 path
8158 }
8159
8160 #[test]
8161 fn reserved_never_spawned_refuses_every_hello() {
8162 let supervisor = SupervisorHandle::default();
8167 supervisor.apply_identity_configuration(&ModuleSpec {
8168 module_id: "never-spawned".to_string(),
8169 program: PathBuf::from("/usr/bin/false"),
8170 args: Vec::new(),
8171 env: Vec::new(),
8172 reserved: true,
8173 reserved_prefixes: Vec::new(),
8174 protocol: ModuleProtocol::Subc,
8175 overlap: Default::default(),
8176 });
8177 assert!(
8178 supervisor
8179 .reserved_hello_rejection("never-spawned", Some("any-forged-nonce"))
8180 .is_some(),
8181 "forged nonce must refuse on a reserved never-spawned id"
8182 );
8183 assert!(
8184 supervisor
8185 .reserved_hello_rejection("never-spawned", None)
8186 .is_some(),
8187 "absent nonce must refuse on a reserved never-spawned id"
8188 );
8189 supervisor.set_spawn_nonce("never-spawned", "minted".to_string());
8191 supervisor.apply_identity_configuration(&ModuleSpec {
8192 module_id: "never-spawned".to_string(),
8193 program: PathBuf::from("/usr/bin/false"),
8194 args: Vec::new(),
8195 env: Vec::new(),
8196 reserved: true,
8197 reserved_prefixes: Vec::new(),
8198 protocol: ModuleProtocol::Subc,
8199 overlap: Default::default(),
8200 });
8201 assert!(supervisor
8202 .reserved_hello_rejection("never-spawned", Some("minted"))
8203 .is_none());
8204 assert!(supervisor
8205 .reserved_hello_rejection("never-spawned", Some("forged"))
8206 .is_some());
8207 }
8208
8209 fn seed_crash_restarts(state: &mut SupervisorSnapshot, count: u32) {
8212 let now = ClockInstant::now();
8213 for _ in 0..count {
8214 state.crash_restarts.push_back(now);
8215 }
8216 }
8217
8218 fn age_oldest_crash_restart_out_of_window(state: &mut SupervisorSnapshot, window: Duration) {
8222 let aged = state
8223 .crash_restarts
8224 .front()
8225 .expect("a crash restart must be recorded before it can be aged")
8226 .checked_sub(window + Duration::from_secs(1))
8227 .expect("the test clock is far enough from its origin to age an instant");
8228 state.crash_restarts[0] = aged;
8229 }
8230
8231 fn snapshot_with_restarts(enabled: bool, count: u32) -> SupervisorSnapshot {
8232 let mut state = SupervisorSnapshot::new(ModuleState::Running, enabled);
8233 seed_crash_restarts(&mut state, count);
8234 state
8235 }
8236
8237 #[test]
8238 fn daemon_owned_recovery_predicate_uses_the_pre_increment_budget() {
8239 let policy = RestartPolicy::new(3, Duration::ZERO);
8240 let now = ClockInstant::now();
8241 assert!(daemon_will_restart(
8242 &mut snapshot_with_restarts(true, 2),
8243 &policy,
8244 now
8245 ));
8246 assert!(!daemon_will_restart(
8247 &mut snapshot_with_restarts(true, 3),
8248 &policy,
8249 now
8250 ));
8251 assert!(!daemon_will_restart(
8252 &mut snapshot_with_restarts(false, 0),
8253 &policy,
8254 now
8255 ));
8256 }
8257
8258 #[test]
8259 fn crash_restart_backoff_escalates_with_in_window_count() {
8260 let policy = RestartPolicy::new(4, Duration::from_millis(100))
8261 .with_max_backoff(Duration::from_secs(30));
8262 let now = ClockInstant::now();
8263 let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
8264 let schedules = (0..4)
8265 .map(|_| {
8266 state
8267 .next_crash_restart(&policy, now)
8268 .expect("the test policy allows four crash restarts")
8269 })
8270 .collect::<Vec<_>>();
8271
8272 assert_eq!(
8273 schedules
8274 .iter()
8275 .map(|schedule| schedule.restart_in_window)
8276 .collect::<Vec<_>>(),
8277 vec![0, 1, 2, 3]
8278 );
8279 assert_eq!(
8280 schedules
8281 .iter()
8282 .map(|schedule| schedule.delay)
8283 .collect::<Vec<_>>(),
8284 vec![
8285 Duration::from_millis(100),
8286 Duration::from_secs(1),
8287 Duration::from_secs(10),
8288 Duration::from_secs(30),
8289 ]
8290 );
8291 }
8292
8293 #[test]
8294 fn crash_restart_backoff_resets_after_ring_clear() {
8295 let policy = RestartPolicy::new(3, Duration::from_millis(100));
8296 let now = ClockInstant::now();
8297 let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
8298 assert_eq!(
8299 state.next_crash_restart(&policy, now).unwrap().delay,
8300 Duration::from_millis(100)
8301 );
8302 assert_eq!(
8303 state.next_crash_restart(&policy, now).unwrap().delay,
8304 Duration::from_secs(1)
8305 );
8306
8307 state.clear_crash_restarts();
8308 let schedule = state
8309 .next_crash_restart(&policy, now)
8310 .expect("a cleared ring must allow another restart");
8311 assert_eq!(schedule.restart_in_window, 0);
8312 assert_eq!(schedule.delay, Duration::from_millis(100));
8313 }
8314
8315 #[test]
8316 fn crash_restart_backoff_ignores_aged_restarts() {
8317 let policy = RestartPolicy::new(3, Duration::from_millis(100));
8318 let now = ClockInstant::now();
8319 let mut state = SupervisorSnapshot::new(ModuleState::Running, true);
8320 state
8321 .next_crash_restart(&policy, now)
8322 .expect("the first restart is allowed");
8323 state
8324 .next_crash_restart(&policy, now)
8325 .expect("the second restart is allowed");
8326 state.crash_restarts[0] = now
8327 .checked_sub(policy.window + Duration::from_secs(1))
8328 .expect("the fake clock can age a restart past the window");
8329
8330 let schedule = state
8331 .next_crash_restart(&policy, now)
8332 .expect("an aged restart must release its slot");
8333 assert_eq!(schedule.restart_in_window, 1);
8334 assert_eq!(schedule.delay, Duration::from_secs(1));
8335 assert_eq!(state.crash_restarts.len(), 2);
8336 }
8337
8338 #[test]
8342 fn a_budget_spent_before_the_window_no_longer_refuses() {
8343 let policy = RestartPolicy::new(3, Duration::ZERO);
8344 let mut state = snapshot_with_restarts(true, 3);
8345 let now = ClockInstant::now();
8346 assert!(!daemon_will_restart(&mut state, &policy, now));
8347
8348 assert!(daemon_will_restart(
8349 &mut state,
8350 &policy,
8351 now + policy.window + Duration::from_secs(1)
8352 ));
8353 assert!(
8354 state.crash_restarts.is_empty(),
8355 "reading the budget must drop the instants that left the window"
8356 );
8357 }
8358
8359 fn module_with_recovery_snapshot(
8360 state: ModuleState,
8361 enabled: bool,
8362 restart_count: u32,
8363 ) -> SupervisedModule {
8364 let registry = Arc::new(Registry::default());
8365 let supervisor =
8366 Supervisor::new(Arc::clone(®istry), RestartPolicy::new(3, Duration::ZERO));
8367 let module = supervisor
8368 .spawn(ModuleSpec {
8369 module_id: "recovery-snapshot".to_string(),
8370 program: fake_aft_stub_path(),
8371 args: Vec::new(),
8372 env: Vec::new(),
8373 reserved: false,
8374 reserved_prefixes: Vec::new(),
8375 protocol: ModuleProtocol::Subc,
8376 overlap: Default::default(),
8377 })
8378 .unwrap();
8379 update_snapshot(
8380 &module.inner.snapshot,
8381 Some("recovery-snapshot"),
8382 |snapshot| {
8383 snapshot.state = state;
8384 snapshot.enabled = enabled;
8385 seed_crash_restarts(snapshot, restart_count);
8386 },
8387 )
8388 .unwrap();
8389 module
8390 }
8391
8392 #[cfg(target_os = "linux")]
8393 #[tokio::test]
8394 async fn no_cgroup_placement_does_not_block_fake_aft_stub_spawn() {
8395 let supervisor = Supervisor::new(Arc::new(Registry::default()), RestartPolicy::default())
8396 .with_cgroup_placement(None);
8397 let result = supervisor.spawn(ModuleSpec {
8398 module_id: "no-cgroup-placement".to_string(),
8399 program: fake_aft_stub_path(),
8400 args: Vec::new(),
8401 env: Vec::new(),
8402 reserved: false,
8403 reserved_prefixes: Vec::new(),
8404 protocol: ModuleProtocol::Subc,
8405 overlap: Default::default(),
8406 });
8407
8408 assert!(
8409 result.is_ok(),
8410 "no delegation must not turn an otherwise valid spawn into a failure: {result:?}"
8411 );
8412 }
8413
8414 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8415 async fn undecided_snapshot_uses_shared_restart_predicate() {
8416 assert!(module_with_recovery_snapshot(ModuleState::Running, true, 2)
8417 .will_recover_after_connection_loss()
8418 .unwrap());
8419 assert!(
8420 !module_with_recovery_snapshot(ModuleState::Running, true, 3)
8421 .will_recover_after_connection_loss()
8422 .unwrap()
8423 );
8424 }
8425
8426 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8427 async fn restarting_snapshot_at_exhausted_budget_is_non_terminal() {
8428 assert!(
8429 module_with_recovery_snapshot(ModuleState::Restarting, true, 3)
8430 .will_recover_after_connection_loss()
8431 .unwrap()
8432 );
8433 }
8434
8435 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8436 async fn terminal_phase_snapshots_are_terminal_before_budget_exhaustion() {
8437 assert!(!module_with_recovery_snapshot(ModuleState::Failed, true, 0)
8438 .will_recover_after_connection_loss()
8439 .unwrap());
8440 assert!(
8441 !module_with_recovery_snapshot(ModuleState::Disabled, true, 0)
8442 .will_recover_after_connection_loss()
8443 .unwrap()
8444 );
8445 }
8446
8447 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8448 async fn warming_snapshot_is_limited_to_startup_phases() {
8449 for state in [
8450 ModuleState::Starting,
8451 ModuleState::Running,
8452 ModuleState::Restarting,
8453 ] {
8454 assert!(
8455 module_with_recovery_snapshot(state, true, 0)
8456 .is_warming()
8457 .unwrap(),
8458 "{state:?} should be warming"
8459 );
8460 }
8461 for state in [
8462 ModuleState::Unresponsive,
8463 ModuleState::Draining,
8464 ModuleState::Stopped,
8465 ModuleState::Failed,
8466 ModuleState::Disabled,
8467 ] {
8468 assert!(
8469 !module_with_recovery_snapshot(state, true, 0)
8470 .is_warming()
8471 .unwrap(),
8472 "{state:?} should not be warming"
8473 );
8474 }
8475 }
8476
8477 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8478 async fn terminal_history_survives_respawn_and_keeps_both_crashes_in_order() {
8479 let registry = Arc::new(Registry::default());
8480 let supervisor =
8481 Supervisor::new(Arc::clone(®istry), RestartPolicy::new(1, Duration::ZERO));
8482 let module = supervisor
8483 .spawn(ModuleSpec {
8484 module_id: "terminal-history".to_string(),
8485 program: fake_aft_stub_path(),
8486 args: Vec::new(),
8487 env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
8488 reserved: false,
8489 reserved_prefixes: Vec::new(),
8490 protocol: ModuleProtocol::Subc,
8491 overlap: Default::default(),
8492 })
8493 .unwrap();
8494
8495 let deadline = Instant::now() + Duration::from_secs(5);
8496 loop {
8497 let history = module.terminal_history();
8498 if history.entries.len() == 2 {
8499 assert_eq!(module.status().unwrap().state, ModuleState::Failed);
8500 assert_eq!(history.dropped, 0);
8501 assert_eq!(
8502 history
8503 .entries
8504 .iter()
8505 .map(|entry| entry.exit_code)
8506 .collect::<Vec<_>>(),
8507 vec![Some(23), Some(23)]
8508 );
8509 assert!(history.entries[0].at_ms <= history.entries[1].at_ms);
8510 return;
8511 }
8512 assert!(
8513 Instant::now() < deadline,
8514 "module did not retain two terminal exits: {history:?}"
8515 );
8516 sleep(Duration::from_millis(10)).await;
8517 }
8518 }
8519
8520 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8524 async fn disable_during_crash_backoff_cancels_pending_respawn() {
8525 let backoff = Duration::from_secs(2);
8526 let supervisor = Supervisor::new(
8527 Arc::new(Registry::default()),
8528 RestartPolicy::new(10, backoff),
8529 );
8530 let module = supervisor
8531 .spawn(ModuleSpec {
8532 module_id: "disable-during-backoff".to_string(),
8533 program: fake_aft_stub_path(),
8534 args: Vec::new(),
8535 env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
8536 reserved: false,
8537 reserved_prefixes: Vec::new(),
8538 protocol: ModuleProtocol::Subc,
8539 overlap: Default::default(),
8540 })
8541 .unwrap();
8542
8543 let deadline = Instant::now() + Duration::from_secs(5);
8545 loop {
8546 if module.status().unwrap().state == ModuleState::Restarting {
8547 break;
8548 }
8549 assert!(
8550 Instant::now() < deadline,
8551 "module never entered the crash backoff"
8552 );
8553 sleep(Duration::from_millis(10)).await;
8554 }
8555
8556 let started = Instant::now();
8557 module.set_enabled(false).await.unwrap();
8558 let waited = started.elapsed();
8559
8560 assert!(
8561 waited < backoff / 2,
8562 "disable waited {waited:?} behind the {backoff:?} crash backoff; the operator command must preempt the pending respawn"
8563 );
8564 assert_eq!(module.status().unwrap().state, ModuleState::Disabled);
8565
8566 sleep(backoff + Duration::from_millis(500)).await;
8568 let status = module.status().unwrap();
8569 assert_eq!(status.state, ModuleState::Disabled);
8570 assert_eq!(
8571 status.spawn_generation, 1,
8572 "module respawned after the operator disabled it"
8573 );
8574 }
8575
8576 #[cfg(unix)]
8579 fn protocol_none_sigterm_exits_clean_spec(
8580 module_id: &str,
8581 dir: &std::path::Path,
8582 ) -> (ModuleSpec, PathBuf, PathBuf) {
8583 let ready = dir.join("ready");
8584 let marker = dir.join("sigterm");
8585 let spec = ModuleSpec {
8586 module_id: module_id.to_string(),
8587 program: fake_aft_stub_path(),
8588 args: Vec::new(),
8589 env: vec![
8590 ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
8591 (
8592 "FAKE_AFT_SIGTERM_MARKER_PATH".to_string(),
8593 marker.display().to_string(),
8594 ),
8595 (
8596 "FAKE_AFT_NEVER_CONNECT_READY_PATH".to_string(),
8597 ready.display().to_string(),
8598 ),
8599 ],
8600 reserved: false,
8601 reserved_prefixes: Vec::new(),
8602 protocol: ModuleProtocol::None,
8603 overlap: Default::default(),
8604 };
8605 (spec, ready, marker)
8606 }
8607
8608 #[cfg(unix)]
8612 async fn wait_for_file(path: &std::path::Path) {
8613 let deadline = Instant::now() + Duration::from_secs(10);
8614 while !path.exists() {
8615 assert!(
8616 Instant::now() < deadline,
8617 "{} never appeared",
8618 path.display()
8619 );
8620 sleep(Duration::from_millis(10)).await;
8621 }
8622 }
8623
8624 #[cfg(unix)]
8628 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8629 async fn protocol_none_unrequested_clean_exit_restarts_as_a_crash() {
8630 let dir = subc_test_support::TestTempDir::new("none-unrequested-clean-exit");
8631 let (spec, ready, marker) =
8632 protocol_none_sigterm_exits_clean_spec("none-unrequested-clean-exit", dir.path());
8633 let supervisor = Supervisor::new(
8634 Arc::new(Registry::default()),
8635 RestartPolicy::new(3, Duration::ZERO),
8636 );
8637 let module = supervisor.spawn(spec).unwrap();
8638 wait_for_file(&ready).await;
8639 let first_pid = module
8640 .status()
8641 .unwrap()
8642 .pid
8643 .expect("a running module reports its pid");
8644
8645 rustix::process::kill_process(
8646 rustix::process::Pid::from_raw(i32::try_from(first_pid).unwrap()).unwrap(),
8647 rustix::process::Signal::TERM,
8648 )
8649 .unwrap();
8650
8651 let deadline = Instant::now() + Duration::from_secs(10);
8652 let respawned = loop {
8653 let status = module.status().unwrap();
8654 if status.state == ModuleState::Running
8655 && status.pid.is_some_and(|pid| pid != first_pid)
8656 {
8657 break status;
8658 }
8659 assert!(
8660 Instant::now() < deadline,
8661 "protocol-none module was not respawned after an unrequested clean exit: {status:?}"
8662 );
8663 sleep(Duration::from_millis(10)).await;
8664 };
8665 assert_eq!(respawned.spawn_generation, 2);
8666 assert!(
8667 marker.exists(),
8668 "the child must have exited through its SIGTERM handler (exit 0), or this proves nothing about clean exits"
8669 );
8670
8671 let history = module.terminal_history();
8672 assert_eq!(history.entries.len(), 1, "{history:?}");
8673 let entry = &history.entries[0];
8674 assert_eq!(entry.exit_code, Some(0));
8675 assert_eq!(entry.exit_kind, subc_control::TerminalExitKind::Clean);
8676 assert_eq!(entry.disposition, TerminalDisposition::Restarting);
8677
8678 module.stop().await.unwrap();
8679 }
8680
8681 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8685 async fn protocol_none_repeated_clean_exits_exhaust_the_restart_budget() {
8686 let supervisor = Supervisor::new(
8687 Arc::new(Registry::default()),
8688 RestartPolicy::new(1, Duration::ZERO),
8689 );
8690 let module = supervisor
8691 .spawn(ModuleSpec {
8692 module_id: "none-clean-exit-budget".to_string(),
8693 program: fake_aft_stub_path(),
8694 args: Vec::new(),
8695 env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
8696 reserved: false,
8697 reserved_prefixes: Vec::new(),
8698 protocol: ModuleProtocol::None,
8699 overlap: Default::default(),
8700 })
8701 .unwrap();
8702
8703 let deadline = Instant::now() + Duration::from_secs(10);
8704 loop {
8705 let status = module.status().unwrap();
8706 if status.state == ModuleState::Failed {
8707 break;
8708 }
8709 assert!(
8710 Instant::now() < deadline,
8711 "module never exhausted its budget: {status:?} {:?}",
8712 module.terminal_history()
8713 );
8714 sleep(Duration::from_millis(10)).await;
8715 }
8716 let history = module.terminal_history();
8717 assert_eq!(
8718 history
8719 .entries
8720 .iter()
8721 .map(|entry| (entry.exit_code, entry.disposition.clone()))
8722 .collect::<Vec<_>>(),
8723 vec![
8724 (Some(0), TerminalDisposition::Restarting),
8725 (Some(0), TerminalDisposition::Failed),
8726 ]
8727 );
8728 let detail = history.entries[1]
8729 .disposition_detail
8730 .as_deref()
8731 .expect("a budget failure names the budget");
8732 assert!(detail.contains("max_restarts=1"), "{detail}");
8733 assert_eq!(module.status().unwrap().spawn_generation, 2);
8734 }
8735
8736 #[cfg(unix)]
8739 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8740 async fn protocol_none_requested_stop_and_disable_do_not_respawn() {
8741 for disable in [false, true] {
8742 let label = if disable {
8743 "none-requested-disable"
8744 } else {
8745 "none-requested-stop"
8746 };
8747 let dir = subc_test_support::TestTempDir::new(label);
8748 let (spec, ready, marker) = protocol_none_sigterm_exits_clean_spec(label, dir.path());
8749 let supervisor = Supervisor::new(
8750 Arc::new(Registry::default()),
8751 RestartPolicy::new(3, Duration::ZERO),
8752 );
8753 let module = supervisor.spawn(spec).unwrap();
8754 wait_for_file(&ready).await;
8755
8756 if disable {
8757 module.set_enabled(false).await.unwrap();
8758 } else {
8759 module.stop().await.unwrap();
8760 }
8761 assert!(
8762 marker.exists(),
8763 "{label}: the child must have left through its SIGTERM handler with exit 0"
8764 );
8765
8766 sleep(Duration::from_millis(500)).await;
8769 let status = module.status().unwrap();
8770 let expected = if disable {
8771 ModuleState::Disabled
8772 } else {
8773 ModuleState::Stopped
8774 };
8775 assert_eq!(status.state, expected, "{label}");
8776 assert_eq!(
8777 status.spawn_generation, 1,
8778 "{label}: respawned after a requested stop"
8779 );
8780 let history = module.terminal_history();
8781 assert_eq!(history.entries.len(), 1, "{label}: {history:?}");
8782 assert_eq!(history.entries[0].exit_code, Some(0), "{label}");
8783 assert_ne!(
8784 history.entries[0].disposition,
8785 TerminalDisposition::Restarting,
8786 "{label}"
8787 );
8788 }
8789 }
8790
8791 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8794 async fn subc_wire_clean_exit_is_still_a_stop() {
8795 let supervisor = Supervisor::new(
8796 Arc::new(Registry::default()),
8797 RestartPolicy::new(3, Duration::ZERO),
8798 );
8799 let module = supervisor
8800 .spawn(ModuleSpec {
8801 module_id: "wire-clean-exit".to_string(),
8802 program: fake_aft_stub_path(),
8803 args: Vec::new(),
8804 env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "0".to_string())],
8805 reserved: false,
8806 reserved_prefixes: Vec::new(),
8807 protocol: ModuleProtocol::Subc,
8808 overlap: Default::default(),
8809 })
8810 .unwrap();
8811
8812 let deadline = Instant::now() + Duration::from_secs(10);
8813 while module.terminal_history().entries.is_empty() {
8814 assert!(Instant::now() < deadline, "module never exited");
8815 sleep(Duration::from_millis(10)).await;
8816 }
8817 sleep(Duration::from_millis(500)).await;
8819 let status = module.status().unwrap();
8820 assert_eq!(status.state, ModuleState::Stopped);
8821 assert_eq!(status.spawn_generation, 1);
8822 let history = module.terminal_history();
8823 assert_eq!(history.entries.len(), 1, "{history:?}");
8824 assert_eq!(history.entries[0].exit_code, Some(0));
8825 assert_eq!(history.entries[0].disposition, TerminalDisposition::Stopped);
8826 }
8827
8828 #[cfg(unix)]
8829 #[tokio::test]
8830 async fn reload_exit_keeps_roster_until_terminal_is_recorded_during_shutdown() {
8831 let dir = subc_test_support::TestTempDir::new("reload-roster-terminal-order");
8832 let record = dir.join("live-children.json");
8833 let supervisor = Supervisor::new(
8834 Arc::new(Registry::default()),
8835 RestartPolicy::new(0, Duration::ZERO),
8836 );
8837 let mut runtime = supervisor.runtime_config();
8838 runtime.child_roster.record_to(record.clone());
8839 let gate = Arc::new(super::ReloadExitRecordGate::default());
8840 runtime.test_reload_exit_record_gate = Some(Arc::clone(&gate));
8841 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8842 let spec = ModuleSpec {
8843 module_id: "reload-exit-roster".into(),
8844 program: fake_aft_stub_path(),
8845 args: Vec::new(),
8846 env: vec![("FAKE_AFT_EXIT_CODE".into(), "23".into())],
8847 reserved: false,
8848 reserved_prefixes: Vec::new(),
8849 protocol: ModuleProtocol::Subc,
8850 overlap: Default::default(),
8851 };
8852 let mut child = None;
8853 let reload = super::finish_reload_child(
8854 &spec,
8855 &runtime,
8856 &supervisor.registry,
8857 &supervisor.process_liveness,
8858 &snapshot,
8859 &mut child,
8860 );
8861 tokio::pin!(reload);
8862 tokio::select! {
8863 result = &mut reload => panic!("reload missed the before-record gate: {result:?}"),
8864 _ = gate.reached.notified() => {}
8865 }
8866 assert!(runtime
8867 .terminal_ring
8868 .lock()
8869 .unwrap()
8870 .snapshot()
8871 .entries
8872 .is_empty());
8873 assert_eq!(
8874 crate::live_children::read_record(&record).unwrap().len(),
8875 1,
8876 "shutdown must still wait for the reaped child until its terminal record exists"
8877 );
8878 runtime.child_roster.close();
8879 gate.resume.notify_one();
8880 assert!(reload.await.is_err());
8881 assert!(crate::live_children::read_record(&record)
8882 .unwrap()
8883 .is_empty());
8884 let history = runtime.terminal_ring.lock().unwrap().snapshot();
8885 assert_eq!(history.entries.len(), 1);
8886 assert_eq!(
8887 history.entries[0].disposition,
8888 TerminalDisposition::DaemonShutdown
8889 );
8890 }
8891
8892 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
8896 async fn every_restart_increment_path_advances_lifetime_count() {
8897 let supervisor = Supervisor::new(
8898 Arc::new(Registry::default()),
8899 RestartPolicy::new(1, Duration::ZERO),
8900 );
8901 let runtime = supervisor.runtime_config();
8902 let spec = ModuleSpec {
8903 module_id: "lifetime-increment-path".to_string(),
8904 program: PathBuf::from("/unused/lifetime-increment-path"),
8905 args: Vec::new(),
8906 env: Vec::new(),
8907 reserved: false,
8908 reserved_prefixes: Vec::new(),
8909 protocol: ModuleProtocol::Subc,
8910 overlap: Default::default(),
8911 };
8912
8913 let crash_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8914 assert!(matches!(
8915 on_child_exit(
8916 &spec,
8917 runtime.restart_policy,
8918 &supervisor.registry,
8919 &crash_snapshot,
8920 &runtime.terminal_ring,
8921 &runtime.spawn_events,
8922 &runtime.child_roster,
8923 ExitReport {
8924 kind: ExitKind::Crash,
8925 code: Some(1),
8926 signal: None,
8927 at_ms: 1,
8928 },
8929 )
8930 .await,
8931 NextAction::Restart { schedule: _ }
8932 ));
8933 let (crash_restarts, crash_lifetime) = {
8934 let state = lock_snapshot(&crash_snapshot).unwrap();
8935 (state.crash_restarts.len(), state.lifetime_restarts)
8936 };
8937 assert_eq!(crash_restarts, 1);
8938 assert_eq!(crash_lifetime, 1);
8939
8940 let health_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8941 let mut health_child = None;
8942 assert!(matches!(
8943 health_restart_child(
8944 &spec,
8945 &runtime,
8946 &supervisor.registry,
8947 &supervisor.process_liveness,
8948 &health_snapshot,
8949 &mut health_child,
8950 SupervisorHealthStatus::Failing,
8951 None,
8952 2,
8953 )
8954 .await,
8955 Ok(())
8956 ));
8957 assert!(health_child.is_none());
8958 assert!(lock_snapshot(&health_snapshot).unwrap().respawn_pending);
8959 assert!(runtime.scheduled_respawn.lock().unwrap().take().is_some());
8960 let (health_restarts, health_lifetime) = {
8961 let state = lock_snapshot(&health_snapshot).unwrap();
8962 (state.crash_restarts.len(), state.lifetime_restarts)
8963 };
8964 assert_eq!(health_restarts, 1);
8965 assert_eq!(health_lifetime, 1);
8966
8967 let reload_snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
8968 let mut reload_child = None;
8969 assert!(matches!(
8970 handle_reload_spawn_failure(
8971 &spec,
8972 &runtime,
8973 &supervisor.process_liveness,
8974 &reload_snapshot,
8975 &mut reload_child,
8976 "forced reload spawn failure".to_string(),
8977 )
8978 .await,
8979 Err(SuperviseError::ReloadFailed { .. })
8980 ));
8981 let (reload_restarts, reload_lifetime) = {
8982 let state = lock_snapshot(&reload_snapshot).unwrap();
8983 (state.crash_restarts.len(), state.lifetime_restarts)
8984 };
8985 assert_eq!(reload_restarts, 1);
8986 assert_eq!(reload_lifetime, 1);
8987 }
8988
8989 #[tokio::test]
8990 async fn deliberately_severed_live_child_records_lifetime_without_spending_restart_budget() {
8991 let supervisor = Supervisor::new(
8992 Arc::new(Registry::default()),
8993 RestartPolicy::new(3, Duration::ZERO),
8994 );
8995 let runtime = supervisor.runtime_config();
8996 let spec = ModuleSpec {
8997 module_id: "deliberately-severed".to_string(),
8998 program: PathBuf::from("/unused/deliberately-severed"),
8999 args: Vec::new(),
9000 env: Vec::new(),
9001 reserved: false,
9002 reserved_prefixes: Vec::new(),
9003 protocol: ModuleProtocol::Subc,
9004 overlap: Default::default(),
9005 };
9006 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
9007 let process = ProcessIdentity {
9008 pid: 41,
9009 start_time: 101,
9010 };
9011 record_deliberate_severance(&snapshot, process).unwrap();
9012 let exit_report = apply_deliberate_severance_marker(
9013 &snapshot,
9014 Some(process),
9015 ExitReport {
9016 kind: ExitKind::Crash,
9017 code: Some(1),
9018 signal: None,
9019 at_ms: 1,
9020 },
9021 );
9022 assert_eq!(exit_report.kind, ExitKind::DeliberateSeverance);
9023
9024 assert!(matches!(
9025 on_child_exit(
9026 &spec,
9027 runtime.restart_policy,
9028 &supervisor.registry,
9029 &snapshot,
9030 &runtime.terminal_ring,
9031 &runtime.spawn_events,
9032 &runtime.child_roster,
9033 exit_report,
9034 )
9035 .await,
9036 NextAction::Restart { schedule: _ }
9037 ));
9038 let state = lock_snapshot(&snapshot).unwrap();
9039 assert_eq!(state.lifetime_restarts, 1);
9040 assert_eq!(state.crash_restarts.len(), 0);
9041 }
9042
9043 #[tokio::test]
9044 async fn genuine_crash_spends_restart_budget_and_records_lifetime() {
9045 let supervisor = Supervisor::new(
9046 Arc::new(Registry::default()),
9047 RestartPolicy::new(3, Duration::ZERO),
9048 );
9049 let runtime = supervisor.runtime_config();
9050 let spec = ModuleSpec {
9051 module_id: "genuine-crash".to_string(),
9052 program: PathBuf::from("/unused/genuine-crash"),
9053 args: Vec::new(),
9054 env: Vec::new(),
9055 reserved: false,
9056 reserved_prefixes: Vec::new(),
9057 protocol: ModuleProtocol::Subc,
9058 overlap: Default::default(),
9059 };
9060 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
9061
9062 assert!(matches!(
9063 on_child_exit(
9064 &spec,
9065 runtime.restart_policy,
9066 &supervisor.registry,
9067 &snapshot,
9068 &runtime.terminal_ring,
9069 &runtime.spawn_events,
9070 &runtime.child_roster,
9071 ExitReport {
9072 kind: ExitKind::Crash,
9073 code: Some(1),
9074 signal: None,
9075 at_ms: 1,
9076 },
9077 )
9078 .await,
9079 NextAction::Restart { schedule: _ }
9080 ));
9081 let state = lock_snapshot(&snapshot).unwrap();
9082 assert_eq!(state.lifetime_restarts, 1);
9083 assert_eq!(state.crash_restarts.len(), 1);
9084 }
9085
9086 fn crash_exit_report(at_ms: u64) -> ExitReport {
9087 ExitReport {
9088 kind: ExitKind::Crash,
9089 code: Some(1),
9090 signal: None,
9091 at_ms,
9092 }
9093 }
9094
9095 fn windowed_crash_spec(module_id: &str) -> ModuleSpec {
9096 ModuleSpec {
9097 module_id: module_id.to_string(),
9098 program: PathBuf::from("/unused").join(module_id),
9099 args: Vec::new(),
9100 env: Vec::new(),
9101 reserved: false,
9102 reserved_prefixes: Vec::new(),
9103 protocol: ModuleProtocol::Subc,
9104 overlap: Default::default(),
9105 }
9106 }
9107
9108 #[tokio::test]
9114 async fn three_crashes_inside_the_window_stop_the_module_and_name_the_window() {
9115 let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::ERROR);
9116 let supervisor = Supervisor::new(
9117 Arc::new(Registry::default()),
9118 RestartPolicy::new(2, Duration::ZERO),
9119 );
9120 let runtime = supervisor.runtime_config();
9121 let spec = windowed_crash_spec("crash-loop-in-window");
9122 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
9123
9124 for attempt in 1..=2 {
9125 assert!(
9126 matches!(
9127 on_child_exit(
9128 &spec,
9129 runtime.restart_policy,
9130 &supervisor.registry,
9131 &snapshot,
9132 &runtime.terminal_ring,
9133 &runtime.spawn_events,
9134 &runtime.child_roster,
9135 crash_exit_report(attempt),
9136 )
9137 .await,
9138 NextAction::Restart { schedule: _ }
9139 ),
9140 "crash {attempt} is inside the budget and must respawn"
9141 );
9142 }
9143
9144 assert!(matches!(
9145 on_child_exit(
9146 &spec,
9147 runtime.restart_policy,
9148 &supervisor.registry,
9149 &snapshot,
9150 &runtime.terminal_ring,
9151 &runtime.spawn_events,
9152 &runtime.child_roster,
9153 crash_exit_report(3),
9154 )
9155 .await,
9156 NextAction::Stop { .. }
9157 ));
9158
9159 {
9160 let state = lock_snapshot(&snapshot).unwrap();
9161 assert_eq!(state.state, ModuleState::Failed);
9162 assert_eq!(state.crash_restarts.len(), 2);
9163 assert_eq!(state.lifetime_restarts, 2);
9164 }
9165
9166 let history = runtime
9167 .terminal_ring
9168 .lock()
9169 .expect("terminal ring is not poisoned")
9170 .snapshot();
9171 let last = history
9172 .entries
9173 .last()
9174 .expect("the refused crash is retained");
9175 assert_eq!(last.disposition, TerminalDisposition::Failed);
9176 assert_eq!(
9177 last.disposition_detail.as_deref(),
9178 Some("crash budget exhausted: max_restarts=2 within window_secs=600")
9179 );
9180
9181 let captured = crate::router::test_log::captured_logs(&logs);
9182 assert!(
9183 captured.contains("crash budget exhausted: max_restarts=2 within window_secs=600"),
9184 "the stop must be logged with its window: {captured}"
9185 );
9186 }
9187
9188 #[tokio::test]
9196 async fn a_crash_older_than_the_window_frees_its_slot_for_a_later_crash() {
9197 let supervisor = Supervisor::new(
9198 Arc::new(Registry::default()),
9199 RestartPolicy::new(2, Duration::ZERO),
9200 );
9201 let runtime = supervisor.runtime_config();
9202 let spec = windowed_crash_spec("crash-across-windows");
9203 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
9204
9205 for attempt in 1..=2 {
9206 assert!(matches!(
9207 on_child_exit(
9208 &spec,
9209 runtime.restart_policy,
9210 &supervisor.registry,
9211 &snapshot,
9212 &runtime.terminal_ring,
9213 &runtime.spawn_events,
9214 &runtime.child_roster,
9215 crash_exit_report(attempt),
9216 )
9217 .await,
9218 NextAction::Restart { schedule: _ }
9219 ));
9220 }
9221
9222 update_snapshot(&snapshot, Some(&spec.module_id), |state| {
9225 age_oldest_crash_restart_out_of_window(state, runtime.restart_policy.window);
9226 })
9227 .unwrap();
9228
9229 assert!(
9230 matches!(
9231 on_child_exit(
9232 &spec,
9233 runtime.restart_policy,
9234 &supervisor.registry,
9235 &snapshot,
9236 &runtime.terminal_ring,
9237 &runtime.spawn_events,
9238 &runtime.child_roster,
9239 crash_exit_report(3),
9240 )
9241 .await,
9242 NextAction::Restart { schedule: _ }
9243 ),
9244 "a crash older than the window must not hold a budget slot"
9245 );
9246
9247 let state = lock_snapshot(&snapshot).unwrap();
9248 assert_eq!(state.state, ModuleState::Restarting);
9249 assert_eq!(
9250 state.crash_restarts.len(),
9251 2,
9252 "the aged instant is dropped and the new one takes its place"
9253 );
9254 assert_eq!(
9255 state.lifetime_restarts, 3,
9256 "the ledger counts every restart, including the ones the window forgot"
9257 );
9258 }
9259
9260 #[tokio::test]
9265 async fn an_operator_restart_clears_the_ring_and_leaves_the_ledger_alone() {
9266 let supervisor = Supervisor::new(
9267 Arc::new(Registry::default()),
9268 RestartPolicy::new(2, Duration::ZERO),
9269 );
9270 let runtime = supervisor.runtime_config();
9271 let spec = windowed_crash_spec("operator-cleared-budget");
9272 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
9273
9274 for attempt in 1..=2 {
9275 assert!(matches!(
9276 on_child_exit(
9277 &spec,
9278 runtime.restart_policy,
9279 &supervisor.registry,
9280 &snapshot,
9281 &runtime.terminal_ring,
9282 &runtime.spawn_events,
9283 &runtime.child_roster,
9284 crash_exit_report(attempt),
9285 )
9286 .await,
9287 NextAction::Restart { schedule: _ }
9288 ));
9289 }
9290
9291 reset_restart_count(&snapshot, &spec.module_id).unwrap();
9292 {
9293 let state = lock_snapshot(&snapshot).unwrap();
9294 assert!(
9295 state.crash_restarts.is_empty(),
9296 "an operator restart returns the full budget"
9297 );
9298 assert_eq!(
9299 state.lifetime_restarts, 2,
9300 "clearing the budget must not unmake the crashes"
9301 );
9302 }
9303
9304 assert!(
9305 matches!(
9306 on_child_exit(
9307 &spec,
9308 runtime.restart_policy,
9309 &supervisor.registry,
9310 &snapshot,
9311 &runtime.terminal_ring,
9312 &runtime.spawn_events,
9313 &runtime.child_roster,
9314 crash_exit_report(3),
9315 )
9316 .await,
9317 NextAction::Restart { schedule: _ }
9318 ),
9319 "the cleared budget must be spendable again"
9320 );
9321 let state = lock_snapshot(&snapshot).unwrap();
9322 assert_eq!(state.crash_restarts.len(), 1);
9323 assert_eq!(state.lifetime_restarts, 3);
9324 }
9325
9326 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
9327 async fn severance_marker_for_a_dead_child_does_not_label_its_successor() {
9328 let severed = ProcessIdentity {
9329 pid: 41,
9330 start_time: 101,
9331 };
9332 let successor = ProcessIdentity {
9333 pid: 41,
9334 start_time: 202,
9335 };
9336 let module = module_with_recovery_snapshot(ModuleState::Running, true, 0);
9337 update_snapshot(&module.inner.snapshot, Some("recovery-snapshot"), |state| {
9338 state.pid = Some(successor.pid);
9339 state.process_start_time = Some(successor.start_time);
9340 })
9341 .unwrap();
9342 assert!(!module.record_deliberate_severance(severed).unwrap());
9343
9344 let exit_report = apply_deliberate_severance_marker(
9345 &module.inner.snapshot,
9346 Some(successor),
9347 ExitReport {
9348 kind: ExitKind::Crash,
9349 code: Some(1),
9350 signal: None,
9351 at_ms: 1,
9352 },
9353 );
9354
9355 assert_eq!(exit_report.kind, ExitKind::Crash);
9356 }
9357
9358 #[tokio::test]
9359 async fn drain_reap_marks_deliberate_severance_and_records_lifetime_without_budget() {
9360 let registry = Registry::default();
9361 let supervisor = Supervisor::new(
9362 Arc::new(Registry::default()),
9363 RestartPolicy::new(3, Duration::ZERO),
9364 );
9365 let runtime = supervisor.runtime_config();
9366 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
9367 let spec = ModuleSpec {
9368 module_id: "drain-deliberate-severance".to_string(),
9369 program: fake_aft_stub_path(),
9370 args: Vec::new(),
9371 env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
9372 reserved: false,
9373 reserved_prefixes: Vec::new(),
9374 protocol: ModuleProtocol::Subc,
9375 overlap: Default::default(),
9376 };
9377 let mut child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
9378 let process = ProcessIdentity {
9379 pid: 41,
9380 start_time: 101,
9381 };
9382 child.process_identity = Some(process);
9383 update_snapshot(&snapshot, Some(&spec.module_id), |state| {
9384 state.pid = Some(process.pid);
9385 state.process_start_time = Some(process.start_time);
9386 })
9387 .unwrap();
9388 record_deliberate_severance(&snapshot, process).unwrap();
9389
9390 drain_child_to_state(
9391 &spec.module_id,
9392 spec.protocol,
9393 StopNotice::SentOverConnection,
9396 ®istry,
9397 None,
9398 &snapshot,
9399 &runtime.terminal_ring,
9400 &runtime.spawn_events,
9401 child,
9402 Duration::from_secs(1),
9403 ModuleState::Stopped,
9404 Some(false),
9405 )
9406 .await
9407 .unwrap();
9408
9409 let state = lock_snapshot(&snapshot).unwrap();
9410 assert_eq!(
9411 state.last_exit.as_ref().map(|exit| exit.kind),
9412 Some(ExitKind::DeliberateSeverance)
9413 );
9414 assert_eq!(state.lifetime_restarts, 1);
9415 assert_eq!(state.crash_restarts.len(), 0);
9416 drop(state);
9417 let history = runtime.terminal_ring.lock().unwrap().snapshot();
9418 assert_eq!(
9419 history.entries[0].exit_kind,
9420 subc_control::TerminalExitKind::DeliberateSeverance
9421 );
9422 }
9423
9424 #[tokio::test]
9425 async fn ordinary_drain_reap_does_not_record_a_lifetime_restart() {
9426 let registry = Registry::default();
9427 let supervisor = Supervisor::new(
9428 Arc::new(Registry::default()),
9429 RestartPolicy::new(3, Duration::ZERO),
9430 );
9431 let runtime = supervisor.runtime_config();
9432 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
9433 let spec = ModuleSpec {
9434 module_id: "ordinary-drain".to_string(),
9435 program: fake_aft_stub_path(),
9436 args: Vec::new(),
9437 env: vec![("FAKE_AFT_EXIT_CODE".to_string(), "23".to_string())],
9438 reserved: false,
9439 reserved_prefixes: Vec::new(),
9440 protocol: ModuleProtocol::Subc,
9441 overlap: Default::default(),
9442 };
9443 let child = spawn_and_mark_running(&spec, &runtime, &snapshot).unwrap();
9444
9445 drain_child_to_state(
9446 &spec.module_id,
9447 spec.protocol,
9448 StopNotice::SentOverConnection,
9451 ®istry,
9452 None,
9453 &snapshot,
9454 &runtime.terminal_ring,
9455 &runtime.spawn_events,
9456 child,
9457 Duration::from_secs(1),
9458 ModuleState::Stopped,
9459 Some(false),
9460 )
9461 .await
9462 .unwrap();
9463
9464 let state = lock_snapshot(&snapshot).unwrap();
9465 assert_eq!(
9466 state.last_exit.as_ref().map(|exit| exit.kind),
9467 Some(ExitKind::Crash)
9468 );
9469 assert_eq!(state.lifetime_restarts, 0);
9470 assert_eq!(state.crash_restarts.len(), 0);
9471 }
9472
9473 #[test]
9474 fn fatal_connection_teardown_cannot_arm_a_marker_for_a_surviving_process() {
9475 assert!(!include_str!("server.rs")
9481 .contains("router.record_deliberate_connection_severance(ctx.connection_id)"));
9482 }
9483
9484 #[test]
9491 fn drained_after_quiescence_wait_passes_ok_through_and_forces_false_on_err() {
9492 assert!(drained_after_quiescence_wait(&Ok(true)));
9493 assert!(!drained_after_quiescence_wait(&Ok(false)));
9494 assert!(!drained_after_quiescence_wait(&Err(
9495 SuperviseError::StatePoisoned { module_id: None }
9496 )));
9497 }
9498
9499 #[test]
9508 fn wait_error_exit_report_records_a_failed_terminal_with_no_code_or_signal() {
9509 let ring = Arc::new(Mutex::new(TerminalRing::new(
9510 TerminalRingConfig::default(),
9511 0,
9512 )));
9513 record_wait_error_terminal("wait-error", &ring, &super::SpawnEventFeed::default());
9514
9515 let snapshot = ring.lock().unwrap().snapshot();
9516 assert_eq!(snapshot.entries.len(), 1);
9517 let entry = &snapshot.entries[0];
9518 assert_eq!(entry.exit_code, None);
9519 assert_eq!(entry.exit_signal, None);
9520 assert_eq!(entry.disposition, TerminalDisposition::Failed);
9521 }
9522
9523 #[test]
9524 fn wait_error_exit_path_preserves_spawn_event_density() {
9525 let feed = super::SpawnEventFeed::default();
9526 feed.configure_incarnation("wait-error-density".to_string());
9527 feed.emit_spawned("wait-error", 41, 1);
9528 let ring = Arc::new(Mutex::new(TerminalRing::new(
9529 TerminalRingConfig::default(),
9530 0,
9531 )));
9532
9533 record_wait_error_terminal("wait-error", &ring, &feed);
9534 feed.emit_spawned("after-wait-error", 42, 2);
9535
9536 let state = feed.0.lock().unwrap();
9537 let sequences = state
9538 .events
9539 .iter()
9540 .map(|event| event.cursor.seq)
9541 .collect::<Vec<_>>();
9542 assert_eq!(sequences, vec![1, 2, 3]);
9543 assert_eq!(state.events[1].kind, SpawnEventKind::Exited);
9544 assert_eq!(state.events[1].exit_code, None);
9545 assert_eq!(state.events[1].exit_signal, None);
9546 }
9547
9548 #[test]
9552 fn wait_error_exit_report_is_classified_as_a_crash() {
9553 assert_eq!(wait_error_exit_report().kind, ExitKind::Crash);
9554 }
9555}
9556
9557#[cfg(test)]
9558mod health_evidence_tests {
9559 use super::{HealthProbeError, HealthProbeEvidence};
9560 use std::collections::HashSet;
9561
9562 #[test]
9570 fn only_a_dead_lane_is_proof_of_death() {
9571 assert!(HealthProbeError::lane_dead("gone").is_proof_of_death());
9572 assert!(!HealthProbeError::no_answer("timed out").is_proof_of_death());
9576 assert!(!HealthProbeError::bad_answer("garbage").is_proof_of_death());
9577 assert!(!HealthProbeError::misconfigured("no table").is_proof_of_death());
9578 }
9579
9580 #[test]
9586 fn every_evidence_class_has_a_distinct_label() {
9587 let labels = [
9588 HealthProbeError::lane_dead("").label(),
9589 HealthProbeError::no_answer("").label(),
9590 HealthProbeError::bad_answer("").label(),
9591 HealthProbeError::misconfigured("").label(),
9592 ];
9593 let unique: HashSet<_> = labels.iter().collect();
9594 assert_eq!(unique.len(), labels.len(), "labels collided: {labels:?}");
9595 }
9596
9597 #[test]
9603 fn classification_preserves_the_original_message() {
9604 let err = HealthProbeError::no_answer("module did not answer within 5s");
9605 assert_eq!(err.to_string(), "module did not answer within 5s");
9606 assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
9607 }
9608}
9609
9610#[cfg(test)]
9611mod health_tombstone_tests {
9612 use std::{path::PathBuf, sync::Arc, time::Duration};
9613
9614 use subc_protocol::{
9615 manifest::Concurrency,
9616 session::{HealthStatus, ModuleControlResponse},
9617 };
9618 use tokio::sync::mpsc;
9619
9620 use super::{
9621 probe_module_health, HealthAction, HealthConfig, HealthProbeEvidence, ModuleProtocol,
9622 ModuleSpec, RestartPolicy, Supervisor, SupervisorRuntimeConfig,
9623 };
9624 use crate::{
9625 control::ControlHandler,
9626 forwarding::{ForwardingTable, ModuleControlRpcCompletion, ModuleControlRpcOutcome},
9627 registry::{ConnectionId, Registry},
9628 router::FrameSink,
9629 };
9630
9631 struct ProbeHarness {
9632 spec: ModuleSpec,
9633 runtime: SupervisorRuntimeConfig,
9634 forwarding: Arc<ForwardingTable>,
9635 module_connection: ConnectionId,
9636 module_rx: mpsc::Receiver<crate::router::OutboundFrame>,
9637 handler: ControlHandler,
9638 module: super::SupervisedModule,
9639 }
9640
9641 fn probe_harness() -> ProbeHarness {
9642 let registry = Arc::new(Registry::default());
9643 let forwarding = Arc::new(ForwardingTable::default());
9644 let supervisor_handle = super::SupervisorHandle::new();
9645 let health = HealthConfig {
9646 cadence: Duration::from_secs(30),
9647 deadline: Duration::from_secs(5),
9648 failure_threshold: 3,
9649 on_degraded: HealthAction::Report,
9650 on_failing: HealthAction::Report,
9651 critical: false,
9652 };
9653 let supervisor = Supervisor::new(Arc::clone(®istry), RestartPolicy::default())
9654 .with_forwarding(Arc::clone(&forwarding))
9655 .with_handle(supervisor_handle.clone())
9656 .with_health_config(health);
9657 let spec = ModuleSpec {
9658 module_id: "late-health-module".to_string(),
9659 program: PathBuf::from("disabled-module"),
9660 args: Vec::new(),
9661 env: Vec::new(),
9662 reserved: false,
9663 reserved_prefixes: Vec::new(),
9664 protocol: ModuleProtocol::Subc,
9665 overlap: Default::default(),
9666 };
9667 let module = supervisor
9668 .supervise_configured(spec.clone(), false)
9669 .unwrap();
9670 let runtime = supervisor.runtime_config();
9671 let handler = ControlHandler::with_forwarding(registry, Arc::clone(&forwarding))
9672 .with_supervisor(supervisor_handle);
9673 let module_connection = ConnectionId::new(700);
9674 let (module_tx, module_rx) = mpsc::channel(8);
9675 forwarding
9676 .register_module_connection(
9677 module_connection,
9678 spec.module_id.clone(),
9679 subc_protocol::PROTOCOL_VERSION,
9680 Concurrency::ModuleManaged,
9681 FrameSink::new(module_tx),
9682 )
9683 .unwrap();
9684
9685 ProbeHarness {
9686 spec,
9687 runtime,
9688 forwarding,
9689 module_connection,
9690 module_rx,
9691 handler,
9692 module,
9693 }
9694 }
9695
9696 async fn finish_after(
9697 harness: &mut ProbeHarness,
9698 stall: Duration,
9699 ) -> ModuleControlRpcCompletion {
9700 assert!(stall > harness.runtime.health.deadline);
9701 let deadline = harness.runtime.health.deadline;
9702 let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
9703 let answer = async {
9704 let frame = harness.module_rx.recv().await.expect("health.check frame");
9705 tokio::time::advance(deadline).await;
9706 tokio::task::yield_now().await;
9707 tokio::time::advance(stall - deadline).await;
9708 harness
9709 .forwarding
9710 .complete_module_control_rpc(
9711 harness.module_connection,
9712 frame.header.corr,
9713 Some("health.check"),
9714 ModuleControlRpcOutcome::Response(ModuleControlResponse::HealthCheck {
9715 status: HealthStatus::Ok,
9716 detail: None,
9717 metrics: None,
9718 }),
9719 )
9720 .unwrap()
9721 };
9722 let (probe_result, completion) = tokio::join!(probe, answer);
9723 let err = probe_result.expect_err("probe must miss its deadline");
9724 assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
9725 completion
9726 }
9727
9728 async fn time_out_without_answer(harness: &mut ProbeHarness) {
9729 let deadline = harness.runtime.health.deadline;
9730 let probe = probe_module_health(&harness.spec.module_id, &harness.runtime, None);
9731 let exhaust_deadline = async {
9732 let _frame = harness.module_rx.recv().await.expect("health.check frame");
9733 tokio::time::advance(deadline).await;
9734 tokio::task::yield_now().await;
9735 };
9736 let (probe_result, ()) = tokio::join!(probe, exhaust_deadline);
9737 let err = probe_result.expect_err("probe must miss its deadline");
9738 assert!(matches!(err.evidence, HealthProbeEvidence::NoAnswer));
9739 }
9740
9741 #[tokio::test(start_paused = true)]
9742 async fn late_health_answers_record_start_anchored_latency_for_two_stalls() {
9743 let mut harness = probe_harness();
9744
9745 let first = finish_after(&mut harness, Duration::from_secs(8)).await;
9746 let first_latency = match &first {
9747 ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
9748 other => panic!("late answer was not retained: {other:?}"),
9749 };
9750 assert!(harness.handler.observe_module_control_completion(first));
9751
9752 let second = finish_after(&mut harness, Duration::from_secs(11)).await;
9753 let second_latency = match &second {
9754 ModuleControlRpcCompletion::LateHealthAnswer { latency, .. } => *latency,
9755 other => panic!("late answer was not retained: {other:?}"),
9756 };
9757 assert!(harness.handler.observe_module_control_completion(second));
9758
9759 assert_eq!(first_latency, Duration::from_secs(8));
9760 assert_eq!(
9761 second_latency - first_latency,
9762 Duration::from_secs(3),
9763 "latency must grow linearly with the additional stall"
9764 );
9765 let health = harness.module.status().unwrap().health;
9766 assert_eq!(health.late_answer_count, 2);
9767 assert_eq!(health.last_late_answer_latency_ms, Some(11_000));
9768 }
9769
9770 #[tokio::test(start_paused = true)]
9778 async fn late_answer_clears_the_consecutive_failure_streak() {
9779 let mut harness = probe_harness();
9780
9781 time_out_without_answer(&mut harness).await;
9783 harness
9784 .module
9785 .record_health_probe_failure_for_test("[no-answer] test miss")
9786 .unwrap();
9787 assert_eq!(
9788 harness.module.status().unwrap().health.consecutive_failures,
9789 1,
9790 "precondition: the miss must be on the streak before the late answer"
9791 );
9792
9793 let late = finish_after(&mut harness, Duration::from_secs(9)).await;
9795 assert!(matches!(
9796 late,
9797 ModuleControlRpcCompletion::LateHealthAnswer { .. }
9798 ));
9799 assert!(harness.handler.observe_module_control_completion(late));
9800
9801 let health = harness.module.status().unwrap().health;
9802 assert_eq!(
9803 health.consecutive_failures, 0,
9804 "a late answer is an answer: the streak must reset"
9805 );
9806 assert_eq!(health.late_answer_count, 1);
9807 }
9808
9809 #[tokio::test(start_paused = true)]
9810 async fn repeated_serial_probe_cycles_keep_one_tombstone_per_endpoint() {
9811 let mut harness = probe_harness();
9812
9813 for _ in 0..20 {
9814 time_out_without_answer(&mut harness).await;
9815 assert_eq!(
9816 harness.forwarding.health_probe_tombstone_count().unwrap(),
9817 1
9818 );
9819 }
9820 }
9821}
9822
9823#[cfg(test)]
9824mod child_env_tests {
9825 use super::{
9826 apply_child_env, apply_spawn_role, apply_wire_spawn_args, ModuleProtocol, ModuleSpec,
9827 SpawnRole, SupervisorHandle, SPAWN_ROLE_SWAP_CANDIDATE, SUBC_ARG, SUBC_LAUNCH_NONCE_ENV,
9828 SUBC_MODULE_ID_ENV, SUBC_SPAWN_ROLE_ENV,
9829 };
9830 use std::{ffi::OsStr, path::PathBuf};
9831 use tokio::process::Command;
9832
9833 fn spec(env: Vec<(String, String)>) -> ModuleSpec {
9834 ModuleSpec {
9835 module_id: "env-plan".to_string(),
9836 program: PathBuf::from("/nonexistent"),
9837 args: Vec::new(),
9838 env,
9839 reserved: false,
9840 reserved_prefixes: Vec::new(),
9841 protocol: ModuleProtocol::Subc,
9842 overlap: Default::default(),
9843 }
9844 }
9845
9846 #[test]
9860 fn ambient_ck_log_is_removed_and_a_configured_one_survives() {
9861 let mut command = Command::new("/nonexistent");
9862 apply_child_env(&mut command, &spec(Vec::new()));
9863 let removed = command
9864 .as_std()
9865 .get_envs()
9866 .any(|(key, value)| key == OsStr::new("CK_LOG") && value.is_none());
9867 assert!(
9868 removed,
9869 "ambient CK_LOG must be explicitly removed for an unconfigured module"
9870 );
9871
9872 let mut configured = Command::new("/nonexistent");
9873 apply_child_env(
9874 &mut configured,
9875 &spec(vec![("CK_LOG".to_string(), "debug".to_string())]),
9876 );
9877 let effective = configured
9878 .as_std()
9879 .get_envs()
9880 .filter(|(key, _)| *key == OsStr::new("CK_LOG"))
9881 .last()
9882 .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
9883 assert_eq!(
9884 effective,
9885 Some(Some("debug".to_string())),
9886 "a module's configured CK_LOG must survive the ambient removal"
9887 );
9888 }
9889
9890 #[test]
9899 fn protocol_none_spawn_carries_no_subc_argument_and_no_nonce() {
9900 let connection_file = std::path::Path::new("/run/subc-connection.json");
9901 let handle = SupervisorHandle::new();
9902
9903 let mut none_spec = spec(Vec::new());
9904 none_spec.protocol = ModuleProtocol::None;
9905 let mut none = Command::new("/nonexistent");
9906 let none_handoff =
9907 apply_wire_spawn_args(&mut none, &none_spec, Some(connection_file), Some(&handle))
9908 .expect("protocol-none spawn args apply");
9909 assert!(
9910 none_handoff.is_none(),
9911 "protocol:none spawn must not receive a nonce descriptor"
9912 );
9913 assert!(
9914 !none.as_std().get_envs().any(|(key, value)| key
9915 == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV)
9916 && value.is_some()),
9917 "protocol:none spawn must not name a nonce descriptor"
9918 );
9919 let none_args: Vec<String> = none
9920 .as_std()
9921 .get_args()
9922 .map(|a| a.to_string_lossy().into_owned())
9923 .collect();
9924 assert!(
9925 !none_args.iter().any(|a| a == SUBC_ARG),
9926 "protocol:none argv must not carry --subc; got {none_args:?}"
9927 );
9928 let none_has_nonce = none
9929 .as_std()
9930 .get_envs()
9931 .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some());
9932 assert!(
9933 !none_has_nonce,
9934 "protocol:none spawn must not receive a launch nonce"
9935 );
9936 let none_has_module_id = none
9937 .as_std()
9938 .get_envs()
9939 .any(|(key, value)| key == OsStr::new(SUBC_MODULE_ID_ENV) && value.is_some());
9940 assert!(
9941 none_has_module_id,
9942 "SUBC_MODULE_ID is inert and stays on every path"
9943 );
9944 assert!(
9945 handle.spawn_nonce(&none_spec.module_id).is_none(),
9946 "no nonce record for a process that will never present one"
9947 );
9948
9949 let wire_spec = spec(Vec::new());
9951 let mut wire = Command::new("/nonexistent");
9952 let wire_handoff =
9953 apply_wire_spawn_args(&mut wire, &wire_spec, Some(connection_file), Some(&handle))
9954 .expect("subc-wire spawn args apply");
9955 let wire_fd_env = wire
9956 .as_std()
9957 .get_envs()
9958 .find(|(key, _)| *key == OsStr::new(subc_os::LAUNCH_NONCE_FD_ENV))
9959 .and_then(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()));
9960 #[cfg(unix)]
9961 assert_eq!(
9962 wire_fd_env,
9963 Some(wire_handoff.expect("a descriptor handoff").fd_env_value()),
9964 "a subc-wire spawn names the pipe it will receive at descriptor 3"
9965 );
9966 #[cfg(not(unix))]
9967 assert!(wire_handoff.is_none() && wire_fd_env.is_none());
9968 let wire_args: Vec<String> = wire
9969 .as_std()
9970 .get_args()
9971 .map(|a| a.to_string_lossy().into_owned())
9972 .collect();
9973 assert_eq!(
9974 wire_args,
9975 vec![
9976 SUBC_ARG.to_string(),
9977 connection_file.to_string_lossy().into_owned()
9978 ],
9979 "a subc-wire spawn still carries --subc <path>"
9980 );
9981 assert_eq!(
9982 wire.as_std()
9983 .get_envs()
9984 .any(|(key, value)| key == OsStr::new(SUBC_LAUNCH_NONCE_ENV) && value.is_some()),
9985 !cfg!(unix),
9986 "only Windows supplies the environment nonce"
9987 );
9988 assert!(handle.spawn_nonce(&wire_spec.module_id).is_some());
9989 }
9990
9991 #[test]
10001 fn plain_spawn_removes_the_spawn_role_even_when_the_spec_sets_it() {
10002 let role = |command: &Command| {
10003 command
10004 .as_std()
10005 .get_envs()
10006 .filter(|(key, _)| *key == OsStr::new(SUBC_SPAWN_ROLE_ENV))
10007 .last()
10008 .map(|(_, value)| value.map(|v| v.to_string_lossy().into_owned()))
10009 };
10010 let forged = spec(vec![(
10011 SUBC_SPAWN_ROLE_ENV.to_string(),
10012 SPAWN_ROLE_SWAP_CANDIDATE.to_string(),
10013 )]);
10014
10015 let mut plain = Command::new("/nonexistent");
10016 apply_child_env(&mut plain, &forged);
10017 apply_spawn_role(&mut plain, SpawnRole::Plain);
10018 assert_eq!(
10019 role(&plain),
10020 Some(None),
10021 "a plain spawn must remove SUBC_SPAWN_ROLE, whatever the spec says"
10022 );
10023
10024 let mut candidate = Command::new("/nonexistent");
10025 apply_child_env(&mut candidate, &spec(Vec::new()));
10026 apply_spawn_role(&mut candidate, SpawnRole::SwapCandidate);
10027 assert_eq!(
10028 role(&candidate),
10029 Some(Some(SPAWN_ROLE_SWAP_CANDIDATE.to_string()))
10030 );
10031 }
10032
10033 #[test]
10039 fn daemon_private_capture_keys_are_not_passed_to_the_child() {
10040 let mut command = Command::new("/nonexistent");
10041 apply_child_env(
10042 &mut command,
10043 &spec(vec![
10044 (super::CAPTURE_KEEP_ENV.to_string(), "5".to_string()),
10045 ("KEPT".to_string(), "yes".to_string()),
10046 ]),
10047 );
10048 let keys: Vec<String> = command
10049 .as_std()
10050 .get_envs()
10051 .filter(|(_, value)| value.is_some())
10052 .map(|(key, _)| key.to_string_lossy().into_owned())
10053 .collect();
10054 assert!(keys.contains(&"KEPT".to_string()), "got {keys:?}");
10055 assert!(
10056 !keys.contains(&super::CAPTURE_KEEP_ENV.to_string()),
10057 "daemon-private capture key leaked to the child: {keys:?}"
10058 );
10059 }
10060}
10061
10062#[cfg(test)]
10063mod jitter_tests {
10064 use super::jittered_health_delay;
10065 use std::{collections::HashSet, time::Duration};
10066
10067 const FLEET: [&str; 14] = [
10076 "aft",
10077 "alfonso-core",
10078 "magic-context",
10079 "broca",
10080 "thalamus",
10081 "quota",
10082 "engram",
10083 "plexus",
10084 "cerebellum",
10085 "astrocyte",
10086 "synapse",
10087 "subc-mcp",
10088 "cortexkit-credentials",
10089 "subc-federation",
10090 ];
10091
10092 #[test]
10100 fn probe_delays_disperse_across_the_fleet() {
10101 let cadence = Duration::from_secs(30);
10102 let delays: HashSet<Duration> = FLEET
10103 .iter()
10104 .map(|id| jittered_health_delay(id, 0, cadence))
10105 .collect();
10106 assert_eq!(
10107 delays.len(),
10108 FLEET.len(),
10109 "every supervised module must land on its own probe offset"
10110 );
10111 }
10112
10113 #[test]
10119 fn jitter_only_delays_and_stays_within_one_tenth_of_cadence() {
10120 let cadence = Duration::from_secs(30);
10121 let span = cadence / 10;
10122 for id in FLEET {
10123 for probe_index in 0..8 {
10124 let delay = jittered_health_delay(id, probe_index, cadence);
10125 assert!(
10126 delay >= cadence,
10127 "{id}#{probe_index}: jitter must not shorten the cadence"
10128 );
10129 assert!(
10130 delay < cadence + span,
10131 "{id}#{probe_index}: jitter must stay inside one tenth of the cadence"
10132 );
10133 }
10134 }
10135 }
10136
10137 #[test]
10143 fn a_module_offset_is_stable_across_restarts() {
10144 let cadence = Duration::from_secs(30);
10145 for id in FLEET {
10146 assert_eq!(
10147 jittered_health_delay(id, 0, cadence),
10148 jittered_health_delay(id, 0, cadence),
10149 "{id}: the same module and probe index must produce the same offset"
10150 );
10151 }
10152 }
10153
10154 #[test]
10156 fn zero_cadence_yields_zero_delay() {
10157 assert_eq!(
10158 jittered_health_delay("aft", 0, Duration::ZERO),
10159 Duration::ZERO
10160 );
10161 }
10162}
10163
10164#[cfg(all(test, target_os = "linux"))]
10165mod cgroup_placement_tests {
10166 use super::{
10167 apply_cgroup_placement, remove_module_cgroup, ModuleProtocol, ModuleSpec, SuperviseError,
10168 SupervisedChild,
10169 };
10170 use crate::stderr_tail::{StderrRing, StderrTailConfig};
10171 use std::{
10172 fs, io,
10173 path::{Path, PathBuf},
10174 sync::{Arc, Mutex},
10175 };
10176 use subc_test_support::TestTempDir;
10177 use tokio::process::Command;
10178
10179 #[test]
10180 fn failed_parent_cgroup_open_is_a_cgroup_supervision_error() {
10181 let path = Path::new("/definitely-missing-subc-cgroup");
10182 let mut command = Command::new("true");
10183 let error = apply_cgroup_placement(
10184 &mut command,
10185 &ModuleSpec {
10186 module_id: "broken-cgroup".to_string(),
10187 program: PathBuf::from("true"),
10188 args: Vec::new(),
10189 env: Vec::new(),
10190 reserved: false,
10191 reserved_prefixes: Vec::new(),
10192 protocol: ModuleProtocol::Subc,
10193 overlap: Default::default(),
10194 },
10195 path,
10196 )
10197 .expect_err("a parent cgroup open failure must reject the supervised spawn");
10198 let reason = error.to_string();
10199
10200 assert!(
10201 matches!(error, SuperviseError::Cgroup { .. }),
10202 "parent cgroup open must be reported as a cgroup supervision error: {reason}"
10203 );
10204 assert!(
10205 reason.contains("/definitely-missing-subc-cgroup/cgroup.procs"),
10206 "parent cgroup open failure must name cgroup.procs: {reason}"
10207 );
10208 }
10209
10210 #[tokio::test]
10211 async fn reaping_a_child_removes_its_empty_module_cgroup() {
10212 let root = TestTempDir::new("supervisor-reap-cgroup");
10213 fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
10214 let placement = subc_cgroup::prepare_at(&root)
10215 .expect("prepare scratch cgroup root")
10216 .expect("scratch root has a cgroup.procs marker");
10217 let module_id = "reaped-module";
10218 let module = placement
10219 .module_path(module_id)
10220 .expect("create scratch module cgroup");
10221 let child = Command::new("true")
10222 .spawn()
10223 .expect("spawn short-lived child");
10224 let pid = child.id().expect("spawned child has pid");
10225 let mut child = SupervisedChild {
10226 child,
10227 module_id: module_id.to_string(),
10228 cgroup_placement: Some(placement),
10229 stdout_pump: None,
10230 stderr_pump: None,
10231 stderr_ring: Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default()))),
10232 spawned_at_ms: 0,
10233 spawned_from: PathBuf::from("true"),
10234 spawned_file_identity: None,
10235 process_start_time: None,
10236 process_identity: None,
10237 pid,
10238 roster_guard: None,
10239 };
10240
10241 child.wait().await.expect("reap short-lived child");
10242
10243 assert!(
10244 !module.exists(),
10245 "reaping the supervised child must remove its empty cgroup"
10246 );
10247 }
10248
10249 #[test]
10250 fn non_empty_cgroup_removal_is_reported_without_blocking_teardown() {
10251 let root = TestTempDir::new("supervisor-non-empty-cgroup");
10252 fs::write(root.join("cgroup.procs"), b"").expect("write scratch cgroup marker");
10253 let placement = subc_cgroup::prepare_at(&root)
10254 .expect("prepare scratch cgroup root")
10255 .expect("scratch root has a cgroup.procs marker");
10256 let module = placement
10257 .module_path("surviving-module")
10258 .expect("create scratch module cgroup");
10259 fs::write(module.join("surviving-process"), b"still present")
10260 .expect("make scratch cgroup non-empty");
10261 let (logs, _guard) = crate::router::test_log::log_capture(tracing::Level::WARN);
10262
10263 remove_module_cgroup(&placement, "surviving-module");
10264
10265 let logs = crate::router::test_log::captured_logs(&logs);
10266 assert!(
10267 module.exists(),
10268 "failed removal must leave the cgroup intact"
10269 );
10270 assert!(
10271 logs.contains("could not remove module cgroup after process exit; continuing teardown")
10272 && logs.contains("surviving-module"),
10273 "best-effort removal must report the failure without returning it: {logs}"
10274 );
10275 }
10276
10277 #[test]
10278 fn cgroup_pre_exec_spawn_failure_names_the_cgroup_path() {
10279 let cgroup_path = PathBuf::from("/sys/fs/cgroup/subc-modules/broken-module");
10280 let reason = SuperviseError::Spawn {
10281 program: PathBuf::from("/bin/true"),
10282 source: io::Error::from_raw_os_error(13),
10283 cgroup_path: Some(cgroup_path.clone()),
10284 }
10285 .to_string();
10286
10287 assert!(
10288 reason.contains(&cgroup_path.display().to_string()),
10289 "a pre_exec spawn failure must name the cgroup path: {reason}"
10290 );
10291 }
10292}
10293
10294#[cfg(test)]
10295mod spawn_subscriber_lag_tests {
10296 use super::*;
10297
10298 #[tokio::test]
10303 async fn lagged_spawn_subscriber_receives_a_terminal_lagged_error_after_its_queued_frames() {
10304 let feed = SpawnEventFeed::default();
10305 feed.configure_incarnation("lag-incarnation".to_string());
10306 let (tx, mut rx) = mpsc::channel(1);
10309 feed.subscribe(ConnectionId::new(1), 7, 1, None, FrameSink::new(tx))
10310 .expect("subscribe");
10311 let emitted = SPAWN_SUBSCRIBER_BUFFER + 16;
10312 for index in 0..emitted {
10313 feed.emit_spawned(&format!("lag-module-{index}"), 1000, 0);
10314 tokio::task::yield_now().await;
10317 }
10318 assert_eq!(
10319 feed.subscriber_count(),
10320 0,
10321 "the lagged subscriber must be removed"
10322 );
10323
10324 let mut data = Vec::new();
10325 let mut last = None;
10326 loop {
10327 let next = tokio::time::timeout(Duration::from_secs(5), rx.recv())
10328 .await
10329 .expect("the forwarder must finish once the subscriber is dropped");
10330 let Some(outbound) = next else { break };
10331 let frame = outbound.frame;
10332 if frame.header.ty == FrameType::StreamData {
10333 assert!(last.is_none(), "no data may follow the terminal frame");
10334 let event: SpawnEvent = serde_json::from_slice(&frame.body).unwrap();
10335 data.push(event.cursor.seq);
10336 } else {
10337 assert!(last.is_none(), "exactly one terminal frame");
10338 last = Some(frame);
10339 }
10340 }
10341 assert!(!data.is_empty(), "queued frames drain before the terminal");
10342 for pair in data.windows(2) {
10343 assert_eq!(
10344 pair[1],
10345 pair[0] + 1,
10346 "queued frames arrive dense and in order"
10347 );
10348 }
10349 let terminal = last.expect("a lagged subscriber must receive a terminal frame");
10350 assert_eq!(terminal.header.ty, FrameType::Error);
10351 assert_eq!(terminal.header.corr, 7);
10352 let body: subc_protocol::ErrorBody = serde_json::from_slice(&terminal.body).unwrap();
10353 assert_eq!(body.code, SPAWN_SUBSCRIBER_LAGGED_CODE);
10354 let detail = body.detail.expect("lagged error carries detail");
10355 assert_eq!(
10356 detail["first_undelivered_cursor"]["seq"],
10357 data.last().unwrap() + 1,
10358 "the named cursor is the first event the subscriber did not receive"
10359 );
10360 assert_eq!(
10361 detail["first_undelivered_cursor"]["daemon_incarnation"],
10362 "lag-incarnation"
10363 );
10364 }
10365}
10366
10367#[cfg(test)]
10368mod terminal_history_read_concurrency_tests {
10369 use super::*;
10370 use crate::terminal_journal::read_pause;
10371 use std::sync::mpsc as std_mpsc;
10372 use subc_test_support::TestTempDir;
10373
10374 fn journaled_ring(
10375 journal: &Arc<crate::terminal_journal::TerminalJournal>,
10376 ) -> Arc<Mutex<TerminalRing>> {
10377 Arc::new(Mutex::new(
10378 TerminalRing::new(TerminalRingConfig::default(), 1)
10379 .with_journal(Some(Arc::clone(journal))),
10380 ))
10381 }
10382
10383 fn crash(at_ms: u64) -> ExitReport {
10384 ExitReport {
10385 kind: ExitKind::Crash,
10386 code: Some(1),
10387 signal: None,
10388 at_ms,
10389 }
10390 }
10391
10392 fn record_within(
10395 module_id: &'static str,
10396 ring: &Arc<Mutex<TerminalRing>>,
10397 at_ms: u64,
10398 bound: Duration,
10399 ) -> bool {
10400 let ring = Arc::clone(ring);
10401 let (done, done_rx) = std_mpsc::channel();
10402 std::thread::spawn(move || {
10403 record_terminal(
10404 module_id,
10405 &ring,
10406 &SpawnEventFeed::default(),
10407 &crash(at_ms),
10408 TerminalDisposition::Restarting,
10409 );
10410 let _ = done.send(());
10411 });
10412 done_rx.recv_timeout(bound).is_ok()
10413 }
10414
10415 #[test]
10420 fn exits_recorded_during_a_paused_history_read_are_not_blocked_or_half_merged() {
10421 let dir = TestTempDir::new("terminal-history-concurrent-read");
10422 let path = dir.join("terminals.jsonl");
10423 let journal = Arc::new(crate::terminal_journal::TerminalJournal::open(
10424 path.clone(),
10425 "daemon".into(),
10426 ));
10427 let reader_ring = journaled_ring(&journal);
10428 let other_ring = journaled_ring(&journal);
10429 assert!(record_within(
10430 "reader-module",
10431 &reader_ring,
10432 10,
10433 Duration::from_secs(5)
10434 ));
10435
10436 let (started, release) = read_pause::install(&path);
10437 let reading = {
10438 let ring = Arc::clone(&reader_ring);
10439 std::thread::spawn(move || durable_terminal_history_of(&ring, "reader-module"))
10440 };
10441 started
10442 .recv_timeout(Duration::from_secs(5))
10443 .expect("the history read reached its pause");
10444
10445 let bound = Duration::from_secs(1);
10446 assert!(
10447 record_within("other-module", &other_ring, 20, bound),
10448 "another module's exit waited on a history read (journal writer held)"
10449 );
10450 assert!(
10451 record_within("reader-module", &reader_ring, 30, bound),
10452 "the read module's own exit waited on its history read (ring held)"
10453 );
10454
10455 drop(release);
10456 let paused = reading.join().unwrap();
10457 assert_eq!(
10458 paused.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
10459 vec![10],
10460 "an exit recorded after the read began lands in neither half of it"
10461 );
10462 assert_eq!(paused.journal_skipped_lines, 0);
10463 assert_eq!(paused.journal_read_errors, 0);
10464
10465 let after = durable_terminal_history_of(&reader_ring, "reader-module");
10466 assert_eq!(
10467 after.entries.iter().map(|e| e.at_ms).collect::<Vec<_>>(),
10468 vec![10, 30],
10469 "the next read merges ring and journal with no duplicate"
10470 );
10471 assert_eq!(after.journal_skipped_lines, 0);
10472 }
10473}
10474
10475#[cfg(test)]
10480mod stderr_settle_tests {
10481 use std::{
10482 future::Future,
10483 io,
10484 pin::Pin,
10485 sync::{Arc, Mutex},
10486 task::{Context, Poll},
10487 time::Duration,
10488 };
10489
10490 use tokio::{
10491 io::{AsyncRead, ReadBuf},
10492 sync::oneshot,
10493 time::Instant,
10494 };
10495
10496 use super::{settle_stderr_pump, StderrPump};
10497 use crate::stderr_tail::{
10498 pump_stderr_to, untimed, CaptureState, OutputSink, StderrRing, StderrTailConfig, TailEntry,
10499 };
10500
10501 const BOUND: Duration = Duration::from_millis(250);
10502
10503 struct HeldReader {
10507 before: Option<Vec<u8>>,
10508 gate: Option<oneshot::Receiver<()>>,
10509 after: io::Cursor<Vec<u8>>,
10510 }
10511
10512 impl AsyncRead for HeldReader {
10513 fn poll_read(
10514 mut self: Pin<&mut Self>,
10515 cx: &mut Context<'_>,
10516 buf: &mut ReadBuf<'_>,
10517 ) -> Poll<io::Result<()>> {
10518 if let Some(bytes) = self.before.take() {
10519 buf.put_slice(&bytes);
10520 return Poll::Ready(Ok(()));
10521 }
10522 if let Some(gate) = self.gate.as_mut() {
10523 match Pin::new(gate).poll(cx) {
10524 Poll::Pending => return Poll::Pending,
10525 Poll::Ready(_) => self.gate = None,
10526 }
10527 }
10528 Pin::new(&mut self.after).poll_read(cx, buf)
10529 }
10530 }
10531
10532 struct DiscardSink;
10533
10534 impl OutputSink for DiscardSink {
10535 fn write_line(&mut self, _line: &[u8]) {}
10536 }
10537
10538 fn line(text: &str) -> TailEntry {
10539 TailEntry::Line {
10540 text: text.to_string(),
10541 truncated: false,
10542 at_ms: None,
10543 }
10544 }
10545
10546 fn lock(ring: &Arc<Mutex<StderrRing>>) -> std::sync::MutexGuard<'_, StderrRing> {
10547 ring.lock().unwrap()
10548 }
10549
10550 fn held_pump(
10554 ring: &Arc<Mutex<StderrRing>>,
10555 before: &str,
10556 after: &str,
10557 ) -> (StderrPump, oneshot::Sender<()>) {
10558 let generation = lock(ring).begin_process();
10559 let (release, gate) = oneshot::channel();
10560 let reader = HeldReader {
10561 before: Some(before.as_bytes().to_vec()),
10562 gate: Some(gate),
10563 after: io::Cursor::new(after.as_bytes().to_vec()),
10564 };
10565 let task = tokio::spawn(pump_stderr_to(
10566 reader,
10567 Arc::clone(ring),
10568 generation,
10569 DiscardSink,
10570 ));
10571 (StderrPump { task, generation }, release)
10572 }
10573
10574 async fn wait_until(ring: &Arc<Mutex<StderrRing>>, done: impl Fn(&StderrRing) -> bool) {
10575 for _ in 0..1000 {
10576 if done(&lock(ring)) {
10577 return;
10578 }
10579 tokio::time::sleep(Duration::from_millis(1)).await;
10580 }
10581 panic!(
10582 "ring never reached the expected state: {:?}",
10583 lock(ring).snapshot(None, None)
10584 );
10585 }
10586
10587 #[tokio::test(start_paused = true)]
10588 async fn a_crash_line_the_reader_had_not_reached_by_the_bound_is_kept_before_the_restart() {
10589 let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
10590 let (pump, release) = held_pump(&ring, "booting\n", "config error: missing storage\n");
10591
10592 settle_stderr_pump("crasher", &ring, pump, BOUND).await;
10593 let before_release = lock(&ring).snapshot(None, None);
10594 assert!(
10595 matches!(before_release.capture, CaptureState::Incomplete { .. }),
10596 "a reader that has not reached EOF cannot claim a whole tail: {before_release:?}"
10597 );
10598
10599 let next = lock(&ring).begin_process();
10602 lock(&ring).push_line_from(next, "next process booting");
10603 release.send(()).unwrap();
10604 wait_until(&ring, |ring| {
10605 ring.snapshot(None, None).capture == CaptureState::Captured
10606 })
10607 .await;
10608
10609 assert_eq!(
10610 untimed(lock(&ring).snapshot(None, None).entries),
10611 vec![
10612 line("booting"),
10613 line("config error: missing storage"),
10614 TailEntry::ProcessStart,
10615 line("next process booting"),
10616 ],
10617 "the crash's last line must survive a slow reader and stay in the crashed process's section"
10618 );
10619 }
10620
10621 #[tokio::test(start_paused = true)]
10622 async fn a_pipe_held_open_by_a_descendant_reads_incomplete_without_delaying_the_restart_past_the_bound(
10623 ) {
10624 let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
10625 let (pump, _held) = held_pump(&ring, "parent exiting\n", "");
10628
10629 let started = Instant::now();
10630 settle_stderr_pump("orphaning", &ring, pump, BOUND).await;
10631 assert_eq!(
10632 started.elapsed(),
10633 BOUND,
10634 "the restart must wait exactly the bound for a pipe that stays open, no longer"
10635 );
10636
10637 let next = lock(&ring).begin_process();
10638 lock(&ring).push_line_from(next, "next process booting");
10639 tokio::time::sleep(Duration::from_secs(60)).await;
10640
10641 let snapshot = lock(&ring).snapshot(None, None);
10642 match &snapshot.capture {
10643 CaptureState::Incomplete { reason } => assert!(
10644 reason.contains("had not reached EOF") && reason.contains("250ms"),
10645 "the reason must say what is missing and after how long: {reason}"
10646 ),
10647 other => panic!("expected Incomplete while the pipe is held open, got {other:?}"),
10648 }
10649 assert_eq!(
10650 untimed(snapshot.entries),
10651 vec![
10652 line("parent exiting"),
10653 TailEntry::ProcessStart,
10654 line("next process booting"),
10655 ]
10656 );
10657 }
10658
10659 #[tokio::test(start_paused = true)]
10660 async fn a_reader_that_reaches_eof_within_the_bound_leaves_the_tail_captured() {
10661 let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
10662 let (pump, release) = held_pump(&ring, "one\n", "two\n");
10663 release.send(()).unwrap();
10664
10665 settle_stderr_pump("clean", &ring, pump, BOUND).await;
10666
10667 let snapshot = lock(&ring).snapshot(None, None);
10668 assert_eq!(snapshot.capture, CaptureState::Captured);
10669 assert_eq!(untimed(snapshot.entries), vec![line("one"), line("two")]);
10670 }
10671}
10672
10673#[cfg(all(test, windows))]
10687mod job_containment_tests {
10688 use super::*;
10689 use std::{
10690 path::{Path, PathBuf},
10691 sync::{Arc, Mutex},
10692 time::{Duration, Instant},
10693 };
10694 use subc_test_support::TestTempDir;
10695
10696 fn stub_path() -> PathBuf {
10702 let mut path = std::env::current_exe().expect("current_exe available in tests");
10703 path.pop();
10704 path.pop();
10705 path.push("fake-aft-stub.exe");
10706 assert!(
10707 path.exists(),
10708 "fake-aft-stub not built at {}: run `cargo test -p subc-core` (which builds \
10709 [[bin]] targets) rather than `cargo test -p subc-core --lib` (which does not)",
10710 path.display()
10711 );
10712 path
10713 }
10714
10715 fn read_grandchild_pid(path: &Path) -> u32 {
10717 let deadline = Instant::now() + Duration::from_secs(10);
10718 loop {
10719 if let Ok(contents) = std::fs::read_to_string(path) {
10720 if let Ok(pid) = contents.trim().parse() {
10721 return pid;
10722 }
10723 }
10724 assert!(
10725 Instant::now() < deadline,
10726 "the stub never recorded a grandchild pid at {}",
10727 path.display()
10728 );
10729 std::thread::sleep(Duration::from_millis(10));
10730 }
10731 }
10732
10733 struct Fixture {
10736 _dir: TestTempDir,
10737 module_id: String,
10738 grandchild: u32,
10739 child: Option<SupervisedChild>,
10740 registry: Arc<Registry>,
10741 snapshot: Arc<Mutex<SupervisorSnapshot>>,
10742 terminal_ring: Arc<Mutex<TerminalRing>>,
10743 spawn_events: SpawnEventFeed,
10744 }
10745
10746 fn fixture(label: &str, module_id: &str) -> Fixture {
10747 let dir = TestTempDir::new(label);
10748 let pid_file = dir.join("grandchild.pid");
10749 let supervisor = Supervisor::new(
10750 Arc::new(Registry::default()),
10751 RestartPolicy::new(3, Duration::ZERO),
10752 );
10753 let runtime = supervisor.runtime_config();
10754 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
10755 let spec = ModuleSpec {
10756 module_id: module_id.to_string(),
10757 program: stub_path(),
10758 args: Vec::new(),
10762 env: vec![
10763 ("FAKE_AFT_NEVER_CONNECT".to_string(), "1".to_string()),
10764 (
10765 "FAKE_AFT_GRANDCHILD_PID_FILE".to_string(),
10766 pid_file.display().to_string(),
10767 ),
10768 ],
10769 reserved: false,
10770 reserved_prefixes: Vec::new(),
10771 protocol: ModuleProtocol::Subc,
10772 overlap: Default::default(),
10773 };
10774 let child = spawn_and_mark_running(&spec, &runtime, &snapshot)
10775 .expect("spawn the supervised fixture");
10776 let grandchild = read_grandchild_pid(&pid_file);
10777 Fixture {
10778 _dir: dir,
10779 module_id: module_id.to_string(),
10780 grandchild,
10781 child: Some(child),
10782 registry: Arc::new(Registry::default()),
10783 snapshot,
10784 terminal_ring: Arc::clone(&runtime.terminal_ring),
10785 spawn_events: SpawnEventFeed::default(),
10786 }
10787 }
10788
10789 impl Fixture {
10790 async fn drain(&mut self) {
10792 let child = self
10793 .child
10794 .take()
10795 .expect("the fixture child is still present");
10796 drain_child_to_state(
10797 &self.module_id,
10798 ModuleProtocol::Subc,
10799 StopNotice::NotSent,
10802 &self.registry,
10803 None,
10804 &self.snapshot,
10805 &self.terminal_ring,
10806 &self.spawn_events,
10807 child,
10808 Duration::from_millis(500),
10809 ModuleState::Stopped,
10810 Some(false),
10811 )
10812 .await
10813 .expect("drain the supervised fixture");
10814 }
10815 }
10816
10817 #[tokio::test]
10823 async fn teardown_reaps_the_grandchild() {
10824 let mut fixture = fixture("teardown-grandchild", "tree-teardown");
10825 let grandchild = fixture.grandchild;
10826
10827 assert!(
10828 subc_jobobject::process_exists(grandchild),
10829 "grandchild {grandchild} must be alive before teardown, or this proves nothing"
10830 );
10831
10832 fixture.drain().await;
10833
10834 assert!(
10835 subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
10836 "grandchild {grandchild} outlived module teardown: the tree was not contained"
10837 );
10838 }
10839
10840 #[test]
10853 fn an_uncontained_grandchild_survives_a_direct_child_kill() {
10854 let dir = TestTempDir::new("teardown-uncontained");
10855 let pid_file = dir.join("grandchild.pid");
10856 let mut child = std::process::Command::new(stub_path())
10857 .env("FAKE_AFT_NEVER_CONNECT", "1")
10858 .env(
10859 "FAKE_AFT_GRANDCHILD_PID_FILE",
10860 pid_file.display().to_string(),
10861 )
10862 .stdin(std::process::Stdio::null())
10863 .stdout(std::process::Stdio::null())
10864 .stderr(std::process::Stdio::null())
10865 .spawn()
10866 .expect("spawn the uncontained fixture");
10867 let grandchild = read_grandchild_pid(&pid_file);
10868
10869 child.kill().expect("kill the direct child");
10871 let _ = child.wait();
10872
10873 assert!(
10874 subc_jobobject::process_exists(grandchild),
10875 "grandchild {grandchild} died with the direct child, so this control no longer \
10876 distinguishes contained from uncontained teardown and the regression test is \
10877 passing vacuously"
10878 );
10879
10880 kill_tree(grandchild);
10883 }
10884
10885 #[tokio::test]
10894 async fn dropping_containment_reaps_the_grandchild() {
10895 let mut fixture = fixture("drop-containment", "tree-drop");
10896 let grandchild = fixture.grandchild;
10897
10898 assert!(subc_jobobject::process_exists(grandchild));
10899
10900 fixture.child.as_mut().expect("child present").job = None;
10902
10903 assert!(
10904 subc_jobobject::wait_for_process_exit(grandchild, Duration::from_secs(10)),
10905 "grandchild {grandchild} survived the containment handle closing, so a daemon \
10906 crash would leave the tree behind"
10907 );
10908 }
10909
10910 fn kill_tree(pid: u32) {
10912 let _ = std::process::Command::new("taskkill.exe")
10913 .args(["/PID", &pid.to_string(), "/T", "/F"])
10914 .stdin(std::process::Stdio::null())
10915 .stdout(std::process::Stdio::null())
10916 .stderr(std::process::Stdio::null())
10917 .status();
10918 assert!(
10919 subc_jobobject::wait_for_process_exit(pid, Duration::from_secs(10)),
10920 "could not clean up grandchild {pid}"
10921 );
10922 }
10923}
10924
10925#[cfg(all(test, unix))]
10929mod launch_nonce_descriptor_tests {
10930 use super::{ChildRoster, ModuleProtocol, ModuleSpec, SupervisorHandle};
10931 use crate::stderr_tail::{StderrRing, StderrTailConfig};
10932 use std::{
10933 path::PathBuf,
10934 sync::{Arc, Mutex},
10935 time::{Duration, Instant},
10936 };
10937 use subc_test_support::TestTempDir;
10938
10939 async fn probe(role: super::SpawnRole) {
10940 let scratch = TestTempDir::new("launch-nonce-descriptor");
10941 let fd_copy = scratch.join("from-descriptor");
10942 let env_copy = scratch.join("environment");
10943 let script = format!(
10944 "env > '{env}.tmp' && mv '{env}.tmp' '{env}'; cat <&3 > '{fd}.tmp' && mv '{fd}.tmp' '{fd}'; sleep 30",
10945 fd = fd_copy.display(), env = env_copy.display(),
10946 );
10947 let xdg = |name: &str| (name.to_string(), scratch.join(name).display().to_string());
10948 let spec = ModuleSpec {
10949 module_id: "nonce-descriptor-probe".to_string(),
10950 program: PathBuf::from("/bin/sh"),
10951 args: vec!["-c".to_string(), script],
10952 env: vec![
10953 xdg("XDG_DATA_HOME"),
10954 xdg("XDG_RUNTIME_DIR"),
10955 xdg("XDG_CONFIG_HOME"),
10956 ("SUBC_LAUNCH_NONCE".to_string(), "must-not-leak".to_string()),
10957 ],
10958 reserved: true,
10959 reserved_prefixes: Vec::new(),
10960 protocol: ModuleProtocol::Subc,
10961 overlap: Default::default(),
10962 };
10963 let handle = SupervisorHandle::new();
10964 let ring = Arc::new(Mutex::new(StderrRing::new(StderrTailConfig::default())));
10965 let roster = ChildRoster::default();
10966 let child = super::spawn_child_in_slot(
10967 &spec,
10968 None,
10969 Some(&handle),
10970 &ring,
10971 None,
10972 &roster,
10973 #[cfg(target_os = "linux")]
10974 None,
10975 role,
10976 matches!(role, super::SpawnRole::SwapCandidate),
10977 )
10978 .expect("spawn probe");
10979 let deadline = Instant::now() + Duration::from_secs(10);
10980 while !(fd_copy.exists() && env_copy.exists()) {
10981 assert!(Instant::now() < deadline, "probe never wrote its copies");
10982 tokio::time::sleep(Duration::from_millis(20)).await;
10983 }
10984 let nonce = std::fs::read_to_string(fd_copy).unwrap();
10985 assert!(!nonce.is_empty());
10986 let environment = std::fs::read_to_string(env_copy).unwrap();
10987 assert!(environment
10988 .lines()
10989 .any(|line| line.starts_with("SUBC_LAUNCH_NONCE_FD=3:")));
10990 let copy = environment
10991 .lines()
10992 .find_map(|line| line.strip_prefix("SUBC_LAUNCH_NONCE="));
10993 assert_eq!(
10994 copy, None,
10995 "Unix children must never receive the environment nonce"
10996 );
10997 if matches!(role, super::SpawnRole::Plain) {
10998 assert_eq!(
10999 handle.spawn_nonce(&spec.module_id).as_deref(),
11000 Some(nonce.as_str())
11001 );
11002 }
11003 drop(child);
11004 }
11005
11006 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
11007 async fn a_spawned_module_receives_its_nonce_only_on_descriptor_3() {
11008 probe(super::SpawnRole::Plain).await;
11009 }
11010
11011 #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
11012 async fn swap_candidate_receives_its_nonce_only_on_descriptor_3() {
11013 probe(super::SpawnRole::SwapCandidate).await;
11014 }
11015}
11016
11017#[cfg(all(test, target_os = "linux"))]
11018mod cgroup_containment_tests {
11019 use super::*;
11020 use subc_test_support::TestTempDir;
11021
11022 fn running(pid: u32) -> bool {
11023 std::fs::read_to_string(format!("/proc/{pid}/stat"))
11025 .ok()
11026 .and_then(|stat| {
11027 stat.rsplit_once(") ")
11028 .map(|(_, rest)| rest.starts_with('Z'))
11029 })
11030 .is_some_and(|zombie| !zombie)
11031 }
11032
11033 #[tokio::test]
11034 async fn linux_teardown_reaps_the_grandchild() {
11035 teardown_tree("linux_teardown_reaps_the_grandchild", false).await;
11036 }
11037
11038 #[tokio::test]
11039 async fn linux_shutdown_straggler_reaps_the_grandchild() {
11040 teardown_tree("linux_shutdown_straggler_reaps_the_grandchild", true).await;
11041 }
11042
11043 async fn teardown_tree(test_name: &str, shutdown: bool) {
11044 let dir = TestTempDir::new(test_name);
11045 let root = PathBuf::from(format!(
11046 "/sys/fs/cgroup/subc-tree-test-{test_name}-{}-{}",
11047 std::process::id(),
11048 unix_ms_now()
11049 ));
11050 if let Err(error) = std::fs::create_dir(&root) {
11051 assert!(
11052 std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
11053 "required cgroup test cannot execute: {error}"
11054 );
11055 eprintln!(
11056 "SKIP {test_name}: no writable delegated cgroup at {}: {error}",
11057 root.display()
11058 );
11059 return;
11060 }
11061 let placement = subc_cgroup::prepare_at(&root)
11062 .expect("prepare isolated kernel cgroup")
11063 .expect("isolated cgroup is delegated");
11064 let module_id = "tree-teardown";
11065 let module = placement
11066 .module_path(module_id)
11067 .expect("create isolated module cgroup");
11068 if !module.join("cgroup.kill").exists() {
11069 std::fs::remove_dir(&module).unwrap();
11070 std::fs::remove_dir(root.join("subc-modules")).unwrap();
11071 std::fs::remove_dir(&root).unwrap();
11072 assert!(
11073 std::env::var_os("SUBC_REQUIRE_CGROUP_TEST").is_none(),
11074 "required cgroup.kill interface unavailable"
11075 );
11076 eprintln!("SKIP {test_name}: cgroup.kill unavailable (kernel < 5.14)");
11077 return;
11078 }
11079 let supervisor = Supervisor::new(
11080 Arc::new(Registry::default()),
11081 RestartPolicy::new(3, Duration::ZERO),
11082 )
11083 .with_cgroup_placement(Some(placement));
11084 let mut runtime = supervisor.runtime_config();
11085 runtime.child_roster = runtime
11086 .child_roster
11087 .for_module(Arc::new(Mutex::new(Duration::from_millis(100))));
11088 let snapshot = Arc::new(Mutex::new(SupervisorSnapshot::starting()));
11089 let pid_file = dir.join("grandchild.pid");
11090 let spec = ModuleSpec {
11091 module_id: module_id.to_string(),
11092 program: PathBuf::from("/bin/sh"),
11093 args: vec![
11094 "-c".into(),
11095 "trap '' TERM; sleep 600 & echo $! > \"$1\"; wait".into(),
11096 "fixture".into(),
11097 pid_file.display().to_string(),
11098 ],
11099 env: ["XDG_DATA_HOME", "XDG_RUNTIME_DIR", "XDG_CONFIG_HOME"]
11100 .into_iter()
11101 .map(|key| (key.to_string(), dir.display().to_string()))
11102 .collect(),
11103 reserved: false,
11104 reserved_prefixes: Vec::new(),
11105 protocol: ModuleProtocol::None,
11106 overlap: Default::default(),
11107 };
11108 let child =
11109 spawn_and_mark_running(&spec, &runtime, &snapshot).expect("spawn supervised tree");
11110 let deadline = tokio::time::Instant::now() + Duration::from_secs(10);
11111 let grandchild: u32 = loop {
11112 if let Ok(contents) = std::fs::read_to_string(&pid_file) {
11113 if let Ok(pid) = contents.trim().parse() {
11114 break pid;
11115 }
11116 }
11117 assert!(
11118 tokio::time::Instant::now() < deadline,
11119 "grandchild pid was not recorded"
11120 );
11121 tokio::time::sleep(Duration::from_millis(10)).await;
11122 };
11123 assert!(
11124 running(grandchild),
11125 "grandchild must be alive before teardown"
11126 );
11127 if shutdown {
11128 let mut child = child;
11129 crate::child_roster::end_children_for_daemon_shutdown(
11130 &runtime.child_roster,
11131 false,
11132 std::future::pending(),
11133 )
11134 .await;
11135 child.wait().await.expect("reap shutdown straggler");
11136 } else {
11137 drain_child_to_state(
11138 module_id,
11139 ModuleProtocol::None,
11140 StopNotice::NotSent,
11141 &Registry::default(),
11142 None,
11143 &snapshot,
11144 &runtime.terminal_ring,
11145 &SpawnEventFeed::default(),
11146 child,
11147 Duration::from_millis(100),
11148 ModuleState::Stopped,
11149 Some(false),
11150 )
11151 .await
11152 .expect("real supervisor teardown");
11153 }
11154 let deadline = tokio::time::Instant::now() + Duration::from_secs(5);
11155 while running(grandchild) && tokio::time::Instant::now() < deadline {
11156 tokio::time::sleep(Duration::from_millis(10)).await;
11157 }
11158 let survived = running(grandchild);
11159 if survived {
11161 let pid = rustix::process::Pid::from_raw(grandchild as i32).unwrap();
11162 let _ = rustix::process::kill_process(pid, rustix::process::Signal::KILL);
11163 tokio::time::sleep(Duration::from_millis(100)).await;
11164 }
11165 if module.exists() {
11166 std::fs::remove_dir(&module).expect("remove empty module cgroup");
11167 }
11168 std::fs::remove_dir(root.join("subc-modules")).unwrap();
11169 std::fs::remove_dir(&root).unwrap();
11170 assert!(
11171 !survived,
11172 "grandchild {grandchild} outlived module teardown"
11173 );
11174 eprintln!("EXECUTED {test_name}: grandchild {grandchild} killed in isolated cgroup");
11175 }
11176}