Skip to main content

mj_controller/pollers/
lifecycle.rs

1use super::*;
2
3pub enum LifecycleSuccess {
4    Created,
5    Resumed {
6        profile_id: String,
7        target_id: String,
8    },
9    Moved(mj_core::state::MoveOutcome),
10    Closed,
11    ForceStopped,
12    DestroyedStopped,
13    ForceDestroyed,
14}
15
16pub struct LifecycleUpdate {
17    pub session_id: String,
18    pub result: std::result::Result<LifecycleSuccess, String>,
19    pub deferred_cleanup: bool,
20}
21
22/// Whether a close stopped partway and left the record mid-close with its
23/// target still present. Such a record cannot be closed again from the start:
24/// its worker is gone, so only recovery can finish it.
25pub fn is_interrupted_close(session: &SessionRecord) -> bool {
26    matches!(
27        session.state,
28        SessionState::Closing | SessionState::Destroying
29    ) && session.target.is_some()
30}
31
32pub fn interrupted_suspend_session_ids(controller: &Controller) -> Vec<String> {
33    controller
34        .state
35        .sessions
36        .values()
37        .filter(|session| is_interrupted_close(session))
38        .map(|session| session.id.clone())
39        .collect()
40}
41
42/// Why a record left in an in-flight lifecycle state has nobody to finish it,
43/// in words the user reads in `mj sessions` and the TUI.
44///
45/// Every in-flight state needs an owner that will complete it. A durable move
46/// intent owns its session, [`is_interrupted_close`] owns a close or teardown
47/// that still holds its target, and
48/// `database::recover_interrupted_checkpointing_sessions` returns an
49/// interrupted `Checkpointing` record to `Running` before the controller
50/// loads. What is left is a record whose operation died with the process, and
51/// it has to say so instead of waiting forever.
52///
53/// `None` means the state needs no reconciliation; callers exclude the owned
54/// sessions before asking.
55pub fn interrupted_lifecycle_cause(session: &SessionRecord) -> Option<String> {
56    match session.state {
57        // Provisioning has no durable operation behind it. Whatever the dead
58        // provision created is not named by this record, so the resource is
59        // recovered through `mj recover scan`, which can see it again once the
60        // record is no longer in flight.
61        SessionState::Provisioning => Some(
62            "the daemon stopped while this session was provisioning; anything it created \
63             is offered by `mj recover scan`"
64                .to_owned(),
65        ),
66        // An interrupted close or teardown that still holds its target is
67        // resumed rather than failed, so only the target-less residue reaches
68        // here: there is nothing left to tear down, and no relay through which
69        // to finish the close the record claims.
70        SessionState::Closing => Some(
71            "the daemon stopped while this session was closing, and it has no target left \
72             to close"
73                .to_owned(),
74        ),
75        SessionState::Destroying => Some(
76            "the daemon stopped while this session was being torn down, and it has no \
77             target left to remove"
78                .to_owned(),
79        ),
80        // A parked sub-agent is settled: its worker was stopped on purpose and
81        // it waits for its parent's next `send_input`.
82        SessionState::Checkpointing
83        | SessionState::Running
84        | SessionState::Disconnected
85        | SessionState::Parked
86        | SessionState::Stopped
87        | SessionState::Lost
88        | SessionState::Error
89        | SessionState::DestroyedWithDataLoss => None,
90    }
91}
92
93/// Every session whose in-flight lifecycle state has no owner, with the cause
94/// to record against it. `owned` names the sessions a durable move intent or
95/// another startup recovery has already claimed.
96pub fn unowned_interrupted_lifecycles(
97    controller: &Controller,
98    owned: &std::collections::BTreeSet<String>,
99) -> Vec<(String, String)> {
100    controller
101        .state
102        .sessions
103        .values()
104        .filter(|session| !owned.contains(&session.id) && !is_interrupted_close(session))
105        .filter_map(|session| {
106            interrupted_lifecycle_cause(session).map(|cause| (session.id.clone(), cause))
107        })
108        .collect()
109}
110
111pub fn reserve_recovery_or_cancel(
112    observer: &crate::recovery_gate::RecoveryObserver,
113    session_id: &str,
114    cancelled: &AtomicBool,
115) -> Result<crate::recovery_gate::RecoveryReservation> {
116    let reservation = observer.reserve(session_id);
117    // The reservation stops the next copy; cancelling preempts the one already
118    // running so a lifecycle operation never queues behind a long or wedged
119    // copy.
120    observer.cancel_busy(session_id);
121    while observer.is_busy(session_id) {
122        if cancelled.load(Ordering::Acquire) {
123            bail!("operation cancelled while waiting for recovery copy");
124        }
125        std::thread::sleep(Duration::from_millis(25));
126    }
127    Ok(reservation)
128}
129
130pub fn project_worker_title(
131    controller: &mut Controller,
132    update: &WorkerPollUpdate,
133) -> Option<Option<String>> {
134    let snapshot = update.view.snapshot.as_ref()?;
135    let session = controller.state.sessions.get_mut(&update.session_id)?;
136    let title = snapshot.resolved_title();
137    if session.acp_session_title == title {
138        return None;
139    }
140    session.acp_session_title = title.clone();
141    Some(title)
142}
143
144pub fn apply_worker_record_update(controller: &mut Controller, update: &WorkerPollUpdate) {
145    let Some(title) = project_worker_title(controller, update) else {
146        return;
147    };
148    let session_id = update.session_id.clone();
149    tokio::spawn(async move {
150        let result = tokio::task::spawn_blocking(move || {
151            crate::database::set_session_acp_title(&session_id, title.as_deref())
152        })
153        .await;
154        match result {
155            Ok(Ok(())) => {}
156            Ok(Err(error)) => tracing::warn!(%error, "could not persist relay title"),
157            Err(error) => tracing::warn!(%error, "relay title persistence task failed"),
158        }
159    });
160}