Skip to main content

mj_controller/controller/
subagent_park.rs

1//! Parking a sub-agent whose turn has ended, and starting it again.
2//!
3//! A sub-agent child runs inside its parent's target, and in a container every
4//! live child's harness holds hundreds of threads against the container's pids
5//! limit (#1161). A child that has finished its turn, and whose parent has
6//! been told so, therefore gives its processes back: its worker is stopped and
7//! its record says [`SessionState::Parked`]. Everything else stays: the
8//! record, the relation to the parent, the borrowed target locator, and the
9//! worker root with its relay journal and the harness's native session id. A
10//! parked child is started again in place, with the same sequence a worker
11//! restart uses, when its parent gives it more input.
12//!
13//! The daemon runs both as lifecycle operations of the child, so they are
14//! serialized with its close, suspend, destroy and each other.
15
16use std::time::Duration;
17
18use anyhow::{Context, Result, ensure};
19
20use mj_core::state::SessionState;
21
22use super::worker_binary::{
23    refresh_installed_worker_binary, replace_installed_worker_launch_config, stop_worker,
24    stop_worker_after_target_recovery,
25};
26use super::worker_restart::{InstalledWorkerRestart, WorkerRestartMessages};
27use super::{Controller, IdleWorkspaceLease};
28use crate::session_manager::SessionManagerControl;
29use crate::targets::{self, CommandExecutor};
30
31/// How long a park waits for the child's session actor to exist.
32const PARK_ACTOR_TIMEOUT: Duration = Duration::from_secs(5);
33
34/// How long a park waits, after recording `Parked`, for the session manager
35/// to drop the child. The manager rereads the store twice a second.
36const PARK_RELEASE_TIMEOUT: Duration = Duration::from_secs(15);
37
38/// What an unpark tells the operator at each step of the restart it runs.
39const RESTART_FROM_PARKED: WorkerRestartMessages = WorkerRestartMessages {
40    stop: "stop the parked sub-agent's worker",
41    replace: "install the current Mjolnir worker binary for the parked sub-agent",
42    start: "start the parked sub-agent's worker",
43    connect: "connect to the parked sub-agent's worker after starting it",
44    project_memory: "project memory will not be synchronized for the restarted sub-agent",
45    native_session: "wait for the parked sub-agent's harness to load its conversation",
46};
47
48/// How one park attempt ended. Only [`ParkOutcome::Parked`] changed anything.
49#[derive(Debug, Clone, Copy, PartialEq, Eq)]
50pub enum ParkOutcome {
51    /// The worker was stopped and the record says `Parked`.
52    Parked,
53    /// The child had work in flight or queued, such as a prompt its parent
54    /// sent after the turn ended, so it was left running.
55    Busy,
56    /// The child is not a running sub-agent any more (it is closing, already
57    /// parked, or gone), so there was nothing to park.
58    NotRunning,
59}
60
61/// An executor for stopping what a failed unpark started. The unpark's own
62/// executor may be the reason it failed, cancelled by a close of the child.
63fn cleanup_executor() -> targets::CancellableProcessExecutor {
64    targets::CancellableProcessExecutor::with_timeout(Duration::from_secs(30))
65}
66
67/// Say plainly that a sub-agent could not start because its target ran out of
68/// process slots, when `error` shows that, with the container's pid counts
69/// when they can be read; return any other error unchanged.
70///
71/// The read is best effort, bounded by [`targets::PIDS_USAGE_READ_TIMEOUT`],
72/// and itself fails in a container that is completely full, which the
73/// message then says.
74pub(super) fn explain_process_exhaustion(
75    error: anyhow::Error,
76    backend: &targets::TargetLocator,
77    session_id: &str,
78) -> anyhow::Error {
79    if !targets::shows_process_exhaustion(&error) {
80        return error;
81    }
82    let usage = targets::is_container(backend).then(|| {
83        let executor =
84            targets::CancellableProcessExecutor::with_timeout(targets::PIDS_USAGE_READ_TIMEOUT);
85        targets::read_pids_usage(&executor, backend, session_id)
86    });
87    anyhow::anyhow!(targets::process_exhaustion_message(usage.as_ref(), &error))
88}
89
90impl Controller {
91    /// Stop a running sub-agent's worker while it is idle, and record it as
92    /// parked.
93    ///
94    /// The worker is reserved the way an idle worker upgrade reserves it: the
95    /// actor's connection is leased only while the worker reports nothing
96    /// running or queued, and the worker holds an idle barrier until it is
97    /// stopped. A prompt that reaches the actor meanwhile waits behind the
98    /// lease. The lease is kept until the session manager has dropped the
99    /// child, which it does once the store says `Parked`, so such a prompt is
100    /// rejected as undelivered rather than sent to a stopped worker; the
101    /// caller that sent it can start the child again and resend it.
102    ///
103    /// Any failure before the record changes leaves the child running: the
104    /// lease is dropped and the actor reconnects, restarting the worker if the
105    /// stop got that far.
106    pub async fn park_subagent_worker(
107        &self,
108        session_id: &str,
109        executor: &(impl CommandExecutor + Sync),
110        manager: &SessionManagerControl,
111    ) -> Result<ParkOutcome> {
112        ensure!(
113            self.state.subagents.contains_key(session_id),
114            "session {session_id} is not a sub-agent"
115        );
116        let Some(session) = self.state.sessions.get(session_id) else {
117            return Ok(ParkOutcome::NotRunning);
118        };
119        if session.state != SessionState::Running {
120            return Ok(ParkOutcome::NotRunning);
121        }
122        let (backend, worker_root) = self.worker_placement(session_id)?;
123        let handle = manager
124            .wait_for_session(session_id, PARK_ACTOR_TIMEOUT)
125            .await?;
126        let Some(mut lease) =
127            IdleWorkspaceLease::acquire_for_upgrade(&handle, session.harness_kind).await?
128        else {
129            return Ok(ParkOutcome::Busy);
130        };
131        if !lease.verify_for_upgrade().await? {
132            return Ok(ParkOutcome::Busy);
133        }
134        stop_worker_after_target_recovery(executor, &backend, session_id, &worker_root)
135            .context("stop the sub-agent's worker to park it")?;
136        let mut record = session.clone();
137        record.state = SessionState::Parked;
138        record.last_error = None;
139        record.updated_at = super::now();
140        crate::database::save_lifecycle_session(&record).context("record the parked sub-agent")?;
141        let released = tokio::time::timeout(PARK_RELEASE_TIMEOUT, async {
142            while manager.session(session_id.to_owned()).await.is_ok() {
143                tokio::time::sleep(Duration::from_millis(50)).await;
144            }
145        })
146        .await;
147        if released.is_err() {
148            tracing::warn!(
149                session_id,
150                "the session manager still held the parked sub-agent; releasing it anyway"
151            );
152        }
153        drop(lease);
154        Ok(ParkOutcome::Parked)
155    }
156
157    /// Start a parked sub-agent's worker again in place and record it as
158    /// running.
159    ///
160    /// The worker binary and launch configuration are refreshed first when
161    /// this controller would now install different ones, then the restart
162    /// sequence runs: start the worker on its existing root, connect with the
163    /// long restart timeout, and wait until the harness has loaded its native
164    /// session and is idle. The container's start admission is held around
165    /// the harness start, as it is for a child's first start.
166    ///
167    /// A failure stops whatever was started and leaves the record `Parked`,
168    /// so the parent can try again. Nothing here connects a session actor:
169    /// the caller runs this while the daemon keeps the manager off the child,
170    /// and the manager attaches once the record says `Running`.
171    pub async fn unpark_subagent_worker(
172        &self,
173        session_id: &str,
174        executor: &(impl CommandExecutor + Sync),
175    ) -> Result<()> {
176        ensure!(
177            self.state.subagents.contains_key(session_id),
178            "session {session_id} is not a sub-agent"
179        );
180        let session = self
181            .state
182            .sessions
183            .get(session_id)
184            .with_context(|| format!("unknown session {session_id}"))?;
185        if session.state == SessionState::Running {
186            return Ok(());
187        }
188        ensure!(
189            session.state == SessionState::Parked,
190            "sub-agent {session_id} is {} and cannot be started again",
191            session.state.as_str()
192        );
193        let (backend, worker_root) = self.worker_placement(session_id)?;
194        let reconnect = targets::reconnect_plan(&backend, session_id)?
195            .commands
196            .into_iter()
197            .next()
198            .context("reconnect plan is empty")?;
199        let launch = self.current_worker_launch_config(session_id, &backend)?;
200        let started = async {
201            refresh_installed_worker_binary(executor, &backend, session_id)
202                .context(RESTART_FROM_PARKED.replace)?;
203            replace_installed_worker_launch_config(executor, &backend, session_id, &launch)
204                .context("install the current Mjolnir worker launch configuration")?;
205            // Held across the harness start, which is the part that does not
206            // survive a crowd of children starting in one container.
207            let gate = super::provisioning::container_start_gate(&backend);
208            let _admitted = match &gate {
209                Some(gate) => gate.acquire().await.ok(),
210                None => None,
211            };
212            self.start_installed_worker(
213                session_id,
214                executor,
215                InstalledWorkerRestart {
216                    backend: &backend,
217                    worker_root: &worker_root,
218                    reconnect: &reconnect,
219                    launch: Some(&launch),
220                    prepared: true,
221                    messages: &RESTART_FROM_PARKED,
222                },
223            )
224            .await
225        }
226        .await;
227        let connection = match started {
228            Ok(connection) => connection,
229            Err(error) => {
230                if let Err(stop_error) = stop_worker(&cleanup_executor(), &backend, &worker_root) {
231                    tracing::warn!(
232                        session_id,
233                        error = format!("{stop_error:#}"),
234                        "could not stop the worker of a sub-agent whose restart failed"
235                    );
236                }
237                return Err(explain_process_exhaustion(error, &backend, session_id));
238            }
239        };
240        // The manager opens its own connection once the record says running.
241        drop(connection);
242        let mut record = session.clone();
243        record.state = SessionState::Running;
244        record.last_error = None;
245        record.updated_at = super::now();
246        if let Err(error) = crate::database::save_lifecycle_session(&record) {
247            if let Err(stop_error) = stop_worker(&cleanup_executor(), &backend, &worker_root) {
248                tracing::warn!(
249                    session_id,
250                    error = format!("{stop_error:#}"),
251                    "could not stop the worker of a sub-agent whose restart was not recorded"
252                );
253            }
254            return Err(error.context("record the restarted sub-agent as running"));
255        }
256        Ok(())
257    }
258}
259
260#[cfg(all(test, unix))]
261mod tests {
262    use std::sync::Mutex;
263
264    use agent_client_protocol::schema::v1::ContentBlock;
265    use mj_core::relay::RelayCommand;
266    use mj_core::state::TargetLocator;
267
268    use super::*;
269    use crate::controller::checkpoint::tests::{
270        LATCH_RELAY_SESSION, ReleaseSupport, latch_relay_target,
271    };
272    use crate::controller::test_support::{IsolatedTest, checkpoint_test_session, test_name};
273    use crate::targets::{CommandOutput, CommandSpec};
274
275    const MARKER: &str = "MJ_TEST_SUBAGENT_PARK_CHILD";
276
277    /// Run the named test alone, with a store of its own. Returns whether
278    /// this process is that run.
279    fn isolated(test: &str) -> bool {
280        if std::env::var_os(MARKER).is_some() {
281            return true;
282        }
283        let directory = tempfile::tempdir().unwrap();
284        IsolatedTest::new(test_name(module_path!(), test))
285            .env(MARKER, "1")
286            .isolated_store(directory.path())
287            .run();
288        false
289    }
290
291    /// Stands in for the target: every command succeeds, and the first one,
292    /// which is the park's stop, also sends the child a prompt through its
293    /// actor, the way a parent's `send_input` can race a park.
294    #[derive(Default)]
295    struct RacingStop {
296        purposes: Mutex<Vec<String>>,
297        racer: Mutex<
298            Option<(
299                crate::session_manager::ManagedSessionHandle,
300                tokio::runtime::Handle,
301            )>,
302        >,
303        raced: Mutex<Option<tokio::task::JoinHandle<Result<u64>>>>,
304    }
305
306    impl CommandExecutor for RacingStop {
307        fn execute(&self, command: &CommandSpec) -> Result<CommandOutput> {
308            self.purposes.lock().unwrap().push(command.purpose.clone());
309            if let Some((handle, runtime)) = self.racer.lock().unwrap().take() {
310                *self.raced.lock().unwrap() = Some(runtime.spawn(async move {
311                    handle
312                        .submit(
313                            "raced-prompt".into(),
314                            RelayCommand::Prompt {
315                                prompt: vec![ContentBlock::from("one more thing")],
316                            },
317                        )
318                        .await
319                }));
320            }
321            Ok(CommandOutput {
322                status: 0,
323                stdout: Vec::new(),
324                stderr: Vec::new(),
325            })
326        }
327    }
328
329    /// A parent, and the stand-in relay's session registered as its child on
330    /// a bare target under `root`, with a report it handed back.
331    fn register_child(root: &std::path::Path) {
332        crate::database::save_session(&checkpoint_test_session("parent-1")).unwrap();
333        let mut child = checkpoint_test_session(LATCH_RELAY_SESSION);
334        child.target = Some(TargetLocator::LocalBare {
335            worker_root: root.join(LATCH_RELAY_SESSION),
336        });
337        crate::database::save_subagent_session(
338            &child,
339            &mj_core::subagent::SubagentRecord {
340                child_session_id: LATCH_RELAY_SESSION.into(),
341                parent_session_id: "parent-1".into(),
342                task_name: "map the parser".into(),
343                profile_id: "codex".into(),
344                model: None,
345                effort: None,
346                working_directory: Default::default(),
347                initial_prompt: "map the parser".into(),
348                request_key: "request-1".into(),
349                created_at: "2026-09-25T00:00:00Z".into(),
350                noticed_turn: None,
351                handback_tool: true,
352            },
353        )
354        .unwrap();
355        assert!(
356            crate::database::record_subagent_handback(
357                LATCH_RELAY_SESSION,
358                &mj_core::subagent::SubagentHandback {
359                    command_id: "task-1".into(),
360                    message: "The parser has three entry points.".into(),
361                    recorded_at_ms: 1,
362                },
363            )
364            .unwrap()
365        );
366    }
367
368    fn loaded_controller() -> Controller {
369        Controller {
370            config: mj_core::config::Config::default(),
371            state: crate::database::load_state().unwrap(),
372        }
373    }
374
375    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
376    async fn parking_stops_an_idle_child_keeps_its_record_and_turns_away_a_racing_prompt() {
377        if !isolated("parking_stops_an_idle_child_keeps_its_record_and_turns_away_a_racing_prompt")
378        {
379            return;
380        }
381        let _writer = crate::database::install_isolated_test_writer();
382        let root = tempfile::tempdir().unwrap();
383        register_child(root.path());
384        let crate::session_manager::SessionManagerChannels {
385            targets,
386            control,
387            updates: _updates,
388            shutdown,
389        } = crate::session_manager::spawn_session_manager().unwrap();
390        targets
391            .send(vec![latch_relay_target(
392                root.path(),
393                None,
394                ReleaseSupport::Supported,
395                false,
396            )])
397            .unwrap();
398        let handle = control
399            .wait_for_session(LATCH_RELAY_SESSION, Duration::from_secs(10))
400            .await
401            .unwrap();
402        let executor = RacingStop::default();
403        *executor.racer.lock().unwrap() = Some((handle, tokio::runtime::Handle::current()));
404        // What the daemon's target refresher does: once the store says the
405        // child is parked, the session manager no longer holds it.
406        let refresher = tokio::spawn(async move {
407            while crate::database::load_session_state(LATCH_RELAY_SESSION).unwrap()
408                != Some(SessionState::Parked)
409            {
410                tokio::time::sleep(Duration::from_millis(25)).await;
411            }
412            targets.send_replace(Vec::new());
413            targets
414        });
415
416        let outcome = loaded_controller()
417            .park_subagent_worker(LATCH_RELAY_SESSION, &executor, &control)
418            .await
419            .unwrap();
420
421        assert_eq!(outcome, ParkOutcome::Parked);
422        assert!(
423            executor
424                .purposes
425                .lock()
426                .unwrap()
427                .iter()
428                .any(|purpose| purpose == "stop Mjolnir worker daemon"),
429            "the child's worker was stopped: {:?}",
430            executor.purposes.lock().unwrap()
431        );
432        // The prompt that arrived during the park never reached the stopped
433        // worker, and its sender is told so for certain, so it can start the
434        // child again and send it once more.
435        let raced = executor.raced.lock().unwrap().take();
436        let raced = raced
437            .expect("the stop raced a prompt")
438            .await
439            .unwrap()
440            .expect_err("a prompt that arrived during the park is turned away");
441        assert!(
442            raced
443                .downcast_ref::<mj_client::session::DeliveryUnconfirmed>()
444                .is_none(),
445            "a turned-away prompt is known not to be delivered: {raced:#}"
446        );
447        // Everything but the worker stays.
448        let stored = crate::database::load_state().unwrap();
449        let child = &stored.sessions[LATCH_RELAY_SESSION];
450        assert_eq!(child.state, SessionState::Parked);
451        assert!(child.target.is_some(), "a parked child keeps its target");
452        assert!(stored.subagents.contains_key(LATCH_RELAY_SESSION));
453        assert_eq!(
454            crate::database::load_subagent_report(LATCH_RELAY_SESSION)
455                .unwrap()
456                .handback
457                .map(|handback| handback.message)
458                .as_deref(),
459            Some("The parser has three entry points.")
460        );
461        assert!(control.session(LATCH_RELAY_SESSION).await.is_err());
462        let _targets = refresher.await.unwrap();
463        shutdown.shutdown().await.unwrap();
464    }
465
466    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
467    async fn a_child_with_work_in_flight_is_not_parked() {
468        if !isolated("a_child_with_work_in_flight_is_not_parked") {
469            return;
470        }
471        let _writer = crate::database::install_isolated_test_writer();
472        let root = tempfile::tempdir().unwrap();
473        register_child(root.path());
474        let channels = crate::session_manager::spawn_session_manager().unwrap();
475        channels
476            .targets
477            .send(vec![latch_relay_target(
478                root.path(),
479                None,
480                ReleaseSupport::Supported,
481                true,
482            )])
483            .unwrap();
484        channels
485            .control
486            .wait_for_session(LATCH_RELAY_SESSION, Duration::from_secs(10))
487            .await
488            .unwrap();
489        let executor = RacingStop::default();
490
491        let outcome = loaded_controller()
492            .park_subagent_worker(LATCH_RELAY_SESSION, &executor, &channels.control)
493            .await
494            .unwrap();
495
496        assert_eq!(outcome, ParkOutcome::Busy);
497        assert!(
498            executor.purposes.lock().unwrap().is_empty(),
499            "nothing stopped"
500        );
501        assert_eq!(
502            crate::database::load_session_state(LATCH_RELAY_SESSION).unwrap(),
503            Some(SessionState::Running)
504        );
505        channels.shutdown.shutdown().await.unwrap();
506    }
507
508    #[test]
509    fn only_a_full_target_is_rewritten_and_a_bare_one_reads_no_container_counts() {
510        let backend = targets::TargetLocator::LocalBare {
511            worker_root: "/tmp/workers/child".into(),
512        };
513        let unrelated =
514            explain_process_exhaustion(anyhow::anyhow!("the harness exited"), &backend, "child");
515        assert_eq!(format!("{unrelated:#}"), "the harness exited");
516
517        let full = explain_process_exhaustion(
518            anyhow::anyhow!("sh: 1: Cannot fork").context("start the parked sub-agent's worker"),
519            &backend,
520            "child",
521        );
522        let message = format!("{full:#}");
523        assert!(
524            message.starts_with("the target machine ran out of process slots"),
525            "{message}"
526        );
527        assert!(
528            message.contains("Close sub-agents you no longer need"),
529            "{message}"
530        );
531        assert!(
532            message.contains("Cannot fork"),
533            "the original error stays: {message}"
534        );
535    }
536}