Skip to main content

mj_controller/controller/
subagent_park.rs

1//! Parking a sub-agent whose turn has ended, and starting it again.
2//!
3//! A sub-agent child runs inside its parent's target, and in a container every
4//! live child's harness holds hundreds of threads against the container's pids
5//! limit (#1161). A child that has finished its turn, and whose parent has
6//! been told so, therefore gives its processes back: its worker is stopped and
7//! its record says [`SessionState::Parked`]. Everything else stays: the
8//! record, the relation to the parent, the borrowed target locator, and the
9//! worker root with its relay journal and the harness's native session id. A
10//! parked child is started again in place, with the same sequence a worker
11//! restart uses, when its parent gives it more input.
12//!
13//! The daemon runs both as lifecycle operations of the child, so they are
14//! serialized with its close, suspend, destroy and each other.
15
16use std::time::Duration;
17
18use anyhow::{Context, Result, ensure};
19
20use mj_core::state::SessionState;
21
22use super::worker_binary::{
23    refresh_installed_worker_binary, replace_installed_worker_launch_config, stop_worker,
24    stop_worker_after_target_recovery,
25};
26use super::worker_restart::{InstalledWorkerRestart, WorkerRestartMessages};
27use super::{Controller, IdleWorkspaceLease};
28use crate::session_manager::SessionManagerControl;
29use crate::targets::{self, CommandExecutor};
30
31/// How long a park waits for the child's session actor to exist.
32const PARK_ACTOR_TIMEOUT: Duration = Duration::from_secs(5);
33
34/// How long a park waits, after recording `Parked`, for the session manager
35/// to drop the child. The manager rereads the store twice a second.
36const PARK_RELEASE_TIMEOUT: Duration = Duration::from_secs(15);
37
38/// What an unpark tells the operator at each step of the restart it runs.
39const RESTART_FROM_PARKED: WorkerRestartMessages = WorkerRestartMessages {
40    stop: "stop the parked sub-agent's worker",
41    replace: "install the current Mjolnir worker binary for the parked sub-agent",
42    start: "start the parked sub-agent's worker",
43    connect: "connect to the parked sub-agent's worker after starting it",
44    project_memory: "project memory will not be synchronized for the restarted sub-agent",
45    native_session: "wait for the parked sub-agent's harness to load its conversation",
46};
47
48/// How one park attempt ended. Only [`ParkOutcome::Parked`] changed anything.
49#[derive(Debug, Clone, Copy, PartialEq, Eq)]
50pub enum ParkOutcome {
51    /// The worker was stopped and the record says `Parked`.
52    Parked,
53    /// The child had work in flight or queued, such as a prompt its parent
54    /// sent after the turn ended, so it was left running.
55    Busy,
56    /// The child is not a running sub-agent any more (it is closing, already
57    /// parked, or gone), so there was nothing to park.
58    NotRunning,
59}
60
61/// An executor for stopping what a failed unpark started. The unpark's own
62/// executor may be the reason it failed, cancelled by a close of the child.
63pub(crate) const FAILED_STARTUP_CLEANUP_TIMEOUT: Duration = Duration::from_secs(30);
64
65pub(crate) fn failed_launch_cleanup_executor() -> targets::CancellableProcessExecutor {
66    targets::CancellableProcessExecutor::with_timeout(FAILED_STARTUP_CLEANUP_TIMEOUT)
67}
68
69/// Say plainly that a sub-agent could not start because its target ran out of
70/// process slots, when `error` shows that, with the container's pid counts
71/// when they can be read; return any other error unchanged.
72///
73/// The read is best effort, bounded by [`targets::PIDS_USAGE_READ_TIMEOUT`],
74/// and itself fails in a container that is completely full, which the
75/// message then says.
76pub(super) fn explain_process_exhaustion(
77    error: anyhow::Error,
78    backend: &targets::TargetLocator,
79    session_id: &str,
80) -> anyhow::Error {
81    if !targets::shows_process_exhaustion(&error) {
82        return error;
83    }
84    let usage = targets::is_container(backend).then(|| {
85        let executor =
86            targets::CancellableProcessExecutor::with_timeout(targets::PIDS_USAGE_READ_TIMEOUT);
87        targets::read_pids_usage(&executor, backend, session_id)
88    });
89    anyhow::anyhow!(targets::process_exhaustion_message(usage.as_ref(), &error))
90}
91
92impl Controller {
93    /// Stop a running sub-agent's worker while it is idle, and record it as
94    /// parked.
95    ///
96    /// The worker is reserved the way an idle worker upgrade reserves it: the
97    /// actor's connection is leased only while the worker reports nothing
98    /// running or queued, and the worker holds an idle barrier until it is
99    /// stopped. A prompt that reaches the actor meanwhile waits behind the
100    /// lease. The lease is kept until the session manager has dropped the
101    /// child, which it does once the store says `Parked`, so such a prompt is
102    /// rejected as undelivered rather than sent to a stopped worker; the
103    /// caller that sent it can start the child again and resend it.
104    ///
105    /// Any failure before the record changes leaves the child running: the
106    /// lease is dropped and the actor reconnects, restarting the worker if the
107    /// stop got that far.
108    pub async fn park_subagent_worker(
109        &self,
110        session_id: &str,
111        executor: &(impl CommandExecutor + Sync),
112        manager: &SessionManagerControl,
113    ) -> Result<ParkOutcome> {
114        self.stop_idle_subagent_worker(session_id, executor, manager, None)
115            .await
116    }
117
118    /// Stop a sub-agent whose first prompt can never run and record it as
119    /// failed with `cause`, so every surface shows the failure instead of an
120    /// idle session holding a worker (I1-2). It is stopped exactly as a park
121    /// stops a child, behind the same idle reservation, and keeps its record,
122    /// relation and target; its close settles without a checkpoint (see
123    /// [`super::has_nothing_to_checkpoint`]).
124    ///
125    /// A child that took work meanwhile is left running and
126    /// [`ParkOutcome::Busy`] is returned. [`ParkOutcome::Parked`] means the
127    /// worker was stopped and the record now says `Error`.
128    pub async fn fail_subagent_start_worker(
129        &self,
130        session_id: &str,
131        cause: &str,
132        executor: &(impl CommandExecutor + Sync),
133        manager: &SessionManagerControl,
134    ) -> Result<ParkOutcome> {
135        self.stop_idle_subagent_worker(session_id, executor, manager, Some(cause))
136            .await
137    }
138
139    /// The stop behind a park and a failed start: the record becomes
140    /// `Parked`, or `Error` with `failure` as its cause.
141    async fn stop_idle_subagent_worker(
142        &self,
143        session_id: &str,
144        executor: &(impl CommandExecutor + Sync),
145        manager: &SessionManagerControl,
146        failure: Option<&str>,
147    ) -> Result<ParkOutcome> {
148        crate::worker_lifecycle::run(session_id, "stop idle subagent worker", executor, async {
149            crate::worker_lifecycle::require(session_id)?.verify_cached_target(&self.state)?;
150            ensure!(
151                self.state.subagents.contains_key(session_id),
152                "session {session_id} is not a sub-agent"
153            );
154            let Some(session) = self.state.sessions.get(session_id) else {
155                return Ok(ParkOutcome::NotRunning);
156            };
157            if session.state != SessionState::Running {
158                return Ok(ParkOutcome::NotRunning);
159            }
160            let (backend, worker_root) = self.worker_placement(session_id)?;
161            let handle = manager
162                .wait_for_session(session_id, PARK_ACTOR_TIMEOUT)
163                .await?;
164            let Some(mut lease) =
165                IdleWorkspaceLease::acquire_for_upgrade(&handle, session.harness_kind).await?
166            else {
167                return Ok(ParkOutcome::Busy);
168            };
169            if !lease.verify_for_upgrade().await? {
170                return Ok(ParkOutcome::Busy);
171            }
172            stop_worker_after_target_recovery(executor, &backend, session_id, &worker_root)
173                .context("stop the sub-agent's worker")?;
174            let mut record = session.clone();
175            record.state = if failure.is_some() {
176                SessionState::Error
177            } else {
178                SessionState::Parked
179            };
180            record.last_error = failure.map(str::to_owned);
181            record.updated_at = super::now();
182            crate::database::save_lifecycle_session(&record)
183                .context("record the stopped sub-agent")?;
184            let released = tokio::time::timeout(PARK_RELEASE_TIMEOUT, async {
185                while manager.session(session_id.to_owned()).await.is_ok() {
186                    tokio::time::sleep(Duration::from_millis(50)).await;
187                }
188            })
189            .await;
190            if released.is_err() {
191                tracing::warn!(
192                    session_id,
193                    "the session manager still held the stopped sub-agent; releasing it anyway"
194                );
195            }
196            drop(lease);
197            Ok(ParkOutcome::Parked)
198        })
199        .await
200    }
201
202    /// Start a parked sub-agent's worker again in place and record it as
203    /// running.
204    ///
205    /// The worker binary and launch configuration are refreshed first when
206    /// this controller would now install different ones, then the restart
207    /// sequence runs: start the worker on its existing root, connect with the
208    /// long restart timeout, and wait until the harness has loaded its native
209    /// session and is idle. The container's start admission is held around
210    /// the harness start, as it is for a child's first start.
211    ///
212    /// A failure stops whatever was started and leaves the record `Parked`,
213    /// so the parent can try again. Nothing here connects a session actor:
214    /// the caller runs this while the daemon keeps the manager off the child,
215    /// and the manager attaches once the record says `Running`.
216    pub async fn unpark_subagent_worker(
217        &self,
218        session_id: &str,
219        executor: &(impl CommandExecutor + Sync),
220    ) -> Result<()> {
221        crate::worker_lifecycle::run(session_id, "unpark subagent worker", executor, async {
222            crate::worker_lifecycle::require(session_id)?.verify_cached_target(&self.state)?;
223            ensure!(
224                self.state.subagents.contains_key(session_id),
225                "session {session_id} is not a sub-agent"
226            );
227            let session = self
228                .state
229                .sessions
230                .get(session_id)
231                .with_context(|| format!("unknown session {session_id}"))?;
232            if session.state == SessionState::Running {
233                return Ok(());
234            }
235            ensure!(
236                session.state == SessionState::Parked,
237                "sub-agent {session_id} is {} and cannot be started again",
238                session.state.as_str()
239            );
240            let (backend, worker_root) = self.worker_placement(session_id)?;
241            let reconnect = targets::reconnect_plan(&backend, session_id)?
242                .commands
243                .into_iter()
244                .next()
245                .context("reconnect plan is empty")?;
246            let launch = self.current_worker_launch_config(session_id, &backend)?;
247            let started = async {
248                refresh_installed_worker_binary(executor, &backend, session_id)
249                    .context(RESTART_FROM_PARKED.replace)?;
250                replace_installed_worker_launch_config(executor, &backend, session_id, &launch)
251                    .context("install the current Mjolnir worker launch configuration")?;
252                // Held across the harness start, which is the part that does not
253                // survive a crowd of children starting in one container.
254                let gate = super::provisioning::container_start_gate(&backend);
255                let _admitted = match &gate {
256                    Some(gate) => gate.acquire().await.ok(),
257                    None => None,
258                };
259                self.start_installed_worker(
260                    session_id,
261                    executor,
262                    InstalledWorkerRestart {
263                        backend: &backend,
264                        worker_root: &worker_root,
265                        reconnect: &reconnect,
266                        launch: Some(&launch),
267                        prepared: true,
268                        messages: &RESTART_FROM_PARKED,
269                    },
270                )
271                .await
272            }
273            .await;
274            let connection = match started {
275                Ok(connection) => connection,
276                Err(error) => {
277                    if let Err(stop_error) = stop_worker(
278                        &crate::worker_lifecycle::require(session_id)?,
279                        &failed_launch_cleanup_executor(),
280                        &backend,
281                        &worker_root,
282                    ) {
283                        tracing::warn!(
284                            session_id,
285                            error = format!("{stop_error:#}"),
286                            "could not stop the worker of a sub-agent whose restart failed"
287                        );
288                    }
289                    if let Some(failure) = error
290                        .downcast_ref::<crate::controller::HarnessPreparationFailure>()
291                    {
292                        let mut record = session.clone();
293                        record.state = SessionState::Error;
294                        record.last_error = Some(failure.to_string());
295                        record.updated_at = super::now();
296                        if let Err(record_error) =
297                            crate::database::save_lifecycle_session(&record)
298                        {
299                            return Err(error.context(format!(
300                                "also failed to record the sub-agent harness preparation failure: {record_error:#}"
301                            )));
302                        }
303                    }
304                    return Err(explain_process_exhaustion(error, &backend, session_id));
305                }
306            };
307            // The manager opens its own connection once the record says running.
308            drop(connection);
309            let mut record = session.clone();
310            record.state = SessionState::Running;
311            record.last_error = None;
312            record.updated_at = super::now();
313            if let Err(error) = crate::database::save_lifecycle_session(&record) {
314                if let Err(stop_error) = stop_worker(
315                    &crate::worker_lifecycle::require(session_id)?,
316                    &failed_launch_cleanup_executor(),
317                    &backend,
318                    &worker_root,
319                ) {
320                    tracing::warn!(
321                        session_id,
322                        error = format!("{stop_error:#}"),
323                        "could not stop the worker of a sub-agent whose restart was not recorded"
324                    );
325                }
326                return Err(error.context("record the restarted sub-agent as running"));
327            }
328            Ok(())
329        })
330        .await
331    }
332}
333
334#[cfg(all(test, unix))]
335mod tests {
336    use std::sync::Mutex;
337
338    use agent_client_protocol::schema::v1::ContentBlock;
339    use mj_core::relay::RelayCommand;
340    use mj_core::state::TargetLocator;
341
342    use super::*;
343    use crate::controller::checkpoint::tests::{
344        LATCH_RELAY_SESSION, ReleaseSupport, latch_relay_target,
345    };
346    use crate::controller::test_support::{IsolatedTest, checkpoint_test_session, test_name};
347    use crate::targets::{CommandOutput, CommandSpec};
348
349    const MARKER: &str = "MJ_TEST_SUBAGENT_PARK_CHILD";
350
351    /// Run the named test alone, with a store of its own. Returns whether
352    /// this process is that run.
353    fn isolated(test: &str) -> bool {
354        if std::env::var_os(MARKER).is_some() {
355            return true;
356        }
357        let directory = tempfile::tempdir().unwrap();
358        IsolatedTest::new(test_name(module_path!(), test))
359            .env(MARKER, "1")
360            .isolated_store(directory.path())
361            .run();
362        false
363    }
364
365    /// Stands in for the target: every command succeeds, and the first one,
366    /// which is the park's stop, also sends the child a prompt through its
367    /// actor, the way a parent's `send_input` can race a park.
368    #[derive(Default)]
369    struct RacingStop {
370        purposes: Mutex<Vec<String>>,
371        racer: Mutex<
372            Option<(
373                crate::session_manager::ManagedSessionHandle,
374                tokio::runtime::Handle,
375            )>,
376        >,
377        raced: Mutex<Option<tokio::task::JoinHandle<Result<u64>>>>,
378    }
379
380    impl CommandExecutor for RacingStop {
381        fn execute(&self, command: &CommandSpec) -> Result<CommandOutput> {
382            self.purposes.lock().unwrap().push(command.purpose.clone());
383            if let Some((handle, runtime)) = self.racer.lock().unwrap().take() {
384                *self.raced.lock().unwrap() = Some(runtime.spawn(async move {
385                    handle
386                        .submit(
387                            "raced-prompt".into(),
388                            RelayCommand::Prompt {
389                                prompt: vec![ContentBlock::from("one more thing")],
390                            },
391                        )
392                        .await
393                }));
394            }
395            Ok(CommandOutput {
396                status: 0,
397                stdout: Vec::new(),
398                stderr: Vec::new(),
399            })
400        }
401    }
402
403    /// A parent, and the stand-in relay's session registered as its child on
404    /// a bare target under `root`, with a report it handed back.
405    fn register_child(root: &std::path::Path) {
406        crate::database::save_session(&checkpoint_test_session("parent-1")).unwrap();
407        let mut child = checkpoint_test_session(LATCH_RELAY_SESSION);
408        child.target = Some(TargetLocator::LocalBare {
409            worker_root: root.join(LATCH_RELAY_SESSION),
410        });
411        crate::database::save_subagent_session(
412            &child,
413            &mj_core::subagent::SubagentRecord {
414                child_session_id: LATCH_RELAY_SESSION.into(),
415                parent_session_id: "parent-1".into(),
416                task_name: "map the parser".into(),
417                profile_id: "codex".into(),
418                model: None,
419                effort: None,
420                working_directory: Default::default(),
421                initial_prompt: "map the parser".into(),
422                request_key: "request-1".into(),
423                created_at: "2026-09-25T00:00:00Z".into(),
424                noticed_turn: None,
425                handback_tool: true,
426            },
427        )
428        .unwrap();
429        assert!(
430            crate::database::record_subagent_handback(
431                LATCH_RELAY_SESSION,
432                &mj_core::subagent::SubagentHandback {
433                    command_id: "task-1".into(),
434                    message: "The parser has three entry points.".into(),
435                    recorded_at_ms: 1,
436                },
437            )
438            .unwrap()
439        );
440    }
441
442    fn loaded_controller() -> Controller {
443        Controller {
444            config: mj_core::config::Config::default(),
445            state: crate::database::load_state().unwrap(),
446        }
447    }
448
449    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
450    async fn parking_stops_an_idle_child_keeps_its_record_and_turns_away_a_racing_prompt() {
451        if !isolated("parking_stops_an_idle_child_keeps_its_record_and_turns_away_a_racing_prompt")
452        {
453            return;
454        }
455        let _writer = crate::database::install_isolated_test_writer();
456        let root = tempfile::tempdir().unwrap();
457        register_child(root.path());
458        let crate::session_manager::SessionManagerChannels {
459            session_cpu: _,
460            targets,
461            control,
462            updates: _updates,
463            shutdown,
464        } = crate::session_manager::spawn_session_manager().unwrap();
465        targets
466            .send(vec![latch_relay_target(
467                root.path(),
468                None,
469                ReleaseSupport::Supported,
470                false,
471            )])
472            .unwrap();
473        let handle = control
474            .wait_for_session(LATCH_RELAY_SESSION, Duration::from_secs(10))
475            .await
476            .unwrap();
477        let executor = RacingStop::default();
478        *executor.racer.lock().unwrap() = Some((handle, tokio::runtime::Handle::current()));
479        // What the daemon's target refresher does: once the store says the
480        // child is parked, the session manager no longer holds it.
481        let refresher = tokio::spawn(async move {
482            while crate::database::load_session_state(LATCH_RELAY_SESSION).unwrap()
483                != Some(SessionState::Parked)
484            {
485                tokio::time::sleep(Duration::from_millis(25)).await;
486            }
487            targets.send_replace(Vec::new());
488            targets
489        });
490
491        let outcome = loaded_controller()
492            .park_subagent_worker(LATCH_RELAY_SESSION, &executor, &control)
493            .await
494            .unwrap();
495
496        assert_eq!(outcome, ParkOutcome::Parked);
497        assert!(
498            executor
499                .purposes
500                .lock()
501                .unwrap()
502                .iter()
503                .any(|purpose| purpose == "stop Mjolnir worker daemon"),
504            "the child's worker was stopped: {:?}",
505            executor.purposes.lock().unwrap()
506        );
507        // The prompt that arrived during the park never reached the stopped
508        // worker, and its sender is told so for certain, so it can start the
509        // child again and send it once more.
510        let raced = executor.raced.lock().unwrap().take();
511        let raced = raced
512            .expect("the stop raced a prompt")
513            .await
514            .unwrap()
515            .expect_err("a prompt that arrived during the park is turned away");
516        assert!(
517            raced
518                .downcast_ref::<mj_client::session::DeliveryUnconfirmed>()
519                .is_none(),
520            "a turned-away prompt is known not to be delivered: {raced:#}"
521        );
522        // Everything but the worker stays.
523        let stored = crate::database::load_state().unwrap();
524        let child = &stored.sessions[LATCH_RELAY_SESSION];
525        assert_eq!(child.state, SessionState::Parked);
526        assert!(child.target.is_some(), "a parked child keeps its target");
527        assert!(stored.subagents.contains_key(LATCH_RELAY_SESSION));
528        assert_eq!(
529            crate::database::load_subagent_report(LATCH_RELAY_SESSION)
530                .unwrap()
531                .handback
532                .map(|handback| handback.message)
533                .as_deref(),
534            Some("The parser has three entry points.")
535        );
536        assert!(control.session(LATCH_RELAY_SESSION).await.is_err());
537        let _targets = refresher.await.unwrap();
538        shutdown.shutdown().await.unwrap();
539    }
540
541    /// I1-2: a child whose first prompt is refused for good has its worker
542    /// stopped and is recorded as failed with the cause, keeping its record,
543    /// relation and target, instead of staying a live idle session.
544    // Hard-won: ba6c34276ced: a refused first prompt left a live idle child whose parent had already seen the error.
545    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
546    async fn a_child_whose_start_failed_is_stopped_and_recorded_as_failed() {
547        if !isolated("a_child_whose_start_failed_is_stopped_and_recorded_as_failed") {
548            return;
549        }
550        let _writer = crate::database::install_isolated_test_writer();
551        let root = tempfile::tempdir().unwrap();
552        register_child(root.path());
553        let crate::session_manager::SessionManagerChannels {
554            session_cpu: _,
555            targets,
556            control,
557            updates: _updates,
558            shutdown,
559        } = crate::session_manager::spawn_session_manager().unwrap();
560        targets
561            .send(vec![latch_relay_target(
562                root.path(),
563                None,
564                ReleaseSupport::Supported,
565                false,
566            )])
567            .unwrap();
568        control
569            .wait_for_session(LATCH_RELAY_SESSION, Duration::from_secs(10))
570            .await
571            .unwrap();
572        let refresher = tokio::spawn(async move {
573            while crate::database::load_session_state(LATCH_RELAY_SESSION).unwrap()
574                != Some(SessionState::Error)
575            {
576                tokio::time::sleep(Duration::from_millis(25)).await;
577            }
578            targets.send_replace(Vec::new());
579            targets
580        });
581        let executor = RacingStop::default();
582        let cause = "this agent does not offer high as a effort";
583
584        let outcome = loaded_controller()
585            .fail_subagent_start_worker(LATCH_RELAY_SESSION, cause, &executor, &control)
586            .await
587            .unwrap();
588
589        assert_eq!(outcome, ParkOutcome::Parked);
590        assert!(
591            executor
592                .purposes
593                .lock()
594                .unwrap()
595                .iter()
596                .any(|purpose| purpose == "stop Mjolnir worker daemon"),
597            "the child's worker was stopped: {:?}",
598            executor.purposes.lock().unwrap()
599        );
600        let stored = crate::database::load_state().unwrap();
601        let child = &stored.sessions[LATCH_RELAY_SESSION];
602        assert_eq!(child.state, SessionState::Error);
603        assert_eq!(child.last_error.as_deref(), Some(cause));
604        assert!(child.target.is_some(), "the failed child keeps its target");
605        assert!(stored.subagents.contains_key(LATCH_RELAY_SESSION));
606        assert!(control.session(LATCH_RELAY_SESSION).await.is_err());
607        let _targets = refresher.await.unwrap();
608        shutdown.shutdown().await.unwrap();
609    }
610
611    #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
612    async fn a_child_with_work_in_flight_is_not_parked() {
613        if !isolated("a_child_with_work_in_flight_is_not_parked") {
614            return;
615        }
616        let _writer = crate::database::install_isolated_test_writer();
617        let root = tempfile::tempdir().unwrap();
618        register_child(root.path());
619        let channels = crate::session_manager::spawn_session_manager().unwrap();
620        channels
621            .targets
622            .send(vec![latch_relay_target(
623                root.path(),
624                None,
625                ReleaseSupport::Supported,
626                true,
627            )])
628            .unwrap();
629        channels
630            .control
631            .wait_for_session(LATCH_RELAY_SESSION, Duration::from_secs(10))
632            .await
633            .unwrap();
634        let executor = RacingStop::default();
635
636        let outcome = loaded_controller()
637            .park_subagent_worker(LATCH_RELAY_SESSION, &executor, &channels.control)
638            .await
639            .unwrap();
640
641        assert_eq!(outcome, ParkOutcome::Busy);
642        assert!(
643            executor.purposes.lock().unwrap().is_empty(),
644            "nothing stopped"
645        );
646        assert_eq!(
647            crate::database::load_session_state(LATCH_RELAY_SESSION).unwrap(),
648            Some(SessionState::Running)
649        );
650        channels.shutdown.shutdown().await.unwrap();
651    }
652
653    // Hard-won: 6927da2ba976: Issue #1161 exhausted a shipped container's process slots and left users with a Cannot fork failure.
654    #[test]
655    fn only_a_full_target_is_rewritten_and_a_bare_one_reads_no_container_counts() {
656        let backend = targets::TargetLocator::LocalBare {
657            worker_root: "/tmp/workers/child".into(),
658        };
659        let unrelated =
660            explain_process_exhaustion(anyhow::anyhow!("the harness exited"), &backend, "child");
661        assert_eq!(format!("{unrelated:#}"), "the harness exited");
662
663        let full = explain_process_exhaustion(
664            anyhow::anyhow!("sh: 1: Cannot fork").context("start the parked sub-agent's worker"),
665            &backend,
666            "child",
667        );
668        let message = format!("{full:#}");
669        assert!(
670            message.starts_with("the target machine ran out of process slots"),
671            "{message}"
672        );
673        assert!(
674            message.contains("Close sub-agents you no longer need"),
675            "{message}"
676        );
677        assert!(
678            message.contains("Cannot fork"),
679            "the original error stays: {message}"
680        );
681    }
682}