Skip to main content

supercode_harness/
agent.rs

1//! The agent loop.
2
3use std::collections::HashSet;
4
5use crate::config::{CachePlan, Config, SteeringMode, ToolAdvertising};
6use crate::error::{Error, Result};
7use crate::provider::{self, ChatRequest, OpenAiProvider, Provider, ToolSchema};
8use crate::reduce::rehydrate::CAP_NOTICE_MARKER;
9use crate::reduce::{self, ReductionLog, ReductionPolicy};
10use crate::tools::{ToolContext, ToolRegistry};
11use supercode_interchange::session::Session;
12use supercode_interchange::sidecar::SidecarWriter;
13use supercode_interchange::{ChatMessage, Role};
14use supercode_runtime::AgentEvent;
15
16/// BP-6: how many skill bodies one user message may pull in through `$slug`
17/// mentions — Claude Code caps skill chaining at six per message (cc§7
18/// "Skill chaining"); the same ceiling bounds the mention path here.
19const MAX_SKILL_LOADS_PER_MESSAGE: usize = 6;
20
21/// BP-5: how many `@path` mentions one message may attach. A prompt is not
22/// a bulk loader; past this the user means `--file`.
23const MAX_FILE_MENTIONS_PER_MESSAGE: usize = 10;
24
25/// BP-5: ceiling on the bytes one `@path` mention contributes.
26const MAX_FILE_MENTION_BYTES: usize = 64 * 1024;
27
28/// Tool name of the `tool_search` agent intrinsic (B6). Never a registered
29/// [`crate::tools::Tool`] — intercepted in [`Agent::run_tool`] before registry
30/// lookup, so it works under any [`ToolAdvertising`] mode.
31const TOOL_SEARCH: &str = "tool_search";
32
33/// Tool name of the `expand_reduction` agent intrinsic (T12/TR-1) — the
34/// model-invocable rehydration counterpart to `tool_search`, same
35/// interception pattern. Advertised whenever a [`ReductionPolicy`] is
36/// installed, regardless of [`ToolAdvertising`] mode (see [`Self::tool_schemas`]).
37const EXPAND_REDUCTION: &str = "expand_reduction";
38
39/// Tool name of the `sidecar_search` agent intrinsic (T12/TR-1).
40const SIDECAR_SEARCH: &str = "sidecar_search";
41
42/// Tool name of the `spawn_subagent` agent intrinsic (P5-3, §2 module 9 D1
43/// "spawn tool"). Same interception pattern as [`TOOL_SEARCH`] — never a
44/// registered [`crate::tools::Tool`], intercepted in [`Agent::run_tool`]
45/// before registry lookup — but ALSO needs full `&mut self` async access
46/// (running a whole child agent loop, or `tokio::spawn`-ing one), which
47/// [`Agent::prepare_tool_call`]'s purely-synchronous intrinsics don't, so
48/// the interception point is `Self::run_tool`'s top, not
49/// `prepare_tool_call`.
50const SPAWN_SUBAGENT: &str = "spawn_subagent";
51
52/// BP-7 (catalog §4a "Review mode"): the `[core.prompts]` key the review
53/// turn's template lives under. One name for both presets — cc spells the
54/// command `/code-review`, cx spells it `/review`, and both resolve to this
55/// template, so the row's evidence is one config key, not two.
56pub const REVIEW_PROMPT_NAME: &str = "code-review";
57
58/// BP-7 (catalog §4a "Side/ephemeral Q&A"): the instruction prefixed to a
59/// side question, so the model knows it is answering ABOUT the session
60/// rather than continuing it. The exchange never enters history either way;
61/// this keeps the answer from reading like the next assistant turn.
62const SIDE_QUESTION_PREAMBLE: &str = "[side question — answer from the conversation above; this exchange is not part of the conversation and you have no tools for it]";
63
64/// Claude Code's native name for [`SPAWN_SUBAGENT`]. It is exposed only when
65/// `Config::subagents_claude_agent_alias` is enabled for a Claude import.
66const CLAUDE_AGENT: &str = "Agent";
67
68/// Claude Code spellings for core filesystem/shell tools. Imported Claude
69/// context frequently continues to call these names even when another model
70/// is driving the turn, so emulation must translate execution as well as
71/// preserve the original call/result names in the transcript.
72const CLAUDE_BASH: &str = "Bash";
73const CLAUDE_READ: &str = "Read";
74const CLAUDE_WRITE: &str = "Write";
75const CLAUDE_EDIT: &str = "Edit";
76const CLAUDE_GLOB: &str = "Glob";
77const CLAUDE_GREP: &str = "Grep";
78
79/// Claude Code scheduler compatibility intrinsics. They edit an imported
80/// [`crate::ClaudeRuntimeManifest`]; actual timer execution belongs to an
81/// embedding scheduler driver, never this agent loop.
82const CLAUDE_CRON_CREATE: &str = "CronCreate";
83const CLAUDE_CRON_DELETE: &str = "CronDelete";
84const CLAUDE_CRON_LIST: &str = "CronList";
85const CLAUDE_SCHEDULE_WAKEUP: &str = "ScheduleWakeup";
86
87/// Shared SDK steering mailbox. `accepting` and `queue` share one lock so a
88/// turn's final boundary can close acceptance atomically with its last drain;
89/// a steer can therefore never be acknowledged into the following turn.
90#[derive(Default)]
91pub(crate) struct SteerInbox {
92    queue: std::collections::VecDeque<QueuedSteer>,
93    accepting: bool,
94}
95
96struct QueuedSteer {
97    message: String,
98    sdk_bound: bool,
99}
100
101impl SteerInbox {
102    pub(crate) fn open(&mut self) {
103        self.queue.clear();
104        self.accepting = true;
105    }
106
107    pub(crate) fn enqueue(&mut self, message: String) -> bool {
108        if !self.accepting {
109            return false;
110        }
111        self.queue.push_back(QueuedSteer {
112            message,
113            sdk_bound: true,
114        });
115        true
116    }
117
118    pub(crate) fn close(&mut self) {
119        self.accepting = false;
120        self.queue.retain(|queued| !queued.sdk_bound);
121    }
122
123    fn drain(&mut self, mode: SteeringMode) -> Option<String> {
124        if self.queue.is_empty() {
125            return None;
126        }
127        match mode {
128            SteeringMode::All => Some(
129                self.queue
130                    .drain(..)
131                    .map(|queued| queued.message)
132                    .collect::<Vec<_>>()
133                    .join("\n\n"),
134            ),
135            SteeringMode::OneAtATime => self.queue.pop_front().map(|queued| queued.message),
136        }
137    }
138
139    fn drain_or_close(&mut self, mode: SteeringMode) -> Option<String> {
140        if !self.queue.iter().any(|queued| queued.sdk_bound) {
141            self.accepting = false;
142            return None;
143        }
144        self.drain(mode)
145    }
146
147    fn queue_unchecked(&mut self, message: String) {
148        self.queue.push_back(QueuedSteer {
149            message,
150            sdk_bound: false,
151        });
152    }
153
154    fn len(&self) -> usize {
155        self.queue.len()
156    }
157}
158
159struct SteerTurnGuard {
160    inbox: std::sync::Arc<std::sync::Mutex<SteerInbox>>,
161}
162
163impl SteerTurnGuard {
164    fn new(inbox: std::sync::Arc<std::sync::Mutex<SteerInbox>>) -> Self {
165        inbox
166            .lock()
167            .unwrap_or_else(std::sync::PoisonError::into_inner)
168            .accepting = true;
169        Self { inbox }
170    }
171}
172
173impl Drop for SteerTurnGuard {
174    fn drop(&mut self) {
175        self.inbox
176            .lock()
177            .unwrap_or_else(std::sync::PoisonError::into_inner)
178            .close();
179    }
180}
181
182/// Tool name of the `subagent_status` agent intrinsic (P5-3, D3
183/// "background+resume"): poll (and reap, once finished) a background child
184/// spawned via [`SPAWN_SUBAGENT`]. Only advertised when
185/// `Config::subagents_background` is on (see [`Agent::tool_schemas`]).
186const SUBAGENT_STATUS: &str = "subagent_status";
187
188/// The agent's one message verb (cc's `SendMessage`, cx's v2 `send_message`):
189/// `to` is either a STILL-RUNNING background child (delivered into its own
190/// inbox) or another session of any harness, which goes through the one send
191/// every sender uses ([`crate::mail_send`]). Only advertised when
192/// `Config::subagents_background` is on.
193const SEND_MESSAGE: &str = "send_message";
194
195/// BP-7 (catalog §4a "Background subagents + resume": "resumable with
196/// context intact"): continue a FINISHED child with its own transcript
197/// restored, rather than starting a fresh one that has to be re-briefed.
198const SUBAGENT_RESUME: &str = "subagent_resume";
199
200/// Tool name of the `background_exec` agent intrinsic (P5-6, §2 module 4
201/// `tools.background` D1 "background exec"). Unlike [`SPAWN_SUBAGENT`], this
202/// needs no async child-agent loop — spawning a process
203/// (`tokio::process::Command::spawn`) is itself synchronous — so, like
204/// [`TOOL_SEARCH`], it is intercepted in [`Agent::prepare_tool_call`], not
205/// [`Agent::run_tool`].
206const BACKGROUND_EXEC: &str = "background_exec";
207
208/// Tool name of the `background_status` agent intrinsic (P5-6, D1 "monitor/
209/// event feed"): poll a background job's run status, drain its newly
210/// captured output as an [`AgentEvent::BackgroundOutput`] event, and reap it
211/// (remove it from [`Agent::background_jobs`]) once it has exited or been
212/// killed.
213const BACKGROUND_STATUS: &str = "background_status";
214
215/// Tool name of the `background_list` agent intrinsic (P5-6, D10
216/// "bg-manager"): list every background job this agent is currently
217/// tracking (running or finished-but-unreaped), without draining output or
218/// reaping anything.
219const BACKGROUND_LIST: &str = "background_list";
220
221/// Tool name of the `background_kill` agent intrinsic (P5-6, D10
222/// "bg-manager"): kill a background job's real OS process
223/// (`tokio::process::Child::start_kill`) and reap it immediately.
224const BACKGROUND_KILL: &str = "background_kill";
225
226/// P4e (§3.1 `core.parallel_tool_calls`): the synchronous outcome of
227/// [`Agent::prepare_tool_call`] — either a result already in hand (an
228/// intrinsic, or a call refused before it ever reached `Tool::execute`), or
229/// a plain registry-tool call ready for the (possibly concurrent) async
230/// `execute()` step.
231enum PreparedCall {
232    /// A final `(output, is_error)` result — no `Tool::execute` call is
233    /// coming for this one.
234    Done((String, bool)),
235    /// Passed every synchronous check; `execute(args, &ctx)` on the named
236    /// registry tool is the only remaining step.
237    Ready {
238        name: String,
239        args: serde_json::Value,
240    },
241}
242
243/// Marker prefix of the notice [`Agent::cap_tool_output`] appends to an
244/// oversized tool result kept in `history` (the recorder receives the full
245/// UX-26 (B7-warn): current wall-clock time as unix milliseconds, the same
246/// unit [`supercode_interchange::sidecar::rfc3339_to_ms`] parses session timestamps into —
247/// lets [`Agent::build_request_messages`] compare "now" against a
248/// cross-process signal (a loaded session's last message timestamp) on
249/// equal footing with an in-process one (this agent's own last annotated
250/// send). Saturates to 0 on a pre-epoch clock rather than panicking (never
251/// happens on real hardware, but `duration_since` can theoretically error).
252fn now_ms() -> i64 {
253    std::time::SystemTime::now()
254        .duration_since(std::time::UNIX_EPOCH)
255        .map(|d| d.as_millis() as i64)
256        .unwrap_or(0)
257}
258
259/// P5-3: process-wide sequence number backing [`next_subagent_id`] —
260/// disambiguates two spawns landing in the same millisecond (which
261/// `now_ms()` alone cannot).
262static SUBAGENT_ID_SEQ: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(0);
263
264/// P5-3: a fresh, process-unique child agent id (`"agent-<hex-ts>-<hex-seq>"`
265/// — the native analog of Claude Code's `agent-<id>` naming, see
266/// `supercode_interchange::session::SessionMeta::agent_id`'s doc comment).
267fn next_subagent_id() -> String {
268    let seq = SUBAGENT_ID_SEQ.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
269    format!("agent-{:x}-{:x}", now_ms(), seq)
270}
271
272/// P5-4: the shape [`Agent::child_approval_handler_factory`]/
273/// [`Agent::set_child_approval_handler_factory`] share — factored into its
274/// own alias (clippy `type_complexity`) rather than spelled out inline at
275/// both use sites.
276type ChildApprovalHandlerFactory = dyn Fn(
277        String,
278        std::sync::Arc<std::sync::Mutex<Vec<crate::subagents::QueuedApproval>>>,
279    ) -> std::sync::Arc<dyn crate::permissions::PermissionsApprovalHandler>
280    + Send
281    + Sync;
282
283/// BP-4 (catalog:109 "Context-usage introspection"): the live
284/// context-window accounting [`Agent::context_usage`] reports — cc's
285/// `/context` grid and cx's `/status` + `get_context_remaining` in one
286/// shape, over the numbers `resume --dry-run`'s preflight already computes.
287///
288/// Every token figure is the SAME estimate the context guard enforces
289/// (`supercode_runtime`), so what this reports and what refuses an oversized
290/// turn can never disagree.
291#[derive(Debug, Clone, PartialEq, Eq, serde::Serialize, serde::Deserialize)]
292pub struct ContextUsage {
293    /// The model the accounting is against.
294    pub model: String,
295    /// Messages in the projected request view (reduction stubs included).
296    pub messages: usize,
297    /// Estimated tokens for those messages.
298    pub message_tokens: u64,
299    /// Tools advertised on the next request.
300    pub tool_count: usize,
301    /// Estimated tokens for the serialized tool-schema array — a real part
302    /// of the wire request, and the half a message-only count misses.
303    pub tool_schema_tokens: u64,
304    /// `message_tokens + tool_schema_tokens`.
305    pub request_tokens: u64,
306    /// `request_tokens` with the guard's safety margin applied — the figure
307    /// the context guard actually compares.
308    pub projected_tokens: u64,
309    /// Headroom the guard reserves for the model's own reply.
310    pub response_reserve_tokens: u64,
311    /// The model's context window, when known.
312    pub context_limit: Option<u64>,
313    /// Tokens still available after the reply reserve, `0` when unknown.
314    pub remaining_tokens: u64,
315    /// `projected_tokens` as a whole percentage of the window (rounded),
316    /// `0` when the window is unknown. An integer so this whole struct
317    /// stays `Eq`-comparable on the frontend wire.
318    pub used_pct: u32,
319    /// Whether the next request would pass the context guard.
320    pub fits: bool,
321}
322
323impl ContextUsage {
324    /// One human line, the shape a `/context` command prints.
325    pub fn summary_line(&self) -> String {
326        match self.context_limit {
327            Some(limit) => format!(
328                "{} · {}% of {} used ({} projected, {} left) · {} messages {} · {} tool schemas {}",
329                self.model,
330                self.used_pct,
331                supercode_runtime::fmt_approx_tokens(limit),
332                supercode_runtime::fmt_approx_tokens(self.projected_tokens),
333                supercode_runtime::fmt_approx_tokens(self.remaining_tokens),
334                self.messages,
335                supercode_runtime::fmt_approx_tokens(self.message_tokens),
336                self.tool_count,
337                supercode_runtime::fmt_approx_tokens(self.tool_schema_tokens),
338            ),
339            None => format!(
340                "{} · context window unknown · {} projected · {} messages {} · {} tool schemas {}",
341                self.model,
342                supercode_runtime::fmt_approx_tokens(self.projected_tokens),
343                self.messages,
344                supercode_runtime::fmt_approx_tokens(self.message_tokens),
345                self.tool_count,
346                supercode_runtime::fmt_approx_tokens(self.tool_schema_tokens),
347            ),
348        }
349    }
350}
351
352/// A stateful agent: configuration, a model transport, a tool set, and the
353/// running conversation. Drive it with [`Agent::send`].
354/// BP-13 — one hop the run loop's failure-fallback pass performed: the
355/// model it was on, the model it moved to, and the provider failure that
356/// made it move.
357#[derive(Debug, Clone, PartialEq, Eq)]
358pub struct FallbackHop {
359    /// The model that failed.
360    pub from: String,
361    /// The next chain entry, which the request was re-sent against.
362    pub to: String,
363    /// The failure, rendered — the record's `reason`.
364    pub reason: String,
365}
366
367/// BP-13 — whether `error` is the kind of failure ANOTHER MODEL could
368/// plausibly answer, i.e. one the fallback chain exists for.
369///
370/// Deliberately narrow: rate limiting (429) and server-side failures (5xx,
371/// which is where "overloaded" lives) are properties of the model/endpoint
372/// that was asked, so asking a different one is a real remedy. Everything
373/// else — a bad request, a refused key, a decode failure, a tool error —
374/// is the CALLER's problem and would fail identically against every entry
375/// in the chain, so walking it would only multiply the same error by three.
376/// The transport's own retry (`OpenAiProvider::send_with_retry`) has
377/// already run and given up by the time this is consulted.
378pub fn is_failover_worthy(error: &Error) -> bool {
379    matches!(error, Error::Provider { status, .. } if *status == 429 || *status >= 500)
380}
381
382pub struct Agent {
383    config: Config,
384    provider: std::sync::Arc<dyn Provider>,
385    registry: ToolRegistry,
386    history: Vec<ChatMessage>,
387    ctx: ToolContext,
388    /// Cumulative output (completion) tokens across every `send` on this agent.
389    total_output_tokens: u64,
390    /// Names of non-core tools discovered via `tool_search` (B6): advertised
391    /// starting with the *next* request once populated.
392    activated_tools: HashSet<String>,
393    /// The live sidecar writer (A3), if this agent is recording. `None` is
394    /// today's behavior, at zero cost: every append point becomes a no-op.
395    recorder: Option<SidecarWriter>,
396    /// BP-8 (catalog:150 "Append-only durable transcript"): the live
397    /// append-only journal, if one is installed
398    /// ([`Self::set_journal`], armed by the caller when
399    /// [`Config::session_append_only`] is on). Behind an `Arc<Mutex<_>>`
400    /// rather than owned outright because the queue doors
401    /// ([`Self::queue_steer`]) take `&self` — a pending input has to be
402    /// recorded from a shared handle while a turn holds `&mut Agent`.
403    /// `None` (the default) is a no-op at every append point: today's
404    /// behavior, no file created.
405    journal: Option<std::sync::Arc<std::sync::Mutex<crate::session_journal::SessionJournal>>>,
406    /// BP-8 (catalog:151 "In-place conversation tree"): the live
407    /// `SessionTree` for this session, materialized when
408    /// [`Config::session_tree_enabled`] is on. Every recorded message
409    /// becomes a node, and [`Self::rewind_conversation`] moves the active
410    /// branch's leaf — the tree is what makes a rewind lossless (the old
411    /// leaf is preserved under a sibling branch) rather than a truncation.
412    /// `None` (the default, and every preset that leaves the module off) is
413    /// zero cost: nothing is built and nothing is persisted.
414    session_tree: Option<supercode_interchange::session_tree::SessionTree>,
415    /// BP-8 (catalog:152 "Rewind/rollback conversation"): tails removed by
416    /// rewinds that have not been undone, newest last. Restored from the
417    /// journal on resume, so "undo the rewind" survives a restart.
418    rewind_undo: Vec<Vec<ChatMessage>>,
419    /// BP-8 (catalog:156): the plan as last written to the journal —
420    /// compared against `ctx.plan` so an unchanged plan is not re-journaled
421    /// on every loop iteration.
422    journaled_plan: Vec<crate::session_journal::PlanEntry>,
423    /// Reversible reduction policy (A5/A7/A10). `None` is today's behavior,
424    /// at zero cost: every provider request is built from `self.history`
425    /// verbatim, exactly as before this landed.
426    reduction_policy: Option<ReductionPolicy>,
427    /// The accumulating reduction log (A5): fed back into
428    /// [`reduce::project_messages`] on every request-build so already-applied
429    /// reductions reproduce verbatim across turns and `send` calls (prefix
430    /// stability). `history` itself is never touched by this — see
431    /// `Self::run_loop`.
432    reduction_log: ReductionLog,
433    /// B7: length of the stable, byte-identical-across-turns prefix at the
434    /// front of [`Self::history`] — this agent's own system message plus
435    /// every message of a previously-imported session — set by
436    /// [`Self::load_session`]. `None` (the default) means no session has been
437    /// loaded, so [`crate::provider::apply_cache_plan`] has nothing to
438    /// annotate even under [`CachePlan::ImportedPrefix`].
439    imported_prefix_len: Option<usize>,
440    /// BP-11: set for the duration of a `/compact` so the `pre_compact`
441    /// observer can tell a manual compaction from an automatic trigger.
442    compacting_manually: bool,
443    /// BP-4 (catalog:90 "Environment context block", cx§2 "re-emitted on
444    /// change"): the `# Environment` block currently spliced into
445    /// `history[0]`, verbatim — `None` when `core.env_context` is off (or
446    /// on a construction path that assembles no prompt). Kept so
447    /// [`Self::refresh_env_context`] can locate and replace exactly this
448    /// text when cwd, approval/sandbox policy or the git branch moves
449    /// mid-session, instead of leaving the model reading a block that
450    /// stopped being true.
451    env_context_live: Option<String>,
452    /// BP-5 (catalog D2 "Per-model-family base-prompt selection"): the base
453    /// system prompt currently spliced at the head of `history[0]`,
454    /// verbatim. Kept for the same reason [`Self::env_context_live`] is:
455    /// when the model changes ([`Self::set_model`]) the family's prompt
456    /// changes with it, and the stale text has to be located and REPLACED
457    /// rather than left in front of the new model.
458    base_prompt_live: String,
459    /// BP-5 (catalog D2 "Shell-output injection in templates/skills"): the
460    /// permission-engine authorization every `` !`cmd` `` in a skill or
461    /// command body runs under. Inert (executes nothing) unless
462    /// `[core.skills] shell_injection` is on.
463    shell_injection: crate::skills::ShellInjection,
464    /// BP-4 (catalog:91): blocks spliced into the context AFTER
465    /// construction — see [`Self::inject_context_block`] and
466    /// [`crate::context_injection`]. Empty by default, at zero cost.
467    spliced_context_blocks: Vec<crate::config::ContextInjectionBlock>,
468    /// TR-7 (T20): the injectable side-call ([`reduce::summarize::SpanSummarizer`])
469    /// used to summarize an A10 `TurnsCleared` span, if one is installed
470    /// ([`Self::set_span_summarizer`]). `None` is today's behavior, at zero
471    /// cost: `Self::build_request_messages` never calls
472    /// [`reduce::prepare_cleared_turns_summary`] without one, so
473    /// `policy.summarize_cleared_turns` being on with no summarizer
474    /// installed behaves exactly like it being off (deterministic stub only)
475    /// — never a panic, never a blocked request.
476    span_summarizer: Option<std::sync::Arc<dyn reduce::summarize::SpanSummarizer + Send + Sync>>,
477    /// TR-8 (T5): the tool-schema tier signature (global knob + per-tool
478    /// overrides) as of the last request this agent built, or `None` before
479    /// the first request. Compared against the CURRENT signature at the top
480    /// of every `Self::build_request_messages` call so a tier change made
481    /// mid-session (via [`Self::set_schema_tier`] /
482    /// [`Self::set_tool_schema_tier`]) is detected and flagged to the B7
483    /// cache planner as a cache-bust event (`provider::tier_change_is_cache_bust`).
484    last_tool_schema_tier_signature: Option<u64>,
485    /// PARITY-18 D4 — the target model's context-window size, if the caller
486    /// has armed the guard via [`Self::set_context_limit`]. `None` (the
487    /// default) means no guard: every request is sent unconditionally.
488    /// CLI entry points arm it for their resolved model; direct SDK callers
489    /// retain explicit control through [`Self::set_context_limit`].
490    /// Once set, `Self::run_loop` re-checks
491    /// [`supercode_runtime::context_guard`] before EVERY request it builds —
492    /// not just the first — so "never sends an over-context request" holds
493    /// for the whole session, not only a one-shot preflight.
494    context_limit: Option<u64>,
495    /// PARITY-18 D3 — becomes `true` the first time `Self::run_loop`
496    /// actually reaches its real send site (immediately before
497    /// [`Provider::complete`]). Exposed via [`Self::request_issued`] so a
498    /// caller can report "request sent" truthfully — never asserted ahead
499    /// of time, so a pre-delivery failure (guard refusal, a build error) or
500    /// an interactive session that quits before any turn completes is
501    /// reported honestly as "not sent".
502    requests_issued: bool,
503    /// UX-26 (B7-warn): unix-ms wall-clock time this agent last knew the
504    /// active [`CachePlan::ImportedPrefix`] breakpoint to be warm. Seeded by
505    /// [`Self::load_session`] from the just-loaded session's OWN last
506    /// message timestamp (`metadata["timestamp"]`, parsed via
507    /// [`supercode_interchange::sidecar::rfc3339_to_ms`]) — a cross-process signal: how long
508    /// the resumed conversation has sat idle since ANY tool last touched it,
509    /// which is exactly when Anthropic's server-side cache entry (if one
510    /// ever existed) was last capable of being warm. Refreshed to "now"
511    /// every time `Self::run_loop` actually sends a cache-annotated
512    /// request (an in-process signal: idle time between this agent's own
513    /// turns). `None` when no imported prefix exists yet, or the loaded
514    /// session's last message carries no parseable timestamp — never
515    /// guessed, so the TTL check in [`provider::cache_cold_reason`] simply
516    /// doesn't fire rather than risk a false positive.
517    last_cache_activity_ms: Option<i64>,
518    /// UX-26: whether a PRIOR request already carried a cache_control
519    /// annotation for the current [`Self::imported_prefix_len`] — i.e.
520    /// whether reuse is genuinely "expected" on the NEXT annotated request.
521    /// `false` until the first annotated request goes out (that one is
522    /// establishing the cache entry, a legitimate write, never a "miss") and
523    /// reset to `false` by [`Self::load_session`] whenever the imported
524    /// prefix itself changes.
525    cache_established: bool,
526    /// UX-26 scratch: this turn's cache-warmth context, computed once at the
527    /// top of `Self::build_request_messages` (before the request is sent,
528    /// while `effective_cache_plan`/`busted` are in scope) and consumed once
529    /// in `Self::run_loop` right after `usage` comes back — never read
530    /// across turns, so a stale value can't leak. `(will_annotate,
531    /// cache_established, idle_secs)` — see [`provider::cache_cold_reason`]
532    /// for what each of the first two independently gates.
533    pending_cache_turn: (bool, bool, Option<i64>),
534    /// P4b: the injectable auto-title side-call ([`Self::set_session_titler`]),
535    /// mirroring `Self::span_summarizer`'s "installing one alone changes
536    /// nothing" contract — `Config::auto_title` is the actual gate a caller
537    /// consults before invoking [`Self::auto_title`].
538    session_titler: Option<std::sync::Arc<dyn crate::session_title::SessionTitler + Send + Sync>>,
539    /// P4b (§1.6, catalog §4a "persisted per-turn usage records"): every
540    /// [`crate::usage_log::UsageRecord`] recorded so far this agent's
541    /// lifetime. Always accumulated (cheap, small) regardless of whether a
542    /// caller ever persists it — see [`Self::usage_records`]/
543    /// [`Self::save_usage_log`].
544    usage_log: Vec<crate::usage_log::UsageRecord>,
545    /// P4b: 0-based index of the NEXT model round-trip, for
546    /// [`crate::usage_log::UsageRecord::turn`].
547    turn_index: usize,
548    /// BP-7 (catalog §4a "Turn/step bracketing records"): the per-round-trip
549    /// marker log — context/usage/finish brackets plus the retry, abort,
550    /// effort and goal markers. Persisted beside the session as
551    /// `<name>.events.jsonl` (see [`Self::save_turn_records`]).
552    turn_records: Vec<crate::turn_record::TurnRecord>,
553    /// BP-7: retries the transport reported, drained after every
554    /// `complete()` so each notice attaches to the round-trip that produced
555    /// it. Only the HTTP provider built by [`Self::new`] writes into this;
556    /// an injected provider simply never records anything.
557    retry_log: std::sync::Arc<crate::provider::RetryLog>,
558    /// BP-7 (catalog §4a "Per-turn cost/usage accounting", "Turn/budget
559    /// caps"): the price to bill this agent's model at, resolved at
560    /// construction from [`Config::price_input_per_mtok`]/
561    /// [`Config::price_output_per_mtok`] or [`crate::pricing`]'s table, and
562    /// re-resolved by [`Self::set_model`]. `None` = unpriceable, so no cost
563    /// is recorded (never a guess).
564    model_price: Option<crate::pricing::ModelPrice>,
565    /// BP-7: dollars this agent has spent across its whole lifetime — the
566    /// counter [`Config::max_budget_usd`] is measured against.
567    total_cost_usd: f64,
568    /// BP-7: tool calls this agent has executed across its whole lifetime —
569    /// the counter [`Config::max_steps`] is measured against.
570    total_steps: usize,
571    /// BP-7 (catalog §4a "Background subagents + resume"): a finished
572    /// child's post-system-prompt transcript, kept after the reap so
573    /// `subagent_resume` can restore its context in-process. A session
574    /// with a subagent store attached also has it on disk; this makes
575    /// resume work for an embedder that never attached one.
576    reaped_subagents: std::collections::HashMap<String, Vec<ChatMessage>>,
577    /// BP-7 (catalog §4a "Goals"): the session's standing objective, when
578    /// `capabilities.todos.goals` is on and one has been set. Restated at
579    /// the TAIL of every request while it stands (see
580    /// [`crate::goals::GoalRecord::reminder`]) and persisted as
581    /// `<session>.goal.json` — never written into `history`, so the
582    /// transcript stays exactly what the conversation was.
583    goal: Option<crate::goals::GoalRecord>,
584    /// P4b (§1.7/§3.1 `core.steering`, pi§3 semantics): queued mid-turn
585    /// steering messages — drained at the top of `Self::run_loop`'s next
586    /// iteration (pi's "steer = after current tool calls"). Empty by
587    /// default, at zero cost: `Self::run_loop` skips the drain entirely
588    /// when empty.
589    steer_queue: std::sync::Arc<std::sync::Mutex<SteerInbox>>,
590    /// P4b: queued follow-up messages — drained only once the loop is
591    /// otherwise idle (pi's "follow-up = at idle"), i.e. exactly the point
592    /// `Self::run_loop` would otherwise return a final answer.
593    follow_up_queue: std::collections::VecDeque<String>,
594    /// P4c (§5.2 P4 "doom-loop breaker", §3.1 `core.doom_loop_threshold`):
595    /// `(tool name, canonical JSON args)` of the most recent tool call, if
596    /// [`Config::doom_loop_threshold`] is armed — `None` before the first
597    /// call this agent has run. See [`Self::check_doom_loop`].
598    doom_loop_last_call: Option<(String, String)>,
599    /// P4c: how many times [`Self::doom_loop_last_call`] has repeated
600    /// consecutively so far (starts at 1 on the call that SET it).
601    doom_loop_streak: u32,
602    /// P4c (§1.10/§3.1 `core.model_switch.allow_switch`): every
603    /// [`crate::model_change::ModelChangeRecord`] [`Self::switch_model`] has
604    /// created so far this agent's lifetime. Always empty when
605    /// `Config::model_switch_allow_switch` is off (the default) or no
606    /// switch has happened yet.
607    model_change_log: Vec<crate::model_change::ModelChangeRecord>,
608    /// P4e (§1.6/§3.1 `core.session.git_metadata`, catalog:331): captured
609    /// once at construction when [`Config::session_git_metadata`] is on;
610    /// `None` when the gate is off (the default) or the best-effort git
611    /// probe found nothing (not a repo, `git` missing). See
612    /// [`Self::git_metadata`]/[`Self::save_git_metadata`].
613    git_metadata: Option<crate::git_metadata::GitMetadataRecord>,
614    /// P5-1 (§2.10, session-scoped "approve for session" cache): populated
615    /// only when a [`crate::permissions::PermissionsApprovalHandler`]
616    /// returns [`crate::permissions::ApprovalOutcome::AllowForSession`] —
617    /// see [`Self::prepare_tool_call`]'s `Config::permissions_enabled`
618    /// branch. Always constructed (cheap, empty) regardless of whether the
619    /// engine is ever active — the same "zero cost when off" posture as
620    /// [`Self::doom_loop_last_call`].
621    permissions_approval_cache: crate::permissions::ApprovalCache,
622    /// P5-1: the non-interactive decision seam a caller installs via
623    /// [`Self::set_permissions_approval_handler`] — mirrors
624    /// `Self::span_summarizer`/[`Self::session_titler`]'s "installing one
625    /// alone changes nothing, `Config::permissions_enabled` is the actual
626    /// gate" pattern. `None` (the default) means every `Ask`-tier decision
627    /// is denied (fail-closed — see
628    /// `crate::permissions::approval::PermissionsApprovalHandler`'s doc
629    /// comment).
630    permissions_approval_handler:
631        Option<std::sync::Arc<dyn crate::permissions::PermissionsApprovalHandler>>,
632    /// P5-2 (§2 module 15 D7 row 4 "prompts-as-commands"): MCP server
633    /// prompts registered via [`Self::register_mcp_prompt`], keyed by their
634    /// ALREADY-NAMESPACED command name (`mcp__<server>__<prompt>` — see
635    /// [`crate::mcp::McpPromptSource`]'s doc comment for why that namespace
636    /// is what keeps an untrusted server's prompt from ever colliding with
637    /// a trusted `Config::prompts` entry). Empty by default, at zero cost:
638    /// [`Self::expand_prompt_async`] only consults this after
639    /// `Config::prompts` finds no match.
640    mcp_prompts: std::collections::HashMap<String, Box<dyn crate::sdk::SdkPromptSource>>,
641    /// BP-6 (catalog D2 "Skills (progressive-disclosure packages)", D7
642    /// "Skill discovery from multiple roots"): the SKILL.md packages
643    /// discovered for this config, frontmatter only — name, description,
644    /// version and the manifest path. Never a body: a body is read from
645    /// disk on invocation (`/name`, `/skill:name`, a `$slug` mention, or
646    /// the `skill` tool) and nowhere else. Empty unless `[core.skills]` is
647    /// on AND names a harness whose roots to read.
648    skills: Vec<crate::skills::LoopSkill>,
649    /// P5-3 (§2 module 9): how deep in the spawn tree THIS agent is — `0`
650    /// for a top-level agent. Set from [`Config::subagent_depth`] at
651    /// construction; `Self::run_spawn_subagent` builds a child `Config`
652    /// with `subagent_depth = self.subagent_depth + 1` and ALSO overwrites
653    /// the freshly-built child `Agent`'s own field to match (belt-and-
654    /// suspenders — the child never has to trust its own `Config` alone).
655    subagent_depth: usize,
656    /// P5-3 (resource bound, "must not fork-bomb"): the shared, tree-wide
657    /// concurrency gauge every spawn (this agent's own, and every
658    /// descendant's) increments/decrements against
659    /// (`crate::subagents::try_acquire`/`ConcurrencyGuard`). A TOP-level
660    /// agent gets a fresh `Arc::new(AtomicUsize::new(0))` at construction;
661    /// `Self::run_spawn_subagent` clones this SAME `Arc` into every child it
662    /// spawns (never a fresh one), so a cap of N holds across the WHOLE
663    /// tree regardless of its branching shape — a parent with 3 children
664    /// each spawning 3 more shares one counter, not nine independent ones.
665    subagent_concurrency_gauge: std::sync::Arc<std::sync::atomic::AtomicUsize>,
666    /// P5-3 (D3 "background+resume"): background subagents this agent has
667    /// spawned and not yet reaped via `subagent_status`, keyed by their
668    /// `child_agent_id`. Each entry's `JoinHandle` moves its own
669    /// [`crate::subagents::ConcurrencyGuard`] into the spawned task, so the
670    /// concurrency slot is held for exactly as long as the child is
671    /// actually running, independent of whether/when the parent polls.
672    background_subagents: std::collections::HashMap<String, BackgroundSubagent>,
673    /// P5-3 (§2.2 C6 "parent-surfaced queue"): approval requests a
674    /// `background_prompts = "parent"` child raised, queued here rather
675    /// than blocking (see [`crate::subagents::QueuedApproval`]'s doc
676    /// comment — each is already resolved `Deny` by the time it lands
677    /// here). Exposed read-only via [`Self::pending_child_approvals`].
678    /// Always constructed (cheap, empty) regardless of whether background
679    /// spawning is ever used, same "zero cost when off" posture as
680    /// [`Self::permissions_approval_cache`].
681    pending_child_approvals:
682        std::sync::Arc<std::sync::Mutex<Vec<crate::subagents::QueuedApproval>>>,
683    /// P5-4 (tui, closes the P5-3 §2.2 C6 deferred chain — see
684    /// [`crate::subagents::ParentQueueApprovalHandler`]'s doc comment for
685    /// the "never blocks" contract this OVERRIDES only when a factory is
686    /// installed): when `Some`, `Self::run_spawn_subagent` uses THIS
687    /// factory — instead of constructing the default never-blocking
688    /// [`crate::subagents::ParentQueueApprovalHandler`] — to build the
689    /// `PermissionsApprovalHandler` a `background_prompts = "parent"`
690    /// child gets. Installed via
691    /// [`Self::set_child_approval_handler_factory`] by a `tui` embedder
692    /// that wants queued child approvals to be genuinely ANSWERABLE
693    /// (blocks the child's tool call until the parent resolves it, or
694    /// denies if the factory's handler's channel is ever dropped/closed —
695    /// still fail-closed, never a hang past process lifetime). `None` (the
696    /// default) preserves P5-3's shipped behavior byte-for-byte: every
697    /// `background_prompts = "parent"` child still gets the immediate-deny
698    /// `ParentQueueApprovalHandler`, and [`Self::pending_child_approvals`]
699    /// stays exactly the read-only audit view it already is.
700    child_approval_handler_factory: Option<std::sync::Arc<ChildApprovalHandlerFactory>>,
701    /// P5-3 (D5 "subagent transcripts… persisted + linked"): an optional
702    /// `(store, this agent's own session name)` pair installed via
703    /// [`Self::set_subagent_store`] — mirrors [`Self::set_recorder`]/
704    /// [`Self::set_span_summarizer`]'s "installing one alone changes
705    /// nothing" pattern. `None` (the default) means a spawned child's
706    /// transcript/lineage is still joined back into THIS agent's context
707    /// (the foreground/background mechanics work either way) but nothing
708    /// is written to a [`crate::store::SessionStore`] — no behavior change
709    /// for any caller that never installs one (e.g. every pre-P5-3 caller).
710    subagent_store: Option<(std::sync::Arc<crate::store::SessionStore>, String)>,
711    /// Imported Claude runtime state. The manifest can be paused or active,
712    /// but this Agent contains no scheduler or timer handle; an embedding
713    /// driver owns execution and persistence.
714    claude_runtime_manifest: Option<crate::claude_runtime_state::ClaudeRuntimeManifest>,
715    /// P5-6 (§2 module 4 `tools.background`, D10 "bg-manager"): background
716    /// OS processes spawned via `background_exec`, keyed by job id, tracked
717    /// until reaped (a terminal `background_status` poll, or an explicit
718    /// `background_kill`) — see [`BackgroundJob`]'s doc comment. Always
719    /// constructed (cheap, empty), same "zero cost when off" posture as
720    /// [`Self::background_subagents`].
721    background_jobs: std::collections::HashMap<String, BackgroundJob>,
722    /// P5-6 (resource bound, mirroring [`Self::subagent_concurrency_gauge`]'s
723    /// own precedent): the shared concurrency gauge every `background_exec`
724    /// call on this agent increments/decrements against
725    /// (`crate::subagents::try_acquire`/`ConcurrencyGuard` — reused
726    /// verbatim, a second independent gauge instance scoped to background
727    /// JOBS rather than subagent SPAWNS).
728    background_concurrency_gauge: std::sync::Arc<std::sync::atomic::AtomicUsize>,
729    /// P5-9 (§2 module 20 `checkpoint`): the write-path-interception
730    /// observer installed on [`Self::ctx`]'s `write_observer` (as a
731    /// `dyn WriteObserver`), held here ADDITIONALLY as its concrete type so
732    /// `Self::run_loop` can call
733    /// [`crate::checkpoint::CheckpointObserver::begin_turn`] once per turn.
734    /// `None` when `Config::checkpoint_enabled` is `false` (the default) or
735    /// the shadow store failed to open — see
736    /// [`crate::checkpoint::observer_for_config`].
737    checkpoint_observer: Option<std::sync::Arc<crate::checkpoint::CheckpointObserver>>,
738    /// P5-11 (§2 module 28 `lsp`): the LSP server registry installed (via
739    /// `crate::lsp::LspDiagnosticsObserver`) on `Self::ctx`'s
740    /// `write_observer` chain, held here ADDITIONALLY as its concrete type
741    /// so `impl Drop for Agent` can reach
742    /// [`crate::lsp::LspManager::kill_all_sync`] (no orphaned language-
743    /// server processes) and a clean-exit caller can reach
744    /// [`crate::lsp::LspManager::shutdown_all`] for a graceful handshake.
745    /// `None` when `Config::lsp_enabled` is `false` (the default).
746    lsp_manager: Option<std::sync::Arc<crate::lsp::LspManager>>,
747}
748
749/// P5-6 (§2 module 4 `tools.background`): one background-spawned OS process
750/// this agent is tracking, awaiting a `background_status`/`background_list`
751/// poll (or `background_kill`/agent drop) to reap or terminate it.
752///
753/// **Real process, not a child agent.** Unlike [`BackgroundSubagent`] (which
754/// wraps a whole recursive child [`Agent`] loop against the SAME mock/real
755/// provider), this wraps a plain OS subprocess spawned via
756/// `crate::tools::build_sandboxed_sh` — the exact function
757/// [`crate::tools::BashTool::execute`] itself calls, so a background
758/// command gets byte-identical sandboxing/cwd/env handling to a foreground
759/// `bash` call (build brief: "reuse the bash tool's execution + sandbox
760/// path").
761struct BackgroundJob {
762    /// The live process handle — kept directly on the job (not moved into a
763    /// spawned task) so [`Agent::run_background_status`]/
764    /// [`Agent::run_background_list`] can call the SYNCHRONOUS,
765    /// non-blocking `Child::try_wait` to observe exit status, and
766    /// [`Agent::run_background_kill`]/[`impl Drop for Agent`] can call the
767    /// SYNCHRONOUS `Child::start_kill` for a REAL process kill — never just
768    /// a `tokio::task::JoinHandle::abort` (which would only cancel a Rust
769    /// future, not the OS process it spawned). `kill_on_drop(true)` was set
770    /// at spawn time as defense-in-depth: even a `BackgroundJob` dropped
771    /// through some path OTHER than the explicit kill call sites below
772    /// still kills its child (a documented tokio behavior; a no-op if the
773    /// process already exited).
774    child: tokio::process::Child,
775    /// The exact command text this job is running — the SAME text that was
776    /// already checked against the permissions engine at spawn time (see
777    /// [`Agent::background_permission_denial`]).
778    command: String,
779    /// The OS process id, captured once at spawn time (before `child` is
780    /// ever mutated) — surfaced in every status/list/kill result, and the
781    /// only thing an OUTSIDE observer (e.g. a test proving real
782    /// termination) needs to check liveness independent of this process's
783    /// own bookkeeping.
784    pid: Option<u32>,
785    /// Bounded, incrementally-appended combined stdout+stderr capture —
786    /// written to by the reader tasks [`Agent::run_background_exec`] spawns
787    /// right after `child.stdout`/`child.stderr` are taken, read by every
788    /// status/list poll. Shared via `Arc` since the reader tasks outlive
789    /// this method call.
790    output: std::sync::Arc<supercode_runtime::background::CapturedOutput>,
791    /// Unix-ms wall-clock time the spawn happened.
792    started_at_ms: i64,
793    /// Set by [`Agent::run_background_kill`] — [`Agent::run_background_status`]/
794    /// [`Agent::run_background_list`] report [`supercode_runtime::background::JobStatus::Killed`]
795    /// unconditionally once this is `true`, rather than racing
796    /// `Child::try_wait` to see whether the kill signal has landed yet.
797    killed: bool,
798    /// The concurrency-gauge slot this job holds for as long as it remains
799    /// in [`Agent::background_jobs`] — dropped (freeing the slot) when this
800    /// `BackgroundJob` is removed from the map (a terminal reap, or an
801    /// explicit kill), exactly mirroring [`BackgroundSubagent`]'s own
802    /// "guard held for as long as it's tracked, not just while the process
803    /// is alive" posture (§2 module 9 precedent, kept consistent here).
804    _guard: crate::subagents::ConcurrencyGuard,
805}
806
807/// P5-6: the non-blocking status read [`Agent::run_background_status`]/
808/// [`Agent::run_background_list`] share — `job.killed` (set by
809/// [`Agent::run_background_kill`]) always wins over a fresh `try_wait`,
810/// since a kill signal racing the OS reaping the process is otherwise
811/// indistinguishable from "still running" for one poll cycle; reporting
812/// `Killed` unconditionally once requested avoids that race entirely. A
813/// `try_wait` error (would only happen if this job's id were somehow
814/// double-reaped, which the map ownership below already prevents) is
815/// treated as "no news yet" — `Running` — rather than inventing a made-up
816/// exit code.
817fn background_job_status(job: &mut BackgroundJob) -> supercode_runtime::background::JobStatus {
818    if job.killed {
819        return supercode_runtime::background::JobStatus::Killed;
820    }
821    match job.child.try_wait() {
822        Ok(Some(status)) => supercode_runtime::background::JobStatus::Exited(status.code()),
823        Ok(None) | Err(_) => supercode_runtime::background::JobStatus::Running,
824    }
825}
826
827/// Fable-5 review (HIGH, "grandchildren orphaned on kill AND agent-drop"):
828/// the shared real-kill body for both [`Agent::run_background_kill`] and
829/// `impl Drop for Agent` — sends `SIGKILL` to `job`'s ENTIRE process group,
830/// not just the one directly-tracked pid, so a surviving `&` job, pipeline
831/// stage, or double-forking daemon spawned by the job is killed too, then
832/// reaps the group leader so it doesn't linger as a zombie.
833///
834/// Relies on the spawn site (`Agent::run_background_exec`) having put the
835/// job in its OWN new process group via `Command::process_group(0)` — which
836/// makes the leader's pgid equal to its own pid, so `job.pid` doubles as the
837/// group id here.
838#[cfg(unix)]
839fn kill_job_process_group(job: &mut BackgroundJob) {
840    if let Some(pid) = job.pid {
841        // SAFETY: `libc::kill` with a negative pid is `killpg` — it only
842        // ever sends a signal (never dereferences memory), so this is safe
843        // regardless of whether the group is still alive. A `-1`/`ESRCH`
844        // return means the leader (and thus the whole group, since a group
845        // can't outlive its leader) already exited — not an error, just
846        // "already dead", exactly like `Child::start_kill`'s own documented
847        // no-op-on-already-exited contract.
848        unsafe {
849            libc::kill(-(pid as libc::pid_t), libc::SIGKILL);
850        }
851    }
852    // Belt-and-suspenders for the leader itself — `kill_on_drop(true)` set
853    // at spawn time is the same outcome via a different (implicit) path —
854    // then reap it so the SIGKILL we just delivered doesn't leave a zombie
855    // behind.
856    let _ = job.child.start_kill();
857    let _ = job.child.try_wait();
858}
859
860/// Non-unix fallback: no portable process-group primitive is wired up here
861/// (same posture as `crate::tools::build_sandboxed_sh`'s own platform
862/// split) — falls back to the pre-fix per-child kill. A background job that
863/// spawns a surviving grandchild process on a non-Unix target is a
864/// documented residual, not silently claimed fixed by this cfg arm.
865#[cfg(not(unix))]
866fn kill_job_process_group(job: &mut BackgroundJob) {
867    let _ = job.child.start_kill();
868}
869
870/// P5-6 (D1 "monitor/event feed", "output captured incrementally +
871/// BOUNDED"): spawn a fire-and-forget reader task that continuously drains
872/// `reader` (a piped `ChildStdout`/`ChildStderr`) into `output`, bounded at
873/// `cap` bytes. Reading NEVER stops at the cap — only what's RETAINED is
874/// bounded ([`supercode_runtime::background::CapturedOutput::append`]'s own contract)
875/// — because a background job's child process would otherwise block
876/// forever writing to a full, undrained OS pipe once this stopped reading
877/// it, silently hanging real work behind an apparently-"running" job. The
878/// task exits on its own once the pipe reaches EOF (the process closed the
879/// descriptor, whether by exiting or being killed) — no explicit
880/// abort/cleanup call site is needed; a detached `tokio::spawn` this short-
881/// lived is not the kind of orphaned-task risk `impl Drop for Agent`'s own
882/// doc comment is about (that one concerns a whole recursive provider-
883/// calling child AGENT loop, not a bounded byte-copy loop that ends the
884/// instant its source pipe closes).
885fn spawn_output_reader<R>(
886    reader: R,
887    output: std::sync::Arc<supercode_runtime::background::CapturedOutput>,
888    cap: usize,
889) -> tokio::task::JoinHandle<()>
890where
891    R: tokio::io::AsyncRead + Unpin + Send + 'static,
892{
893    tokio::spawn(async move {
894        use tokio::io::AsyncReadExt;
895        let mut reader = reader;
896        let mut buf = [0u8; 8192];
897        loop {
898            match reader.read(&mut buf).await {
899                Ok(0) => break,
900                Ok(n) => {
901                    let chunk = String::from_utf8_lossy(&buf[..n]);
902                    output.append(&chunk, cap);
903                }
904                Err(_) => break,
905            }
906        }
907    })
908}
909
910/// BP-7 (catalog §4a "Named agent definitions as data"): merge
911/// `<cwd>/.claude/agents/*.md` into `config.subagents_definitions`.
912///
913/// Runs for every `Agent` whose `subagents` module is on, whatever preset
914/// it came from — before BP-7 the `.md` loader was reachable only from the
915/// Claude emulate/resume path, so a cc-parity or cx-parity session ignored
916/// definitions sitting right there in the repo.
917///
918/// * A no-op when the module is off (the default), and for every spawned
919///   CHILD (`subagent_depth > 0`), which already inherits its parent's
920///   resolved definitions verbatim.
921/// * A config-table entry WINS over a discovered file of the same name:
922///   `[capabilities.subagents.agents.<name>]` is explicit configuration,
923///   the file is discovery.
924/// * A malformed file is skipped with a warning, never a failed
925///   construction: `Agent::with_provider` has no error channel, and a
926///   broken agent file in some repo must not make the harness unusable
927///   there. (The emulate/resume path keeps its own strict behavior, where
928///   a definition the resumed session may depend on going missing IS worth
929///   failing over.)
930fn merge_project_agent_definitions(config: &mut Config) {
931    if !config.subagents_enabled || config.subagent_depth > 0 {
932        return;
933    }
934    match crate::claude_compat::load_project_agents(&config.cwd) {
935        Ok(agents) => {
936            for agent in agents {
937                config
938                    .subagents_definitions
939                    .entry(agent.definition.name.clone())
940                    .or_insert(agent.definition);
941            }
942        }
943        Err(e) => {
944            tracing::debug!(
945                error = %e,
946                "skipping .claude/agents discovery: a definition file could not be parsed"
947            );
948        }
949    }
950}
951
952/// P5-3: one background-spawned child this agent is tracking, awaiting a
953/// `subagent_status` poll (or agent drop) to reap it.
954struct BackgroundSubagent {
955    /// Resolves to `(child_agent_id, child's final result, the child's own
956    /// post-system-prompt history — for D5 transcript persistence once
957    /// reaped)` — the concurrency-guard slot for this child is held INSIDE
958    /// the spawned future (moved in at spawn time), so it releases the
959    /// instant the child's own run loop finishes, not when the parent gets
960    /// around to polling.
961    handle: tokio::task::JoinHandle<(String, Result<String>, Vec<ChatMessage>)>,
962    /// The task/prompt text the child was spawned with (surfaced by a
963    /// `"pending"` status poll, since the handle alone can't answer "what
964    /// is it doing").
965    task: String,
966    /// The named `agent_type` spawned, if any.
967    agent_type: Option<String>,
968    /// Unix-ms wall-clock time the spawn happened.
969    started_at_ms: i64,
970    /// BP-7 (catalog §4a "Background subagents + resume"): the child's own
971    /// steering inbox, captured before the child moved into its task.
972    ///
973    /// This IS the mailbox. `SteerInbox` was built (P4b) to be writable
974    /// while an active turn holds `&mut Agent` — exactly the property a
975    /// message-to-a-running-child needs — so the mailbox is that existing
976    /// seam reached from outside, not a second delivery channel with its
977    /// own ordering rules. A message lands at the top of the child's next
978    /// loop iteration, per `Config::steering_mode`.
979    mailbox: std::sync::Arc<std::sync::Mutex<SteerInbox>>,
980}
981
982/// P5-3 safety hardening (Fable-5 review, MEDIUM-LOW "orphaned billed
983/// spend"): a dropped parent must not leave a detached background child
984/// running against a REAL provider. Without this, a parent dropped
985/// mid-run (the caller's own process exits the scope, panics, or simply
986/// stops polling) leaves every still-running `BackgroundSubagent::handle`
987/// as an orphaned `tokio::spawn` task: nothing had ever awaited or
988/// aborted it, so it runs to its own (`max_iterations`-bounded)
989/// completion regardless — bounded but real provider spend nobody is
990/// paying attention to.
991///
992/// `.abort()` on a [`tokio::task::JoinHandle`] is safe to call
993/// unconditionally, including on an ALREADY-finished task (a documented
994/// no-op there — see tokio's `JoinHandle::abort` docs) — so this never
995/// needs to distinguish "still running" from "already done"; a background
996/// child that already finished and is merely awaiting a `subagent_status`
997/// reap is untouched in practice (aborting a finished task changes
998/// nothing observable). For a task still mid-flight, tokio cancels it at
999/// its next `.await` point, which drops that future in place — including
1000/// the `_guard: ConcurrencyGuard` moved into it at spawn time (see
1001/// `Self::run_spawn_subagent`'s `tokio::spawn` body) — so the
1002/// concurrency-gauge slot is released exactly the same way a normal
1003/// completion releases it (`ConcurrencyGuard`'s own `Drop`, in
1004/// `crate::subagents`). No separate cleanup call site to forget.
1005///
1006/// Deliberately does NOT touch [`Self::pending_child_approvals]` or
1007/// `Self::subagent_store` — this is purely "stop burning provider
1008/// calls on behalf of a caller who's gone", not a transcript-persistence
1009/// path (a child aborted mid-flight has no finished result to persist;
1010/// see this build's named residual on abort-time transcript loss).
1011impl Drop for Agent {
1012    fn drop(&mut self) {
1013        for (child_id, bg) in self.background_subagents.drain() {
1014            // Named, not silent: a child that was still running gets its
1015            // provider calls cut off here — worth a trace even though
1016            // there's no transcript left to persist (the future is
1017            // dropped mid-flight, before it ever returns a result).
1018            if !bg.handle.is_finished() {
1019                tracing::debug!(
1020                    child_id = %child_id,
1021                    "parent Agent dropped: aborting still-running background subagent \
1022                     to stop further provider spend"
1023                );
1024            }
1025            bg.handle.abort();
1026        }
1027        // P5-6 (§2 module 4 `tools.background`, build brief "on agent drop
1028        // / session end, jobs MUST be killed... real process kill via the
1029        // child handle's kill(), not just tokio task abort"): a REAL OS
1030        // process, not a Rust task — `Child::start_kill` (synchronous, no
1031        // `.await` needed, so callable from this non-async `Drop::drop`)
1032        // sends the actual kill signal; a no-op, per its own docs, on a
1033        // job that already exited. `kill_on_drop(true)` (set at spawn
1034        // time) is a second, independent line of defense for the same
1035        // outcome, but this explicit loop is what makes the guarantee
1036        // provable/traceable rather than relying solely on an implicit
1037        // tokio runtime behavior.
1038        for (job_id, mut job) in self.background_jobs.drain() {
1039            if !job.killed {
1040                tracing::debug!(
1041                    job_id = %job_id,
1042                    command = %job.command,
1043                    "parent Agent dropped: killing still-tracked background job's real \
1044                     OS process (and its whole process group — see \
1045                     `kill_job_process_group`)"
1046                );
1047            }
1048            kill_job_process_group(&mut job);
1049        }
1050        // P5-11 (§2 module 28 `lsp`, build brief "no orphaned language-
1051        // server processes"): a REAL OS process, same rationale as the
1052        // background-job loop just above — `kill_all_sync` is
1053        // synchronous (`Child::start_kill`, no `.await` needed, so
1054        // callable from this non-async `Drop::drop`), SIGKILLs each
1055        // server's WHOLE process group (unix — same `kill_job_process_group`
1056        // mechanism as the background-job loop above, so worker
1057        // grandchildren like rust-analyzer's proc-macro server or
1058        // typescript-language-server's `tsserver` are killed too, not just
1059        // the one directly-tracked pid), and is provable/traceable rather
1060        // than relying solely on `kill_on_drop(true)`'s implicit tokio
1061        // runtime behavior (which remains a second, independent line of
1062        // defense on every spawned `LspClient`).
1063        if let Some(lsp) = &self.lsp_manager {
1064            lsp.kill_all_sync();
1065        }
1066    }
1067}
1068
1069/// P4 (§1.8 credential-helper indirection, D6 row): run an `api_key_cmd`
1070/// through the shell and return its trimmed stdout. Runs via `sh -c` (POSIX
1071/// shell, matching pi's `!command` precedent) so the configured string can
1072/// use pipes/substitution, e.g. `pass show api-key`. Never panics or
1073/// propagates an error: a spawn failure or non-zero exit is reported via
1074/// `tracing::warn!` and returns an empty `String`, which
1075/// `Agent::new`'s resolution chain treats exactly like an unset helper —
1076/// falling through to `Config::api_key_env`.
1077fn run_api_key_cmd(cmd: &str) -> String {
1078    match std::process::Command::new("sh").arg("-c").arg(cmd).output() {
1079        Ok(out) if out.status.success() => String::from_utf8_lossy(&out.stdout).trim().to_string(),
1080        Ok(out) => {
1081            tracing::warn!(
1082                "api_key_cmd exited with status {:?}; falling back to api_key_env",
1083                out.status.code()
1084            );
1085            String::new()
1086        }
1087        Err(e) => {
1088            tracing::warn!("api_key_cmd failed to run ({e}); falling back to api_key_env");
1089            String::new()
1090        }
1091    }
1092}
1093
1094/// BP-9 (§3.1 `core.api_key_command`, D6 row "Credential helpers /
1095/// keyring"): run an ARGV credential helper and return its trimmed stdout.
1096/// No shell is involved — `argv[0]` is exec'd with the rest as arguments —
1097/// so a helper path with spaces, or an argument containing `$`/`;`, means
1098/// what it says. Same never-panics, fall-through-on-failure contract as
1099/// [`run_api_key_cmd`]: an empty result is treated as "no helper".
1100pub(crate) fn run_api_key_command(argv: &[String]) -> String {
1101    let Some((program, args)) = argv.split_first() else {
1102        return String::new();
1103    };
1104    match std::process::Command::new(program).args(args).output() {
1105        Ok(out) if out.status.success() => String::from_utf8_lossy(&out.stdout).trim().to_string(),
1106        Ok(out) => {
1107            tracing::warn!(
1108                "api_key_command exited with status {:?}; trying the next credential source",
1109                out.status.code()
1110            );
1111            String::new()
1112        }
1113        Err(e) => {
1114            tracing::warn!(
1115                "api_key_command failed to run ({e}); trying the next credential source"
1116            );
1117            String::new()
1118        }
1119    }
1120}
1121
1122/// P4c (§1.2/§3.1 `core.shell_env_snapshot`, SPLIT CC+CX row, catalog:338):
1123/// capture the user's interactive login-shell environment ONCE, best-effort.
1124/// Runs `$SHELL -lc env` (falling back to `sh -lc env` when `$SHELL` is
1125/// unset) — a LOGIN shell (`-l`) sources the user's rc files, which is
1126/// exactly the sourcing `bash` calls should no longer need to repeat once
1127/// this snapshot is in hand. Never panics: any failure (spawn error,
1128/// non-zero exit, unparseable output) returns an empty map, which
1129/// `ToolContext::shell_env`'s "no-op when `None`/empty" contract already
1130/// treats as harmless.
1131fn capture_shell_env() -> std::collections::HashMap<String, String> {
1132    let shell = std::env::var("SHELL").unwrap_or_else(|_| "sh".to_string());
1133    let out = match std::process::Command::new(&shell)
1134        .arg("-lc")
1135        .arg("env")
1136        .output()
1137    {
1138        Ok(o) if o.status.success() => o.stdout,
1139        Ok(o) => {
1140            tracing::warn!(
1141                "shell_env_snapshot: `{shell} -lc env` exited with status {:?}; snapshot is empty",
1142                o.status.code()
1143            );
1144            return std::collections::HashMap::new();
1145        }
1146        Err(e) => {
1147            tracing::warn!(
1148                "shell_env_snapshot: failed to run `{shell} -lc env` ({e}); snapshot is empty"
1149            );
1150            return std::collections::HashMap::new();
1151        }
1152    };
1153    let text = String::from_utf8_lossy(&out);
1154    let mut map = std::collections::HashMap::new();
1155    for line in text.lines() {
1156        if let Some((k, v)) = line.split_once('=') {
1157            if !k.is_empty() {
1158                map.insert(k.to_string(), v.to_string());
1159            }
1160        }
1161    }
1162    map
1163}
1164
1165/// Build the [`ToolContext`] an [`Agent`] hands to every tool call, folding
1166/// in every P4c per-tool config knob (§1.2) alongside the pre-existing
1167/// `cwd`/`sandbox` — shared by [`Agent::with_parts`]/[`Agent::with_provider_arc`]
1168/// so the two construction paths can never drift apart on which config
1169/// fields reach the context. Also builds (P5-9) the
1170/// [`crate::checkpoint::CheckpointObserver`], if `config.checkpoint_enabled`
1171/// — installed on the returned context's `write_observer` AND returned
1172/// separately (as the concrete type) so `Agent::run_loop` can call
1173/// [`crate::checkpoint::CheckpointObserver::begin_turn`] once per turn.
1174/// `None`/no-op end to end when the module is off — see
1175/// [`crate::checkpoint::observer_for_config`]'s own doc comment for the
1176/// default-off byte-identity guarantee.
1177///
1178/// P5-11 (§2 modules 28/29, D-5 "shared write-path interception seam"):
1179/// `crate::formatters::observer_for_config`/`crate::lsp::manager_for_config`
1180/// are folded into the SAME `write_observer` slot via
1181/// [`crate::tools::WriteObserverChain`], in the design's required order —
1182/// `checkpoint -> formatters -> lsp` (checkpoint's pre-image capture must
1183/// see the file before ANY mutation; lsp's diagnostics must see the file
1184/// AFTER formatting, never before). When 0 or 1 of the three modules is
1185/// active, this degrades to exactly what P5-9 shipped (`None`, or the
1186/// single concrete observer installed directly) — no chain wrapper is
1187/// introduced unless there is actually more than one observer to order,
1188/// keeping every single-module (or all-off) configuration byte-identical
1189/// to before this function grew multi-observer support. The `lsp` manager
1190/// is ALSO returned separately (like `checkpoint_observer`), so
1191/// `Agent`'s `Drop` impl can reach `crate::lsp::LspManager::kill_all_sync`
1192/// regardless of how the chain is shaped.
1193/// BP-2: `pub(crate)` so a parity test can build the SAME `ToolContext` an
1194/// `Agent` would from a resolved preset's `Config` and drive a registry
1195/// tool through it — a tool's behavior under a preset is exactly the
1196/// composition of the two, and a test that hand-assembled a context would
1197/// be proving an unwired function.
1198pub(crate) fn build_tool_context(
1199    config: &Config,
1200) -> (
1201    ToolContext,
1202    Option<std::sync::Arc<crate::checkpoint::CheckpointObserver>>,
1203    Option<std::sync::Arc<crate::lsp::LspManager>>,
1204) {
1205    let shell_env = if config.shell_env_snapshot {
1206        Some(std::sync::Arc::new(capture_shell_env()))
1207    } else {
1208        None
1209    };
1210    let checkpoint_observer = crate::checkpoint::observer_for_config(config);
1211    let format_observer = crate::formatters::observer_for_config(config);
1212    let lsp_manager = crate::lsp::manager_for_config(config);
1213    let lsp_observer = lsp_manager
1214        .clone()
1215        .map(|m| std::sync::Arc::new(crate::lsp::LspDiagnosticsObserver::new(m)));
1216    let mut observers: Vec<std::sync::Arc<dyn crate::tools::WriteObserver>> = Vec::new();
1217    if let Some(cp) = &checkpoint_observer {
1218        observers.push(cp.clone() as std::sync::Arc<dyn crate::tools::WriteObserver>);
1219    }
1220    if let Some(f) = &format_observer {
1221        observers.push(f.clone() as std::sync::Arc<dyn crate::tools::WriteObserver>);
1222    }
1223    if let Some(l) = &lsp_observer {
1224        observers.push(l.clone() as std::sync::Arc<dyn crate::tools::WriteObserver>);
1225    }
1226    let write_observer: Option<std::sync::Arc<dyn crate::tools::WriteObserver>> =
1227        match observers.len() {
1228            0 => None,
1229            1 => observers.into_iter().next(),
1230            _ => Some(std::sync::Arc::new(crate::tools::WriteObserverChain::new(
1231                observers,
1232            ))),
1233        };
1234    let ctx = ToolContext {
1235        cwd: config.cwd.clone(),
1236        // BP-10 (catalog row "Additional working directories"): the
1237        // `--add-dir`/`core.additional_dirs` roots reach the TOOLS now,
1238        // not just the project-context walk — `ToolContext::check_write`,
1239        // the OS backstop's writable set, and the permissions engine's
1240        // path rules all read them. Empty (the default) is byte-identical
1241        // to confining everything to `cwd`.
1242        extra_roots: config.additional_dirs.clone(),
1243        sandbox: config.sandbox,
1244        multimodal_read: config.read_file_multimodal,
1245        // BP-2 (§1.2 `core.tools.read_file.line_numbers`, catalog:26).
1246        read_line_numbers: config.read_file_line_numbers,
1247        require_read_before_edit: config.edit_file_require_read_before_edit,
1248        // BP-2: path → content hash at read time (`ToolContext::read_state`).
1249        read_paths: std::sync::Arc::new(std::sync::Mutex::new(std::collections::HashMap::new())),
1250        notebook_aware: config.edit_file_notebook_aware,
1251        shell_env,
1252        nested_instructions: config.nested_instructions,
1253        injected_instruction_dirs: std::sync::Arc::new(std::sync::Mutex::new(HashSet::new())),
1254        // BP-5: filled in by `Agent::with_parts` (the one construction path
1255        // that assembles a prompt, and therefore the one that knows which
1256        // rules were held back); empty everywhere else.
1257        path_rules: std::sync::Arc::new(Vec::new()),
1258        injected_rule_files: std::sync::Arc::new(std::sync::Mutex::new(HashSet::new())),
1259        // P5-1 (§2 module 12 carry-forward): now sourced from real config
1260        // (`capabilities.permissions.sandbox.network.*`, wired by
1261        // `configfile::materialize_config`) instead of always `None`. `None`
1262        // (the default, unchanged when the config never sets it) is still
1263        // byte-identical to today's behavior.
1264        network_policy: config.network_policy.clone(),
1265        // BP-10 (catalog row "Allow/ask/deny rule language", the DOMAIN
1266        // subject): the config's own rule arrays reach the network surface
1267        // too, so a `domain(...)` rule is evaluated by the SAME engine that
1268        // evaluates `bash(...)`/`write(...)` at the dispatch gate — not by
1269        // a second matcher over a second list. `None` when the permissions
1270        // module is off, which is byte-identical to before.
1271        permission_rules: config.permissions_enabled.then(|| {
1272            std::sync::Arc::new(crate::permissions::RuleSet {
1273                deny: config.tool_deny_patterns.clone(),
1274                ask: config.permissions_ask_patterns.clone(),
1275                allow: config.tool_allow_patterns.clone(),
1276            })
1277        }),
1278        // P4e (S3.1 `core.tools.bash.timeout_secs`, S14): folds the `bash`
1279        // `ToolOverride`'s `timeout_secs`, if set, into the context every
1280        // `BashTool::execute` call receives -- `None` (no override
1281        // configured) is byte-identical to today's behavior.
1282        bash_timeout_secs: config
1283            .tool_overrides
1284            .get("bash")
1285            .and_then(|o| o.timeout_secs),
1286        write_observer,
1287        // P5-10 (§2 module 12): sourced from real config
1288        // (`capabilities.permissions.sandbox.{enabled,escalation,env_policy}`,
1289        // wired by `configfile::materialize_config`). `sandbox_approval_handler`
1290        // starts `None` here (no handler is installed yet at `Agent`
1291        // construction time) and is kept in sync by
1292        // `Agent::set_permissions_approval_handler` — see that method's doc
1293        // comment.
1294        sandbox_os_enabled: config.sandbox_os_enabled,
1295        sandbox_escalation: config.sandbox_escalation,
1296        sandbox_env_policy: config.sandbox_env_policy,
1297        sandbox_approval_handler: None,
1298        // BP-3: both handler seams start `None` (nothing is installed at
1299        // construction time) and are filled by
1300        // `Agent::set_permissions_approval_handler` /
1301        // `Agent::set_user_question_handler`, exactly like
1302        // `sandbox_approval_handler` above. The two shared states are
1303        // always present but inert: plan mode starts off (contributing no
1304        // rules), and the budget starts unpublished.
1305        question_handler: None,
1306        approval_handler: None,
1307        plan_mode: std::sync::Arc::new(crate::tools::PlanModeState::new()),
1308        context_budget: std::sync::Arc::new(crate::tools::ContextBudget::new()),
1309        // BP-8 (catalog:156): the shared plan the agent journals and
1310        // persists. Always present, empty and inert until `update_plan`
1311        // writes one.
1312        plan: std::sync::Arc::new(std::sync::Mutex::new(Vec::new())),
1313    };
1314    (ctx, checkpoint_observer, lsp_manager)
1315}
1316
1317/// P4b (§1.4/§3.1, catalog §4a "Global/user-level instruction file tier"):
1318/// where the user/global instruction tier lives — `$SUPERCODE_HOME`, else
1319/// `$XDG_CONFIG_HOME/supercode`, else `~/.config/supercode`. Deliberately
1320/// duplicates `crates/cli/src/userconfig.rs::config_home`'s exact precedence
1321/// rather than depending on the `cli` crate from `core` (wrong dependency
1322/// direction — `cli` depends on `core`, never the reverse). `pub(crate)`:
1323/// also the DEFAULT shadow-store root `crate::checkpoint::observer_for_config`
1324/// (P5-9) derives from when `Config::checkpoint_dir` is unset — one
1325/// `$SUPERCODE_HOME` resolver, not a second hand-rolled one.
1326pub(crate) fn global_instructions_dir() -> std::path::PathBuf {
1327    if let Ok(h) = std::env::var("SUPERCODE_HOME") {
1328        if !h.is_empty() {
1329            return std::path::PathBuf::from(h);
1330        }
1331    }
1332    if let Ok(xdg) = std::env::var("XDG_CONFIG_HOME") {
1333        if !xdg.is_empty() {
1334            return std::path::PathBuf::from(xdg).join("supercode");
1335        }
1336    }
1337    let home = supercode_interchange::user_home()
1338        .map(|home| home.to_string_lossy().into_owned())
1339        .ok_or(std::env::VarError::NotPresent)
1340        .unwrap_or_else(|_| ".".into());
1341    std::path::PathBuf::from(home)
1342        .join(".config")
1343        .join("supercode")
1344}
1345
1346/// P4b (§1.4, catalog §4a "Instruction imports"): is `rel` (an `@`-import
1347/// target found inside a PROJECT-sourced instruction file) LEXICALLY safe to
1348/// resolve? Mirrors `configfile::is_safe_project_dir`'s posture (LOW-1
1349/// precedent): rejects absolute paths, `~`-relative paths, and any `..`
1350/// component — an untrusted repo's own CLAUDE.md/AGENTS.md must not be able
1351/// to `@import` its way to an arbitrary file on disk (e.g. `@/etc/passwd`,
1352/// `@../../.ssh/id_rsa`). Global-tier files (the user's own machine, same
1353/// trust level as the user's shell) are NOT run through this check.
1354///
1355/// This is a cheap PRE-FILTER only — it operates on the literal token text
1356/// and cannot see through a symlink committed in the repo whose *target*
1357/// escapes the root while the *link itself* has a clean, traversal-free
1358/// relative name (e.g. `@link.md` where `link.md -> /etc/passwd`). See
1359/// [`import_target_is_contained`] for the canonicalizing check that closes
1360/// that gap; the project-scoped resolution path runs both.
1361fn import_path_is_safe(rel: &str) -> bool {
1362    if rel.is_empty() || rel.contains('\0') {
1363        return false;
1364    }
1365    let path = std::path::Path::new(rel);
1366    if path.is_absolute() || rel.starts_with('~') {
1367        return false;
1368    }
1369    !path
1370        .components()
1371        .any(|c| matches!(c, std::path::Component::ParentDir))
1372}
1373
1374/// P4b security fix (Fable-5 review, MEDIUM: symlink bypass of the
1375/// project-scoped `@`-import boundary): does `candidate` — after resolving
1376/// symlinks — stay inside `root` — also after resolving symlinks? This is
1377/// what actually enforces [`import_path_is_safe`]'s doc-comment guarantee
1378/// ("must not be able to `@import` its way to an arbitrary file on disk"):
1379/// the lexical check alone rejects `@/etc/passwd` and `@../../secret`, but a
1380/// repo can commit a symlink (e.g. `link.md -> /etc/passwd`) whose own
1381/// relative name is perfectly clean, defeating a purely lexical check.
1382///
1383/// Both sides are canonicalized before the comparison — not just
1384/// `candidate` — because `root` itself can legitimately be a symlink (a
1385/// tempdir under macOS's `/tmp` -> `/private/tmp`, or any other symlinked
1386/// project checkout); comparing a canonicalized candidate against a
1387/// non-canonicalized root would falsely reject genuinely-in-root files.
1388///
1389/// Fails CLOSED: a `canonicalize()` failure (broken symlink, a target that
1390/// doesn't exist, a permission error) returns `false` — never inlined,
1391/// mirroring [`expand_instruction_imports`]'s existing "unreadable file ⇒
1392/// left as literal text" posture rather than panicking or defaulting open.
1393pub(crate) fn import_target_is_contained(
1394    candidate: &std::path::Path,
1395    root: &std::path::Path,
1396) -> bool {
1397    let (Ok(real_root), Ok(real_candidate)) = (
1398        std::fs::canonicalize(root),
1399        std::fs::canonicalize(candidate),
1400    ) else {
1401        return false;
1402    };
1403    real_candidate.starts_with(&real_root)
1404}
1405
1406/// P4b (§1.4/§3.1 `core.instruction_imports`, catalog:85): inline `@path`
1407/// import tokens found in `text` with the referenced file's own (trimmed)
1408/// content, resolved relative to `dir` (the directory the CONTAINING file
1409/// lives in — so a chain of imports each resolves relative to its own
1410/// location, not the original file's). `depth` bounds recursion (CC's own
1411/// default of 4, cited in the cc-parity preset) so a cyclical or
1412/// deeply-nested import chain can't blow the stack or loop forever.
1413/// `project_scoped` gates [`import_path_is_safe`] AND
1414/// [`import_target_is_contained`] — see their doc comments; `root` is the
1415/// containment boundary those checks canonicalize against (the SAME root
1416/// for every level of a nested import chain, even though `dir` itself walks
1417/// deeper with each level — an import three levels deep must still resolve
1418/// under the original project root, not merely under its own immediate
1419/// parent). Ignored when `!project_scoped` (the global/user tier, trusted,
1420/// unrestricted — see [`append_instruction_file`]'s doc comment).
1421/// Any token that isn't `@`-prefixed, doesn't resolve to a readable file, or
1422/// (project-scoped) fails the safety/containment check is left as literal
1423/// text — an import is best-effort, never a hard error that could make
1424/// instruction loading fail outright.
1425fn expand_instruction_imports(
1426    text: &str,
1427    dir: &std::path::Path,
1428    root: &std::path::Path,
1429    project_scoped: bool,
1430    depth: u8,
1431) -> String {
1432    if depth >= 4 {
1433        return text.to_string();
1434    }
1435    let mut out = String::with_capacity(text.len());
1436    for token in split_preserving_whitespace(text) {
1437        if let Some(rel) = token.strip_prefix('@') {
1438            if !rel.is_empty()
1439                && !rel.contains(char::is_whitespace)
1440                && (!project_scoped || import_path_is_safe(rel))
1441            {
1442                let candidate = dir.join(rel);
1443                if !project_scoped || import_target_is_contained(&candidate, root) {
1444                    if let Ok(imported) = std::fs::read_to_string(&candidate) {
1445                        let imported = imported.trim();
1446                        if !imported.is_empty() {
1447                            let imported_dir = candidate.parent().unwrap_or(dir);
1448                            out.push_str(&expand_instruction_imports(
1449                                imported,
1450                                imported_dir,
1451                                root,
1452                                project_scoped,
1453                                depth + 1,
1454                            ));
1455                            continue;
1456                        }
1457                    }
1458                }
1459            }
1460        }
1461        out.push_str(token);
1462    }
1463    out
1464}
1465
1466/// Split `text` into tokens that, concatenated, reproduce it exactly —
1467/// alternating runs of non-whitespace and whitespace. Used by
1468/// [`expand_instruction_imports`] so `@import` tokens can be located and
1469/// replaced without disturbing surrounding formatting/whitespace.
1470fn split_preserving_whitespace(text: &str) -> Vec<&str> {
1471    let mut out = Vec::new();
1472    let mut start = 0;
1473    let mut in_ws = None;
1474    for (i, c) in text.char_indices() {
1475        let ws = c.is_whitespace();
1476        match in_ws {
1477            None => in_ws = Some(ws),
1478            Some(prev) if prev != ws => {
1479                out.push(&text[start..i]);
1480                start = i;
1481                in_ws = Some(ws);
1482            }
1483            _ => {}
1484        }
1485    }
1486    if start < text.len() {
1487        out.push(&text[start..]);
1488    }
1489    out
1490}
1491
1492/// BP-4 (catalog:87 "Instruction-file hygiene controls", cc§2
1493/// `claudeMdExcludes`): whether `path` is excluded from instruction loading
1494/// by [`Config::project_doc_excludes`]. A pattern matches when it
1495/// [`crate::config::glob_match`]es the file's NAME (`CLAUDE.md`), its full
1496/// path, or its path relative to `root` — the three spellings cc's own
1497/// "glob/absolute-path list" accepts. Empty (the default) excludes nothing.
1498fn instruction_file_excluded(
1499    config: &Config,
1500    path: &std::path::Path,
1501    root: &std::path::Path,
1502) -> bool {
1503    if config.project_doc_excludes.is_empty() {
1504        return false;
1505    }
1506    let full = path.to_string_lossy().to_string();
1507    let name = path
1508        .file_name()
1509        .map(|n| n.to_string_lossy().to_string())
1510        .unwrap_or_default();
1511    let rel = path
1512        .strip_prefix(root)
1513        .ok()
1514        .map(|p| p.to_string_lossy().to_string());
1515    config.project_doc_excludes.iter().any(|pat| {
1516        crate::config::glob_match(pat, &full)
1517            || crate::config::glob_match(pat, &name)
1518            || rel
1519                .as_deref()
1520                .is_some_and(|r| crate::config::glob_match(pat, r))
1521    })
1522}
1523
1524/// BP-4 (catalog:87, cc§2 "HTML comment stripping"): drop block-level
1525/// `<!-- … -->` spans from an instruction file's text so maintainer notes
1526/// cost no tokens, exactly as cc does before injection. Unterminated
1527/// openers drop the remainder (the same reading a markdown renderer takes).
1528/// Off by default ([`Config::project_doc_strip_comments`]) — cx does NOT
1529/// strip, so this is a per-preset hygiene lever, not a universal one.
1530fn strip_html_comments(text: &str) -> String {
1531    let mut out = String::with_capacity(text.len());
1532    let mut rest = text;
1533    while let Some(open) = rest.find("<!--") {
1534        out.push_str(&rest[..open]);
1535        match rest[open..].find("-->") {
1536            Some(close) => rest = &rest[open + close + 3..],
1537            None => return out,
1538        }
1539    }
1540    out.push_str(rest);
1541    out
1542}
1543
1544/// P4b (§1.4): append one instruction file's (trimmed, import-expanded)
1545/// content to `blob` as a labeled section, exactly like the pre-P4b inline
1546/// loop did — a no-op when `path` doesn't exist or is empty (the common
1547/// case). `project_scoped` distinguishes the project tier (imports bounded
1548/// to `root`, canonicalized-and-contained — see
1549/// [`import_target_is_contained`]) from the global tier (imports
1550/// unrestricted, same trust level as the user's own machine — `root` is
1551/// unused in that case). `root` is normally `path`'s own parent (the tier
1552/// root `path` was discovered under, e.g. an ancestor of `cwd` or an
1553/// `additional_dirs` entry) — see [`assemble_project_instructions`]'s call
1554/// sites.
1555///
1556/// BP-4 adds the hygiene controls (catalog:87): the exclude list
1557/// ([`instruction_file_excluded`]), HTML-comment stripping
1558/// ([`strip_html_comments`]) and the [`InstructionBudget`] —
1559/// [`Config::project_doc_max_bytes`] spent INCREMENTALLY as files are
1560/// concatenated root→cwd, which is how cx's own cap works on its root-down
1561/// concat, rather than one chop at the end (that chop would silently eat
1562/// the trailing per-file notices it had just written).
1563fn append_instruction_file(
1564    blob: &mut String,
1565    config: &Config,
1566    path: &std::path::Path,
1567    root: &std::path::Path,
1568    label: &str,
1569    project_scoped: bool,
1570    budget: &mut InstructionBudget,
1571) {
1572    if budget.exhausted() || instruction_file_excluded(config, path, root) {
1573        return;
1574    }
1575    let Ok(text) = std::fs::read_to_string(path) else {
1576        return;
1577    };
1578    let stripped;
1579    let text = if config.project_doc_strip_comments {
1580        stripped = strip_html_comments(&text);
1581        stripped.trim()
1582    } else {
1583        text.trim()
1584    };
1585    if text.is_empty() {
1586        return;
1587    }
1588    let dir = path.parent().unwrap_or(std::path::Path::new("."));
1589    let mut content = if config.instruction_imports {
1590        expand_instruction_imports(text, dir, root, project_scoped, 0)
1591    } else {
1592        text.to_string()
1593    };
1594    if !budget.take(&mut content) {
1595        return;
1596    }
1597    blob.push_str(&format!("\n\n# {label}\n{content}"));
1598}
1599
1600/// Truncate `s` to at most `max` BYTES, backing off to the nearest char
1601/// boundary — shared by the per-file and aggregate instruction caps.
1602fn truncate_at_char_boundary(s: &mut String, max: usize) {
1603    let mut end = max;
1604    while end > 0 && !s.is_char_boundary(end) {
1605        end -= 1;
1606    }
1607    s.truncate(end);
1608}
1609
1610/// BP-4 (catalog:87 "Instruction-file hygiene controls", cx§2
1611/// `project_doc_max_bytes`): the instruction-content byte budget, spent as
1612/// files are concatenated root→cwd.
1613///
1614/// `None` (the default, and cc-parity's explicit `= 0`) is uncapped, so
1615/// [`Self::take`] is a no-op and assembly is byte-identical to a config
1616/// that never heard of the cap. With a cap set, each file is truncated to
1617/// whatever budget REMAINS (per-file notice), and once the budget is gone
1618/// the remaining files are skipped entirely (aggregate notice, emitted once
1619/// by [`Self::aggregate_notice`]) — the total instruction CONTENT can
1620/// therefore never exceed the cap, and the notices survive because nothing
1621/// chops the assembled blob afterwards.
1622struct InstructionBudget {
1623    remaining: Option<usize>,
1624    hit: bool,
1625}
1626
1627impl InstructionBudget {
1628    fn new(config: &Config) -> Self {
1629        InstructionBudget {
1630            remaining: config.project_doc_max_bytes,
1631            hit: false,
1632        }
1633    }
1634
1635    /// True once the cap has consumed the whole budget — later files are
1636    /// skipped rather than partially appended.
1637    fn exhausted(&self) -> bool {
1638        self.remaining == Some(0)
1639    }
1640
1641    /// Charge `content` against the budget, truncating it (and appending a
1642    /// per-file notice) when it doesn't fit. Returns whether anything is
1643    /// left to append.
1644    fn take(&mut self, content: &mut String) -> bool {
1645        let Some(remaining) = self.remaining else {
1646            return true;
1647        };
1648        if content.len() <= remaining {
1649            self.remaining = Some(remaining - content.len());
1650            return true;
1651        }
1652        self.hit = true;
1653        self.remaining = Some(0);
1654        if remaining == 0 {
1655            return false;
1656        }
1657        truncate_at_char_boundary(content, remaining);
1658        content.push_str("\n[supercode: file truncated at core.project_doc_max_bytes]");
1659        true
1660    }
1661
1662    /// The one aggregate notice, appended after assembly when the cap bound
1663    /// anywhere — the statement that the assembled block is not the whole
1664    /// instruction set.
1665    fn aggregate_notice(&self) -> &'static str {
1666        if self.hit {
1667            "\n\n[supercode: instruction content truncated at core.project_doc_max_bytes]"
1668        } else {
1669            ""
1670        }
1671    }
1672}
1673
1674/// BP-4 (catalog:81 "Project instruction files w/ directory walk"; cc§2
1675/// "Directory-walk loading", cx§2 "walk project root (git root) down to
1676/// cwd"): the ancestor chain instruction files are discovered on, ordered
1677/// OUTERMOST FIRST so the nearest directory wins precedence by appearing
1678/// last in the concatenated blob (the root→cwd ordering both inventories
1679/// document).
1680///
1681/// The walk starts at [`Config::cwd`] and climbs until it has included the
1682/// project root [`crate::config::project_root_for`] identifies (`.git` by
1683/// default — cx's `project_root_markers`, §3.1), or until the filesystem
1684/// root, whichever comes first. [`MAX_INSTRUCTION_WALK_DEPTH`] bounds it
1685/// unconditionally, so a marker-less path deep under `/` can never turn
1686/// prompt assembly into an unbounded stat storm.
1687pub(crate) fn instruction_walk_roots(config: &Config) -> Vec<std::path::PathBuf> {
1688    // BP-9's shared answer to "where does the project stop?" — the same
1689    // walk the `env_context` git probe and the CLI's `.supercode.toml`
1690    // discovery use, so one `project_root_markers` value cannot mean three
1691    // different things. `None` (no marker anywhere, or an empty list) means
1692    // no root was found, and the climb below then stops at the filesystem
1693    // root under `MAX_INSTRUCTION_WALK_DEPTH`.
1694    let root = crate::config::project_root_for(&config.cwd, &config.project_root_markers);
1695    let mut chain: Vec<std::path::PathBuf> = Vec::new();
1696    let mut dir = config.cwd.clone();
1697    loop {
1698        let at_root = root.as_deref() == Some(dir.as_path());
1699        chain.push(dir.clone());
1700        if at_root || chain.len() >= MAX_INSTRUCTION_WALK_DEPTH {
1701            break;
1702        }
1703        match dir.parent() {
1704            Some(parent) if parent != dir => dir = parent.to_path_buf(),
1705            _ => break,
1706        }
1707    }
1708    chain.reverse();
1709    chain
1710}
1711
1712/// Hard bound on [`instruction_walk_roots`]'s ancestor climb.
1713const MAX_INSTRUCTION_WALK_DEPTH: usize = 64;
1714
1715/// P4b (§1.4, obligation 4 assembly site): the full instruction-file blob —
1716/// global/user tier (catalog §4a "Global/user-level instruction file tier")
1717/// FIRST, then the project tier — capped by
1718/// [`Config::project_doc_max_bytes`] if set (catalog §4a "hygiene caps
1719/// (`project_doc_max_bytes` analog)").
1720///
1721/// BP-4 (catalog:81): the project tier is no longer `cwd` alone. It is the
1722/// ANCESTOR WALK [`instruction_walk_roots`] returns (cwd's chain up to the
1723/// git root, outermost first) followed by `additional_dirs` — root-first
1724/// ordering throughout, so the nearest directory wins by appearing later,
1725/// which is exactly how both cc§2 ("concatenated root→cwd, closest read
1726/// last") and cx§2 ("nearer-to-cwd wins by appearing later") describe their
1727/// own walks. `cwd` is the last element of the walk chain, so a config
1728/// whose cwd IS the project root assembles byte-identically to the pre-BP-4
1729/// loop.
1730/// BP-5 (catalog D2 "Per-model-family base-prompt selection", cx§2
1731/// "Per-model base instructions": "the system prompt is selected per model
1732/// family from bundled markdown … the active `base_instructions` are
1733/// persisted verbatim into the rollout `session_meta`"): the base system
1734/// prompt for the model this config runs.
1735///
1736/// `[capabilities.model_catalog] base_prompts` maps a model-id glob to that
1737/// family's prompt; the most specific match wins
1738/// ([`crate::model_catalog::base_prompt_for`]). No table and no match both
1739/// give [`Config::system_prompt`] verbatim, so this is a no-op for every
1740/// config that does not set the table.
1741fn base_prompt_for_config(config: &Config) -> String {
1742    crate::model_catalog::base_prompt_for(&config.model_family_prompts, &config.model)
1743        .map(str::to_string)
1744        .unwrap_or_else(|| config.system_prompt.clone())
1745}
1746
1747fn assemble_project_instructions(config: &Config) -> String {
1748    let mut blob = String::new();
1749    let mut budget = InstructionBudget::new(config);
1750    let global_dir = global_instructions_dir();
1751    for name in ["CLAUDE.md", "AGENTS.md"] {
1752        append_instruction_file(
1753            &mut blob,
1754            config,
1755            &global_dir.join(name),
1756            // Global tier is trusted/unrestricted (project_scoped=false
1757            // below) — `root` is never consulted, but pass `global_dir`
1758            // rather than a bogus value for clarity.
1759            &global_dir,
1760            name,
1761            false,
1762            &mut budget,
1763        );
1764    }
1765    // BP-10 (catalog row "Project/workspace trust gate", cc§4/cx§4:
1766    // "Prompt before loading project-local config/code"): the PROJECT tier
1767    // is trust-gated. The global/user tier above is not — it is the user's
1768    // own machine, the same trust level as their shell, exactly as
1769    // `append_instruction_file`'s `project_scoped = false` argument
1770    // already says.
1771    //
1772    // `crate::trust::is_trusted` asks the `Config::trust_handler` door
1773    // once per project and records the answer; with no door installed
1774    // `TrustSurface::Instructions` resolves to LOADED, which is both the
1775    // pre-BP-10 behavior and what a headless run of either upstream
1776    // harness does — see `crate::trust`'s doc comment for why the
1777    // undecided answer differs between text and code.
1778    if !crate::trust::is_trusted(config, crate::trust::TrustSurface::Instructions) {
1779        blob.push_str(
1780            "\n[project instruction files were not loaded: this workspace is not trusted              (capabilities.trust)]\n",
1781        );
1782        blob.push_str(budget.aggregate_notice());
1783        return blob;
1784    }
1785    let walk = instruction_walk_roots(config);
1786    for root in walk.iter().chain(config.additional_dirs.iter()) {
1787        for name in ["CLAUDE.md", "AGENTS.md"] {
1788            append_instruction_file(
1789                &mut blob,
1790                config,
1791                &root.join(name),
1792                // Project tier: `@`-imports from THIS file must stay under
1793                // THIS root (canonicalized) — see
1794                // `import_target_is_contained`.
1795                root,
1796                name,
1797                true,
1798                &mut budget,
1799            );
1800        }
1801        // A repository-native agent package is an additional project
1802        // instruction tier. It is subject to the same `project_context`
1803        // switch, import containment, and aggregate byte cap as root
1804        // AGENTS.md/CLAUDE.md; loading it never executes package code.
1805        for path in crate::agent_package::workspace_package_instruction_files(root) {
1806            append_instruction_file(
1807                &mut blob,
1808                config,
1809                &path,
1810                root,
1811                "Volter Harness agent package instructions",
1812                true,
1813                &mut budget,
1814            );
1815        }
1816    }
1817    blob.push_str(budget.aggregate_notice());
1818    blob
1819}
1820
1821/// P4b (§1.4/§3.1 `core.env_context`, catalog §4a "Environment context block
1822/// injection"): cwd, platform, date, and a best-effort git branch/dirty
1823/// status (silently absent when `cwd` isn't a git repo or `git` isn't on
1824/// `PATH` — never blocks agent construction).
1825///
1826/// BP-4 (catalog:90): plus the APPROVAL/SANDBOX POLICY line the row's own
1827/// semantics name ("cwd/git/platform/date/**policy**") and cx's
1828/// `<environment_context>` supplies — the model is told which approval mode
1829/// and which filesystem confinement it is operating under, which is what
1830/// makes "ask before you do X" instructions legible to it. The block is
1831/// re-derivable at any moment from `config` alone, which is what lets
1832/// [`Agent::refresh_env_context`] re-emit it mid-session on change.
1833/// BP-6 (catalog D2 "Skills (progressive-disclosure packages)", §1.4
1834/// obligation 4): the `# Skills` prompt section — the discovered SKILL.md
1835/// packages' names and descriptions, plus the `[core.prompts]` template
1836/// names, and NOTHING else. A skill's body is deliberately absent: it costs
1837/// its tokens only when something actually invokes it (`docs:skills`
1838/// "body loads only when used"; cx§7; pi§2 "progressive disclosure").
1839///
1840/// A skill whose frontmatter hides it from the model (`enabled: false`,
1841/// `disable-model-invocation: true`) is left OUT of the index while staying
1842/// user-invocable — cc§7 "Invocation control", pi§2.
1843///
1844/// Empty string when there is nothing to list, so an agent with neither
1845/// skills nor templates keeps the prompt it had before this existed.
1846fn skills_prompt_section(config: &Config, skills: &[crate::skills::LoopSkill]) -> String {
1847    let listed: Vec<&crate::skills::LoopSkill> = skills
1848        .iter()
1849        .filter(|skill| skill.model_invocable)
1850        .collect();
1851    let mut templates: Vec<&str> = config.prompts.keys().map(String::as_str).collect();
1852    templates.sort_unstable();
1853    if listed.is_empty() && templates.is_empty() {
1854        return String::new();
1855    }
1856    let mut out = String::from("\n\n# Skills\n");
1857    if !listed.is_empty() {
1858        out.push_str(
1859            "Installed skill packages. Only each skill's name and description are listed \
1860             here; call the `skill` tool with a name below to load that skill's full \
1861             instructions when it applies, then follow them.\n",
1862        );
1863        for skill in listed {
1864            out.push_str(&skill.index_line());
1865            out.push('\n');
1866        }
1867    }
1868    if !templates.is_empty() {
1869        if !out.ends_with("# Skills\n") {
1870            out.push('\n');
1871        }
1872        out.push_str("Prompt templates (invoke via `/name args`):\n");
1873        for name in templates {
1874            out.push_str(&format!("- {name}\n"));
1875        }
1876    }
1877    out
1878}
1879
1880fn env_context_block(config: &Config) -> String {
1881    let mut lines = vec![
1882        format!("cwd: {}", config.cwd.display()),
1883        format!("platform: {}", std::env::consts::OS),
1884        format!(
1885            "date: {}",
1886            supercode_interchange::sidecar::now_rfc3339()
1887                .get(..10)
1888                .unwrap_or("")
1889        ),
1890        format!(
1891            "approval policy: {} · sandbox: {}",
1892            approval_policy_label(config.approval),
1893            sandbox_policy_label(config.sandbox),
1894        ),
1895    ];
1896    // BP-9 (§3.1 `core.project_root_markers`, catalog:232): the git probe
1897    // runs at the PROJECT ROOT the markers define, not at whatever
1898    // subdirectory the process happens to sit in — the marker knob's whole
1899    // job is deciding where "the project" starts. Falls back to `cwd` when
1900    // no ancestor carries a marker (or the list is empty), which is
1901    // byte-identical to the pre-BP-9 behavior.
1902    let root = crate::config::project_root_for(&config.cwd, &config.project_root_markers)
1903        .unwrap_or_else(|| config.cwd.clone());
1904    if let Some(status) = env_context_git_status(&root) {
1905        lines.push(status);
1906    }
1907    format!("\n\n# Environment\n{}", lines.join("\n"))
1908}
1909
1910/// The `[capabilities.permissions] approval` spelling of a policy — the same
1911/// token the config schema accepts (`configfile::parse_approval_str`), so
1912/// the block reports the policy in the vocabulary the user configured it in.
1913fn approval_policy_label(policy: crate::config::ApprovalPolicy) -> &'static str {
1914    match policy {
1915        crate::config::ApprovalPolicy::Never => "never",
1916        crate::config::ApprovalPolicy::OnRequest => "on-request",
1917        crate::config::ApprovalPolicy::Untrusted => "untrusted",
1918        crate::config::ApprovalPolicy::ModelRequested => "model-requested",
1919    }
1920}
1921
1922/// The `[capabilities.permissions] sandbox` spelling of a tier — see
1923/// [`approval_policy_label`].
1924fn sandbox_policy_label(policy: crate::tools::SandboxPolicy) -> &'static str {
1925    match policy {
1926        crate::tools::SandboxPolicy::ReadOnly => "read-only",
1927        crate::tools::SandboxPolicy::WorkspaceWrite => "workspace-write",
1928        crate::tools::SandboxPolicy::DangerFullAccess => "danger-full-access",
1929    }
1930}
1931
1932/// Best-effort `git branch (dirty|clean)` for [`env_context_block`]. `None`
1933/// on anything short of a clean success (not a repo, `git` missing, a
1934/// detached/errored state) — this is informational context, never worth
1935/// failing agent construction over.
1936fn env_context_git_status(cwd: &std::path::Path) -> Option<String> {
1937    let branch_out = std::process::Command::new("git")
1938        .args(["rev-parse", "--abbrev-ref", "HEAD"])
1939        .current_dir(cwd)
1940        .output()
1941        .ok()?;
1942    if !branch_out.status.success() {
1943        return None;
1944    }
1945    let branch = String::from_utf8_lossy(&branch_out.stdout)
1946        .trim()
1947        .to_string();
1948    if branch.is_empty() {
1949        return None;
1950    }
1951    let dirty = std::process::Command::new("git")
1952        .args(["status", "--porcelain"])
1953        .current_dir(cwd)
1954        .output()
1955        .ok()
1956        .map(|o| !o.stdout.is_empty())
1957        .unwrap_or(false);
1958    Some(format!(
1959        "git branch: {branch} ({})",
1960        if dirty { "dirty" } else { "clean" }
1961    ))
1962}
1963
1964impl Agent {
1965    /// Build an agent backed by an OpenAI-compatible endpoint (OpenRouter by
1966    /// default). The API key is taken from [`Config::api_key`], then
1967    /// [`Config::api_key_cmd`] (P4: a credential-helper command, run via the
1968    /// shell — see `run_api_key_cmd`), then the configured environment
1969    /// variable ([`Config::api_key_env`]).
1970    pub fn new(config: Config) -> Result<Self> {
1971        let api_key = match &config.api_key {
1972            Some(k) if !k.is_empty() => k.clone(),
1973            // BP-9: the ARGV helper (`core.api_key_command`) is consulted
1974            // first — it is the form with no shell in the path, so a config
1975            // that sets both gets the one with fewer ways to surprise its
1976            // author. Empty/failed → fall through, same as `api_key_cmd`.
1977            _ => match config
1978                .api_key_command
1979                .as_deref()
1980                .filter(|argv| !argv.is_empty())
1981                .map(run_api_key_command)
1982                .filter(|k| !k.is_empty())
1983                .or_else(|| {
1984                    config
1985                        .api_key_cmd
1986                        .as_deref()
1987                        .filter(|c| !c.is_empty())
1988                        .map(run_api_key_cmd)
1989                }) {
1990                // P4 (§1.8 credential-helper indirection, D6 row): the
1991                // helper ran and produced a non-empty key — use it. A
1992                // failed/empty helper falls through to `api_key_env` rather
1993                // than erroring outright, same "try the next source"
1994                // posture as every other layer in this resolution chain.
1995                Some(k) if !k.is_empty() => k,
1996                _ => std::env::var(&config.api_key_env)
1997                    .ok()
1998                    .filter(|k| !k.is_empty())
1999                    .ok_or_else(|| Error::MissingApiKey(config.api_key_env.clone()))?,
2000            },
2001        };
2002        // P4b (§1.1/§3.1 `core.retry`, pi§3 shape): `Config.retry_*` now
2003        // reaches the pre-existing transport-layer retry mechanism (see
2004        // `provider::HttpOptions::from_retry_config`'s doc comment for the
2005        // exact "byte-identical when unset" contract).
2006        let http_options = provider::HttpOptions::from_retry_config(
2007            config.retry_enabled,
2008            config.retry_max_retries,
2009            config.retry_base_delay_ms,
2010        );
2011        // BP-7 (catalog §4a "Turn/budget caps"): a spend cap armed against
2012        // a model this build cannot price is refused HERE rather than
2013        // accepted and silently never enforced. See
2014        // `Config::max_budget_usd`.
2015        if config.max_budget_usd.is_some_and(|b| b > 0.0)
2016            && crate::pricing::resolve(
2017                &config.model,
2018                config.price_input_per_mtok,
2019                config.price_output_per_mtok,
2020            )
2021            .is_none()
2022        {
2023            return Err(Error::UnpriceableBudget {
2024                model: config.model.clone(),
2025            });
2026        }
2027        // BP-7 (catalog §4a "Auto-retry on transient provider errors"): the
2028        // shared log the transport's retry loop reports into and
2029        // `Self::run_loop` drains after every completion.
2030        let retry_log = std::sync::Arc::new(crate::provider::RetryLog::default());
2031        let provider = OpenAiProvider::new_with_options(
2032            config.base_url.clone(),
2033            api_key,
2034            config.extra_headers.clone(),
2035            http_options,
2036        )
2037        .with_retry_log(retry_log.clone());
2038        // P3 (design §5.2): `ToolRegistry::from_config` replaces the
2039        // unconditional `with_builtins()` call — a no-op when
2040        // `config.module_registry` is off (the default, §5.3 risk 2).
2041        let registry = ToolRegistry::from_config(&config);
2042        let mut agent = Self::with_parts(config, Box::new(provider), registry);
2043        agent.retry_log = retry_log;
2044        Ok(agent)
2045    }
2046
2047    /// Build an agent with an explicit provider and the built-in tools. Handy
2048    /// for tests (inject a mock provider) or custom transports.
2049    pub fn with_provider(config: Config, provider: Box<dyn Provider>) -> Self {
2050        let registry = ToolRegistry::from_config(&config);
2051        Self::with_parts(config, provider, registry)
2052    }
2053
2054    /// Build an agent from all three parts.
2055    pub fn with_parts(
2056        mut config: Config,
2057        provider: Box<dyn Provider>,
2058        mut registry: ToolRegistry,
2059    ) -> Self {
2060        // P5-12 (§2 module 18 `plugins`, D-10): register every trusted,
2061        // loaded plugin's declared tools — the same "unconditional, config-
2062        // gated" wiring `build_tool_context` just below gives
2063        // checkpoint/formatters/lsp. `crate::plugins::register_into` is a
2064        // true no-op (no filesystem read, no subprocess) whenever
2065        // `config.plugins_enabled` is `false` (the default) — byte-identical
2066        // to before this module existed. Runs here (the one tail every
2067        // `Agent` construction path funnels through — `new`/`with_provider`
2068        // both call this) rather than in `ToolRegistry::from_config`, so it
2069        // is NOT entangled with that function's unrelated `module_registry`
2070        // experimental gate.
2071        crate::plugins::register_into(&config, &mut registry);
2072        let (mut ctx, checkpoint_observer, lsp_manager) = build_tool_context(&config);
2073        // Auto-load project context files (CLAUDE.md / AGENTS.md) from the
2074        // working directory (and any extra roots), appending them to the system
2075        // prompt — the analog of how Claude Code / Codex discover them.
2076        // P4b (§1.4): also the global/user tier + instruction imports + the
2077        // `project_doc_max_bytes` hygiene cap — see `assemble_project_instructions`.
2078        // BP-5 (catalog D2 "Per-model-family base-prompt selection", cx§2
2079        // "Per-model base instructions"): the base prompt is chosen for the
2080        // model in force, not fixed before the model is known — the exact
2081        // residue the ledger row named. `base_prompt_for_config` is
2082        // `config.system_prompt` verbatim for every config that sets no
2083        // family table, so this is a no-op by default.
2084        let base_prompt_live = base_prompt_for_config(&config);
2085        // BP-5 (catalog D2 "Output style / personality module"): a custom
2086        // style may REPLACE the base coding instructions rather than append
2087        // to them (cc§7 `keep-coding-instructions`); every other style is
2088        // appended at the end of the assembled prompt, below.
2089        let output_style = crate::output_style::resolve(&config);
2090        let mut system = match output_style.as_ref() {
2091            Some(style) if style.replaces_base => style.text.clone(),
2092            _ => base_prompt_live.clone(),
2093        };
2094        if config.load_project_context {
2095            system.push_str(&assemble_project_instructions(&config));
2096        }
2097        // BP-5 (catalog D2 "Path-scoped rules", cc§2 `.claude/rules`): the
2098        // UNSCOPED rules join the instruction blob here. A rule carrying a
2099        // `paths:` selector deliberately does not — it waits for a tool to
2100        // touch a matching file (`tools::builtins::path_rules_notice`).
2101        let path_rules = crate::path_rules::load(&config);
2102        system.push_str(&crate::path_rules::always_on_text(&path_rules));
2103        // P4b (§1.4/§3.1 `core.env_context`, catalog §4a "Environment
2104        // context block injection"): `false` (the default) is a no-op —
2105        // byte-identical to today's behavior. BP-4 keeps the rendered block
2106        // on the agent (`env_context_live`) so `refresh_env_context` can
2107        // find and REPLACE exactly this text when cwd/policy/branch move,
2108        // rather than leaving a stale block in the prompt forever.
2109        let env_context_live = if config.env_context {
2110            let block = env_context_block(&config);
2111            system.push_str(&block);
2112            Some(block)
2113        } else {
2114            None
2115        };
2116        // P4e (§1.4/§3.1 `core.context_injections`, catalog:91 "Synthetic
2117        // context-injection blocks"): same assembly site, right after
2118        // `env_context`. `false` (the default) is a no-op — byte-identical
2119        // to today's behavior. BP-4 routes it through
2120        // `crate::context_injection`, so the gate now delivers the built-in
2121        // ambient blocks the row is about (and stays extensible at runtime
2122        // through `Self::inject_context_block`) instead of only whatever
2123        // static list an embedder happened to populate.
2124        system.push_str(&crate::context_injection::assemble(&config, &[]));
2125        // P3 (design §5.2, §1.4 obligation 4, D-7): the skills prompt
2126        // section is a MODULE-GATED prompt section, the design's own
2127        // illustration of "a disabled module contributes no prompt
2128        // sections" — only assembled at all under
2129        // `[experimental] module_registry = true` (§5.3 risk 2: flag-off is
2130        // byte-for-byte today's behavior, and today's behavior never emits
2131        // this section, since it doesn't exist pre-P3). Gated further by
2132        // D-7 itself: `core.skills` requires a read pathway (`read_file` or
2133        // `bash`) — absent either, no section is appended, matching the
2134        // hard-dependency shape `configfile::validate_modules` enforces at
2135        // resolve time.
2136        //
2137        // BP-6 (catalog D2 "Skills (progressive-disclosure packages)"): the
2138        // section is now the discovered SKILL.md INDEX — each package's
2139        // frontmatter `name` and `description`, nothing else. A body is
2140        // never assembled here; it is read on invocation only, which is
2141        // what "progressive disclosure" means. The `[core.prompts]`
2142        // template names keep their own sub-list below it.
2143        let skills = crate::skills::load_for_config(&config);
2144        if config.module_registry && config.skills_enabled {
2145            let has_read_pathway = config
2146                .core_tools_enabled
2147                .iter()
2148                .any(|t| t == "read_file" || t == "bash");
2149            if has_read_pathway {
2150                system.push_str(&skills_prompt_section(&config, &skills));
2151            }
2152        }
2153        // BP-5: the style layer lands LAST, where cc puts it ("output styles
2154        // append custom instructions to the END of the system prompt").
2155        // Empty for a neutral style (`default`/`none`) and for a style that
2156        // already replaced the base above.
2157        if let Some(style) = output_style.as_ref().filter(|s| !s.replaces_base) {
2158            system.push_str(&style.section());
2159        }
2160        // BP-5: the SCOPED rules travel with the tool context, which is
2161        // where a "a tool touched a matching file" event can see them.
2162        ctx.path_rules = std::sync::Arc::new(path_rules);
2163        let history = vec![ChatMessage::system(system)];
2164        // P4e (§1.6/§3.1 `core.session.git_metadata`, catalog:331): captured
2165        // once here, alongside `env_context`'s own git probe — `false` (the
2166        // default) is a no-op, byte-identical to today's behavior.
2167        let git_metadata = if config.session_git_metadata {
2168            crate::git_metadata::capture(&config.cwd, now_ms())
2169        } else {
2170            None
2171        };
2172        // BP-7 (catalog §4a "Named agent definitions as data"): discover
2173        // `<cwd>/.claude/agents/*.md` for EVERY harness that turns the
2174        // subagents module on, not only the Claude emulate/resume path —
2175        // that restriction was the second half of the ledger row's
2176        // residue.
2177        merge_project_agent_definitions(&mut config);
2178        // P5-3: captured before `config` moves into the literal below (a
2179        // `usize` field READ, not a move, but it must happen before the
2180        // `config` shorthand field consumes the binding).
2181        let subagent_depth = config.subagent_depth;
2182        // BP-5 (catalog D2 "Shell-output injection in templates/skills"):
2183        // the authorization every `` !`cmd` `` in a skill/command body is
2184        // evaluated under — this config's own permission rules, resolved
2185        // once. Disabled unless `[core.skills] shell_injection` is on.
2186        let shell_injection = crate::skills::ShellInjection::from_config(&config);
2187        // BP-8 (catalog:151): `[capabilities.session_tree] enabled` finally
2188        // has a reader. An armed tree starts empty and grows one node per
2189        // recorded message — the degenerate single-path case, byte-for-byte
2190        // the same conversation, until a rewind or branch actually forks it.
2191        let session_tree = if config.session_tree_enabled {
2192            Some(supercode_interchange::session_tree::SessionTree::new())
2193        } else {
2194            None
2195        };
2196        // BP-7: resolved once here so the request path never re-does the
2197        // lookup, and so `Self::model_price` is `None` exactly when this
2198        // build cannot price the model.
2199        let model_price = crate::pricing::resolve(
2200            &config.model,
2201            config.price_input_per_mtok,
2202            config.price_output_per_mtok,
2203        );
2204        // BP-10: same reason — built before `config` moves into the
2205        // literal. `Config::permissions_approvals_persist` off (the
2206        // default) makes this the pre-BP-10 in-memory cache and touches no
2207        // filesystem.
2208        let permissions_approval_cache = crate::permissions::cache_for_config(&config);
2209        Agent {
2210            config,
2211            provider: std::sync::Arc::from(provider),
2212            registry,
2213            history,
2214            ctx,
2215            total_output_tokens: 0,
2216            activated_tools: HashSet::new(),
2217            recorder: None,
2218            journal: None,
2219            session_tree,
2220            rewind_undo: Vec::new(),
2221            journaled_plan: Vec::new(),
2222            reduction_policy: None,
2223            reduction_log: ReductionLog::default(),
2224            imported_prefix_len: None,
2225            compacting_manually: false,
2226            env_context_live,
2227            base_prompt_live,
2228            shell_injection,
2229            spliced_context_blocks: Vec::new(),
2230            span_summarizer: None,
2231            last_tool_schema_tier_signature: None,
2232            context_limit: None,
2233            requests_issued: false,
2234            last_cache_activity_ms: None,
2235            cache_established: false,
2236            pending_cache_turn: (false, false, None),
2237            session_titler: None,
2238            usage_log: Vec::new(),
2239            turn_index: 0,
2240            turn_records: Vec::new(),
2241            retry_log: std::sync::Arc::new(crate::provider::RetryLog::default()),
2242            model_price,
2243            total_cost_usd: 0.0,
2244            total_steps: 0,
2245            reaped_subagents: std::collections::HashMap::new(),
2246            goal: None,
2247            steer_queue: std::sync::Arc::new(std::sync::Mutex::new(SteerInbox::default())),
2248            follow_up_queue: std::collections::VecDeque::new(),
2249            doom_loop_last_call: None,
2250            doom_loop_streak: 0,
2251            model_change_log: Vec::new(),
2252            git_metadata,
2253            permissions_approval_cache,
2254            permissions_approval_handler: None,
2255            mcp_prompts: std::collections::HashMap::new(),
2256            skills,
2257            subagent_depth,
2258            subagent_concurrency_gauge: std::sync::Arc::new(std::sync::atomic::AtomicUsize::new(0)),
2259            background_subagents: std::collections::HashMap::new(),
2260            pending_child_approvals: std::sync::Arc::new(std::sync::Mutex::new(Vec::new())),
2261            child_approval_handler_factory: None,
2262            subagent_store: None,
2263            claude_runtime_manifest: None,
2264            background_jobs: std::collections::HashMap::new(),
2265            background_concurrency_gauge: std::sync::Arc::new(std::sync::atomic::AtomicUsize::new(
2266                0,
2267            )),
2268            checkpoint_observer,
2269            lsp_manager,
2270        }
2271    }
2272
2273    /// A handle to this agent's model transport, for sharing with subagents.
2274    pub fn provider_arc(&self) -> std::sync::Arc<dyn Provider> {
2275        self.provider.clone()
2276    }
2277
2278    /// Read-only access to this agent's resolved [`Config`] — e.g. so a
2279    /// caller (`crates/cli`'s `attach_mcp`) can consult
2280    /// [`Config::module_registry`]/[`Config::module_activation`] AFTER
2281    /// construction without having to separately thread the config through
2282    /// every call site that builds an `Agent` and later needs it again.
2283    /// Same trust boundary as every other already-public `Agent` accessor
2284    /// (`history`, `provider_arc`) — the caller is the same process that
2285    /// built this `Config` in the first place, not a new exposure surface.
2286    pub fn config(&self) -> &Config {
2287        &self.config
2288    }
2289
2290    /// P5-9 (§2 module 20 `checkpoint`): this agent's checkpoint engine, if
2291    /// `Config::checkpoint_enabled` is `true` and the shadow store opened
2292    /// successfully — `None` otherwise (the default-off case, or a
2293    /// graceful-degrade after an I/O failure). A caller (CLI/TUI/embedder)
2294    /// uses this to `list`/`turn_diff`/`restore` WITHOUT re-deriving the
2295    /// shadow-store root itself. Deliberately named `checkpoint_observer`,
2296    /// not `checkpoint` — [`Self::checkpoint`] already names the unrelated
2297    /// in-memory conversation-position marker (see that method's doc
2298    /// comment).
2299    pub fn checkpoint_observer(&self) -> Option<&crate::checkpoint::CheckpointObserver> {
2300        self.checkpoint_observer.as_deref()
2301    }
2302
2303    /// P5-11 (§2 module 28 `lsp`): this agent's LSP server registry, if
2304    /// `Config::lsp_enabled` is `true` — `None` otherwise (the default-off
2305    /// case). `impl Drop for Agent` already covers production teardown via
2306    /// [`crate::lsp::LspManager::kill_all_sync`] (a real, group-killing OS
2307    /// process kill — see `crate::lsp`'s module doc). This accessor exists
2308    /// for an OPTIONAL caller (CLI/TUI/embedder) that manages its own
2309    /// `Agent` lifecycle and additionally wants to reach
2310    /// [`crate::lsp::LspManager::shutdown_all`] for a graceful LSP
2311    /// `shutdown`/`exit` handshake BEFORE dropping the agent — nothing
2312    /// calls `shutdown_all` automatically today.
2313    pub fn lsp_manager(&self) -> Option<&crate::lsp::LspManager> {
2314        self.lsp_manager.as_deref()
2315    }
2316
2317    /// Spawn a subagent that shares this agent's model transport, runs `task`
2318    /// to completion with its own fresh conversation (seeded with `system`), and
2319    /// returns its final answer. The analog of `Agent` / `spawn_agent`.
2320    pub async fn run_subagent(
2321        &self,
2322        system: impl Into<String>,
2323        task: impl Into<String>,
2324    ) -> Result<String> {
2325        let mut sub_config = Config::builder()
2326            .model(self.config.model.clone())
2327            .system_prompt(system)
2328            .cwd(self.config.cwd.clone())
2329            .sandbox(self.config.sandbox)
2330            .max_iterations(self.config.max_iterations)
2331            .build();
2332        sub_config.base_url = self.config.base_url.clone();
2333        let mut sub = Agent::with_provider_arc(sub_config, self.provider.clone());
2334        sub.send(task).await
2335    }
2336
2337    /// Like [`Self::with_provider`] but sharing an existing transport handle.
2338    pub fn with_provider_arc(mut config: Config, provider: std::sync::Arc<dyn Provider>) -> Self {
2339        let (ctx, checkpoint_observer, lsp_manager) = build_tool_context(&config);
2340        let history = vec![ChatMessage::system(config.system_prompt.clone())];
2341        // P3 (design §5.2): see the `Self::new` doc note — a no-op when
2342        // `config.module_registry` is off (the default).
2343        let mut registry = ToolRegistry::from_config(&config);
2344        // P5-12: see `Self::with_parts`'s identical call — a no-op when
2345        // `config.plugins_enabled` is `false` (the default).
2346        crate::plugins::register_into(&config, &mut registry);
2347        // P4e: see `Self::with_parts`'s identical capture.
2348        let git_metadata = if config.session_git_metadata {
2349            crate::git_metadata::capture(&config.cwd, now_ms())
2350        } else {
2351            None
2352        };
2353        // BP-7 (catalog §4a "Named agent definitions as data"): discover
2354        // `<cwd>/.claude/agents/*.md` for EVERY harness that turns the
2355        // subagents module on, not only the Claude emulate/resume path —
2356        // that restriction was the second half of the ledger row's
2357        // residue.
2358        merge_project_agent_definitions(&mut config);
2359        // P5-3: captured before `config` moves into the literal below (a
2360        // `usize` field READ, not a move, but it must happen before the
2361        // `config` shorthand field consumes the binding).
2362        let subagent_depth = config.subagent_depth;
2363        // BP-6: this constructor assembles no prompt sections at all (it
2364        // takes `config.system_prompt` verbatim), so there is no skills
2365        // INDEX here — but the discovered set still rides along, so an
2366        // explicit invocation (`/name`, `$slug`, the `skill` tool) resolves
2367        // the same packages the registry's own `skill` tool holds.
2368        let skills = crate::skills::load_for_config(&config);
2369        let base_prompt_live = config.system_prompt.clone();
2370        let shell_injection = crate::skills::ShellInjection::from_config(&config);
2371        // BP-8 (catalog:151): `[capabilities.session_tree] enabled` finally
2372        // has a reader. An armed tree starts empty and grows one node per
2373        // recorded message — the degenerate single-path case, byte-for-byte
2374        // the same conversation, until a rewind or branch actually forks it.
2375        let session_tree = if config.session_tree_enabled {
2376            Some(supercode_interchange::session_tree::SessionTree::new())
2377        } else {
2378            None
2379        };
2380        // BP-7: resolved once here so the request path never re-does the
2381        // lookup, and so `Self::model_price` is `None` exactly when this
2382        // build cannot price the model.
2383        let model_price = crate::pricing::resolve(
2384            &config.model,
2385            config.price_input_per_mtok,
2386            config.price_output_per_mtok,
2387        );
2388        // BP-10: see the sibling constructor — built before `config` moves.
2389        let permissions_approval_cache = crate::permissions::cache_for_config(&config);
2390        Agent {
2391            config,
2392            provider,
2393            registry,
2394            history,
2395            ctx,
2396            total_output_tokens: 0,
2397            activated_tools: HashSet::new(),
2398            recorder: None,
2399            journal: None,
2400            session_tree,
2401            rewind_undo: Vec::new(),
2402            journaled_plan: Vec::new(),
2403            reduction_policy: None,
2404            reduction_log: ReductionLog::default(),
2405            imported_prefix_len: None,
2406            compacting_manually: false,
2407            env_context_live: None,
2408            // BP-5: this constructor assembles no prompt sections (see the
2409            // skills note above) — `config.system_prompt` IS the whole
2410            // system message, so that is what a later `set_model` would
2411            // have to replace.
2412            base_prompt_live,
2413            shell_injection,
2414            spliced_context_blocks: Vec::new(),
2415            span_summarizer: None,
2416            last_tool_schema_tier_signature: None,
2417            context_limit: None,
2418            requests_issued: false,
2419            last_cache_activity_ms: None,
2420            cache_established: false,
2421            pending_cache_turn: (false, false, None),
2422            session_titler: None,
2423            usage_log: Vec::new(),
2424            turn_index: 0,
2425            turn_records: Vec::new(),
2426            retry_log: std::sync::Arc::new(crate::provider::RetryLog::default()),
2427            model_price,
2428            total_cost_usd: 0.0,
2429            total_steps: 0,
2430            reaped_subagents: std::collections::HashMap::new(),
2431            goal: None,
2432            steer_queue: std::sync::Arc::new(std::sync::Mutex::new(SteerInbox::default())),
2433            follow_up_queue: std::collections::VecDeque::new(),
2434            doom_loop_last_call: None,
2435            doom_loop_streak: 0,
2436            model_change_log: Vec::new(),
2437            git_metadata,
2438            permissions_approval_cache,
2439            permissions_approval_handler: None,
2440            mcp_prompts: std::collections::HashMap::new(),
2441            skills,
2442            subagent_depth,
2443            subagent_concurrency_gauge: std::sync::Arc::new(std::sync::atomic::AtomicUsize::new(0)),
2444            background_subagents: std::collections::HashMap::new(),
2445            pending_child_approvals: std::sync::Arc::new(std::sync::Mutex::new(Vec::new())),
2446            child_approval_handler_factory: None,
2447            subagent_store: None,
2448            claude_runtime_manifest: None,
2449            background_jobs: std::collections::HashMap::new(),
2450            background_concurrency_gauge: std::sync::Arc::new(std::sync::atomic::AtomicUsize::new(
2451                0,
2452            )),
2453            checkpoint_observer,
2454            lsp_manager,
2455        }
2456    }
2457
2458    /// Run a prompt on a background task, returning a handle that resolves to
2459    /// the final answer (and the agent, so the caller can continue it). The
2460    /// analog of background/async agent runs.
2461    pub fn run_in_background(
2462        mut self,
2463        prompt: impl Into<String>,
2464    ) -> tokio::task::JoinHandle<(Self, Result<String>)>
2465    where
2466        Self: Send + 'static,
2467    {
2468        let prompt = prompt.into();
2469        tokio::spawn(async move {
2470            let result = self.send(prompt).await;
2471            (self, result)
2472        })
2473    }
2474
2475    /// Build an agent and seed it with a previously-recorded [`Session`] so it
2476    /// can continue where Claude Code or Codex left off.
2477    pub fn resume(config: Config, session: Session) -> Result<Self> {
2478        let mut agent = Agent::new(config)?;
2479        agent.load_session(session);
2480        Ok(agent)
2481    }
2482
2483    /// Like [`Self::resume`], but also begins recording (A2/A3): a fresh
2484    /// native-v2 sidecar is created at `sidecar_path` from `session` (header +
2485    /// `session.raw` verbatim — the imported prefix's own fidelity), and every
2486    /// subsequent turn this agent produces is appended to it at full fidelity,
2487    /// independent of whatever `cap_tool_output`/`maybe_compact` (D6) do to
2488    /// `history`.
2489    ///
2490    /// Invariant this establishes ONLY once [`Self::set_reduction_policy`] is
2491    /// also called (the D6/A7 supersession gate, `Self::run_loop`): at any
2492    /// instant, `Session::from_native_str(sidecar).messages` equals
2493    /// `session.messages` (the imported prefix) followed by every message
2494    /// appended since — i.e. `self.history()[1..]` (`history[0]` is this
2495    /// agent's own system prompt, per [`Self::load_session`]; it is never
2496    /// part of `session` and is never written to the sidecar). Recording
2497    /// alone (no policy) leaves the gate off: `cap_tool_output` still runs on
2498    /// oversized tool results, and `history` can diverge from the sidecar for
2499    /// them — honestly, via the notice's "full output in session sidecar"
2500    /// label, never silently.
2501    pub fn resume_recorded(
2502        config: Config,
2503        session: Session,
2504        sidecar_path: &std::path::Path,
2505    ) -> Result<Self> {
2506        let mut agent = Agent::new(config)?;
2507        let recorder = SidecarWriter::create(sidecar_path, &session)?;
2508        agent.load_session(session);
2509        agent.recorder = Some(recorder);
2510        Ok(agent)
2511    }
2512
2513    /// Install (or replace) this agent's sidecar recorder (A3).
2514    pub fn set_recorder(&mut self, w: SidecarWriter) {
2515        self.recorder = Some(w);
2516    }
2517
2518    // ---- BP-8: the append-only journal (catalog:150/152/154/156) --------
2519
2520    /// Install (or replace) this agent's append-only session journal — the
2521    /// durable, flush-per-record log of every message it produces plus
2522    /// every queue/rewind/plan operation performed on it. Installing one
2523    /// alone changes nothing about the conversation; it only makes the
2524    /// session survive a crash mid-turn.
2525    pub fn set_journal(&mut self, journal: crate::session_journal::SessionJournal) {
2526        self.journal = Some(std::sync::Arc::new(std::sync::Mutex::new(journal)));
2527    }
2528
2529    /// Whether an append-only journal is installed.
2530    pub fn has_journal(&self) -> bool {
2531        self.journal.is_some()
2532    }
2533
2534    /// Append one operation to the journal, if installed. Best-effort by
2535    /// design: losing a durability record must never fail the turn it
2536    /// describes, so the failure is logged and the loop continues — the
2537    /// same contract the compaction-marker `record` call keeps.
2538    fn journal_op(&self, op: crate::session_journal::JournalOp) {
2539        let Some(journal) = &self.journal else { return };
2540        let mut guard = journal
2541            .lock()
2542            .unwrap_or_else(std::sync::PoisonError::into_inner);
2543        if let Err(error) = guard.append(op) {
2544            tracing::warn!("failed to append a session-journal record: {error}");
2545        }
2546    }
2547
2548    /// Declare the durable view caught up: `<name>.jsonl` now holds
2549    /// `messages` messages and every journal record before this point is
2550    /// already in it. Everything journaled AFTER the last such record is
2551    /// exactly what a crash would have lost — see
2552    /// [`crate::session_journal::JournalState::unpersisted`].
2553    pub fn journal_checkpoint(&self, messages: usize) {
2554        self.journal_op(crate::session_journal::JournalOp::Checkpoint { messages });
2555    }
2556
2557    /// BP-13: record one per-turn usage entry in the append-only journal.
2558    pub fn journal_usage(&self, record: &crate::usage_log::UsageRecord) {
2559        self.journal_op(crate::session_journal::JournalOp::Usage {
2560            record: record.clone(),
2561        });
2562    }
2563
2564    /// BP-13: record one mid-session model change in the append-only
2565    /// journal — the ONE persisted home for a routing record (BP-8's
2566    /// journal), never a second file.
2567    pub fn journal_model_change(&self, record: &crate::model_change::ModelChangeRecord) {
2568        self.journal_op(crate::session_journal::JournalOp::ModelChange {
2569            record: record.clone(),
2570        });
2571    }
2572
2573    /// BP-8 (catalog:151): this session's conversation tree, when the
2574    /// module is on.
2575    pub fn session_tree(&self) -> Option<&supercode_interchange::session_tree::SessionTree> {
2576        self.session_tree.as_ref()
2577    }
2578
2579    /// Install a tree loaded from the store (a resume), replacing whatever
2580    /// this agent built. A no-op when the module is off — a session whose
2581    /// preset does not enable `session_tree` must not acquire one through
2582    /// the back door of an old sidecar.
2583    pub fn set_session_tree(&mut self, tree: supercode_interchange::session_tree::SessionTree) {
2584        if self.config.session_tree_enabled {
2585            self.session_tree = Some(tree);
2586        }
2587    }
2588
2589    /// BP-8: rebuild the tree from the current linear history — used after
2590    /// a resume that loaded a transcript but had no `.tree.json` to restore
2591    /// (every session recorded before the module was on).
2592    pub fn rebuild_session_tree_from_history(&mut self) {
2593        if !self.config.session_tree_enabled {
2594            return;
2595        }
2596        let linear: Vec<ChatMessage> = self.history.iter().skip(1).cloned().collect();
2597        self.session_tree =
2598            Some(supercode_interchange::session_tree::SessionTree::from_linear(&linear, now_ms()));
2599    }
2600
2601    /// BP-8 (catalog:152 "Rewind/rollback conversation"): move THIS
2602    /// conversation back to an earlier point — the whole row, not the
2603    /// last-exchange special case [`Self::rewind_to`] serves and not
2604    /// `sessions fork --at`, which makes a different session.
2605    ///
2606    /// `keep` is a message count (index into `history`), so `keep = 1`
2607    /// leaves only the system message. Three things happen, in this order:
2608    ///
2609    /// 1. the removed tail is pushed onto an undo stack, so
2610    ///    [`Self::undo_rewind`] can put it back;
2611    /// 2. a [`crate::session_journal::JournalOp::Rewind`] record is
2612    ///    APPENDED — nothing is deleted from disk, so the rewound-away
2613    ///    messages remain recoverable from the log;
2614    /// 3. when the tree module is on, the active branch's leaf moves to the
2615    ///    node at `keep`, and the old leaf is preserved under a fresh
2616    ///    sibling branch — the next message appended forks there rather
2617    ///    than overwriting.
2618    ///
2619    /// Returns what it did. Rewinding to a point at or past the end is a
2620    /// no-op with `removed = 0`, never an error.
2621    pub fn rewind_conversation(&mut self, keep: usize) -> RewindOutcome {
2622        let keep = keep.max(1).min(self.history.len());
2623        let removed: Vec<ChatMessage> = self.history.split_off(keep);
2624        if removed.is_empty() {
2625            return RewindOutcome {
2626                kept: self.history.len(),
2627                removed: 0,
2628                preserved_branch: None,
2629            };
2630        }
2631        let removed_count = removed.len();
2632        self.rewind_undo.push(removed);
2633        // `keep` counts the system message; the journal records only
2634        // `history[1..]`, so its own view is one shorter.
2635        self.journal_op(crate::session_journal::JournalOp::Rewind { to: keep - 1 });
2636        let preserved_branch = self.session_tree.as_mut().and_then(|tree| {
2637            let path = tree.active_path().unwrap_or_default();
2638            // `keep - 1` messages remain after the system message, so the
2639            // new leaf is the node at index `keep - 2`.
2640            match keep.checked_sub(2).and_then(|i| path.get(i).cloned()) {
2641                Some(node) => tree.rewind(&node, now_ms()).ok().flatten(),
2642                None => None,
2643            }
2644        });
2645        RewindOutcome {
2646            kept: self.history.len(),
2647            removed: removed_count,
2648            preserved_branch,
2649        }
2650    }
2651
2652    /// BP-8: invert the most recent [`Self::rewind_conversation`] — the
2653    /// messages come back, and the inversion is itself an appended journal
2654    /// record. `false` when there is nothing to undo.
2655    pub fn undo_rewind(&mut self) -> bool {
2656        let Some(mut tail) = self.rewind_undo.pop() else {
2657            return false;
2658        };
2659        self.history.append(&mut tail);
2660        self.journal_op(crate::session_journal::JournalOp::Unrewind);
2661        if self.config.session_tree_enabled {
2662            self.rebuild_session_tree_from_history();
2663        }
2664        true
2665    }
2666
2667    /// BP-8 (catalog:150): append messages recovered from the journal
2668    /// after a crash — they were already recorded, so this deliberately
2669    /// does NOT re-journal them; it puts the live conversation back where
2670    /// the interrupted process left it.
2671    pub fn append_recovered_messages(&mut self, messages: &[ChatMessage]) {
2672        for msg in messages {
2673            if let Some(tree) = self.session_tree.as_mut() {
2674                tree.append_message(msg.clone(), now_ms());
2675            }
2676            self.history.push(msg.clone());
2677        }
2678    }
2679
2680    /// BP-8: how many rewinds are currently undoable.
2681    pub fn undoable_rewinds(&self) -> usize {
2682        self.rewind_undo.len()
2683    }
2684
2685    /// BP-8: restore the undo stack a previous process left in the journal,
2686    /// so `/rewind undo` works across a restart.
2687    pub fn restore_rewind_undo(&mut self, stack: Vec<Vec<ChatMessage>>) {
2688        self.rewind_undo = stack;
2689    }
2690
2691    /// BP-8 (catalog:154 "Queued-prompt persistence"): re-queue pending
2692    /// inputs recovered from the journal WITHOUT re-recording them — they
2693    /// are already in the log, and journaling them again would double them
2694    /// on the next restart.
2695    pub fn restore_queues(&mut self, steer: &[String], follow_up: &[String]) {
2696        for message in steer {
2697            self.steer_queue
2698                .lock()
2699                .unwrap_or_else(std::sync::PoisonError::into_inner)
2700                .queue_unchecked(message.clone());
2701        }
2702        for message in follow_up {
2703            self.follow_up_queue.push_back(message.clone());
2704        }
2705    }
2706
2707    /// BP-8 (catalog:156 "Todos/plan persisted per session"): the session's
2708    /// current `update_plan` checklist.
2709    pub fn plan(&self) -> Vec<crate::session_journal::PlanEntry> {
2710        self.ctx.plan_snapshot()
2711    }
2712
2713    /// BP-8: restore a plan read back from the store on resume. Marked as
2714    /// already-journaled, so a resume that changes nothing writes nothing.
2715    pub fn set_plan(&mut self, steps: Vec<crate::session_journal::PlanEntry>) {
2716        self.ctx.set_plan(steps.clone());
2717        self.journaled_plan = steps;
2718    }
2719
2720    /// BP-8 (catalog:154): record that `count` pending inputs left `queue`
2721    /// and became conversation. A no-op when nothing was taken, or when
2722    /// queue persistence is off.
2723    fn journal_queue_drain(&self, queue: crate::session_journal::QueueKind, count: usize) {
2724        if count == 0 || !self.config.session_queue_persist {
2725            return;
2726        }
2727        self.journal_op(crate::session_journal::JournalOp::Dequeue { queue, count });
2728    }
2729
2730    /// BP-8: journal the plan if `update_plan` changed it since the last
2731    /// time this ran. Called at every loop boundary — a plan that a crash
2732    /// would otherwise strand in the tool's memory is on disk within one
2733    /// iteration of being written.
2734    fn journal_plan_if_changed(&mut self) {
2735        if !self.config.todos_persist {
2736            return;
2737        }
2738        let current = self.ctx.plan_snapshot();
2739        if current == self.journaled_plan {
2740            return;
2741        }
2742        self.journaled_plan.clone_from(&current);
2743        self.journal_op(crate::session_journal::JournalOp::Plan { steps: current });
2744    }
2745
2746    /// Install (or replace) this agent's reduction policy (A5/A7/A10). Once
2747    /// set, every provider request is built from a *projected* view of
2748    /// `history[1..]` (`reduce::project_messages`) rather than `history`
2749    /// verbatim — `history` itself is never shrunk or mutated by this; only
2750    /// the request view does.
2751    pub fn set_reduction_policy(&mut self, policy: ReductionPolicy) {
2752        self.reduction_policy = Some(policy);
2753    }
2754
2755    /// This agent's reduction policy, if one is installed.
2756    pub fn reduction_policy(&self) -> Option<&ReductionPolicy> {
2757        self.reduction_policy.as_ref()
2758    }
2759
2760    /// Change the global tool-schema tier (TR-8/T5) mid-session. Takes effect
2761    /// starting with the NEXT request this agent builds. Under
2762    /// [`CachePlan::ImportedPrefix`], the first request built after a change
2763    /// is flagged as a cache-bust event and its cache-control annotation is
2764    /// skipped for that one request (see [`provider::tier_change_is_cache_bust`],
2765    /// consulted in `Self::build_request_messages`) — normal annotation
2766    /// resumes on the next request if the tier doesn't change again.
2767    pub fn set_schema_tier(&mut self, tier: crate::tools::SchemaTier) {
2768        self.config.tool_schema_tier = tier;
2769    }
2770
2771    /// Override the schema tier for a single tool (TR-8/T5) mid-session, same
2772    /// cache-bust interaction as [`Self::set_schema_tier`].
2773    pub fn set_tool_schema_tier(
2774        &mut self,
2775        name: impl Into<String>,
2776        tier: crate::tools::SchemaTier,
2777    ) {
2778        self.config
2779            .tool_overrides
2780            .entry(name.into())
2781            .or_default()
2782            .schema_tier = Some(tier);
2783    }
2784
2785    /// A deterministic fingerprint of the current tool-schema tier
2786    /// configuration (global knob + every per-tool override), used to detect
2787    /// a mid-session tier change (TR-8/T5, dev/05). Order-independent over
2788    /// `tool_overrides` (sorted by name before hashing) so insertion order
2789    /// never spuriously changes the signature.
2790    fn schema_tier_signature(&self) -> u64 {
2791        use std::hash::{Hash, Hasher};
2792        let mut hasher = std::collections::hash_map::DefaultHasher::new();
2793        self.config.tool_schema_tier.hash(&mut hasher);
2794        let mut overrides: Vec<(&str, crate::tools::SchemaTier)> = self
2795            .config
2796            .tool_overrides
2797            .iter()
2798            .filter_map(|(name, o)| o.schema_tier.map(|t| (name.as_str(), t)))
2799            .collect();
2800        overrides.sort_by_key(|(name, _)| *name);
2801        for (name, tier) in overrides {
2802            name.hash(&mut hasher);
2803            tier.hash(&mut hasher);
2804        }
2805        hasher.finish()
2806    }
2807
2808    /// Install (or replace) this agent's TR-7 span summarizer — the
2809    /// injectable side-call `Self::build_request_messages` uses to turn an
2810    /// A10 `TurnsCleared` span into an LLM-written summary paragraph when
2811    /// `policy.summarize_cleared_turns` is on. Installing one alone changes
2812    /// nothing: [`ReductionPolicy::summarize_cleared_turns`] (off by
2813    /// default) is the actual gate, so tests/callers that want the
2814    /// deterministic stub can simply never call this.
2815    pub fn set_span_summarizer(
2816        &mut self,
2817        summarizer: impl reduce::summarize::SpanSummarizer + Send + Sync + 'static,
2818    ) {
2819        self.span_summarizer = Some(std::sync::Arc::new(summarizer));
2820    }
2821
2822    /// Install an already-shared summarizer — same seam as
2823    /// [`Self::set_span_summarizer`], for callers (and tests) that need to
2824    /// keep their own handle on it.
2825    pub fn set_span_summarizer_arc(
2826        &mut self,
2827        summarizer: std::sync::Arc<dyn reduce::summarize::SpanSummarizer + Send + Sync>,
2828    ) {
2829        self.span_summarizer = Some(summarizer);
2830    }
2831
2832    /// Prepare TR-7 metadata with this agent's installed summarizer for a
2833    /// projection performed by an outer driver before session history/log
2834    /// are loaded (the CLI foreign-resume preflight). `None` preserves the
2835    /// deterministic fallback when the gate is off, no summarizer exists,
2836    /// the span is below the cost floor, or the side-call fails.
2837    pub fn prepare_cleared_turns_summary(
2838        &self,
2839        msgs: &[ChatMessage],
2840        policy: &ReductionPolicy,
2841        prior: &ReductionLog,
2842    ) -> Option<reduce::PreparedClearSummary> {
2843        let summarizer = self.span_summarizer.as_deref()?;
2844        reduce::prepare_cleared_turns_summary(msgs, policy, prior, summarizer)
2845    }
2846
2847    /// P5-4: install (or replace) this agent's [`crate::EventSink`] AFTER
2848    /// construction — `Config::event_sink` is otherwise only set at
2849    /// `Config`-build time (before `Agent::new`), which is too early for a
2850    /// `tui` embedder that only knows it's activating (and needs to
2851    /// replace whatever print-mode/REPL sink was already installed with
2852    /// one that feeds its own render loop instead of writing straight to
2853    /// stdout) once it already holds a live `Agent`. Mirrors [`Self::
2854    /// set_permissions_approval_handler`]'s "installing one alone changes
2855    /// nothing beyond what already consults `Config::event_sink`" pattern
2856    /// — this is a plain replacement, not a new activation gate.
2857    pub fn set_event_sink(&mut self, sink: crate::EventSink) {
2858        self.config.event_sink = Some(sink);
2859    }
2860
2861    /// P5-1: install (or replace) this agent's permissions-engine approval
2862    /// handler — see [`crate::permissions::PermissionsApprovalHandler`].
2863    /// This is the non-interactive decision seam a CLI/TUI/SDK embedder
2864    /// implements for the `Ask`-tier prompt; the TUI's actual interactive
2865    /// UI is a separate module (P5 row 4), not built here. Installing one
2866    /// alone changes nothing: [`Config::permissions_enabled`] (off by
2867    /// default) is the actual gate — with no handler installed, every
2868    /// `Ask`-tier decision denies (fail-closed, see that trait's doc
2869    /// comment).
2870    ///
2871    /// P5-10 (§2 module 12, `escalation = "ask"`): the SAME handler also
2872    /// backs a sandbox-unenforceable `ask` decision
2873    /// (`crate::sandbox::decide_fs`'s `approval` parameter) — one installed
2874    /// seam serves both `permissions.rules`' `Ask` tier and
2875    /// `permissions.sandbox`'s `escalation = "ask"`, rather than requiring
2876    /// an embedder to install two near-identical handlers. Kept in sync on
2877    /// `self.ctx` (not just `self.permissions_approval_handler`) because
2878    /// `BashTool::execute`/`PersistentShellTool::execute` only ever see
2879    /// `&ToolContext`, never `&Agent` — see `ToolContext::
2880    /// sandbox_approval_handler`'s doc comment.
2881    /// BP-10 (catalog row "Session approval caching"): this agent's
2882    /// approval cache — the door an embedder/TUI uses to inspect or REVOKE
2883    /// remembered grants (`ApprovalCache::clear` forgets every one, in
2884    /// memory and on disk, and the next matching call asks again). Also
2885    /// how a test proves a grant really did survive the process:
2886    /// `store_path()` names the file a second agent reads back.
2887    pub fn permissions_approval_cache(&self) -> &crate::permissions::ApprovalCache {
2888        &self.permissions_approval_cache
2889    }
2890
2891    pub fn set_permissions_approval_handler(
2892        &mut self,
2893        handler: impl crate::permissions::PermissionsApprovalHandler + 'static,
2894    ) {
2895        let handler: std::sync::Arc<dyn crate::permissions::PermissionsApprovalHandler> =
2896            std::sync::Arc::new(handler);
2897        self.permissions_approval_handler = Some(handler.clone());
2898        self.ctx.sandbox_approval_handler =
2899            Some(crate::sandbox::SandboxApprovalHandler(handler.clone()));
2900        // BP-3 (§2 module 8): `exit_plan_mode` presents its plan on this
2901        // same door — one approval seam for the session, not a second one
2902        // the operator would have to answer separately.
2903        self.ctx.approval_handler = Some(crate::tools::ToolApprovalHandler(handler));
2904    }
2905
2906    /// BP-3 (§2 module 6 `tools.question`): install the door `ask_user`
2907    /// asks the human through — the `elicitation/create` handler the design
2908    /// names as the module's protocol side. SDK-owned frontends pass the
2909    /// broker-backed handler (`crate::server::FrontendRequestBridge::
2910    /// elicitation_handler`), which is what makes the question a real
2911    /// frontend request the turn waits on. `None` (the default, nothing
2912    /// installed) leaves the tool deny-default: it reports that nobody can
2913    /// be asked instead of blocking.
2914    ///
2915    /// Installing one alone changes nothing about whether the tool EXISTS —
2916    /// `[capabilities.tools_question]` is that gate, applied by
2917    /// `crate::tools::ToolRegistry::from_config`.
2918    pub fn set_user_question_handler(
2919        &mut self,
2920        handler: std::sync::Arc<dyn crate::mcp::McpElicitationHandler>,
2921    ) {
2922        self.ctx.question_handler = Some(crate::tools::UserQuestionHandler(handler));
2923    }
2924
2925    /// BP-3 (§2 module 8): the shared plan-mode state, so a frontend (the
2926    /// REPL's `/plan`, a TUI toggle) can enter or leave the read-only
2927    /// research phase the same tools and permission gate see.
2928    pub fn plan_mode(&self) -> &std::sync::Arc<crate::tools::PlanModeState> {
2929        &self.ctx.plan_mode
2930    }
2931
2932    /// Install the compatibility approval seam used when the composable
2933    /// permissions engine is disabled. SDK-owned interactive frontends call
2934    /// this alongside [`Self::set_permissions_approval_handler`] so the same
2935    /// authenticated request channel works under either policy engine; the
2936    /// selected engine remains entirely a configuration decision.
2937    pub fn set_legacy_approval_handler(&mut self, handler: crate::config::ApprovalHandler) {
2938        self.config.approval_handler = Some(handler);
2939    }
2940
2941    /// P5-3 (§2 module 9 D5 "subagent transcripts… persisted + linked"):
2942    /// install a [`crate::store::SessionStore`] (+ this agent's own session
2943    /// name in it) so `spawn_subagent` persists each child's transcript
2944    /// (via [`crate::store::SessionStore::save_subagent_transcript`]) and
2945    /// lineage record (via
2946    /// [`crate::store::SessionStore::save_subagent_lineage`]) once the
2947    /// child finishes. Installing one alone changes nothing about whether
2948    /// spawning WORKS — [`Config::subagents_enabled`] is the actual gate;
2949    /// this only controls whether a completed spawn's transcript additionally
2950    /// lands on disk.
2951    pub fn set_subagent_store(
2952        &mut self,
2953        store: std::sync::Arc<crate::store::SessionStore>,
2954        session_name: impl Into<String>,
2955    ) {
2956        self.subagent_store = Some((store, session_name.into()));
2957    }
2958
2959    /// Seed the Claude runtime manifest reconstructed during resume.
2960    ///
2961    /// Installing state enables the matching Claude runtime tool schemas so
2962    /// a disk-reloaded continuation does not lose that vocabulary, but never
2963    /// starts a timer by itself. The supplied execution posture is preserved:
2964    /// an embedding scheduler may deliberately activate before installing it.
2965    pub fn set_claude_runtime_manifest(
2966        &mut self,
2967        manifest: crate::claude_runtime_state::ClaudeRuntimeManifest,
2968    ) {
2969        // A persisted manifest is itself the compatibility capability marker.
2970        // Reopening a Supercode session must not retain its timers while
2971        // silently dropping Claude's Cron*/ScheduleWakeup vocabulary.
2972        self.config.claude_runtime_tools_enabled = true;
2973        self.claude_runtime_manifest = Some(manifest);
2974    }
2975
2976    /// Reinstall project-scoped Claude named-agent definitions when a
2977    /// Supercode continuation carrying a Claude runtime manifest is reopened
2978    /// from disk. The manifest is the durable capability marker; definitions
2979    /// themselves remain authoritative in `<cwd>/.claude/agents/*.md`.
2980    pub fn restore_claude_project_agents(&mut self) -> Result<usize> {
2981        let definitions = crate::claude_compat::load_project_agents(&self.config.cwd)?;
2982        crate::claude_compat::enable_claude_subagent_compatibility(&mut self.config);
2983        for imported in &definitions {
2984            self.config.subagents_definitions.insert(
2985                imported.definition.name.clone(),
2986                imported.definition.clone(),
2987            );
2988        }
2989        Ok(definitions.len())
2990    }
2991
2992    /// Current imported Claude runtime state, including paused mutations made
2993    /// by `Cron*`/`ScheduleWakeup`, for persistence by the embedding loop.
2994    pub fn claude_runtime_manifest(
2995        &self,
2996    ) -> Option<&crate::claude_runtime_state::ClaudeRuntimeManifest> {
2997        self.claude_runtime_manifest.as_ref()
2998    }
2999
3000    /// Mutable access for an embedding scheduler driver to atomically claim
3001    /// due events and persist the resulting manifest. Merely borrowing this
3002    /// state does not start a timer; execution remains the driver's explicit
3003    /// responsibility.
3004    pub fn claude_runtime_manifest_mut(
3005        &mut self,
3006    ) -> Option<&mut crate::claude_runtime_state::ClaudeRuntimeManifest> {
3007        self.claude_runtime_manifest.as_mut()
3008    }
3009
3010    /// P5-4 (tui, closes the P5-3 §2.2 C6 deferred chain): install a
3011    /// factory this agent's `Self::run_spawn_subagent` calls (with the
3012    /// fresh child's own id and this agent's shared
3013    /// [`Self::pending_child_approvals`] queue) to build the
3014    /// `PermissionsApprovalHandler` a `background_prompts = "parent"`
3015    /// child gets, INSTEAD of the default
3016    /// [`crate::subagents::ParentQueueApprovalHandler`]. Installing one
3017    /// alone changes nothing about whether background spawning works —
3018    /// [`Config::subagents_background_prompts`] being
3019    /// [`crate::subagents::BackgroundPromptsPolicy::Parent`] is the actual
3020    /// gate that reaches this factory at all; a `Parent`-policy child
3021    /// spawned before this is installed (or on an agent that never installs
3022    /// it) still gets the immediate-deny default, unchanged.
3023    ///
3024    /// **Security note.** The factory only controls WHICH handler answers
3025    /// an `Ask`-tier request — it can never widen what gets asked in the
3026    /// first place: [`crate::permissions::approval::resolve_ask`] only
3027    /// calls a handler's `ask` when the rule engine has already resolved
3028    /// the call to `Ask` (`Deny` short-circuits before any handler is
3029    /// consulted; `Allow` never needs one), so a parent's "allow" answer
3030    /// here can only grant what the policy already routed to a prompt —
3031    /// never override a `Deny` the engine already decided.
3032    pub fn set_child_approval_handler_factory(
3033        &mut self,
3034        factory: impl Fn(
3035                String,
3036                std::sync::Arc<std::sync::Mutex<Vec<crate::subagents::QueuedApproval>>>,
3037            ) -> std::sync::Arc<dyn crate::permissions::PermissionsApprovalHandler>
3038            + Send
3039            + Sync
3040            + 'static,
3041    ) {
3042        self.child_approval_handler_factory = Some(std::sync::Arc::new(factory));
3043    }
3044
3045    /// P5-3 (§2.2 C6 "parent-surfaced queue"): every approval request a
3046    /// `background_prompts = "parent"` child has raised so far, oldest
3047    /// first — a read-only audit view, not a mutable queue the caller
3048    /// answers. Under P5-3's own default handler (no P5-4 TUI factory
3049    /// installed) every entry here WAS already resolved `Deny` (a
3050    /// background call can't wait for an answer with no handler
3051    /// installed) — but once a `crate::tui::TuiChildApprovalHandler`
3052    /// factory is installed (P5-4,
3053    /// [`Self::set_child_approval_handler_factory`]), the underlying call
3054    /// genuinely blocks and may resolve `Allow`/`AllowForSession`; this
3055    /// method still records the SAME entry for the audit trail either
3056    /// way, so "queued here" no longer implies "was denied" in general —
3057    /// see [`crate::subagents::QueuedApproval`]'s doc comment.
3058    pub fn pending_child_approvals(&self) -> Vec<crate::subagents::QueuedApproval> {
3059        self.pending_child_approvals
3060            .lock()
3061            .map(|q| q.clone())
3062            .unwrap_or_default()
3063    }
3064
3065    /// P4b: install (or replace) this agent's auto-title side-call — see
3066    /// [`crate::session_title::SessionTitler`]. Installing one alone changes
3067    /// nothing: [`Config::auto_title`] (off by default) is the actual gate a
3068    /// caller should consult before calling [`Self::auto_title`].
3069    pub fn set_session_titler(
3070        &mut self,
3071        titler: impl crate::session_title::SessionTitler + Send + Sync + 'static,
3072    ) {
3073        self.session_titler = Some(std::sync::Arc::new(titler));
3074    }
3075
3076    /// P4b: produce a title for this agent's current conversation via the
3077    /// installed [`Self::set_session_titler`] side-call. Returns `None` (never
3078    /// panics, never blocks longer than the titler itself does) if no
3079    /// titler is installed, or the side-call itself declined (see
3080    /// [`crate::session_title::auto_title`]). Does NOT consult
3081    /// [`Config::auto_title`] itself — that gate is the caller's
3082    /// responsibility, matching `Self::span_summarizer`'s precedent of
3083    /// keeping the mechanism and the policy gate separate.
3084    pub fn auto_title(&self) -> Option<String> {
3085        let titler = self.session_titler.as_deref()?;
3086        crate::session_title::auto_title(&self.history, titler)
3087    }
3088
3089    /// P4b (§1.6, catalog §4a "persisted per-turn usage records"): every
3090    /// [`crate::usage_log::UsageRecord`] this agent has accumulated so far.
3091    pub fn usage_records(&self) -> &[crate::usage_log::UsageRecord] {
3092        &self.usage_log
3093    }
3094
3095    /// P4b: persist this agent's accumulated usage log to `store` under
3096    /// `name` — a thin wrapper over [`crate::store::SessionStore::save_usage_log`]
3097    /// so callers don't need to import both types.
3098    pub fn save_usage_log(&self, store: &crate::store::SessionStore, name: &str) -> Result<()> {
3099        store.save_usage_log(name, &self.usage_log)
3100    }
3101
3102    /// BP-7 (catalog §4a "Turn/step bracketing records"): every
3103    /// [`crate::turn_record::TurnRecord`] this agent has accumulated —
3104    /// the context/usage/finish brackets of each model round-trip plus the
3105    /// retry, abort, effort and goal markers between them.
3106    pub fn turn_records(&self) -> &[crate::turn_record::TurnRecord] {
3107        &self.turn_records
3108    }
3109
3110    /// BP-7: persist the marker log to `store` under `name`
3111    /// (`<name>.events.jsonl`), the same thin-wrapper shape
3112    /// [`Self::save_usage_log`] has.
3113    pub fn save_turn_records(&self, store: &crate::store::SessionStore, name: &str) -> Result<()> {
3114        store.save_turn_records(name, &self.turn_records)
3115    }
3116
3117    /// BP-7 (catalog §4a "Per-turn cost/usage accounting"): dollars this
3118    /// agent has spent so far. `0.0` when the model is unpriceable — read
3119    /// [`Self::model_priced`] to tell "free" from "unknown".
3120    pub fn total_cost_usd(&self) -> f64 {
3121        self.total_cost_usd
3122    }
3123
3124    /// BP-7: whether this build can price this agent's model, i.e. whether
3125    /// [`Self::total_cost_usd`] is a real figure rather than a floor.
3126    pub fn model_priced(&self) -> bool {
3127        self.model_price.is_some()
3128    }
3129
3130    /// BP-7 (catalog §4a "Turn/budget caps"): tool calls this agent has
3131    /// executed so far — the counter [`Config::max_steps`] bounds.
3132    pub fn total_steps(&self) -> usize {
3133        self.total_steps
3134    }
3135
3136    /// BP-7 (catalog §4a "Interrupt/abort with state preserved"): record
3137    /// that the in-flight turn was interrupted.
3138    ///
3139    /// Called by whoever owns the cancellation (the CLI's Ctrl-C race), NOT
3140    /// by the loop itself: a cancelled `send` future is dropped mid-await,
3141    /// so the loop never runs another line. The partial work already
3142    /// appended to the transcript stands; this marker is what makes the
3143    /// interruption a persisted FACT — the residue the ledger row named —
3144    /// rather than something a reader has to infer from a dangling tool
3145    /// call on reload. Emits [`AgentEvent::TurnAborted`] as the live
3146    /// counterpart.
3147    pub fn note_abort(&mut self, source: &str) {
3148        let messages = self.history.len();
3149        self.emit(AgentEvent::TurnAborted {
3150            source: source.to_string(),
3151        });
3152        self.push_turn_marker(crate::turn_record::TurnMarker::Aborted {
3153            source: source.to_string(),
3154            messages,
3155        });
3156    }
3157
3158    // ---- BP-7: goals (catalog §4a "Goals — persistent objective across
3159    // turns"; §2 module 7 `todos`, §3.1 `capabilities.todos.goals`) ----
3160
3161    /// Set (or revise) this session's standing objective.
3162    ///
3163    /// Returns `false`, changing nothing, when `capabilities.todos.goals`
3164    /// is off — the module gate, not a silent success. A goal restates
3165    /// itself at the tail of every request until [`Self::clear_goal`], and
3166    /// each change appends a `goal` marker to the turn-record log.
3167    pub fn set_goal(&mut self, objective: impl Into<String>) -> bool {
3168        if !self.config.goals_enabled {
3169            return false;
3170        }
3171        let objective = objective.into();
3172        let now = now_ms();
3173        match &mut self.goal {
3174            Some(goal) => goal.revise(objective.clone(), now),
3175            slot @ None => *slot = Some(crate::goals::GoalRecord::new(objective.clone(), now)),
3176        }
3177        self.push_turn_marker(crate::turn_record::TurnMarker::Goal { objective });
3178        true
3179    }
3180
3181    /// This session's standing objective, if one is set.
3182    pub fn goal(&self) -> Option<&crate::goals::GoalRecord> {
3183        self.goal.as_ref()
3184    }
3185
3186    /// Drop the standing objective. `true` when there was one to drop.
3187    pub fn clear_goal(&mut self) -> bool {
3188        if self.goal.take().is_none() {
3189            return false;
3190        }
3191        self.push_turn_marker(crate::turn_record::TurnMarker::Goal {
3192            objective: String::new(),
3193        });
3194        true
3195    }
3196
3197    /// BP-7: persist (or, when cleared, remove) the standing objective
3198    /// beside the session — same thin-wrapper shape as
3199    /// [`Self::save_usage_log`].
3200    pub fn save_goal(&self, store: &crate::store::SessionStore, name: &str) -> Result<()> {
3201        match &self.goal {
3202            Some(goal) => store.save_goal(name, goal),
3203            None => store.clear_goal(name),
3204        }
3205    }
3206
3207    /// BP-7: adopt a goal loaded from the store (a resumed session picks up
3208    /// exactly where it left off). Bypasses the module gate on purpose: a
3209    /// goal already persisted is data to restore, not a new capability
3210    /// being turned on, and dropping it silently would lose session state.
3211    pub fn restore_goal(&mut self, goal: Option<crate::goals::GoalRecord>) {
3212        self.goal = goal;
3213    }
3214
3215    // ---- BP-7: extended-thinking control (catalog §4a "Extended thinking
3216    // control": "Reasoning on/off/levels mid-session") ----
3217
3218    /// The reasoning-effort level in force for the NEXT request, or `None`
3219    /// when extended thinking is off.
3220    pub fn effort(&self) -> Option<&str> {
3221        self.config.effort.as_deref()
3222    }
3223
3224    /// Change the reasoning-effort level mid-session.
3225    ///
3226    /// `Some(level)` sets the level; `None` turns extended thinking OFF —
3227    /// the on/off toggle the ledger row named as distinct from the level.
3228    /// `run_loop` reads `self.config.effort` fresh when it builds each
3229    /// `ChatRequest`, so this takes effect on the very next request with no
3230    /// other copy to update (the same contract [`Self::set_model`] has).
3231    /// The change is appended to the turn-record log as an `effort` marker,
3232    /// the extended-thinking analog of the `model_change` log.
3233    ///
3234    /// Returns the PREVIOUS setting.
3235    pub fn set_effort(&mut self, effort: Option<String>) -> Option<String> {
3236        let previous = self.config.effort.clone();
3237        if previous == effort {
3238            return previous;
3239        }
3240        self.config.effort = effort.clone();
3241        self.push_turn_marker(crate::turn_record::TurnMarker::Effort {
3242            from: previous.clone(),
3243            to: effort,
3244        });
3245        previous
3246    }
3247
3248    // ---- BP-7: review mode (catalog §4a "Review mode — dedicated
3249    // code-review flow"; §3.1 `[core.prompts]`) ----
3250
3251    /// The purpose-built review turn's prompt: the `code-review` template
3252    /// from [`Config::prompts`] with `{args}` replaced by `args`.
3253    ///
3254    /// `None` when the resolved config carries no `code-review` template —
3255    /// the preset decides whether this harness has a review mode, and the
3256    /// template IS the report format (both parity presets pin one).
3257    pub fn review_prompt(&self, args: &str) -> Option<String> {
3258        self.config
3259            .prompts
3260            .get(REVIEW_PROMPT_NAME)
3261            .map(|template| template.replace("{args}", args.trim()))
3262    }
3263
3264    /// Run the review turn: an ordinary [`Self::send`] of
3265    /// [`Self::review_prompt`], so the review's request, tools, transcript
3266    /// and records are the session's own — a purpose-built TURN, not a
3267    /// second agent.
3268    pub async fn review(&mut self, args: &str) -> Result<String> {
3269        let prompt = self.review_prompt(args).ok_or_else(|| {
3270            Error::Other(format!(
3271                "no `{REVIEW_PROMPT_NAME}` prompt template is configured for this harness"
3272            ))
3273        })?;
3274        self.send(prompt).await
3275    }
3276
3277    // ---- BP-7: side/ephemeral Q&A (catalog §4a "Side/ephemeral Q&A":
3278    // "Tool-less question over full context, never enters history") ----
3279
3280    /// Answer `question` over this session's FULL current context without
3281    /// recording anything.
3282    ///
3283    /// Three properties, all load-bearing and all asserted by this build's
3284    /// tests: the request carries the whole conversation as the next turn
3285    /// would see it; it advertises NO tools, so the model can only answer;
3286    /// and neither `history`, the sidecar recorder, the usage log nor the
3287    /// turn-record log is touched — `&self`, not `&mut self`, is the type
3288    /// system saying so. cc's `/btw` and cx's `/side`.
3289    pub async fn side_question(&self, question: &str) -> Result<String> {
3290        let mut messages = self.history.clone();
3291        if let Some(goal) = &self.goal {
3292            messages.push(ChatMessage::system(goal.reminder()));
3293        }
3294        messages.push(ChatMessage::user(format!(
3295            "{SIDE_QUESTION_PREAMBLE}
3296
3297{question}"
3298        )));
3299        let mut req = ChatRequest {
3300            model: self.config.model.clone(),
3301            messages,
3302            tools: Vec::new(),
3303            temperature: self.config.temperature,
3304            max_tokens: self.config.max_tokens,
3305            effort: self.config.effort.clone(),
3306            response_format: None,
3307            service_tier: None,
3308            thinking_budget: None,
3309            extra_body: self.config.extra_body.clone(),
3310        };
3311        // BP-13: a side question is still a request to THIS model, so it
3312        // carries the same routing decisions the loop's own requests do.
3313        self.apply_routing(&mut req);
3314        let (assistant, _usage) = self.provider.complete(&req, &|_: &str| {}).await?;
3315        Ok(assistant.content.unwrap_or_default())
3316    }
3317
3318    /// BP-7: append one marker against the NEXT round-trip's index — the
3319    /// right frame for a marker written between turns (a goal change, an
3320    /// effort change, an abort).
3321    fn push_turn_marker(&mut self, marker: crate::turn_record::TurnMarker) {
3322        self.push_turn_marker_at(self.turn_index, marker);
3323    }
3324
3325    /// BP-7: append one marker against an explicit round-trip index — used
3326    /// inside `Self::run_loop`, where markers are written on both sides of
3327    /// the `turn_index` advance and must all carry the round-trip they
3328    /// describe.
3329    fn push_turn_marker_at(&mut self, turn: usize, marker: crate::turn_record::TurnMarker) {
3330        self.turn_records.push(crate::turn_record::TurnRecord::new(
3331            turn,
3332            &self.config.model,
3333            now_ms(),
3334            marker,
3335        ));
3336    }
3337
3338    /// P4b (§1.7, pi§3 semantics): queue a mid-turn steering message —
3339    /// delivered "after current tool calls" (pi's phrasing): at the top of
3340    /// `Self::run_loop`'s NEXT iteration, before the next model request is
3341    /// built, regardless of whether this turn is still mid-flight with
3342    /// pending tool calls. Drained per [`Config::steering_mode`].
3343    pub fn queue_steer(&self, message: impl Into<String>) {
3344        let message = message.into();
3345        // BP-8 (catalog:154 "Queued-prompt persistence"): the input is
3346        // recorded BEFORE it is queued, so the window in which a crash
3347        // could lose it is zero. A no-op when `core.session.queue_persist`
3348        // is off (cx-parity: stock Codex has no queue-operation records).
3349        if self.config.session_queue_persist {
3350            self.journal_op(crate::session_journal::JournalOp::Enqueue {
3351                queue: crate::session_journal::QueueKind::Steer,
3352                text: message.clone(),
3353            });
3354        }
3355        self.steer_queue
3356            .lock()
3357            .unwrap_or_else(std::sync::PoisonError::into_inner)
3358            .queue_unchecked(message);
3359    }
3360
3361    /// Crate-internal shared steering handle used by the canonical SDK
3362    /// runtime. It remains writable while an active turn holds `&mut Agent`,
3363    /// allowing local and remote frontends to steer without owning the loop.
3364    pub(crate) fn steer_queue_handle(&self) -> std::sync::Arc<std::sync::Mutex<SteerInbox>> {
3365        self.steer_queue.clone()
3366    }
3367
3368    /// P4b: queue a follow-up message — delivered "at idle" (pi's phrasing):
3369    /// only once `Self::run_loop` would otherwise return a final answer
3370    /// (no more tool calls pending). Drained per [`Config::follow_up_mode`].
3371    pub fn queue_follow_up(&mut self, message: impl Into<String>) {
3372        let message = message.into();
3373        // BP-8 (catalog:154): same record-then-queue order as
3374        // [`Self::queue_steer`].
3375        if self.config.session_queue_persist {
3376            self.journal_op(crate::session_journal::JournalOp::Enqueue {
3377                queue: crate::session_journal::QueueKind::FollowUp,
3378                text: message.clone(),
3379            });
3380        }
3381        self.follow_up_queue.push_back(message);
3382    }
3383
3384    /// P4b: how many steering messages are currently queued (mid-turn +
3385    /// follow-up combined) — mostly for tests/diagnostics.
3386    pub fn queued_steer_count(&self) -> usize {
3387        self.steer_queue
3388            .lock()
3389            .unwrap_or_else(std::sync::PoisonError::into_inner)
3390            .len()
3391            + self.follow_up_queue.len()
3392    }
3393
3394    /// The accumulating reduction log (A5) — every reduction applied to any
3395    /// projected request view so far. Combined with a full-fidelity sidecar
3396    /// Session, this is enough to `reduce::invert` any projected view back to
3397    /// the exact original.
3398    pub fn reduction_log(&self) -> &ReductionLog {
3399        &self.reduction_log
3400    }
3401
3402    /// PARITY-18 D4 — arm the per-send context guard: `Self::run_loop`
3403    /// will refuse (via [`Error::ContextLimitExceeded`]) to build and issue
3404    /// ANY request — the first or any later turn — whose
3405    /// [`supercode_runtime::context_guard`] verdict is "does not fit" against
3406    /// `limit`. Call this once the target model's context-window size is
3407    /// known (`resume --reduced`'s preflight already computes it). Leaving
3408    /// this unset (the default) is a no-op: no guard runs, exactly today's
3409    /// pre-PARITY-18 behavior.
3410    pub fn set_context_limit(&mut self, limit: u64) {
3411        self.context_limit = Some(limit);
3412    }
3413
3414    /// This agent's armed context limit, if [`Self::set_context_limit`] has
3415    /// been called.
3416    pub fn context_limit(&self) -> Option<u64> {
3417        self.context_limit
3418    }
3419
3420    /// The model identifier this agent sends on its next request
3421    /// ([`Config::model`], as of construction/resume or the last
3422    /// [`Self::set_model`] call).
3423    pub fn model(&self) -> &str {
3424        &self.config.model
3425    }
3426
3427    /// UX-30 dev/02 — switch the model this agent sends, starting with the
3428    /// NEXT request it builds (and every one after, until changed again).
3429    /// `Self::run_loop` reads `self.config.model` fresh on every request
3430    /// (see its `ChatRequest` construction), so this alone is enough —
3431    /// there is no cached/baked-in copy anywhere else to also update.
3432    /// Takes effect immediately; safe to call only between turns (the
3433    /// REPL's `/model` picker runs at the prompt, never mid-turn). Touches
3434    /// nothing else: history, the sidecar, and reduction state are exactly
3435    /// as untouched as [`Self::set_schema_tier`] leaves them for a
3436    /// mid-session tier change.
3437    ///
3438    /// P4c-review note: this is the LOW-LEVEL primitive — it swaps
3439    /// [`Config::model`] and nothing else. It does NOT run dep 8's
3440    /// reasoning-artifact filter
3441    /// ([`reduce::rehydrate::filter_reasoning_artifacts`]) and does NOT
3442    /// create a [`crate::model_change::ModelChangeRecord`], so calling it
3443    /// directly for a mid-session handoff between two DIFFERENT models
3444    /// leaves model-A's reasoning artifacts in `history` for model-B to
3445    /// inherit. [`Self::switch_model`] is the safe superset — gated by
3446    /// [`Config::model_switch_allow_switch`], it filters and records the
3447    /// switch before delegating to this method — and is what callers
3448    /// performing a governed mid-session model switch should use instead.
3449    pub fn set_model(&mut self, model: impl Into<String>) {
3450        self.config.model = model.into();
3451        // BP-5 (catalog D2 "Per-model-family base-prompt selection"): the
3452        // family's base prompt follows the model. Codex re-selects
3453        // `base_instructions` when the model changes; leaving model-A's
3454        // base prompt in front of model-B is exactly the mismatch the row
3455        // exists to prevent. Same locate-and-replace mechanism
3456        // `refresh_env_context` uses, and a no-op whenever the selection
3457        // did not actually change (always, for a config with no family
3458        // table).
3459        self.refresh_base_prompt();
3460        // BP-7: the price follows the model, or the per-turn cost figure
3461        // would keep billing the OLD model's rates after a switch.
3462        self.model_price = crate::pricing::resolve(
3463            &self.config.model,
3464            self.config.price_input_per_mtok,
3465            self.config.price_output_per_mtok,
3466        );
3467    }
3468
3469    /// P4c (§1.10/§3.1 `core.model_switch.allow_switch`, D9 row, dep 8,
3470    /// design's "core NEW-significant" item): the mid-session model
3471    /// switch — a superset of [`Self::set_model`] gated by
3472    /// [`Config::model_switch_allow_switch`].
3473    ///
3474    /// **`allow_switch = false` (the default): EXACTLY [`Self::set_model`]**
3475    /// — same single field write, nothing else touched, no
3476    /// [`crate::model_change::ModelChangeRecord`] created. Byte-identical to
3477    /// calling `set_model` directly.
3478    ///
3479    /// **`allow_switch = true`:** additionally, before the swap takes
3480    /// effect, runs [`reduce::rehydrate::filter_reasoning_artifacts`] over
3481    /// [`Self::history`] — model-A's reasoning/thinking artifacts (any
3482    /// [`supercode_interchange::ChatMessage::metadata`] key in
3483    /// [`reduce::rehydrate::REASONING_METADATA_KEYS`], any `content_parts`
3484    /// block whose `"type"` is in
3485    /// [`reduce::rehydrate::REASONING_CONTENT_PART_TYPES`]) are stripped
3486    /// BEFORE model-B ever builds a request from this history — then
3487    /// appends a typed, translatable [`crate::model_change::ModelChangeRecord`]
3488    /// to [`Self::model_change_records`] (persist it via
3489    /// [`Self::save_model_change_log`]). A switch TO the current model
3490    /// (`model == Self::model()`) is treated as a no-op — still exactly
3491    /// `set_model`'s mechanics, no record for a switch that didn't actually
3492    /// change anything (and nothing to filter FOR, since there was no
3493    /// handoff).
3494    pub fn switch_model(&mut self, model: impl Into<String>) {
3495        let to = model.into();
3496        if !self.config.model_switch_allow_switch || self.config.model == to {
3497            self.set_model(to);
3498            return;
3499        }
3500        let from = self.config.model.clone();
3501        self.record_model_change(&from, &to, None);
3502    }
3503
3504    /// BP-13 — the ONE place a mid-session model change is performed and
3505    /// recorded, shared by [`Self::switch_model`] (a user asked) and the
3506    /// run loop's fallback pass (a provider failed).
3507    ///
3508    /// It does four things, in this order, and nothing else: strips model-A
3509    /// reasoning artifacts out of the live history (dep 8 — model B must
3510    /// never inherit them), moves [`Config::model`], appends the typed
3511    /// [`crate::model_change::ModelChangeRecord`], and writes that same
3512    /// record into the append-only session journal (BP-8) — which is where
3513    /// every persisted routing record lives; there is no second file. The
3514    /// change is also EMITTED, so a surface that renders events shows the
3515    /// switch instead of silently answering as a different model.
3516    pub fn record_model_change(&mut self, from: &str, to: &str, reason: Option<&str>) {
3517        if from == to {
3518            return;
3519        }
3520        let touched = reduce::rehydrate::filter_reasoning_artifacts(&mut self.history);
3521        self.set_model(to.to_string());
3522        let record = crate::model_change::ModelChangeRecord::new(
3523            self.turn_index,
3524            from,
3525            to,
3526            true,
3527            touched,
3528            now_ms(),
3529        )
3530        .with_reason(reason.map(str::to_string));
3531        self.journal_model_change(&record);
3532        self.model_change_log.push(record);
3533        self.emit(AgentEvent::ModelChanged {
3534            from: from.to_string(),
3535            to: to.to_string(),
3536            reason: reason.map(str::to_string),
3537        });
3538        // A switch also re-injects the switch NOTICE when the config asks
3539        // for one (Codex's own mid-session behavior: the conversation is
3540        // told the model changed, so the new model reads the handoff rather
3541        // than inferring it from a style break).
3542        if self.config.model_switch_notice {
3543            let notice = ChatMessage::user(format!(
3544                "[model changed: {from} -> {to}{}]",
3545                match reason {
3546                    Some(r) => format!(" ({r})"),
3547                    None => String::new(),
3548                }
3549            ));
3550            let _ = self.record(&notice);
3551            self.history.push(notice);
3552        }
3553    }
3554
3555    /// BP-13 (catalog D9 "Fast mode / service tiers"): set or clear the
3556    /// session-level service-tier override. `Some(tier)` WINS over the
3557    /// `[capabilities.model_catalog] service_tier` rule for every
3558    /// subsequent request (it is the live toggle the user just pulled);
3559    /// `None` puts the configured rule back in charge. Takes effect on the
3560    /// next request the loop builds, like [`Self::set_model`].
3561    pub fn set_service_tier(&mut self, tier: Option<String>) {
3562        self.config.service_tier = tier;
3563    }
3564
3565    /// BP-13 — apply the routing table to a request that already names its
3566    /// model: effort LEVEL (per-model override of `[core] effort`, clamped
3567    /// by whichever effort cap applies), thinking-token BUDGET, and service
3568    /// TIER (the live `/fast` override winning over the configured rule).
3569    /// Called for every request the loop builds AND again for every
3570    /// fallback hop, so a hop to a different model gets that model's
3571    /// routing rather than the previous model's.
3572    fn apply_routing(&self, req: &mut ChatRequest) {
3573        let routing = &self.config.model_routing;
3574        let rules = routing.rules_for(&req.model);
3575        // Plan mode's own effort tier, when the mode is live, is the
3576        // session level for this request — Codex's `/plan` is effort
3577        // steering (cx§6), so planning need not think at the executing
3578        // level. It is still clamped by whatever effort cap applies,
3579        // because `effective_effort` does the clamping, not this line.
3580        let session_effort = match (
3581            self.ctx.plan_mode.is_active(),
3582            self.config.plan_mode_effort.as_deref(),
3583        ) {
3584            (true, Some(effort)) => Some(effort),
3585            _ => self.config.effort.as_deref(),
3586        };
3587        req.effort = routing.effective_effort(&req.model, session_effort);
3588        req.thinking_budget = rules.thinking_budget;
3589        req.service_tier = self.config.service_tier.clone().or(rules.service_tier);
3590    }
3591
3592    /// BP-13 — send `req`, walking [`Config::model_fallback`] when the
3593    /// failure is one another model could plausibly answer.
3594    ///
3595    /// Returns the final outcome plus the hops actually taken, so the
3596    /// caller (which owns `&mut self`) can record each one. Each hop
3597    /// re-applies routing for the new model and strips model-A reasoning
3598    /// artifacts from the request's own message copy before model B sees
3599    /// them — the same dep-8 guarantee [`Self::record_model_change`] gives
3600    /// the live history.
3601    async fn complete_with_fallback(
3602        &self,
3603        req: &mut ChatRequest,
3604        on_delta: &(dyn for<'a> Fn(&'a str) + Send + Sync),
3605    ) -> (Result<(ChatMessage, provider::Usage)>, Vec<FallbackHop>) {
3606        let mut hops = Vec::new();
3607        let mut result = self.provider.complete(req, on_delta).await;
3608        for next in &self.config.model_fallback {
3609            let Err(error) = &result else {
3610                break;
3611            };
3612            if !is_failover_worthy(error) {
3613                break;
3614            }
3615            if next.is_empty() || next == &req.model {
3616                continue;
3617            }
3618            let reason = error.to_string();
3619            let from = std::mem::replace(&mut req.model, next.clone());
3620            reduce::rehydrate::filter_reasoning_artifacts(&mut req.messages);
3621            self.apply_routing(req);
3622            hops.push(FallbackHop {
3623                from,
3624                to: next.clone(),
3625                reason,
3626            });
3627            result = self.provider.complete(req, on_delta).await;
3628        }
3629        (result, hops)
3630    }
3631
3632    /// P4c: every [`crate::model_change::ModelChangeRecord`] this agent has
3633    /// accumulated so far (via [`Self::switch_model`] with `allow_switch`
3634    /// on). Empty when the knob is off or no switch has happened yet.
3635    pub fn model_change_records(&self) -> &[crate::model_change::ModelChangeRecord] {
3636        &self.model_change_log
3637    }
3638
3639    /// P4c: persist this agent's accumulated model-change log to `store`
3640    /// under `name` — the [`crate::model_change::ModelChangeRecord`] analog
3641    /// of [`Self::save_usage_log`].
3642    pub fn save_model_change_log(
3643        &self,
3644        store: &crate::store::SessionStore,
3645        name: &str,
3646    ) -> Result<()> {
3647        store.save_model_change_log(name, &self.model_change_log)
3648    }
3649
3650    /// P4e (§1.6/§3.1 `core.session.git_metadata`, catalog:331): this
3651    /// agent's captured git provenance, if [`Config::session_git_metadata`]
3652    /// was on at construction and the best-effort probe found a repo.
3653    pub fn git_metadata(&self) -> Option<&crate::git_metadata::GitMetadataRecord> {
3654        self.git_metadata.as_ref()
3655    }
3656
3657    /// P4e: persist this agent's captured git metadata to `store` under
3658    /// `name` — a thin wrapper over
3659    /// [`crate::store::SessionStore::save_git_metadata`], the
3660    /// [`crate::git_metadata::GitMetadataRecord`] analog of
3661    /// [`Self::save_usage_log`]. A no-op (`Ok(())`, nothing written) when
3662    /// [`Self::git_metadata`] is `None`.
3663    pub fn save_git_metadata(&self, store: &crate::store::SessionStore, name: &str) -> Result<()> {
3664        match &self.git_metadata {
3665            Some(record) => store.save_git_metadata(name, record),
3666            None => Ok(()),
3667        }
3668    }
3669
3670    /// P4e DEFECT-FIX (independent Fable-5 review of P4e: `core.session.persist`
3671    /// had a `Config` field and CLI plumbing at `ConfigProfile` → `Config` but
3672    /// no consumer at all): whether a CLI caller's session-store save sites
3673    /// (`persist_session`, `persist_full_view`) should actually write to
3674    /// disk. `true` (the default) is byte-identical to pre-fix behavior —
3675    /// every session persists. `false` makes a session ephemeral: it runs
3676    /// exactly as before, but no `<name>.jsonl`/sidecar family is ever
3677    /// written for it. A plain getter, same posture as [`Self::model`] —
3678    /// this crate itself never reads or enforces it; the CLI's save sites do.
3679    pub fn session_persist(&self) -> bool {
3680        self.config.session_persist
3681    }
3682
3683    /// P4e DEFECT-FIX (independent Fable-5 review of P4e: `core.session.name`
3684    /// had a `Config` field and CLI plumbing but no consumer): the
3685    /// caller-configured session name, if `[core.session] name` was set.
3686    /// `None` (the default) leaves session naming exactly as before —
3687    /// `mint_session_name`'s auto-generated `<tag>-<adjective>-<noun>` shape.
3688    /// A plain getter, same posture as [`Self::session_persist`].
3689    pub fn session_name(&self) -> Option<&str> {
3690        self.config.session_name.as_deref()
3691    }
3692
3693    /// PARITY-18 D3 — whether this agent has actually issued at least one
3694    /// live request to its [`Provider`] so far (set the instant
3695    /// `Self::run_loop` reaches its real send site, regardless of whether
3696    /// that call then succeeds or fails). Callers should report
3697    /// "request sent" from THIS, never from having merely passed the
3698    /// context guard or having called [`Self::send`] — either of those can
3699    /// happen with zero requests actually issued (a guard refusal, an
3700    /// interactive session quit before any turn completes).
3701    pub fn request_issued(&self) -> bool {
3702        self.requests_issued
3703    }
3704
3705    /// P5-2 (§2.2 C2): whether this agent currently considers its
3706    /// [`CachePlan::ImportedPrefix`] cache entry warm — mirrors
3707    /// [`Self::request_issued`]'s read-only-observability precedent, so a
3708    /// caller (or a test) can confirm [`Self::register_tool`]'s C2
3709    /// invalidation actually took effect without reaching into private
3710    /// state.
3711    pub fn cache_established(&self) -> bool {
3712        self.cache_established
3713    }
3714
3715    /// B7: length of the imported-prefix protected by [`CachePlan::ImportedPrefix`]
3716    /// (this agent's own system message plus every message of a
3717    /// previously-imported session), set by [`Self::load_session`]. `None`
3718    /// until a session has been loaded.
3719    pub fn imported_prefix_len(&self) -> Option<usize> {
3720        self.imported_prefix_len
3721    }
3722
3723    /// Replace this agent's accumulating reduction log (C4: `/expand`/`/reduce`
3724    /// mutate the log directly via `reduce::invert_one`/`reduce::project_messages`
3725    /// and must feed the result back here so the *next* request build or
3726    /// persist sees the updated state instead of silently recomputing from an
3727    /// empty log). Also lets a caller (`resume_cmd`, C1) seed the log with the
3728    /// initial projection it already computed for the entry banner, so
3729    /// `reduction_log()` reflects reality even before this agent's first
3730    /// `send()` (which is otherwise the only place `build_request_messages`
3731    /// populates it).
3732    pub fn set_reduction_log(&mut self, log: ReductionLog) {
3733        self.reduction_log = log;
3734    }
3735
3736    /// Replace the conversation with a loaded session, keeping this agent's own
3737    /// system prompt at the front. The session's own system/developer turns are
3738    /// preserved after it for context.
3739    pub fn load_session(&mut self, session: Session) {
3740        let system = self.history.first().cloned();
3741        self.history.clear();
3742        if let Some(sys) = system {
3743            self.history.push(sys);
3744        }
3745        self.history.extend(session.messages);
3746        // B7: the whole of `history` at this point — this agent's own system
3747        // message plus every imported message — is the stable prefix a
3748        // resumed session resends byte-identically every turn.
3749        self.imported_prefix_len = Some(self.history.len());
3750        // UX-26 (B7-warn): a freshly loaded prefix has no established cache
3751        // entry of THIS agent's own making yet (even if this agent was
3752        // resumed once before — that earlier prefix is gone). Seed the
3753        // activity clock from the loaded session's own last message
3754        // timestamp (walking backward past any trailing message that
3755        // carries none), so a session that's been sitting idle since
3756        // Claude Code/Codex/a prior supercode run last touched it is
3757        // correctly treated as already-cold on its very first turn here —
3758        // `None` (no timestamp anywhere in the loaded messages) leaves the
3759        // TTL check disarmed rather than guessing.
3760        self.cache_established = false;
3761        self.last_cache_activity_ms = self
3762            .history
3763            .iter()
3764            .rev()
3765            .find_map(|m| m.metadata.get("timestamp"))
3766            .and_then(|ts| supercode_interchange::sidecar::rfc3339_to_ms(ts));
3767    }
3768
3769    /// Append `msg` to the sidecar recorder (A3), if one is installed — a
3770    /// no-op, at zero cost, when `recorder` is `None` (today's behavior).
3771    fn record(&mut self, msg: &ChatMessage) -> Result<()> {
3772        if let Some(recorder) = self.recorder.as_mut() {
3773            recorder.append(msg)?;
3774        }
3775        // BP-8 (catalog:150): the append-only half — written and FLUSHED
3776        // here, at the moment the message exists, not at the end of the
3777        // turn. A journal failure is logged, never fatal: durability
3778        // bookkeeping must not be able to fail a turn.
3779        if let Some(journal) = &self.journal {
3780            let mut guard = journal
3781                .lock()
3782                .unwrap_or_else(std::sync::PoisonError::into_inner);
3783            if let Err(error) = guard.append_message(msg) {
3784                tracing::warn!("failed to journal a message: {error}");
3785            }
3786        }
3787        // BP-8 (catalog:151): the same message becomes a tree node, so the
3788        // tree and the linear history never disagree about what was said.
3789        if let Some(tree) = self.session_tree.as_mut() {
3790            tree.append_message(msg.clone(), now_ms());
3791        }
3792        Ok(())
3793    }
3794
3795    /// Persist the live conversation to `path` as JSONL (one [`ChatMessage`]
3796    /// per line) so the session can be resumed later — supercode's own sessions
3797    /// become first-class, resumable artifacts.
3798    pub fn save_transcript(&self, path: impl AsRef<std::path::Path>) -> Result<()> {
3799        let mut out = String::new();
3800        for m in &self.history {
3801            out.push_str(&serde_json::to_string(m).map_err(Error::Decode)?);
3802            out.push('\n');
3803        }
3804        std::fs::write(path, out)?;
3805        Ok(())
3806    }
3807
3808    /// Restore a conversation previously written with [`Self::save_transcript`],
3809    /// replacing the current history.
3810    pub fn load_transcript(&mut self, path: impl AsRef<std::path::Path>) -> Result<()> {
3811        let text = std::fs::read_to_string(path)?;
3812        let mut history = Vec::new();
3813        for line in text.lines().map(str::trim).filter(|l| !l.is_empty()) {
3814            history.push(serde_json::from_str::<ChatMessage>(line).map_err(Error::Decode)?);
3815        }
3816        self.history = history;
3817        Ok(())
3818    }
3819
3820    /// Take a checkpoint of the current conversation position. Pass it to
3821    /// [`Self::rewind_to`] to discard everything sent since (the rewind/undo
3822    /// analog of `fork`/checkpoint).
3823    pub fn checkpoint(&self) -> usize {
3824        self.history.len()
3825    }
3826
3827    /// Rewind the conversation to a [`Self::checkpoint`], discarding later turns.
3828    pub fn rewind_to(&mut self, checkpoint: usize) {
3829        self.history.truncate(checkpoint.min(self.history.len()));
3830    }
3831
3832    /// Send a message with file inputs attached — the `--file` / `-i` analog.
3833    /// Each file's contents are injected into the prompt: UTF-8 text inline,
3834    /// binary (e.g. images) noted with a size marker. (Native image *vision*
3835    /// would additionally require multimodal content parts.)
3836    pub async fn send_with_files(
3837        &mut self,
3838        text: impl Into<String>,
3839        files: &[std::path::PathBuf],
3840    ) -> Result<String> {
3841        let mut prompt = text.into();
3842        for path in files {
3843            let block = match std::fs::read(path) {
3844                Ok(bytes) => match String::from_utf8(bytes.clone()) {
3845                    Ok(s) => format!("\n\n[file: {}]\n{}", path.display(), s),
3846                    Err(_) => format!(
3847                        "\n\n[file: {} — {} bytes, binary content omitted]",
3848                        path.display(),
3849                        bytes.len()
3850                    ),
3851                },
3852                Err(e) => format!("\n\n[file: {} — could not read: {e}]", path.display()),
3853            };
3854            prompt.push_str(&block);
3855        }
3856        let expanded = self.expand_prompt_async(&prompt).await;
3857        let msg = ChatMessage::user(expanded);
3858        self.guard_candidate_message(&msg)?;
3859        self.record(&msg)?;
3860        self.history.push(msg);
3861        self.run_loop().await
3862    }
3863
3864    /// Send a message with image inputs to a vision model — the `-i/--image`
3865    /// analog. `image_urls` may be `https://…` links or `data:image/…;base64,…`
3866    /// URLs; they're attached as multimodal `image_url` content parts.
3867    pub async fn send_with_images(
3868        &mut self,
3869        text: impl Into<String>,
3870        image_urls: &[String],
3871    ) -> Result<String> {
3872        let expanded = self.expand_prompt_async(&text.into()).await;
3873        let msg = ChatMessage::user_with_images(expanded, image_urls);
3874        self.guard_candidate_message(&msg)?;
3875        self.record(&msg)?;
3876        self.history.push(msg);
3877        self.run_loop().await
3878    }
3879
3880    /// Expand a `/<name> <args>` slash command against the registered prompt
3881    /// templates (`{args}` is replaced with the trailing text). Non-matching
3882    /// input is returned unchanged.
3883    /// BP-6 additionally resolves SKILL.md invocations here, after the
3884    /// template table misses: `/skill:name args` (pi§2 "Skill commands"),
3885    /// `/name args` when the config follows Claude Code (cc§7: "a `SKILL.md`
3886    /// in a directory = a `/name` command"), and `$slug` mentions (cx§7).
3887    /// `$ARGUMENTS` in the body is replaced with the trailing text.
3888    pub fn expand_prompt(&self, input: &str) -> String {
3889        // BP-5 (catalog D2 "@-file mentions / attachments"): `@path`
3890        // expansion happens FIRST, so a mention works in a bare message, in
3891        // a slash-command's arguments, and in the text a `$slug` mention
3892        // appends to — one rule, every prompt shape.
3893        let input = &self.expand_file_mentions(input);
3894        let trimmed = input.trim_start();
3895        let Some(rest) = trimmed.strip_prefix('/') else {
3896            return self.expand_skill_mentions(input);
3897        };
3898        let (name, args) = match rest.split_once(char::is_whitespace) {
3899            Some((n, a)) => (n, a.trim()),
3900            None => (rest, ""),
3901        };
3902        match self.config.prompts.get(name) {
3903            Some(template) => template.replace("{args}", args),
3904            None => match self.expand_skill_command(name, args) {
3905                Some(expanded) => expanded,
3906                None => self.expand_skill_mentions(input),
3907            },
3908        }
3909    }
3910
3911    /// BP-5 (catalog D2 "@-file mentions / attachments"; cc§2 "`@` in the
3912    /// prompt triggers file-path autocomplete and injects file context …
3913    /// Read deny rules best-effort apply to `@file` mentions"; cx§2
3914    /// "`@`-mentions (files)"): replace each `@path` token in `input` with
3915    /// that file's contents.
3916    ///
3917    /// **Deny-rule aware, through the one permissions engine.** Each
3918    /// mention is resolved with
3919    /// [`crate::permissions::evaluate_path_safe`] — the same
3920    /// traversal/symlink-resolving check a `read_file` tool call goes
3921    /// through — against this config's own rules and protected-path floor.
3922    /// Anything short of `Allow` inlines the refusal instead of the file, so
3923    /// `@.env` under a preset whose protected paths cover it says so rather
3924    /// than quietly leaking it.
3925    ///
3926    /// A token that names nothing readable is left exactly as the user typed
3927    /// it: an email address, a decorator, or a `@`-prefixed word in prose is
3928    /// not a file mention, and must survive untouched.
3929    /// Off by default (`[core.file_mentions]`).
3930    fn expand_file_mentions(&self, input: &str) -> String {
3931        if !self.config.file_mentions || !input.contains('@') {
3932            return input.to_string();
3933        }
3934        let mut attachments = String::new();
3935        let mut seen: Vec<String> = Vec::new();
3936        for token in input.split_whitespace() {
3937            let Some(rel) = token.strip_prefix('@') else {
3938                continue;
3939            };
3940            let rel = rel.trim_end_matches([',', ';', ':', '.', ')', ']', '"', '\'']);
3941            if rel.is_empty() || seen.iter().any(|s| s == rel) {
3942                continue;
3943            }
3944            let path = if std::path::Path::new(rel).is_absolute() {
3945                std::path::PathBuf::from(rel)
3946            } else {
3947                self.config.cwd.join(rel)
3948            };
3949            if !path.is_file() {
3950                continue;
3951            }
3952            seen.push(rel.to_string());
3953            attachments.push_str(&self.render_mention(rel, &path));
3954            if seen.len() >= MAX_FILE_MENTIONS_PER_MESSAGE {
3955                break;
3956            }
3957        }
3958        if attachments.is_empty() {
3959            return input.to_string();
3960        }
3961        format!("{input}{attachments}")
3962    }
3963
3964    /// One mention's block: the permission verdict first, then the bytes.
3965    /// Text is inlined; a binary file is named with its size, the same
3966    /// shape [`Self::send_with_files`] already uses for an explicit
3967    /// attachment, so a mention and a `--file` read the same way.
3968    fn render_mention(&self, shown: &str, path: &std::path::Path) -> String {
3969        use crate::permissions::{Decision, PathKind};
3970        let rules = crate::permissions::rules_for_config(&self.config);
3971        // BP-10's multi-root form: a mention is checked against every
3972        // granted root (cwd + `additional_dirs`), folded to the strictest —
3973        // the same call the tool-dispatch gate makes for a `read_file`
3974        // path, so a mention can never reach a file a read could not.
3975        let mut roots = vec![self.config.cwd.clone()];
3976        roots.extend(self.config.additional_dirs.iter().cloned());
3977        let decision = crate::permissions::evaluate_path_safe_roots(
3978            &rules,
3979            PathKind::Read,
3980            &roots,
3981            &path.to_string_lossy(),
3982            Decision::Allow,
3983        );
3984        if decision != Decision::Allow {
3985            return format!(
3986                "\n\n[file: {shown} — not attached; the permission rules for this session \
3987                 resolve reading it to {decision:?}]"
3988            );
3989        }
3990        match std::fs::read(path) {
3991            Ok(bytes) => match String::from_utf8(bytes) {
3992                Ok(text) => {
3993                    let mut text = text;
3994                    if text.len() > MAX_FILE_MENTION_BYTES {
3995                        let mut cut = MAX_FILE_MENTION_BYTES;
3996                        while cut > 0 && !text.is_char_boundary(cut) {
3997                            cut -= 1;
3998                        }
3999                        text.truncate(cut);
4000                        text.push_str("\n[file truncated]");
4001                    }
4002                    format!("\n\n[file: {shown}]\n{text}")
4003                }
4004                Err(e) => format!(
4005                    "\n\n[file: {shown} — {} bytes, binary content omitted]",
4006                    e.into_bytes().len()
4007                ),
4008            },
4009            Err(e) => format!("\n\n[file: {shown} — could not read: {e}]"),
4010        }
4011    }
4012
4013    /// The SKILL.md packages this agent discovered (frontmatter only) — the
4014    /// exact set its prompt index lists and its `skill` tool can load.
4015    pub fn skills(&self) -> &[crate::skills::LoopSkill] {
4016        &self.skills
4017    }
4018
4019    /// BP-6: resolve a slash command against the discovered skills.
4020    ///
4021    /// `/skill:<name>` is pi's own form and is accepted under every config
4022    /// (it can never collide with a template name, which cannot contain a
4023    /// colon-prefixed `skill` segment by construction). The BARE `/<name>`
4024    /// form is Claude Code's — there, a skill IS a slash command — so it is
4025    /// honored only when the config reads Claude Code's roots; under
4026    /// `cx-parity`, where Codex has no skill slash commands, `/deploy` stays
4027    /// the literal text the user typed.
4028    fn expand_skill_command(&self, name: &str, args: &str) -> Option<String> {
4029        if self.skills.is_empty() {
4030            return None;
4031        }
4032        let bare = match name.strip_prefix("skill:") {
4033            Some(rest) => rest,
4034            None if self.config.skills_harness.as_deref()
4035                == Some(crate::HarnessId::CLAUDE_CODE) =>
4036            {
4037                name
4038            }
4039            None => return None,
4040        };
4041        let skill = self.find_skill(bare)?;
4042        skill
4043            .body_with_shell(args, &self.shell_injection)
4044            .ok()
4045            .map(|body| crate::skills::render_skill(skill, &body))
4046    }
4047
4048    /// BP-6: `$slug` mentions (cx§7 `TOOL_MENTION_SIGIL = '$'`) and — only
4049    /// under `[core.skills] implicit_match` — a description match.
4050    ///
4051    /// The user's own text is never replaced: a loaded body is APPENDED, the
4052    /// way Codex splices a skill into the turn. Mentions are only honored
4053    /// for a config that reads Codex's roots; `$WORD` is ordinary shell text
4054    /// everywhere else.
4055    fn expand_skill_mentions(&self, input: &str) -> String {
4056        if self.skills.is_empty() {
4057            return input.to_string();
4058        }
4059        let mut loaded: Vec<String> = Vec::new();
4060        let mut names: Vec<String> = Vec::new();
4061        if self.config.skills_harness.as_deref() == Some(crate::HarnessId::CODEX) {
4062            for token in input.split_whitespace() {
4063                let Some(slug) = token.strip_prefix('$') else {
4064                    continue;
4065                };
4066                let slug =
4067                    slug.trim_matches(|c: char| !c.is_alphanumeric() && c != '-' && c != ':');
4068                if slug.is_empty() {
4069                    continue;
4070                }
4071                let Some(skill) = self.find_skill(slug) else {
4072                    continue;
4073                };
4074                if names.contains(&skill.name) || loaded.len() >= MAX_SKILL_LOADS_PER_MESSAGE {
4075                    continue;
4076                }
4077                if let Ok(body) = skill.body_with_shell("", &self.shell_injection) {
4078                    names.push(skill.name.clone());
4079                    loaded.push(crate::skills::render_skill(skill, &body));
4080                }
4081            }
4082        }
4083        if loaded.is_empty() && self.config.skills_implicit_match {
4084            if let Some(skill) = crate::skills::implicit_skill_match(&self.skills, input) {
4085                if let Ok(body) = skill.body_with_shell("", &self.shell_injection) {
4086                    loaded.push(crate::skills::render_skill(skill, &body));
4087                }
4088            }
4089        }
4090        if loaded.is_empty() {
4091            return input.to_string();
4092        }
4093        format!("{input}\n\n{}", loaded.join("\n\n"))
4094    }
4095
4096    /// Resolve one invocation name against the discovered set — the same
4097    /// resolver the `skill` tool uses, so every door agrees on what a name
4098    /// means.
4099    fn find_skill(&self, name: &str) -> Option<&crate::skills::LoopSkill> {
4100        crate::skills::find_skill(&self.skills, name)
4101    }
4102
4103    /// P5-2 (§2 module 15 D7 row 4 "prompts-as-commands"): like
4104    /// [`Self::expand_prompt`], but also consults MCP-server-sourced
4105    /// prompts registered via [`Self::register_mcp_prompt`] when the local
4106    /// `Config::prompts` table has no match — a live `prompts/get`
4107    /// round-trip, which is why this is async and [`Self::expand_prompt`]
4108    /// itself stays synchronous (its public sync signature is unchanged,
4109    /// for every existing caller that doesn't need MCP prompts).
4110    ///
4111    /// **Argument mapping (a scope decision, not a protocol requirement —
4112    /// the MCP spec leaves "how does free CLI text become named prompt
4113    /// arguments" to the client):** a prompt with zero or one declared
4114    /// arguments gets the whole trailing text (empty string if the prompt
4115    /// takes no arguments and none was given); a prompt with two or more
4116    /// declared arguments expects `key=value` pairs, whitespace-separated
4117    /// (`/mcp__server__prompt lang=rust topic=async`) — an unparseable pair
4118    /// (no `=`) is simply skipped, never a hard error (matches this
4119    /// method's "non-matching input passes through" fail-open posture for
4120    /// the LOCAL-prompt case above).
4121    pub async fn expand_prompt_async(&self, input: &str) -> String {
4122        let local = self.expand_prompt(input);
4123        if local != input {
4124            return local; // a local `Config::prompts` template matched
4125        }
4126        let trimmed = input.trim_start();
4127        let Some(rest) = trimmed.strip_prefix('/') else {
4128            return input.to_string();
4129        };
4130        let (name, args) = match rest.split_once(char::is_whitespace) {
4131            Some((n, a)) => (n, a.trim()),
4132            None => (rest, ""),
4133        };
4134        let Some(source) = self.mcp_prompts.get(name) else {
4135            return input.to_string();
4136        };
4137        let arg_map = match source.arg_names() {
4138            [] => std::collections::BTreeMap::new(),
4139            [single] => {
4140                let mut m = std::collections::BTreeMap::new();
4141                if !args.is_empty() {
4142                    m.insert(single.clone(), args.to_string());
4143                }
4144                m
4145            }
4146            _ => args
4147                .split_whitespace()
4148                .filter_map(|pair| pair.split_once('='))
4149                .map(|(k, v)| (k.to_string(), v.to_string()))
4150                .collect(),
4151        };
4152        match source.render(arg_map).await {
4153            Ok(rendered) => rendered,
4154            Err(e) => format!("Error: mcp prompt `{name}` failed: {e}"),
4155        }
4156    }
4157
4158    /// P5-2 (§2 module 15 D7 row 4): register an MCP server's prompt as a
4159    /// slash-command source — `command_name` MUST already be the
4160    /// namespaced `mcp__<server>__<prompt>` form
4161    /// ([`crate::mcp::McpServerHandle::prompts`] produces exactly that
4162    /// shape); this method does not re-namespace or validate it, so a
4163    /// caller that hands it a bare name defeats the collision protection
4164    /// [`crate::mcp::McpPromptSource`]'s doc comment describes. Overwrites
4165    /// any prior registration under the same command name (re-attaching
4166    /// the same server replaces its own earlier prompt list; this can
4167    /// never touch a NON-`mcp__`-prefixed key, i.e. never a local
4168    /// `Config::prompts` entry).
4169    pub fn register_mcp_prompt(
4170        &mut self,
4171        command_name: impl Into<String>,
4172        source: impl crate::sdk::SdkPromptSource + 'static,
4173    ) {
4174        self.mcp_prompts
4175            .insert(command_name.into(), Box::new(source));
4176    }
4177
4178    /// P5-2 (§2 module 15 D7 row 5 "instructions"): fold an MCP server's
4179    /// `initialize`-time instructions (or any other free-text note) into
4180    /// this agent's system message — the context-assembly site every other
4181    /// `core.*`/`capabilities.*` prompt-section append already uses
4182    /// (`Self::with_parts`), except this one fires AFTER construction
4183    /// (attaching MCP servers happens once the agent already exists — see
4184    /// `crates/cli/src/main.rs`'s `attach_mcp`). A no-op if `history` is
4185    /// somehow empty or its first message isn't a system message (never
4186    /// true for an `Agent` built via `Self::new`/`Self::with_parts`, but
4187    /// checked rather than assumed).
4188    pub fn append_system_note(&mut self, text: &str) {
4189        if let Some(system) = self.history.first_mut() {
4190            if system.role == Role::System {
4191                system
4192                    .content
4193                    .get_or_insert_with(String::new)
4194                    .push_str(text);
4195            }
4196        }
4197    }
4198
4199    /// BP-4 (catalog:90, cx§2 `<environment_context>` "re-emitted on
4200    /// change"): re-derive the `# Environment` block and, if anything in it
4201    /// moved — cwd, the approval/sandbox policy, the git branch or its
4202    /// dirty state, the date — replace the stale copy in the system message
4203    /// with the fresh one. Returns whether the block changed.
4204    ///
4205    /// A no-op (and free — no git subprocess) when `core.env_context` is
4206    /// off, which is the default and every non-parity config. Replacing in
4207    /// place rather than appending a second block is deliberate: two
4208    /// `# Environment` sections disagreeing about cwd is worse context than
4209    /// one stale one, and the system message is re-sent on every request,
4210    /// so the rewrite IS the re-emission the model sees.
4211    pub fn refresh_env_context(&mut self) -> bool {
4212        if !self.config.env_context {
4213            return false;
4214        }
4215        let fresh = env_context_block(&self.config);
4216        let Some(stale) = self.env_context_live.clone() else {
4217            // Nothing was spliced at construction (e.g. `with_provider_arc`);
4218            // splice it now rather than silently never emitting one.
4219            self.append_system_note(&fresh);
4220            self.env_context_live = Some(fresh);
4221            return true;
4222        };
4223        if stale == fresh {
4224            return false;
4225        }
4226        if let Some(system) = self.history.first_mut() {
4227            if system.role == Role::System {
4228                if let Some(content) = system.content.as_mut() {
4229                    if let Some(at) = content.find(&stale) {
4230                        content.replace_range(at..at + stale.len(), &fresh);
4231                        self.env_context_live = Some(fresh);
4232                        return true;
4233                    }
4234                }
4235            }
4236        }
4237        false
4238    }
4239
4240    /// BP-5 (catalog D2 "Per-model-family base-prompt selection"): re-select
4241    /// the base prompt for the model now in force and replace the stale one
4242    /// in place. Returns whether the system message changed.
4243    ///
4244    /// A no-op — not even a string search — when the selection is unchanged,
4245    /// which is every config that sets no `base_prompts` table.
4246    fn refresh_base_prompt(&mut self) -> bool {
4247        let fresh = base_prompt_for_config(&self.config);
4248        if fresh == self.base_prompt_live {
4249            return false;
4250        }
4251        let stale = std::mem::replace(&mut self.base_prompt_live, fresh.clone());
4252        if stale.is_empty() {
4253            return false;
4254        }
4255        if let Some(system) = self.history.first_mut() {
4256            if system.role == Role::System {
4257                if let Some(content) = system.content.as_mut() {
4258                    if let Some(at) = content.find(&stale) {
4259                        content.replace_range(at..at + stale.len(), &fresh);
4260                        return true;
4261                    }
4262                }
4263            }
4264        }
4265        false
4266    }
4267
4268    /// BP-5: assemble the request from an already-built message list and
4269    /// tool-schema list. The ONE place a [`ChatRequest`] is constructed from
4270    /// this agent's config, so the request `Self::run_loop` issues and the
4271    /// request [`Self::model_input`] renders cannot drift apart.
4272    fn chat_request(&self, messages: Vec<ChatMessage>, tools: Vec<ToolSchema>) -> ChatRequest {
4273        let mut req = ChatRequest {
4274            model: self.config.model.clone(),
4275            messages,
4276            tools,
4277            temperature: self.config.temperature,
4278            max_tokens: self.config.max_tokens,
4279            effort: self.config.effort.clone(),
4280            response_format: self.config.response_format.clone(),
4281            service_tier: None,
4282            thinking_budget: None,
4283            extra_body: self.config.extra_body.clone(),
4284        };
4285        // BP-13 (catalog Domain 9): the per-request routing decisions —
4286        // effort LEVEL, thinking-token BUDGET and service TIER — all come
4287        // out of `Config::model_routing` keyed by the model this request is
4288        // actually going to. Applied HERE so `model_input`'s rendering and
4289        // the loop's own send can never disagree about what would be sent,
4290        // and so a mid-session switch re-decides all three for the new
4291        // model on the next pass.
4292        self.apply_routing(&mut req);
4293        req
4294    }
4295
4296    /// BP-5 (catalog D2 "Prompt-input debugging": *render the exact
4297    /// model-visible input for inspection*; cx§2 `codex debug prompt-input`,
4298    /// which "renders the exact model-visible input list as JSON"): the
4299    /// request this agent would send next.
4300    ///
4301    /// Built by the SAME two calls the loop makes
4302    /// ([`Self::build_request_messages`], [`Self::tool_schemas`]) and
4303    /// assembled by the SAME [`Self::chat_request`] — it is the real
4304    /// request, not a reconstruction of one. `&mut self` because
4305    /// `build_request_messages` is: rendering the input is exactly as
4306    /// stateful as building it for a send.
4307    pub fn model_input(&mut self) -> ChatRequest {
4308        let tools = self.tool_schemas();
4309        let messages = self.build_request_messages();
4310        self.chat_request(messages, tools)
4311    }
4312
4313    /// BP-5: [`Self::model_input`] for a turn that has not been sent —
4314    /// `prompt` is expanded exactly as [`Self::send`] would expand it
4315    /// (slash templates, skills, `@path` mentions, MCP prompts) and appended
4316    /// to the conversation IN MEMORY, then the request is rendered.
4317    ///
4318    /// Deliberately not recorded: this door inspects an input, it does not
4319    /// take a turn. Nothing is written to the session store, no journal
4320    /// entry is made, and no request is issued.
4321    pub async fn model_input_for(&mut self, prompt: &str) -> ChatRequest {
4322        let expanded = self.expand_prompt_async(prompt).await;
4323        self.history.push(ChatMessage::user(expanded));
4324        self.model_input()
4325    }
4326
4327    /// BP-5: a [`ChatRequest`] as the JSON a human (or `jq`) inspects — the
4328    /// system prompt, every message in order, and every advertised tool
4329    /// schema, plus the sampling controls that travel with them.
4330    pub fn render_model_input(req: &ChatRequest) -> serde_json::Value {
4331        serde_json::json!({
4332            "model": req.model,
4333            "temperature": req.temperature,
4334            "max_tokens": req.max_tokens,
4335            "effort": req.effort,
4336            "response_format": req.response_format,
4337            // Serialized through `ChatMessage`'s OWN wire serializer and
4338            // `ToolSchema`'s own — i.e. the exact bytes the provider is
4339            // handed, not a second rendering of them.
4340            "messages": serde_json::to_value(&req.messages).unwrap_or(serde_json::Value::Null),
4341            "tools": serde_json::to_value(&req.tools).unwrap_or(serde_json::Value::Null),
4342        })
4343    }
4344
4345    /// BP-5 (catalog D2 "Per-model-family base-prompt selection"): the base
4346    /// system prompt currently in force for this agent's model.
4347    pub fn base_prompt(&self) -> &str {
4348        &self.base_prompt_live
4349    }
4350
4351    /// BP-4 (catalog:91 "Synthetic context-injection blocks"): splice one
4352    /// named ambient block into the live context — the seam a hook's
4353    /// `additionalContext`, a frontend nudge or an orchestrator's brief
4354    /// enters through, mid-session, after construction.
4355    ///
4356    /// Requires `core.context_injections` (returns `false` otherwise): the
4357    /// gate governs the whole registry, not just its startup half. The
4358    /// block is appended to the system message and remembered, so it is
4359    /// carried by every later request and re-rendered by
4360    /// [`crate::context_injection::assemble`] wherever the prompt is
4361    /// rebuilt.
4362    pub fn inject_context_block(
4363        &mut self,
4364        name: impl Into<String>,
4365        content: impl Into<String>,
4366    ) -> bool {
4367        if !self.config.context_injections {
4368            return false;
4369        }
4370        let block = crate::config::ContextInjectionBlock::new(name, content);
4371        let rendered = crate::context_injection::render(std::slice::from_ref(&block));
4372        self.spliced_context_blocks.push(block);
4373        self.append_system_note(&rendered);
4374        true
4375    }
4376
4377    /// The blocks spliced in since construction — see
4378    /// [`Self::inject_context_block`].
4379    pub fn spliced_context_blocks(&self) -> &[crate::config::ContextInjectionBlock] {
4380        &self.spliced_context_blocks
4381    }
4382
4383    /// Compact the conversation if it has grown past the configured
4384    /// threshold.
4385    ///
4386    /// **Re-founded (A10):** with a [`ReductionPolicy`] installed
4387    /// ([`Self::set_reduction_policy`]), this no longer touches `self.history`
4388    /// at all. It derives `policy.clear_turns_older_than` from
4389    /// `compact_after_messages` so the *next* projected request view
4390    /// (`reduce::project_messages`, built in `Self::run_loop`) collapses the
4391    /// old turns into one reversible `TurnsCleared` stub instead —
4392    /// `history()` and the sidecar keep every message forever; only the view
4393    /// shrinks. Returns whether the live (unreduced) history currently
4394    /// exceeds the threshold, i.e. whether a clearing will actually be
4395    /// visible in the next projected view.
4396    ///
4397    /// **Legacy path (no policy) — LOSSY, kept only for byte-identical
4398    /// backward compatibility (D6):** destructively rewrites `self.history`,
4399    /// permanently discarding the dropped middle turns (replaced by a single
4400    /// non-reversible summary marker that becomes their SOLE remaining copy —
4401    /// exactly the lossy compaction this reduction layer differentiates
4402    /// against). Once a sidecar/recorder or a [`ReductionPolicy`] is in play,
4403    /// prefer installing a policy so this method takes the re-founded path
4404    /// above instead.
4405    pub fn maybe_compact(&mut self) -> bool {
4406        // P4e (§1.5/§3.1 `core.compaction.enabled`, "no master gate exists
4407        // yet"): checked FIRST, before either trigger — `false` disables
4408        // every auto-compaction trigger unconditionally (message-count AND
4409        // pressure), composing with them rather than replacing their own
4410        // logic. `true` (the default, matching today's pre-P4e behavior,
4411        // where nothing ever gated compaction) falls straight through to
4412        // the existing trigger checks below, unchanged.
4413        if !self.config.compaction_enabled {
4414            return false;
4415        }
4416        let threshold = self.config.compact_after_messages;
4417        // P4b (§1.5/§3.1 `core.compaction.reserve_tokens`, pi§2 shape): a
4418        // SECOND, independent trigger — context-window pressure — alongside
4419        // (not instead of) the message-count one above. `None` (the
4420        // default) is byte-identical to today's message-count-only
4421        // behavior; this whole block is a no-op then.
4422        let message_trigger = threshold.is_some_and(|t| self.history.len() > t);
4423        let pressure_trigger = self.compaction_pressure_triggered();
4424        if threshold.is_none() && self.config.compaction_reserve_tokens.is_none() {
4425            return false;
4426        }
4427        if !message_trigger && !pressure_trigger {
4428            return false;
4429        }
4430        if let Some(policy) = self.reduction_policy.as_mut() {
4431            if let Some(t) = threshold {
4432                policy.clear_turns_older_than = Some(t);
4433            }
4434            // P4b scope note: the token-PRESSURE trigger's "how much to
4435            // clear" derivation (below, for the legacy in-place path) has no
4436            // `ReductionPolicy`/A10 analog yet — that mechanism decides its
4437            // own clearing window once `clear_turns_older_than` is set, so
4438            // pressure firing alone (no message threshold configured) has
4439            // nothing new to hand it in this pass. Report the message-count
4440            // verdict only, matching today's pre-P4b behavior exactly when
4441            // only `threshold` is set.
4442            return message_trigger;
4443        }
4444        // Legacy in-place compaction (no `ReductionPolicy` installed) below.
4445        // `keep_recent`: the message-count trigger's own `threshold / 2`
4446        // shape when it's what fired (or both fired); otherwise (pressure
4447        // fired alone) a token-budget-derived count.
4448        let keep_recent = if message_trigger {
4449            (threshold.unwrap() / 2).max(2)
4450        } else {
4451            self.keep_recent_count_by_tokens()
4452        };
4453        self.compact_in_place(keep_recent, None)
4454    }
4455
4456    /// BP-4 (catalog:98 "Manual compact with focus instructions", cc§2 /
4457    /// cx§2 `/compact [instructions]`): compact NOW, regardless of whether
4458    /// either automatic trigger has fired — the mechanism behind the REPL's
4459    /// `/compact [focus]`.
4460    ///
4461    /// `focus` is this invocation's steering text: it overrides the standing
4462    /// `core.compaction.focus_instructions` for this compaction only, is
4463    /// carried into the SUMMARIZER's input (so the model-written summary
4464    /// preserves what the user asked for), and is stated on the marker. An
4465    /// empty/whitespace `focus` falls back to the configured standing value,
4466    /// which is what a bare `/compact` means.
4467    ///
4468    /// Returns whether anything was compacted (`false` when the history is
4469    /// already at or below the keep-window, or when a [`ReductionPolicy`] is
4470    /// installed — under a policy the reversible A10 path owns clearing, and
4471    /// a manual compact would be the lossy one).
4472    pub fn compact_now(&mut self, focus: Option<&str>) -> bool {
4473        self.compacting_manually = true;
4474        let compacted = self.compact_now_inner(focus);
4475        self.compacting_manually = false;
4476        compacted
4477    }
4478
4479    fn compact_now_inner(&mut self, focus: Option<&str>) -> bool {
4480        if self.reduction_policy.is_some() {
4481            return false;
4482        }
4483        // A manual compact must actually compact. The token budget alone
4484        // (`core.compaction.keep_recent_tokens`, 20k) keeps EVERYTHING on
4485        // any ordinary conversation, which is right for the pressure
4486        // trigger (it fires only when the window is nearly full) and wrong
4487        // for `/compact`, whose whole point is compacting before the
4488        // pressure arrives. So the keep-window is the tighter of the two:
4489        // the token budget, and the message-count trigger's own established
4490        // "keep the most recent half" shape (`maybe_compact`'s
4491        // `threshold / 2`, floor 2).
4492        let keep_recent = self
4493            .keep_recent_count_by_tokens()
4494            .min((self.history.len() / 2).max(2));
4495        let focus = focus.map(str::trim).filter(|f| !f.is_empty());
4496        self.compact_in_place(keep_recent, focus)
4497    }
4498
4499    /// The legacy (no-[`ReductionPolicy`]) in-place compaction both
4500    /// [`Self::maybe_compact`] and [`Self::compact_now`] run: collapse
4501    /// `history[first..cut)` into one marker, keeping the newest
4502    /// `keep_recent` messages.
4503    fn compact_in_place(&mut self, keep_recent: usize, focus_override: Option<&str>) -> bool {
4504        if self.history.len() <= keep_recent {
4505            return false;
4506        }
4507        // Indices: 0 is the system prompt; collapse [first .. len-keep_recent).
4508        // `first` is 1 (only the system prompt is ever auto-preserved) unless
4509        // B7's coordination clamp widens it.
4510        let mut first = 1usize;
4511        // B7 coordination clamp: this legacy (no-`ReductionPolicy`) path
4512        // mutates `self.history` directly, so — unlike the re-founded A10
4513        // path (clamped inside `reduce::project_messages`, threaded from
4514        // `build_request_messages`) — it must clamp itself. Widening `first`
4515        // (not `cut`) is what actually protects the imported prefix: the
4516        // drop range is `[first, cut)`, so raising `cut` alone would only
4517        // drop MORE messages, not fewer. `imported_prefix_len` is already an
4518        // absolute `history` index count (it protects `history[0..len]`), so
4519        // no offset conversion is needed here.
4520        if matches!(self.config.cache_plan, CachePlan::ImportedPrefix) {
4521            if let Some(protected) = self.imported_prefix_len {
4522                first = first.max(protected);
4523            }
4524        }
4525        let mut cut = self.history.len() - keep_recent;
4526        if cut <= first {
4527            return false;
4528        }
4529        // Never begin the kept window on a tool result: its originating
4530        // assistant turn (with the matching `tool_calls`) is about to be
4531        // dropped, which would orphan the tool message and make the replayed
4532        // conversation invalid. Advance past any leading tool results.
4533        while cut < self.history.len() && self.history[cut].role == Role::Tool {
4534            cut += 1;
4535        }
4536        if cut >= self.history.len() {
4537            return false;
4538        }
4539        let dropped = cut - first;
4540        // BP-11: the compaction is decided from here on — the one point both
4541        // the automatic triggers and `/compact` pass through — so this is
4542        // where `pre_compact` observers hear about it.
4543        self.fire_lifecycle(&crate::config::LifecycleEvent::PreCompact {
4544            messages: self.history.len(),
4545            dropped,
4546            manual: focus_override.is_some() || self.compacting_manually,
4547        });
4548        // P4b (§1.5/§3.1 `core.compaction.focus_instructions`, catalog D2
4549        // "no instruction steering" gap): appended to the marker whenever
4550        // set, regardless of which trigger fired. `None` (the default)
4551        // leaves this byte-identical to the pre-P4b marker text.
4552        //
4553        // BP-4: `focus_override` — the per-invocation `/compact <focus>`
4554        // text — wins over the standing config value for THIS compaction,
4555        // which is what "`/compact [instructions]` steers what's preserved"
4556        // means. Neither is required.
4557        let focus: Option<String> = focus_override.map(str::to_string).or_else(|| {
4558            self.config
4559                .compaction_focus_instructions
4560                .clone()
4561                .filter(|f| !f.is_empty())
4562        });
4563        // BP-1 (§1.5/§3.1 `core.compaction.summarize`): the verb the marker
4564        // uses is now the config's to state. `true` (the default, and what
4565        // every preset sets) keeps the historical "summarized" text
4566        // byte-identical; `false` says only what actually happened to the
4567        // span, so a config that turns summarization off does not leave a
4568        // marker claiming a summary exists.
4569        let verb = if self.config.compaction_summarize {
4570            "summarized"
4571        } else {
4572            "cleared"
4573        };
4574        // BP-4 (catalog:107 "LLM summaries of cleared spans", design §1.5:
4575        // obligation 5 is "auto-compaction … + a persisted marker + AN LLM
4576        // SUMMARY OF THE COMPACTED SPAN", knob `[core.compaction] summarize`
4577        // — "the summary side-call depends on a utility model … core falls
4578        // back to the main model"). The side-call is therefore CORE, not a
4579        // reduction-module privilege: when `core.compaction.summarize` is on
4580        // and a summarizer is installed, the span is summarized by the model
4581        // and the marker carries that summary instead of only a count.
4582        //
4583        // Every failure mode degrades to the count-only marker: no
4584        // summarizer installed, an `Err` from the side-call, or an empty
4585        // reply. It never blocks or fails compaction — the same contract
4586        // TR-7's own side-call site keeps.
4587        let summary_body = if self.config.compaction_summarize {
4588            self.summarize_span(first..cut, focus.as_deref())
4589        } else {
4590            None
4591        };
4592        // BP-4 (catalog:99 "Compaction markers persisted in transcript"):
4593        // the marker states where the originals went, which is the whole
4594        // point of a boundary record — a reader must be able to tell a
4595        // reversible compaction from a lossy one without knowing which
4596        // modules were on.
4597        let retention = if self.recorder.is_some() {
4598            "The compacted messages remain in this session's transcript sidecar."
4599        } else {
4600            "No transcript sidecar is attached, so this marker is the only remaining record of them."
4601        };
4602        let mut summary_text = format!(
4603            "[earlier conversation compacted: {dropped} message(s) {verb} to save context]\n{retention}"
4604        );
4605        if let Some(focus) = &focus {
4606            summary_text.push_str(&format!("\n\nFocus: {focus}"));
4607        }
4608        if let Some(body) = &summary_body {
4609            summary_text.push_str(&format!("\n\nSummary of the compacted span:\n{body}"));
4610        }
4611        let summary = ChatMessage::system(summary_text);
4612        // BP-4 (catalog:99): PERSIST the boundary. Before this the legacy
4613        // path rewrote `self.history` and never called `record`, so the
4614        // marker existed only in the live window and a resumed session had
4615        // no on-disk trace that a compaction ever happened. A recorder
4616        // failure is logged, never fatal — losing the boundary record must
4617        // not lose the compaction.
4618        if let Err(error) = self.record(&summary) {
4619            tracing::warn!("failed to persist the compaction marker: {error}");
4620        }
4621        let mut new_history = Vec::with_capacity(first + keep_recent + 2);
4622        new_history.extend(self.history[..first].iter().cloned());
4623        new_history.push(summary);
4624        new_history.extend(self.history.split_off(cut));
4625        self.history = new_history;
4626        self.fire_lifecycle(&crate::config::LifecycleEvent::PostCompact {
4627            messages: self.history.len(),
4628            dropped,
4629        });
4630        // BP-8 (catalog:150): compaction RESHAPES the live view rather than
4631        // appending to it, so the journal's "everything since the last
4632        // checkpoint is unpersisted" accounting has to be re-based here —
4633        // otherwise a crash-recovery replay would re-append messages this
4634        // compaction deliberately set aside. The set-aside messages' own
4635        // bytes stay in the log above, untouched.
4636        self.journal_checkpoint(self.history.len());
4637        true
4638    }
4639
4640    /// BP-4 (catalog:107): run the installed [`reduce::summarize::SpanSummarizer`]
4641    /// over `history[span]`, with `focus` (the `/compact <focus>` text)
4642    /// carried into the summarizer's INPUT so the model-written summary
4643    /// preserves what the user asked to keep.
4644    ///
4645    /// `None` — never an error — whenever no summarizer is installed, the
4646    /// span renders empty, the side-call fails, or it returns nothing. The
4647    /// caller falls back to the count-only marker.
4648    fn summarize_span(&self, span: std::ops::Range<usize>, focus: Option<&str>) -> Option<String> {
4649        let summarizer = self.span_summarizer.as_deref()?;
4650        let mut span_text = String::new();
4651        // The focus rides at the head of the span text (the trait's one
4652        // input) as an explicit, labeled line rather than a silent prompt
4653        // mutation: the fixed prompt's "do not state anything not present
4654        // in the span" still holds, because the focus IS present in it.
4655        if let Some(focus) = focus {
4656            span_text.push_str(&format!("[compaction focus requested: {focus}]\n\n"));
4657        }
4658        for msg in self.history.get(span)? {
4659            let role = match msg.role {
4660                Role::System => "system",
4661                Role::User => "user",
4662                Role::Assistant => "assistant",
4663                Role::Tool => "tool",
4664            };
4665            span_text.push_str(role);
4666            span_text.push_str(": ");
4667            span_text.push_str(msg.content.as_deref().unwrap_or(""));
4668            span_text.push('\n');
4669        }
4670        match summarizer.summarize(&span_text) {
4671            Ok(text) if !text.trim().is_empty() => Some(text.trim().to_string()),
4672            Ok(_) => None,
4673            Err(error) => {
4674                tracing::warn!("compaction span summarizer failed: {error}");
4675                None
4676            }
4677        }
4678    }
4679
4680    /// BP-4 (catalog:106 "Handoff (fresh objective + curated keep-set)",
4681    /// cx§1 `new_context`): reset the live working view to a fresh
4682    /// objective plus a curated keep-set, in-session.
4683    ///
4684    /// The new view is: the system prompt (plus any imported prefix a
4685    /// `CachePlan::ImportedPrefix` config protects — same clamp compaction
4686    /// uses), then a handoff marker stating the objective and what was set
4687    /// aside, then the most recent `keep_recent` messages (`None` = the
4688    /// token-budget-derived count `core.compaction.keep_recent_tokens`
4689    /// already governs, so the keep-set is curated by the same budget the
4690    /// rest of the compaction machinery uses, not by a magic number). The
4691    /// keep-set never begins on a tool result, so no tool message is left
4692    /// orphaned from its originating assistant turn.
4693    ///
4694    /// Returns how many messages were set aside. Like compaction, the
4695    /// marker is PERSISTED through the recorder, so a resumed session can
4696    /// see where the handoff happened; and like compaction, the set-aside
4697    /// messages remain in the transcript sidecar whenever one is attached.
4698    ///
4699    /// Scope note: this is the in-session `new_context` mechanism, NOT
4700    /// `Config::handoff_enabled`'s reversible ReductionLog snapshot (the
4701    /// offline `supercode handoff` projection) — that one is the reduction
4702    /// module's, and stays there.
4703    pub fn new_context(&mut self, objective: &str, keep_recent: Option<usize>) -> usize {
4704        let keep_recent = keep_recent.unwrap_or_else(|| self.keep_recent_count_by_tokens());
4705        let mut first = 1usize;
4706        if matches!(self.config.cache_plan, CachePlan::ImportedPrefix) {
4707            if let Some(protected) = self.imported_prefix_len {
4708                first = first.max(protected);
4709            }
4710        }
4711        let mut cut = self.history.len().saturating_sub(keep_recent).max(first);
4712        while cut < self.history.len() && self.history[cut].role == Role::Tool {
4713            cut += 1;
4714        }
4715        let dropped = cut.saturating_sub(first);
4716        let objective = objective.trim();
4717        let retention = if self.recorder.is_some() {
4718            "They remain in this session's transcript sidecar."
4719        } else {
4720            "No transcript sidecar is attached, so they are not retained."
4721        };
4722        let marker = ChatMessage::system(format!(
4723            "[handoff: a fresh working context starts here]\nObjective: {objective}\n\
4724             {dropped} earlier message(s) were set aside; the most recent {kept} were kept. \
4725             {retention}",
4726            kept = self.history.len() - cut,
4727        ));
4728        if let Err(error) = self.record(&marker) {
4729            tracing::warn!("failed to persist the handoff marker: {error}");
4730        }
4731        let mut new_history = Vec::with_capacity(first + keep_recent + 2);
4732        new_history.extend(self.history[..first].iter().cloned());
4733        new_history.push(marker);
4734        new_history.extend(self.history.split_off(cut));
4735        self.history = new_history;
4736        // BP-8: same re-basing as `maybe_compact` — see its comment.
4737        self.journal_checkpoint(self.history.len());
4738        dropped
4739    }
4740
4741    /// P4b (§1.5/§3.1 `core.compaction.reserve_tokens`, pi§2 shape:
4742    /// `contextTokens > contextWindow - reserveTokens`): whether the
4743    /// estimated token size of the live history is within `reserve_tokens`
4744    /// of the model's context window. `false` when
4745    /// [`Config::compaction_reserve_tokens`] is unset (the default).
4746    fn compaction_pressure_triggered(&self) -> bool {
4747        let Some(reserve) = self.config.compaction_reserve_tokens else {
4748            return false;
4749        };
4750        let limit = provider::model_context_limit(&self.config.model)
4751            .unwrap_or(provider::UNKNOWN_MODEL_CONTEXT_FLOOR);
4752        let used = supercode_runtime::estimate_view_tokens(&self.history);
4753        used.saturating_add(reserve) > limit
4754    }
4755
4756    /// P4b (§1.5/§3.1 `core.compaction.keep_recent_tokens`): how many of the
4757    /// most recent messages (walking backward from the end of `self.history`,
4758    /// skipping the system prompt) fit within the configured token budget
4759    /// (default 20,000, pi§6 precedent). Always keeps at least 2 messages,
4760    /// matching the message-count trigger's own floor.
4761    fn keep_recent_count_by_tokens(&self) -> usize {
4762        let budget = self.config.compaction_keep_recent_tokens.unwrap_or(20_000);
4763        let mut used = 0u64;
4764        let mut count = 0usize;
4765        for msg in self.history.iter().skip(1).rev() {
4766            let t = supercode_runtime::estimate_view_tokens(std::slice::from_ref(msg));
4767            if used.saturating_add(t) > budget && count > 0 {
4768                break;
4769            }
4770            used = used.saturating_add(t);
4771            count += 1;
4772        }
4773        count.max(2)
4774    }
4775
4776    /// Register an additional tool (e.g. your own capability).
4777    ///
4778    /// P5-2 (§2.2 C2 "connect invalidates cache prefix"): registering a
4779    /// tool AFTER this agent has already issued a request
4780    /// ([`Self::request_issued`]) changes the tools schema every
4781    /// subsequent request carries — the exact prefix-churn shape C2
4782    /// describes, MCP-sourced or not. Resets [`Self::cache_established`] so
4783    /// the next cache-warmth check (`provider::cache_cold_reason`) doesn't
4784    /// wrongly assume the entry is still warm. A no-op call before the
4785    /// first request (the common case: `attach_mcp` registers tools once at
4786    /// startup, before any turn runs) changes nothing — byte-identical to
4787    /// today.
4788    pub fn register_tool(&mut self, tool: impl crate::tools::Tool + 'static) {
4789        self.registry.register(tool);
4790        if self.requests_issued {
4791            self.cache_established = false;
4792        }
4793    }
4794
4795    /// The current conversation, including the system prompt.
4796    pub fn history(&self) -> &[ChatMessage] {
4797        &self.history
4798    }
4799
4800    /// Send a user message and run the loop until the model produces a final
4801    /// answer (text with no tool calls) or the iteration budget is exhausted.
4802    pub async fn send(&mut self, user_input: impl Into<String>) -> Result<String> {
4803        let expanded = self.expand_prompt_async(&user_input.into()).await;
4804        let msg = ChatMessage::user(expanded);
4805        self.guard_candidate_message(&msg)?;
4806        self.record(&msg)?;
4807        self.history.push(msg);
4808        self.run_loop().await
4809    }
4810
4811    /// BP-4 (catalog:109 "Context-usage introspection", cc§2 `/context`
4812    /// grid, cx§8 `/status` + `get_context_remaining`): the LIVE
4813    /// context-window accounting for this session — the same numbers
4814    /// `resume --dry-run`'s preflight already computes
4815    /// (`tokens::estimate_request_tokens` / `tokens::context_guard`), read
4816    /// out mid-session instead of only before one.
4817    ///
4818    /// Pure: it projects the request view exactly as
4819    /// [`Self::guard_candidate_message`] does (reduction stubs included,
4820    /// cache annotation included) without mutating the reduction log, so
4821    /// asking "how full am I?" can never change what the next request
4822    /// carries.
4823    pub fn context_usage(&self) -> ContextUsage {
4824        let messages = self.projected_view(None);
4825        let tools = self.tool_schemas();
4826        let message_tokens = supercode_runtime::estimate_view_tokens(&messages);
4827        let request_tokens = supercode_runtime::estimate_request_tokens(&messages, &tools);
4828        let limit = self.context_limit.or_else(|| {
4829            crate::provider::model_context_limit(&self.config.model)
4830                .or(Some(crate::provider::UNKNOWN_MODEL_CONTEXT_FLOOR))
4831        });
4832        let projected_tokens = supercode_runtime::with_guard_margin(request_tokens);
4833        let reserve = supercode_runtime::CONTEXT_RESPONSE_RESERVE_TOKENS;
4834        let (fits, remaining_tokens, used_pct) = match limit {
4835            Some(limit) => (
4836                projected_tokens.saturating_add(reserve) <= limit,
4837                limit
4838                    .saturating_sub(reserve)
4839                    .saturating_sub(projected_tokens),
4840                if limit == 0 {
4841                    0
4842                } else {
4843                    (projected_tokens as f64 / limit as f64 * 100.0).round() as u32
4844                },
4845            ),
4846            None => (true, 0, 0),
4847        };
4848        ContextUsage {
4849            model: self.config.model.clone(),
4850            messages: messages.len(),
4851            message_tokens,
4852            tool_count: tools.len(),
4853            tool_schema_tokens: request_tokens.saturating_sub(message_tokens),
4854            request_tokens,
4855            projected_tokens,
4856            response_reserve_tokens: reserve,
4857            context_limit: limit,
4858            remaining_tokens,
4859            used_pct,
4860            fits,
4861        }
4862    }
4863
4864    /// The messages a request would carry right now — the read-only half of
4865    /// [`Self::guard_candidate_message`]/[`Self::build_request_messages`],
4866    /// with `candidate` optionally appended as a not-yet-committed turn.
4867    /// Never mutates `self`.
4868    fn projected_view(&self, candidate: Option<&ChatMessage>) -> Vec<ChatMessage> {
4869        let messages = match &self.reduction_policy {
4870            None => {
4871                let mut messages = self.history.clone();
4872                if let Some(candidate) = candidate {
4873                    messages.push(candidate.clone());
4874                }
4875                messages
4876            }
4877            Some(policy) => {
4878                let has_system = self.history.first().is_some_and(|m| m.role == Role::System);
4879                let mut reducible = self.history[usize::from(has_system)..].to_vec();
4880                if let Some(candidate) = candidate {
4881                    reducible.push(candidate.clone());
4882                }
4883                let mut prepared = policy.clone();
4884                reduce::prepare_read_freshness(&mut prepared, &reducible);
4885                let (view, _) =
4886                    reduce::project_messages(&reducible, &prepared, &self.reduction_log);
4887                let mut messages = Vec::with_capacity(view.len() + usize::from(has_system));
4888                if has_system {
4889                    messages.push(self.history[0].clone());
4890                }
4891                messages.extend(view);
4892                messages
4893            }
4894        };
4895        provider::apply_cache_plan(&messages, self.config.cache_plan, self.imported_prefix_len)
4896    }
4897
4898    /// Refuse an oversized new user turn before it mutates canonical history
4899    /// or an attached sidecar. The in-loop guard remains authoritative for
4900    /// every actual request; this preflight closes the first-request seam
4901    /// where `send*` used to record/push the message before that guard ran.
4902    fn guard_candidate_message(&self, msg: &ChatMessage) -> Result<()> {
4903        let Some(limit) = self.context_limit else {
4904            return Ok(());
4905        };
4906
4907        let messages = self.projected_view(Some(msg));
4908        let tools = self.tool_schemas();
4909        let (fits, projected_tokens) = supercode_runtime::context_guard(&messages, &tools, limit);
4910        if !fits {
4911            return Err(Error::ContextLimitExceeded {
4912                projected_tokens,
4913                reserve_tokens: supercode_runtime::CONTEXT_RESPONSE_RESERVE_TOKENS,
4914                context_limit: limit,
4915                model: self.config.model.clone(),
4916            });
4917        }
4918        Ok(())
4919    }
4920
4921    /// The messages a provider request should carry for the CURRENT turn
4922    /// (A5/A7/A8/A10): with no [`ReductionPolicy`] installed, exactly
4923    /// `self.history.clone()` — byte-identical to every version of this
4924    /// method before reduction landed. With a policy installed, `history[0]`
4925    /// (this agent's own system prompt, never a reduction target) followed by
4926    /// [`reduce::project_messages`]'s projected view of `history[1..]`, fed
4927    /// with `self.reduction_log` so already-applied reductions reproduce
4928    /// verbatim across turns (prefix stability, A5) — the updated log is
4929    /// stored back onto `self` so the NEXT call (this turn, next turn, or a
4930    /// later `send`) sees the same accumulating state. `self.history` itself
4931    /// is never read back into or mutated by this: it stays the full
4932    /// canonical view, in lockstep with the sidecar (A3).
4933    ///
4934    /// When `policy.elide_stale_reads` is set, this re-runs
4935    /// [`reduce::probe_read_freshness`] (the one place A8's disk I/O happens)
4936    /// against `history[1..]` before projecting, so every request sees
4937    /// up-to-date freshness verdicts — `project_messages` itself stays pure.
4938    ///
4939    /// Finally, B7's [`provider::apply_cache_plan`] runs over the assembled
4940    /// view (regardless of whether a [`ReductionPolicy`] is installed) — a
4941    /// pure, cloning annotation step, so this method's `&mut self` mutations
4942    /// above (`self.reduction_log`) are already committed before it runs and
4943    /// its own output is never written back onto `self.history` or the log:
4944    /// purity for B7's cache breakpoints holds independently of A5's.
4945    fn build_request_messages(&mut self) -> Vec<ChatMessage> {
4946        let messages = match self.reduction_policy.clone() {
4947            None => self.history.clone(),
4948            Some(mut policy) => {
4949                reduce::prepare_read_freshness(&mut policy, &self.history[1..]);
4950                // B7 coordination clamp: while `CachePlan::ImportedPrefix` is
4951                // active, A10 turn-clearing must never establish a range
4952                // that dips into the imported prefix (protects the cache
4953                // breakpoint the request build will place there below).
4954                // `imported_prefix_len` counts `history[0]` (this agent's own
4955                // system message) plus the imported messages, but
4956                // `project_messages` only ever sees `history[1..]` — hence
4957                // the `- 1`.
4958                if matches!(self.config.cache_plan, CachePlan::ImportedPrefix) {
4959                    policy.protect_imported_prefix =
4960                        self.imported_prefix_len.map(|n| n.saturating_sub(1));
4961                }
4962                // TR-7 (T20): the one side-call site, run BEFORE
4963                // `project_messages` (which stays pure/I-O-free) — mirrors
4964                // `elide_stale_reads`/`probe_read_freshness` immediately
4965                // above. Only ever does anything when both the policy gate
4966                // AND a summarizer are present; either being absent means
4967                // `cleared_turns_summary` stays `None` and `project_messages`
4968                // renders the deterministic stub, same as before TR-7
4969                // existed.
4970                if policy.summarize_cleared_turns {
4971                    if let Some(summarizer) = self.span_summarizer.as_deref() {
4972                        policy.cleared_turns_summary = reduce::prepare_cleared_turns_summary(
4973                            &self.history[1..],
4974                            &policy,
4975                            &self.reduction_log,
4976                            summarizer,
4977                        );
4978                    }
4979                }
4980                let (view, log) =
4981                    reduce::project_messages(&self.history[1..], &policy, &self.reduction_log);
4982                self.reduction_log = log;
4983                let mut messages = Vec::with_capacity(view.len() + 1);
4984                messages.push(self.history[0].clone());
4985                messages.extend(view);
4986                messages
4987            }
4988        };
4989        // TR-8 (T5): a tool-schema tier change since the last request is a
4990        // cache-bust event under `CachePlan::ImportedPrefix` — the `tools`
4991        // array is part of the cache key alongside `messages`, so flag it by
4992        // skipping this one request's cache annotation rather than claiming
4993        // a prefix hit that won't actually land. Recorded unconditionally
4994        // (even under `CachePlan::Off`) so the signature stays current
4995        // regardless of which plan is active.
4996        let tier_sig = self.schema_tier_signature();
4997        let busted =
4998            provider::tier_change_is_cache_bust(self.last_tool_schema_tier_signature, tier_sig);
4999        self.last_tool_schema_tier_signature = Some(tier_sig);
5000        let effective_cache_plan = if busted {
5001            CachePlan::Off
5002        } else {
5003            self.config.cache_plan
5004        };
5005        // UX-26 (B7-warn): mirror `apply_cache_plan`'s own placement gate
5006        // (`ImportedPrefix` AND a non-zero prefix) to know whether THIS
5007        // request will actually carry a `cache_control` annotation. `busted`
5008        // requests (schema-tier change) and `CachePlan::Off` never annotate,
5009        // so `provider::cache_cold_reason` can never flag them — there was
5010        // nothing to reuse, by construction. `idle_secs` is computed
5011        // whenever a signal exists at all (even before this agent's first
5012        // annotated send — see `Self::last_cache_activity_ms`'s doc comment
5013        // on why the pre-establishment case matters); `cache_established`
5014        // additionally gates the usage-ratio check specifically (see
5015        // `provider::cache_cold_reason`'s doc comment for why those two
5016        // checks need independent gates).
5017        let will_annotate = matches!(effective_cache_plan, CachePlan::ImportedPrefix)
5018            && self.imported_prefix_len.is_some_and(|n| n > 0);
5019        let idle_secs = self
5020            .last_cache_activity_ms
5021            .map(|last| (now_ms() - last).max(0) / 1000);
5022        self.pending_cache_turn = (will_annotate, self.cache_established, idle_secs);
5023        let mut messages =
5024            provider::apply_cache_plan(&messages, effective_cache_plan, self.imported_prefix_len);
5025        // BP-7 (catalog §4a "Goals"): the standing objective, restated at
5026        // the TAIL of the request — after the cache annotation, which sits
5027        // on the PREFIX, so a goal that changes mid-session never busts the
5028        // cached prefix. Request-view only: `history` is untouched, so the
5029        // persisted transcript is exactly the conversation and a translator
5030        // never has to invent a message for a harness-tracked goal.
5031        if let Some(goal) = &self.goal {
5032            messages.push(ChatMessage::system(goal.reminder()));
5033        }
5034        messages
5035    }
5036
5037    /// Run the model/tool loop over the current history until a final answer or
5038    /// the iteration budget is exhausted. (Shared by `send`, `send_with_files`,
5039    /// and `send_with_images`.)
5040    /// P4b (§1.7, pi§3 semantics): pop the next message(s) to deliver from
5041    /// `queue` per `mode` — `All` drains everything and joins it with a
5042    /// blank line, `OneAtATime` pops exactly one. `None` when `queue` is
5043    /// empty (the default state, at zero cost).
5044    fn drain_steer_queue(
5045        queue: &mut std::collections::VecDeque<String>,
5046        mode: SteeringMode,
5047    ) -> Option<String> {
5048        if queue.is_empty() {
5049            return None;
5050        }
5051        match mode {
5052            SteeringMode::All => Some(queue.drain(..).collect::<Vec<_>>().join("\n\n")),
5053            SteeringMode::OneAtATime => queue.pop_front(),
5054        }
5055    }
5056
5057    async fn run_loop(&mut self) -> Result<String> {
5058        let _steer_turn = SteerTurnGuard::new(self.steer_queue.clone());
5059        let mut output_tokens_used: u64 = 0;
5060
5061        // BP-7 (catalog §4a "Turn/budget caps"): the SPEND cap, checked
5062        // before this `send` can issue anything. Unlike
5063        // `max_total_output_tokens` (a per-`send` allowance, unchanged),
5064        // spend accumulates over the agent's whole lifetime — a dollar
5065        // budget that resets on every prompt is not a budget. A cap reached
5066        // MID-loop ends that loop cleanly with a `spend_budget` finish
5067        // marker (below); a cap already exhausted at entry is an error,
5068        // because there is nothing to return.
5069        if let Some(budget) = self.config.max_budget_usd.filter(|b| *b > 0.0) {
5070            if self.total_cost_usd >= budget {
5071                return Err(Error::BudgetExhausted {
5072                    spent_usd: self.total_cost_usd,
5073                    budget_usd: budget,
5074                });
5075            }
5076        }
5077
5078        // P5-9 (§2 module 20, cc's "per-prompt file-history-snapshot"):
5079        // open a fresh checkpoint for THIS turn — `run_loop` is called
5080        // exactly once per `send`/`send_with_files`/`send_with_images`
5081        // call (never recursively for the same turn), so this fires once
5082        // per user prompt, matching the design's per-prompt granularity.
5083        // `self.history.last()` is the user message that call just pushed.
5084        // `None` (`checkpoint_observer` unset, the default) is a no-op —
5085        // zero cost, no disk touched.
5086        if let Some(cp) = &self.checkpoint_observer {
5087            let label = self
5088                .history
5089                .last()
5090                .and_then(|m| m.content.as_deref())
5091                .unwrap_or("")
5092                .to_string();
5093            cp.begin_turn(&label);
5094        }
5095
5096        // BP-4 (catalog:90, cx§2 "re-emitted on change"): once per user
5097        // turn — not per loop iteration — re-derive the environment block
5098        // so a cwd change, an approval/sandbox policy change or a branch
5099        // switch since the last turn reaches the model instead of leaving
5100        // it reading the startup snapshot. A no-op, with no subprocess, for
5101        // every config that doesn't set `core.env_context`.
5102        self.refresh_env_context();
5103
5104        for _ in 0..self.config.max_iterations {
5105            // BP-8 (catalog:156): flush a plan `update_plan` wrote during
5106            // the previous iteration's tool calls. A no-op when
5107            // `todos.persist` is off or the plan did not change.
5108            self.journal_plan_if_changed();
5109            // BP-7: the index of the round-trip this iteration is about to
5110            // make. Captured here because `self.turn_index` advances the
5111            // moment the usage record is written, and every marker in this
5112            // iteration — including the ones written after that point —
5113            // must carry the SAME index, or the marker log would not join
5114            // to the usage log on `turn`.
5115            let round_trip = self.turn_index;
5116            self.maybe_compact();
5117            // P4b (§1.7, pi§3 "steer = after current tool calls"): drain any
5118            // queued mid-turn steering message(s) BEFORE building the next
5119            // request — the top of every loop iteration is exactly "after
5120            // whatever tool calls the previous iteration just ran" (or, on
5121            // the very first iteration, before anything has happened yet,
5122            // which is an equally valid "deliver immediately" reading).
5123            // Empty queue (today's default state) is a no-op.
5124            let (steer_msg, steer_taken) = {
5125                let mut inbox = self
5126                    .steer_queue
5127                    .lock()
5128                    .unwrap_or_else(std::sync::PoisonError::into_inner);
5129                let before = inbox.len();
5130                let drained = inbox.drain(self.config.steering_mode);
5131                let taken = before - inbox.len();
5132                (drained, taken)
5133            };
5134            if let Some(steer_msg) = steer_msg {
5135                // BP-8 (catalog:154): the queue record's other half —
5136                // without it a replayed journal would keep re-delivering an
5137                // input the conversation already consumed.
5138                self.journal_queue_drain(crate::session_journal::QueueKind::Steer, steer_taken);
5139                let msg = ChatMessage::user(steer_msg);
5140                self.record(&msg)?;
5141                self.history.push(msg);
5142            }
5143            // Recomputed every iteration (not hoisted): under `Deferred`
5144            // advertising, a `tool_search` call earlier in this same loop
5145            // activates tools that must be advertised starting with the very
5146            // next request (B6).
5147            let tools = self.tool_schemas();
5148            let messages = self.build_request_messages();
5149
5150            // PARITY-18 D4 — re-check the context guard before EVERY
5151            // request this loop builds, not just the caller's one-shot
5152            // preflight: interactive turns 2+, `/expand all`, and any
5153            // mid-loop tool round-trip that grows `messages` can push a
5154            // barely-passing session over the limit between sends. Only
5155            // armed when a caller has opted in via `set_context_limit`.
5156            // Uses the exact same `tokens::context_guard`
5157            // formula the CLI preflight uses, so the two can never disagree.
5158            if let Some(limit) = self.context_limit {
5159                let (fits, projected_tokens) =
5160                    supercode_runtime::context_guard(&messages, &tools, limit);
5161                if !fits {
5162                    return Err(Error::ContextLimitExceeded {
5163                        projected_tokens,
5164                        reserve_tokens: supercode_runtime::CONTEXT_RESPONSE_RESERVE_TOKENS,
5165                        context_limit: limit,
5166                        model: self.config.model.clone(),
5167                    });
5168                }
5169            }
5170
5171            // BP-7 (catalog §4a "Turn/step bracketing records"): the
5172            // OPENING bracket, written before the request is issued so it
5173            // survives a request that never returns (a cancelled turn keeps
5174            // its `context` marker with no `usage`/`finish` after it).
5175            // Uses `tokens::estimate_request_tokens` — the same estimator
5176            // the context guard above uses, so the two can never disagree.
5177            self.push_turn_marker_at(
5178                round_trip,
5179                crate::turn_record::TurnMarker::Context {
5180                    messages: messages.len(),
5181                    tools: tools.len(),
5182                    estimated_tokens: supercode_runtime::estimate_request_tokens(&messages, &tools),
5183                },
5184            );
5185
5186            let mut req = self.chat_request(messages, tools);
5187
5188            let fallback_hops: Vec<FallbackHop>;
5189            let completion = {
5190                let sink = self.config.event_sink.as_ref();
5191                let on_delta = move |s: &str| {
5192                    if let Some(sink) = sink {
5193                        sink(AgentEvent::TextDelta(s.to_string()));
5194                    }
5195                };
5196                // PARITY-18 D3 — the real send site: flip the flag
5197                // immediately before issuing the request, regardless of
5198                // whether `complete` then succeeds or fails, so
5199                // `request_issued()` truthfully reflects "a live request
5200                // was attempted" rather than "the run reached this line and
5201                // later succeeded."
5202                self.requests_issued = true;
5203                // P4b (§1.1/§3.1 `core.retry`, pi§3 shape): retry-with-
5204                // backoff already lives at the TRANSPORT layer
5205                // (`provider::OpenAiProvider::send_with_retry`, pre-existing
5206                // — connection failures and 5xx responses are retried
5207                // there); `Config.retry_*` (see `Agent::new`) makes that
5208                // EXISTING mechanism config-file-settable instead of
5209                // duplicating a second retry loop here, which would nest
5210                // retries confusingly on top of the transport's own.
5211                // BP-13 (D9 "Failure fallback model chains"): the chain is
5212                // EXECUTED here, not merely resolved. On a failure another
5213                // model could plausibly answer (overload / rate limit /
5214                // unavailability — `is_failover_worthy`), the request is
5215                // re-sent against the next entry of
5216                // `Config::model_fallback`, with routing re-applied for
5217                // that model. A 4xx that is not a rate limit is the caller's
5218                // problem, not the model's, and is never retried elsewhere.
5219                // The transport-level retry above has already run and given
5220                // up by the time a hop is considered.
5221                let (result, hops) = self.complete_with_fallback(&mut req, &on_delta).await;
5222                fallback_hops = hops;
5223                result
5224            };
5225            // BP-7: drained whether the request succeeded or failed, and
5226            // BEFORE the `?` — a request that exhausted its retries and
5227            // then errored is exactly the case a retry record exists for.
5228            for notice in self.retry_log.drain() {
5229                self.emit(AgentEvent::ProviderRetry {
5230                    attempt: notice.attempt,
5231                    delay_ms: notice.delay_ms,
5232                    reason: notice.reason.clone(),
5233                });
5234                self.push_turn_marker_at(
5235                    round_trip,
5236                    crate::turn_record::TurnMarker::Retry {
5237                        attempt: notice.attempt,
5238                        delay_ms: notice.delay_ms,
5239                        reason: notice.reason,
5240                    },
5241                );
5242            }
5243            // The switch the fallback pass performed is a real mid-session
5244            // model change: it moves `Config::model` for every subsequent
5245            // request and is recorded exactly like a user-driven `/model`
5246            // switch (typed record + journal line), never as a silent retry.
5247            for hop in fallback_hops {
5248                self.record_model_change(&hop.from, &hop.to, Some(hop.reason.as_str()));
5249            }
5250            let (mut assistant, usage) = completion?;
5251            // Persist the actual generating model on the message itself.
5252            // A resumed foreign session keeps its original model in
5253            // `SessionMeta`; using only that session-level value on export
5254            // misattributes every Supercode continuation turn to the source
5255            // harness model. Per-message provenance lets native exporters
5256            // preserve the boundary accurately (for example, Claude history
5257            // followed by a GLM continuation).
5258            assistant
5259                .metadata
5260                .insert("model".to_string(), self.config.model.clone());
5261            output_tokens_used += usage.completion_tokens;
5262            self.total_output_tokens += usage.completion_tokens;
5263
5264            // UX-26 (B7-warn): consult the verdict computed at build time
5265            // (before this request was sent) now that `usage` — the only
5266            // piece that couldn't be known pre-send — is in hand. Gated on
5267            // `Config::cache_warnings` (default on; `--no-cache-warnings` /
5268            // `SUPERCODE_CACHE_WARNINGS=0` at the CLI layer, dev/03) so this
5269            // stays a zero-behavior-change no-op for every caller that
5270            // hasn't opted into `CachePlan::ImportedPrefix` in the first
5271            // place (`pending_cache_turn.0` is `false` whenever
5272            // `CachePlan::Off`, so the predicate always returns `None` then
5273            // regardless of this flag).
5274            let (will_annotate, cache_established, idle_secs) = self.pending_cache_turn;
5275            // UX-26 T2 (accuracy fold-in): `CacheColdReason::message` asserts
5276            // Anthropic-specific facts (a fixed 5-minute ephemeral TTL, and
5277            // cache-read-ratio semantics that assume Anthropic's exact-count
5278            // billing) that are only true for Anthropic-family models. This
5279            // is a WARNING-only gate, deliberately not folded into
5280            // `will_annotate`/the breakpoint-placement gate above: whether a
5281            // `cache_control` breakpoint is safe/inert to send to a
5282            // non-Anthropic model through OpenRouter is a separate cache-
5283            // behavior question this ticket doesn't touch (see
5284            // `.volter/tracker/markdown/UX-26.md`'s T2 note) — narrowing only
5285            // the warning keeps this fix scoped to warning ACCURACY, with
5286            // zero change to what gets sent on the wire.
5287            let warning_applies_to_this_model =
5288                provider::is_anthropic_family_model(&self.config.model);
5289            if self.config.cache_warnings && warning_applies_to_this_model {
5290                if let Some(reason) =
5291                    provider::cache_cold_reason(will_annotate, cache_established, idle_secs, &usage)
5292                {
5293                    self.emit(AgentEvent::CacheWarning {
5294                        message: reason.message(),
5295                    });
5296                }
5297            }
5298            // Refresh the activity clock / establish-once flag for the NEXT
5299            // turn's comparison, but only when THIS request actually carried
5300            // the annotation — an unannotated (busted/Off) request neither
5301            // warms nor cools a cache entry it never touched.
5302            if will_annotate {
5303                self.last_cache_activity_ms = Some(now_ms());
5304                self.cache_established = true;
5305            }
5306
5307            // UX-23: emitted before `TurnCompleted` so a `--trace`/
5308            // `stream-json` consumer sees "this round-trip cost N tokens"
5309            // land right alongside the round-trip it describes, rather than
5310            // needing to correlate it with a later event.
5311            self.emit(AgentEvent::Usage(usage.clone()));
5312            self.emit(AgentEvent::TurnCompleted);
5313            // P4b (§1.6, catalog §4a "persisted per-turn usage records"):
5314            // EventSink already streamed `Usage` above — this durably
5315            // accumulates the same data as a typed record (see
5316            // `Self::usage_records`/`Self::save_usage_log`), never a lossy
5317            // display-only channel.
5318            // BP-7 (catalog §4a "Per-turn cost/usage accounting"): the
5319            // record now carries the round-trip's DOLLAR cost too — the
5320            // half the row's semantics name alongside tokens — whenever
5321            // this build can price the model.
5322            // BP-13 (D9 "Model-served-vs-requested provenance"): the record
5323            // now carries BOTH sides — the model this agent asked for and,
5324            // when the provider reported one, the model that actually
5325            // answered. They can genuinely differ (a gateway aliasing a
5326            // name to a dated snapshot, a fallback hop, a routed tier), and
5327            // a record that can only ever state the request cannot show it.
5328            let served = assistant
5329                .metadata
5330                .get(crate::provider::SERVED_MODEL_KEY)
5331                .cloned();
5332            let record = crate::usage_log::UsageRecord::from_usage(
5333                self.turn_index,
5334                &self.config.model,
5335                &usage,
5336                now_ms(),
5337            )
5338            .priced(self.model_price)
5339            .with_served_model(served);
5340            self.total_cost_usd += record.cost_usd.unwrap_or(0.0);
5341            // BP-7: the round-trip's usage bracket, written from the same
5342            // point as the usage record so the two logs never disagree.
5343            self.push_turn_marker_at(
5344                round_trip,
5345                crate::turn_record::TurnMarker::Usage {
5346                    prompt_tokens: record.prompt_tokens,
5347                    completion_tokens: record.completion_tokens,
5348                    total_tokens: record.total_tokens,
5349                    cached_tokens: record.cached_tokens,
5350                    cost_usd: record.cost_usd,
5351                },
5352            );
5353            self.journal_usage(&record);
5354            self.usage_log.push(record);
5355            self.turn_index += 1;
5356            self.record(&assistant)?;
5357            self.history.push(assistant.clone());
5358
5359            let calls = assistant.tool_calls().to_vec();
5360            if calls.is_empty() {
5361                // Close steering acceptance under the same lock as the last
5362                // drain. A message accepted before this boundary extends the
5363                // current turn; anything later is rejected by the SDK and
5364                // can never leak into a future turn.
5365                let (steer_msg, steer_taken) = {
5366                    let mut inbox = self
5367                        .steer_queue
5368                        .lock()
5369                        .unwrap_or_else(std::sync::PoisonError::into_inner);
5370                    let before = inbox.len();
5371                    let drained = inbox.drain_or_close(self.config.steering_mode);
5372                    let taken = before - inbox.len();
5373                    (drained, taken)
5374                };
5375                if let Some(steer_msg) = steer_msg {
5376                    self.journal_queue_drain(crate::session_journal::QueueKind::Steer, steer_taken);
5377                    let msg = ChatMessage::user(steer_msg);
5378                    self.record(&msg)?;
5379                    self.history.push(msg);
5380                    continue;
5381                }
5382                // P4b (§1.7, pi§3 "follow-up = at idle"): a queued follow-up
5383                // message takes priority over the stop-gate — it's more
5384                // input to answer, not a veto of an answer already given.
5385                let follow_up_before = self.follow_up_queue.len();
5386                if let Some(follow_up_msg) =
5387                    Self::drain_steer_queue(&mut self.follow_up_queue, self.config.follow_up_mode)
5388                {
5389                    self.journal_queue_drain(
5390                        crate::session_journal::QueueKind::FollowUp,
5391                        follow_up_before - self.follow_up_queue.len(),
5392                    );
5393                    let msg = ChatMessage::user(follow_up_msg);
5394                    self.record(&msg)?;
5395                    self.history.push(msg);
5396                    continue;
5397                }
5398                // P4b (§1.9/§3.1 `[core] stop_gate`, D3 "stop/completion
5399                // gating"): consulted exactly once per iteration that would
5400                // otherwise return — computed into an owned `Option<String>`
5401                // so the immutable borrow of `self.config.stop_gate` ends
5402                // before the `self.record`/`self.history.push` calls below
5403                // need `&mut self`.
5404                let final_content = assistant.content.clone().unwrap_or_default();
5405                let veto_reason: Option<String> = self
5406                    .config
5407                    .stop_gate
5408                    .as_ref()
5409                    .and_then(|gate| gate(&final_content));
5410                if let Some(reason) = veto_reason {
5411                    let msg = ChatMessage::user(reason);
5412                    self.record(&msg)?;
5413                    self.history.push(msg);
5414                    continue;
5415                }
5416                // BP-8 (catalog:156): the last iteration's tool calls are
5417                // the ones the top-of-loop flush above never sees.
5418                self.journal_plan_if_changed();
5419                self.push_turn_marker_at(
5420                    round_trip,
5421                    crate::turn_record::TurnMarker::Finish {
5422                        reason: crate::turn_record::FinishReason::EndTurn,
5423                    },
5424                );
5425                return Ok(assistant.content.unwrap_or_default());
5426            }
5427
5428            // BP-7: this round-trip ended by asking for tool calls; the
5429            // loop continues. The budget arms below mark the LOOP's end
5430            // separately when one of them stops it here.
5431            self.push_turn_marker_at(
5432                round_trip,
5433                crate::turn_record::TurnMarker::Finish {
5434                    reason: crate::turn_record::FinishReason::ToolCalls,
5435                },
5436            );
5437
5438            // Output-token budget (output only — input tokens are not counted,
5439            // so this does not bound cost): stop spawning further model turns
5440            // once the cumulative output-token budget for this `send` is
5441            // exhausted.
5442            if let Some(budget) = self.config.max_total_output_tokens {
5443                if output_tokens_used >= budget {
5444                    // The assistant turn we just pushed carries unanswered
5445                    // tool_calls. Leaving them dangling yields an invalid
5446                    // history (assistant tool_calls with no tool results) that
5447                    // the provider rejects on the next `send`/resume. Emit
5448                    // synthetic results so the transcript stays well-formed.
5449                    for call in &calls {
5450                        let msg = ChatMessage::tool_result(
5451                            call.id.clone(),
5452                            call.function.name.clone(),
5453                            "[skipped: output token budget reached]".to_string(),
5454                        );
5455                        self.record(&msg)?;
5456                        self.history.push(msg);
5457                    }
5458                    self.push_turn_marker_at(
5459                        round_trip,
5460                        crate::turn_record::TurnMarker::Finish {
5461                            reason: crate::turn_record::FinishReason::OutputTokenBudget,
5462                        },
5463                    );
5464                    return Ok(assistant.content.clone().unwrap_or_default());
5465                }
5466            }
5467
5468            // BP-7 (catalog §4a "Turn/budget caps"): the SPEND and STEP
5469            // caps, at the same point and with the same shape as the
5470            // output-token cap above — checked before this turn's tool
5471            // calls run, with synthetic results so the transcript stays
5472            // well-formed for a resume.
5473            let spend_exhausted = self
5474                .config
5475                .max_budget_usd
5476                .is_some_and(|b| b > 0.0 && self.total_cost_usd >= b);
5477            let steps_exhausted = self
5478                .config
5479                .max_steps
5480                .is_some_and(|n| n > 0 && self.total_steps + calls.len() > n);
5481            if spend_exhausted || steps_exhausted {
5482                let (label, reason) = if spend_exhausted {
5483                    (
5484                        "[skipped: spend budget reached]",
5485                        crate::turn_record::FinishReason::SpendBudget,
5486                    )
5487                } else {
5488                    (
5489                        "[skipped: step budget reached]",
5490                        crate::turn_record::FinishReason::StepBudget,
5491                    )
5492                };
5493                for call in &calls {
5494                    let msg = ChatMessage::tool_result(
5495                        call.id.clone(),
5496                        call.function.name.clone(),
5497                        label.to_string(),
5498                    );
5499                    self.record(&msg)?;
5500                    self.history.push(msg);
5501                }
5502                self.push_turn_marker_at(
5503                    round_trip,
5504                    crate::turn_record::TurnMarker::Finish { reason },
5505                );
5506                return Ok(assistant.content.clone().unwrap_or_default());
5507            }
5508            self.total_steps += calls.len();
5509
5510            // P4e (§3.1 `core.parallel_tool_calls`, catalog:59): off (the
5511            // default) or a single call takes the EXACT pre-P4e sequential
5512            // path below, byte-identical. Only `true` with 2+ calls in this
5513            // turn takes `Self::run_tools_concurrently` — see its doc
5514            // comment for exactly what does and doesn't run concurrently.
5515            if self.config.parallel_tool_calls && calls.len() > 1 {
5516                for call in &calls {
5517                    self.emit(AgentEvent::tool_started(call));
5518                }
5519                let results = self.run_tools_concurrently(&calls).await;
5520                for (call, (output, is_error)) in calls.iter().zip(results) {
5521                    self.emit(AgentEvent::ToolCallCompleted {
5522                        id: call.id.clone(),
5523                        name: call.function.name.clone(),
5524                        output: output.clone(),
5525                        is_error,
5526                    });
5527                    self.apply_tool_result(call, output, is_error)?;
5528                }
5529            } else {
5530                for call in &calls {
5531                    self.emit(AgentEvent::tool_started(call));
5532                    let (output, is_error) = self.run_tool(call).await;
5533                    self.emit(AgentEvent::ToolCallCompleted {
5534                        id: call.id.clone(),
5535                        name: call.function.name.clone(),
5536                        output: output.clone(),
5537                        is_error,
5538                    });
5539                    self.apply_tool_result(call, output, is_error)?;
5540                }
5541            }
5542            // BP-3 (catalog row "Context-budget tools"): a `new_context`
5543            // call parks its request on the shared budget; this is where
5544            // the agent — the one owner of `history` — applies it, so the
5545            // NEXT request built by this loop is already the fresh window.
5546            // No parked request (every session that never calls the tool)
5547            // is a single `Option` check.
5548            self.apply_pending_new_context();
5549        }
5550
5551        self.push_turn_marker_at(
5552            self.turn_index.saturating_sub(1),
5553            crate::turn_record::TurnMarker::Finish {
5554                reason: crate::turn_record::FinishReason::MaxIterations,
5555            },
5556        );
5557        Err(Error::MaxIterations(self.config.max_iterations))
5558    }
5559
5560    /// BP-3: apply a parked [`crate::tools::NewContextRequest`], if any.
5561    ///
5562    /// The rewrite itself is BP-4's [`Self::new_context`] — the SAME
5563    /// mechanism the operator's `/handoff` runs, so the model's door and the
5564    /// human's door can never drift into two different notions of "a fresh
5565    /// window". This function is only the hand-off point between the tool
5566    /// that asked and the agent that owns `history`.
5567    fn apply_pending_new_context(&mut self) {
5568        let Some(request) = self.ctx.context_budget.take_new_context() else {
5569            return;
5570        };
5571        self.new_context(&request.objective, request.keep_recent);
5572    }
5573
5574    /// The exact post-execution handling every tool result gets, regardless
5575    /// of whether it was produced by the sequential loop or
5576    /// [`Self::run_tools_concurrently`] — factored out of `Self::run_loop`'s
5577    /// tool-dispatch section (P4e) so both paths share one copy: multimodal
5578    /// image-marker detection, A7 output capping (gated exactly as before),
5579    /// TR-10 error stamping, and the `record`/`history` append. Always
5580    /// called in ORIGINAL call order, one call at a time, so the lossless
5581    /// sidecar's append-order invariant (S1.13) holds regardless of which
5582    /// dispatch path produced the result.
5583    fn apply_tool_result(
5584        &mut self,
5585        call: &supercode_interchange::ToolCall,
5586        output: String,
5587        is_error: bool,
5588    ) -> Result<()> {
5589        // P4c (§1.2 `core.tools.read_file.multimodal` / `view_image`):
5590        // a successful tool result carrying the image-data-URL
5591        // marker becomes a `content_parts` image block instead of
5592        // plain text — checked BEFORE `cap_tool_output` (a data URL
5593        // is not meaningfully "capped" by a byte-length text notice)
5594        // and recorded identically on both the full and history
5595        // copies, mirroring `ImageRedacted`'s "images are their own
5596        // axis, orthogonal to A7 text truncation" treatment
5597        // (reduce.rs). An ERRORED call never carries the marker (a
5598        // tool only emits it on success), so `is_error` is not
5599        // re-checked here.
5600        if let Some(data_url) = output.strip_prefix(crate::tools::MULTIMODAL_IMAGE_MARKER) {
5601            let notice = format!("[{}: image content attached below]", call.function.name);
5602            let full_result = ChatMessage::tool_result_with_image(
5603                call.id.clone(),
5604                call.function.name.clone(),
5605                notice.clone(),
5606                data_url.to_string(),
5607            );
5608            let hist_result = ChatMessage::tool_result_with_image(
5609                call.id.clone(),
5610                call.function.name.clone(),
5611                notice,
5612                data_url.to_string(),
5613            );
5614            self.record(&full_result)?;
5615            self.history.push(hist_result);
5616            return Ok(());
5617        }
5618        // Record the FULL output before capping (A3): what the
5619        // sidecar keeps must never be the already-lossy, truncated
5620        // copy (#8/#40) — `history` alone governs what shrinks.
5621        let mut full_result =
5622            ChatMessage::tool_result(call.id.clone(), call.function.name.clone(), output.clone());
5623        // D6/A7 supersession gate (TR-12 land-blocker fix): `history`
5624        // is the exact slice `reduce::project_messages` mints A7/A10
5625        // reduction hashes from (`Self::build_request_messages`
5626        // below). Capping it here — as this unconditionally used to
5627        // do — would silently shrink the bytes those hashes cover, so
5628        // a hash minted now could never recompute the same way once
5629        // the sidecar is reloaded from disk later (`verify_log`/
5630        // `invert`, offline). Gate `cap_tool_output` off in exactly
5631        // the combination where reductions can be minted over
5632        // `history` AND the full bytes are durably retained: a
5633        // recorder AND a `ReductionPolicy` both installed. A7 then
5634        // owns tool-output bounding, reversibly, at projection time
5635        // (SPEC.md D6/A7) — `history`/the sidecar keep everything,
5636        // only the request view shrinks. With a policy but no
5637        // recorder (constructible via `set_reduction_policy` alone),
5638        // nothing durable backs the full bytes, so capping stays on —
5639        // the same honest-labeling spirit as `cap_tool_output`'s own
5640        // retention branch below, just applied at the gate instead of
5641        // the notice text. With no policy at all, this is untouched:
5642        // today's byte-identical legacy cap.
5643        let for_history = if self.recorder.is_some() && self.reduction_policy.is_some() {
5644            output
5645        } else {
5646            self.cap_tool_output(output)
5647        };
5648        let mut hist_result =
5649            ChatMessage::tool_result(call.id.clone(), call.function.name.clone(), for_history);
5650        if is_error {
5651            // TR-10: the reduction layer's success/failure boundary
5652            // (`ReductionKind::ToolInputElided` must never target an
5653            // errored call — TR-6's territory) has no other
5654            // structural signal on `ChatMessage`; stamp both the
5655            // recorded copy (so it survives a sidecar round-trip via
5656            // `NativeTurn`) and the live-history copy (so an
5657            // in-process `project_messages` sees it immediately).
5658            reduce::mark_tool_error(&mut full_result);
5659            reduce::mark_tool_error(&mut hist_result);
5660        }
5661        self.record(&full_result)?;
5662        self.history.push(hist_result);
5663        Ok(())
5664    }
5665
5666    /// Truncate an oversized tool result so a single runaway command can't blow
5667    /// up the context window. Cuts on a char boundary and appends a notice.
5668    fn cap_tool_output(&self, output: String) -> String {
5669        let Some(max) = self.config.max_tool_output_bytes else {
5670            return output;
5671        };
5672        if max == 0 || output.len() <= max {
5673            return output;
5674        }
5675        // Find the largest char boundary <= max.
5676        let mut end = max;
5677        while end > 0 && !output.is_char_boundary(end) {
5678            end -= 1;
5679        }
5680        let total = output.len();
5681        let mut s = output[..end].to_string();
5682        // BP-2 (catalog:58, `core.tool_output_spill`): write the full bytes
5683        // to a per-session file the model can read back. Off (the default)
5684        // leaves the notice byte-identical to before.
5685        let spill = if self.config.tool_output_spill {
5686            self.spill_tool_output(&output)
5687        } else {
5688            None
5689        };
5690        // Honest retention labeling (D6, B10-AC4): only claim the sidecar has
5691        // the full output when a recorder is actually installed — or, BP-2,
5692        // that the spill file has it when one was actually written.
5693        let retention = if self.recorder.is_some() {
5694            "full output in session sidecar"
5695        } else if spill.is_some() {
5696            "full output on disk"
5697        } else {
5698            "full output not retained"
5699        };
5700        let recovery = match &spill {
5701            // The door is named in the notice, so it works WITHOUT
5702            // `capabilities.reduction`: under a preset with a read tool
5703            // that is `read_file`; under a shell-only preset (cx-parity,
5704            // whose whole read pathway is the shell) it is `cat`.
5705            Some(path) => {
5706                let door = if self.registry.get("read_file").is_some() {
5707                    "read it with `read_file`"
5708                } else {
5709                    "read it with `cat`"
5710                };
5711                format!("; full output spilled to {} — {door}", path.display())
5712            }
5713            None => String::new(),
5714        };
5715        s.push_str(&format!(
5716            "{CAP_NOTICE_MARKER}{total} bytes total, showing first {end}; {retention}{recovery}]"
5717        ));
5718        s
5719    }
5720
5721    /// BP-2 (catalog:58 "Oversized output truncated; full content kept
5722    /// reachable"): write `full` to this session's spill directory and
5723    /// return the path, or `None` if it could not be written (a spill is a
5724    /// recovery convenience — it must never fail the tool call).
5725    ///
5726    /// The file is named by content hash, so the same output spilled twice
5727    /// costs one file and a re-run of an identical command reuses it.
5728    fn spill_tool_output(&self, full: &str) -> Option<std::path::PathBuf> {
5729        let dir = self.spill_dir();
5730        std::fs::create_dir_all(&dir).ok()?;
5731        let digest = blake3::hash(full.as_bytes()).to_hex();
5732        let path = dir.join(format!("tool-output-{}.txt", &digest[..16]));
5733        if !path.exists() {
5734            std::fs::write(&path, full).ok()?;
5735        }
5736        Some(path)
5737    }
5738
5739    /// BP-2: where this agent's spilled outputs live — beside the session
5740    /// sidecar when one is recording (per-SESSION, the same identity the
5741    /// sidecar has), else a per-PROCESS temp directory, which is as
5742    /// specific as an agent with no sidecar can honestly be.
5743    fn spill_dir(&self) -> std::path::PathBuf {
5744        if let Some(recorder) = &self.recorder {
5745            let path = recorder.path();
5746            if let (Some(parent), Some(stem)) = (path.parent(), path.file_stem()) {
5747                return parent.join(format!("{}.spill", stem.to_string_lossy()));
5748            }
5749        }
5750        std::env::temp_dir().join(format!("supercode-spill-{}", std::process::id()))
5751    }
5752
5753    /// P5-3 note on the signature: written as a plain fn returning an
5754    /// explicitly boxed future (`Pin<Box<dyn Future + Send>>`) rather than
5755    /// as `async fn`. `spawn_subagent` makes this function genuinely
5756    /// recursive at the TYPE level: `run_tool` -> `run_spawn_subagent` ->
5757    /// (a child) `Agent::send` -> `run_loop` -> `run_tool` again — an
5758    /// `async fn`'s return type is an anonymous, compiler-inferred
5759    /// self-referential state machine, and inferring one that embeds
5760    /// itself (even indirectly, through several other functions) is a
5761    /// compile error (an infinitely-sized/cyclic opaque type). Declaring
5762    /// `run_tool`'s return type EXPLICITLY as a boxed trait object breaks
5763    /// the cycle: every other function on the call graph now embeds a
5764    /// concrete, already-known type here instead of one the compiler would
5765    /// otherwise need to (cyclically) infer. Callers are unaffected —
5766    /// `self.run_tool(call).await` reads identically either way.
5767    fn run_tool<'a>(
5768        &'a mut self,
5769        call: &'a supercode_interchange::ToolCall,
5770    ) -> std::pin::Pin<Box<dyn std::future::Future<Output = (String, bool)> + Send + 'a>> {
5771        Box::pin(async move {
5772            let translated_builtin = if self.config.claude_runtime_tools_enabled {
5773                match self.translate_claude_builtin_call(call) {
5774                    Ok(translated) => translated,
5775                    Err(error) => return (format!("Error: {error}"), true),
5776                }
5777            } else {
5778                None
5779            };
5780            let call = translated_builtin.as_ref().unwrap_or(call);
5781            if self.config.claude_runtime_tools_enabled
5782                && matches!(
5783                    call.function.name.as_str(),
5784                    CLAUDE_CRON_CREATE
5785                        | CLAUDE_CRON_DELETE
5786                        | CLAUDE_CRON_LIST
5787                        | CLAUDE_SCHEDULE_WAKEUP
5788                )
5789            {
5790                return self.run_claude_runtime_tool(call);
5791            }
5792            // P5-3: `spawn_subagent`/`subagent_status` need full async
5793            // `&mut self` access (running a child agent's loop, or
5794            // awaiting an already-finished background `JoinHandle`) —
5795            // `prepare_tool_call` is purely synchronous, so these are
5796            // intercepted HERE, one level above it, rather than inside it
5797            // like `TOOL_SEARCH`/`EXPAND_REDUCTION`/`SIDECAR_SEARCH`.
5798            if call.function.name == CLAUDE_AGENT && self.config.subagents_claude_agent_alias {
5799                return match self.translate_claude_agent_call(call) {
5800                    Ok(translated) => self.run_spawn_subagent(&translated).await,
5801                    Err(error) => (format!("Error: {error}"), true),
5802                };
5803            }
5804            if call.function.name == SPAWN_SUBAGENT {
5805                return self.run_spawn_subagent(call).await;
5806            }
5807            // P5-3 safety-hardening fix (Fable-5 review, LOW "wrong error
5808            // when disabled"): gated on `subagents_enabled`, matching
5809            // `run_spawn_subagent`'s own already-correct disabled behavior
5810            // (that one gates INTERNALLY, at its own top; this one gates
5811            // HERE, at the interception point, because unlike
5812            // `spawn_subagent` it has no other reason to run any logic at
5813            // all when subagents are off). When disabled, a hallucinated
5814            // `subagent_status` call must NOT be intercepted — it falls
5815            // through to `prepare_tool_call`'s normal unknown-tool path
5816            // below, which returns `Error::UnknownTool("subagent_status")`,
5817            // byte-identical to the pre-P5-3 (and disabled-spawn_subagent)
5818            // error text — never `Error::SubagentNotFound`'s "unknown
5819            // subagent id" text, which would wrongly imply subagents are on
5820            // but this particular id is bogus.
5821            if call.function.name == SUBAGENT_STATUS && self.config.subagents_enabled {
5822                return self.run_subagent_status(call).await;
5823            }
5824            // BP-7: same interception shape and same `subagents_enabled`
5825            // gate as `SUBAGENT_STATUS` above — when the module is off a
5826            // hallucinated call falls through to the ordinary unknown-tool
5827            // error rather than a misleading "unknown subagent id".
5828            if call.function.name == SEND_MESSAGE && self.config.subagents_enabled {
5829                return self.run_send_message(call).await;
5830            }
5831            if call.function.name == SUBAGENT_RESUME && self.config.subagents_enabled {
5832                return self.run_subagent_resume(call).await;
5833            }
5834            match self.prepare_tool_call(call) {
5835                PreparedCall::Done(result) => result,
5836                PreparedCall::Ready { name, args } => {
5837                    // `prepare_tool_call` already confirmed the registry has
5838                    // this tool.
5839                    let tool = self.registry.get(&name).expect("prepared as Ready");
5840                    let (output, is_error) = match tool.execute(args, &self.ctx).await {
5841                        Ok(out) => (out, false),
5842                        Err(e) => (format!("Error: {e}"), true),
5843                    };
5844                    if let Some(hook) = &self.config.post_tool_hook {
5845                        hook(&name, &output, is_error);
5846                    }
5847                    (output, is_error)
5848                }
5849            }
5850        })
5851    }
5852
5853    /// P4e (§3.1 `core.parallel_tool_calls`, catalog:59): the SYNCHRONOUS
5854    /// half of dispatching one tool call — everything `Self::run_tool` did
5855    /// BEFORE its single `tool.execute(...).await`, factored out so
5856    /// [`Self::run_tools_concurrently`] can run these cheap, stateful,
5857    /// `&mut self` checks (agent intrinsics, unknown-tool, approval,
5858    /// doom-loop, pre-tool-hook) SEQUENTIALLY and in ORIGINAL call order —
5859    /// exactly as `run_tool` always has — before handing the remaining
5860    /// calls' `execute()` futures to `join_all`. `Self::run_tool` itself is
5861    /// now a thin wrapper over this (a pure refactor: byte-identical
5862    /// observable behavior, verified by the existing test suite).
5863    fn prepare_tool_call(&mut self, call: &supercode_interchange::ToolCall) -> PreparedCall {
5864        let name = &call.function.name;
5865        // BP-3 (catalog row "Context-budget tools"): hand the model's
5866        // `get_context_remaining` the agent's OWN accounting — BP-4's
5867        // [`Self::context_usage`], the same struct `/context` prints and
5868        // the same estimates the context guard enforces, so what the model
5869        // reads and what refuses an oversized turn can never disagree.
5870        // Computed at the moment the question is asked (the freshest
5871        // possible view) and ONLY then: `context_usage` projects the whole
5872        // request view, which is not a cost to pay on unrelated calls.
5873        if name == crate::tools::GET_CONTEXT_REMAINING {
5874            if let Ok(usage) = serde_json::to_value(self.context_usage()) {
5875                self.ctx.context_budget.publish(usage);
5876            }
5877        }
5878        if name == TOOL_SEARCH {
5879            // Agent intrinsic (B6): intercepted before registry lookup, since
5880            // `Tool::execute` has no access to the registry or `activated_tools`.
5881            return PreparedCall::Done(self.run_tool_search(call));
5882        }
5883        if name == EXPAND_REDUCTION {
5884            // Agent intrinsic (T12/TR-1): intercepted before registry lookup,
5885            // same reason — resolves against `self.reduction_log`/`self.history`,
5886            // which `Tool::execute` has no access to.
5887            return PreparedCall::Done(self.run_expand_reduction(call));
5888        }
5889        if name == SIDECAR_SEARCH {
5890            return PreparedCall::Done(self.run_sidecar_search(call));
5891        }
5892        // P5-6 (§2 module 4 `tools.background`): gated at the interception
5893        // point itself (not internally, at each method's own top) —
5894        // mirroring `SUBAGENT_STATUS`'s own fix (Fable-5 review, LOW "wrong
5895        // error when disabled"): a hallucinated call when the module is off
5896        // must fall through to the plain `Error::UnknownTool` path below,
5897        // never a background-specific error that would wrongly imply the
5898        // module is on. Unlike `SPAWN_SUBAGENT`/`SUBAGENT_STATUS`, none of
5899        // these four need async `&mut self` access (spawning a process,
5900        // `Child::try_wait`, and `Child::start_kill` are all synchronous),
5901        // so they're intercepted here in `prepare_tool_call` rather than in
5902        // `Self::run_tool`.
5903        if self.config.tools_background_enabled {
5904            if name == BACKGROUND_EXEC {
5905                return PreparedCall::Done(self.run_background_exec(call));
5906            }
5907            if name == BACKGROUND_STATUS {
5908                return PreparedCall::Done(self.run_background_status(call));
5909            }
5910            if name == BACKGROUND_LIST {
5911                return PreparedCall::Done(self.run_background_list(call));
5912            }
5913            if name == BACKGROUND_KILL {
5914                return PreparedCall::Done(self.run_background_kill(call));
5915            }
5916        }
5917        if self.registry.get(name).is_none() {
5918            let err = Error::UnknownTool(name.clone());
5919            return PreparedCall::Done((format!("Error: {err}"), true));
5920        }
5921
5922        // P5-1 (§2 modules 10-11, integration point named in
5923        // COMPOSABLE-HARNESS-DESIGN.md's activation set): the permissions
5924        // ENGINE governs the gate when `capabilities.permissions.enabled`
5925        // is on; every other config resolves this to `false`
5926        // (`Config::default`), which takes the `else` branch below —
5927        // the EXACT pre-P5-1 code, untouched, so the default posture
5928        // (approval=never/sandbox=none) and every existing test's observed
5929        // behavior is byte-for-byte unchanged.
5930        if self.config.permissions_enabled {
5931            // The engine needs the command/path TEXT the legacy tool-name-
5932            // only gate below never looked at, so args must be parsed
5933            // BEFORE the gate here (not after, like the legacy branch).
5934            let args = match call.function.parsed_arguments() {
5935                Ok(v) => v,
5936                Err(e) => {
5937                    let err = Error::InvalidArguments {
5938                        tool: name.clone(),
5939                        message: e.to_string(),
5940                    };
5941                    return PreparedCall::Done((format!("Error: {err}"), true));
5942                }
5943            };
5944            // P4c doom-loop breaker, unchanged, still before any gate.
5945            if let Some(reason) = self.check_doom_loop(name, &args) {
5946                return PreparedCall::Done((format!("Error: {reason}"), true));
5947            }
5948            // BP-10: the hook runs BEFORE the engine on this path, so its
5949            // rewrite is what the rules see and its allow/ask/deny is a
5950            // tier inside them — see `run_pre_tool_hook`.
5951            let (args, hook_decision) = match self.run_pre_tool_hook(name, args) {
5952                Ok(pair) => pair,
5953                Err(done) => return done,
5954            };
5955            if let Some(reason) = self.permissions_gate_denial(name, &args, hook_decision) {
5956                return PreparedCall::Done((format!("Error: {reason}"), true));
5957            }
5958            PreparedCall::Ready {
5959                name: name.clone(),
5960                args,
5961            }
5962        } else {
5963            // ---- pre-P5-1 gate, byte-for-byte unchanged ----
5964            // Approval gate: if the policy requires it, consult the handler
5965            // (absent handler denies, so an OnRequest/Untrusted policy is
5966            // fail-closed).
5967            if self.config.needs_approval(name) {
5968                let approved = self
5969                    .config
5970                    .approval_handler
5971                    .as_ref()
5972                    .map(|h| h(call))
5973                    .unwrap_or(false);
5974                if !approved {
5975                    return PreparedCall::Done((
5976                        format!("Error: tool `{name}` was not approved for execution"),
5977                        true,
5978                    ));
5979                }
5980            }
5981            let args = match call.function.parsed_arguments() {
5982                Ok(v) => v,
5983                Err(e) => {
5984                    let err = Error::InvalidArguments {
5985                        tool: name.clone(),
5986                        message: e.to_string(),
5987                    };
5988                    return PreparedCall::Done((format!("Error: {err}"), true));
5989                }
5990            };
5991            self.finish_prepare(name.clone(), args)
5992        }
5993    }
5994
5995    /// P5-1: the shared tail of [`Self::prepare_tool_call`] — doom-loop
5996    /// check, pre-tool hook, `Ready` construction — factored out so both
5997    /// the legacy gate and the new permissions-engine gate run the exact
5998    /// same downstream checks in the exact same order (§5.3 risk 1: the
5999    /// permissions engine changes WHO gets to run, never what happens once
6000    /// they're approved).
6001    fn finish_prepare(&mut self, name: String, args: serde_json::Value) -> PreparedCall {
6002        // P4c (§5.2 P4 "doom-loop breaker", oc UNIQUE `doom_loop` row,
6003        // catalog D3): a default, always-available veto point distinct from
6004        // `Config.pre_tool_hook` (a single user-installable slot — the
6005        // breaker must coexist with a caller's own hook, not compete for the
6006        // one slot). `None`/`Some(0|1)` is a no-op — byte-identical to
6007        // today (no repetition tracking, no call is ever refused on this
6008        // basis).
6009        if let Some(reason) = self.check_doom_loop(&name, &args) {
6010            return PreparedCall::Done((format!("Error: {reason}"), true));
6011        }
6012        // Pre-tool hook may block the call. BP-10: the pre-P5-1 gate has
6013        // no permissions engine for a hook's `Allow`/`Ask` to be a tier
6014        // OF, so only the deny half can mean anything here — an
6015        // `updated_args` rewrite still applies (it is a property of the
6016        // call, not of any gate), and `Allow`/`Ask` are no-ops, exactly
6017        // as `None` was before BP-10.
6018        let mut args = args;
6019        if let Some(hook) = &self.config.pre_tool_hook {
6020            let outcome = hook(&name, &args);
6021            if let Some(rewritten) = outcome.updated_args {
6022                args = rewritten;
6023            }
6024            if outcome.decision == crate::config::HookDecision::Deny {
6025                let reason = outcome.reason.unwrap_or_else(|| "denied".to_string());
6026                return PreparedCall::Done((
6027                    format!("Error: blocked by pre-tool hook: {reason}"),
6028                    true,
6029                ));
6030            }
6031        }
6032        PreparedCall::Ready { name, args }
6033    }
6034
6035    /// BP-10 (catalog row "Hook/plugin permission veto"): fire the pre-tool
6036    /// hook for the permissions-engine path, where it runs BEFORE the gate
6037    /// (CC's own order: a `PreToolUse` hook answers the permission question
6038    /// rather than being asked after it). Returns the possibly-REWRITTEN
6039    /// arguments plus the [`crate::config::HookDecision`] the engine folds
6040    /// in, or the finished denial when the hook refused outright.
6041    ///
6042    /// The rewrite lands BEFORE the gate deliberately: the engine must
6043    /// evaluate what will actually run, so a hook cannot launder a denied
6044    /// command by rewriting it past the rules.
6045    #[allow(clippy::type_complexity)]
6046    fn run_pre_tool_hook(
6047        &self,
6048        name: &str,
6049        args: serde_json::Value,
6050    ) -> std::result::Result<(serde_json::Value, crate::config::HookDecision), PreparedCall> {
6051        let Some(hook) = &self.config.pre_tool_hook else {
6052            return Ok((args, crate::config::HookDecision::Pass));
6053        };
6054        let outcome = hook(name, &args);
6055        let args = outcome.updated_args.unwrap_or(args);
6056        if outcome.decision == crate::config::HookDecision::Deny {
6057            let reason = outcome.reason.unwrap_or_else(|| "denied".to_string());
6058            return Err(PreparedCall::Done((
6059                format!("Error: blocked by pre-tool hook: {reason}"),
6060                true,
6061            )));
6062        }
6063        Ok((args, outcome.decision))
6064    }
6065
6066    /// D-2 (Fable-5 delta review — LOW-MEDIUM, "over-grant residual"): the
6067    /// built-in tools whose `command`/`path`/`patch` arg IS semantically
6068    /// the whole call — the ONLY tools [`Self::permissions_gate_denial`]
6069    /// is allowed to turn into an [`permissions::ApprovalRequest::subject`]
6070    /// (see that method's own doc comment on the `subject` line for the
6071    /// full story). Bash-family (`bash`, and `shell` —
6072    /// [`crate::tools::builtins::PersistentShellTool`]'s registered name,
6073    /// what the review's "persistent-shell" refers to), the file tools
6074    /// (`read_file`/`write_file`/`edit_file`/`view_image`, whose `path` IS
6075    /// the subject), and `apply_patch` (whose `patch` envelope is handled
6076    /// separately but is unconditionally this tool only, see the `patch`
6077    /// local a few lines below). Deliberately NOT `list_dir`/`glob`/
6078    /// `search` — this crate's F2 fix (`ApprovalCache::key_for_request`)
6079    /// already falls back to a full-args digest for anything not on this
6080    /// list, which is a strictly SAFER (if slightly less cache-granular)
6081    /// default than guessing at more built-ins that weren't part of this
6082    /// finding.
6083    const SUBJECT_BEARING_BUILTIN_TOOLS: &'static [&'static str] = &[
6084        "bash",
6085        "shell",
6086        "read_file",
6087        "write_file",
6088        "edit_file",
6089        "view_image",
6090        "apply_patch",
6091    ];
6092
6093    /// P5-1: evaluate `name`'s call (with parsed `args`) against the
6094    /// permissions engine (`crate::permissions`) — builds the
6095    /// [`crate::permissions::RuleSet`] from `Config`'s deny/ask/allow
6096    /// pattern lists (folding [`Config::permissions_protected_paths`] into
6097    /// the `deny` tier, module 13), picks a command-, path-, or name-only
6098    /// evaluation depending on what `args` carries, resolves an `Ask`
6099    /// decision via the session cache + THIS agent's own installed
6100    /// [`crate::permissions::PermissionsApprovalHandler`], and returns
6101    /// `Some(reason)` when the call is refused (`None` = proceed). Thin
6102    /// wrapper over [`Self::permissions_gate_denial_impl`] — see that
6103    /// method's doc comment for why the handler is a parameter there.
6104    fn permissions_gate_denial(
6105        &self,
6106        name: &str,
6107        args: &serde_json::Value,
6108        hook: crate::config::HookDecision,
6109    ) -> Option<String> {
6110        self.permissions_gate_denial_impl(
6111            name,
6112            args,
6113            self.permissions_approval_handler.as_deref(),
6114            hook,
6115        )
6116    }
6117
6118    /// P5-6 (§2.2 C6, build brief "wire to the P5-1 engine's non-
6119    /// interactive fail-closed path"): the SAME rule-evaluation body as
6120    /// [`Self::permissions_gate_denial`], but the approval `handler` is a
6121    /// PARAMETER instead of always reading `self.permissions_approval_handler`
6122    /// — `Agent::background_permission_denial` calls this with `handler:
6123    /// None` (or a [`crate::subagents::ParentQueueApprovalHandler`]) so a
6124    /// `background_exec` call's `Ask`-tier decisions resolve exactly like a
6125    /// P5-3 background child's do (`crate::permissions::resolve_ask`'s
6126    /// pre-existing "no handler ⇒ deny" contract), REGARDLESS of whether
6127    /// this agent itself has an interactive handler installed for its own
6128    /// foreground calls — a background job must never block on a prompt it
6129    /// has no way to answer, even if the agent hosting it could otherwise
6130    /// answer one. The rule SET and default-policy baseline are otherwise
6131    /// identical to a foreground call's — only how an `Ask` decision
6132    /// resolves ever differs, and only in the strictly-narrower direction
6133    /// (never escalates past what a foreground call of the same command is
6134    /// allowed).
6135    fn permissions_gate_denial_impl(
6136        &self,
6137        name: &str,
6138        args: &serde_json::Value,
6139        handler: Option<&dyn crate::permissions::PermissionsApprovalHandler>,
6140        hook: crate::config::HookDecision,
6141    ) -> Option<String> {
6142        use crate::permissions::{self, Decision, PathKind};
6143
6144        // `Config.tool_deny_patterns`/`tool_allow_patterns` (P4a) ARE the
6145        // engine's deny/allow tiers — the same `capabilities.permissions.
6146        // rules.deny`/`.allow` keys, one source of truth, no duplication.
6147        // Protected paths (module 13) are an unconditional deny floor,
6148        // folded in here rather than checked separately, so they benefit
6149        // from the SAME first-match deny-wins priority every other deny
6150        // rule gets.
6151        // BP-5: the rule set itself is `permissions::rules_for_config` —
6152        // ONE construction shared with every other surface that has to ask
6153        // this engine a question (see that function's doc comment). Plan
6154        // mode's narrowing is layered on top of it here, because it is
6155        // this agent's live state, not the config's.
6156        let mut rules = permissions::rules_for_config(&self.config);
6157        // BP-3 (§2 module 8 `plan_mode`, dependency edge `plan_mode →
6158        // permissions.rules|sandbox`): the read-only research phase IS a
6159        // narrowing of this rule set — while the mode is active the write
6160        // and execution tools join the deny tier, and get the same
6161        // first-match, never-overridable treatment every other deny rule
6162        // gets. Inactive (the default, and the only state a config without
6163        // the module can reach) contributes NOTHING, so the rule set is
6164        // byte-identical to before.
6165        rules
6166            .deny
6167            .extend(crate::tools::plan_mode::deny_rules(&self.ctx.plan_mode));
6168
6169        // The baseline decision when NO rule matches at all — derived from
6170        // `ApprovalPolicy`, the same per-policy shape
6171        // `Config::needs_approval` uses for the legacy gate (see
6172        // `ApprovalPolicy::ModelRequested`'s doc comment for why this
6173        // richer gate approximates Codex's real "mostly silent" posture
6174        // instead of that method's conservative OnRequest-alike treatment
6175        // — explicit deny/ask rules still apply on top regardless).
6176        let default = permissions::default_decision(&self.config, name);
6177
6178        // BP-10: path rules are evaluated relative to EVERY granted root
6179        // (cwd + `core.additional_dirs`/`--add-dir`), folded to the
6180        // strictest — see `permissions::evaluate_path_safe_roots`'s doc
6181        // comment for why a grant must not also remove the root-relative
6182        // protected-path floor inside the granted directory. With no extra
6183        // dirs (the default) this is the single `cwd` list, byte-identical
6184        // to before.
6185        let mut roots = vec![self.config.cwd.clone()];
6186        roots.extend(self.config.additional_dirs.iter().cloned());
6187
6188        let command = args.get("command").and_then(|v| v.as_str());
6189        let path = args.get("path").and_then(|v| v.as_str());
6190        // F4 (Fable-5 adversarial review): `apply_patch`'s args carry a
6191        // patch ENVELOPE body (`args["patch"]`), not a `command` or a
6192        // `path` — the two branches above never fire for it, which is
6193        // exactly how a patch touching a protected path bypassed
6194        // `protected_paths` entirely. Only consulted for the `apply_patch`
6195        // tool specifically (a `patch`-shaped arg on some other tool is not
6196        // this envelope format and isn't given this treatment).
6197        let patch = (name == "apply_patch")
6198            .then(|| args.get("patch").and_then(|v| v.as_str()))
6199            .flatten();
6200        let decision = if let Some(command) = command {
6201            permissions::evaluate_command(&rules, name, command, default)
6202        } else if let Some(patch) = patch {
6203            // Same dual-check shape as the path branch below (pseudo-tool
6204            // `write(...)` rules from `protected_paths`, AND a rule
6205            // authored against the real `apply_patch` tool name), applied
6206            // to EVERY path the envelope's ops touch (`Add`/`Delete`/
6207            // `Update`'s `path`, plus `*** Move to:`). A patch that fails
6208            // to parse can't be proven to avoid a protected path — fail
6209            // closed to at least `Ask`, the same floor an unparseable bash
6210            // command gets in `permissions::evaluate_command`, rather than
6211            // silently let it through on `default`.
6212            let mut d = rules.evaluate(name, None).unwrap_or(default);
6213            match crate::tools::patch_target_paths(patch) {
6214                Ok(paths) => {
6215                    for p in &paths {
6216                        // SECURITY (CRITICAL fix): route both checks through
6217                        // the safe-path-resolving variants — a patch target
6218                        // like `x/../.git/config` must be caught exactly
6219                        // like a `write_file`/`edit_file` `path` argument
6220                        // would be (see `evaluate_path_safe`'s doc comment).
6221                        let pseudo = permissions::evaluate_path_safe_roots(
6222                            &rules,
6223                            PathKind::Write,
6224                            &roots,
6225                            p,
6226                            Decision::Allow,
6227                        );
6228                        let real_tool = permissions::evaluate_path_subject_safe_roots(
6229                            &rules,
6230                            name,
6231                            &roots,
6232                            p,
6233                            Decision::Allow,
6234                        );
6235                        d = d.stricter(pseudo).stricter(real_tool);
6236                    }
6237                }
6238                Err(_) => {
6239                    d = d.stricter(Decision::Ask);
6240                }
6241            }
6242            d
6243        } else if let Some(path) = path {
6244            let kind = if matches!(name, "write_file" | "edit_file") {
6245                PathKind::Write
6246            } else {
6247                PathKind::Read
6248            };
6249            // TWO independent sources of path-shaped rules can apply to the
6250            // same call, and BOTH must be checked:
6251            // (a) the `read(...)`/`write(...)` pseudo-tool (tool-agnostic —
6252            //     applies no matter WHICH tool touches the path; this is
6253            //     what `Config::permissions_protected_paths`/module 13
6254            //     expands into, via `protected_path_deny_rules`);
6255            // (b) a rule authored against the REAL tool name with the path
6256            //     as its subject — design §4.4's own oc-parity worked
6257            //     example writes exactly this shape (`"read_file(*.env)"`,
6258            //     not a pseudo-tool), matching how `bash(cmdglob)` rules
6259            //     are authored. `RuleSet::evaluate`'s bare-tool-name-glob
6260            //     branch (no parens) ALSO fires here regardless of
6261            //     `subject`, so this one call additionally covers a
6262            //     blanket "deny this tool entirely" rule — no separate
6263            //     `rules.evaluate(name, None)` call is needed.
6264            // SECURITY (CRITICAL fix, guarantor audit): both checks now
6265            // route through the safe-path-resolving variants (see
6266            // `evaluate_path_safe`'s doc comment) instead of glob-matching
6267            // the raw model-supplied `path` string directly — this is what
6268            // closes the traversal bypass (`write_file
6269            // path="x/../.git/config"`) and the analogous symlink escape.
6270            let pseudo_decision =
6271                permissions::evaluate_path_safe_roots(&rules, kind, &roots, path, default);
6272            let real_tool_decision =
6273                permissions::evaluate_path_subject_safe_roots(&rules, name, &roots, path, default);
6274            pseudo_decision.stricter(real_tool_decision)
6275        } else {
6276            rules.evaluate(name, None).unwrap_or(default)
6277        };
6278
6279        // BP-10 (catalog row "Hook/plugin permission veto"): the hook's
6280        // verdict is a TIER of this engine, folded in the one direction
6281        // that is always safe — `Ask` tightens (`stricter` never loosens),
6282        // `Deny` is a floor. `Allow` deliberately does NOT change the
6283        // decision here: it answers the `Ask` tier below (the
6284        // `PermissionRequest`-class reply CC's hooks give on the user's
6285        // behalf), so a `deny` rule still refuses the call outright — a
6286        // hook may skip a prompt, never a floor.
6287        let decision = match hook {
6288            crate::config::HookDecision::Deny => Decision::Deny,
6289            crate::config::HookDecision::Ask => decision.stricter(Decision::Ask),
6290            crate::config::HookDecision::Allow | crate::config::HookDecision::Pass => decision,
6291        };
6292
6293        // BP-10 (catalog row "Sandbox-escalation path", cx§4
6294        // `sandbox_permissions: "require_escalated"` + justification): the
6295        // model's channel to ASK for an unsandboxed run is a rule inside
6296        // this one engine, not a switch beside it. A call carrying
6297        // `with_escalated_permissions: true` is forced to at least the
6298        // `Ask` tier — never below whatever the rules already decided, so
6299        // a denied command cannot escalate its way out (`stricter` only
6300        // tightens), and never silently allowed under
6301        // `ApprovalPolicy::ModelRequested`/`Never`, whose `Allow` baseline
6302        // is exactly what made "the model requests escalation" a no-op
6303        // before. The `justification` rides in `raw_args` below, so the
6304        // approval door shows the user the model's own reason.
6305        let decision = if args
6306            .get("with_escalated_permissions")
6307            .and_then(|v| v.as_bool())
6308            .unwrap_or(false)
6309        {
6310            decision.stricter(Decision::Ask)
6311        } else {
6312            decision
6313        };
6314
6315        // D-2 (Fable-5 delta review — LOW-MEDIUM): `command`/`path`/`patch`
6316        // above are extracted (and used to DRIVE the decision above) for
6317        // ANY tool that happens to carry one of those arg names — that
6318        // part is unchanged and correct (a rule authored against, say, an
6319        // MCP tool's own name legitimately wants to glob-match its
6320        // `command`-shaped arg too). But the narrower single-field
6321        // `subject` handed to the cache/handler below must NOT do the
6322        // same for a non-built-in tool: an MCP (or other) tool's
6323        // `command`/`path` is just one field among potentially several
6324        // that together define what the call actually does — collapsing
6325        // an `AllowForSession` grant down to that one field would silently
6326        // auto-allow a later call with the SAME `command` but different
6327        // OTHER args (e.g. `{"command":"sync","target":"staging"}`
6328        // auto-allowing `{"command":"sync","target":"production"}`).
6329        // Restricting this to the known built-ins whose `subject` really
6330        // IS the whole call leaves every other tool with `subject: None`,
6331        // which routes it through `ApprovalCache::key_for_request`'s
6332        // full-args-digest fallback (F2) instead.
6333        let subject = Self::SUBJECT_BEARING_BUILTIN_TOOLS
6334            .contains(&name)
6335            .then(|| command.or(path).or(patch))
6336            .flatten();
6337        let req = permissions::ApprovalRequest {
6338            tool: name,
6339            subject,
6340            raw_args: args,
6341        };
6342        let approved = permissions::decision_to_approved(decision, || {
6343            // BP-10: a hook `Allow` answers this ask without a prompt (and
6344            // without a cache entry — the hook is consulted on every call,
6345            // so caching its answer would be a second, staler copy of the
6346            // same decision).
6347            if hook == crate::config::HookDecision::Allow {
6348                return true;
6349            }
6350            permissions::resolve_ask(&self.permissions_approval_cache, handler, &req)
6351        });
6352        if approved {
6353            None
6354        } else {
6355            Some(format!(
6356                "tool `{name}` was not approved for execution (permissions engine: {decision:?})"
6357            ))
6358        }
6359    }
6360
6361    /// P5-6 (§2.2 C6, build brief "a bg `rm -rf` subject to the same deny
6362    /// rules... must never escalate past what a foreground exec of the
6363    /// same command is allowed"): the permission gate `background_exec`
6364    /// runs BEFORE spawning anything. Evaluated against the tool name
6365    /// `"bash"` (not `"background_exec"`) deliberately — so any
6366    /// `bash(...)`-authored deny/ask/allow rule (or protected-path floor)
6367    /// applies to a background command byte-for-byte identically to a
6368    /// foreground `bash` call, the SAME rule set + default baseline
6369    /// [`Self::permissions_gate_denial`] would use for one.
6370    ///
6371    /// The one deliberate difference (C6 itself): an `Ask`-tier decision
6372    /// NEVER reaches an interactive handler here — a background job has no
6373    /// way to block on a prompt it can't answer. When
6374    /// [`Config::subagents_background_prompts`] is
6375    /// [`crate::subagents::BackgroundPromptsPolicy::Parent`], the denied
6376    /// request is additionally queued onto [`Self::pending_child_approvals`]
6377    /// (via [`crate::subagents::ParentQueueApprovalHandler`], reused
6378    /// verbatim — the SAME "parent-surfaced queue" §2.2 C6 names for
6379    /// `subagents.background`, with the job id standing in for a child
6380    /// agent id) for later inspection; any other configuration (including
6381    /// no `background_prompts` set at all) resolves via `handler: None` —
6382    /// [`crate::permissions::resolve_ask`]'s pre-existing "no handler ⇒
6383    /// deny" fail-closed default, identical to `subagents`'s own
6384    /// `AutoPolicy` reading. Either way, `Ask` always denies; only `Allow`
6385    /// (from the rule engine itself, or a PRIOR interactively-granted
6386    /// `AllowForSession` cache entry) ever lets a background command run —
6387    /// so this can only ever be as-or-more restrictive than a foreground
6388    /// call, never looser, regardless of configuration.
6389    ///
6390    /// Covers BOTH gate generations: when [`Config::permissions_enabled`]
6391    /// is on, the P5-1 engine (above) is used; otherwise the legacy
6392    /// [`Config::needs_approval`] gate is consulted but its
6393    /// `approval_handler` closure is NEVER invoked (that closure could
6394    /// itself block, e.g. a real interactive prompt) — an approval-required
6395    /// legacy policy simply denies a background command outright, the same
6396    /// never-hang guarantee under the older gate.
6397    fn background_permission_denial(
6398        &self,
6399        command: &str,
6400        job_id: &str,
6401        hook: crate::config::HookDecision,
6402    ) -> Option<String> {
6403        let args = serde_json::json!({ "command": command });
6404        if self.config.permissions_enabled {
6405            if let Some(crate::subagents::BackgroundPromptsPolicy::Parent) =
6406                self.config.subagents_background_prompts
6407            {
6408                let handler = crate::subagents::ParentQueueApprovalHandler {
6409                    child_agent_id: format!("bg:{job_id}"),
6410                    queue: self.pending_child_approvals.clone(),
6411                };
6412                self.permissions_gate_denial_impl("bash", &args, Some(&handler), hook)
6413            } else {
6414                self.permissions_gate_denial_impl("bash", &args, None, hook)
6415            }
6416        } else if self.config.needs_approval("bash") {
6417            Some(
6418                "tool `bash` requires approval, which a background job cannot request \
6419                 interactively (§2.2 C6: auto-policy denies)"
6420                    .to_string(),
6421            )
6422        } else {
6423            None
6424        }
6425    }
6426
6427    /// P4e (§3.1 `core.parallel_tool_calls`, catalog:59 "Independent
6428    /// sibling calls run concurrently"): runs `calls`' `Tool::execute()`
6429    /// futures CONCURRENTLY via `futures::future::join_all`, for whichever
6430    /// calls [`Self::prepare_tool_call`] resolves to [`PreparedCall::Ready`]
6431    /// — i.e. every plain (non-intrinsic) registry-tool call that passes
6432    /// its synchronous approval/doom-loop/pre-tool-hook checks. A call that
6433    /// resolves to [`PreparedCall::Done`] (an intrinsic, an unknown tool, a
6434    /// denied/blocked call) is NOT parallelized — its result is already in
6435    /// hand from the synchronous prepare pass. Every prepare check still
6436    /// runs sequentially, in original call order, before ANY `execute()`
6437    /// future starts (only the actual tool I/O overlaps) — so doom-loop
6438    /// bookkeeping and pre-tool-hook vetoes see the exact same call order
6439    /// they would under the sequential path. Returns results in the SAME
6440    /// order as `calls`, so callers can always `zip` the two. Post-tool
6441    /// hooks fire per call, in original order, once every result is in
6442    /// hand — a caller-visible timing difference from the sequential path
6443    /// ONLY when this method runs at all (i.e. only when
6444    /// `Config::parallel_tool_calls` is on): hooks see "this batch
6445    /// finished" ordering rather than "this one call finished" ordering.
6446    /// Documented, not a bug.
6447    async fn run_tools_concurrently(
6448        &mut self,
6449        calls: &[supercode_interchange::ToolCall],
6450    ) -> Vec<(String, bool)> {
6451        // P5-3: `spawn_subagent`/`subagent_status` need sequential `&mut
6452        // self` access `prepare_tool_call`'s synchronous-only signature
6453        // can't give them (see `Self::run_tool`'s identical interception).
6454        // A batch that includes one falls back to dispatching the WHOLE
6455        // batch sequentially via `Self::run_tool` — a documented, narrow
6456        // simplification (not a partial-parallelization attempt) rather
6457        // than restructuring `PreparedCall` to carry a future; a batch with
6458        // no subagent intrinsic is completely unaffected and still
6459        // parallelizes exactly as before.
6460        if calls.iter().any(|c| {
6461            c.function.name == SPAWN_SUBAGENT
6462                || c.function.name == SUBAGENT_STATUS
6463                || c.function.name == SEND_MESSAGE
6464                || c.function.name == SUBAGENT_RESUME
6465                || (self.config.claude_runtime_tools_enabled
6466                    && matches!(
6467                        c.function.name.as_str(),
6468                        CLAUDE_CRON_CREATE
6469                            | CLAUDE_CRON_DELETE
6470                            | CLAUDE_CRON_LIST
6471                            | CLAUDE_SCHEDULE_WAKEUP
6472                    ))
6473        }) {
6474            let mut out = Vec::with_capacity(calls.len());
6475            for call in calls {
6476                out.push(self.run_tool(call).await);
6477            }
6478            return out;
6479        }
6480        let prepared: Vec<PreparedCall> = calls.iter().map(|c| self.prepare_tool_call(c)).collect();
6481        let mut slots: Vec<Option<(String, bool)>> = prepared
6482            .iter()
6483            .map(|p| match p {
6484                PreparedCall::Done(r) => Some(r.clone()),
6485                PreparedCall::Ready { .. } => None,
6486            })
6487            .collect();
6488
6489        let ready_idxs: Vec<usize> = prepared
6490            .iter()
6491            .enumerate()
6492            .filter(|(_, p)| matches!(p, PreparedCall::Ready { .. }))
6493            .map(|(i, _)| i)
6494            .collect();
6495
6496        if !ready_idxs.is_empty() {
6497            let futs = ready_idxs.iter().map(|&i| {
6498                let PreparedCall::Ready { name, args } = &prepared[i] else {
6499                    unreachable!("filtered to Ready above")
6500                };
6501                // `self.registry.get` borrows `self.registry` immutably;
6502                // `self.ctx` is `Clone` (P4c precedent) so each future owns
6503                // its own copy rather than borrowing `self` across the
6504                // `.await` inside `join_all`.
6505                let tool = self.registry.get(name).expect("prepared as Ready");
6506                let args = args.clone();
6507                let ctx = self.ctx.clone();
6508                async move {
6509                    match tool.execute(args, &ctx).await {
6510                        Ok(out) => (out, false),
6511                        Err(e) => (format!("Error: {e}"), true),
6512                    }
6513                }
6514            });
6515            let results = futures::future::join_all(futs).await;
6516            for (idx, result) in ready_idxs.iter().zip(results) {
6517                slots[*idx] = Some(result);
6518            }
6519        }
6520
6521        let out: Vec<(String, bool)> = slots
6522            .into_iter()
6523            .map(|s| s.expect("every call resolved to Some above"))
6524            .collect();
6525        // Post-tool hook, in original order — only for calls that actually
6526        // reached `execute()` (matches `run_tool`'s existing behavior: an
6527        // intrinsic/denied/blocked call never fires the post-tool hook).
6528        let ready_set: std::collections::HashSet<usize> = ready_idxs.into_iter().collect();
6529        for (i, call) in calls.iter().enumerate() {
6530            if !ready_set.contains(&i) {
6531                continue;
6532            }
6533            let (output, is_error) = &out[i];
6534            if let Some(hook) = &self.config.post_tool_hook {
6535                hook(&call.function.name, output, *is_error);
6536            }
6537        }
6538        out
6539    }
6540
6541    /// P4c (§5.2 P4 "doom-loop breaker", §3.1 `core.doom_loop_threshold`):
6542    /// update the consecutive-identical-call streak for `(name, args)` and
6543    /// return `Some(reason)` the moment the streak reaches
6544    /// `Config.doom_loop_threshold` (a call whose name AND JSON-canonical
6545    /// arguments are byte-identical to the immediately preceding call
6546    /// extends the streak; anything else resets it to 1). `None`
6547    /// (`Config.doom_loop_threshold` unset, or `Some(n)` with `n < 2` — a
6548    /// threshold below 2 can never fire since the FIRST call already
6549    /// "repeats zero times") never touches the streak fields at all.
6550    fn check_doom_loop(&mut self, name: &str, args: &serde_json::Value) -> Option<String> {
6551        let threshold = self.config.doom_loop_threshold?;
6552        if threshold < 2 {
6553            return None;
6554        }
6555        // `serde_json::Value::Object` is a `BTreeMap` in this workspace (no
6556        // `preserve_order` feature), so `to_string()` is already
6557        // key-order-canonical — two calls that differ only in argument key
6558        // order are still treated as identical.
6559        let key = (name.to_string(), args.to_string());
6560        if self.doom_loop_last_call.as_ref() == Some(&key) {
6561            self.doom_loop_streak += 1;
6562        } else {
6563            self.doom_loop_last_call = Some(key);
6564            self.doom_loop_streak = 1;
6565        }
6566        if self.doom_loop_streak >= threshold {
6567            Some(format!(
6568                "doom-loop breaker: `{name}` called with identical arguments {} times in a row \
6569                 — try a different approach instead of repeating the same call",
6570                self.doom_loop_streak
6571            ))
6572        } else {
6573            None
6574        }
6575    }
6576
6577    /// Whether `name` is in the eagerly-advertised "core" set for the current
6578    /// [`ToolAdvertising`] mode: every enabled tool under `Full`, or the
6579    /// explicit `core` allowlist under `Deferred`.
6580    fn is_core_tool(&self, name: &str) -> bool {
6581        match &self.config.tool_advertising {
6582            ToolAdvertising::Full => true,
6583            ToolAdvertising::Deferred { core } => core.iter().any(|c| c == name),
6584        }
6585    }
6586
6587    /// The schema advertised on the wire for `t`: the raw (as-shipped)
6588    /// schema with TR-8/T5's per-tool schema tier applied. This is what
6589    /// [`Self::tool_schemas`] sends every request.
6590    fn schema_for(&self, t: &dyn crate::tools::Tool) -> ToolSchema {
6591        let raw = self.raw_schema_for(t);
6592        let tier = self.config.schema_tier_for(t.name());
6593        let (description, parameters) =
6594            crate::tools::tiers::minify(&raw.description, &raw.parameters, tier);
6595        ToolSchema {
6596            name: raw.name,
6597            description,
6598            parameters,
6599        }
6600    }
6601
6602    /// The ORIGINAL, as-shipped schema for `t` — never tier-minified. This is
6603    /// the full contract [`Self::run_tool_search`] hands back on activation
6604    /// (TR-8/T5 dev/03: the B6 fetch path is the invert of tiering, so a
6605    /// model that fetched a tool via `tool_search` always sees the complete
6606    /// schema, byte-equal to `t.description()`/`t.parameters()` — modulo the
6607    /// pre-existing [`crate::Config::tool_description`] override, which is
6608    /// orthogonal to tiering).
6609    fn raw_schema_for(&self, t: &dyn crate::tools::Tool) -> ToolSchema {
6610        ToolSchema {
6611            name: t.name().to_string(),
6612            description: self
6613                .config
6614                .tool_description(t.name(), t.description())
6615                .to_string(),
6616            parameters: t.parameters(),
6617        }
6618    }
6619
6620    /// The synthetic `tool_search` schema advertised under `Deferred` (B6).
6621    fn tool_search_schema() -> ToolSchema {
6622        ToolSchema {
6623            name: TOOL_SEARCH.to_string(),
6624            description: "Search for additional tools not currently advertised (the deferred \
6625                MCP surface and any other non-core tools). Matches keywords case-insensitively \
6626                against each tool's name and description. Matched tools become callable starting \
6627                with your NEXT message, not this one."
6628                .to_string(),
6629            parameters: serde_json::json!({
6630                "type": "object",
6631                "properties": {
6632                    "query": {
6633                        "type": "string",
6634                        "description": "Keyword(s) to search for in tool names and descriptions."
6635                    },
6636                    "max_results": {
6637                        "type": "integer",
6638                        "description": "Maximum number of matching tools to return."
6639                    }
6640                },
6641                "required": ["query"],
6642                "additionalProperties": false
6643            }),
6644        }
6645    }
6646
6647    /// The tool-schema array this agent would advertise on its NEXT
6648    /// request, exactly as `Self::run_loop` computes it. Public
6649    /// (PARITY-18 D1) so a caller can measure the real request-token cost
6650    /// of an agent's tool surface — including the current
6651    /// [`crate::config::ToolAdvertising`] mode's core/deferred split and
6652    /// the synthetic `tool_search`/`expand_reduction`/`sidecar_search`
6653    /// schemas — BEFORE ever calling [`Self::send`], e.g. for a preflight
6654    /// context-guard check.
6655    pub fn tool_schemas(&self) -> Vec<ToolSchema> {
6656        let mut out = match &self.config.tool_advertising {
6657            ToolAdvertising::Full => self
6658                .registry
6659                .iter()
6660                .filter(|t| self.config.tool_enabled(t.name()))
6661                .map(|t| self.schema_for(t))
6662                .collect(),
6663            ToolAdvertising::Deferred { .. } => {
6664                let mut out: Vec<ToolSchema> = self
6665                    .registry
6666                    .iter()
6667                    .filter(|t| self.config.tool_enabled(t.name()))
6668                    .filter(|t| {
6669                        self.is_core_tool(t.name()) || self.activated_tools.contains(t.name())
6670                    })
6671                    .map(|t| self.schema_for(t))
6672                    .collect();
6673                out.push(Self::tool_search_schema());
6674                out
6675            }
6676        };
6677        // T12/TR-1: `expand_reduction`/`sidecar_search` are orthogonal to
6678        // `tool_advertising` (which governs the ordinary tool surface) —
6679        // advertised whenever a `ReductionPolicy` is installed, regardless of
6680        // Full/Deferred, since only a reduced session ever has anything to
6681        // expand or search (SPEC.md TR-1 dev/01).
6682        if self.reduction_policy.is_some() {
6683            out.push(Self::expand_reduction_schema());
6684            out.push(Self::sidecar_search_schema());
6685        }
6686        // P5-3 (§2 module 9): `spawn_subagent`/`subagent_status` are
6687        // orthogonal to `tool_advertising` too, same reasoning as
6688        // `expand_reduction`/`sidecar_search` above — advertised whenever
6689        // `Config::subagents_enabled` is on, Full or Deferred alike.
6690        // `false` (the default) never appends either, so a config that
6691        // never turns the module on gets byte-identical tool schemas to
6692        // today.
6693        if self.config.subagents_enabled {
6694            out.push(self.spawn_subagent_schema());
6695            if self.config.subagents_claude_agent_alias {
6696                out.push(self.claude_agent_schema());
6697            }
6698            if self.config.subagents_background {
6699                out.push(Self::subagent_status_schema());
6700                // BP-7 (catalog §4a "Background subagents + resume"): the
6701                // two halves the row named as missing — a mailbox into a
6702                // still-running child, and a resume of a finished one with
6703                // its context intact.
6704                out.push(Self::send_message_schema());
6705                out.push(Self::subagent_resume_schema());
6706            }
6707        }
6708        if self.config.claude_runtime_tools_enabled {
6709            out.extend(self.claude_builtin_tool_schemas());
6710            out.push(Self::claude_cron_create_schema());
6711            out.push(Self::claude_cron_delete_schema());
6712            out.push(Self::claude_cron_list_schema());
6713            out.push(Self::claude_schedule_wakeup_schema());
6714        }
6715        // P5-6 (§2 module 4 `tools.background`): same orthogonal-to-
6716        // `tool_advertising` treatment, advertised whenever
6717        // `Config::tools_background_enabled` is on. `false` (the default)
6718        // never appends any of the four, so a config that never turns the
6719        // module on gets byte-identical tool schemas to today.
6720        if self.config.tools_background_enabled {
6721            out.push(Self::background_exec_schema());
6722            out.push(Self::background_status_schema());
6723            out.push(Self::background_list_schema());
6724            out.push(Self::background_kill_schema());
6725        }
6726        // BP-10 (catalog row "Tool hiding via policy", cc§4 "bare-name
6727        // deny"): a policy deny does not merely REFUSE the call at
6728        // dispatch — it removes the tool from the model's view. Applied
6729        // once, here, over the finished array, so every family appended
6730        // above (`spawn_subagent`, `background_*`, the Claude aliases,
6731        // `expand_reduction`, …) is hidden by the same one rule, not by a
6732        // per-family repeat of it. See [`Self::policy_hides_tool`] for
6733        // which deny tier is consulted and why.
6734        out.retain(|schema| !self.policy_hides_tool(&schema.name));
6735        out
6736    }
6737
6738    /// BP-10: whether the CONFIG-DECLARED deny tier hides `name` from the
6739    /// model's tool surface entirely (cc§4: CC's bare-name deny "removes
6740    /// the tool from the model's view", where an ordinary rule only
6741    /// refuses the call).
6742    ///
6743    /// The ONE engine decides: this is
6744    /// [`crate::permissions::RuleSet::evaluate`] with `subject: None`, so
6745    /// exactly the patterns that can be satisfied by a tool NAME ALONE
6746    /// (`"bash"`, `"mcp_*"`, `"*"`) hide; a rule that names a
6747    /// command/path constraint (`"bash(rm -rf*)"`, `"write(.git/**)"`) is
6748    /// not satisfiable without a subject and therefore never hides a tool
6749    /// — the same `rule_matches` contract the dispatch gate uses.
6750    ///
6751    /// **Which deny tier.** `Config::tool_deny_patterns` — the
6752    /// `capabilities.permissions.rules.deny` array — and NOT the two
6753    /// runtime narrowings the dispatch gate folds in beside it:
6754    /// `protected_paths` expands to `read(...)`/`write(...)` patterns that
6755    /// carry a subject by construction (so they could never match here
6756    /// anyway), and `plan_mode::deny_rules` is a MODE, not a policy — CC's
6757    /// plan mode refuses a write, it does not make Write disappear and
6758    /// reappear as the mode toggles mid-session. Hiding is a property of
6759    /// the configured policy, which is fixed for the run.
6760    ///
6761    /// Gated on [`Config::permissions_enabled`]: a config that never turns
6762    /// the module on gets byte-identical schemas to before this existed.
6763    fn policy_hides_tool(&self, name: &str) -> bool {
6764        if !self.config.permissions_enabled || self.config.tool_deny_patterns.is_empty() {
6765            return false;
6766        }
6767        let rules = crate::permissions::RuleSet {
6768            deny: self.config.tool_deny_patterns.clone(),
6769            ..Default::default()
6770        };
6771        rules.evaluate(name, None) == Some(crate::permissions::Decision::Deny)
6772    }
6773
6774    fn claude_builtin_tool_schemas(&self) -> Vec<ToolSchema> {
6775        let mut schemas = Vec::new();
6776        let mut push = |alias: &str, native: &str, description: &str, parameters| {
6777            if self.registry.get(native).is_some() && self.config.tool_enabled(native) {
6778                schemas.push(ToolSchema {
6779                    name: alias.to_string(),
6780                    description: description.to_string(),
6781                    parameters,
6782                });
6783            }
6784        };
6785        push(
6786            CLAUDE_BASH,
6787            "bash",
6788            "Claude Code-compatible shell command execution.",
6789            serde_json::json!({
6790                "type": "object",
6791                "properties": {
6792                    "command": {"type": "string"},
6793                    "timeout": {"type": "integer", "description": "Timeout in milliseconds."},
6794                    "description": {"type": "string"}
6795                },
6796                "required": ["command"],
6797                "additionalProperties": true
6798            }),
6799        );
6800        push(
6801            CLAUDE_READ,
6802            "read_file",
6803            "Claude Code-compatible file reader.",
6804            serde_json::json!({
6805                "type": "object",
6806                "properties": {
6807                    "file_path": {"type": "string"},
6808                    "offset": {"type": "integer"},
6809                    "limit": {"type": "integer"}
6810                },
6811                "required": ["file_path"],
6812                "additionalProperties": false
6813            }),
6814        );
6815        push(
6816            CLAUDE_WRITE,
6817            "write_file",
6818            "Claude Code-compatible file writer.",
6819            serde_json::json!({
6820                "type": "object",
6821                "properties": {"file_path": {"type": "string"}, "content": {"type": "string"}},
6822                "required": ["file_path", "content"],
6823                "additionalProperties": false
6824            }),
6825        );
6826        push(
6827            CLAUDE_EDIT,
6828            "edit_file",
6829            "Claude Code-compatible exact file edit.",
6830            serde_json::json!({
6831                "type": "object",
6832                "properties": {
6833                    "file_path": {"type": "string"},
6834                    "old_string": {"type": "string"},
6835                    "new_string": {"type": "string"},
6836                    "replace_all": {"type": "boolean"}
6837                },
6838                "required": ["file_path", "old_string", "new_string"],
6839                "additionalProperties": false
6840            }),
6841        );
6842        push(
6843            CLAUDE_GLOB,
6844            "glob",
6845            "Claude Code-compatible file glob.",
6846            serde_json::json!({
6847                "type": "object",
6848                "properties": {"pattern": {"type": "string"}, "path": {"type": "string"}},
6849                "required": ["pattern"],
6850                "additionalProperties": false
6851            }),
6852        );
6853        push(
6854            CLAUDE_GREP,
6855            "search",
6856            "Claude Code-compatible content search.",
6857            serde_json::json!({
6858                "type": "object",
6859                "properties": {"pattern": {"type": "string"}, "path": {"type": "string"}},
6860                "required": ["pattern"],
6861                "additionalProperties": true
6862            }),
6863        );
6864        schemas
6865    }
6866
6867    fn translate_claude_builtin_call(
6868        &self,
6869        call: &supercode_interchange::ToolCall,
6870    ) -> Result<Option<supercode_interchange::ToolCall>> {
6871        let native = match call.function.name.as_str() {
6872            CLAUDE_BASH => "bash",
6873            CLAUDE_READ => "read_file",
6874            CLAUDE_WRITE => "write_file",
6875            CLAUDE_EDIT => "edit_file",
6876            CLAUDE_GLOB => "glob",
6877            CLAUDE_GREP => "search",
6878            _ => return Ok(None),
6879        };
6880        let mut args = call.function.parsed_arguments()?;
6881        let object = args
6882            .as_object_mut()
6883            .ok_or_else(|| Error::InvalidArguments {
6884                tool: call.function.name.clone(),
6885                message: "expected a JSON object".to_string(),
6886            })?;
6887        if let Some(path) = object.remove("file_path") {
6888            object.entry("path".to_string()).or_insert(path);
6889        }
6890        if call.function.name == CLAUDE_BASH {
6891            if let Some(timeout) = object.remove("timeout") {
6892                object.entry("timeout_ms".to_string()).or_insert(timeout);
6893            }
6894        }
6895        if call.function.name == CLAUDE_GLOB {
6896            if let Some(path) = object
6897                .remove("path")
6898                .and_then(|value| value.as_str().map(str::to_owned))
6899            {
6900                if let Some(pattern) = object.get_mut("pattern") {
6901                    if let Some(value) = pattern.as_str() {
6902                        if !std::path::Path::new(value).is_absolute() {
6903                            *pattern = serde_json::Value::String(
6904                                std::path::Path::new(&path)
6905                                    .join(value)
6906                                    .to_string_lossy()
6907                                    .into_owned(),
6908                            );
6909                        }
6910                    }
6911                }
6912            }
6913        }
6914        let mut translated = call.clone();
6915        translated.function.name = native.to_string();
6916        translated.function.arguments = serde_json::to_string(&args)?;
6917        Ok(Some(translated))
6918    }
6919
6920    fn claude_cron_create_schema() -> ToolSchema {
6921        ToolSchema {
6922            name: CLAUDE_CRON_CREATE.to_string(),
6923            description: "Record a Claude-compatible cron job in the imported runtime manifest. \
6924                The job inherits the manifest's ACTIVE or PAUSED posture; an embedding scheduler, \
6925                not this agent loop, owns execution."
6926                .to_string(),
6927            parameters: serde_json::json!({
6928                "type": "object",
6929                "properties": {
6930                    "cron": {"type": "string", "description": "Cron expression to preserve."},
6931                    "prompt": {"type": "string", "description": "Prompt associated with the job."},
6932                    "recurring": {"type": "boolean", "default": false},
6933                    "durable": {"type": "boolean", "default": false}
6934                },
6935                "required": ["cron", "prompt"],
6936                "additionalProperties": false
6937            }),
6938        }
6939    }
6940
6941    fn claude_cron_delete_schema() -> ToolSchema {
6942        ToolSchema {
6943            name: CLAUDE_CRON_DELETE.to_string(),
6944            description: "Delete a Claude-compatible cron job from the imported manifest. \
6945                This updates state only; an embedding scheduler owns execution."
6946                .to_string(),
6947            parameters: serde_json::json!({
6948                "type": "object",
6949                "properties": {"id": {"type": "string"}},
6950                "required": ["id"],
6951                "additionalProperties": false
6952            }),
6953        }
6954    }
6955
6956    fn claude_cron_list_schema() -> ToolSchema {
6957        ToolSchema {
6958            name: CLAUDE_CRON_LIST.to_string(),
6959            description: "List imported Claude cron jobs and their explicit ACTIVE or PAUSED \
6960                manifest posture. This agent loop itself does not run a scheduler."
6961                .to_string(),
6962            parameters: serde_json::json!({
6963                "type": "object",
6964                "properties": {},
6965                "additionalProperties": false
6966            }),
6967        }
6968    }
6969
6970    fn claude_schedule_wakeup_schema() -> ToolSchema {
6971        ToolSchema {
6972            name: CLAUDE_SCHEDULE_WAKEUP.to_string(),
6973            description: "Replace the one-shot wakeup stored in the imported Claude manifest. \
6974                The wakeup inherits the manifest's ACTIVE or PAUSED posture; an embedding scheduler \
6975                owns timer execution."
6976                .to_string(),
6977            parameters: serde_json::json!({
6978                "type": "object",
6979                "properties": {
6980                    "delaySeconds": {"type": "integer", "minimum": 0},
6981                    "reason": {"type": "string"},
6982                    "prompt": {"type": "string"}
6983                },
6984                "required": ["delaySeconds"],
6985                "additionalProperties": false
6986            }),
6987        }
6988    }
6989
6990    /// The `background_exec` schema (P5-6, §2 module 4, D1 "background
6991    /// exec").
6992    fn background_exec_schema() -> ToolSchema {
6993        ToolSchema {
6994            name: BACKGROUND_EXEC.to_string(),
6995            description: "Run a shell command in the BACKGROUND: spawns it as a detached \
6996                process and returns a `job_id` IMMEDIATELY, before the command finishes — this \
6997                call never returns the command's output. Poll `background_status` with the \
6998                `job_id` to check progress and retrieve captured output; use `background_kill` \
6999                to cancel it early. The command goes through the exact same sandbox/permission \
7000                checks as a foreground `bash` call, and any check that would need an \
7001                interactive approval is denied automatically (a background job cannot wait for \
7002                one)."
7003                .to_string(),
7004            parameters: serde_json::json!({
7005                "type": "object",
7006                "properties": {
7007                    "command": {
7008                        "type": "string",
7009                        "description": "Shell command to run in the background via `sh -c`."
7010                    }
7011                },
7012                "required": ["command"],
7013                "additionalProperties": false
7014            }),
7015        }
7016    }
7017
7018    /// The `background_status` schema (P5-6, D1 "monitor/event feed").
7019    fn background_status_schema() -> ToolSchema {
7020        ToolSchema {
7021            name: BACKGROUND_STATUS.to_string(),
7022            description: "Check on a background job spawned via background_exec: its \
7023                running/exited/killed status, exit code (once known), and the command's \
7024                captured stdout/stderr so far (bounded — very large output is truncated with a \
7025                marker). Once the job has exited or been killed, this call also reaps it (it \
7026                will no longer appear in background_list or accept further status polls)."
7027                .to_string(),
7028            parameters: serde_json::json!({
7029                "type": "object",
7030                "properties": {
7031                    "job_id": {
7032                        "type": "string",
7033                        "description": "The id `background_exec` returned when this job was \
7034                            started."
7035                    }
7036                },
7037                "required": ["job_id"],
7038                "additionalProperties": false
7039            }),
7040        }
7041    }
7042
7043    /// The `background_list` schema (P5-6, D10 "bg-manager").
7044    fn background_list_schema() -> ToolSchema {
7045        ToolSchema {
7046            name: BACKGROUND_LIST.to_string(),
7047            description: "List every background job currently tracked (running, or finished \
7048                but not yet polled via background_status) — job id, command, status, pid, and \
7049                start time for each. Does not retrieve output or reap anything."
7050                .to_string(),
7051            parameters: serde_json::json!({
7052                "type": "object",
7053                "properties": {},
7054                "additionalProperties": false
7055            }),
7056        }
7057    }
7058
7059    /// The `background_kill` schema (P5-6, D10 "bg-manager").
7060    fn background_kill_schema() -> ToolSchema {
7061        ToolSchema {
7062            name: BACKGROUND_KILL.to_string(),
7063            description: "Kill a background job's real process immediately (a no-op, not an \
7064                error, if it already exited on its own) and reap it."
7065                .to_string(),
7066            parameters: serde_json::json!({
7067                "type": "object",
7068                "properties": {
7069                    "job_id": {
7070                        "type": "string",
7071                        "description": "The id `background_exec` returned when this job was \
7072                            started."
7073                    }
7074                },
7075                "required": ["job_id"],
7076                "additionalProperties": false
7077            }),
7078        }
7079    }
7080
7081    /// The `spawn_subagent` schema (P5-3, §2 module 9 D1 "spawn tool").
7082    /// Lists every configured `agent_type` name so the model knows what's
7083    /// available, but `agent_type` stays optional — an ad-hoc spawn with an
7084    /// inline `system_prompt` is always allowed too.
7085    fn spawn_subagent_schema(&self) -> ToolSchema {
7086        let mut names: Vec<&str> = self
7087            .config
7088            .subagents_definitions
7089            .keys()
7090            .map(String::as_str)
7091            .collect();
7092        names.sort_unstable();
7093        let agent_type_desc = if names.is_empty() {
7094            "Optional named subagent type to run (none configured — omit this and pass \
7095             `system_prompt` instead)."
7096                .to_string()
7097        } else {
7098            format!(
7099                "Optional named subagent type to run: {}. Omit to run an ad-hoc subagent with \
7100                 your own `system_prompt` instead.",
7101                names.join(", ")
7102            )
7103        };
7104        let background_desc = if self.config.subagents_background {
7105            "Run this subagent in the background instead of waiting for it — this call \
7106             returns immediately with a `subagent_id`; poll `subagent_status` with that id for \
7107             the result."
7108        } else {
7109            "Background subagents are disabled for this agent — this must be omitted or false."
7110        };
7111        ToolSchema {
7112            name: SPAWN_SUBAGENT.to_string(),
7113            description: "Spawn a subagent to work on a self-contained task and (by default) \
7114                wait for its final answer, which is returned as this call's result. The \
7115                subagent runs its own independent reasoning/tool loop; it does not see your \
7116                conversation except for the `task` text you give it here."
7117                .to_string(),
7118            parameters: serde_json::json!({
7119                "type": "object",
7120                "properties": {
7121                    "task": {
7122                        "type": "string",
7123                        "description": "The self-contained task/prompt for the subagent."
7124                    },
7125                    "agent_type": {
7126                        "type": "string",
7127                        "description": agent_type_desc
7128                    },
7129                    "system_prompt": {
7130                        "type": "string",
7131                        "description": "Inline system prompt for an ad-hoc subagent (ignored \
7132                            if `agent_type` is given — the named type's own prompt is used \
7133                            instead)."
7134                    },
7135                    "background": {
7136                        "type": "boolean",
7137                        "description": background_desc
7138                    }
7139                },
7140                "required": ["task"],
7141                "additionalProperties": false
7142            }),
7143        }
7144    }
7145
7146    /// Claude Code-compatible alias for [`Self::spawn_subagent_schema`].
7147    fn claude_agent_schema(&self) -> ToolSchema {
7148        let mut names: Vec<String> = self.config.subagents_definitions.keys().cloned().collect();
7149        names.push("general-purpose".into());
7150        names.sort_unstable();
7151        names.dedup();
7152        ToolSchema {
7153            name: CLAUDE_AGENT.to_string(),
7154            description: "Claude Code-compatible subagent dispatcher. Runs a named or ad-hoc \
7155                child agent; children default to background execution in this compatibility mode."
7156                .to_string(),
7157            parameters: serde_json::json!({
7158                "type": "object",
7159                "properties": {
7160                    "prompt": {"type": "string", "description": "Self-contained child task."},
7161                    "subagent_type": {
7162                        "type": "string",
7163                        "description": format!("Named agent type. Available: {}", names.join(", "))
7164                    },
7165                    "description": {
7166                        "type": "string",
7167                        "description": "Short human-facing task label; preserved as descriptive input."
7168                    },
7169                    "model": {
7170                        "type": "string",
7171                        "description": "Optional model alias or full provider slug for this child."
7172                    },
7173                    "run_in_background": {
7174                        "type": "boolean",
7175                        "description": "Whether to return immediately with a child id (default true)."
7176                    }
7177                },
7178                "required": ["prompt"],
7179                "additionalProperties": false
7180            }),
7181        }
7182    }
7183
7184    /// Translate Claude's `Agent` arguments to the native subagent intrinsic.
7185    fn translate_claude_agent_call(
7186        &self,
7187        call: &supercode_interchange::ToolCall,
7188    ) -> Result<supercode_interchange::ToolCall> {
7189        let args = call
7190            .function
7191            .parsed_arguments()
7192            .map_err(|error| Error::InvalidArguments {
7193                tool: CLAUDE_AGENT.to_string(),
7194                message: error.to_string(),
7195            })?;
7196        let object = args.as_object().ok_or_else(|| Error::InvalidArguments {
7197            tool: CLAUDE_AGENT.to_string(),
7198            message: "arguments must be an object".to_string(),
7199        })?;
7200        let mut translated = serde_json::Map::new();
7201        if let Some(value) = object.get("prompt") {
7202            translated.insert("task".to_string(), value.clone());
7203        }
7204        if let Some(value) = object.get("subagent_type") {
7205            // `general-purpose` is a built-in Claude agent, not a project
7206            // definition file. Supercode's equivalent is an ad-hoc child
7207            // using the inherited default system prompt, represented by an
7208            // omitted `agent_type`.
7209            if value.as_str() != Some("general-purpose") {
7210                translated.insert("agent_type".to_string(), value.clone());
7211            }
7212        }
7213        if let Some(value) = object.get("model") {
7214            translated.insert("model".to_string(), value.clone());
7215        }
7216        translated.insert(
7217            "background".to_string(),
7218            object
7219                .get("run_in_background")
7220                .cloned()
7221                .unwrap_or(serde_json::Value::Bool(true)),
7222        );
7223        Ok(supercode_interchange::ToolCall {
7224            id: call.id.clone(),
7225            kind: call.kind.clone(),
7226            function: supercode_interchange::FunctionCall {
7227                name: SPAWN_SUBAGENT.to_string(),
7228                arguments: serde_json::Value::Object(translated).to_string(),
7229            },
7230        })
7231    }
7232
7233    /// Execute Claude's scheduling vocabulary against the imported manifest.
7234    ///
7235    /// This is intentionally a state editor, not a scheduler: it owns no
7236    /// timer/task handle, nothing downstream of it fires, and every
7237    /// successful response says so, so the model is never told a job it just
7238    /// created will run here.
7239    fn run_claude_runtime_tool(
7240        &mut self,
7241        call: &supercode_interchange::ToolCall,
7242    ) -> (String, bool) {
7243        let args = match call.function.parsed_arguments() {
7244            Ok(value) if value.is_object() => value,
7245            Ok(_) => {
7246                return (
7247                    format!("Error: {} arguments must be an object", call.function.name),
7248                    true,
7249                )
7250            }
7251            Err(error) => return (format!("Error: {error}"), true),
7252        };
7253        let object = args.as_object().expect("checked object above");
7254
7255        let Some(manifest) = self.claude_runtime_manifest.as_mut() else {
7256            return (
7257                "Error: Claude runtime compatibility was enabled without an imported runtime \
7258                 manifest; refusing to invent scheduler state"
7259                    .to_string(),
7260                true,
7261            );
7262        };
7263        // Every imported schedule is carried and inert. `state` is the single
7264        // fact the model is told about it, in the manifest's own vocabulary.
7265        let state = "paused";
7266
7267        match call.function.name.as_str() {
7268            CLAUDE_CRON_LIST => {
7269                let jobs: Vec<serde_json::Value> = manifest
7270                    .active_crons
7271                    .iter()
7272                    .map(|job| {
7273                        serde_json::json!({
7274                            "id": job.id,
7275                            "cron": job.schedule,
7276                            "prompt": job.prompt,
7277                            "recurring": job.recurring,
7278                            "durable": job.durable_requested,
7279                            "state": state
7280                        })
7281                    })
7282                    .collect();
7283                let notice = "Imported jobs are preserved but no scheduler is running.";
7284                (
7285                    serde_json::json!({
7286                        "execution_state": state,
7287                        "execution_notice": notice,
7288                        "jobs": jobs
7289                    })
7290                    .to_string(),
7291                    false,
7292                )
7293            }
7294            CLAUDE_CRON_CREATE => {
7295                let Some(schedule) = object.get("cron").and_then(serde_json::Value::as_str) else {
7296                    return ("Error: CronCreate requires string `cron`".to_string(), true);
7297                };
7298                let Some(prompt) = object.get("prompt").and_then(serde_json::Value::as_str) else {
7299                    return (
7300                        "Error: CronCreate requires string `prompt`".to_string(),
7301                        true,
7302                    );
7303                };
7304                let recurring = object
7305                    .get("recurring")
7306                    .and_then(serde_json::Value::as_bool)
7307                    .unwrap_or(false);
7308                let durable_requested = object
7309                    .get("durable")
7310                    .and_then(serde_json::Value::as_bool)
7311                    .unwrap_or(false);
7312                let mut sequence = 1_u64;
7313                let id = loop {
7314                    let candidate = format!("sc{sequence:06}");
7315                    if !manifest.active_crons.iter().any(|job| job.id == candidate) {
7316                        break candidate;
7317                    }
7318                    sequence += 1;
7319                };
7320                let kind = if recurring { "recurring " } else { "" };
7321                let result = format!(
7322                    "Scheduled {kind}job {id} ({schedule}) in PAUSED state. The job is preserved \
7323                     in the continuation manifest but no scheduler is running and it will not execute."
7324                );
7325                manifest
7326                    .active_crons
7327                    .push(crate::claude_runtime_state::ClaudeCronJob {
7328                        id: id.clone(),
7329                        tool_use_id: call.id.clone(),
7330                        schedule: schedule.to_string(),
7331                        recurring,
7332                        durable_requested,
7333                        prompt: prompt.to_string(),
7334                        // The creation instant is a fact about this
7335                        // continuation, recorded like every other manifest
7336                        // field. Nothing consults it as a due time.
7337                        created_at: Some(supercode_interchange::sidecar::ms_to_rfc3339(now_ms())),
7338                        expires_after_seconds: None,
7339                        creation_result: result.clone(),
7340                    });
7341                manifest
7342                    .active_crons
7343                    .sort_by(|left, right| left.id.cmp(&right.id));
7344                (result, false)
7345            }
7346            CLAUDE_CRON_DELETE => {
7347                let Some(id) = object.get("id").and_then(serde_json::Value::as_str) else {
7348                    return ("Error: CronDelete requires string `id`".to_string(), true);
7349                };
7350                let Some(index) = manifest.active_crons.iter().position(|job| job.id == id) else {
7351                    return (
7352                        format!("Error: unknown {state} Claude cron job `{id}`"),
7353                        true,
7354                    );
7355                };
7356                manifest.active_crons.remove(index);
7357                (
7358                    format!("Cancelled job {id}. The job was PAUSED; no execution occurred."),
7359                    false,
7360                )
7361            }
7362            CLAUDE_SCHEDULE_WAKEUP => {
7363                let Some(delay_seconds) = object
7364                    .get("delaySeconds")
7365                    .and_then(serde_json::Value::as_u64)
7366                else {
7367                    return (
7368                        "Error: ScheduleWakeup requires integer `delaySeconds`".to_string(),
7369                        true,
7370                    );
7371                };
7372                let reason = object
7373                    .get("reason")
7374                    .and_then(serde_json::Value::as_str)
7375                    .map(str::to_string);
7376                let prompt = object
7377                    .get("prompt")
7378                    .and_then(serde_json::Value::as_str)
7379                    .map(str::to_string);
7380                let now = now_ms();
7381                let created_at = Some(supercode_interchange::sidecar::ms_to_rfc3339(now));
7382                // The instant the wakeup asks for, recorded as the request
7383                // made it. No timer consults it here.
7384                let delay_ms = i64::try_from(delay_seconds)
7385                    .unwrap_or(i64::MAX)
7386                    .saturating_mul(1_000);
7387                let scheduled_for =
7388                    supercode_interchange::sidecar::ms_to_rfc3339(now.saturating_add(delay_ms));
7389                let result = format!(
7390                    "Next wakeup recorded for {scheduled_for} (in {delay_seconds}s) in PAUSED \
7391                     state. The request replaced the prior wakeup in the manifest, but no timer \
7392                     is running and it will not execute."
7393                );
7394                manifest.pending_wakeups.clear();
7395                manifest
7396                    .pending_wakeups
7397                    .push(crate::claude_runtime_state::ClaudeWakeup {
7398                        tool_use_id: call.id.clone(),
7399                        delay_seconds,
7400                        reason,
7401                        prompt,
7402                        created_at,
7403                        scheduled_for: Some(scheduled_for),
7404                        creation_result: result.clone(),
7405                    });
7406                (result, false)
7407            }
7408            _ => unreachable!("runtime tool dispatch is name-gated"),
7409        }
7410    }
7411
7412    /// The `subagent_status` schema (P5-3, D3 "background+resume").
7413    fn subagent_status_schema() -> ToolSchema {
7414        ToolSchema {
7415            name: SUBAGENT_STATUS.to_string(),
7416            description: "Check on (and, once finished, retrieve the result of) a background \
7417                subagent spawned via spawn_subagent with background=true. Pass the \
7418                `subagent_id` that spawn returned."
7419                .to_string(),
7420            parameters: serde_json::json!({
7421                "type": "object",
7422                "properties": {
7423                    "subagent_id": {
7424                        "type": "string",
7425                        "description": "The id `spawn_subagent` returned when this subagent \
7426                            was spawned."
7427                    }
7428                },
7429                "required": ["subagent_id"],
7430                "additionalProperties": false
7431            }),
7432        }
7433    }
7434
7435    /// The `send_message` schema.
7436    fn send_message_schema() -> ToolSchema {
7437        ToolSchema {
7438            name: SEND_MESSAGE.to_string(),
7439            description: "Send a message to another agent: one of your background subagents \
7440                that is STILL RUNNING (by the id `spawn_subagent` returned; it receives it at the \
7441                start of its next step), or another session of any harness, on this machine or an \
7442                enrolled one (by its name, name@machine, or sc: address; `supercode message list` \
7443                shows them). A session's reply comes back to you as a message. Send only what \
7444                asks something or carries a result; no acknowledgements."
7445                .to_string(),
7446            parameters: serde_json::json!({
7447                "type": "object",
7448                "properties": {
7449                    "to": {
7450                        "type": "string",
7451                        "description": "A running subagent's id, or a session's name, name@machine or sc: address."
7452                    },
7453                    "message": {
7454                        "type": "string",
7455                        "description": "What to tell it."
7456                    },
7457                    "notify_when_idle": {
7458                        "type": "boolean",
7459                        "description": "For a session: also get one notice when its next turn ends."
7460                    }
7461                },
7462                "required": ["to", "message"],
7463                "additionalProperties": false
7464            }),
7465        }
7466    }
7467
7468    /// BP-7: the `subagent_resume` schema.
7469    fn subagent_resume_schema() -> ToolSchema {
7470        ToolSchema {
7471            name: SUBAGENT_RESUME.to_string(),
7472            description: "Continue a subagent that has already FINISHED, with its own                 previous conversation restored, so it keeps everything it learned instead                 of being briefed again from scratch. Pass the id it was spawned with and                 the next task."
7473                .to_string(),
7474            parameters: serde_json::json!({
7475                "type": "object",
7476                "properties": {
7477                    "subagent_id": {
7478                        "type": "string",
7479                        "description": "The id of a subagent that has already finished."
7480                    },
7481                    "task": {
7482                        "type": "string",
7483                        "description": "What the resumed subagent should do next."
7484                    }
7485                },
7486                "required": ["subagent_id", "task"],
7487                "additionalProperties": false
7488            }),
7489        }
7490    }
7491
7492    /// BP-7 (catalog §4a "Background subagents + resume"): deliver a
7493    /// message into a still-running background child's mailbox.
7494    ///
7495    /// The mailbox is the child's own `SteerInbox` — the seam P4b built for
7496    /// mid-turn steering, which is writable while the child's turn holds
7497    /// `&mut Agent`. So delivery ordering is already defined: the message
7498    /// arrives at the top of the child's next loop iteration, i.e. after
7499    /// whatever tool calls it is currently running, per its
7500    /// `steering_mode`. A child that has already FINISHED is refused with
7501    /// a pointer at `subagent_resume`, which is the operation for that
7502    /// case — never silently dropped.
7503    async fn run_send_message(&mut self, call: &supercode_interchange::ToolCall) -> (String, bool) {
7504        let args = match call.function.parsed_arguments() {
7505            Ok(v) => v,
7506            Err(e) => {
7507                let err = Error::InvalidArguments {
7508                    tool: SEND_MESSAGE.to_string(),
7509                    message: e.to_string(),
7510                };
7511                return (format!("Error: {err}"), true);
7512            }
7513        };
7514        let Some(id) = args.get("to").and_then(serde_json::Value::as_str) else {
7515            let err = Error::InvalidArguments {
7516                tool: SEND_MESSAGE.to_string(),
7517                message: "`to` is required".to_string(),
7518            };
7519            return (format!("Error: {err}"), true);
7520        };
7521        let message = args
7522            .get("message")
7523            .and_then(serde_json::Value::as_str)
7524            .unwrap_or("");
7525        if message.is_empty() {
7526            let err = Error::InvalidArguments {
7527                tool: SEND_MESSAGE.to_string(),
7528                message: "`message` is required and must be non-empty".to_string(),
7529            };
7530            return (format!("Error: {err}"), true);
7531        }
7532        let Some(entry) = self.background_subagents.get(id) else {
7533            // Not one of this agent's children: another session.
7534            let notify_when_idle = args
7535                .get("notify_when_idle")
7536                .and_then(serde_json::Value::as_bool)
7537                .unwrap_or(false);
7538            return send_to_session(id, message, notify_when_idle).await;
7539        };
7540        if entry.handle.is_finished() {
7541            let out = serde_json::json!({
7542                "subagent_id": id,
7543                "status": "finished",
7544                "delivered": false,
7545                "hint": "this subagent already finished — collect it with subagent_status,                          then continue it with subagent_resume",
7546            });
7547            return (out.to_string(), false);
7548        }
7549        entry
7550            .mailbox
7551            .lock()
7552            .unwrap_or_else(std::sync::PoisonError::into_inner)
7553            .queue_unchecked(message.to_string());
7554        let out = serde_json::json!({
7555            "subagent_id": id,
7556            "status": "running",
7557            "delivered": true,
7558        });
7559        (out.to_string(), false)
7560    }
7561
7562    /// BP-7 (catalog §4a "Background subagents + resume": "resumable with
7563    /// context intact"): continue a finished child over its OWN transcript.
7564    ///
7565    /// The context comes from the reap (kept in-process) or, for a session
7566    /// that attached a subagent store, from that child's persisted
7567    /// `<parent>.subagents/<id>.sidecar.jsonl`. Either way the resumed
7568    /// child is rebuilt through `build_child_config` from the SAME named
7569    /// definition it was spawned with, so its permission posture on resume
7570    /// is the one it had originally — never a fresh, looser default.
7571    async fn run_subagent_resume(
7572        &mut self,
7573        call: &supercode_interchange::ToolCall,
7574    ) -> (String, bool) {
7575        let args = match call.function.parsed_arguments() {
7576            Ok(v) => v,
7577            Err(e) => {
7578                let err = Error::InvalidArguments {
7579                    tool: SUBAGENT_RESUME.to_string(),
7580                    message: e.to_string(),
7581                };
7582                return (format!("Error: {err}"), true);
7583            }
7584        };
7585        let Some(id) = args
7586            .get("subagent_id")
7587            .and_then(serde_json::Value::as_str)
7588            .map(String::from)
7589        else {
7590            let err = Error::InvalidArguments {
7591                tool: SUBAGENT_RESUME.to_string(),
7592                message: "`subagent_id` is required".to_string(),
7593            };
7594            return (format!("Error: {err}"), true);
7595        };
7596        let task = args
7597            .get("task")
7598            .and_then(serde_json::Value::as_str)
7599            .unwrap_or("")
7600            .to_string();
7601        if task.is_empty() {
7602            let err = Error::InvalidArguments {
7603                tool: SUBAGENT_RESUME.to_string(),
7604                message: "`task` is required and must be non-empty".to_string(),
7605            };
7606            return (format!("Error: {err}"), true);
7607        }
7608        if self
7609            .background_subagents
7610            .get(&id)
7611            .is_some_and(|e| !e.handle.is_finished())
7612        {
7613            let err = Error::tool(
7614                SUBAGENT_RESUME,
7615                format!(
7616                    "subagent `{id}` is still running — send it a message with                      send_message, or collect it with subagent_status first"
7617                ),
7618            );
7619            return (format!("Error: {err}"), true);
7620        }
7621        let lineage = self
7622            .subagent_store
7623            .as_ref()
7624            .and_then(|(store, parent)| store.load_subagent_lineage(parent, &id).ok().flatten());
7625        let Some(prior) = self.prior_subagent_transcript(&id) else {
7626            let err = Error::SubagentNotFound(id.clone());
7627            return (format!("Error: {err}"), true);
7628        };
7629
7630        let agent_type = lineage.as_ref().and_then(|l| l.agent_type.clone());
7631        let definition = agent_type
7632            .as_ref()
7633            .and_then(|name| self.config.subagents_definitions.get(name).cloned());
7634        let Some(guard) = crate::subagents::try_acquire(
7635            &self.subagent_concurrency_gauge,
7636            self.config.subagents_max_concurrent,
7637        ) else {
7638            let err = Error::SubagentConcurrencyExceeded {
7639                max_concurrent: self.config.subagents_max_concurrent,
7640            };
7641            return (format!("Error: {err}"), true);
7642        };
7643        let child_config = self.build_child_config(
7644            definition.as_ref(),
7645            None,
7646            lineage
7647                .as_ref()
7648                .map(|l| l.model.clone())
7649                .or_else(|| definition.as_ref().and_then(|d| d.model.clone())),
7650        );
7651        let mut child = Agent::with_provider_arc(child_config, self.provider.clone());
7652        child.subagent_depth = self.subagent_depth + 1;
7653        child.subagent_concurrency_gauge = self.subagent_concurrency_gauge.clone();
7654        // Context intact: the child's own prior messages, appended after
7655        // its (re-derived, identical) system prompt.
7656        child.history.extend(prior);
7657
7658        let result = child.send(task).await;
7659        let transcript = child.history()[1..].to_vec();
7660        if let Some(lineage) = &lineage {
7661            self.persist_subagent_transcript(&id, lineage, &transcript);
7662        }
7663        self.reaped_subagents.insert(id.clone(), transcript);
7664        drop(guard);
7665        match result {
7666            Ok(text) => {
7667                let out = serde_json::json!({
7668                    "subagent_id": id,
7669                    "status": "done",
7670                    "resumed": true,
7671                    "result": text,
7672                });
7673                (out.to_string(), false)
7674            }
7675            Err(e) => {
7676                let out = serde_json::json!({
7677                    "subagent_id": id,
7678                    "status": "error",
7679                    "resumed": true,
7680                    "message": e.to_string(),
7681                });
7682                (out.to_string(), true)
7683            }
7684        }
7685    }
7686
7687    /// BP-7 (catalog §4a "Named agent definitions as data"): the child
7688    /// `Config` a `spawn_subagent` of `agent_type` would build — the
7689    /// resolved posture a named definition actually produces, including
7690    /// its [`crate::subagents::AgentPermissions`] bundle applied through
7691    /// the tightening-only rules. `None` when no definition of that name
7692    /// is registered or discovered.
7693    ///
7694    /// Exposed so a caller (and this build's tests) can ask what a named
7695    /// agent WOULD run as without spawning it and paying for a turn.
7696    pub fn child_config_for_agent_type(&self, agent_type: &str) -> Option<Config> {
7697        let definition = self.config.subagents_definitions.get(agent_type)?.clone();
7698        Some(self.build_child_config(Some(&definition), None, definition.model.clone()))
7699    }
7700
7701    /// BP-7 (catalog §4a "Background subagents + resume"): the ids of
7702    /// children that have finished and been reaped, and can therefore be
7703    /// continued with [`SUBAGENT_RESUME`].
7704    pub fn reaped_subagent_ids(&self) -> Vec<String> {
7705        let mut ids: Vec<String> = self.reaped_subagents.keys().cloned().collect();
7706        ids.sort();
7707        ids
7708    }
7709
7710    /// BP-7: a finished child's own messages — from the in-process reap
7711    /// cache first, then this session's subagent store.
7712    fn prior_subagent_transcript(&self, id: &str) -> Option<Vec<ChatMessage>> {
7713        if let Some(messages) = self.reaped_subagents.get(id) {
7714            return Some(messages.clone());
7715        }
7716        let (store, parent) = self.subagent_store.as_ref()?;
7717        let jsonl = store.load_subagent_transcript(parent, id).ok()??;
7718        let session = supercode_interchange::session::Session::from_sidecar_str(&jsonl).ok()?;
7719        Some(
7720            session
7721                .messages
7722                .into_iter()
7723                .filter(|m| m.role != supercode_interchange::Role::System)
7724                .collect(),
7725        )
7726    }
7727
7728    /// Build the CHILD `Config` a `spawn_subagent` call constructs its
7729    /// [`Agent`] from. The whole point of this method (§5.3-style
7730    /// "monotonic posture", build-brief "a subagent inherits or narrows —
7731    /// never widens — the parent's permission posture"): every field that
7732    /// governs what the child is ALLOWED to do (sandbox, approval,
7733    /// tool_overrides, deny/allow patterns, protected paths, the subagents
7734    /// caps themselves) is copied VERBATIM from `self.config` — never
7735    /// loosened — and the only NARROWING lever is `definition.tools`
7736    /// (intersected with whatever the parent already had enabled, never
7737    /// unioned in anything new).
7738    ///
7739    /// P5-3 safety hardening (Fable-5 review, LOW-MEDIUM "child safety-limit
7740    /// inheritance"): the monotonic-posture guarantee above was, before this
7741    /// fix, scoped to PERMISSION fields only — a child could still silently
7742    /// get a LOOSER safety BUDGET/BREAKER than its parent, because
7743    /// `max_total_output_tokens`/`max_tool_output_bytes`/`max_tokens`/
7744    /// `doom_loop_threshold`/`edit_file_require_read_before_edit` were never
7745    /// copied and so fell back to `Config::default()`'s (looser/uncapped)
7746    /// values on every spawn regardless of what the parent had configured.
7747    /// These are now copied verbatim alongside the permission-posture
7748    /// fields — a parent that capped its own output/tool-output/doom-loop
7749    /// exposure, or required read-before-edit, gets a child that is bound
7750    /// by the exact same ceiling, never a wider one.
7751    ///
7752    /// **Full field-by-field accounting** (every [`Config`] field, so this
7753    /// doc comment stays the single place that answers "did we forget
7754    /// one?"): fields already copied above/below this note (permission
7755    /// posture: `sandbox`/`approval`/`tool_overrides`/`auto_approved_tools`/
7756    /// `tool_deny_patterns`/`tool_allow_patterns`/`permissions_enabled`/
7757    /// `permissions_ask_patterns`/`permissions_protected_paths`/
7758    /// `network_policy`/`core_tools_enabled`/`module_registry`/
7759    /// `module_activation`/every `subagents_*` field; safety limits:
7760    /// `max_iterations`/`max_total_output_tokens`/`max_tool_output_bytes`/
7761    /// `max_tokens`/`doom_loop_threshold`/`edit_file_require_read_before_edit`;
7762    /// identity/transport: `model`/`system_prompt`/`cwd`/`base_url`/
7763    /// `api_key`/`api_key_env`/`api_key_cmd`) are the ones that gate
7764    /// harm/spend/hazard exposure. Every OTHER field is deliberately left at
7765    /// `Config::default()` because none of them is a safety ceiling the
7766    /// child could "loosen" by missing it:
7767    /// - `temperature`/`effort`/`response_format`/`extra_body`/`extra_headers`/
7768    ///   `tool_advertising`/`tool_schema_tier`/`cache_plan`/`cache_warnings`/
7769    ///   `reduction_policy`/
7770    ///   `session_*`/`small_model`/`model_fallback`/`env_context`/
7771    ///   `project_root_markers`/`project_doc_max_bytes`/`instruction_imports`/
7772    ///   `retry_*`/`compaction_*`/`auto_title`/`steering_mode`/
7773    ///   `follow_up_mode`/`read_file_multimodal`/`edit_file_notebook_aware`/
7774    ///   `shell_env_snapshot`/`nested_instructions`/`model_switch_allow_switch`/
7775    ///   `context_injections`/`context_injection_blocks`/`parallel_tool_calls`
7776    ///   are behavior/cost-shaping or presentation knobs, not hard guards —
7777    ///   a child defaulting on any of these can do LESS (e.g. no multimodal
7778    ///   read, no notebook-aware edits, no proactive compaction) or the same,
7779    ///   never something the parent hadn't already exposed it to. Several
7780    ///   default to their OFF/conservative state (`false`/`None`), which is
7781    ///   the tight direction, not the loose one.
7782    /// - `additional_dirs`: governs which extra roots are reachable at all
7783    ///   (`presets.rs`'s `[core] additional_dirs` note) — a child that
7784    ///   doesn't inherit it has FEWER reachable roots than its parent, i.e.
7785    ///   strictly tighter, never looser.
7786    /// - `load_project_context`: whether instruction files are auto-loaded
7787    ///   into the system prompt — a read-time convenience, not an access
7788    ///   grant (`sandbox`/`permissions_protected_paths` already gate actual
7789    ///   file access).
7790    /// - `prompts`: named `/slash` command templates for THIS agent's own
7791    ///   user-facing input surface, not something the model can invoke
7792    ///   against the child's tool surface.
7793    /// - `stop_gate`/`post_tool_hook`/`approval_handler`/`event_sink`:
7794    ///   code-only `Box<dyn Fn>` callbacks (see the `pre_tool_hook` note
7795    ///   immediately below — same non-`Clone` shape) that are observational
7796    ///   or terminate-only, not a call-time veto over what a tool is allowed
7797    ///   to do; `approval_handler` specifically is ALREADY documented at
7798    ///   this method's call site (`Self::run_spawn_subagent`) as
7799    ///   intentionally never set here — a foreground child gets no handler
7800    ///   by design, an embedder installs its own after spawn if it wants
7801    ///   one.
7802    ///
7803    /// **`pre_tool_hook` cannot propagate, and this is deliberate + named,
7804    /// not a silent gap**: `Config::pre_tool_hook` is a `Box<dyn Fn(&str,
7805    /// &serde_json::Value) -> Option<String> + Send + Sync>` — an
7806    /// embedder's own call-time veto over every tool call. `Box<dyn Fn>` is
7807    /// not `Clone` (there is no generic way to duplicate an opaque closure),
7808    /// so it genuinely CANNOT be copied into a child `Config` the way every
7809    /// `Clone`-able field above is — there is no fix that makes this one
7810    /// "verbatim copy" like the others. An embedder relying on a
7811    /// `pre_tool_hook` veto reaching spawned children as well as the parent
7812    /// MUST re-install one on the child explicitly (e.g. via a
7813    /// `spawn_subagent`-adjacent hook of their own, or by not relying on
7814    /// `pre_tool_hook` alone for anything safety-critical across a spawn
7815    /// boundary) — named here so this is a documented contract, not a gap
7816    /// an embedder discovers by a child silently misbehaving.
7817    fn build_child_config(
7818        &self,
7819        definition: Option<&crate::subagents::NamedAgentDefinition>,
7820        inline_system_prompt: Option<String>,
7821        model_override: Option<String>,
7822    ) -> Config {
7823        let system_prompt = definition
7824            .map(|d| d.system_prompt.clone())
7825            .filter(|s| !s.is_empty())
7826            .or(inline_system_prompt)
7827            .unwrap_or_else(|| self.config.system_prompt.clone());
7828        let model = model_override.unwrap_or_else(|| self.config.model.clone());
7829
7830        let mut child = Config::builder()
7831            .model(model)
7832            .system_prompt(system_prompt)
7833            .cwd(self.config.cwd.clone())
7834            // Monotonic: verbatim, never loosened.
7835            .sandbox(self.config.sandbox)
7836            .approval(self.config.approval)
7837            .max_iterations(self.config.max_iterations)
7838            .build();
7839        child.base_url = self.config.base_url.clone();
7840        child.api_key = self.config.api_key.clone();
7841        child.api_key_env = self.config.api_key_env.clone();
7842        child.api_key_cmd = self.config.api_key_cmd.clone();
7843        // P5-3 safety hardening (Fable-5 review, LOW-MEDIUM "child
7844        // safety-limit inheritance"): the monotonic-posture spirit extends
7845        // to safety BUDGETS/BREAKERS, not just permissions — a child must
7846        // not get a looser cap/breaker than its parent by simply falling
7847        // back to `Config::default()`'s (looser) values. See this method's
7848        // doc comment for the full field-by-field accounting.
7849        child.max_total_output_tokens = self.config.max_total_output_tokens;
7850        child.max_tool_output_bytes = self.config.max_tool_output_bytes;
7851        child.max_tokens = self.config.max_tokens;
7852        child.doom_loop_threshold = self.config.doom_loop_threshold;
7853        child.edit_file_require_read_before_edit = self.config.edit_file_require_read_before_edit;
7854        // Monotonic tool posture: start from the PARENT's own overrides
7855        // (so anything the parent already disabled stays disabled), then
7856        // narrow further if a named definition restricts the tool set.
7857        child.tool_overrides = self.config.tool_overrides.clone();
7858        child.auto_approved_tools = self.config.auto_approved_tools.clone();
7859        child.tool_deny_patterns = self.config.tool_deny_patterns.clone();
7860        child.tool_allow_patterns = self.config.tool_allow_patterns.clone();
7861        child.permissions_enabled = self.config.permissions_enabled;
7862        child.permissions_ask_patterns = self.config.permissions_ask_patterns.clone();
7863        child.permissions_protected_paths = self.config.permissions_protected_paths.clone();
7864        child.network_policy = self.config.network_policy.clone();
7865        // P5-10 (§2 module 12): same monotonic-posture treatment as
7866        // `sandbox`/`approval` above — a subagent must inherit its
7867        // parent's OS-sandbox posture verbatim, never a looser
7868        // `Config::default()` fallback (`sandbox_os_enabled: None`,
7869        // `escalation: Deny`, `env_policy: Inherit` would otherwise be
7870        // right back to "confine only when the tier itself says so" for a
7871        // child whose parent explicitly forced the backstop on/off).
7872        child.sandbox_os_enabled = self.config.sandbox_os_enabled;
7873        child.sandbox_escalation = self.config.sandbox_escalation;
7874        child.sandbox_env_policy = self.config.sandbox_env_policy;
7875        if let Some(def) = definition {
7876            if let Some(allowed) = &def.tools {
7877                for name in &self.config.core_tools_enabled {
7878                    if !allowed.iter().any(|t| t == name) {
7879                        child
7880                            .tool_overrides
7881                            .entry(name.clone())
7882                            .or_default()
7883                            .enabled = Some(false);
7884                    }
7885                }
7886            }
7887            // BP-7 (catalog §4a "Named agent definitions as data": the
7888            // `permissions` component of `prompt+model+tools+permissions`).
7889            // Every arm below can only TIGHTEN — the two policy values go
7890            // through the SAME strictness ranks `configfile::
7891            // clamp_project_permissions` uses for the untrusted project
7892            // layer (a looser value is ignored, never honored), the
7893            // auto-approve list is INTERSECTED with the parent's, and the
7894            // deny list is a union. A definition may come from a
7895            // `.claude/agents/*.md` file in the repo, so it sits at the
7896            // project trust tier and must never be an escalation door.
7897            if let Some(perms) = &def.permissions {
7898                if let Some(approval) = perms.approval {
7899                    if crate::configfile::approval_rank(approval)
7900                        < crate::configfile::approval_rank(child.approval)
7901                    {
7902                        child.approval = approval;
7903                    }
7904                }
7905                if let Some(sandbox) = perms.sandbox {
7906                    if crate::configfile::sandbox_rank(sandbox)
7907                        < crate::configfile::sandbox_rank(child.sandbox)
7908                    {
7909                        child.sandbox = sandbox;
7910                    }
7911                }
7912                if let Some(allowed) = &perms.auto_approved_tools {
7913                    child
7914                        .auto_approved_tools
7915                        .retain(|tool| allowed.iter().any(|a| a == tool));
7916                }
7917                for pattern in &perms.deny {
7918                    if !child.tool_deny_patterns.iter().any(|p| p == pattern) {
7919                        child.tool_deny_patterns.push(pattern.clone());
7920                    }
7921                }
7922            }
7923        }
7924        child.core_tools_enabled = self.config.core_tools_enabled.clone();
7925        child.module_registry = self.config.module_registry;
7926        child.module_activation = self.config.module_activation.clone();
7927        // The subagents module itself never widens either: a child spawned
7928        // at depth d+1 inherits the SAME caps (never a looser depth/
7929        // concurrency/background posture than its own parent).
7930        child.subagents_enabled = self.config.subagents_enabled;
7931        child.subagents_max_depth = self.config.subagents_max_depth;
7932        child.subagents_max_concurrent = self.config.subagents_max_concurrent;
7933        child.subagents_background = self.config.subagents_background;
7934        child.subagents_background_prompts = self.config.subagents_background_prompts;
7935        child.subagents_claude_agent_alias = self.config.subagents_claude_agent_alias;
7936        child.subagents_definitions = self.config.subagents_definitions.clone();
7937        child.subagent_depth = self.subagent_depth + 1;
7938        child
7939    }
7940
7941    /// Execute the `spawn_subagent` intrinsic (P5-3, §2 module 9). See
7942    /// `Self::build_child_config` for the monotonic-posture guarantee and
7943    /// `crate::subagents` for the depth/concurrency resource bounds and the
7944    /// §2.2 C6 background-policy enforcement.
7945    async fn run_spawn_subagent(
7946        &mut self,
7947        call: &supercode_interchange::ToolCall,
7948    ) -> (String, bool) {
7949        // BP-11: `subagent_start`/`subagent_stop` bracket a call that passed
7950        // the same validation the runner applies (subagents on, non-empty
7951        // task); a refused call fires neither.
7952        let task = if self.config.subagents_enabled {
7953            call.function
7954                .parsed_arguments()
7955                .ok()
7956                .and_then(|v| {
7957                    v.get("task")
7958                        .and_then(serde_json::Value::as_str)
7959                        .map(str::to_string)
7960                })
7961                .filter(|t| !t.is_empty())
7962        } else {
7963            None
7964        };
7965        if let Some(task) = &task {
7966            self.fire_lifecycle(&crate::config::LifecycleEvent::SubagentStart {
7967                task: task.clone(),
7968            });
7969        }
7970        let (output, is_error) = self.run_spawn_subagent_inner(call).await;
7971        if let Some(task) = task {
7972            self.fire_lifecycle(&crate::config::LifecycleEvent::SubagentStop {
7973                task,
7974                is_error,
7975                output_len: output.len(),
7976            });
7977        }
7978        (output, is_error)
7979    }
7980
7981    /// Hands a lifecycle moment to the installed observer, if any (BP-11).
7982    fn fire_lifecycle(&self, event: &crate::config::LifecycleEvent) {
7983        if let Some(hook) = self.config.lifecycle_hook.as_ref() {
7984            hook(event);
7985        }
7986    }
7987
7988    /// Installs the lifecycle observer (compaction and subagent boundaries).
7989    pub fn set_lifecycle_hook(&mut self, hook: crate::config::LifecycleHook) {
7990        self.config.lifecycle_hook = Some(hook);
7991    }
7992
7993    async fn run_spawn_subagent_inner(
7994        &mut self,
7995        call: &supercode_interchange::ToolCall,
7996    ) -> (String, bool) {
7997        if !self.config.subagents_enabled {
7998            let err = Error::UnknownTool(SPAWN_SUBAGENT.to_string());
7999            return (format!("Error: {err}"), true);
8000        }
8001        let args = match call.function.parsed_arguments() {
8002            Ok(v) => v,
8003            Err(e) => {
8004                let err = Error::InvalidArguments {
8005                    tool: SPAWN_SUBAGENT.to_string(),
8006                    message: e.to_string(),
8007                };
8008                return (format!("Error: {err}"), true);
8009            }
8010        };
8011        let task = args
8012            .get("task")
8013            .and_then(serde_json::Value::as_str)
8014            .unwrap_or("")
8015            .to_string();
8016        if task.is_empty() {
8017            let err = Error::InvalidArguments {
8018                tool: SPAWN_SUBAGENT.to_string(),
8019                message: "`task` is required and must be non-empty".to_string(),
8020            };
8021            return (format!("Error: {err}"), true);
8022        }
8023        let agent_type = args
8024            .get("agent_type")
8025            .and_then(serde_json::Value::as_str)
8026            .map(String::from);
8027        let inline_system_prompt = args
8028            .get("system_prompt")
8029            .and_then(serde_json::Value::as_str)
8030            .map(String::from);
8031        let background = args
8032            .get("background")
8033            .and_then(serde_json::Value::as_bool)
8034            .unwrap_or(false);
8035        let requested_model = args
8036            .get("model")
8037            .and_then(serde_json::Value::as_str)
8038            .map(|model| crate::model_catalog::resolve_alias(model));
8039
8040        let definition = match &agent_type {
8041            Some(name) => match self.config.subagents_definitions.get(name) {
8042                Some(d) => Some(d.clone()),
8043                None => {
8044                    let err = Error::SubagentDefinitionNotFound(name.clone());
8045                    return (format!("Error: {err}"), true);
8046                }
8047            },
8048            None => None,
8049        };
8050
8051        if background {
8052            if !self.config.subagents_background {
8053                let err = Error::tool(
8054                    SPAWN_SUBAGENT,
8055                    "background=true requires capabilities.subagents.background = true",
8056                );
8057                return (format!("Error: {err}"), true);
8058            }
8059            // §2.2 C6, defensive re-check (belt-and-suspenders — see
8060            // `Error::SubagentBackgroundPolicyMissing`'s doc comment for why
8061            // this can't just trust the resolver already checked it).
8062            if self.config.subagents_background_prompts.is_none() {
8063                let err = Error::SubagentBackgroundPolicyMissing;
8064                return (format!("Error: {err}"), true);
8065            }
8066        }
8067
8068        // Resource bounds (fail-closed): depth first (cheap, no side
8069        // effect on failure), THEN concurrency (holds a slot — must be the
8070        // LAST check before actually spawning, so a refused spawn never
8071        // leaves a stray slot held).
8072        if let Err(e) =
8073            crate::subagents::check_depth(self.subagent_depth, self.config.subagents_max_depth)
8074        {
8075            return (format!("Error: {e}"), true);
8076        }
8077        let Some(guard) = crate::subagents::try_acquire(
8078            &self.subagent_concurrency_gauge,
8079            self.config.subagents_max_concurrent,
8080        ) else {
8081            let err = Error::SubagentConcurrencyExceeded {
8082                max_concurrent: self.config.subagents_max_concurrent,
8083            };
8084            return (format!("Error: {err}"), true);
8085        };
8086
8087        let child_id = next_subagent_id();
8088        let child_config = self.build_child_config(
8089            definition.as_ref(),
8090            inline_system_prompt,
8091            requested_model.or_else(|| definition.as_ref().and_then(|d| d.model.clone())),
8092        );
8093        let child_model = child_config.model.clone();
8094        let mut child = Agent::with_provider_arc(child_config, self.provider.clone());
8095        child.subagent_depth = self.subagent_depth + 1;
8096        child.subagent_concurrency_gauge = self.subagent_concurrency_gauge.clone();
8097
8098        // §2.2 C6: a background child NEVER gets a BLOCKING-BY-DEFAULT
8099        // interactive approval handler — either no handler at all
8100        // (`AutoPolicy`: the engine's pre-existing "no handler ⇒ deny"
8101        // fail-closed default), or (`Parent`) the never-blocking
8102        // `ParentQueueApprovalHandler`, UNLESS a `tui` embedder has
8103        // installed [`Self::child_approval_handler_factory`] (P5-4), in
8104        // which case THAT builds the handler instead — see
8105        // [`Self::set_child_approval_handler_factory`]'s doc comment for
8106        // why this can't escalate past what the rule engine already routed
8107        // to `Ask`. A foreground child also gets no handler here (today's
8108        // existing default posture; an embedder that wants an interactive
8109        // child installs its own via `set_permissions_approval_handler`
8110        // after this call returns, out of this method's scope).
8111        if background {
8112            if let Some(crate::subagents::BackgroundPromptsPolicy::Parent) =
8113                self.config.subagents_background_prompts
8114            {
8115                let handler: std::sync::Arc<dyn crate::permissions::PermissionsApprovalHandler> =
8116                    match &self.child_approval_handler_factory {
8117                        Some(factory) => {
8118                            factory(child_id.clone(), self.pending_child_approvals.clone())
8119                        }
8120                        None => std::sync::Arc::new(crate::subagents::ParentQueueApprovalHandler {
8121                            child_agent_id: child_id.clone(),
8122                            queue: self.pending_child_approvals.clone(),
8123                        }),
8124                    };
8125                child.ctx.sandbox_approval_handler =
8126                    Some(crate::sandbox::SandboxApprovalHandler(handler.clone()));
8127                child.permissions_approval_handler = Some(handler);
8128            }
8129        }
8130
8131        let lineage = crate::subagents::SubagentLineage {
8132            child_agent_id: child_id.clone(),
8133            parent_session_id: self.subagent_store.as_ref().map(|(_, name)| name.clone()),
8134            parent_tool_use_id: call.id.clone(),
8135            depth: self.subagent_depth + 1,
8136            agent_type: agent_type.clone(),
8137            task: task.clone(),
8138            background,
8139            spawned_at_ms: now_ms(),
8140            model: child_model,
8141        };
8142        if let Some((store, parent_name)) = &self.subagent_store {
8143            let _ = store.save_subagent_lineage(parent_name, &child_id, &lineage);
8144        }
8145
8146        if background {
8147            let spawned_task_text = task.clone();
8148            // BP-7: captured BEFORE `child` moves into the task — this is
8149            // the handle `send_message` writes into.
8150            let mailbox = child.steer_queue_handle();
8151            self.background_subagents.insert(
8152                child_id.clone(),
8153                BackgroundSubagent {
8154                    handle: tokio::spawn(async move {
8155                        // The concurrency slot lives for exactly as long as
8156                        // this future runs — moved in here, dropped when the
8157                        // child's `send` (and this future) finishes.
8158                        let _guard = guard;
8159                        let result = child.send(spawned_task_text).await;
8160                        let transcript = child.history()[1..].to_vec();
8161                        (child_id, result, transcript)
8162                    }),
8163                    task,
8164                    agent_type,
8165                    started_at_ms: lineage.spawned_at_ms,
8166                    mailbox,
8167                },
8168            );
8169            let out = serde_json::json!({
8170                "subagent_id": lineage.child_agent_id,
8171                "status": "spawned",
8172                "background": true,
8173            });
8174            return (out.to_string(), false);
8175        }
8176
8177        // Foreground: run to completion now, guard held until this
8178        // function returns (then drops, freeing the slot).
8179        let result = child.send(task).await;
8180        let transcript = child.history()[1..].to_vec();
8181        self.persist_subagent_transcript(&child_id, &lineage, &transcript);
8182        // BP-7: kept in-process so `subagent_resume` can restore this
8183        // child's context even with no session store attached.
8184        self.reaped_subagents
8185            .insert(child_id.clone(), transcript.clone());
8186        drop(guard);
8187        match result {
8188            Ok(text) => (text, false),
8189            Err(e) => (format!("Error: subagent `{child_id}` failed: {e}"), true),
8190        }
8191    }
8192
8193    /// Execute the `subagent_status` intrinsic (P5-3, D3
8194    /// "background+resume"): poll a background child; once its `JoinHandle`
8195    /// is finished, reap it (removing it from `Self::background_subagents`
8196    /// and persisting its transcript, same as the foreground path).
8197    async fn run_subagent_status(
8198        &mut self,
8199        call: &supercode_interchange::ToolCall,
8200    ) -> (String, bool) {
8201        let args = match call.function.parsed_arguments() {
8202            Ok(v) => v,
8203            Err(e) => {
8204                let err = Error::InvalidArguments {
8205                    tool: SUBAGENT_STATUS.to_string(),
8206                    message: e.to_string(),
8207                };
8208                return (format!("Error: {err}"), true);
8209            }
8210        };
8211        let Some(id) = args.get("subagent_id").and_then(serde_json::Value::as_str) else {
8212            let err = Error::InvalidArguments {
8213                tool: SUBAGENT_STATUS.to_string(),
8214                message: "`subagent_id` is required".to_string(),
8215            };
8216            return (format!("Error: {err}"), true);
8217        };
8218        let Some(entry) = self.background_subagents.get(id) else {
8219            let err = Error::SubagentNotFound(id.to_string());
8220            return (format!("Error: {err}"), true);
8221        };
8222        if !entry.handle.is_finished() {
8223            let out = serde_json::json!({
8224                "subagent_id": id,
8225                "status": "pending",
8226                "task": entry.task,
8227                "agent_type": entry.agent_type,
8228                "started_at_ms": entry.started_at_ms,
8229            });
8230            return (out.to_string(), false);
8231        }
8232        // Finished — reap it. `.await` on an already-finished handle
8233        // resolves immediately (never actually blocks).
8234        let entry = self
8235            .background_subagents
8236            .remove(id)
8237            .expect("checked Some above");
8238        let (child_id, result, transcript) = match entry.handle.await {
8239            Ok(v) => v,
8240            Err(join_err) => {
8241                let err = Error::tool(
8242                    SUBAGENT_STATUS,
8243                    format!("subagent `{id}` task panicked: {join_err}"),
8244                );
8245                return (format!("Error: {err}"), true);
8246            }
8247        };
8248        // Re-derive the lineage record for persistence (cheap; the fields
8249        // are all still in hand) — mirrors the foreground path's single
8250        // `persist_subagent_transcript` call site.
8251        if let Some((store, parent_name)) = self.subagent_store.clone() {
8252            if let Ok(Some(lineage)) = store.load_subagent_lineage(&parent_name, &child_id) {
8253                self.persist_subagent_transcript(&child_id, &lineage, &transcript);
8254            }
8255        }
8256        // BP-7: see the foreground path's identical line.
8257        self.reaped_subagents
8258            .insert(child_id.clone(), transcript.clone());
8259        match result {
8260            Ok(text) => {
8261                let out = serde_json::json!({
8262                    "subagent_id": child_id,
8263                    "status": "done",
8264                    "result": text,
8265                });
8266                (out.to_string(), false)
8267            }
8268            Err(e) => {
8269                let out = serde_json::json!({
8270                    "subagent_id": child_id,
8271                    "status": "error",
8272                    "message": e.to_string(),
8273                });
8274                (out.to_string(), true)
8275            }
8276        }
8277    }
8278
8279    /// Execute the `background_exec` intrinsic (P5-6, §2 module 4, D1
8280    /// "background exec"): spawn `args.command` as a detached OS process
8281    /// via `crate::tools::build_sandboxed_sh` — the SAME sandboxed-spawn
8282    /// path [`crate::tools::BashTool::execute`] uses — and return its job
8283    /// id IMMEDIATELY, never the command's output. Gated by the same
8284    /// permission check a foreground `bash` call gets
8285    /// ([`Self::background_permission_denial`]), then a fail-closed
8286    /// concurrency cap ([`Config::tools_background_max_concurrent`]), THEN
8287    /// the actual spawn — in that order, so a refused call never holds a
8288    /// concurrency slot and never touches the process table.
8289    fn run_background_exec(&mut self, call: &supercode_interchange::ToolCall) -> (String, bool) {
8290        let args = match call.function.parsed_arguments() {
8291            Ok(v) => v,
8292            Err(e) => {
8293                let err = Error::InvalidArguments {
8294                    tool: BACKGROUND_EXEC.to_string(),
8295                    message: e.to_string(),
8296                };
8297                return (format!("Error: {err}"), true);
8298            }
8299        };
8300        let command = args
8301            .get("command")
8302            .and_then(serde_json::Value::as_str)
8303            .unwrap_or("")
8304            .to_string();
8305        if command.is_empty() {
8306            let err = Error::InvalidArguments {
8307                tool: BACKGROUND_EXEC.to_string(),
8308                message: "`command` is required and must be non-empty".to_string(),
8309            };
8310            return (format!("Error: {err}"), true);
8311        }
8312
8313        // A job id up front (before spawning) — used both as the audit
8314        // handle for a §2.2 C6 `Parent`-policy queued denial (this call may
8315        // never actually reach the spawn below) and, if the call proceeds,
8316        // as `Self::background_jobs`'s real key.
8317        let job_id = supercode_runtime::background::next_job_id(now_ms());
8318
8319        // Fable-5 review (LOW, "pre_tool_hook + doom-loop don't cover
8320        // background_exec"): this intrinsic is intercepted in
8321        // `Self::prepare_tool_call` and returns before `Self::finish_prepare`
8322        // ever runs, so — unlike a foreground `bash` call — it was reaching
8323        // this real spawn below WITHOUT ever offering `Config.pre_tool_hook`
8324        // a chance to veto it. `background_exec` runs a REAL command (unlike
8325        // the purely in-process meta-intrinsics `tool_search`/
8326        // `expand_reduction`/`sidecar_search`, which have no such gap to
8327        // close), so it belongs behind the same security-relevant veto a
8328        // foreground call gets. Scoped to this one call site — the other
8329        // meta-intrinsics are unchanged. The doom-loop counter
8330        // (`Self::check_doom_loop`) is deliberately NOT wired here: it is a
8331        // foreground repetition breaker keyed on `(self.doom_loop_last_call,
8332        // self.doom_loop_streak)`, a single piece of state shared with the
8333        // ordinary tool-call loop — folding background jobs into that same
8334        // streak would make an interleaved foreground/background pattern
8335        // trip (or fail to trip) the breaker in ways that have nothing to
8336        // do with the foreground loop actually repeating itself; the
8337        // pre_tool_hook veto below is the security-relevant half of this
8338        // fix, the doom-loop breaker is not.
8339        // BP-10: the hook now runs BEFORE this path's permissions gate, the
8340        // same order the foreground path uses — so a rewrite is what the
8341        // rules evaluate and what actually runs, and the hook's
8342        // `Allow`/`Ask` are tiers inside the engine rather than a second
8343        // verdict beside it.
8344        let mut command = command;
8345        let mut hook_decision = crate::config::HookDecision::Pass;
8346        if let Some(hook) = &self.config.pre_tool_hook {
8347            let outcome = hook(BACKGROUND_EXEC, &args);
8348            if outcome.decision == crate::config::HookDecision::Deny {
8349                let reason = outcome.reason.unwrap_or_else(|| "denied".to_string());
8350                return (format!("Error: blocked by pre-tool hook: {reason}"), true);
8351            }
8352            if let Some(rewritten) = outcome.updated_args {
8353                command = rewritten
8354                    .get("command")
8355                    .and_then(|v| v.as_str())
8356                    .unwrap_or(&command)
8357                    .to_string();
8358            }
8359            hook_decision = outcome.decision;
8360        }
8361
8362        if let Some(reason) = self.background_permission_denial(&command, &job_id, hook_decision) {
8363            return (format!("Error: {reason}"), true);
8364        }
8365
8366        let Some(guard) = crate::subagents::try_acquire(
8367            &self.background_concurrency_gauge,
8368            self.config.tools_background_max_concurrent,
8369        ) else {
8370            let err = Error::BackgroundJobConcurrencyExceeded {
8371                max_concurrent: self.config.tools_background_max_concurrent,
8372            };
8373            return (format!("Error: {err}"), true);
8374        };
8375
8376        let mut cmd = match crate::tools::build_sandboxed_sh(&command, &self.ctx) {
8377            Ok(cmd) => cmd,
8378            Err(e) => return (format!("Error: {e}"), true),
8379        };
8380        cmd.current_dir(&self.ctx.cwd)
8381            .stdin(std::process::Stdio::null())
8382            .stdout(std::process::Stdio::piped())
8383            .stderr(std::process::Stdio::piped())
8384            // Defense-in-depth for the "must be killed on drop" guarantee —
8385            // see `impl Drop for Agent`'s doc comment; the EXPLICIT
8386            // `start_kill()` loop there is what makes the guarantee
8387            // provable, this is a second, independent line of defense for
8388            // the same outcome.
8389            .kill_on_drop(true);
8390        // Fable-5 review (HIGH, "grandchildren orphaned on kill AND
8391        // agent-drop"): `Child::start_kill` only signals the DIRECT child.
8392        // A background command that spawns a surviving subprocess (a `&`
8393        // job, a pipeline, a double-forking daemon — or, on macOS, the
8394        // `sandbox-exec` wrapper itself in `build_sandboxed_sh`, whose real
8395        // `sh` and ITS children are all grandchildren of the tracked pid)
8396        // leaves those processes running, reparented to init, after the
8397        // tracked job is "killed". Putting this job in its OWN new process
8398        // group (`pgid == its own pid`, since every descendant inherits the
8399        // group unless it explicitly opts out) lets `kill_job_process_group`
8400        // below signal the WHOLE tree at kill/drop time, not just the one
8401        // pid we happen to be tracking. No portable equivalent on Windows —
8402        // see `kill_job_process_group`'s `#[cfg(not(unix))]` fallback.
8403        #[cfg(unix)]
8404        cmd.process_group(0);
8405        // P4c (`core.shell_env_snapshot`)/P5-10 (`env_policy`):
8406        // `build_sandboxed_sh` (above) already applied both via its own
8407        // `apply_sandbox_env_policy` last step — no separate `ctx.shell_env`
8408        // application here (that would re-add a secret `Filtered`/`None`
8409        // just stripped, on top of the already-`env_clear`'d command).
8410
8411        let mut child = match cmd.spawn() {
8412            Ok(c) => c,
8413            Err(e) => {
8414                drop(guard);
8415                let err = Error::tool(
8416                    BACKGROUND_EXEC,
8417                    format!("failed to spawn background command: {e}"),
8418                );
8419                return (format!("Error: {err}"), true);
8420            }
8421        };
8422        let pid = child.id();
8423        let output = std::sync::Arc::new(supercode_runtime::background::CapturedOutput::new());
8424        let cap = self.config.tools_background_max_output_bytes;
8425        // Fire-and-forget: the reader tasks outlive this method call and
8426        // exit on their own at pipe EOF — see `spawn_output_reader`'s doc
8427        // comment. Bound to named (not `_`) locals only to keep clippy's
8428        // `let_underscore_future` lint quiet; neither handle is awaited or
8429        // aborted anywhere.
8430        if let Some(stdout) = child.stdout.take() {
8431            let _stdout_reader = spawn_output_reader(stdout, output.clone(), cap);
8432        }
8433        if let Some(stderr) = child.stderr.take() {
8434            let _stderr_reader = spawn_output_reader(stderr, output.clone(), cap);
8435        }
8436
8437        let started_at_ms = now_ms();
8438        self.background_jobs.insert(
8439            job_id.clone(),
8440            BackgroundJob {
8441                child,
8442                command: command.clone(),
8443                pid,
8444                output,
8445                started_at_ms,
8446                killed: false,
8447                _guard: guard,
8448            },
8449        );
8450
8451        let out = serde_json::json!({
8452            "job_id": job_id,
8453            "status": "running",
8454            "pid": pid,
8455            "command": command,
8456        });
8457        (out.to_string(), false)
8458    }
8459
8460    /// Execute the `background_status` intrinsic (P5-6, D1 "monitor/event
8461    /// feed"): non-blocking poll of one job's run status (via
8462    /// `Child::try_wait`), drain its output captured since the LAST poll
8463    /// and emit it as an [`AgentEvent::BackgroundOutput`] event (the
8464    /// "event feed" — a real `EventSink` consumer sees each poll's new
8465    /// output live), and return the full captured output (bounded, per
8466    /// [`Config::tools_background_max_output_bytes`]) so far either way.
8467    /// Once the job is terminal (exited or killed), this reaps it — removes
8468    /// it from [`Self::background_jobs`], freeing its concurrency slot —
8469    /// same "poll once more to reap" contract [`Self::run_subagent_status`]
8470    /// already established for background subagents.
8471    fn run_background_status(&mut self, call: &supercode_interchange::ToolCall) -> (String, bool) {
8472        let args = match call.function.parsed_arguments() {
8473            Ok(v) => v,
8474            Err(e) => {
8475                let err = Error::InvalidArguments {
8476                    tool: BACKGROUND_STATUS.to_string(),
8477                    message: e.to_string(),
8478                };
8479                return (format!("Error: {err}"), true);
8480            }
8481        };
8482        let Some(job_id) = args.get("job_id").and_then(serde_json::Value::as_str) else {
8483            let err = Error::InvalidArguments {
8484                tool: BACKGROUND_STATUS.to_string(),
8485                message: "`job_id` is required".to_string(),
8486            };
8487            return (format!("Error: {err}"), true);
8488        };
8489        let job_id = job_id.to_string();
8490
8491        // Scoped so the mutable borrow of `self.background_jobs` ends
8492        // before `self.emit(...)`/`self.background_jobs.remove(...)` below
8493        // need their own (mutable) access to `self`.
8494        let (command, pid, started_at_ms, status, output_so_far, truncated, delta) = {
8495            let Some(job) = self.background_jobs.get_mut(&job_id) else {
8496                let err = Error::BackgroundJobNotFound(job_id);
8497                return (format!("Error: {err}"), true);
8498            };
8499            let status = background_job_status(job);
8500            let (output_so_far, truncated) = job.output.snapshot();
8501            let delta = job.output.drain_new();
8502            (
8503                job.command.clone(),
8504                job.pid,
8505                job.started_at_ms,
8506                status,
8507                output_so_far,
8508                truncated,
8509                delta,
8510            )
8511        };
8512
8513        if !delta.is_empty() {
8514            self.emit(AgentEvent::BackgroundOutput {
8515                job_id: job_id.clone(),
8516                chunk: delta,
8517                truncated,
8518            });
8519        }
8520
8521        let exit_code = match status {
8522            supercode_runtime::background::JobStatus::Exited(code) => code,
8523            _ => None,
8524        };
8525        let out = serde_json::json!({
8526            "job_id": job_id,
8527            "command": command,
8528            "status": status.as_str(),
8529            "exit_code": exit_code,
8530            "pid": pid,
8531            "started_at_ms": started_at_ms,
8532            "output": output_so_far,
8533            "output_truncated": truncated,
8534        });
8535        if !matches!(status, supercode_runtime::background::JobStatus::Running) {
8536            self.background_jobs.remove(&job_id);
8537        }
8538        (out.to_string(), false)
8539    }
8540
8541    /// Execute the `background_list` intrinsic (P5-6, D10 "bg-manager"):
8542    /// list every background job this agent is currently tracking, without
8543    /// draining output or reaping anything (a read-only listing —
8544    /// `background_status` is the reaping poll).
8545    fn run_background_list(&mut self, _call: &supercode_interchange::ToolCall) -> (String, bool) {
8546        let mut jobs = Vec::new();
8547        for (job_id, job) in self.background_jobs.iter_mut() {
8548            let status = background_job_status(job);
8549            jobs.push(serde_json::json!({
8550                "job_id": job_id,
8551                "command": job.command,
8552                "status": status.as_str(),
8553                "pid": job.pid,
8554                "started_at_ms": job.started_at_ms,
8555            }));
8556        }
8557        let out = serde_json::json!({ "jobs": jobs });
8558        (out.to_string(), false)
8559    }
8560
8561    /// Execute the `background_kill` intrinsic (P5-6, D10 "bg-manager",
8562    /// build brief "kill/cancel a job"): request REAL termination of a
8563    /// background job's OS process AND its whole process group (see
8564    /// [`kill_job_process_group`] — Fable-5 review, HIGH, "grandchildren
8565    /// orphaned on kill"; a documented no-op if the process already
8566    /// exited) and reap it immediately, freeing its concurrency slot.
8567    fn run_background_kill(&mut self, call: &supercode_interchange::ToolCall) -> (String, bool) {
8568        let args = match call.function.parsed_arguments() {
8569            Ok(v) => v,
8570            Err(e) => {
8571                let err = Error::InvalidArguments {
8572                    tool: BACKGROUND_KILL.to_string(),
8573                    message: e.to_string(),
8574                };
8575                return (format!("Error: {err}"), true);
8576            }
8577        };
8578        let Some(job_id) = args.get("job_id").and_then(serde_json::Value::as_str) else {
8579            let err = Error::InvalidArguments {
8580                tool: BACKGROUND_KILL.to_string(),
8581                message: "`job_id` is required".to_string(),
8582            };
8583            return (format!("Error: {err}"), true);
8584        };
8585        let job_id = job_id.to_string();
8586        let Some(mut job) = self.background_jobs.remove(&job_id) else {
8587            let err = Error::BackgroundJobNotFound(job_id);
8588            return (format!("Error: {err}"), true);
8589        };
8590        kill_job_process_group(&mut job);
8591        job.killed = true;
8592        let out = serde_json::json!({
8593            "job_id": job_id,
8594            "status": "killed",
8595            "pid": job.pid,
8596        });
8597        // `job` (and its `ConcurrencyGuard`) drops here, freeing the slot.
8598        (out.to_string(), false)
8599    }
8600
8601    /// P5-3 (D5 "subagent transcripts… persisted + linked"): write a
8602    /// finished child's transcript to `Self::subagent_store`, if one is
8603    /// installed — a no-op otherwise (see that field's doc comment). Builds
8604    /// the child's `Session` the same way `to_native_jsonl_v2`'s doc
8605    /// comment describes (an empty imported prefix + `transcript` as
8606    /// `appended` `NativeTurn`s), with `meta.agent_id`/`parent_tool_use_id`/
8607    /// `lineage` populated from `lineage` so the native-v2 header carries
8608    /// the full lineage record on disk (see `Session::to_native_jsonl_v2`'s
8609    /// P5-3 doc note).
8610    ///
8611    /// P5-3 safety-hardening fix (Fable-5 review, LOW "translation-fidelity
8612    /// cosmetic"): `Session::from_claude_code_str("")` is used ONLY to get
8613    /// a blank `raw`/`messages` skeleton cheaply (an empty string parses
8614    /// identically under any loader) — it is NOT claiming this child's
8615    /// session actually came from Claude Code. Before this fix, that
8616    /// borrowed constructor's `meta.source` (`SessionSource::ClaudeCode`)
8617    /// leaked straight through to the persisted sidecar's `source` header,
8618    /// mislabeling a native `spawn_subagent` child as an imported CC
8619    /// session. Corrected to `SessionSource::Native` immediately after —
8620    /// see that variant's doc comment.
8621    fn persist_subagent_transcript(
8622        &self,
8623        child_id: &str,
8624        lineage: &crate::subagents::SubagentLineage,
8625        transcript: &[ChatMessage],
8626    ) {
8627        let Some((store, parent_name)) = &self.subagent_store else {
8628            return;
8629        };
8630        let mut session = match Session::from_claude_code_str("") {
8631            Ok(s) => s,
8632            Err(_) => return,
8633        };
8634        session.meta.source = supercode_interchange::session::SessionSource::Native;
8635        session.meta.agent_id = Some(lineage.child_agent_id.clone());
8636        session.meta.parent_tool_use_id = Some(lineage.parent_tool_use_id.clone());
8637        session.meta.lineage = lineage.to_lineage_map();
8638        let sidecar_jsonl = session.to_native_jsonl_v2(transcript);
8639        let _ = store.save_subagent_transcript(parent_name, child_id, &sidecar_jsonl);
8640        let _ = store.save_subagent_lineage(parent_name, child_id, lineage);
8641    }
8642
8643    /// Execute the `tool_search` intrinsic (B6): case-insensitive keyword
8644    /// match over `name` + `description` of every registered, enabled,
8645    /// non-core, not-yet-activated tool (builtin and `mcp__*` alike). Matches
8646    /// are activated (advertised starting with the next request) and
8647    /// returned as a JSON array of their full [`ToolSchema`]s.
8648    fn run_tool_search(&mut self, call: &supercode_interchange::ToolCall) -> (String, bool) {
8649        let args = match call.function.parsed_arguments() {
8650            Ok(v) => v,
8651            Err(e) => {
8652                let err = Error::InvalidArguments {
8653                    tool: TOOL_SEARCH.to_string(),
8654                    message: e.to_string(),
8655                };
8656                return (format!("Error: {err}"), true);
8657            }
8658        };
8659        let query = args
8660            .get("query")
8661            .and_then(serde_json::Value::as_str)
8662            .unwrap_or("")
8663            .to_lowercase();
8664        let max_results = args
8665            .get("max_results")
8666            .and_then(serde_json::Value::as_u64)
8667            .map(|n| n as usize);
8668
8669        let mut matches: Vec<ToolSchema> = self
8670            .registry
8671            .iter()
8672            .filter(|t| self.config.tool_enabled(t.name()))
8673            .filter(|t| !self.is_core_tool(t.name()))
8674            .filter(|t| !self.activated_tools.contains(t.name()))
8675            .filter(|t| {
8676                query.is_empty()
8677                    || t.name().to_lowercase().contains(&query)
8678                    || self
8679                        .config
8680                        .tool_description(t.name(), t.description())
8681                        .to_lowercase()
8682                        .contains(&query)
8683            })
8684            // TR-8/T5 dev/03: the on-demand fetch always returns the ORIGINAL
8685            // full schema, never the tier-minified one — that's the invert.
8686            .map(|t| self.raw_schema_for(t))
8687            .collect();
8688
8689        if let Some(max) = max_results {
8690            matches.truncate(max);
8691        }
8692
8693        for m in &matches {
8694            self.activated_tools.insert(m.name.clone());
8695        }
8696
8697        let result = serde_json::to_string(&matches).unwrap_or_else(|_| "[]".to_string());
8698        (result, false)
8699    }
8700
8701    /// The `expand_reduction` schema (T12/TR-1), advertised whenever a
8702    /// [`ReductionPolicy`] is installed.
8703    ///
8704    /// The description deliberately never spells the literal stub sentinel
8705    /// prefix: A11's export leak guard is unconditional, so an assistant
8706    /// turn that quoted a stub line verbatim (which teaching the syntax
8707    /// invites) would permanently fail export for that session. Stubs are
8708    /// described abstractly and the model is told to pass ids only.
8709    fn expand_reduction_schema() -> ToolSchema {
8710        ToolSchema {
8711            name: EXPAND_REDUCTION.to_string(),
8712            description: "Fetch back the original content hidden behind a reduction stub in \
8713                your current view — a truncated tool output, cleared old turns, or an elided \
8714                file read that was hidden to save context. Each stub line names a reduction id \
8715                like r0042-9f3c: pass ONLY that id here, and never quote or repeat a stub line \
8716                itself in your replies. The original is durably kept in the session sidecar. \
8717                Pass `byte_range` to fetch a slice of a large one at a time instead of all of \
8718                it at once; ranged results are prefixed with a `bytes start..end of total` \
8719                header so you can plan the next slice."
8720                .to_string(),
8721            parameters: serde_json::json!({
8722                "type": "object",
8723                "properties": {
8724                    "reduction_id": {
8725                        "type": "string",
8726                        "description": "The reduction id named in the stub line, e.g. \
8727                            \"r0042-9f3c\". Pass the id alone."
8728                    },
8729                    "byte_range": {
8730                        "type": "array",
8731                        "items": {"type": "integer"},
8732                        "minItems": 2,
8733                        "maxItems": 2,
8734                        "description": "Optional [start, end) byte offsets within the original \
8735                            content to fetch instead of all of it. Exactly two non-negative \
8736                            integers with start <= end."
8737                    }
8738                },
8739                "required": ["reduction_id"],
8740                "additionalProperties": false
8741            }),
8742        }
8743    }
8744
8745    /// The `sidecar_search` schema (T12/TR-1), advertised whenever a
8746    /// [`ReductionPolicy`] is installed. Same no-literal-sentinel rule as
8747    /// [`Self::expand_reduction_schema`].
8748    fn sidecar_search_schema() -> ToolSchema {
8749        ToolSchema {
8750            name: SIDECAR_SEARCH.to_string(),
8751            description: "Search content currently hidden from your view by reduction stubs \
8752                (large tool outputs, cleared old turns, elided file reads) for a substring or \
8753                regex. Only hidden content is searched, never what you can already see. \
8754                Returns match snippets with each match's reduction_id for use with \
8755                expand_reduction; refer to results by their reduction id rather than quoting \
8756                stub lines. Results are capped — if `truncated` is true, narrow the query."
8757                .to_string(),
8758            parameters: serde_json::json!({
8759                "type": "object",
8760                "properties": {
8761                    "query": {
8762                        "type": "string",
8763                        "description": "Non-empty substring or regex to search for \
8764                            (case-insensitive)."
8765                    }
8766                },
8767                "required": ["query"],
8768                "additionalProperties": false
8769            }),
8770        }
8771    }
8772
8773    /// Reload the recorder's full recorded messages from disk (TR-1's
8774    /// `recorded` resolution source). Since TR-12's D6/A7 supersession gate
8775    /// (`Self::run_loop`), a `expand_reduction`/`sidecar_search` call only
8776    /// ever exists alongside an active [`ReductionPolicy`] (see
8777    /// [`EXPAND_REDUCTION`]'s doc), and pairing one with a recorder — as the
8778    /// CLI's reduced mode always does — means the gate is already on and
8779    /// `history[1..]` holds the same full bytes as this reload: this upgrade
8780    /// is then a dormant no-op (`reduce::rehydrate::prefer_recorded` sees
8781    /// `recorded == minted` and keeps `minted`). It stops being a no-op —
8782    /// defense in depth, not the common path — for a **legacy** sidecar
8783    /// recorded before this gate existed, or for a policy-without-recorder
8784    /// agent (gate off, so `history[1..]` still carries
8785    /// [`Self::cap_tool_output`]-capped copies): only there can `history[1..]`
8786    /// diverge from the sidecar, and only there does consulting this reload
8787    /// actually recover bytes `history[1..]` alone couldn't. `Ok(None)` when
8788    /// no recorder is attached (rehydration then resolves from history alone,
8789    /// whose capped copies — if any — carry their own honest cap notice). A
8790    /// disk-level reload is fine here regardless: these intrinsic calls are
8791    /// rare, model-initiated events, not per-request work.
8792    fn recorded_messages(&self) -> std::result::Result<Option<Vec<ChatMessage>>, String> {
8793        let Some(recorder) = &self.recorder else {
8794            return Ok(None);
8795        };
8796        let raw = std::fs::read_to_string(recorder.path())
8797            .map_err(|e| format!("failed to read the session sidecar: {e}"))?;
8798        let session = Session::from_sidecar_str(&raw)
8799            .map_err(|e| format!("failed to parse the session sidecar: {e}"))?;
8800        Ok(Some(session.messages))
8801    }
8802
8803    /// Execute the `expand_reduction` intrinsic (T12/TR-1): resolves against
8804    /// `self.reduction_log` + `self.history[1..]` (the hash-minting source),
8805    /// upgraded to the recorder's full recorded bytes for cap-diverged
8806    /// content ([`Self::recorded_messages`]; the two-source contract is
8807    /// documented on `reduce::rehydrate`). `byte_range` is validated
8808    /// strictly — any malformed shape is a model-recoverable error naming
8809    /// the expected form and the original's true size, never a silent
8810    /// whole-content (or empty) return.
8811    fn run_expand_reduction(&mut self, call: &supercode_interchange::ToolCall) -> (String, bool) {
8812        let args = match call.function.parsed_arguments() {
8813            Ok(v) => v,
8814            Err(e) => {
8815                let err = Error::InvalidArguments {
8816                    tool: EXPAND_REDUCTION.to_string(),
8817                    message: e.to_string(),
8818                };
8819                return (format!("Error: {err}"), true);
8820            }
8821        };
8822        let Some(id) = args.get("reduction_id").and_then(serde_json::Value::as_str) else {
8823            return (
8824                "Error: expand_reduction requires a `reduction_id` string argument".to_string(),
8825                true,
8826            );
8827        };
8828        let recorded = match self.recorded_messages() {
8829            Ok(r) => r,
8830            Err(e) => return (format!("Error: expand_reduction: {e}"), true),
8831        };
8832        let recorded = recorded.as_deref();
8833
8834        // B3: strict shape validation — exactly two non-negative integers.
8835        // Anything else errors (with the true total when resolvable) rather
8836        // than silently degrading to a whole-content expand.
8837        let byte_range = match args.get("byte_range") {
8838            None | Some(serde_json::Value::Null) => None,
8839            Some(v) => {
8840                let parsed = v
8841                    .as_array()
8842                    .filter(|a| a.len() == 2)
8843                    .and_then(|a| Some((a[0].as_u64()? as usize, a[1].as_u64()? as usize)));
8844                match parsed {
8845                    Some(range) => Some(range),
8846                    None => {
8847                        let total = reduce::rehydrate::reduction_total_bytes(
8848                            &self.reduction_log,
8849                            &self.history[1..],
8850                            recorded,
8851                            id,
8852                        )
8853                        .map(|n| format!("; the original is {n} bytes"))
8854                        .unwrap_or_default();
8855                        return (
8856                            format!(
8857                                "Error: expand_reduction: malformed byte_range {v} — expected \
8858                                 [start, end): exactly two non-negative integers with \
8859                                 start <= end{total}"
8860                            ),
8861                            true,
8862                        );
8863                    }
8864                }
8865            }
8866        };
8867        match reduce::rehydrate::expand_reduction(
8868            &self.reduction_log,
8869            &self.history[1..],
8870            recorded,
8871            id,
8872            byte_range,
8873        ) {
8874            // A ranged result carries a provenance header naming the slice
8875            // and the true total, so the model can plan its next slice; a
8876            // whole-content expand stays byte-exact (TR-1 dev/01).
8877            Ok(outcome) => match outcome.range {
8878                Some((start, end)) => (
8879                    format!(
8880                        "[{id}: bytes {start}..{end} of {total}]\n{content}",
8881                        total = outcome.total_bytes,
8882                        content = outcome.content
8883                    ),
8884                    false,
8885                ),
8886                None => (outcome.content, false),
8887            },
8888            Err(e) => (format!("Error: {e}"), true),
8889        }
8890    }
8891
8892    /// Execute the `sidecar_search` intrinsic (T12/TR-1); same two-source
8893    /// resolution as [`Self::run_expand_reduction`]. The result is bounded
8894    /// by construction (`reduce::rehydrate::SidecarSearchResult`'s caps), so
8895    /// a broad query can never re-inflate the context or bloat the sidecar
8896    /// the recorder appends this result to.
8897    fn run_sidecar_search(&mut self, call: &supercode_interchange::ToolCall) -> (String, bool) {
8898        let args = match call.function.parsed_arguments() {
8899            Ok(v) => v,
8900            Err(e) => {
8901                let err = Error::InvalidArguments {
8902                    tool: SIDECAR_SEARCH.to_string(),
8903                    message: e.to_string(),
8904                };
8905                return (format!("Error: {err}"), true);
8906            }
8907        };
8908        let query = args
8909            .get("query")
8910            .and_then(serde_json::Value::as_str)
8911            .unwrap_or("");
8912        if query.trim().is_empty() {
8913            return (
8914                "Error: sidecar_search requires a non-empty `query` string argument".to_string(),
8915                true,
8916            );
8917        }
8918        let recorded = match self.recorded_messages() {
8919            Ok(r) => r,
8920            Err(e) => return (format!("Error: sidecar_search: {e}"), true),
8921        };
8922        match reduce::rehydrate::sidecar_search(
8923            &self.reduction_log,
8924            &self.history[1..],
8925            recorded.as_deref(),
8926            query,
8927        ) {
8928            Ok(result) => (
8929                serde_json::to_string(&result).unwrap_or_else(|_| "{}".to_string()),
8930                false,
8931            ),
8932            Err(e) => (format!("Error: {e}"), true),
8933        }
8934    }
8935
8936    fn emit(&self, event: AgentEvent) {
8937        if let Some(sink) = &self.config.event_sink {
8938            sink(event);
8939        }
8940    }
8941
8942    /// Number of non-system messages exchanged so far.
8943    pub fn turn_count(&self) -> usize {
8944        self.history
8945            .iter()
8946            .filter(|m| m.role != Role::System)
8947            .count()
8948    }
8949
8950    /// Cumulative output (completion) tokens reported by the provider across
8951    /// every `send` on this agent. Zero if the provider reports no usage.
8952    pub fn total_output_tokens(&self) -> u64 {
8953        self.total_output_tokens
8954    }
8955}
8956
8957/// Deliver an agent's `send_message` to another session through the one
8958/// send every sender uses. The sender is this process's own session (a
8959/// runtime supercode hosts, found from its ancestry), so replies can come
8960/// back.
8961#[cfg(feature = "adapter-api")]
8962async fn send_to_session(to: &str, message: &str, notify_when_idle: bool) -> (String, bool) {
8963    let homes = crate::HarnessHomes::default();
8964    let caller =
8965        match crate::mail_route::resolve_caller(&homes, &crate::mail_route::process_ancestry()) {
8966            Ok(caller) => caller,
8967            Err(_) => {
8968                return (
8969                    format!(
8970                    "Not sent: `{to}` is not one of your running subagents, and this session has \
8971                     no address other sessions can reply to (supercode does not host it), so it \
8972                     can message only its own subagents."
8973                ),
8974                    true,
8975                )
8976            }
8977        };
8978    let options = crate::mail_send::SendOptions {
8979        notify_when_idle,
8980        ..Default::default()
8981    };
8982    match crate::mail_send::send(&homes, &caller, to, message, options).await {
8983        Ok(outcome) => (outcome.text, outcome.code != 0),
8984        Err(error) => (format!("Not sent to {to}: {error}"), true),
8985    }
8986}
8987
8988#[cfg(not(feature = "adapter-api"))]
8989async fn send_to_session(to: &str, _message: &str, _notify_when_idle: bool) -> (String, bool) {
8990    (
8991        format!("Not sent: `{to}` is not one of your running subagents, and this build has no cross-session messaging."),
8992        true,
8993    )
8994}
8995
8996#[cfg(test)]
8997mod bp2_spill_tests {
8998    //! BP-2 (`.volter/tracker/markdown/BP-2.md`, catalog:58): under the
8999    //! parity presets a capped tool output stays RECOVERABLE by the model —
9000    //! without `capabilities.reduction`, which both presets leave off.
9001
9002    use super::*;
9003    use crate::configfile::{resolve, ResolveOptions};
9004
9005    /// Never called — these tests drive `cap_tool_output` directly.
9006    #[derive(Debug)]
9007    struct NeverCalledProvider;
9008
9009    #[async_trait::async_trait]
9010    impl Provider for NeverCalledProvider {
9011        async fn complete(
9012            &self,
9013            _req: &ChatRequest,
9014            _on_delta: &(dyn for<'a> Fn(&'a str) + Send + Sync),
9015        ) -> crate::Result<(ChatMessage, crate::provider::Usage)> {
9016            unreachable!("BP-2 spill tests never issue a request")
9017        }
9018    }
9019
9020    fn resolved(preset: &str) -> crate::configfile::Resolved {
9021        let toml = crate::presets::lookup(preset).unwrap();
9022        resolve(toml, None, &ResolveOptions { strict: true })
9023            .unwrap_or_else(|e| panic!("{preset} resolves: {e}"))
9024    }
9025
9026    /// The residue this closes: the cap notice said the full output was
9027    /// "not retained" / "in session sidecar" and the only door to it —
9028    /// `expand_reduction` — is advertised solely when a `ReductionPolicy`
9029    /// is installed, which `capabilities.reduction = false` never does. So
9030    /// under both parity presets a truncated output was simply lost.
9031    ///
9032    /// Now the notice NAMES a spill file, and the door is the preset's own
9033    /// read pathway: `read_file` under cc-parity, the shell under
9034    /// cx-parity (which registers no file tools at all).
9035    #[tokio::test]
9036    async fn parity_presets_spill_capped_output_and_name_a_door_the_preset_has() {
9037        for (preset, expected_door) in [
9038            ("cc-parity", "read it with `read_file`"),
9039            ("cx-parity", "read it with `cat`"),
9040        ] {
9041            let r = resolved(preset);
9042            assert!(
9043                r.config.tool_output_spill,
9044                "{preset} must set `core.tool_output_spill`"
9045            );
9046            assert_eq!(
9047                r.modules.get("reduction"),
9048                Some(&false),
9049                "{preset} leaves `capabilities.reduction` off — the spill must not depend on it"
9050            );
9051            let mut config = resolved(preset).config;
9052            config.max_tool_output_bytes = Some(1024);
9053            let registry = crate::tools::ToolRegistry::from_config(&config);
9054            let agent = Agent::with_parts(config, Box::new(NeverCalledProvider), registry);
9055            assert!(
9056                agent.reduction_policy.is_none(),
9057                "no reduction policy is installed under {preset}"
9058            );
9059
9060            let full = "R".repeat(50_000);
9061            let capped = agent.cap_tool_output(full.clone());
9062            assert!(capped.len() < full.len(), "{preset}: output must be capped");
9063            assert!(capped.contains(expected_door), "{preset}: {capped:?}");
9064
9065            // The path in the notice must actually hold the full bytes.
9066            let marker = capped.split("spilled to ").nth(1).unwrap_or_default();
9067            let path = marker.split(" — ").next().unwrap_or_default();
9068            assert!(!path.is_empty(), "{preset}: no spill path in {capped:?}");
9069            assert_eq!(
9070                std::fs::read_to_string(path).unwrap(),
9071                full,
9072                "{preset}: the spill file must hold the FULL output"
9073            );
9074
9075            // And the model can actually walk through that door: the
9076            // preset's own read pathway returns the spilled content.
9077            let ctx = build_tool_context(agent.config()).0;
9078            let recovered = match registry_read_tool(&agent) {
9079                Some(("read_file", tool)) => tool
9080                    .execute(serde_json::json!({"path": path}), &ctx)
9081                    .await
9082                    .unwrap(),
9083                Some(("bash", tool)) => tool
9084                    .execute(serde_json::json!({"command": format!("cat {path}")}), &ctx)
9085                    .await
9086                    .unwrap(),
9087                _ => panic!("{preset}: no read door registered"),
9088            };
9089            assert!(
9090                recovered.contains(&"R".repeat(2000)),
9091                "{preset}: the door must return the spilled output"
9092            );
9093            let _ = std::fs::remove_file(path);
9094        }
9095    }
9096
9097    /// The preset's read pathway: `read_file` where it exists, else the
9098    /// shell — the same choice the cap notice's wording makes.
9099    fn registry_read_tool<'a>(
9100        agent: &'a Agent,
9101    ) -> Option<(&'static str, &'a dyn crate::tools::Tool)> {
9102        if let Some(tool) = agent.registry.get("read_file") {
9103            return Some(("read_file", tool));
9104        }
9105        agent.registry.get("bash").map(|tool| ("bash", tool))
9106    }
9107
9108    /// Off (the default, every non-parity preset and every SDK embedder):
9109    /// no spill file, and the notice is byte-identical to before BP-2.
9110    #[test]
9111    fn spill_off_leaves_the_notice_unchanged_and_writes_nothing() {
9112        let config = Config::builder().max_tool_output_bytes(1024).build();
9113        assert!(!config.tool_output_spill);
9114        let agent = Agent::with_provider(config, Box::new(NeverCalledProvider));
9115        let capped = agent.cap_tool_output("S".repeat(50_000));
9116        assert!(capped.contains("full output not retained"), "{capped:?}");
9117        assert!(!capped.contains("spilled to"), "{capped:?}");
9118    }
9119}
9120
9121#[cfg(test)]
9122mod bp3_new_core_tool_tests {
9123    //! BP-3 (`.volter/tracker/markdown/BP-3.md`): the two behaviours the
9124    //! new tools can only have INSIDE the agent — plan mode narrowing the
9125    //! permissions engine, and `new_context` re-founding the request view
9126    //! through `reduce`'s handoff projection — driven over the RESOLVED
9127    //! parity presets.
9128
9129    use super::*;
9130    use crate::configfile::{resolve, ResolveOptions};
9131
9132    #[derive(Debug)]
9133    struct NeverCalledProvider;
9134
9135    #[async_trait::async_trait]
9136    impl Provider for NeverCalledProvider {
9137        async fn complete(
9138            &self,
9139            _req: &ChatRequest,
9140            _on_delta: &(dyn for<'a> Fn(&'a str) + Send + Sync),
9141        ) -> crate::Result<(ChatMessage, crate::provider::Usage)> {
9142            unreachable!("BP-3 tests never issue a request")
9143        }
9144    }
9145
9146    fn resolved(preset: &str) -> Config {
9147        let toml = crate::presets::lookup(preset).unwrap();
9148        resolve(toml, None, &ResolveOptions { strict: true })
9149            .unwrap_or_else(|e| panic!("{preset} resolves: {e}"))
9150            .config
9151    }
9152
9153    fn agent_for(preset: &str) -> Agent {
9154        Agent::with_provider(resolved(preset), Box::new(NeverCalledProvider))
9155    }
9156
9157    /// An approval door that says yes to everything — so a denial in these
9158    /// tests can only come from the DENY tier, never from cc-parity's
9159    /// `approval = "untrusted"` ask default.
9160    struct AlwaysAllow;
9161    impl crate::permissions::PermissionsApprovalHandler for AlwaysAllow {
9162        fn ask(
9163            &self,
9164            _req: &crate::permissions::ApprovalRequest,
9165        ) -> crate::permissions::ApprovalOutcome {
9166            crate::permissions::ApprovalOutcome::Allow
9167        }
9168    }
9169
9170    fn call(name: &str, args: serde_json::Value) -> supercode_interchange::ToolCall {
9171        supercode_interchange::ToolCall {
9172            id: format!("call-{name}"),
9173            kind: "function".to_string(),
9174            function: supercode_interchange::FunctionCall {
9175                name: name.to_string(),
9176                arguments: args.to_string(),
9177            },
9178        }
9179    }
9180
9181    /// The row `plan-mode-read-only-research-phase` claims: under the
9182    /// resolved `cc-parity` preset, entering plan mode makes the
9183    /// permissions engine REFUSE write and execution tools — and the
9184    /// refusal survives an approval door that allows everything, because
9185    /// the mode contributes DENY rules, the tier no approval can override.
9186    #[test]
9187    fn cc_parity_plan_mode_denies_writes_through_the_permissions_engine() {
9188        let mut agent = agent_for("cc-parity");
9189        assert!(
9190            agent.config.permissions_enabled,
9191            "cc-parity runs the permissions engine; plan mode narrows it"
9192        );
9193        agent.set_permissions_approval_handler(AlwaysAllow);
9194
9195        let write = serde_json::json!({"path": "notes.txt", "content": "x"});
9196        let bash = serde_json::json!({"command": "echo hi"});
9197        assert!(
9198            agent
9199                .permissions_gate_denial("write_file", &write, crate::config::HookDecision::Pass)
9200                .is_none(),
9201            "outside plan mode an allowed write must pass"
9202        );
9203        assert!(agent
9204            .permissions_gate_denial("bash", &bash, crate::config::HookDecision::Pass)
9205            .is_none());
9206
9207        agent.plan_mode().enter(Some("research first"));
9208
9209        let denial = agent
9210            .permissions_gate_denial("write_file", &write, crate::config::HookDecision::Pass)
9211            .expect("plan mode must refuse a write");
9212        assert!(denial.contains("Deny"), "{denial}");
9213        assert!(agent
9214            .permissions_gate_denial("bash", &bash, crate::config::HookDecision::Pass)
9215            .is_some());
9216        assert!(agent
9217            .permissions_gate_denial(
9218                "apply_patch",
9219                &serde_json::json!({"patch": "*** Begin Patch\n*** End Patch"}),
9220                crate::config::HookDecision::Pass
9221            )
9222            .is_some());
9223
9224        // The research surface, and the way out, stay open.
9225        for (tool, args) in [
9226            ("read_file", serde_json::json!({"path": "notes.txt"})),
9227            ("glob", serde_json::json!({"pattern": "*.rs"})),
9228            ("exit_plan_mode", serde_json::json!({"plan": "the plan"})),
9229            ("ask_user", serde_json::json!({"questions": []})),
9230        ] {
9231            assert!(
9232                agent
9233                    .permissions_gate_denial(tool, &args, crate::config::HookDecision::Pass)
9234                    .is_none(),
9235                "plan mode must leave `{tool}` reachable"
9236            );
9237        }
9238
9239        agent.plan_mode().exit();
9240        assert!(
9241            agent
9242                .permissions_gate_denial("write_file", &write, crate::config::HookDecision::Pass)
9243                .is_none(),
9244            "leaving plan mode restores the write surface"
9245        );
9246    }
9247
9248    /// The `context-budget-tools` row's read half: the figure
9249    /// `get_context_remaining` reports is the agent's OWN accounting,
9250    /// computed at the moment the tool asks for it.
9251    #[test]
9252    fn cx_parity_publishes_its_context_accounting_when_the_budget_tool_runs() {
9253        let mut agent = agent_for("cx-parity");
9254        assert!(agent.ctx.context_budget.snapshot().is_none(), "nothing yet");
9255        agent
9256            .history
9257            .push(ChatMessage::user("x".repeat(4000).to_string()));
9258        let _ = agent.prepare_tool_call(&call("current_time", serde_json::json!({})));
9259        assert!(
9260            agent.ctx.context_budget.snapshot().is_none(),
9261            "an unrelated tool call must not pay for the accounting"
9262        );
9263
9264        let _ = agent.prepare_tool_call(&call("get_context_remaining", serde_json::json!({})));
9265        let published = agent
9266            .ctx
9267            .context_budget
9268            .snapshot()
9269            .expect("the budget tool's own call publishes it");
9270        // What the model reads IS `Agent::context_usage()` — the same
9271        // struct `/context` prints and the guard enforces, not a second
9272        // estimate that could disagree with it.
9273        assert_eq!(
9274            published,
9275            serde_json::to_value(agent.context_usage()).unwrap()
9276        );
9277        assert!(published["context_limit"].as_u64().unwrap() > 0);
9278        assert!(published["remaining_tokens"].as_u64().unwrap() > 0);
9279    }
9280
9281    /// The `context-budget-tools` row's write half, over the resolved
9282    /// `cx-parity` preset: a parked `new_context` request re-founds the
9283    /// request view through `reduce`'s handoff projection — objective in
9284    /// the leading system message, the tail kept, the rest covered by
9285    /// `TurnsCleared` spans that land in this agent's own reduction log.
9286    #[test]
9287    fn cx_parity_new_context_rebuilds_the_window_through_the_same_handoff_the_operator_gets() {
9288        let mut agent = agent_for("cx-parity");
9289        agent.history.push(ChatMessage::system("system"));
9290        for i in 0..12 {
9291            agent.history.push(ChatMessage::user(format!("turn {i}")));
9292        }
9293        let before = agent.history.clone();
9294
9295        agent
9296            .ctx
9297            .context_budget
9298            .request_new_context(crate::tools::NewContextRequest {
9299                objective: "finish the parser".to_string(),
9300                keep_recent: Some(2),
9301            });
9302        agent.apply_pending_new_context();
9303
9304        // Exactly what `/handoff` produces — the model's door and the
9305        // operator's door run one mechanism, so this compares against it.
9306        let mut expected =
9307            Agent::with_provider(resolved("cx-parity"), Box::new(NeverCalledProvider));
9308        expected.history = before.clone();
9309        expected.new_context("finish the parser", Some(2));
9310        assert_eq!(agent.history, expected.history);
9311
9312        assert!(
9313            agent.history.len() < before.len(),
9314            "the window must actually shrink: {} -> {}",
9315            before.len(),
9316            agent.history.len()
9317        );
9318        assert_eq!(agent.history[0], before[0], "the system prompt survives");
9319        let marker = agent.history[1].content.clone().unwrap_or_default();
9320        assert!(marker.contains("fresh working context"), "{marker}");
9321        assert!(marker.contains("finish the parser"), "{marker}");
9322        assert_eq!(
9323            agent.history[agent.history.len() - 2..],
9324            before[before.len() - 2..],
9325            "the requested tail is kept verbatim"
9326        );
9327        assert!(
9328            agent.ctx.context_budget.take_new_context().is_none(),
9329            "the request is consumed exactly once"
9330        );
9331    }
9332
9333    /// `new_context` never trades recoverability for a smaller window: with
9334    /// no sidecar recorder the request is refused, the reason is handed
9335    /// back to the model, and the transcript keeps every turn.
9336    #[test]
9337    fn cx_parity_new_context_states_the_retention_it_actually_has() {
9338        let mut agent = agent_for("cx-parity");
9339        agent.history.push(ChatMessage::system("system"));
9340        for i in 0..12 {
9341            agent.history.push(ChatMessage::user(format!("turn {i}")));
9342        }
9343        agent
9344            .ctx
9345            .context_budget
9346            .request_new_context(crate::tools::NewContextRequest {
9347                objective: "finish the parser".to_string(),
9348                keep_recent: Some(2),
9349            });
9350        agent.apply_pending_new_context();
9351
9352        // No recorder is installed here, and the marker says so rather than
9353        // implying the set-aside turns are still somewhere.
9354        let marker = agent.history[1].content.clone().unwrap_or_default();
9355        assert!(
9356            marker.contains("No transcript sidecar is attached"),
9357            "the marker must not overstate retention: {marker}"
9358        );
9359    }
9360}
9361
9362/// BP-8 (catalog:152): what [`Agent::rewind_conversation`] did.
9363#[derive(Debug, Clone, PartialEq, Eq)]
9364pub struct RewindOutcome {
9365    /// Messages remaining, including the system message at index 0.
9366    pub kept: usize,
9367    /// Messages removed from the live conversation (still on disk, in the
9368    /// journal, and still in the tree under `preserved_branch`).
9369    pub removed: usize,
9370    /// When the tree module is on and the rewind actually moved the leaf:
9371    /// the sibling branch the old leaf was preserved under, so the rewound
9372    /// path stays independently addressable.
9373    pub preserved_branch: Option<String>,
9374}
9375
9376#[cfg(test)]
9377mod bp1_compaction_tests {
9378    //! BP-1 (`.volter/tracker/markdown/BP-1.md` AC2): `cx-parity` fires
9379    //! [`Agent::maybe_compact`] at its own trigger, and
9380    //! `core.compaction.summarize` reaches [`Config`] and is what the
9381    //! compaction marker says.
9382
9383    use super::*;
9384    use crate::configfile::{resolve, ResolveOptions};
9385
9386    /// Never called — these tests drive `maybe_compact` directly, which
9387    /// makes no request.
9388    #[derive(Debug)]
9389    struct NeverCalledProvider;
9390
9391    #[async_trait::async_trait]
9392    impl Provider for NeverCalledProvider {
9393        async fn complete(
9394            &self,
9395            _req: &ChatRequest,
9396            _on_delta: &(dyn for<'a> Fn(&'a str) + Send + Sync),
9397        ) -> crate::Result<(ChatMessage, crate::provider::Usage)> {
9398            unreachable!("BP-1 compaction tests never issue a request")
9399        }
9400    }
9401
9402    fn resolved_cx_parity() -> Config {
9403        let toml = crate::presets::lookup("cx-parity").unwrap();
9404        resolve(toml, None, &ResolveOptions { strict: true })
9405            .expect("cx-parity resolves")
9406            .config
9407    }
9408
9409    /// Enough history to sit inside `reserve_tokens` of ANY model context
9410    /// window (`estimate_view_tokens` is size-proportional, and the largest
9411    /// window in the catalog is far below this).
9412    fn stuff_history(agent: &mut Agent) {
9413        agent.history.push(ChatMessage::system("system"));
9414        for i in 0..400 {
9415            agent
9416                .history
9417                .push(ChatMessage::user(format!("turn {i}: {}", "x".repeat(8000))));
9418        }
9419    }
9420
9421    /// The defect: `cx-parity` armed NEITHER compaction trigger, so
9422    /// `maybe_compact` returned `false` on its
9423    /// `threshold.is_none() && compaction_reserve_tokens.is_none()` guard
9424    /// no matter how large the conversation grew.
9425    #[test]
9426    fn cx_parity_fires_maybe_compact_at_its_pressure_trigger() {
9427        let config = resolved_cx_parity();
9428        assert!(config.compaction_enabled);
9429        assert_eq!(config.compaction_reserve_tokens, Some(16384));
9430        assert!(config.compaction_summarize);
9431
9432        let mut agent = Agent::with_provider(config, Box::new(NeverCalledProvider));
9433        stuff_history(&mut agent);
9434        let before = agent.history.len();
9435        assert!(
9436            agent.maybe_compact(),
9437            "cx-parity must compact under context pressure"
9438        );
9439        assert!(agent.history.len() < before, "history must actually shrink");
9440        let marker = agent
9441            .history
9442            .iter()
9443            .find(|m| {
9444                m.content
9445                    .as_deref()
9446                    .is_some_and(|c| c.contains("earlier conversation compacted"))
9447            })
9448            .expect("a compaction marker must be present");
9449        assert!(
9450            marker
9451                .content
9452                .as_deref()
9453                .unwrap()
9454                .contains("summarized to save context"),
9455            "cx-parity sets `core.compaction.summarize = true`"
9456        );
9457    }
9458
9459    /// `core.compaction.summarize = false` reaches `Config` too, and the
9460    /// marker then states only what actually happened to the span.
9461    #[test]
9462    fn compaction_summarize_false_reaches_config_and_changes_the_marker() {
9463        let toml = "extends = \"cx-parity\"\n[core.compaction]\nsummarize = false\n";
9464        let config = resolve(toml, None, &ResolveOptions::default())
9465            .expect("resolves")
9466            .config;
9467        assert!(!config.compaction_summarize);
9468
9469        let mut agent = Agent::with_provider(config, Box::new(NeverCalledProvider));
9470        stuff_history(&mut agent);
9471        assert!(agent.maybe_compact());
9472        let marker = agent
9473            .history
9474            .iter()
9475            .find(|m| {
9476                m.content
9477                    .as_deref()
9478                    .is_some_and(|c| c.contains("earlier conversation compacted"))
9479            })
9480            .expect("a compaction marker must be present");
9481        let text = marker.content.as_deref().unwrap();
9482        assert!(text.contains("cleared to save context"), "got: {text}");
9483        assert!(!text.contains("summarized"));
9484    }
9485}
9486
9487#[cfg(test)]
9488mod api_key_cmd_tests {
9489    //! P4 (design §5.2, §1.8 D6 row): `api_key_cmd` credential-helper
9490    //! resolution. `Agent::new` never makes a network call, so these tests
9491    //! exercise the real resolution chain end-to-end without mocking.
9492
9493    use super::*;
9494
9495    /// Default-off: with no `api_key`/`api_key_cmd` set and an env var that
9496    /// isn't set either, resolution fails exactly as it always has —
9497    /// `api_key_cmd` being a brand-new field changes nothing when unset.
9498    #[test]
9499    fn default_none_falls_through_to_missing_api_key_error() {
9500        let config = Config::builder()
9501            .api_key_env("SUPERCODE_TEST_UNSET_VAR_API_KEY_CMD")
9502            .build();
9503        assert!(config.api_key.is_none());
9504        assert!(config.api_key_cmd.is_none());
9505        let err = Agent::new(config).err().expect("no key source configured");
9506        assert!(matches!(err, Error::MissingApiKey(_)));
9507    }
9508
9509    /// Happy path: `api_key_cmd` alone (no `api_key`, no matching env var)
9510    /// is enough for `Agent::new` to succeed — the helper's stdout is
9511    /// resolved and used.
9512    #[test]
9513    fn api_key_cmd_alone_resolves_successfully() {
9514        let config = Config::builder()
9515            .api_key_cmd("echo sk-test-from-helper")
9516            .api_key_env("SUPERCODE_TEST_UNSET_VAR_API_KEY_CMD_2")
9517            .build();
9518        assert!(Agent::new(config).is_ok());
9519    }
9520
9521    /// A failing helper command (non-zero exit, or empty stdout) falls
9522    /// through to `api_key_env` rather than propagating the helper's own
9523    /// failure — same "try the next source" posture as every other layer.
9524    #[test]
9525    fn api_key_cmd_failure_falls_through_to_env() {
9526        std::env::set_var(
9527            "SUPERCODE_TEST_API_KEY_CMD_FALLBACK",
9528            "sk-from-env-fallback",
9529        );
9530        let config = Config::builder()
9531            .api_key_cmd("exit 1")
9532            .api_key_env("SUPERCODE_TEST_API_KEY_CMD_FALLBACK")
9533            .build();
9534        assert!(Agent::new(config).is_ok());
9535        std::env::remove_var("SUPERCODE_TEST_API_KEY_CMD_FALLBACK");
9536    }
9537
9538    /// A failing helper AND no fallback env var still produces the same
9539    /// `MissingApiKey` error today's no-key path always produced — the new
9540    /// source never turns a hard failure into a silent empty key.
9541    #[test]
9542    fn api_key_cmd_failure_with_no_fallback_still_errors() {
9543        let config = Config::builder()
9544            .api_key_cmd("exit 1")
9545            .api_key_env("SUPERCODE_TEST_UNSET_VAR_API_KEY_CMD_3")
9546            .build();
9547        let err = Agent::new(config)
9548            .err()
9549            .expect("helper failed, no env fallback");
9550        assert!(matches!(err, Error::MissingApiKey(_)));
9551    }
9552
9553    /// `run_api_key_cmd` directly: happy path trims trailing whitespace/
9554    /// newline from the command's stdout.
9555    #[test]
9556    fn run_api_key_cmd_trims_output() {
9557        assert_eq!(run_api_key_cmd("echo '  sk-abc123  '"), "sk-abc123");
9558    }
9559
9560    /// `run_api_key_cmd` directly: a nonexistent binary fails to spawn and
9561    /// returns an empty string rather than panicking.
9562    #[test]
9563    fn run_api_key_cmd_spawn_failure_returns_empty() {
9564        // `sh -c` itself always spawns; feed it a command that can't run.
9565        assert_eq!(
9566            run_api_key_cmd("/no/such/binary/at/all --flag"),
9567            String::new()
9568        );
9569    }
9570}
9571
9572#[cfg(test)]
9573mod bp4_prompt_context_tests {
9574    //! BP-4 (`.volter/tracker/markdown/BP-4.md`): the prompt/context knobs
9575    //! the parity presets never set, proved over the RESOLVED `cc-parity` /
9576    //! `cx-parity` configs (not over hand-built `Config`s — a preset that
9577    //! doesn't arm the knob would pass that weaker test).
9578
9579    use super::*;
9580    use crate::configfile::{resolve, ResolveOptions};
9581
9582    fn resolved(preset: &str) -> Config {
9583        let toml = crate::presets::lookup(preset).unwrap();
9584        resolve(toml, None, &ResolveOptions { strict: true })
9585            .unwrap_or_else(|e| panic!("{preset} resolves: {e}"))
9586            .config
9587    }
9588
9589    /// Never called — these tests assemble prompts and drive
9590    /// `refresh_env_context`/`inject_context_block`, none of which issue a
9591    /// request.
9592    #[derive(Debug)]
9593    struct NoProvider;
9594
9595    #[async_trait::async_trait]
9596    impl Provider for NoProvider {
9597        async fn complete(
9598            &self,
9599            _req: &ChatRequest,
9600            _on_delta: &(dyn for<'a> Fn(&'a str) + Send + Sync),
9601        ) -> crate::Result<(ChatMessage, crate::provider::Usage)> {
9602            unreachable!("BP-4 prompt/context tests never issue a request")
9603        }
9604    }
9605
9606    fn scratch(tag: &str) -> std::path::PathBuf {
9607        let dir = std::env::temp_dir().join(format!(
9608            "supercode-bp4-{tag}-{}-{}",
9609            std::process::id(),
9610            std::time::SystemTime::now()
9611                .duration_since(std::time::UNIX_EPOCH)
9612                .unwrap()
9613                .as_nanos()
9614        ));
9615        std::fs::create_dir_all(&dir).unwrap();
9616        dir
9617    }
9618
9619    /// Row `project-instruction-files-w-directory-walk`: a CLAUDE.md above
9620    /// the working directory is discovered, ordered root→cwd (nearest wins
9621    /// by appearing last), and the climb STOPS at the `.git` root — the
9622    /// directory above it is never read.
9623    #[test]
9624    fn presets_walk_ancestors_up_to_the_git_root_nearest_last() {
9625        for preset in ["cc-parity", "cx-parity"] {
9626            let base = scratch("walk");
9627            let above = base.join("above");
9628            let root = above.join("repo");
9629            let deep = root.join("crates").join("thing");
9630            std::fs::create_dir_all(&deep).unwrap();
9631            std::fs::create_dir_all(root.join(".git")).unwrap();
9632            std::fs::write(above.join("CLAUDE.md"), "ABOVE-THE-ROOT-MARKER").unwrap();
9633            std::fs::write(above.join("AGENTS.md"), "ABOVE-THE-ROOT-MARKER").unwrap();
9634            std::fs::write(root.join("CLAUDE.md"), "REPO-ROOT-MARKER").unwrap();
9635            std::fs::write(root.join("AGENTS.md"), "REPO-ROOT-MARKER").unwrap();
9636            std::fs::write(deep.join("CLAUDE.md"), "NEAREST-DIR-MARKER").unwrap();
9637            std::fs::write(deep.join("AGENTS.md"), "NEAREST-DIR-MARKER").unwrap();
9638
9639            let mut config = resolved(preset);
9640            config.cwd = deep.clone();
9641            let blob = assemble_project_instructions(&config);
9642
9643            let root_at = blob
9644                .find("REPO-ROOT-MARKER")
9645                .unwrap_or_else(|| panic!("{preset}: the ancestor repo root was not walked"));
9646            let near_at = blob
9647                .find("NEAREST-DIR-MARKER")
9648                .unwrap_or_else(|| panic!("{preset}: cwd's own file was not loaded"));
9649            assert!(
9650                root_at < near_at,
9651                "{preset}: nearest-to-cwd must win by appearing LAST (root→cwd)"
9652            );
9653            assert!(
9654                !blob.contains("ABOVE-THE-ROOT-MARKER"),
9655                "{preset}: the walk must stop at the `.git` root"
9656            );
9657            let _ = std::fs::remove_dir_all(&base);
9658        }
9659    }
9660
9661    /// The walk is bounded even with no root marker anywhere: it terminates
9662    /// at the filesystem root instead of looping.
9663    #[test]
9664    fn walk_terminates_without_a_root_marker() {
9665        let base = scratch("nomarker");
9666        let deep = base.join("a").join("b").join("c");
9667        std::fs::create_dir_all(&deep).unwrap();
9668        let mut config = resolved("cx-parity");
9669        config.cwd = deep.clone();
9670        let roots = instruction_walk_roots(&config);
9671        assert!(roots.len() <= MAX_INSTRUCTION_WALK_DEPTH);
9672        assert_eq!(roots.last().unwrap(), &deep, "cwd is the LAST root");
9673        let _ = std::fs::remove_dir_all(&base);
9674    }
9675
9676    /// Row `instruction-file-hygiene-controls`, cx half: `cx-parity` sets
9677    /// the documented 32 KiB `project_doc_max_bytes`, and it binds per file
9678    /// AND over the aggregate, each with its own notice.
9679    #[test]
9680    fn cx_parity_enforces_the_documented_instruction_byte_cap() {
9681        let config = resolved("cx-parity");
9682        assert_eq!(
9683            config.project_doc_max_bytes,
9684            Some(32_768),
9685            "cx-parity must arm cx§2's documented 32 KiB cap"
9686        );
9687
9688        let base = scratch("cap");
9689        std::fs::create_dir_all(base.join(".git")).unwrap();
9690        std::fs::write(base.join("CLAUDE.md"), "x".repeat(40_000)).unwrap();
9691        std::fs::write(base.join("AGENTS.md"), "y".repeat(40_000)).unwrap();
9692        let mut config = config;
9693        config.cwd = base.clone();
9694        let blob = assemble_project_instructions(&config);
9695        assert!(
9696            blob.contains("[supercode: file truncated at core.project_doc_max_bytes]"),
9697            "the per-file cap must fire with a notice"
9698        );
9699        assert!(
9700            blob.contains(
9701                "[supercode: instruction content truncated at core.project_doc_max_bytes]"
9702            ),
9703            "the aggregate cap must fire with a notice"
9704        );
9705        assert!(blob.len() < 33_200, "aggregate blob stayed over the cap");
9706        let _ = std::fs::remove_dir_all(&base);
9707    }
9708
9709    /// Row `instruction-file-hygiene-controls`, cc half: `cc-parity` arms
9710    /// CC's own two levers — HTML-comment stripping and `claudeMdExcludes`
9711    /// — and arms NO byte cap, because CC documents none.
9712    #[test]
9713    fn cc_parity_strips_html_comments_and_honours_excludes() {
9714        let mut config = resolved("cc-parity");
9715        assert!(config.project_doc_strip_comments, "cc strips `<!-- … -->`");
9716        assert_eq!(
9717            config.project_doc_max_bytes, None,
9718            "`project_doc_max_bytes = 0` is §3.1's spelling for uncapped"
9719        );
9720
9721        let base = scratch("hygiene");
9722        std::fs::create_dir_all(base.join(".git")).unwrap();
9723        std::fs::write(
9724            base.join("CLAUDE.md"),
9725            "KEEP-THIS<!-- MAINTAINER-NOTE -->AND-THIS",
9726        )
9727        .unwrap();
9728        std::fs::write(base.join("AGENTS.md"), "EXCLUDED-FILE-MARKER").unwrap();
9729        config.cwd = base.clone();
9730        config.project_doc_excludes = vec!["AGENTS.md".to_string()];
9731        let blob = assemble_project_instructions(&config);
9732        assert!(blob.contains("KEEP-THIS") && blob.contains("AND-THIS"));
9733        assert!(
9734            !blob.contains("MAINTAINER-NOTE"),
9735            "block HTML comments must be stripped before injection"
9736        );
9737        assert!(
9738            !blob.contains("EXCLUDED-FILE-MARKER"),
9739            "an excluded instruction file must never be read into the prompt"
9740        );
9741        let _ = std::fs::remove_dir_all(&base);
9742    }
9743
9744    /// A project layer must not be able to suppress the user's own global
9745    /// instruction files by adding an exclude pattern (§3.3 trust boundary).
9746    #[test]
9747    fn project_layer_cannot_set_instruction_excludes() {
9748        let hc = crate::configfile::HarnessConfig::from_toml_str(
9749            "schema_version = 1\n[core]\nproject_doc_excludes = [\"CLAUDE.md\"]\n",
9750        )
9751        .unwrap();
9752        let (sanitized, dropped) = crate::configfile::sanitize_for_project(&hc);
9753        assert!(sanitized.core.project_doc_excludes.is_none());
9754        assert!(dropped.iter().any(|d| d == "core.project_doc_excludes"));
9755    }
9756
9757    /// Row `environment-context-block`: both presets emit the policy line
9758    /// the row's semantics name, and the block is RE-EMITTED when the thing
9759    /// it describes moves (cx§2 "re-emitted on change").
9760    #[test]
9761    fn env_context_block_carries_policy_and_re_emits_on_change() {
9762        for preset in ["cc-parity", "cx-parity"] {
9763            let base = scratch("env");
9764            // A SIBLING, not a child: `contains` assertions below must not
9765            // be satisfiable by a path prefix.
9766            let here = base.join("here");
9767            let other = base.join("elsewhere");
9768            std::fs::create_dir_all(&here).unwrap();
9769            std::fs::create_dir_all(&other).unwrap();
9770
9771            let mut config = resolved(preset);
9772            config.cwd = here.clone();
9773            assert!(config.env_context, "{preset} must set core.env_context");
9774            let expected_policy = format!(
9775                "approval policy: {} · sandbox: {}",
9776                approval_policy_label(config.approval),
9777                sandbox_policy_label(config.sandbox),
9778            );
9779
9780            let mut agent = Agent::with_provider(config, Box::new(NoProvider));
9781            let system = agent.history[0].content.clone().unwrap_or_default();
9782            assert!(
9783                system.contains(&expected_policy),
9784                "{preset}: the environment block must state the approval/sandbox policy — {system}"
9785            );
9786            assert!(system.contains(&format!("cwd: {}", here.display())));
9787
9788            // Nothing moved ⇒ no churn (the prompt cache is not busted for
9789            // free).
9790            assert!(!agent.refresh_env_context(), "{preset}: spurious re-emit");
9791
9792            // cwd + policy move mid-session.
9793            agent.config.cwd = other.clone();
9794            agent.config.approval = crate::config::ApprovalPolicy::Untrusted;
9795            assert!(
9796                agent.refresh_env_context(),
9797                "{preset}: change not re-emitted"
9798            );
9799            let system = agent.history[0].content.clone().unwrap_or_default();
9800            assert!(
9801                system.contains(&format!("cwd: {}", other.display())),
9802                "{preset}: the fresh cwd must reach the model"
9803            );
9804            assert!(system.contains("approval policy: untrusted"));
9805            assert!(
9806                !system.contains(&format!("cwd: {}", here.display())),
9807                "{preset}: the stale block must be REPLACED, not duplicated"
9808            );
9809            assert_eq!(
9810                system.matches("# Environment").count(),
9811                1,
9812                "{preset}: exactly one environment block"
9813            );
9814            let _ = std::fs::remove_dir_all(&base);
9815        }
9816    }
9817
9818    /// Row `synthetic-context-injection-blocks`: both presets arm the
9819    /// registry, the built-in blocks reach the assembled system prompt, and
9820    /// a block spliced mid-session reaches it too.
9821    #[test]
9822    fn presets_splice_builtin_and_runtime_context_blocks() {
9823        for preset in ["cc-parity", "cx-parity"] {
9824            let config = resolved(preset);
9825            assert!(
9826                config.context_injections,
9827                "{preset} must set core.context_injections"
9828            );
9829            let mut agent = Agent::with_provider(config, Box::new(NoProvider));
9830            let system = agent.history[0].content.clone().unwrap_or_default();
9831            assert!(
9832                system.contains("# Task list"),
9833                "{preset}: a built-in ambient block must reach the prompt"
9834            );
9835
9836            assert!(agent.inject_context_block("Mid session", "SPLICED-BODY-MARKER"));
9837            let system = agent.history[0].content.clone().unwrap_or_default();
9838            assert!(
9839                system.contains("# Mid session") && system.contains("SPLICED-BODY-MARKER"),
9840                "{preset}: a runtime splice must reach the prompt"
9841            );
9842            assert_eq!(agent.spliced_context_blocks().len(), 1);
9843        }
9844    }
9845
9846    /// A deterministic stand-in for the CLI's real provider-backed
9847    /// summarizer: it records exactly what it was asked to summarize, so the
9848    /// test can prove the `/compact <focus>` text reached the summarizer's
9849    /// INPUT and not only the marker.
9850    #[derive(Debug, Default)]
9851    struct RecordingSummarizer {
9852        seen: std::sync::Mutex<Vec<String>>,
9853    }
9854
9855    impl reduce::summarize::SpanSummarizer for RecordingSummarizer {
9856        fn summarize(&self, span_text: &str) -> reduce::Result<String> {
9857            self.seen
9858                .lock()
9859                .unwrap_or_else(std::sync::PoisonError::into_inner)
9860                .push(span_text.to_string());
9861            Ok("MODEL-WRITTEN-SUMMARY".to_string())
9862        }
9863
9864        fn model_id(&self) -> &str {
9865            "test-summarizer"
9866        }
9867    }
9868
9869    fn stuffed_agent(preset: &str) -> Agent {
9870        let config = resolved(preset);
9871        let mut agent = Agent::with_provider(config, Box::new(NoProvider));
9872        for i in 0..40 {
9873            agent.history.push(ChatMessage::user(format!("turn {i}")));
9874            agent
9875                .history
9876                .push(ChatMessage::assistant(format!("reply {i}")));
9877        }
9878        agent
9879    }
9880
9881    /// Row `manual-compact-with-focus-instructions`: `/compact <focus>`
9882    /// compacts on demand (no trigger needed) and the focus text lands in
9883    /// BOTH the summarizer's input and the marker.
9884    #[test]
9885    fn presets_manual_compact_carries_focus_into_the_summarizer_and_the_marker() {
9886        for preset in ["cc-parity", "cx-parity"] {
9887            let config = resolved(preset);
9888            assert!(
9889                config.compaction_focus_instructions.is_some(),
9890                "{preset} must state core.compaction.focus_instructions"
9891            );
9892            let mut agent = stuffed_agent(preset);
9893            let summarizer = std::sync::Arc::new(RecordingSummarizer::default());
9894            agent.set_span_summarizer_arc(summarizer.clone());
9895
9896            let before = agent.history().len();
9897            assert!(
9898                agent.compact_now(Some("keep the migration steps")),
9899                "{preset}: /compact must compact on demand"
9900            );
9901            assert!(agent.history().len() < before, "{preset}: nothing dropped");
9902
9903            let marker = agent
9904                .history()
9905                .iter()
9906                .find_map(|m| m.content.as_deref())
9907                .filter(|c| c.contains("earlier conversation compacted"))
9908                .or_else(|| {
9909                    agent
9910                        .history()
9911                        .iter()
9912                        .filter_map(|m| m.content.as_deref())
9913                        .find(|c| c.contains("earlier conversation compacted"))
9914                })
9915                .unwrap_or_else(|| panic!("{preset}: no compaction marker"))
9916                .to_string();
9917            assert!(
9918                marker.contains("Focus: keep the migration steps"),
9919                "{preset}: {marker}"
9920            );
9921
9922            let seen = summarizer
9923                .seen
9924                .lock()
9925                .unwrap_or_else(std::sync::PoisonError::into_inner);
9926            assert_eq!(seen.len(), 1, "{preset}: exactly one side-call");
9927            assert!(
9928                seen[0].contains("keep the migration steps"),
9929                "{preset}: the focus must reach the summarizer INPUT — {}",
9930                &seen[0][..seen[0].len().min(200)]
9931            );
9932        }
9933    }
9934
9935    /// Row `llm-summaries-of-cleared-spans`: the model-written summary is
9936    /// produced under the presets WITHOUT `capabilities.reduction` — design
9937    /// §1.5 puts "an LLM summary of the compacted span" in core obligation
9938    /// 5, knob `[core.compaction] summarize`.
9939    #[test]
9940    fn presets_summarize_the_cleared_span_without_the_reduction_module() {
9941        for preset in ["cc-parity", "cx-parity"] {
9942            let toml = crate::presets::lookup(preset).unwrap();
9943            let r = resolve(toml, None, &ResolveOptions { strict: true }).unwrap();
9944            assert_eq!(
9945                r.modules.get("reduction"),
9946                Some(&false),
9947                "{preset}: this row must hold with the reduction module OFF"
9948            );
9949            assert!(r.config.compaction_summarize);
9950
9951            let mut agent = stuffed_agent(preset);
9952            agent.set_span_summarizer_arc(std::sync::Arc::new(RecordingSummarizer::default()));
9953            assert!(agent.compact_now(None));
9954            let marker = agent
9955                .history()
9956                .iter()
9957                .filter_map(|m| m.content.as_deref())
9958                .find(|c| c.contains("earlier conversation compacted"))
9959                .unwrap_or_else(|| panic!("{preset}: no compaction marker"));
9960            assert!(
9961                marker.contains("MODEL-WRITTEN-SUMMARY"),
9962                "{preset}: the marker must carry the model-written summary — {marker}"
9963            );
9964        }
9965    }
9966
9967    /// BP-11 (catalog "Lifecycle hooks, config-registered"): the compaction
9968    /// boundary is observable — `pre_compact` fires once the compaction is
9969    /// decided (with the manual/auto trigger named) and `post_compact` once
9970    /// the window has been rewritten, under both parity presets.
9971    #[test]
9972    fn compaction_fires_pre_and_post_lifecycle_events_under_both_presets() {
9973        use crate::config::LifecycleEvent;
9974        for preset in ["cc-parity", "cx-parity"] {
9975            let seen = std::sync::Arc::new(std::sync::Mutex::new(Vec::new()));
9976            let mut agent = stuffed_agent(preset);
9977            let sink = seen.clone();
9978            agent.set_lifecycle_hook(Box::new(move |event| {
9979                sink.lock().unwrap().push(event.clone());
9980            }));
9981            let before = agent.history().len();
9982            assert!(agent.compact_now(Some("keep the plan")), "{preset}");
9983            let after = agent.history().len();
9984            let seen = seen.lock().unwrap();
9985            assert_eq!(seen.len(), 2, "{preset}: exactly pre + post — {seen:?}");
9986            match &seen[0] {
9987                LifecycleEvent::PreCompact {
9988                    messages,
9989                    dropped,
9990                    manual,
9991                } => {
9992                    assert_eq!(*messages, before, "{preset}");
9993                    assert!(*dropped > 0, "{preset}");
9994                    assert!(*manual, "{preset}: /compact is the manual trigger");
9995                }
9996                other => panic!("{preset}: first event must be PreCompact, got {other:?}"),
9997            }
9998            match &seen[1] {
9999                LifecycleEvent::PostCompact { messages, dropped } => {
10000                    assert_eq!(*messages, after, "{preset}");
10001                    assert_eq!(
10002                        *dropped,
10003                        before - after + 1,
10004                        "{preset}: dropped span + 1 marker"
10005                    );
10006                }
10007                other => panic!("{preset}: second event must be PostCompact, got {other:?}"),
10008            }
10009        }
10010    }
10011
10012    /// The automatic trigger reports itself as such, and a window too small
10013    /// to compact fires nothing at all (no pre without a post).
10014    #[test]
10015    fn automatic_compaction_reports_the_auto_trigger_and_a_no_op_fires_nothing() {
10016        use crate::config::LifecycleEvent;
10017        let seen = std::sync::Arc::new(std::sync::Mutex::new(Vec::new()));
10018        let mut agent = stuffed_agent("cc-parity");
10019        let sink = seen.clone();
10020        agent.set_lifecycle_hook(Box::new(move |event| {
10021            sink.lock().unwrap().push(event.clone());
10022        }));
10023        agent.config.compact_after_messages = Some(10);
10024        assert!(agent.maybe_compact());
10025        assert!(matches!(
10026            seen.lock().unwrap()[0],
10027            LifecycleEvent::PreCompact { manual: false, .. }
10028        ));
10029        seen.lock().unwrap().clear();
10030        let mut small = Agent::with_provider(resolved("cc-parity"), Box::new(NoProvider));
10031        let sink = seen.clone();
10032        small.set_lifecycle_hook(Box::new(move |event| {
10033            sink.lock().unwrap().push(event.clone());
10034        }));
10035        assert!(!small.compact_now(None));
10036        assert!(seen.lock().unwrap().is_empty());
10037    }
10038
10039    /// With no summarizer installed the marker degrades to the count-only
10040    /// form — the side-call never blocks or fails compaction.
10041    #[test]
10042    fn compaction_without_a_summarizer_keeps_the_count_only_marker() {
10043        let mut agent = stuffed_agent("cc-parity");
10044        assert!(agent.compact_now(None));
10045        let marker = agent
10046            .history()
10047            .iter()
10048            .filter_map(|m| m.content.as_deref())
10049            .find(|c| c.contains("earlier conversation compacted"))
10050            .unwrap();
10051        assert!(!marker.contains("Summary of the compacted span"));
10052    }
10053
10054    /// Row `compaction-markers-persisted-in-transcript`: under the presets
10055    /// (reduction OFF) the boundary marker is written to the session
10056    /// sidecar, and says where the originals went.
10057    #[test]
10058    fn presets_persist_the_compaction_marker_to_the_transcript() {
10059        for preset in ["cc-parity", "cx-parity"] {
10060            let dir = scratch("marker");
10061            let path = dir.join("session.jsonl");
10062            let empty = supercode_interchange::session::Session::from_claude_code_str("").unwrap();
10063            let writer =
10064                supercode_interchange::sidecar::SidecarWriter::create(&path, &empty).unwrap();
10065            let mut agent = stuffed_agent(preset);
10066            agent.set_recorder(writer);
10067            assert!(agent.reduction_policy().is_none(), "{preset}");
10068
10069            assert!(agent.compact_now(None));
10070            let on_disk = std::fs::read_to_string(&path).unwrap();
10071            assert!(
10072                on_disk.contains("earlier conversation compacted"),
10073                "{preset}: the marker must reach the transcript on disk"
10074            );
10075            assert!(
10076                on_disk.contains("remain in this session's transcript sidecar"),
10077                "{preset}: the marker must say where the originals went"
10078            );
10079            let _ = std::fs::remove_dir_all(&dir);
10080        }
10081    }
10082
10083    /// Row `handoff-fresh-objective-curated-keep-set`: an in-session
10084    /// `new_context` — fresh objective, curated recent keep-set, persisted
10085    /// marker — under cx-parity, where `capabilities.reduction` is off.
10086    #[test]
10087    fn cx_parity_handoff_seeds_a_fresh_objective_with_a_curated_keep_set() {
10088        let mut agent = stuffed_agent("cx-parity");
10089        assert!(!agent.config().handoff_enabled, "reduction handoff is off");
10090        agent.history.push(ChatMessage::user("LAST-USER-TURN"));
10091        let before = agent.history().len();
10092
10093        let dropped = agent.new_context("ship the migration", Some(3));
10094        assert!(dropped > 0, "messages must be set aside");
10095        assert!(agent.history().len() < before);
10096        let system_prompt = agent.history()[0].content.clone().unwrap_or_default();
10097        assert!(
10098            system_prompt.contains("supercode") || !system_prompt.is_empty(),
10099            "the system prompt survives a handoff"
10100        );
10101        let marker = agent
10102            .history()
10103            .iter()
10104            .filter_map(|m| m.content.as_deref())
10105            .find(|c| c.contains("[handoff:"))
10106            .expect("handoff marker");
10107        assert!(marker.contains("Objective: ship the migration"));
10108        assert!(
10109            agent
10110                .history()
10111                .iter()
10112                .any(|m| m.content.as_deref() == Some("LAST-USER-TURN")),
10113            "the curated keep-set must carry the most recent turns"
10114        );
10115    }
10116
10117    /// Row `context-usage-introspection`: a live breakdown, from the same
10118    /// estimator the context guard enforces, without sending anything.
10119    #[test]
10120    fn presets_report_live_context_usage() {
10121        for preset in ["cc-parity", "cx-parity"] {
10122            let agent = stuffed_agent(preset);
10123            let usage = agent.context_usage();
10124            assert_eq!(usage.messages, agent.history().len(), "{preset}");
10125            assert!(usage.message_tokens > 0, "{preset}");
10126            assert_eq!(
10127                usage.request_tokens,
10128                usage.message_tokens + usage.tool_schema_tokens,
10129                "{preset}: the breakdown must add up"
10130            );
10131            assert!(usage.projected_tokens >= usage.request_tokens, "{preset}");
10132            assert!(usage.context_limit.is_some(), "{preset}: window known");
10133            assert!(usage.fits, "{preset}");
10134            let line = usage.summary_line();
10135            assert!(line.contains('%') && line.contains(&usage.model), "{line}");
10136
10137            // Pure: asking must not change what the next request carries.
10138            let again = agent.context_usage();
10139            assert_eq!(usage, again, "{preset}");
10140        }
10141    }
10142}
10143
10144#[cfg(test)]
10145#[path = "agent_bp10_tests.rs"]
10146mod bp10_permissions_tests;