Skip to main content

supercode_harness/
agent.rs

1//! The agent loop.
2
3use std::collections::HashSet;
4
5use crate::config::{CachePlan, Config, SteeringMode, ToolAdvertising};
6use crate::error::{Error, Result};
7use crate::provider::{self, ChatRequest, OpenAiProvider, Provider, ToolSchema};
8use crate::reduce::rehydrate::CAP_NOTICE_MARKER;
9use crate::reduce::{self, ReductionLog, ReductionPolicy};
10use crate::tools::{ToolContext, ToolRegistry};
11use supercode_interchange::session::Session;
12use supercode_interchange::sidecar::SidecarWriter;
13use supercode_interchange::{ChatMessage, Role};
14use supercode_runtime::AgentEvent;
15
16/// BP-6: how many skill bodies one user message may pull in through `$slug`
17/// mentions — Claude Code caps skill chaining at six per message (cc§7
18/// "Skill chaining"); the same ceiling bounds the mention path here.
19const MAX_SKILL_LOADS_PER_MESSAGE: usize = 6;
20
21/// BP-5: how many `@path` mentions one message may attach. A prompt is not
22/// a bulk loader; past this the user means `--file`.
23const MAX_FILE_MENTIONS_PER_MESSAGE: usize = 10;
24
25/// BP-5: ceiling on the bytes one `@path` mention contributes.
26const MAX_FILE_MENTION_BYTES: usize = 64 * 1024;
27
28/// Tool name of the `tool_search` agent intrinsic (B6). Never a registered
29/// [`crate::tools::Tool`] — intercepted in [`Agent::run_tool`] before registry
30/// lookup, so it works under any [`ToolAdvertising`] mode.
31const TOOL_SEARCH: &str = "tool_search";
32
33/// Tool name of the `expand_reduction` agent intrinsic (T12/TR-1) — the
34/// model-invocable rehydration counterpart to `tool_search`, same
35/// interception pattern. Advertised whenever a [`ReductionPolicy`] is
36/// installed, regardless of [`ToolAdvertising`] mode (see [`Self::tool_schemas`]).
37const EXPAND_REDUCTION: &str = "expand_reduction";
38
39/// Tool name of the `sidecar_search` agent intrinsic (T12/TR-1).
40const SIDECAR_SEARCH: &str = "sidecar_search";
41
42/// Tool name of the `spawn_subagent` agent intrinsic (P5-3, §2 module 9 D1
43/// "spawn tool"). Same interception pattern as [`TOOL_SEARCH`] — never a
44/// registered [`crate::tools::Tool`], intercepted in [`Agent::run_tool`]
45/// before registry lookup — but ALSO needs full `&mut self` async access
46/// (running a whole child agent loop, or `tokio::spawn`-ing one), which
47/// [`Agent::prepare_tool_call`]'s purely-synchronous intrinsics don't, so
48/// the interception point is `Self::run_tool`'s top, not
49/// `prepare_tool_call`.
50const SPAWN_SUBAGENT: &str = "spawn_subagent";
51
52/// BP-7 (catalog §4a "Review mode"): the `[core.prompts]` key the review
53/// turn's template lives under. One name for both presets — cc spells the
54/// command `/code-review`, cx spells it `/review`, and both resolve to this
55/// template, so the row's evidence is one config key, not two.
56pub const REVIEW_PROMPT_NAME: &str = "code-review";
57
58/// BP-7 (catalog §4a "Side/ephemeral Q&A"): the instruction prefixed to a
59/// side question, so the model knows it is answering ABOUT the session
60/// rather than continuing it. The exchange never enters history either way;
61/// this keeps the answer from reading like the next assistant turn.
62const SIDE_QUESTION_PREAMBLE: &str = "[side question — answer from the conversation above; this exchange is not part of the conversation and you have no tools for it]";
63
64/// Claude Code's native name for [`SPAWN_SUBAGENT`]. It is exposed only when
65/// `Config::subagents_claude_agent_alias` is enabled for a Claude import.
66const CLAUDE_AGENT: &str = "Agent";
67
68/// Claude Code spellings for core filesystem/shell tools. Imported Claude
69/// context frequently continues to call these names even when another model
70/// is driving the turn, so emulation must translate execution as well as
71/// preserve the original call/result names in the transcript.
72const CLAUDE_BASH: &str = "Bash";
73const CLAUDE_READ: &str = "Read";
74const CLAUDE_WRITE: &str = "Write";
75const CLAUDE_EDIT: &str = "Edit";
76const CLAUDE_GLOB: &str = "Glob";
77const CLAUDE_GREP: &str = "Grep";
78
79/// Claude Code scheduler compatibility intrinsics. They edit an imported
80/// [`crate::ClaudeRuntimeManifest`]; actual timer execution belongs to an
81/// embedding scheduler driver, never this agent loop.
82const CLAUDE_CRON_CREATE: &str = "CronCreate";
83const CLAUDE_CRON_DELETE: &str = "CronDelete";
84const CLAUDE_CRON_LIST: &str = "CronList";
85const CLAUDE_SCHEDULE_WAKEUP: &str = "ScheduleWakeup";
86
87/// Shared SDK steering mailbox. `accepting` and `queue` share one lock so a
88/// turn's final boundary can close acceptance atomically with its last drain;
89/// a steer can therefore never be acknowledged into the following turn.
90#[derive(Default)]
91pub(crate) struct SteerInbox {
92    queue: std::collections::VecDeque<QueuedSteer>,
93    accepting: bool,
94}
95
96struct QueuedSteer {
97    message: String,
98    sdk_bound: bool,
99}
100
101impl SteerInbox {
102    pub(crate) fn open(&mut self) {
103        self.queue.clear();
104        self.accepting = true;
105    }
106
107    pub(crate) fn enqueue(&mut self, message: String) -> bool {
108        if !self.accepting {
109            return false;
110        }
111        self.queue.push_back(QueuedSteer {
112            message,
113            sdk_bound: true,
114        });
115        true
116    }
117
118    pub(crate) fn close(&mut self) {
119        self.accepting = false;
120        self.queue.retain(|queued| !queued.sdk_bound);
121    }
122
123    fn drain(&mut self, mode: SteeringMode) -> Option<String> {
124        if self.queue.is_empty() {
125            return None;
126        }
127        match mode {
128            SteeringMode::All => Some(
129                self.queue
130                    .drain(..)
131                    .map(|queued| queued.message)
132                    .collect::<Vec<_>>()
133                    .join("\n\n"),
134            ),
135            SteeringMode::OneAtATime => self.queue.pop_front().map(|queued| queued.message),
136        }
137    }
138
139    fn drain_or_close(&mut self, mode: SteeringMode) -> Option<String> {
140        if !self.queue.iter().any(|queued| queued.sdk_bound) {
141            self.accepting = false;
142            return None;
143        }
144        self.drain(mode)
145    }
146
147    fn queue_unchecked(&mut self, message: String) {
148        self.queue.push_back(QueuedSteer {
149            message,
150            sdk_bound: false,
151        });
152    }
153
154    fn len(&self) -> usize {
155        self.queue.len()
156    }
157}
158
159struct SteerTurnGuard {
160    inbox: std::sync::Arc<std::sync::Mutex<SteerInbox>>,
161}
162
163impl SteerTurnGuard {
164    fn new(inbox: std::sync::Arc<std::sync::Mutex<SteerInbox>>) -> Self {
165        inbox
166            .lock()
167            .unwrap_or_else(std::sync::PoisonError::into_inner)
168            .accepting = true;
169        Self { inbox }
170    }
171}
172
173impl Drop for SteerTurnGuard {
174    fn drop(&mut self) {
175        self.inbox
176            .lock()
177            .unwrap_or_else(std::sync::PoisonError::into_inner)
178            .close();
179    }
180}
181
182/// Tool name of the `subagent_status` agent intrinsic (P5-3, D3
183/// "background+resume"): poll (and reap, once finished) a background child
184/// spawned via [`SPAWN_SUBAGENT`]. Only advertised when
185/// `Config::subagents_background` is on (see [`Agent::tool_schemas`]).
186const SUBAGENT_STATUS: &str = "subagent_status";
187
188/// The agent's one message verb (cc's `SendMessage`, cx's v2 `send_message`):
189/// `to` is either a STILL-RUNNING background child (delivered into its own
190/// inbox) or another session of any harness, which goes through the one send
191/// every sender uses ([`crate::mail_send`]). Only advertised when
192/// `Config::subagents_background` is on.
193const SEND_MESSAGE: &str = "send_message";
194
195/// BP-7 (catalog §4a "Background subagents + resume": "resumable with
196/// context intact"): continue a FINISHED child with its own transcript
197/// restored, rather than starting a fresh one that has to be re-briefed.
198const SUBAGENT_RESUME: &str = "subagent_resume";
199
200/// Tool name of the `background_exec` agent intrinsic (P5-6, §2 module 4
201/// `tools.background` D1 "background exec"). Unlike [`SPAWN_SUBAGENT`], this
202/// needs no async child-agent loop — spawning a process
203/// (`tokio::process::Command::spawn`) is itself synchronous — so, like
204/// [`TOOL_SEARCH`], it is intercepted in [`Agent::prepare_tool_call`], not
205/// [`Agent::run_tool`].
206const BACKGROUND_EXEC: &str = "background_exec";
207
208/// Tool name of the `background_status` agent intrinsic (P5-6, D1 "monitor/
209/// event feed"): poll a background job's run status, drain its newly
210/// captured output as an [`AgentEvent::BackgroundOutput`] event, and reap it
211/// (remove it from [`Agent::background_jobs`]) once it has exited or been
212/// killed.
213const BACKGROUND_STATUS: &str = "background_status";
214
215/// Tool name of the `background_list` agent intrinsic (P5-6, D10
216/// "bg-manager"): list every background job this agent is currently
217/// tracking (running or finished-but-unreaped), without draining output or
218/// reaping anything.
219const BACKGROUND_LIST: &str = "background_list";
220
221/// Tool name of the `background_kill` agent intrinsic (P5-6, D10
222/// "bg-manager"): kill a background job's real OS process
223/// (`tokio::process::Child::start_kill`) and reap it immediately.
224const BACKGROUND_KILL: &str = "background_kill";
225
226/// P4e (§3.1 `core.parallel_tool_calls`): the synchronous outcome of
227/// [`Agent::prepare_tool_call`] — either a result already in hand (an
228/// intrinsic, or a call refused before it ever reached `Tool::execute`), or
229/// a plain registry-tool call ready for the (possibly concurrent) async
230/// `execute()` step.
231enum PreparedCall {
232    /// A final `(output, is_error)` result — no `Tool::execute` call is
233    /// coming for this one.
234    Done((String, bool)),
235    /// Passed every synchronous check; `execute(args, &ctx)` on the named
236    /// registry tool is the only remaining step.
237    Ready {
238        name: String,
239        args: serde_json::Value,
240    },
241}
242
243/// Marker prefix of the notice [`Agent::cap_tool_output`] appends to an
244/// oversized tool result kept in `history` (the recorder receives the full
245/// UX-26 (B7-warn): current wall-clock time as unix milliseconds, the same
246/// unit [`supercode_interchange::sidecar::rfc3339_to_ms`] parses session timestamps into —
247/// lets [`Agent::build_request_messages`] compare "now" against a
248/// cross-process signal (a loaded session's last message timestamp) on
249/// equal footing with an in-process one (this agent's own last annotated
250/// send). Saturates to 0 on a pre-epoch clock rather than panicking (never
251/// happens on real hardware, but `duration_since` can theoretically error).
252fn now_ms() -> i64 {
253    std::time::SystemTime::now()
254        .duration_since(std::time::UNIX_EPOCH)
255        .map(|d| d.as_millis() as i64)
256        .unwrap_or(0)
257}
258
259/// P5-3: process-wide sequence number backing [`next_subagent_id`] —
260/// disambiguates two spawns landing in the same millisecond (which
261/// `now_ms()` alone cannot).
262static SUBAGENT_ID_SEQ: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(0);
263
264/// P5-3: a fresh, process-unique child agent id (`"agent-<hex-ts>-<hex-seq>"`
265/// — the native analog of Claude Code's `agent-<id>` naming, see
266/// `supercode_interchange::session::SessionMeta::agent_id`'s doc comment).
267fn next_subagent_id() -> String {
268    let seq = SUBAGENT_ID_SEQ.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
269    format!("agent-{:x}-{:x}", now_ms(), seq)
270}
271
272/// P5-4: the shape [`Agent::child_approval_handler_factory`]/
273/// [`Agent::set_child_approval_handler_factory`] share — factored into its
274/// own alias (clippy `type_complexity`) rather than spelled out inline at
275/// both use sites.
276type ChildApprovalHandlerFactory = dyn Fn(
277        String,
278        std::sync::Arc<std::sync::Mutex<Vec<crate::subagents::QueuedApproval>>>,
279    ) -> std::sync::Arc<dyn crate::permissions::PermissionsApprovalHandler>
280    + Send
281    + Sync;
282
283/// BP-4 (catalog:109 "Context-usage introspection"): the live
284/// context-window accounting [`Agent::context_usage`] reports — cc's
285/// `/context` grid and cx's `/status` + `get_context_remaining` in one
286/// shape, over the numbers `resume --dry-run`'s preflight already computes.
287///
288/// Every token figure is the SAME estimate the context guard enforces
289/// (`supercode_runtime`), so what this reports and what refuses an oversized
290/// turn can never disagree.
291#[derive(Debug, Clone, PartialEq, Eq, serde::Serialize, serde::Deserialize)]
292pub struct ContextUsage {
293    /// The model the accounting is against.
294    pub model: String,
295    /// Messages in the projected request view (reduction stubs included).
296    pub messages: usize,
297    /// Estimated tokens for those messages.
298    pub message_tokens: u64,
299    /// Tools advertised on the next request.
300    pub tool_count: usize,
301    /// Estimated tokens for the serialized tool-schema array — a real part
302    /// of the wire request, and the half a message-only count misses.
303    pub tool_schema_tokens: u64,
304    /// `message_tokens + tool_schema_tokens`.
305    pub request_tokens: u64,
306    /// `request_tokens` with the guard's safety margin applied — the figure
307    /// the context guard actually compares.
308    pub projected_tokens: u64,
309    /// Headroom the guard reserves for the model's own reply.
310    pub response_reserve_tokens: u64,
311    /// The model's context window, when known.
312    pub context_limit: Option<u64>,
313    /// Tokens still available after the reply reserve, `0` when unknown.
314    pub remaining_tokens: u64,
315    /// `projected_tokens` as a whole percentage of the window (rounded),
316    /// `0` when the window is unknown. An integer so this whole struct
317    /// stays `Eq`-comparable on the frontend wire.
318    pub used_pct: u32,
319    /// Whether the next request would pass the context guard.
320    pub fits: bool,
321}
322
323impl ContextUsage {
324    /// One human line, the shape a `/context` command prints.
325    pub fn summary_line(&self) -> String {
326        match self.context_limit {
327            Some(limit) => format!(
328                "{} · {}% of {} used ({} projected, {} left) · {} messages {} · {} tool schemas {}",
329                self.model,
330                self.used_pct,
331                supercode_runtime::fmt_approx_tokens(limit),
332                supercode_runtime::fmt_approx_tokens(self.projected_tokens),
333                supercode_runtime::fmt_approx_tokens(self.remaining_tokens),
334                self.messages,
335                supercode_runtime::fmt_approx_tokens(self.message_tokens),
336                self.tool_count,
337                supercode_runtime::fmt_approx_tokens(self.tool_schema_tokens),
338            ),
339            None => format!(
340                "{} · context window unknown · {} projected · {} messages {} · {} tool schemas {}",
341                self.model,
342                supercode_runtime::fmt_approx_tokens(self.projected_tokens),
343                self.messages,
344                supercode_runtime::fmt_approx_tokens(self.message_tokens),
345                self.tool_count,
346                supercode_runtime::fmt_approx_tokens(self.tool_schema_tokens),
347            ),
348        }
349    }
350}
351
352/// A stateful agent: configuration, a model transport, a tool set, and the
353/// running conversation. Drive it with [`Agent::send`].
354/// BP-13 — one hop the run loop's failure-fallback pass performed: the
355/// model it was on, the model it moved to, and the provider failure that
356/// made it move.
357#[derive(Debug, Clone, PartialEq, Eq)]
358pub struct FallbackHop {
359    /// The model that failed.
360    pub from: String,
361    /// The next chain entry, which the request was re-sent against.
362    pub to: String,
363    /// The failure, rendered — the record's `reason`.
364    pub reason: String,
365}
366
367/// BP-13 — whether `error` is the kind of failure ANOTHER MODEL could
368/// plausibly answer, i.e. one the fallback chain exists for.
369///
370/// Deliberately narrow: rate limiting (429) and server-side failures (5xx,
371/// which is where "overloaded" lives) are properties of the model/endpoint
372/// that was asked, so asking a different one is a real remedy. Everything
373/// else — a bad request, a refused key, a decode failure, a tool error —
374/// is the CALLER's problem and would fail identically against every entry
375/// in the chain, so walking it would only multiply the same error by three.
376/// The transport's own retry (`OpenAiProvider::send_with_retry`) has
377/// already run and given up by the time this is consulted.
378pub fn is_failover_worthy(error: &Error) -> bool {
379    matches!(error, Error::Provider { status, .. } if *status == 429 || *status >= 500)
380}
381
382pub struct Agent {
383    config: Config,
384    provider: std::sync::Arc<dyn Provider>,
385    registry: ToolRegistry,
386    history: Vec<ChatMessage>,
387    ctx: ToolContext,
388    /// Cumulative output (completion) tokens across every `send` on this agent.
389    total_output_tokens: u64,
390    /// Names of non-core tools discovered via `tool_search` (B6): advertised
391    /// starting with the *next* request once populated.
392    activated_tools: HashSet<String>,
393    /// The live sidecar writer (A3), if this agent is recording. `None` is
394    /// today's behavior, at zero cost: every append point becomes a no-op.
395    recorder: Option<SidecarWriter>,
396    /// BP-8 (catalog:150 "Append-only durable transcript"): the live
397    /// append-only journal, if one is installed
398    /// ([`Self::set_journal`], armed by the caller when
399    /// [`Config::session_append_only`] is on). Behind an `Arc<Mutex<_>>`
400    /// rather than owned outright because the queue doors
401    /// ([`Self::queue_steer`]) take `&self` — a pending input has to be
402    /// recorded from a shared handle while a turn holds `&mut Agent`.
403    /// `None` (the default) is a no-op at every append point: today's
404    /// behavior, no file created.
405    journal: Option<std::sync::Arc<std::sync::Mutex<crate::session_journal::SessionJournal>>>,
406    /// BP-8 (catalog:151 "In-place conversation tree"): the live
407    /// `SessionTree` for this session, materialized when
408    /// [`Config::session_tree_enabled`] is on. Every recorded message
409    /// becomes a node, and [`Self::rewind_conversation`] moves the active
410    /// branch's leaf — the tree is what makes a rewind lossless (the old
411    /// leaf is preserved under a sibling branch) rather than a truncation.
412    /// `None` (the default, and every preset that leaves the module off) is
413    /// zero cost: nothing is built and nothing is persisted.
414    session_tree: Option<supercode_interchange::session_tree::SessionTree>,
415    /// BP-8 (catalog:152 "Rewind/rollback conversation"): tails removed by
416    /// rewinds that have not been undone, newest last. Restored from the
417    /// journal on resume, so "undo the rewind" survives a restart.
418    rewind_undo: Vec<Vec<ChatMessage>>,
419    /// BP-8 (catalog:156): the plan as last written to the journal —
420    /// compared against `ctx.plan` so an unchanged plan is not re-journaled
421    /// on every loop iteration.
422    journaled_plan: Vec<crate::session_journal::PlanEntry>,
423    /// Reversible reduction policy (A5/A7/A10). `None` is today's behavior,
424    /// at zero cost: every provider request is built from `self.history`
425    /// verbatim, exactly as before this landed.
426    reduction_policy: Option<ReductionPolicy>,
427    /// The accumulating reduction log (A5): fed back into
428    /// [`reduce::project_messages`] on every request-build so already-applied
429    /// reductions reproduce verbatim across turns and `send` calls (prefix
430    /// stability). `history` itself is never touched by this — see
431    /// `Self::run_loop`.
432    reduction_log: ReductionLog,
433    /// B7: length of the stable, byte-identical-across-turns prefix at the
434    /// front of [`Self::history`] — this agent's own system message plus
435    /// every message of a previously-imported session — set by
436    /// [`Self::load_session`]. `None` (the default) means no session has been
437    /// loaded, so [`crate::provider::apply_cache_plan`] has nothing to
438    /// annotate even under [`CachePlan::ImportedPrefix`].
439    imported_prefix_len: Option<usize>,
440    /// BP-11: set for the duration of a `/compact` so the `pre_compact`
441    /// observer can tell a manual compaction from an automatic trigger.
442    compacting_manually: bool,
443    /// BP-4 (catalog:90 "Environment context block", cx§2 "re-emitted on
444    /// change"): the `# Environment` block currently spliced into
445    /// `history[0]`, verbatim — `None` when `core.env_context` is off (or
446    /// on a construction path that assembles no prompt). Kept so
447    /// [`Self::refresh_env_context`] can locate and replace exactly this
448    /// text when cwd, approval/sandbox policy or the git branch moves
449    /// mid-session, instead of leaving the model reading a block that
450    /// stopped being true.
451    env_context_live: Option<String>,
452    /// BP-5 (catalog D2 "Per-model-family base-prompt selection"): the base
453    /// system prompt currently spliced at the head of `history[0]`,
454    /// verbatim. Kept for the same reason [`Self::env_context_live`] is:
455    /// when the model changes ([`Self::set_model`]) the family's prompt
456    /// changes with it, and the stale text has to be located and REPLACED
457    /// rather than left in front of the new model.
458    base_prompt_live: String,
459    /// BP-5 (catalog D2 "Shell-output injection in templates/skills"): the
460    /// permission-engine authorization every `` !`cmd` `` in a skill or
461    /// command body runs under. Inert (executes nothing) unless
462    /// `[core.skills] shell_injection` is on.
463    shell_injection: crate::skills::ShellInjection,
464    /// BP-4 (catalog:91): blocks spliced into the context AFTER
465    /// construction — see [`Self::inject_context_block`] and
466    /// [`crate::context_injection`]. Empty by default, at zero cost.
467    spliced_context_blocks: Vec<crate::config::ContextInjectionBlock>,
468    /// TR-7 (T20): the injectable side-call ([`reduce::summarize::SpanSummarizer`])
469    /// used to summarize an A10 `TurnsCleared` span, if one is installed
470    /// ([`Self::set_span_summarizer`]). `None` is today's behavior, at zero
471    /// cost: `Self::build_request_messages` never calls
472    /// [`reduce::prepare_cleared_turns_summary`] without one, so
473    /// `policy.summarize_cleared_turns` being on with no summarizer
474    /// installed behaves exactly like it being off (deterministic stub only)
475    /// — never a panic, never a blocked request.
476    span_summarizer: Option<std::sync::Arc<dyn reduce::summarize::SpanSummarizer + Send + Sync>>,
477    /// TR-8 (T5): the tool-schema tier signature (global knob + per-tool
478    /// overrides) as of the last request this agent built, or `None` before
479    /// the first request. Compared against the CURRENT signature at the top
480    /// of every `Self::build_request_messages` call so a tier change made
481    /// mid-session (via [`Self::set_schema_tier`] /
482    /// [`Self::set_tool_schema_tier`]) is detected and flagged to the B7
483    /// cache planner as a cache-bust event (`provider::tier_change_is_cache_bust`).
484    last_tool_schema_tier_signature: Option<u64>,
485    /// PARITY-18 D4 — the target model's context-window size, if the caller
486    /// has armed the guard via [`Self::set_context_limit`]. `None` (the
487    /// default) means no guard: every request is sent unconditionally.
488    /// CLI entry points arm it for their resolved model; direct SDK callers
489    /// retain explicit control through [`Self::set_context_limit`].
490    /// Once set, `Self::run_loop` re-checks
491    /// [`supercode_runtime::context_guard`] before EVERY request it builds —
492    /// not just the first — so "never sends an over-context request" holds
493    /// for the whole session, not only a one-shot preflight.
494    context_limit: Option<u64>,
495    /// PARITY-18 D3 — becomes `true` the first time `Self::run_loop`
496    /// actually reaches its real send site (immediately before
497    /// [`Provider::complete`]). Exposed via [`Self::request_issued`] so a
498    /// caller can report "request sent" truthfully — never asserted ahead
499    /// of time, so a pre-delivery failure (guard refusal, a build error) or
500    /// an interactive session that quits before any turn completes is
501    /// reported honestly as "not sent".
502    requests_issued: bool,
503    /// UX-26 (B7-warn): unix-ms wall-clock time this agent last knew the
504    /// active [`CachePlan::ImportedPrefix`] breakpoint to be warm. Seeded by
505    /// [`Self::load_session`] from the just-loaded session's OWN last
506    /// message timestamp (`metadata["timestamp"]`, parsed via
507    /// [`supercode_interchange::sidecar::rfc3339_to_ms`]) — a cross-process signal: how long
508    /// the resumed conversation has sat idle since ANY tool last touched it,
509    /// which is exactly when Anthropic's server-side cache entry (if one
510    /// ever existed) was last capable of being warm. Refreshed to "now"
511    /// every time `Self::run_loop` actually sends a cache-annotated
512    /// request (an in-process signal: idle time between this agent's own
513    /// turns). `None` when no imported prefix exists yet, or the loaded
514    /// session's last message carries no parseable timestamp — never
515    /// guessed, so the TTL check in [`provider::cache_cold_reason`] simply
516    /// doesn't fire rather than risk a false positive.
517    last_cache_activity_ms: Option<i64>,
518    /// UX-26: whether a PRIOR request already carried a cache_control
519    /// annotation for the current [`Self::imported_prefix_len`] — i.e.
520    /// whether reuse is genuinely "expected" on the NEXT annotated request.
521    /// `false` until the first annotated request goes out (that one is
522    /// establishing the cache entry, a legitimate write, never a "miss") and
523    /// reset to `false` by [`Self::load_session`] whenever the imported
524    /// prefix itself changes.
525    cache_established: bool,
526    /// UX-26 scratch: this turn's cache-warmth context, computed once at the
527    /// top of `Self::build_request_messages` (before the request is sent,
528    /// while `effective_cache_plan`/`busted` are in scope) and consumed once
529    /// in `Self::run_loop` right after `usage` comes back — never read
530    /// across turns, so a stale value can't leak. `(will_annotate,
531    /// cache_established, idle_secs)` — see [`provider::cache_cold_reason`]
532    /// for what each of the first two independently gates.
533    pending_cache_turn: (bool, bool, Option<i64>),
534    /// P4b: the injectable auto-title side-call ([`Self::set_session_titler`]),
535    /// mirroring `Self::span_summarizer`'s "installing one alone changes
536    /// nothing" contract — `Config::auto_title` is the actual gate a caller
537    /// consults before invoking [`Self::auto_title`].
538    session_titler: Option<std::sync::Arc<dyn crate::session_title::SessionTitler + Send + Sync>>,
539    /// P4b (§1.6, catalog §4a "persisted per-turn usage records"): every
540    /// [`crate::usage_log::UsageRecord`] recorded so far this agent's
541    /// lifetime. Always accumulated (cheap, small) regardless of whether a
542    /// caller ever persists it — see [`Self::usage_records`]/
543    /// [`Self::save_usage_log`].
544    usage_log: Vec<crate::usage_log::UsageRecord>,
545    /// P4b: 0-based index of the NEXT model round-trip, for
546    /// [`crate::usage_log::UsageRecord::turn`].
547    turn_index: usize,
548    /// BP-7 (catalog §4a "Turn/step bracketing records"): the per-round-trip
549    /// marker log — context/usage/finish brackets plus the retry, abort,
550    /// effort and goal markers. Persisted beside the session as
551    /// `<name>.events.jsonl` (see [`Self::save_turn_records`]).
552    turn_records: Vec<crate::turn_record::TurnRecord>,
553    /// BP-7: retries the transport reported, drained after every
554    /// `complete()` so each notice attaches to the round-trip that produced
555    /// it. Only the HTTP provider built by [`Self::new`] writes into this;
556    /// an injected provider simply never records anything.
557    retry_log: std::sync::Arc<crate::provider::RetryLog>,
558    /// BP-7 (catalog §4a "Per-turn cost/usage accounting", "Turn/budget
559    /// caps"): the price to bill this agent's model at, resolved at
560    /// construction from [`Config::price_input_per_mtok`]/
561    /// [`Config::price_output_per_mtok`] or [`crate::pricing`]'s table, and
562    /// re-resolved by [`Self::set_model`]. `None` = unpriceable, so no cost
563    /// is recorded (never a guess).
564    model_price: Option<crate::pricing::ModelPrice>,
565    /// BP-7: dollars this agent has spent across its whole lifetime — the
566    /// counter [`Config::max_budget_usd`] is measured against.
567    total_cost_usd: f64,
568    /// BP-7: tool calls this agent has executed across its whole lifetime —
569    /// the counter [`Config::max_steps`] is measured against.
570    total_steps: usize,
571    /// BP-7 (catalog §4a "Background subagents + resume"): a finished
572    /// child's post-system-prompt transcript, kept after the reap so
573    /// `subagent_resume` can restore its context in-process. A session
574    /// with a subagent store attached also has it on disk; this makes
575    /// resume work for an embedder that never attached one.
576    reaped_subagents: std::collections::HashMap<String, Vec<ChatMessage>>,
577    /// BP-7 (catalog §4a "Goals"): the session's standing objective, when
578    /// `capabilities.todos.goals` is on and one has been set. Restated at
579    /// the TAIL of every request while it stands (see
580    /// [`crate::goals::GoalRecord::reminder`]) and persisted as
581    /// `<session>.goal.json` — never written into `history`, so the
582    /// transcript stays exactly what the conversation was.
583    goal: Option<crate::goals::GoalRecord>,
584    /// P4b (§1.7/§3.1 `core.steering`, pi§3 semantics): queued mid-turn
585    /// steering messages — drained at the top of `Self::run_loop`'s next
586    /// iteration (pi's "steer = after current tool calls"). Empty by
587    /// default, at zero cost: `Self::run_loop` skips the drain entirely
588    /// when empty.
589    steer_queue: std::sync::Arc<std::sync::Mutex<SteerInbox>>,
590    /// P4b: queued follow-up messages — drained only once the loop is
591    /// otherwise idle (pi's "follow-up = at idle"), i.e. exactly the point
592    /// `Self::run_loop` would otherwise return a final answer.
593    follow_up_queue: std::collections::VecDeque<String>,
594    /// P4c (§5.2 P4 "doom-loop breaker", §3.1 `core.doom_loop_threshold`):
595    /// `(tool name, canonical JSON args)` of the most recent tool call, if
596    /// [`Config::doom_loop_threshold`] is armed — `None` before the first
597    /// call this agent has run. See [`Self::check_doom_loop`].
598    doom_loop_last_call: Option<(String, String)>,
599    /// P4c: how many times [`Self::doom_loop_last_call`] has repeated
600    /// consecutively so far (starts at 1 on the call that SET it).
601    doom_loop_streak: u32,
602    /// P4c (§1.10/§3.1 `core.model_switch.allow_switch`): every
603    /// [`crate::model_change::ModelChangeRecord`] [`Self::switch_model`] has
604    /// created so far this agent's lifetime. Always empty when
605    /// `Config::model_switch_allow_switch` is off (the default) or no
606    /// switch has happened yet.
607    model_change_log: Vec<crate::model_change::ModelChangeRecord>,
608    /// P4e (§1.6/§3.1 `core.session.git_metadata`, catalog:331): captured
609    /// once at construction when [`Config::session_git_metadata`] is on;
610    /// `None` when the gate is off (the default) or the best-effort git
611    /// probe found nothing (not a repo, `git` missing). See
612    /// [`Self::git_metadata`]/[`Self::save_git_metadata`].
613    git_metadata: Option<crate::git_metadata::GitMetadataRecord>,
614    /// P5-1 (§2.10, session-scoped "approve for session" cache): populated
615    /// only when a [`crate::permissions::PermissionsApprovalHandler`]
616    /// returns [`crate::permissions::ApprovalOutcome::AllowForSession`] —
617    /// see [`Self::prepare_tool_call`]'s `Config::permissions_enabled`
618    /// branch. Always constructed (cheap, empty) regardless of whether the
619    /// engine is ever active — the same "zero cost when off" posture as
620    /// [`Self::doom_loop_last_call`].
621    permissions_approval_cache: crate::permissions::ApprovalCache,
622    /// P5-1: the non-interactive decision seam a caller installs via
623    /// [`Self::set_permissions_approval_handler`] — mirrors
624    /// `Self::span_summarizer`/[`Self::session_titler`]'s "installing one
625    /// alone changes nothing, `Config::permissions_enabled` is the actual
626    /// gate" pattern. `None` (the default) means every `Ask`-tier decision
627    /// is denied (fail-closed — see
628    /// `crate::permissions::approval::PermissionsApprovalHandler`'s doc
629    /// comment).
630    permissions_approval_handler:
631        Option<std::sync::Arc<dyn crate::permissions::PermissionsApprovalHandler>>,
632    /// P5-2 (§2 module 15 D7 row 4 "prompts-as-commands"): MCP server
633    /// prompts registered via [`Self::register_mcp_prompt`], keyed by their
634    /// ALREADY-NAMESPACED command name (`mcp__<server>__<prompt>` — see
635    /// [`crate::mcp::McpPromptSource`]'s doc comment for why that namespace
636    /// is what keeps an untrusted server's prompt from ever colliding with
637    /// a trusted `Config::prompts` entry). Empty by default, at zero cost:
638    /// [`Self::expand_prompt_async`] only consults this after
639    /// `Config::prompts` finds no match.
640    mcp_prompts: std::collections::HashMap<String, Box<dyn crate::sdk::SdkPromptSource>>,
641    /// BP-6 (catalog D2 "Skills (progressive-disclosure packages)", D7
642    /// "Skill discovery from multiple roots"): the SKILL.md packages
643    /// discovered for this config, frontmatter only — name, description,
644    /// version and the manifest path. Never a body: a body is read from
645    /// disk on invocation (`/name`, `/skill:name`, a `$slug` mention, or
646    /// the `skill` tool) and nowhere else. Empty unless `[core.skills]` is
647    /// on AND names a harness whose roots to read.
648    skills: Vec<crate::skills::LoopSkill>,
649    /// P5-3 (§2 module 9): how deep in the spawn tree THIS agent is — `0`
650    /// for a top-level agent. Set from [`Config::subagent_depth`] at
651    /// construction; `Self::run_spawn_subagent` builds a child `Config`
652    /// with `subagent_depth = self.subagent_depth + 1` and ALSO overwrites
653    /// the freshly-built child `Agent`'s own field to match (belt-and-
654    /// suspenders — the child never has to trust its own `Config` alone).
655    subagent_depth: usize,
656    /// P5-3 (resource bound, "must not fork-bomb"): the shared, tree-wide
657    /// concurrency gauge every spawn (this agent's own, and every
658    /// descendant's) increments/decrements against
659    /// (`crate::subagents::try_acquire`/`ConcurrencyGuard`). A TOP-level
660    /// agent gets a fresh `Arc::new(AtomicUsize::new(0))` at construction;
661    /// `Self::run_spawn_subagent` clones this SAME `Arc` into every child it
662    /// spawns (never a fresh one), so a cap of N holds across the WHOLE
663    /// tree regardless of its branching shape — a parent with 3 children
664    /// each spawning 3 more shares one counter, not nine independent ones.
665    subagent_concurrency_gauge: std::sync::Arc<std::sync::atomic::AtomicUsize>,
666    /// P5-3 (D3 "background+resume"): background subagents this agent has
667    /// spawned and not yet reaped via `subagent_status`, keyed by their
668    /// `child_agent_id`. Each entry's `JoinHandle` moves its own
669    /// [`crate::subagents::ConcurrencyGuard`] into the spawned task, so the
670    /// concurrency slot is held for exactly as long as the child is
671    /// actually running, independent of whether/when the parent polls.
672    background_subagents: std::collections::HashMap<String, BackgroundSubagent>,
673    /// P5-3 (§2.2 C6 "parent-surfaced queue"): approval requests a
674    /// `background_prompts = "parent"` child raised, queued here rather
675    /// than blocking (see [`crate::subagents::QueuedApproval`]'s doc
676    /// comment — each is already resolved `Deny` by the time it lands
677    /// here). Exposed read-only via [`Self::pending_child_approvals`].
678    /// Always constructed (cheap, empty) regardless of whether background
679    /// spawning is ever used, same "zero cost when off" posture as
680    /// [`Self::permissions_approval_cache`].
681    pending_child_approvals:
682        std::sync::Arc<std::sync::Mutex<Vec<crate::subagents::QueuedApproval>>>,
683    /// P5-4 (tui, closes the P5-3 §2.2 C6 deferred chain — see
684    /// [`crate::subagents::ParentQueueApprovalHandler`]'s doc comment for
685    /// the "never blocks" contract this OVERRIDES only when a factory is
686    /// installed): when `Some`, `Self::run_spawn_subagent` uses THIS
687    /// factory — instead of constructing the default never-blocking
688    /// [`crate::subagents::ParentQueueApprovalHandler`] — to build the
689    /// `PermissionsApprovalHandler` a `background_prompts = "parent"`
690    /// child gets. Installed via
691    /// [`Self::set_child_approval_handler_factory`] by a `tui` embedder
692    /// that wants queued child approvals to be genuinely ANSWERABLE
693    /// (blocks the child's tool call until the parent resolves it, or
694    /// denies if the factory's handler's channel is ever dropped/closed —
695    /// still fail-closed, never a hang past process lifetime). `None` (the
696    /// default) preserves P5-3's shipped behavior byte-for-byte: every
697    /// `background_prompts = "parent"` child still gets the immediate-deny
698    /// `ParentQueueApprovalHandler`, and [`Self::pending_child_approvals`]
699    /// stays exactly the read-only audit view it already is.
700    child_approval_handler_factory: Option<std::sync::Arc<ChildApprovalHandlerFactory>>,
701    /// P5-3 (D5 "subagent transcripts… persisted + linked"): an optional
702    /// `(store, this agent's own session name)` pair installed via
703    /// [`Self::set_subagent_store`] — mirrors [`Self::set_recorder`]/
704    /// [`Self::set_span_summarizer`]'s "installing one alone changes
705    /// nothing" pattern. `None` (the default) means a spawned child's
706    /// transcript/lineage is still joined back into THIS agent's context
707    /// (the foreground/background mechanics work either way) but nothing
708    /// is written to a [`crate::store::SessionStore`] — no behavior change
709    /// for any caller that never installs one (e.g. every pre-P5-3 caller).
710    subagent_store: Option<(std::sync::Arc<crate::store::SessionStore>, String)>,
711    /// Imported Claude runtime state. The manifest can be paused or active,
712    /// but this Agent contains no scheduler or timer handle; an embedding
713    /// driver owns execution and persistence.
714    claude_runtime_manifest: Option<crate::claude_runtime_state::ClaudeRuntimeManifest>,
715    /// P5-6 (§2 module 4 `tools.background`, D10 "bg-manager"): background
716    /// OS processes spawned via `background_exec`, keyed by job id, tracked
717    /// until reaped (a terminal `background_status` poll, or an explicit
718    /// `background_kill`) — see [`BackgroundJob`]'s doc comment. Always
719    /// constructed (cheap, empty), same "zero cost when off" posture as
720    /// [`Self::background_subagents`].
721    background_jobs: std::collections::HashMap<String, BackgroundJob>,
722    /// P5-6 (resource bound, mirroring [`Self::subagent_concurrency_gauge`]'s
723    /// own precedent): the shared concurrency gauge every `background_exec`
724    /// call on this agent increments/decrements against
725    /// (`crate::subagents::try_acquire`/`ConcurrencyGuard` — reused
726    /// verbatim, a second independent gauge instance scoped to background
727    /// JOBS rather than subagent SPAWNS).
728    background_concurrency_gauge: std::sync::Arc<std::sync::atomic::AtomicUsize>,
729    /// P5-9 (§2 module 20 `checkpoint`): the write-path-interception
730    /// observer installed on [`Self::ctx`]'s `write_observer` (as a
731    /// `dyn WriteObserver`), held here ADDITIONALLY as its concrete type so
732    /// `Self::run_loop` can call
733    /// [`crate::checkpoint::CheckpointObserver::begin_turn`] once per turn.
734    /// `None` when `Config::checkpoint_enabled` is `false` (the default) or
735    /// the shadow store failed to open — see
736    /// [`crate::checkpoint::observer_for_config`].
737    checkpoint_observer: Option<std::sync::Arc<crate::checkpoint::CheckpointObserver>>,
738    /// P5-11 (§2 module 28 `lsp`): the LSP server registry installed (via
739    /// `crate::lsp::LspDiagnosticsObserver`) on `Self::ctx`'s
740    /// `write_observer` chain, held here ADDITIONALLY as its concrete type
741    /// so `impl Drop for Agent` can reach
742    /// [`crate::lsp::LspManager::kill_all_sync`] (no orphaned language-
743    /// server processes) and a clean-exit caller can reach
744    /// [`crate::lsp::LspManager::shutdown_all`] for a graceful handshake.
745    /// `None` when `Config::lsp_enabled` is `false` (the default).
746    lsp_manager: Option<std::sync::Arc<crate::lsp::LspManager>>,
747}
748
749/// P5-6 (§2 module 4 `tools.background`): one background-spawned OS process
750/// this agent is tracking, awaiting a `background_status`/`background_list`
751/// poll (or `background_kill`/agent drop) to reap or terminate it.
752///
753/// **Real process, not a child agent.** Unlike [`BackgroundSubagent`] (which
754/// wraps a whole recursive child [`Agent`] loop against the SAME mock/real
755/// provider), this wraps a plain OS subprocess spawned via
756/// `crate::tools::build_sandboxed_sh` — the exact function
757/// [`crate::tools::BashTool::execute`] itself calls, so a background
758/// command gets byte-identical sandboxing/cwd/env handling to a foreground
759/// `bash` call (build brief: "reuse the bash tool's execution + sandbox
760/// path").
761struct BackgroundJob {
762    /// The live process handle — kept directly on the job (not moved into a
763    /// spawned task) so [`Agent::run_background_status`]/
764    /// [`Agent::run_background_list`] can call the SYNCHRONOUS,
765    /// non-blocking `Child::try_wait` to observe exit status, and
766    /// [`Agent::run_background_kill`]/[`impl Drop for Agent`] can call the
767    /// SYNCHRONOUS `Child::start_kill` for a REAL process kill — never just
768    /// a `tokio::task::JoinHandle::abort` (which would only cancel a Rust
769    /// future, not the OS process it spawned). `kill_on_drop(true)` was set
770    /// at spawn time as defense-in-depth: even a `BackgroundJob` dropped
771    /// through some path OTHER than the explicit kill call sites below
772    /// still kills its child (a documented tokio behavior; a no-op if the
773    /// process already exited).
774    child: tokio::process::Child,
775    /// The exact command text this job is running — the SAME text that was
776    /// already checked against the permissions engine at spawn time (see
777    /// [`Agent::background_permission_denial`]).
778    command: String,
779    /// The OS process id, captured once at spawn time (before `child` is
780    /// ever mutated) — surfaced in every status/list/kill result, and the
781    /// only thing an OUTSIDE observer (e.g. a test proving real
782    /// termination) needs to check liveness independent of this process's
783    /// own bookkeeping.
784    pid: Option<u32>,
785    /// Bounded, incrementally-appended combined stdout+stderr capture —
786    /// written to by the reader tasks [`Agent::run_background_exec`] spawns
787    /// right after `child.stdout`/`child.stderr` are taken, read by every
788    /// status/list poll. Shared via `Arc` since the reader tasks outlive
789    /// this method call.
790    output: std::sync::Arc<supercode_runtime::background::CapturedOutput>,
791    /// Unix-ms wall-clock time the spawn happened.
792    started_at_ms: i64,
793    /// Set by [`Agent::run_background_kill`] — [`Agent::run_background_status`]/
794    /// [`Agent::run_background_list`] report [`supercode_runtime::background::JobStatus::Killed`]
795    /// unconditionally once this is `true`, rather than racing
796    /// `Child::try_wait` to see whether the kill signal has landed yet.
797    killed: bool,
798    /// The concurrency-gauge slot this job holds for as long as it remains
799    /// in [`Agent::background_jobs`] — dropped (freeing the slot) when this
800    /// `BackgroundJob` is removed from the map (a terminal reap, or an
801    /// explicit kill), exactly mirroring [`BackgroundSubagent`]'s own
802    /// "guard held for as long as it's tracked, not just while the process
803    /// is alive" posture (§2 module 9 precedent, kept consistent here).
804    _guard: crate::subagents::ConcurrencyGuard,
805}
806
807/// P5-6: the non-blocking status read [`Agent::run_background_status`]/
808/// [`Agent::run_background_list`] share — `job.killed` (set by
809/// [`Agent::run_background_kill`]) always wins over a fresh `try_wait`,
810/// since a kill signal racing the OS reaping the process is otherwise
811/// indistinguishable from "still running" for one poll cycle; reporting
812/// `Killed` unconditionally once requested avoids that race entirely. A
813/// `try_wait` error (would only happen if this job's id were somehow
814/// double-reaped, which the map ownership below already prevents) is
815/// treated as "no news yet" — `Running` — rather than inventing a made-up
816/// exit code.
817fn background_job_status(job: &mut BackgroundJob) -> supercode_runtime::background::JobStatus {
818    if job.killed {
819        return supercode_runtime::background::JobStatus::Killed;
820    }
821    match job.child.try_wait() {
822        Ok(Some(status)) => supercode_runtime::background::JobStatus::Exited(status.code()),
823        Ok(None) | Err(_) => supercode_runtime::background::JobStatus::Running,
824    }
825}
826
827/// Fable-5 review (HIGH, "grandchildren orphaned on kill AND agent-drop"):
828/// the shared real-kill body for both [`Agent::run_background_kill`] and
829/// `impl Drop for Agent` — sends `SIGKILL` to `job`'s ENTIRE process group,
830/// not just the one directly-tracked pid, so a surviving `&` job, pipeline
831/// stage, or double-forking daemon spawned by the job is killed too, then
832/// reaps the group leader so it doesn't linger as a zombie.
833///
834/// Relies on the spawn site (`Agent::run_background_exec`) having put the
835/// job in its OWN new process group via `Command::process_group(0)` — which
836/// makes the leader's pgid equal to its own pid, so `job.pid` doubles as the
837/// group id here.
838#[cfg(unix)]
839fn kill_job_process_group(job: &mut BackgroundJob) {
840    if let Some(pid) = job.pid {
841        // SAFETY: `libc::kill` with a negative pid is `killpg` — it only
842        // ever sends a signal (never dereferences memory), so this is safe
843        // regardless of whether the group is still alive. A `-1`/`ESRCH`
844        // return means the leader (and thus the whole group, since a group
845        // can't outlive its leader) already exited — not an error, just
846        // "already dead", exactly like `Child::start_kill`'s own documented
847        // no-op-on-already-exited contract.
848        unsafe {
849            libc::kill(-(pid as libc::pid_t), libc::SIGKILL);
850        }
851    }
852    // Belt-and-suspenders for the leader itself — `kill_on_drop(true)` set
853    // at spawn time is the same outcome via a different (implicit) path —
854    // then reap it so the SIGKILL we just delivered doesn't leave a zombie
855    // behind.
856    let _ = job.child.start_kill();
857    let _ = job.child.try_wait();
858}
859
860/// Non-unix fallback: no portable process-group primitive is wired up here
861/// (same posture as `crate::tools::build_sandboxed_sh`'s own platform
862/// split) — falls back to the pre-fix per-child kill. A background job that
863/// spawns a surviving grandchild process on a non-Unix target is a
864/// documented residual, not silently claimed fixed by this cfg arm.
865#[cfg(not(unix))]
866fn kill_job_process_group(job: &mut BackgroundJob) {
867    let _ = job.child.start_kill();
868}
869
870/// P5-6 (D1 "monitor/event feed", "output captured incrementally +
871/// BOUNDED"): spawn a fire-and-forget reader task that continuously drains
872/// `reader` (a piped `ChildStdout`/`ChildStderr`) into `output`, bounded at
873/// `cap` bytes. Reading NEVER stops at the cap — only what's RETAINED is
874/// bounded ([`supercode_runtime::background::CapturedOutput::append`]'s own contract)
875/// — because a background job's child process would otherwise block
876/// forever writing to a full, undrained OS pipe once this stopped reading
877/// it, silently hanging real work behind an apparently-"running" job. The
878/// task exits on its own once the pipe reaches EOF (the process closed the
879/// descriptor, whether by exiting or being killed) — no explicit
880/// abort/cleanup call site is needed; a detached `tokio::spawn` this short-
881/// lived is not the kind of orphaned-task risk `impl Drop for Agent`'s own
882/// doc comment is about (that one concerns a whole recursive provider-
883/// calling child AGENT loop, not a bounded byte-copy loop that ends the
884/// instant its source pipe closes).
885fn spawn_output_reader<R>(
886    reader: R,
887    output: std::sync::Arc<supercode_runtime::background::CapturedOutput>,
888    cap: usize,
889) -> tokio::task::JoinHandle<()>
890where
891    R: tokio::io::AsyncRead + Unpin + Send + 'static,
892{
893    tokio::spawn(async move {
894        use tokio::io::AsyncReadExt;
895        let mut reader = reader;
896        let mut buf = [0u8; 8192];
897        loop {
898            match reader.read(&mut buf).await {
899                Ok(0) => break,
900                Ok(n) => {
901                    let chunk = String::from_utf8_lossy(&buf[..n]);
902                    output.append(&chunk, cap);
903                }
904                Err(_) => break,
905            }
906        }
907    })
908}
909
910/// BP-7 (catalog §4a "Named agent definitions as data"): merge
911/// `<cwd>/.claude/agents/*.md` into `config.subagents_definitions`.
912///
913/// Runs for every `Agent` whose `subagents` module is on, whatever preset
914/// it came from — before BP-7 the `.md` loader was reachable only from the
915/// Claude emulate/resume path, so a cc-parity or cx-parity session ignored
916/// definitions sitting right there in the repo.
917///
918/// * A no-op when the module is off (the default), and for every spawned
919///   CHILD (`subagent_depth > 0`), which already inherits its parent's
920///   resolved definitions verbatim.
921/// * A config-table entry WINS over a discovered file of the same name:
922///   `[capabilities.subagents.agents.<name>]` is explicit configuration,
923///   the file is discovery.
924/// * A malformed file is skipped with a warning, never a failed
925///   construction: `Agent::with_provider` has no error channel, and a
926///   broken agent file in some repo must not make the harness unusable
927///   there. (The emulate/resume path keeps its own strict behavior, where
928///   a definition the resumed session may depend on going missing IS worth
929///   failing over.)
930fn merge_project_agent_definitions(config: &mut Config) {
931    if !config.subagents_enabled || config.subagent_depth > 0 {
932        return;
933    }
934    match crate::claude_compat::load_project_agents(&config.cwd) {
935        Ok(agents) => {
936            for agent in agents {
937                config
938                    .subagents_definitions
939                    .entry(agent.definition.name.clone())
940                    .or_insert(agent.definition);
941            }
942        }
943        Err(e) => {
944            tracing::debug!(
945                error = %e,
946                "skipping .claude/agents discovery: a definition file could not be parsed"
947            );
948        }
949    }
950}
951
952/// P5-3: one background-spawned child this agent is tracking, awaiting a
953/// `subagent_status` poll (or agent drop) to reap it.
954struct BackgroundSubagent {
955    /// Resolves to `(child_agent_id, child's final result, the child's own
956    /// post-system-prompt history — for D5 transcript persistence once
957    /// reaped)` — the concurrency-guard slot for this child is held INSIDE
958    /// the spawned future (moved in at spawn time), so it releases the
959    /// instant the child's own run loop finishes, not when the parent gets
960    /// around to polling.
961    handle: tokio::task::JoinHandle<(String, Result<String>, Vec<ChatMessage>)>,
962    /// The task/prompt text the child was spawned with (surfaced by a
963    /// `"pending"` status poll, since the handle alone can't answer "what
964    /// is it doing").
965    task: String,
966    /// The named `agent_type` spawned, if any.
967    agent_type: Option<String>,
968    /// Unix-ms wall-clock time the spawn happened.
969    started_at_ms: i64,
970    /// BP-7 (catalog §4a "Background subagents + resume"): the child's own
971    /// steering inbox, captured before the child moved into its task.
972    ///
973    /// This IS the mailbox. `SteerInbox` was built (P4b) to be writable
974    /// while an active turn holds `&mut Agent` — exactly the property a
975    /// message-to-a-running-child needs — so the mailbox is that existing
976    /// seam reached from outside, not a second delivery channel with its
977    /// own ordering rules. A message lands at the top of the child's next
978    /// loop iteration, per `Config::steering_mode`.
979    mailbox: std::sync::Arc<std::sync::Mutex<SteerInbox>>,
980}
981
982/// P5-3 safety hardening (Fable-5 review, MEDIUM-LOW "orphaned billed
983/// spend"): a dropped parent must not leave a detached background child
984/// running against a REAL provider. Without this, a parent dropped
985/// mid-run (the caller's own process exits the scope, panics, or simply
986/// stops polling) leaves every still-running `BackgroundSubagent::handle`
987/// as an orphaned `tokio::spawn` task: nothing had ever awaited or
988/// aborted it, so it runs to its own (`max_iterations`-bounded)
989/// completion regardless — bounded but real provider spend nobody is
990/// paying attention to.
991///
992/// `.abort()` on a [`tokio::task::JoinHandle`] is safe to call
993/// unconditionally, including on an ALREADY-finished task (a documented
994/// no-op there — see tokio's `JoinHandle::abort` docs) — so this never
995/// needs to distinguish "still running" from "already done"; a background
996/// child that already finished and is merely awaiting a `subagent_status`
997/// reap is untouched in practice (aborting a finished task changes
998/// nothing observable). For a task still mid-flight, tokio cancels it at
999/// its next `.await` point, which drops that future in place — including
1000/// the `_guard: ConcurrencyGuard` moved into it at spawn time (see
1001/// `Self::run_spawn_subagent`'s `tokio::spawn` body) — so the
1002/// concurrency-gauge slot is released exactly the same way a normal
1003/// completion releases it (`ConcurrencyGuard`'s own `Drop`, in
1004/// `crate::subagents`). No separate cleanup call site to forget.
1005///
1006/// Deliberately does NOT touch [`Self::pending_child_approvals]` or
1007/// `Self::subagent_store` — this is purely "stop burning provider
1008/// calls on behalf of a caller who's gone", not a transcript-persistence
1009/// path (a child aborted mid-flight has no finished result to persist;
1010/// see this build's named residual on abort-time transcript loss).
1011impl Drop for Agent {
1012    fn drop(&mut self) {
1013        for (child_id, bg) in self.background_subagents.drain() {
1014            // Named, not silent: a child that was still running gets its
1015            // provider calls cut off here — worth a trace even though
1016            // there's no transcript left to persist (the future is
1017            // dropped mid-flight, before it ever returns a result).
1018            if !bg.handle.is_finished() {
1019                tracing::debug!(
1020                    child_id = %child_id,
1021                    "parent Agent dropped: aborting still-running background subagent \
1022                     to stop further provider spend"
1023                );
1024            }
1025            bg.handle.abort();
1026        }
1027        // P5-6 (§2 module 4 `tools.background`, build brief "on agent drop
1028        // / session end, jobs MUST be killed... real process kill via the
1029        // child handle's kill(), not just tokio task abort"): a REAL OS
1030        // process, not a Rust task — `Child::start_kill` (synchronous, no
1031        // `.await` needed, so callable from this non-async `Drop::drop`)
1032        // sends the actual kill signal; a no-op, per its own docs, on a
1033        // job that already exited. `kill_on_drop(true)` (set at spawn
1034        // time) is a second, independent line of defense for the same
1035        // outcome, but this explicit loop is what makes the guarantee
1036        // provable/traceable rather than relying solely on an implicit
1037        // tokio runtime behavior.
1038        for (job_id, mut job) in self.background_jobs.drain() {
1039            if !job.killed {
1040                tracing::debug!(
1041                    job_id = %job_id,
1042                    command = %job.command,
1043                    "parent Agent dropped: killing still-tracked background job's real \
1044                     OS process (and its whole process group — see \
1045                     `kill_job_process_group`)"
1046                );
1047            }
1048            kill_job_process_group(&mut job);
1049        }
1050        // P5-11 (§2 module 28 `lsp`, build brief "no orphaned language-
1051        // server processes"): a REAL OS process, same rationale as the
1052        // background-job loop just above — `kill_all_sync` is
1053        // synchronous (`Child::start_kill`, no `.await` needed, so
1054        // callable from this non-async `Drop::drop`), SIGKILLs each
1055        // server's WHOLE process group (unix — same `kill_job_process_group`
1056        // mechanism as the background-job loop above, so worker
1057        // grandchildren like rust-analyzer's proc-macro server or
1058        // typescript-language-server's `tsserver` are killed too, not just
1059        // the one directly-tracked pid), and is provable/traceable rather
1060        // than relying solely on `kill_on_drop(true)`'s implicit tokio
1061        // runtime behavior (which remains a second, independent line of
1062        // defense on every spawned `LspClient`).
1063        if let Some(lsp) = &self.lsp_manager {
1064            lsp.kill_all_sync();
1065        }
1066    }
1067}
1068
1069/// P4 (§1.8 credential-helper indirection, D6 row): run an `api_key_cmd`
1070/// through the shell and return its trimmed stdout. Runs via `sh -c` (POSIX
1071/// shell, matching pi's `!command` precedent) so the configured string can
1072/// use pipes/substitution, e.g. `pass show api-key`. Never panics or
1073/// propagates an error: a spawn failure or non-zero exit is reported via
1074/// `tracing::warn!` and returns an empty `String`, which
1075/// `Agent::new`'s resolution chain treats exactly like an unset helper —
1076/// falling through to `Config::api_key_env`.
1077fn run_api_key_cmd(cmd: &str) -> String {
1078    match std::process::Command::new("sh").arg("-c").arg(cmd).output() {
1079        Ok(out) if out.status.success() => String::from_utf8_lossy(&out.stdout).trim().to_string(),
1080        Ok(out) => {
1081            tracing::warn!(
1082                "api_key_cmd exited with status {:?}; falling back to api_key_env",
1083                out.status.code()
1084            );
1085            String::new()
1086        }
1087        Err(e) => {
1088            tracing::warn!("api_key_cmd failed to run ({e}); falling back to api_key_env");
1089            String::new()
1090        }
1091    }
1092}
1093
1094/// BP-9 (§3.1 `core.api_key_command`, D6 row "Credential helpers /
1095/// keyring"): run an ARGV credential helper and return its trimmed stdout.
1096/// No shell is involved — `argv[0]` is exec'd with the rest as arguments —
1097/// so a helper path with spaces, or an argument containing `$`/`;`, means
1098/// what it says. Same never-panics, fall-through-on-failure contract as
1099/// [`run_api_key_cmd`]: an empty result is treated as "no helper".
1100pub(crate) fn run_api_key_command(argv: &[String]) -> String {
1101    let Some((program, args)) = argv.split_first() else {
1102        return String::new();
1103    };
1104    match std::process::Command::new(program).args(args).output() {
1105        Ok(out) if out.status.success() => String::from_utf8_lossy(&out.stdout).trim().to_string(),
1106        Ok(out) => {
1107            tracing::warn!(
1108                "api_key_command exited with status {:?}; trying the next credential source",
1109                out.status.code()
1110            );
1111            String::new()
1112        }
1113        Err(e) => {
1114            tracing::warn!(
1115                "api_key_command failed to run ({e}); trying the next credential source"
1116            );
1117            String::new()
1118        }
1119    }
1120}
1121
1122/// P4c (§1.2/§3.1 `core.shell_env_snapshot`, SPLIT CC+CX row, catalog:338):
1123/// capture the user's interactive login-shell environment ONCE, best-effort.
1124/// Runs `$SHELL -lc env` (falling back to `sh -lc env` when `$SHELL` is
1125/// unset) — a LOGIN shell (`-l`) sources the user's rc files, which is
1126/// exactly the sourcing `bash` calls should no longer need to repeat once
1127/// this snapshot is in hand. Never panics: any failure (spawn error,
1128/// non-zero exit, unparseable output) returns an empty map, which
1129/// `ToolContext::shell_env`'s "no-op when `None`/empty" contract already
1130/// treats as harmless.
1131fn capture_shell_env() -> std::collections::HashMap<String, String> {
1132    let shell = std::env::var("SHELL").unwrap_or_else(|_| "sh".to_string());
1133    let out = match std::process::Command::new(&shell)
1134        .arg("-lc")
1135        .arg("env")
1136        .output()
1137    {
1138        Ok(o) if o.status.success() => o.stdout,
1139        Ok(o) => {
1140            tracing::warn!(
1141                "shell_env_snapshot: `{shell} -lc env` exited with status {:?}; snapshot is empty",
1142                o.status.code()
1143            );
1144            return std::collections::HashMap::new();
1145        }
1146        Err(e) => {
1147            tracing::warn!(
1148                "shell_env_snapshot: failed to run `{shell} -lc env` ({e}); snapshot is empty"
1149            );
1150            return std::collections::HashMap::new();
1151        }
1152    };
1153    let text = String::from_utf8_lossy(&out);
1154    let mut map = std::collections::HashMap::new();
1155    for line in text.lines() {
1156        if let Some((k, v)) = line.split_once('=') {
1157            if !k.is_empty() {
1158                map.insert(k.to_string(), v.to_string());
1159            }
1160        }
1161    }
1162    map
1163}
1164
1165/// Build the [`ToolContext`] an [`Agent`] hands to every tool call, folding
1166/// in every P4c per-tool config knob (§1.2) alongside the pre-existing
1167/// `cwd`/`sandbox` — shared by [`Agent::with_parts`]/[`Agent::with_provider_arc`]
1168/// so the two construction paths can never drift apart on which config
1169/// fields reach the context. Also builds (P5-9) the
1170/// [`crate::checkpoint::CheckpointObserver`], if `config.checkpoint_enabled`
1171/// — installed on the returned context's `write_observer` AND returned
1172/// separately (as the concrete type) so `Agent::run_loop` can call
1173/// [`crate::checkpoint::CheckpointObserver::begin_turn`] once per turn.
1174/// `None`/no-op end to end when the module is off — see
1175/// [`crate::checkpoint::observer_for_config`]'s own doc comment for the
1176/// default-off byte-identity guarantee.
1177///
1178/// P5-11 (§2 modules 28/29, D-5 "shared write-path interception seam"):
1179/// `crate::formatters::observer_for_config`/`crate::lsp::manager_for_config`
1180/// are folded into the SAME `write_observer` slot via
1181/// [`crate::tools::WriteObserverChain`], in the design's required order —
1182/// `checkpoint -> formatters -> lsp` (checkpoint's pre-image capture must
1183/// see the file before ANY mutation; lsp's diagnostics must see the file
1184/// AFTER formatting, never before). When 0 or 1 of the three modules is
1185/// active, this degrades to exactly what P5-9 shipped (`None`, or the
1186/// single concrete observer installed directly) — no chain wrapper is
1187/// introduced unless there is actually more than one observer to order,
1188/// keeping every single-module (or all-off) configuration byte-identical
1189/// to before this function grew multi-observer support. The `lsp` manager
1190/// is ALSO returned separately (like `checkpoint_observer`), so
1191/// `Agent`'s `Drop` impl can reach `crate::lsp::LspManager::kill_all_sync`
1192/// regardless of how the chain is shaped.
1193/// BP-2: `pub(crate)` so a parity test can build the SAME `ToolContext` an
1194/// `Agent` would from a resolved preset's `Config` and drive a registry
1195/// tool through it — a tool's behavior under a preset is exactly the
1196/// composition of the two, and a test that hand-assembled a context would
1197/// be proving an unwired function.
1198pub(crate) fn build_tool_context(
1199    config: &Config,
1200) -> (
1201    ToolContext,
1202    Option<std::sync::Arc<crate::checkpoint::CheckpointObserver>>,
1203    Option<std::sync::Arc<crate::lsp::LspManager>>,
1204) {
1205    let shell_env = if config.shell_env_snapshot {
1206        Some(std::sync::Arc::new(capture_shell_env()))
1207    } else {
1208        None
1209    };
1210    let checkpoint_observer = crate::checkpoint::observer_for_config(config);
1211    let format_observer = crate::formatters::observer_for_config(config);
1212    let lsp_manager = crate::lsp::manager_for_config(config);
1213    let lsp_observer = lsp_manager
1214        .clone()
1215        .map(|m| std::sync::Arc::new(crate::lsp::LspDiagnosticsObserver::new(m)));
1216    let mut observers: Vec<std::sync::Arc<dyn crate::tools::WriteObserver>> = Vec::new();
1217    if let Some(cp) = &checkpoint_observer {
1218        observers.push(cp.clone() as std::sync::Arc<dyn crate::tools::WriteObserver>);
1219    }
1220    if let Some(f) = &format_observer {
1221        observers.push(f.clone() as std::sync::Arc<dyn crate::tools::WriteObserver>);
1222    }
1223    if let Some(l) = &lsp_observer {
1224        observers.push(l.clone() as std::sync::Arc<dyn crate::tools::WriteObserver>);
1225    }
1226    let write_observer: Option<std::sync::Arc<dyn crate::tools::WriteObserver>> =
1227        match observers.len() {
1228            0 => None,
1229            1 => observers.into_iter().next(),
1230            _ => Some(std::sync::Arc::new(crate::tools::WriteObserverChain::new(
1231                observers,
1232            ))),
1233        };
1234    let ctx = ToolContext {
1235        cwd: config.cwd.clone(),
1236        // BP-10 (catalog row "Additional working directories"): the
1237        // `--add-dir`/`core.additional_dirs` roots reach the TOOLS now,
1238        // not just the project-context walk — `ToolContext::check_write`,
1239        // the OS backstop's writable set, and the permissions engine's
1240        // path rules all read them. Empty (the default) is byte-identical
1241        // to confining everything to `cwd`.
1242        extra_roots: config.additional_dirs.clone(),
1243        sandbox: config.sandbox,
1244        multimodal_read: config.read_file_multimodal,
1245        // BP-2 (§1.2 `core.tools.read_file.line_numbers`, catalog:26).
1246        read_line_numbers: config.read_file_line_numbers,
1247        require_read_before_edit: config.edit_file_require_read_before_edit,
1248        // BP-2: path → content hash at read time (`ToolContext::read_state`).
1249        read_paths: std::sync::Arc::new(std::sync::Mutex::new(std::collections::HashMap::new())),
1250        notebook_aware: config.edit_file_notebook_aware,
1251        shell_env,
1252        nested_instructions: config.nested_instructions,
1253        injected_instruction_dirs: std::sync::Arc::new(std::sync::Mutex::new(HashSet::new())),
1254        // BP-5: filled in by `Agent::with_parts` (the one construction path
1255        // that assembles a prompt, and therefore the one that knows which
1256        // rules were held back); empty everywhere else.
1257        path_rules: std::sync::Arc::new(Vec::new()),
1258        injected_rule_files: std::sync::Arc::new(std::sync::Mutex::new(HashSet::new())),
1259        // P5-1 (§2 module 12 carry-forward): now sourced from real config
1260        // (`capabilities.permissions.sandbox.network.*`, wired by
1261        // `configfile::materialize_config`) instead of always `None`. `None`
1262        // (the default, unchanged when the config never sets it) is still
1263        // byte-identical to today's behavior.
1264        network_policy: config.network_policy.clone(),
1265        // BP-10 (catalog row "Allow/ask/deny rule language", the DOMAIN
1266        // subject): the config's own rule arrays reach the network surface
1267        // too, so a `domain(...)` rule is evaluated by the SAME engine that
1268        // evaluates `bash(...)`/`write(...)` at the dispatch gate — not by
1269        // a second matcher over a second list. `None` when the permissions
1270        // module is off, which is byte-identical to before.
1271        permission_rules: config.permissions_enabled.then(|| {
1272            std::sync::Arc::new(crate::permissions::RuleSet {
1273                deny: config.tool_deny_patterns.clone(),
1274                ask: config.permissions_ask_patterns.clone(),
1275                allow: config.tool_allow_patterns.clone(),
1276            })
1277        }),
1278        // P4e (S3.1 `core.tools.bash.timeout_secs`, S14): folds the `bash`
1279        // `ToolOverride`'s `timeout_secs`, if set, into the context every
1280        // `BashTool::execute` call receives -- `None` (no override
1281        // configured) is byte-identical to today's behavior.
1282        bash_timeout_secs: config
1283            .tool_overrides
1284            .get("bash")
1285            .and_then(|o| o.timeout_secs),
1286        write_observer,
1287        // P5-10 (§2 module 12): sourced from real config
1288        // (`capabilities.permissions.sandbox.{enabled,escalation,env_policy}`,
1289        // wired by `configfile::materialize_config`). `sandbox_approval_handler`
1290        // starts `None` here (no handler is installed yet at `Agent`
1291        // construction time) and is kept in sync by
1292        // `Agent::set_permissions_approval_handler` — see that method's doc
1293        // comment.
1294        sandbox_os_enabled: config.sandbox_os_enabled,
1295        sandbox_escalation: config.sandbox_escalation,
1296        sandbox_env_policy: config.sandbox_env_policy,
1297        sandbox_approval_handler: None,
1298        // BP-3: both handler seams start `None` (nothing is installed at
1299        // construction time) and are filled by
1300        // `Agent::set_permissions_approval_handler` /
1301        // `Agent::set_user_question_handler`, exactly like
1302        // `sandbox_approval_handler` above. The two shared states are
1303        // always present but inert: plan mode starts off (contributing no
1304        // rules), and the budget starts unpublished.
1305        question_handler: None,
1306        approval_handler: None,
1307        plan_mode: std::sync::Arc::new(crate::tools::PlanModeState::new()),
1308        context_budget: std::sync::Arc::new(crate::tools::ContextBudget::new()),
1309        // BP-8 (catalog:156): the shared plan the agent journals and
1310        // persists. Always present, empty and inert until `update_plan`
1311        // writes one.
1312        plan: std::sync::Arc::new(std::sync::Mutex::new(Vec::new())),
1313    };
1314    (ctx, checkpoint_observer, lsp_manager)
1315}
1316
1317/// P4b (§1.4/§3.1, catalog §4a "Global/user-level instruction file tier"):
1318/// where the user/global instruction tier lives — `$SUPERCODE_HOME`, else
1319/// `$XDG_CONFIG_HOME/supercode`, else `~/.config/supercode`. Deliberately
1320/// duplicates `crates/cli/src/userconfig.rs::config_home`'s exact precedence
1321/// rather than depending on the `cli` crate from `core` (wrong dependency
1322/// direction — `cli` depends on `core`, never the reverse). `pub(crate)`:
1323/// also the DEFAULT shadow-store root `crate::checkpoint::observer_for_config`
1324/// (P5-9) derives from when `Config::checkpoint_dir` is unset — one
1325/// `$SUPERCODE_HOME` resolver, not a second hand-rolled one.
1326pub(crate) fn global_instructions_dir() -> std::path::PathBuf {
1327    if let Ok(h) = std::env::var("SUPERCODE_HOME") {
1328        if !h.is_empty() {
1329            return std::path::PathBuf::from(h);
1330        }
1331    }
1332    if let Ok(xdg) = std::env::var("XDG_CONFIG_HOME") {
1333        if !xdg.is_empty() {
1334            return std::path::PathBuf::from(xdg).join("supercode");
1335        }
1336    }
1337    let home = std::env::var("HOME").unwrap_or_else(|_| ".".into());
1338    std::path::PathBuf::from(home)
1339        .join(".config")
1340        .join("supercode")
1341}
1342
1343/// P4b (§1.4, catalog §4a "Instruction imports"): is `rel` (an `@`-import
1344/// target found inside a PROJECT-sourced instruction file) LEXICALLY safe to
1345/// resolve? Mirrors `configfile::is_safe_project_dir`'s posture (LOW-1
1346/// precedent): rejects absolute paths, `~`-relative paths, and any `..`
1347/// component — an untrusted repo's own CLAUDE.md/AGENTS.md must not be able
1348/// to `@import` its way to an arbitrary file on disk (e.g. `@/etc/passwd`,
1349/// `@../../.ssh/id_rsa`). Global-tier files (the user's own machine, same
1350/// trust level as the user's shell) are NOT run through this check.
1351///
1352/// This is a cheap PRE-FILTER only — it operates on the literal token text
1353/// and cannot see through a symlink committed in the repo whose *target*
1354/// escapes the root while the *link itself* has a clean, traversal-free
1355/// relative name (e.g. `@link.md` where `link.md -> /etc/passwd`). See
1356/// [`import_target_is_contained`] for the canonicalizing check that closes
1357/// that gap; the project-scoped resolution path runs both.
1358fn import_path_is_safe(rel: &str) -> bool {
1359    if rel.is_empty() || rel.contains('\0') {
1360        return false;
1361    }
1362    let path = std::path::Path::new(rel);
1363    if path.is_absolute() || rel.starts_with('~') {
1364        return false;
1365    }
1366    !path
1367        .components()
1368        .any(|c| matches!(c, std::path::Component::ParentDir))
1369}
1370
1371/// P4b security fix (Fable-5 review, MEDIUM: symlink bypass of the
1372/// project-scoped `@`-import boundary): does `candidate` — after resolving
1373/// symlinks — stay inside `root` — also after resolving symlinks? This is
1374/// what actually enforces [`import_path_is_safe`]'s doc-comment guarantee
1375/// ("must not be able to `@import` its way to an arbitrary file on disk"):
1376/// the lexical check alone rejects `@/etc/passwd` and `@../../secret`, but a
1377/// repo can commit a symlink (e.g. `link.md -> /etc/passwd`) whose own
1378/// relative name is perfectly clean, defeating a purely lexical check.
1379///
1380/// Both sides are canonicalized before the comparison — not just
1381/// `candidate` — because `root` itself can legitimately be a symlink (a
1382/// tempdir under macOS's `/tmp` -> `/private/tmp`, or any other symlinked
1383/// project checkout); comparing a canonicalized candidate against a
1384/// non-canonicalized root would falsely reject genuinely-in-root files.
1385///
1386/// Fails CLOSED: a `canonicalize()` failure (broken symlink, a target that
1387/// doesn't exist, a permission error) returns `false` — never inlined,
1388/// mirroring [`expand_instruction_imports`]'s existing "unreadable file ⇒
1389/// left as literal text" posture rather than panicking or defaulting open.
1390pub(crate) fn import_target_is_contained(
1391    candidate: &std::path::Path,
1392    root: &std::path::Path,
1393) -> bool {
1394    let (Ok(real_root), Ok(real_candidate)) = (
1395        std::fs::canonicalize(root),
1396        std::fs::canonicalize(candidate),
1397    ) else {
1398        return false;
1399    };
1400    real_candidate.starts_with(&real_root)
1401}
1402
1403/// P4b (§1.4/§3.1 `core.instruction_imports`, catalog:85): inline `@path`
1404/// import tokens found in `text` with the referenced file's own (trimmed)
1405/// content, resolved relative to `dir` (the directory the CONTAINING file
1406/// lives in — so a chain of imports each resolves relative to its own
1407/// location, not the original file's). `depth` bounds recursion (CC's own
1408/// default of 4, cited in the cc-parity preset) so a cyclical or
1409/// deeply-nested import chain can't blow the stack or loop forever.
1410/// `project_scoped` gates [`import_path_is_safe`] AND
1411/// [`import_target_is_contained`] — see their doc comments; `root` is the
1412/// containment boundary those checks canonicalize against (the SAME root
1413/// for every level of a nested import chain, even though `dir` itself walks
1414/// deeper with each level — an import three levels deep must still resolve
1415/// under the original project root, not merely under its own immediate
1416/// parent). Ignored when `!project_scoped` (the global/user tier, trusted,
1417/// unrestricted — see [`append_instruction_file`]'s doc comment).
1418/// Any token that isn't `@`-prefixed, doesn't resolve to a readable file, or
1419/// (project-scoped) fails the safety/containment check is left as literal
1420/// text — an import is best-effort, never a hard error that could make
1421/// instruction loading fail outright.
1422fn expand_instruction_imports(
1423    text: &str,
1424    dir: &std::path::Path,
1425    root: &std::path::Path,
1426    project_scoped: bool,
1427    depth: u8,
1428) -> String {
1429    if depth >= 4 {
1430        return text.to_string();
1431    }
1432    let mut out = String::with_capacity(text.len());
1433    for token in split_preserving_whitespace(text) {
1434        if let Some(rel) = token.strip_prefix('@') {
1435            if !rel.is_empty()
1436                && !rel.contains(char::is_whitespace)
1437                && (!project_scoped || import_path_is_safe(rel))
1438            {
1439                let candidate = dir.join(rel);
1440                if !project_scoped || import_target_is_contained(&candidate, root) {
1441                    if let Ok(imported) = std::fs::read_to_string(&candidate) {
1442                        let imported = imported.trim();
1443                        if !imported.is_empty() {
1444                            let imported_dir = candidate.parent().unwrap_or(dir);
1445                            out.push_str(&expand_instruction_imports(
1446                                imported,
1447                                imported_dir,
1448                                root,
1449                                project_scoped,
1450                                depth + 1,
1451                            ));
1452                            continue;
1453                        }
1454                    }
1455                }
1456            }
1457        }
1458        out.push_str(token);
1459    }
1460    out
1461}
1462
1463/// Split `text` into tokens that, concatenated, reproduce it exactly —
1464/// alternating runs of non-whitespace and whitespace. Used by
1465/// [`expand_instruction_imports`] so `@import` tokens can be located and
1466/// replaced without disturbing surrounding formatting/whitespace.
1467fn split_preserving_whitespace(text: &str) -> Vec<&str> {
1468    let mut out = Vec::new();
1469    let mut start = 0;
1470    let mut in_ws = None;
1471    for (i, c) in text.char_indices() {
1472        let ws = c.is_whitespace();
1473        match in_ws {
1474            None => in_ws = Some(ws),
1475            Some(prev) if prev != ws => {
1476                out.push(&text[start..i]);
1477                start = i;
1478                in_ws = Some(ws);
1479            }
1480            _ => {}
1481        }
1482    }
1483    if start < text.len() {
1484        out.push(&text[start..]);
1485    }
1486    out
1487}
1488
1489/// BP-4 (catalog:87 "Instruction-file hygiene controls", cc§2
1490/// `claudeMdExcludes`): whether `path` is excluded from instruction loading
1491/// by [`Config::project_doc_excludes`]. A pattern matches when it
1492/// [`crate::config::glob_match`]es the file's NAME (`CLAUDE.md`), its full
1493/// path, or its path relative to `root` — the three spellings cc's own
1494/// "glob/absolute-path list" accepts. Empty (the default) excludes nothing.
1495fn instruction_file_excluded(
1496    config: &Config,
1497    path: &std::path::Path,
1498    root: &std::path::Path,
1499) -> bool {
1500    if config.project_doc_excludes.is_empty() {
1501        return false;
1502    }
1503    let full = path.to_string_lossy().to_string();
1504    let name = path
1505        .file_name()
1506        .map(|n| n.to_string_lossy().to_string())
1507        .unwrap_or_default();
1508    let rel = path
1509        .strip_prefix(root)
1510        .ok()
1511        .map(|p| p.to_string_lossy().to_string());
1512    config.project_doc_excludes.iter().any(|pat| {
1513        crate::config::glob_match(pat, &full)
1514            || crate::config::glob_match(pat, &name)
1515            || rel
1516                .as_deref()
1517                .is_some_and(|r| crate::config::glob_match(pat, r))
1518    })
1519}
1520
1521/// BP-4 (catalog:87, cc§2 "HTML comment stripping"): drop block-level
1522/// `<!-- … -->` spans from an instruction file's text so maintainer notes
1523/// cost no tokens, exactly as cc does before injection. Unterminated
1524/// openers drop the remainder (the same reading a markdown renderer takes).
1525/// Off by default ([`Config::project_doc_strip_comments`]) — cx does NOT
1526/// strip, so this is a per-preset hygiene lever, not a universal one.
1527fn strip_html_comments(text: &str) -> String {
1528    let mut out = String::with_capacity(text.len());
1529    let mut rest = text;
1530    while let Some(open) = rest.find("<!--") {
1531        out.push_str(&rest[..open]);
1532        match rest[open..].find("-->") {
1533            Some(close) => rest = &rest[open + close + 3..],
1534            None => return out,
1535        }
1536    }
1537    out.push_str(rest);
1538    out
1539}
1540
1541/// P4b (§1.4): append one instruction file's (trimmed, import-expanded)
1542/// content to `blob` as a labeled section, exactly like the pre-P4b inline
1543/// loop did — a no-op when `path` doesn't exist or is empty (the common
1544/// case). `project_scoped` distinguishes the project tier (imports bounded
1545/// to `root`, canonicalized-and-contained — see
1546/// [`import_target_is_contained`]) from the global tier (imports
1547/// unrestricted, same trust level as the user's own machine — `root` is
1548/// unused in that case). `root` is normally `path`'s own parent (the tier
1549/// root `path` was discovered under, e.g. an ancestor of `cwd` or an
1550/// `additional_dirs` entry) — see [`assemble_project_instructions`]'s call
1551/// sites.
1552///
1553/// BP-4 adds the hygiene controls (catalog:87): the exclude list
1554/// ([`instruction_file_excluded`]), HTML-comment stripping
1555/// ([`strip_html_comments`]) and the [`InstructionBudget`] —
1556/// [`Config::project_doc_max_bytes`] spent INCREMENTALLY as files are
1557/// concatenated root→cwd, which is how cx's own cap works on its root-down
1558/// concat, rather than one chop at the end (that chop would silently eat
1559/// the trailing per-file notices it had just written).
1560fn append_instruction_file(
1561    blob: &mut String,
1562    config: &Config,
1563    path: &std::path::Path,
1564    root: &std::path::Path,
1565    label: &str,
1566    project_scoped: bool,
1567    budget: &mut InstructionBudget,
1568) {
1569    if budget.exhausted() || instruction_file_excluded(config, path, root) {
1570        return;
1571    }
1572    let Ok(text) = std::fs::read_to_string(path) else {
1573        return;
1574    };
1575    let stripped;
1576    let text = if config.project_doc_strip_comments {
1577        stripped = strip_html_comments(&text);
1578        stripped.trim()
1579    } else {
1580        text.trim()
1581    };
1582    if text.is_empty() {
1583        return;
1584    }
1585    let dir = path.parent().unwrap_or(std::path::Path::new("."));
1586    let mut content = if config.instruction_imports {
1587        expand_instruction_imports(text, dir, root, project_scoped, 0)
1588    } else {
1589        text.to_string()
1590    };
1591    if !budget.take(&mut content) {
1592        return;
1593    }
1594    blob.push_str(&format!("\n\n# {label}\n{content}"));
1595}
1596
1597/// Truncate `s` to at most `max` BYTES, backing off to the nearest char
1598/// boundary — shared by the per-file and aggregate instruction caps.
1599fn truncate_at_char_boundary(s: &mut String, max: usize) {
1600    let mut end = max;
1601    while end > 0 && !s.is_char_boundary(end) {
1602        end -= 1;
1603    }
1604    s.truncate(end);
1605}
1606
1607/// BP-4 (catalog:87 "Instruction-file hygiene controls", cx§2
1608/// `project_doc_max_bytes`): the instruction-content byte budget, spent as
1609/// files are concatenated root→cwd.
1610///
1611/// `None` (the default, and cc-parity's explicit `= 0`) is uncapped, so
1612/// [`Self::take`] is a no-op and assembly is byte-identical to a config
1613/// that never heard of the cap. With a cap set, each file is truncated to
1614/// whatever budget REMAINS (per-file notice), and once the budget is gone
1615/// the remaining files are skipped entirely (aggregate notice, emitted once
1616/// by [`Self::aggregate_notice`]) — the total instruction CONTENT can
1617/// therefore never exceed the cap, and the notices survive because nothing
1618/// chops the assembled blob afterwards.
1619struct InstructionBudget {
1620    remaining: Option<usize>,
1621    hit: bool,
1622}
1623
1624impl InstructionBudget {
1625    fn new(config: &Config) -> Self {
1626        InstructionBudget {
1627            remaining: config.project_doc_max_bytes,
1628            hit: false,
1629        }
1630    }
1631
1632    /// True once the cap has consumed the whole budget — later files are
1633    /// skipped rather than partially appended.
1634    fn exhausted(&self) -> bool {
1635        self.remaining == Some(0)
1636    }
1637
1638    /// Charge `content` against the budget, truncating it (and appending a
1639    /// per-file notice) when it doesn't fit. Returns whether anything is
1640    /// left to append.
1641    fn take(&mut self, content: &mut String) -> bool {
1642        let Some(remaining) = self.remaining else {
1643            return true;
1644        };
1645        if content.len() <= remaining {
1646            self.remaining = Some(remaining - content.len());
1647            return true;
1648        }
1649        self.hit = true;
1650        self.remaining = Some(0);
1651        if remaining == 0 {
1652            return false;
1653        }
1654        truncate_at_char_boundary(content, remaining);
1655        content.push_str("\n[supercode: file truncated at core.project_doc_max_bytes]");
1656        true
1657    }
1658
1659    /// The one aggregate notice, appended after assembly when the cap bound
1660    /// anywhere — the statement that the assembled block is not the whole
1661    /// instruction set.
1662    fn aggregate_notice(&self) -> &'static str {
1663        if self.hit {
1664            "\n\n[supercode: instruction content truncated at core.project_doc_max_bytes]"
1665        } else {
1666            ""
1667        }
1668    }
1669}
1670
1671/// BP-4 (catalog:81 "Project instruction files w/ directory walk"; cc§2
1672/// "Directory-walk loading", cx§2 "walk project root (git root) down to
1673/// cwd"): the ancestor chain instruction files are discovered on, ordered
1674/// OUTERMOST FIRST so the nearest directory wins precedence by appearing
1675/// last in the concatenated blob (the root→cwd ordering both inventories
1676/// document).
1677///
1678/// The walk starts at [`Config::cwd`] and climbs until it has included the
1679/// project root [`crate::config::project_root_for`] identifies (`.git` by
1680/// default — cx's `project_root_markers`, §3.1), or until the filesystem
1681/// root, whichever comes first. [`MAX_INSTRUCTION_WALK_DEPTH`] bounds it
1682/// unconditionally, so a marker-less path deep under `/` can never turn
1683/// prompt assembly into an unbounded stat storm.
1684pub(crate) fn instruction_walk_roots(config: &Config) -> Vec<std::path::PathBuf> {
1685    // BP-9's shared answer to "where does the project stop?" — the same
1686    // walk the `env_context` git probe and the CLI's `.supercode.toml`
1687    // discovery use, so one `project_root_markers` value cannot mean three
1688    // different things. `None` (no marker anywhere, or an empty list) means
1689    // no root was found, and the climb below then stops at the filesystem
1690    // root under `MAX_INSTRUCTION_WALK_DEPTH`.
1691    let root = crate::config::project_root_for(&config.cwd, &config.project_root_markers);
1692    let mut chain: Vec<std::path::PathBuf> = Vec::new();
1693    let mut dir = config.cwd.clone();
1694    loop {
1695        let at_root = root.as_deref() == Some(dir.as_path());
1696        chain.push(dir.clone());
1697        if at_root || chain.len() >= MAX_INSTRUCTION_WALK_DEPTH {
1698            break;
1699        }
1700        match dir.parent() {
1701            Some(parent) if parent != dir => dir = parent.to_path_buf(),
1702            _ => break,
1703        }
1704    }
1705    chain.reverse();
1706    chain
1707}
1708
1709/// Hard bound on [`instruction_walk_roots`]'s ancestor climb.
1710const MAX_INSTRUCTION_WALK_DEPTH: usize = 64;
1711
1712/// P4b (§1.4, obligation 4 assembly site): the full instruction-file blob —
1713/// global/user tier (catalog §4a "Global/user-level instruction file tier")
1714/// FIRST, then the project tier — capped by
1715/// [`Config::project_doc_max_bytes`] if set (catalog §4a "hygiene caps
1716/// (`project_doc_max_bytes` analog)").
1717///
1718/// BP-4 (catalog:81): the project tier is no longer `cwd` alone. It is the
1719/// ANCESTOR WALK [`instruction_walk_roots`] returns (cwd's chain up to the
1720/// git root, outermost first) followed by `additional_dirs` — root-first
1721/// ordering throughout, so the nearest directory wins by appearing later,
1722/// which is exactly how both cc§2 ("concatenated root→cwd, closest read
1723/// last") and cx§2 ("nearer-to-cwd wins by appearing later") describe their
1724/// own walks. `cwd` is the last element of the walk chain, so a config
1725/// whose cwd IS the project root assembles byte-identically to the pre-BP-4
1726/// loop.
1727/// BP-5 (catalog D2 "Per-model-family base-prompt selection", cx§2
1728/// "Per-model base instructions": "the system prompt is selected per model
1729/// family from bundled markdown … the active `base_instructions` are
1730/// persisted verbatim into the rollout `session_meta`"): the base system
1731/// prompt for the model this config runs.
1732///
1733/// `[capabilities.model_catalog] base_prompts` maps a model-id glob to that
1734/// family's prompt; the most specific match wins
1735/// ([`crate::model_catalog::base_prompt_for`]). No table and no match both
1736/// give [`Config::system_prompt`] verbatim, so this is a no-op for every
1737/// config that does not set the table.
1738fn base_prompt_for_config(config: &Config) -> String {
1739    crate::model_catalog::base_prompt_for(&config.model_family_prompts, &config.model)
1740        .map(str::to_string)
1741        .unwrap_or_else(|| config.system_prompt.clone())
1742}
1743
1744fn assemble_project_instructions(config: &Config) -> String {
1745    let mut blob = String::new();
1746    let mut budget = InstructionBudget::new(config);
1747    let global_dir = global_instructions_dir();
1748    for name in ["CLAUDE.md", "AGENTS.md"] {
1749        append_instruction_file(
1750            &mut blob,
1751            config,
1752            &global_dir.join(name),
1753            // Global tier is trusted/unrestricted (project_scoped=false
1754            // below) — `root` is never consulted, but pass `global_dir`
1755            // rather than a bogus value for clarity.
1756            &global_dir,
1757            name,
1758            false,
1759            &mut budget,
1760        );
1761    }
1762    // BP-10 (catalog row "Project/workspace trust gate", cc§4/cx§4:
1763    // "Prompt before loading project-local config/code"): the PROJECT tier
1764    // is trust-gated. The global/user tier above is not — it is the user's
1765    // own machine, the same trust level as their shell, exactly as
1766    // `append_instruction_file`'s `project_scoped = false` argument
1767    // already says.
1768    //
1769    // `crate::trust::is_trusted` asks the `Config::trust_handler` door
1770    // once per project and records the answer; with no door installed
1771    // `TrustSurface::Instructions` resolves to LOADED, which is both the
1772    // pre-BP-10 behavior and what a headless run of either upstream
1773    // harness does — see `crate::trust`'s doc comment for why the
1774    // undecided answer differs between text and code.
1775    if !crate::trust::is_trusted(config, crate::trust::TrustSurface::Instructions) {
1776        blob.push_str(
1777            "\n[project instruction files were not loaded: this workspace is not trusted              (capabilities.trust)]\n",
1778        );
1779        blob.push_str(budget.aggregate_notice());
1780        return blob;
1781    }
1782    let walk = instruction_walk_roots(config);
1783    for root in walk.iter().chain(config.additional_dirs.iter()) {
1784        for name in ["CLAUDE.md", "AGENTS.md"] {
1785            append_instruction_file(
1786                &mut blob,
1787                config,
1788                &root.join(name),
1789                // Project tier: `@`-imports from THIS file must stay under
1790                // THIS root (canonicalized) — see
1791                // `import_target_is_contained`.
1792                root,
1793                name,
1794                true,
1795                &mut budget,
1796            );
1797        }
1798        // A repository-native agent package is an additional project
1799        // instruction tier. It is subject to the same `project_context`
1800        // switch, import containment, and aggregate byte cap as root
1801        // AGENTS.md/CLAUDE.md; loading it never executes package code.
1802        for path in crate::agent_package::workspace_package_instruction_files(root) {
1803            append_instruction_file(
1804                &mut blob,
1805                config,
1806                &path,
1807                root,
1808                "Volter Harness agent package instructions",
1809                true,
1810                &mut budget,
1811            );
1812        }
1813    }
1814    blob.push_str(budget.aggregate_notice());
1815    blob
1816}
1817
1818/// P4b (§1.4/§3.1 `core.env_context`, catalog §4a "Environment context block
1819/// injection"): cwd, platform, date, and a best-effort git branch/dirty
1820/// status (silently absent when `cwd` isn't a git repo or `git` isn't on
1821/// `PATH` — never blocks agent construction).
1822///
1823/// BP-4 (catalog:90): plus the APPROVAL/SANDBOX POLICY line the row's own
1824/// semantics name ("cwd/git/platform/date/**policy**") and cx's
1825/// `<environment_context>` supplies — the model is told which approval mode
1826/// and which filesystem confinement it is operating under, which is what
1827/// makes "ask before you do X" instructions legible to it. The block is
1828/// re-derivable at any moment from `config` alone, which is what lets
1829/// [`Agent::refresh_env_context`] re-emit it mid-session on change.
1830/// BP-6 (catalog D2 "Skills (progressive-disclosure packages)", §1.4
1831/// obligation 4): the `# Skills` prompt section — the discovered SKILL.md
1832/// packages' names and descriptions, plus the `[core.prompts]` template
1833/// names, and NOTHING else. A skill's body is deliberately absent: it costs
1834/// its tokens only when something actually invokes it (`docs:skills`
1835/// "body loads only when used"; cx§7; pi§2 "progressive disclosure").
1836///
1837/// A skill whose frontmatter hides it from the model (`enabled: false`,
1838/// `disable-model-invocation: true`) is left OUT of the index while staying
1839/// user-invocable — cc§7 "Invocation control", pi§2.
1840///
1841/// Empty string when there is nothing to list, so an agent with neither
1842/// skills nor templates keeps the prompt it had before this existed.
1843fn skills_prompt_section(config: &Config, skills: &[crate::skills::LoopSkill]) -> String {
1844    let listed: Vec<&crate::skills::LoopSkill> = skills
1845        .iter()
1846        .filter(|skill| skill.model_invocable)
1847        .collect();
1848    let mut templates: Vec<&str> = config.prompts.keys().map(String::as_str).collect();
1849    templates.sort_unstable();
1850    if listed.is_empty() && templates.is_empty() {
1851        return String::new();
1852    }
1853    let mut out = String::from("\n\n# Skills\n");
1854    if !listed.is_empty() {
1855        out.push_str(
1856            "Installed skill packages. Only each skill's name and description are listed \
1857             here; call the `skill` tool with a name below to load that skill's full \
1858             instructions when it applies, then follow them.\n",
1859        );
1860        for skill in listed {
1861            out.push_str(&skill.index_line());
1862            out.push('\n');
1863        }
1864    }
1865    if !templates.is_empty() {
1866        if !out.ends_with("# Skills\n") {
1867            out.push('\n');
1868        }
1869        out.push_str("Prompt templates (invoke via `/name args`):\n");
1870        for name in templates {
1871            out.push_str(&format!("- {name}\n"));
1872        }
1873    }
1874    out
1875}
1876
1877fn env_context_block(config: &Config) -> String {
1878    let mut lines = vec![
1879        format!("cwd: {}", config.cwd.display()),
1880        format!("platform: {}", std::env::consts::OS),
1881        format!(
1882            "date: {}",
1883            supercode_interchange::sidecar::now_rfc3339()
1884                .get(..10)
1885                .unwrap_or("")
1886        ),
1887        format!(
1888            "approval policy: {} · sandbox: {}",
1889            approval_policy_label(config.approval),
1890            sandbox_policy_label(config.sandbox),
1891        ),
1892    ];
1893    // BP-9 (§3.1 `core.project_root_markers`, catalog:232): the git probe
1894    // runs at the PROJECT ROOT the markers define, not at whatever
1895    // subdirectory the process happens to sit in — the marker knob's whole
1896    // job is deciding where "the project" starts. Falls back to `cwd` when
1897    // no ancestor carries a marker (or the list is empty), which is
1898    // byte-identical to the pre-BP-9 behavior.
1899    let root = crate::config::project_root_for(&config.cwd, &config.project_root_markers)
1900        .unwrap_or_else(|| config.cwd.clone());
1901    if let Some(status) = env_context_git_status(&root) {
1902        lines.push(status);
1903    }
1904    format!("\n\n# Environment\n{}", lines.join("\n"))
1905}
1906
1907/// The `[capabilities.permissions] approval` spelling of a policy — the same
1908/// token the config schema accepts (`configfile::parse_approval_str`), so
1909/// the block reports the policy in the vocabulary the user configured it in.
1910fn approval_policy_label(policy: crate::config::ApprovalPolicy) -> &'static str {
1911    match policy {
1912        crate::config::ApprovalPolicy::Never => "never",
1913        crate::config::ApprovalPolicy::OnRequest => "on-request",
1914        crate::config::ApprovalPolicy::Untrusted => "untrusted",
1915        crate::config::ApprovalPolicy::ModelRequested => "model-requested",
1916    }
1917}
1918
1919/// The `[capabilities.permissions] sandbox` spelling of a tier — see
1920/// [`approval_policy_label`].
1921fn sandbox_policy_label(policy: crate::tools::SandboxPolicy) -> &'static str {
1922    match policy {
1923        crate::tools::SandboxPolicy::ReadOnly => "read-only",
1924        crate::tools::SandboxPolicy::WorkspaceWrite => "workspace-write",
1925        crate::tools::SandboxPolicy::DangerFullAccess => "danger-full-access",
1926    }
1927}
1928
1929/// Best-effort `git branch (dirty|clean)` for [`env_context_block`]. `None`
1930/// on anything short of a clean success (not a repo, `git` missing, a
1931/// detached/errored state) — this is informational context, never worth
1932/// failing agent construction over.
1933fn env_context_git_status(cwd: &std::path::Path) -> Option<String> {
1934    let branch_out = std::process::Command::new("git")
1935        .args(["rev-parse", "--abbrev-ref", "HEAD"])
1936        .current_dir(cwd)
1937        .output()
1938        .ok()?;
1939    if !branch_out.status.success() {
1940        return None;
1941    }
1942    let branch = String::from_utf8_lossy(&branch_out.stdout)
1943        .trim()
1944        .to_string();
1945    if branch.is_empty() {
1946        return None;
1947    }
1948    let dirty = std::process::Command::new("git")
1949        .args(["status", "--porcelain"])
1950        .current_dir(cwd)
1951        .output()
1952        .ok()
1953        .map(|o| !o.stdout.is_empty())
1954        .unwrap_or(false);
1955    Some(format!(
1956        "git branch: {branch} ({})",
1957        if dirty { "dirty" } else { "clean" }
1958    ))
1959}
1960
1961impl Agent {
1962    /// Build an agent backed by an OpenAI-compatible endpoint (OpenRouter by
1963    /// default). The API key is taken from [`Config::api_key`], then
1964    /// [`Config::api_key_cmd`] (P4: a credential-helper command, run via the
1965    /// shell — see `run_api_key_cmd`), then the configured environment
1966    /// variable ([`Config::api_key_env`]).
1967    pub fn new(config: Config) -> Result<Self> {
1968        let api_key = match &config.api_key {
1969            Some(k) if !k.is_empty() => k.clone(),
1970            // BP-9: the ARGV helper (`core.api_key_command`) is consulted
1971            // first — it is the form with no shell in the path, so a config
1972            // that sets both gets the one with fewer ways to surprise its
1973            // author. Empty/failed → fall through, same as `api_key_cmd`.
1974            _ => match config
1975                .api_key_command
1976                .as_deref()
1977                .filter(|argv| !argv.is_empty())
1978                .map(run_api_key_command)
1979                .filter(|k| !k.is_empty())
1980                .or_else(|| {
1981                    config
1982                        .api_key_cmd
1983                        .as_deref()
1984                        .filter(|c| !c.is_empty())
1985                        .map(run_api_key_cmd)
1986                }) {
1987                // P4 (§1.8 credential-helper indirection, D6 row): the
1988                // helper ran and produced a non-empty key — use it. A
1989                // failed/empty helper falls through to `api_key_env` rather
1990                // than erroring outright, same "try the next source"
1991                // posture as every other layer in this resolution chain.
1992                Some(k) if !k.is_empty() => k,
1993                _ => std::env::var(&config.api_key_env)
1994                    .ok()
1995                    .filter(|k| !k.is_empty())
1996                    .ok_or_else(|| Error::MissingApiKey(config.api_key_env.clone()))?,
1997            },
1998        };
1999        // P4b (§1.1/§3.1 `core.retry`, pi§3 shape): `Config.retry_*` now
2000        // reaches the pre-existing transport-layer retry mechanism (see
2001        // `provider::HttpOptions::from_retry_config`'s doc comment for the
2002        // exact "byte-identical when unset" contract).
2003        let http_options = provider::HttpOptions::from_retry_config(
2004            config.retry_enabled,
2005            config.retry_max_retries,
2006            config.retry_base_delay_ms,
2007        );
2008        // BP-7 (catalog §4a "Turn/budget caps"): a spend cap armed against
2009        // a model this build cannot price is refused HERE rather than
2010        // accepted and silently never enforced. See
2011        // `Config::max_budget_usd`.
2012        if config.max_budget_usd.is_some_and(|b| b > 0.0)
2013            && crate::pricing::resolve(
2014                &config.model,
2015                config.price_input_per_mtok,
2016                config.price_output_per_mtok,
2017            )
2018            .is_none()
2019        {
2020            return Err(Error::UnpriceableBudget {
2021                model: config.model.clone(),
2022            });
2023        }
2024        // BP-7 (catalog §4a "Auto-retry on transient provider errors"): the
2025        // shared log the transport's retry loop reports into and
2026        // `Self::run_loop` drains after every completion.
2027        let retry_log = std::sync::Arc::new(crate::provider::RetryLog::default());
2028        let provider = OpenAiProvider::new_with_options(
2029            config.base_url.clone(),
2030            api_key,
2031            config.extra_headers.clone(),
2032            http_options,
2033        )
2034        .with_retry_log(retry_log.clone());
2035        // P3 (design §5.2): `ToolRegistry::from_config` replaces the
2036        // unconditional `with_builtins()` call — a no-op when
2037        // `config.module_registry` is off (the default, §5.3 risk 2).
2038        let registry = ToolRegistry::from_config(&config);
2039        let mut agent = Self::with_parts(config, Box::new(provider), registry);
2040        agent.retry_log = retry_log;
2041        Ok(agent)
2042    }
2043
2044    /// Build an agent with an explicit provider and the built-in tools. Handy
2045    /// for tests (inject a mock provider) or custom transports.
2046    pub fn with_provider(config: Config, provider: Box<dyn Provider>) -> Self {
2047        let registry = ToolRegistry::from_config(&config);
2048        Self::with_parts(config, provider, registry)
2049    }
2050
2051    /// Build an agent from all three parts.
2052    pub fn with_parts(
2053        mut config: Config,
2054        provider: Box<dyn Provider>,
2055        mut registry: ToolRegistry,
2056    ) -> Self {
2057        // P5-12 (§2 module 18 `plugins`, D-10): register every trusted,
2058        // loaded plugin's declared tools — the same "unconditional, config-
2059        // gated" wiring `build_tool_context` just below gives
2060        // checkpoint/formatters/lsp. `crate::plugins::register_into` is a
2061        // true no-op (no filesystem read, no subprocess) whenever
2062        // `config.plugins_enabled` is `false` (the default) — byte-identical
2063        // to before this module existed. Runs here (the one tail every
2064        // `Agent` construction path funnels through — `new`/`with_provider`
2065        // both call this) rather than in `ToolRegistry::from_config`, so it
2066        // is NOT entangled with that function's unrelated `module_registry`
2067        // experimental gate.
2068        crate::plugins::register_into(&config, &mut registry);
2069        let (mut ctx, checkpoint_observer, lsp_manager) = build_tool_context(&config);
2070        // Auto-load project context files (CLAUDE.md / AGENTS.md) from the
2071        // working directory (and any extra roots), appending them to the system
2072        // prompt — the analog of how Claude Code / Codex discover them.
2073        // P4b (§1.4): also the global/user tier + instruction imports + the
2074        // `project_doc_max_bytes` hygiene cap — see `assemble_project_instructions`.
2075        // BP-5 (catalog D2 "Per-model-family base-prompt selection", cx§2
2076        // "Per-model base instructions"): the base prompt is chosen for the
2077        // model in force, not fixed before the model is known — the exact
2078        // residue the ledger row named. `base_prompt_for_config` is
2079        // `config.system_prompt` verbatim for every config that sets no
2080        // family table, so this is a no-op by default.
2081        let base_prompt_live = base_prompt_for_config(&config);
2082        // BP-5 (catalog D2 "Output style / personality module"): a custom
2083        // style may REPLACE the base coding instructions rather than append
2084        // to them (cc§7 `keep-coding-instructions`); every other style is
2085        // appended at the end of the assembled prompt, below.
2086        let output_style = crate::output_style::resolve(&config);
2087        let mut system = match output_style.as_ref() {
2088            Some(style) if style.replaces_base => style.text.clone(),
2089            _ => base_prompt_live.clone(),
2090        };
2091        if config.load_project_context {
2092            system.push_str(&assemble_project_instructions(&config));
2093        }
2094        // BP-5 (catalog D2 "Path-scoped rules", cc§2 `.claude/rules`): the
2095        // UNSCOPED rules join the instruction blob here. A rule carrying a
2096        // `paths:` selector deliberately does not — it waits for a tool to
2097        // touch a matching file (`tools::builtins::path_rules_notice`).
2098        let path_rules = crate::path_rules::load(&config);
2099        system.push_str(&crate::path_rules::always_on_text(&path_rules));
2100        // P4b (§1.4/§3.1 `core.env_context`, catalog §4a "Environment
2101        // context block injection"): `false` (the default) is a no-op —
2102        // byte-identical to today's behavior. BP-4 keeps the rendered block
2103        // on the agent (`env_context_live`) so `refresh_env_context` can
2104        // find and REPLACE exactly this text when cwd/policy/branch move,
2105        // rather than leaving a stale block in the prompt forever.
2106        let env_context_live = if config.env_context {
2107            let block = env_context_block(&config);
2108            system.push_str(&block);
2109            Some(block)
2110        } else {
2111            None
2112        };
2113        // P4e (§1.4/§3.1 `core.context_injections`, catalog:91 "Synthetic
2114        // context-injection blocks"): same assembly site, right after
2115        // `env_context`. `false` (the default) is a no-op — byte-identical
2116        // to today's behavior. BP-4 routes it through
2117        // `crate::context_injection`, so the gate now delivers the built-in
2118        // ambient blocks the row is about (and stays extensible at runtime
2119        // through `Self::inject_context_block`) instead of only whatever
2120        // static list an embedder happened to populate.
2121        system.push_str(&crate::context_injection::assemble(&config, &[]));
2122        // P3 (design §5.2, §1.4 obligation 4, D-7): the skills prompt
2123        // section is a MODULE-GATED prompt section, the design's own
2124        // illustration of "a disabled module contributes no prompt
2125        // sections" — only assembled at all under
2126        // `[experimental] module_registry = true` (§5.3 risk 2: flag-off is
2127        // byte-for-byte today's behavior, and today's behavior never emits
2128        // this section, since it doesn't exist pre-P3). Gated further by
2129        // D-7 itself: `core.skills` requires a read pathway (`read_file` or
2130        // `bash`) — absent either, no section is appended, matching the
2131        // hard-dependency shape `configfile::validate_modules` enforces at
2132        // resolve time.
2133        //
2134        // BP-6 (catalog D2 "Skills (progressive-disclosure packages)"): the
2135        // section is now the discovered SKILL.md INDEX — each package's
2136        // frontmatter `name` and `description`, nothing else. A body is
2137        // never assembled here; it is read on invocation only, which is
2138        // what "progressive disclosure" means. The `[core.prompts]`
2139        // template names keep their own sub-list below it.
2140        let skills = crate::skills::load_for_config(&config);
2141        if config.module_registry && config.skills_enabled {
2142            let has_read_pathway = config
2143                .core_tools_enabled
2144                .iter()
2145                .any(|t| t == "read_file" || t == "bash");
2146            if has_read_pathway {
2147                system.push_str(&skills_prompt_section(&config, &skills));
2148            }
2149        }
2150        // BP-5: the style layer lands LAST, where cc puts it ("output styles
2151        // append custom instructions to the END of the system prompt").
2152        // Empty for a neutral style (`default`/`none`) and for a style that
2153        // already replaced the base above.
2154        if let Some(style) = output_style.as_ref().filter(|s| !s.replaces_base) {
2155            system.push_str(&style.section());
2156        }
2157        // BP-5: the SCOPED rules travel with the tool context, which is
2158        // where a "a tool touched a matching file" event can see them.
2159        ctx.path_rules = std::sync::Arc::new(path_rules);
2160        let history = vec![ChatMessage::system(system)];
2161        // P4e (§1.6/§3.1 `core.session.git_metadata`, catalog:331): captured
2162        // once here, alongside `env_context`'s own git probe — `false` (the
2163        // default) is a no-op, byte-identical to today's behavior.
2164        let git_metadata = if config.session_git_metadata {
2165            crate::git_metadata::capture(&config.cwd, now_ms())
2166        } else {
2167            None
2168        };
2169        // BP-7 (catalog §4a "Named agent definitions as data"): discover
2170        // `<cwd>/.claude/agents/*.md` for EVERY harness that turns the
2171        // subagents module on, not only the Claude emulate/resume path —
2172        // that restriction was the second half of the ledger row's
2173        // residue.
2174        merge_project_agent_definitions(&mut config);
2175        // P5-3: captured before `config` moves into the literal below (a
2176        // `usize` field READ, not a move, but it must happen before the
2177        // `config` shorthand field consumes the binding).
2178        let subagent_depth = config.subagent_depth;
2179        // BP-5 (catalog D2 "Shell-output injection in templates/skills"):
2180        // the authorization every `` !`cmd` `` in a skill/command body is
2181        // evaluated under — this config's own permission rules, resolved
2182        // once. Disabled unless `[core.skills] shell_injection` is on.
2183        let shell_injection = crate::skills::ShellInjection::from_config(&config);
2184        // BP-8 (catalog:151): `[capabilities.session_tree] enabled` finally
2185        // has a reader. An armed tree starts empty and grows one node per
2186        // recorded message — the degenerate single-path case, byte-for-byte
2187        // the same conversation, until a rewind or branch actually forks it.
2188        let session_tree = if config.session_tree_enabled {
2189            Some(supercode_interchange::session_tree::SessionTree::new())
2190        } else {
2191            None
2192        };
2193        // BP-7: resolved once here so the request path never re-does the
2194        // lookup, and so `Self::model_price` is `None` exactly when this
2195        // build cannot price the model.
2196        let model_price = crate::pricing::resolve(
2197            &config.model,
2198            config.price_input_per_mtok,
2199            config.price_output_per_mtok,
2200        );
2201        // BP-10: same reason — built before `config` moves into the
2202        // literal. `Config::permissions_approvals_persist` off (the
2203        // default) makes this the pre-BP-10 in-memory cache and touches no
2204        // filesystem.
2205        let permissions_approval_cache = crate::permissions::cache_for_config(&config);
2206        Agent {
2207            config,
2208            provider: std::sync::Arc::from(provider),
2209            registry,
2210            history,
2211            ctx,
2212            total_output_tokens: 0,
2213            activated_tools: HashSet::new(),
2214            recorder: None,
2215            journal: None,
2216            session_tree,
2217            rewind_undo: Vec::new(),
2218            journaled_plan: Vec::new(),
2219            reduction_policy: None,
2220            reduction_log: ReductionLog::default(),
2221            imported_prefix_len: None,
2222            compacting_manually: false,
2223            env_context_live,
2224            base_prompt_live,
2225            shell_injection,
2226            spliced_context_blocks: Vec::new(),
2227            span_summarizer: None,
2228            last_tool_schema_tier_signature: None,
2229            context_limit: None,
2230            requests_issued: false,
2231            last_cache_activity_ms: None,
2232            cache_established: false,
2233            pending_cache_turn: (false, false, None),
2234            session_titler: None,
2235            usage_log: Vec::new(),
2236            turn_index: 0,
2237            turn_records: Vec::new(),
2238            retry_log: std::sync::Arc::new(crate::provider::RetryLog::default()),
2239            model_price,
2240            total_cost_usd: 0.0,
2241            total_steps: 0,
2242            reaped_subagents: std::collections::HashMap::new(),
2243            goal: None,
2244            steer_queue: std::sync::Arc::new(std::sync::Mutex::new(SteerInbox::default())),
2245            follow_up_queue: std::collections::VecDeque::new(),
2246            doom_loop_last_call: None,
2247            doom_loop_streak: 0,
2248            model_change_log: Vec::new(),
2249            git_metadata,
2250            permissions_approval_cache,
2251            permissions_approval_handler: None,
2252            mcp_prompts: std::collections::HashMap::new(),
2253            skills,
2254            subagent_depth,
2255            subagent_concurrency_gauge: std::sync::Arc::new(std::sync::atomic::AtomicUsize::new(0)),
2256            background_subagents: std::collections::HashMap::new(),
2257            pending_child_approvals: std::sync::Arc::new(std::sync::Mutex::new(Vec::new())),
2258            child_approval_handler_factory: None,
2259            subagent_store: None,
2260            claude_runtime_manifest: None,
2261            background_jobs: std::collections::HashMap::new(),
2262            background_concurrency_gauge: std::sync::Arc::new(std::sync::atomic::AtomicUsize::new(
2263                0,
2264            )),
2265            checkpoint_observer,
2266            lsp_manager,
2267        }
2268    }
2269
2270    /// A handle to this agent's model transport, for sharing with subagents.
2271    pub fn provider_arc(&self) -> std::sync::Arc<dyn Provider> {
2272        self.provider.clone()
2273    }
2274
2275    /// Read-only access to this agent's resolved [`Config`] — e.g. so a
2276    /// caller (`crates/cli`'s `attach_mcp`) can consult
2277    /// [`Config::module_registry`]/[`Config::module_activation`] AFTER
2278    /// construction without having to separately thread the config through
2279    /// every call site that builds an `Agent` and later needs it again.
2280    /// Same trust boundary as every other already-public `Agent` accessor
2281    /// (`history`, `provider_arc`) — the caller is the same process that
2282    /// built this `Config` in the first place, not a new exposure surface.
2283    pub fn config(&self) -> &Config {
2284        &self.config
2285    }
2286
2287    /// P5-9 (§2 module 20 `checkpoint`): this agent's checkpoint engine, if
2288    /// `Config::checkpoint_enabled` is `true` and the shadow store opened
2289    /// successfully — `None` otherwise (the default-off case, or a
2290    /// graceful-degrade after an I/O failure). A caller (CLI/TUI/embedder)
2291    /// uses this to `list`/`turn_diff`/`restore` WITHOUT re-deriving the
2292    /// shadow-store root itself. Deliberately named `checkpoint_observer`,
2293    /// not `checkpoint` — [`Self::checkpoint`] already names the unrelated
2294    /// in-memory conversation-position marker (see that method's doc
2295    /// comment).
2296    pub fn checkpoint_observer(&self) -> Option<&crate::checkpoint::CheckpointObserver> {
2297        self.checkpoint_observer.as_deref()
2298    }
2299
2300    /// P5-11 (§2 module 28 `lsp`): this agent's LSP server registry, if
2301    /// `Config::lsp_enabled` is `true` — `None` otherwise (the default-off
2302    /// case). `impl Drop for Agent` already covers production teardown via
2303    /// [`crate::lsp::LspManager::kill_all_sync`] (a real, group-killing OS
2304    /// process kill — see `crate::lsp`'s module doc). This accessor exists
2305    /// for an OPTIONAL caller (CLI/TUI/embedder) that manages its own
2306    /// `Agent` lifecycle and additionally wants to reach
2307    /// [`crate::lsp::LspManager::shutdown_all`] for a graceful LSP
2308    /// `shutdown`/`exit` handshake BEFORE dropping the agent — nothing
2309    /// calls `shutdown_all` automatically today.
2310    pub fn lsp_manager(&self) -> Option<&crate::lsp::LspManager> {
2311        self.lsp_manager.as_deref()
2312    }
2313
2314    /// Spawn a subagent that shares this agent's model transport, runs `task`
2315    /// to completion with its own fresh conversation (seeded with `system`), and
2316    /// returns its final answer. The analog of `Agent` / `spawn_agent`.
2317    pub async fn run_subagent(
2318        &self,
2319        system: impl Into<String>,
2320        task: impl Into<String>,
2321    ) -> Result<String> {
2322        let mut sub_config = Config::builder()
2323            .model(self.config.model.clone())
2324            .system_prompt(system)
2325            .cwd(self.config.cwd.clone())
2326            .sandbox(self.config.sandbox)
2327            .max_iterations(self.config.max_iterations)
2328            .build();
2329        sub_config.base_url = self.config.base_url.clone();
2330        let mut sub = Agent::with_provider_arc(sub_config, self.provider.clone());
2331        sub.send(task).await
2332    }
2333
2334    /// Like [`Self::with_provider`] but sharing an existing transport handle.
2335    pub fn with_provider_arc(mut config: Config, provider: std::sync::Arc<dyn Provider>) -> Self {
2336        let (ctx, checkpoint_observer, lsp_manager) = build_tool_context(&config);
2337        let history = vec![ChatMessage::system(config.system_prompt.clone())];
2338        // P3 (design §5.2): see the `Self::new` doc note — a no-op when
2339        // `config.module_registry` is off (the default).
2340        let mut registry = ToolRegistry::from_config(&config);
2341        // P5-12: see `Self::with_parts`'s identical call — a no-op when
2342        // `config.plugins_enabled` is `false` (the default).
2343        crate::plugins::register_into(&config, &mut registry);
2344        // P4e: see `Self::with_parts`'s identical capture.
2345        let git_metadata = if config.session_git_metadata {
2346            crate::git_metadata::capture(&config.cwd, now_ms())
2347        } else {
2348            None
2349        };
2350        // BP-7 (catalog §4a "Named agent definitions as data"): discover
2351        // `<cwd>/.claude/agents/*.md` for EVERY harness that turns the
2352        // subagents module on, not only the Claude emulate/resume path —
2353        // that restriction was the second half of the ledger row's
2354        // residue.
2355        merge_project_agent_definitions(&mut config);
2356        // P5-3: captured before `config` moves into the literal below (a
2357        // `usize` field READ, not a move, but it must happen before the
2358        // `config` shorthand field consumes the binding).
2359        let subagent_depth = config.subagent_depth;
2360        // BP-6: this constructor assembles no prompt sections at all (it
2361        // takes `config.system_prompt` verbatim), so there is no skills
2362        // INDEX here — but the discovered set still rides along, so an
2363        // explicit invocation (`/name`, `$slug`, the `skill` tool) resolves
2364        // the same packages the registry's own `skill` tool holds.
2365        let skills = crate::skills::load_for_config(&config);
2366        let base_prompt_live = config.system_prompt.clone();
2367        let shell_injection = crate::skills::ShellInjection::from_config(&config);
2368        // BP-8 (catalog:151): `[capabilities.session_tree] enabled` finally
2369        // has a reader. An armed tree starts empty and grows one node per
2370        // recorded message — the degenerate single-path case, byte-for-byte
2371        // the same conversation, until a rewind or branch actually forks it.
2372        let session_tree = if config.session_tree_enabled {
2373            Some(supercode_interchange::session_tree::SessionTree::new())
2374        } else {
2375            None
2376        };
2377        // BP-7: resolved once here so the request path never re-does the
2378        // lookup, and so `Self::model_price` is `None` exactly when this
2379        // build cannot price the model.
2380        let model_price = crate::pricing::resolve(
2381            &config.model,
2382            config.price_input_per_mtok,
2383            config.price_output_per_mtok,
2384        );
2385        // BP-10: see the sibling constructor — built before `config` moves.
2386        let permissions_approval_cache = crate::permissions::cache_for_config(&config);
2387        Agent {
2388            config,
2389            provider,
2390            registry,
2391            history,
2392            ctx,
2393            total_output_tokens: 0,
2394            activated_tools: HashSet::new(),
2395            recorder: None,
2396            journal: None,
2397            session_tree,
2398            rewind_undo: Vec::new(),
2399            journaled_plan: Vec::new(),
2400            reduction_policy: None,
2401            reduction_log: ReductionLog::default(),
2402            imported_prefix_len: None,
2403            compacting_manually: false,
2404            env_context_live: None,
2405            // BP-5: this constructor assembles no prompt sections (see the
2406            // skills note above) — `config.system_prompt` IS the whole
2407            // system message, so that is what a later `set_model` would
2408            // have to replace.
2409            base_prompt_live,
2410            shell_injection,
2411            spliced_context_blocks: Vec::new(),
2412            span_summarizer: None,
2413            last_tool_schema_tier_signature: None,
2414            context_limit: None,
2415            requests_issued: false,
2416            last_cache_activity_ms: None,
2417            cache_established: false,
2418            pending_cache_turn: (false, false, None),
2419            session_titler: None,
2420            usage_log: Vec::new(),
2421            turn_index: 0,
2422            turn_records: Vec::new(),
2423            retry_log: std::sync::Arc::new(crate::provider::RetryLog::default()),
2424            model_price,
2425            total_cost_usd: 0.0,
2426            total_steps: 0,
2427            reaped_subagents: std::collections::HashMap::new(),
2428            goal: None,
2429            steer_queue: std::sync::Arc::new(std::sync::Mutex::new(SteerInbox::default())),
2430            follow_up_queue: std::collections::VecDeque::new(),
2431            doom_loop_last_call: None,
2432            doom_loop_streak: 0,
2433            model_change_log: Vec::new(),
2434            git_metadata,
2435            permissions_approval_cache,
2436            permissions_approval_handler: None,
2437            mcp_prompts: std::collections::HashMap::new(),
2438            skills,
2439            subagent_depth,
2440            subagent_concurrency_gauge: std::sync::Arc::new(std::sync::atomic::AtomicUsize::new(0)),
2441            background_subagents: std::collections::HashMap::new(),
2442            pending_child_approvals: std::sync::Arc::new(std::sync::Mutex::new(Vec::new())),
2443            child_approval_handler_factory: None,
2444            subagent_store: None,
2445            claude_runtime_manifest: None,
2446            background_jobs: std::collections::HashMap::new(),
2447            background_concurrency_gauge: std::sync::Arc::new(std::sync::atomic::AtomicUsize::new(
2448                0,
2449            )),
2450            checkpoint_observer,
2451            lsp_manager,
2452        }
2453    }
2454
2455    /// Run a prompt on a background task, returning a handle that resolves to
2456    /// the final answer (and the agent, so the caller can continue it). The
2457    /// analog of background/async agent runs.
2458    pub fn run_in_background(
2459        mut self,
2460        prompt: impl Into<String>,
2461    ) -> tokio::task::JoinHandle<(Self, Result<String>)>
2462    where
2463        Self: Send + 'static,
2464    {
2465        let prompt = prompt.into();
2466        tokio::spawn(async move {
2467            let result = self.send(prompt).await;
2468            (self, result)
2469        })
2470    }
2471
2472    /// Build an agent and seed it with a previously-recorded [`Session`] so it
2473    /// can continue where Claude Code or Codex left off.
2474    pub fn resume(config: Config, session: Session) -> Result<Self> {
2475        let mut agent = Agent::new(config)?;
2476        agent.load_session(session);
2477        Ok(agent)
2478    }
2479
2480    /// Like [`Self::resume`], but also begins recording (A2/A3): a fresh
2481    /// native-v2 sidecar is created at `sidecar_path` from `session` (header +
2482    /// `session.raw` verbatim — the imported prefix's own fidelity), and every
2483    /// subsequent turn this agent produces is appended to it at full fidelity,
2484    /// independent of whatever `cap_tool_output`/`maybe_compact` (D6) do to
2485    /// `history`.
2486    ///
2487    /// Invariant this establishes ONLY once [`Self::set_reduction_policy`] is
2488    /// also called (the D6/A7 supersession gate, `Self::run_loop`): at any
2489    /// instant, `Session::from_native_str(sidecar).messages` equals
2490    /// `session.messages` (the imported prefix) followed by every message
2491    /// appended since — i.e. `self.history()[1..]` (`history[0]` is this
2492    /// agent's own system prompt, per [`Self::load_session`]; it is never
2493    /// part of `session` and is never written to the sidecar). Recording
2494    /// alone (no policy) leaves the gate off: `cap_tool_output` still runs on
2495    /// oversized tool results, and `history` can diverge from the sidecar for
2496    /// them — honestly, via the notice's "full output in session sidecar"
2497    /// label, never silently.
2498    pub fn resume_recorded(
2499        config: Config,
2500        session: Session,
2501        sidecar_path: &std::path::Path,
2502    ) -> Result<Self> {
2503        let mut agent = Agent::new(config)?;
2504        let recorder = SidecarWriter::create(sidecar_path, &session)?;
2505        agent.load_session(session);
2506        agent.recorder = Some(recorder);
2507        Ok(agent)
2508    }
2509
2510    /// Install (or replace) this agent's sidecar recorder (A3).
2511    pub fn set_recorder(&mut self, w: SidecarWriter) {
2512        self.recorder = Some(w);
2513    }
2514
2515    // ---- BP-8: the append-only journal (catalog:150/152/154/156) --------
2516
2517    /// Install (or replace) this agent's append-only session journal — the
2518    /// durable, flush-per-record log of every message it produces plus
2519    /// every queue/rewind/plan operation performed on it. Installing one
2520    /// alone changes nothing about the conversation; it only makes the
2521    /// session survive a crash mid-turn.
2522    pub fn set_journal(&mut self, journal: crate::session_journal::SessionJournal) {
2523        self.journal = Some(std::sync::Arc::new(std::sync::Mutex::new(journal)));
2524    }
2525
2526    /// Whether an append-only journal is installed.
2527    pub fn has_journal(&self) -> bool {
2528        self.journal.is_some()
2529    }
2530
2531    /// Append one operation to the journal, if installed. Best-effort by
2532    /// design: losing a durability record must never fail the turn it
2533    /// describes, so the failure is logged and the loop continues — the
2534    /// same contract the compaction-marker `record` call keeps.
2535    fn journal_op(&self, op: crate::session_journal::JournalOp) {
2536        let Some(journal) = &self.journal else { return };
2537        let mut guard = journal
2538            .lock()
2539            .unwrap_or_else(std::sync::PoisonError::into_inner);
2540        if let Err(error) = guard.append(op) {
2541            tracing::warn!("failed to append a session-journal record: {error}");
2542        }
2543    }
2544
2545    /// Declare the durable view caught up: `<name>.jsonl` now holds
2546    /// `messages` messages and every journal record before this point is
2547    /// already in it. Everything journaled AFTER the last such record is
2548    /// exactly what a crash would have lost — see
2549    /// [`crate::session_journal::JournalState::unpersisted`].
2550    pub fn journal_checkpoint(&self, messages: usize) {
2551        self.journal_op(crate::session_journal::JournalOp::Checkpoint { messages });
2552    }
2553
2554    /// BP-13: record one per-turn usage entry in the append-only journal.
2555    pub fn journal_usage(&self, record: &crate::usage_log::UsageRecord) {
2556        self.journal_op(crate::session_journal::JournalOp::Usage {
2557            record: record.clone(),
2558        });
2559    }
2560
2561    /// BP-13: record one mid-session model change in the append-only
2562    /// journal — the ONE persisted home for a routing record (BP-8's
2563    /// journal), never a second file.
2564    pub fn journal_model_change(&self, record: &crate::model_change::ModelChangeRecord) {
2565        self.journal_op(crate::session_journal::JournalOp::ModelChange {
2566            record: record.clone(),
2567        });
2568    }
2569
2570    /// BP-8 (catalog:151): this session's conversation tree, when the
2571    /// module is on.
2572    pub fn session_tree(&self) -> Option<&supercode_interchange::session_tree::SessionTree> {
2573        self.session_tree.as_ref()
2574    }
2575
2576    /// Install a tree loaded from the store (a resume), replacing whatever
2577    /// this agent built. A no-op when the module is off — a session whose
2578    /// preset does not enable `session_tree` must not acquire one through
2579    /// the back door of an old sidecar.
2580    pub fn set_session_tree(&mut self, tree: supercode_interchange::session_tree::SessionTree) {
2581        if self.config.session_tree_enabled {
2582            self.session_tree = Some(tree);
2583        }
2584    }
2585
2586    /// BP-8: rebuild the tree from the current linear history — used after
2587    /// a resume that loaded a transcript but had no `.tree.json` to restore
2588    /// (every session recorded before the module was on).
2589    pub fn rebuild_session_tree_from_history(&mut self) {
2590        if !self.config.session_tree_enabled {
2591            return;
2592        }
2593        let linear: Vec<ChatMessage> = self.history.iter().skip(1).cloned().collect();
2594        self.session_tree =
2595            Some(supercode_interchange::session_tree::SessionTree::from_linear(&linear, now_ms()));
2596    }
2597
2598    /// BP-8 (catalog:152 "Rewind/rollback conversation"): move THIS
2599    /// conversation back to an earlier point — the whole row, not the
2600    /// last-exchange special case [`Self::rewind_to`] serves and not
2601    /// `sessions fork --at`, which makes a different session.
2602    ///
2603    /// `keep` is a message count (index into `history`), so `keep = 1`
2604    /// leaves only the system message. Three things happen, in this order:
2605    ///
2606    /// 1. the removed tail is pushed onto an undo stack, so
2607    ///    [`Self::undo_rewind`] can put it back;
2608    /// 2. a [`crate::session_journal::JournalOp::Rewind`] record is
2609    ///    APPENDED — nothing is deleted from disk, so the rewound-away
2610    ///    messages remain recoverable from the log;
2611    /// 3. when the tree module is on, the active branch's leaf moves to the
2612    ///    node at `keep`, and the old leaf is preserved under a fresh
2613    ///    sibling branch — the next message appended forks there rather
2614    ///    than overwriting.
2615    ///
2616    /// Returns what it did. Rewinding to a point at or past the end is a
2617    /// no-op with `removed = 0`, never an error.
2618    pub fn rewind_conversation(&mut self, keep: usize) -> RewindOutcome {
2619        let keep = keep.max(1).min(self.history.len());
2620        let removed: Vec<ChatMessage> = self.history.split_off(keep);
2621        if removed.is_empty() {
2622            return RewindOutcome {
2623                kept: self.history.len(),
2624                removed: 0,
2625                preserved_branch: None,
2626            };
2627        }
2628        let removed_count = removed.len();
2629        self.rewind_undo.push(removed);
2630        // `keep` counts the system message; the journal records only
2631        // `history[1..]`, so its own view is one shorter.
2632        self.journal_op(crate::session_journal::JournalOp::Rewind { to: keep - 1 });
2633        let preserved_branch = self.session_tree.as_mut().and_then(|tree| {
2634            let path = tree.active_path().unwrap_or_default();
2635            // `keep - 1` messages remain after the system message, so the
2636            // new leaf is the node at index `keep - 2`.
2637            match keep.checked_sub(2).and_then(|i| path.get(i).cloned()) {
2638                Some(node) => tree.rewind(&node, now_ms()).ok().flatten(),
2639                None => None,
2640            }
2641        });
2642        RewindOutcome {
2643            kept: self.history.len(),
2644            removed: removed_count,
2645            preserved_branch,
2646        }
2647    }
2648
2649    /// BP-8: invert the most recent [`Self::rewind_conversation`] — the
2650    /// messages come back, and the inversion is itself an appended journal
2651    /// record. `false` when there is nothing to undo.
2652    pub fn undo_rewind(&mut self) -> bool {
2653        let Some(mut tail) = self.rewind_undo.pop() else {
2654            return false;
2655        };
2656        self.history.append(&mut tail);
2657        self.journal_op(crate::session_journal::JournalOp::Unrewind);
2658        if self.config.session_tree_enabled {
2659            self.rebuild_session_tree_from_history();
2660        }
2661        true
2662    }
2663
2664    /// BP-8 (catalog:150): append messages recovered from the journal
2665    /// after a crash — they were already recorded, so this deliberately
2666    /// does NOT re-journal them; it puts the live conversation back where
2667    /// the interrupted process left it.
2668    pub fn append_recovered_messages(&mut self, messages: &[ChatMessage]) {
2669        for msg in messages {
2670            if let Some(tree) = self.session_tree.as_mut() {
2671                tree.append_message(msg.clone(), now_ms());
2672            }
2673            self.history.push(msg.clone());
2674        }
2675    }
2676
2677    /// BP-8: how many rewinds are currently undoable.
2678    pub fn undoable_rewinds(&self) -> usize {
2679        self.rewind_undo.len()
2680    }
2681
2682    /// BP-8: restore the undo stack a previous process left in the journal,
2683    /// so `/rewind undo` works across a restart.
2684    pub fn restore_rewind_undo(&mut self, stack: Vec<Vec<ChatMessage>>) {
2685        self.rewind_undo = stack;
2686    }
2687
2688    /// BP-8 (catalog:154 "Queued-prompt persistence"): re-queue pending
2689    /// inputs recovered from the journal WITHOUT re-recording them — they
2690    /// are already in the log, and journaling them again would double them
2691    /// on the next restart.
2692    pub fn restore_queues(&mut self, steer: &[String], follow_up: &[String]) {
2693        for message in steer {
2694            self.steer_queue
2695                .lock()
2696                .unwrap_or_else(std::sync::PoisonError::into_inner)
2697                .queue_unchecked(message.clone());
2698        }
2699        for message in follow_up {
2700            self.follow_up_queue.push_back(message.clone());
2701        }
2702    }
2703
2704    /// BP-8 (catalog:156 "Todos/plan persisted per session"): the session's
2705    /// current `update_plan` checklist.
2706    pub fn plan(&self) -> Vec<crate::session_journal::PlanEntry> {
2707        self.ctx.plan_snapshot()
2708    }
2709
2710    /// BP-8: restore a plan read back from the store on resume. Marked as
2711    /// already-journaled, so a resume that changes nothing writes nothing.
2712    pub fn set_plan(&mut self, steps: Vec<crate::session_journal::PlanEntry>) {
2713        self.ctx.set_plan(steps.clone());
2714        self.journaled_plan = steps;
2715    }
2716
2717    /// BP-8 (catalog:154): record that `count` pending inputs left `queue`
2718    /// and became conversation. A no-op when nothing was taken, or when
2719    /// queue persistence is off.
2720    fn journal_queue_drain(&self, queue: crate::session_journal::QueueKind, count: usize) {
2721        if count == 0 || !self.config.session_queue_persist {
2722            return;
2723        }
2724        self.journal_op(crate::session_journal::JournalOp::Dequeue { queue, count });
2725    }
2726
2727    /// BP-8: journal the plan if `update_plan` changed it since the last
2728    /// time this ran. Called at every loop boundary — a plan that a crash
2729    /// would otherwise strand in the tool's memory is on disk within one
2730    /// iteration of being written.
2731    fn journal_plan_if_changed(&mut self) {
2732        if !self.config.todos_persist {
2733            return;
2734        }
2735        let current = self.ctx.plan_snapshot();
2736        if current == self.journaled_plan {
2737            return;
2738        }
2739        self.journaled_plan.clone_from(&current);
2740        self.journal_op(crate::session_journal::JournalOp::Plan { steps: current });
2741    }
2742
2743    /// Install (or replace) this agent's reduction policy (A5/A7/A10). Once
2744    /// set, every provider request is built from a *projected* view of
2745    /// `history[1..]` (`reduce::project_messages`) rather than `history`
2746    /// verbatim — `history` itself is never shrunk or mutated by this; only
2747    /// the request view does.
2748    pub fn set_reduction_policy(&mut self, policy: ReductionPolicy) {
2749        self.reduction_policy = Some(policy);
2750    }
2751
2752    /// This agent's reduction policy, if one is installed.
2753    pub fn reduction_policy(&self) -> Option<&ReductionPolicy> {
2754        self.reduction_policy.as_ref()
2755    }
2756
2757    /// Change the global tool-schema tier (TR-8/T5) mid-session. Takes effect
2758    /// starting with the NEXT request this agent builds. Under
2759    /// [`CachePlan::ImportedPrefix`], the first request built after a change
2760    /// is flagged as a cache-bust event and its cache-control annotation is
2761    /// skipped for that one request (see [`provider::tier_change_is_cache_bust`],
2762    /// consulted in `Self::build_request_messages`) — normal annotation
2763    /// resumes on the next request if the tier doesn't change again.
2764    pub fn set_schema_tier(&mut self, tier: crate::tools::SchemaTier) {
2765        self.config.tool_schema_tier = tier;
2766    }
2767
2768    /// Override the schema tier for a single tool (TR-8/T5) mid-session, same
2769    /// cache-bust interaction as [`Self::set_schema_tier`].
2770    pub fn set_tool_schema_tier(
2771        &mut self,
2772        name: impl Into<String>,
2773        tier: crate::tools::SchemaTier,
2774    ) {
2775        self.config
2776            .tool_overrides
2777            .entry(name.into())
2778            .or_default()
2779            .schema_tier = Some(tier);
2780    }
2781
2782    /// A deterministic fingerprint of the current tool-schema tier
2783    /// configuration (global knob + every per-tool override), used to detect
2784    /// a mid-session tier change (TR-8/T5, dev/05). Order-independent over
2785    /// `tool_overrides` (sorted by name before hashing) so insertion order
2786    /// never spuriously changes the signature.
2787    fn schema_tier_signature(&self) -> u64 {
2788        use std::hash::{Hash, Hasher};
2789        let mut hasher = std::collections::hash_map::DefaultHasher::new();
2790        self.config.tool_schema_tier.hash(&mut hasher);
2791        let mut overrides: Vec<(&str, crate::tools::SchemaTier)> = self
2792            .config
2793            .tool_overrides
2794            .iter()
2795            .filter_map(|(name, o)| o.schema_tier.map(|t| (name.as_str(), t)))
2796            .collect();
2797        overrides.sort_by_key(|(name, _)| *name);
2798        for (name, tier) in overrides {
2799            name.hash(&mut hasher);
2800            tier.hash(&mut hasher);
2801        }
2802        hasher.finish()
2803    }
2804
2805    /// Install (or replace) this agent's TR-7 span summarizer — the
2806    /// injectable side-call `Self::build_request_messages` uses to turn an
2807    /// A10 `TurnsCleared` span into an LLM-written summary paragraph when
2808    /// `policy.summarize_cleared_turns` is on. Installing one alone changes
2809    /// nothing: [`ReductionPolicy::summarize_cleared_turns`] (off by
2810    /// default) is the actual gate, so tests/callers that want the
2811    /// deterministic stub can simply never call this.
2812    pub fn set_span_summarizer(
2813        &mut self,
2814        summarizer: impl reduce::summarize::SpanSummarizer + Send + Sync + 'static,
2815    ) {
2816        self.span_summarizer = Some(std::sync::Arc::new(summarizer));
2817    }
2818
2819    /// Install an already-shared summarizer — same seam as
2820    /// [`Self::set_span_summarizer`], for callers (and tests) that need to
2821    /// keep their own handle on it.
2822    pub fn set_span_summarizer_arc(
2823        &mut self,
2824        summarizer: std::sync::Arc<dyn reduce::summarize::SpanSummarizer + Send + Sync>,
2825    ) {
2826        self.span_summarizer = Some(summarizer);
2827    }
2828
2829    /// Prepare TR-7 metadata with this agent's installed summarizer for a
2830    /// projection performed by an outer driver before session history/log
2831    /// are loaded (the CLI foreign-resume preflight). `None` preserves the
2832    /// deterministic fallback when the gate is off, no summarizer exists,
2833    /// the span is below the cost floor, or the side-call fails.
2834    pub fn prepare_cleared_turns_summary(
2835        &self,
2836        msgs: &[ChatMessage],
2837        policy: &ReductionPolicy,
2838        prior: &ReductionLog,
2839    ) -> Option<reduce::PreparedClearSummary> {
2840        let summarizer = self.span_summarizer.as_deref()?;
2841        reduce::prepare_cleared_turns_summary(msgs, policy, prior, summarizer)
2842    }
2843
2844    /// P5-4: install (or replace) this agent's [`crate::EventSink`] AFTER
2845    /// construction — `Config::event_sink` is otherwise only set at
2846    /// `Config`-build time (before `Agent::new`), which is too early for a
2847    /// `tui` embedder that only knows it's activating (and needs to
2848    /// replace whatever print-mode/REPL sink was already installed with
2849    /// one that feeds its own render loop instead of writing straight to
2850    /// stdout) once it already holds a live `Agent`. Mirrors [`Self::
2851    /// set_permissions_approval_handler`]'s "installing one alone changes
2852    /// nothing beyond what already consults `Config::event_sink`" pattern
2853    /// — this is a plain replacement, not a new activation gate.
2854    pub fn set_event_sink(&mut self, sink: crate::EventSink) {
2855        self.config.event_sink = Some(sink);
2856    }
2857
2858    /// P5-1: install (or replace) this agent's permissions-engine approval
2859    /// handler — see [`crate::permissions::PermissionsApprovalHandler`].
2860    /// This is the non-interactive decision seam a CLI/TUI/SDK embedder
2861    /// implements for the `Ask`-tier prompt; the TUI's actual interactive
2862    /// UI is a separate module (P5 row 4), not built here. Installing one
2863    /// alone changes nothing: [`Config::permissions_enabled`] (off by
2864    /// default) is the actual gate — with no handler installed, every
2865    /// `Ask`-tier decision denies (fail-closed, see that trait's doc
2866    /// comment).
2867    ///
2868    /// P5-10 (§2 module 12, `escalation = "ask"`): the SAME handler also
2869    /// backs a sandbox-unenforceable `ask` decision
2870    /// (`crate::sandbox::decide_fs`'s `approval` parameter) — one installed
2871    /// seam serves both `permissions.rules`' `Ask` tier and
2872    /// `permissions.sandbox`'s `escalation = "ask"`, rather than requiring
2873    /// an embedder to install two near-identical handlers. Kept in sync on
2874    /// `self.ctx` (not just `self.permissions_approval_handler`) because
2875    /// `BashTool::execute`/`PersistentShellTool::execute` only ever see
2876    /// `&ToolContext`, never `&Agent` — see `ToolContext::
2877    /// sandbox_approval_handler`'s doc comment.
2878    /// BP-10 (catalog row "Session approval caching"): this agent's
2879    /// approval cache — the door an embedder/TUI uses to inspect or REVOKE
2880    /// remembered grants (`ApprovalCache::clear` forgets every one, in
2881    /// memory and on disk, and the next matching call asks again). Also
2882    /// how a test proves a grant really did survive the process:
2883    /// `store_path()` names the file a second agent reads back.
2884    pub fn permissions_approval_cache(&self) -> &crate::permissions::ApprovalCache {
2885        &self.permissions_approval_cache
2886    }
2887
2888    pub fn set_permissions_approval_handler(
2889        &mut self,
2890        handler: impl crate::permissions::PermissionsApprovalHandler + 'static,
2891    ) {
2892        let handler: std::sync::Arc<dyn crate::permissions::PermissionsApprovalHandler> =
2893            std::sync::Arc::new(handler);
2894        self.permissions_approval_handler = Some(handler.clone());
2895        self.ctx.sandbox_approval_handler =
2896            Some(crate::sandbox::SandboxApprovalHandler(handler.clone()));
2897        // BP-3 (§2 module 8): `exit_plan_mode` presents its plan on this
2898        // same door — one approval seam for the session, not a second one
2899        // the operator would have to answer separately.
2900        self.ctx.approval_handler = Some(crate::tools::ToolApprovalHandler(handler));
2901    }
2902
2903    /// BP-3 (§2 module 6 `tools.question`): install the door `ask_user`
2904    /// asks the human through — the `elicitation/create` handler the design
2905    /// names as the module's protocol side. SDK-owned frontends pass the
2906    /// broker-backed handler (`crate::server::FrontendRequestBridge::
2907    /// elicitation_handler`), which is what makes the question a real
2908    /// frontend request the turn waits on. `None` (the default, nothing
2909    /// installed) leaves the tool deny-default: it reports that nobody can
2910    /// be asked instead of blocking.
2911    ///
2912    /// Installing one alone changes nothing about whether the tool EXISTS —
2913    /// `[capabilities.tools_question]` is that gate, applied by
2914    /// `crate::tools::ToolRegistry::from_config`.
2915    pub fn set_user_question_handler(
2916        &mut self,
2917        handler: std::sync::Arc<dyn crate::mcp::McpElicitationHandler>,
2918    ) {
2919        self.ctx.question_handler = Some(crate::tools::UserQuestionHandler(handler));
2920    }
2921
2922    /// BP-3 (§2 module 8): the shared plan-mode state, so a frontend (the
2923    /// REPL's `/plan`, a TUI toggle) can enter or leave the read-only
2924    /// research phase the same tools and permission gate see.
2925    pub fn plan_mode(&self) -> &std::sync::Arc<crate::tools::PlanModeState> {
2926        &self.ctx.plan_mode
2927    }
2928
2929    /// Install the compatibility approval seam used when the composable
2930    /// permissions engine is disabled. SDK-owned interactive frontends call
2931    /// this alongside [`Self::set_permissions_approval_handler`] so the same
2932    /// authenticated request channel works under either policy engine; the
2933    /// selected engine remains entirely a configuration decision.
2934    pub fn set_legacy_approval_handler(&mut self, handler: crate::config::ApprovalHandler) {
2935        self.config.approval_handler = Some(handler);
2936    }
2937
2938    /// P5-3 (§2 module 9 D5 "subagent transcripts… persisted + linked"):
2939    /// install a [`crate::store::SessionStore`] (+ this agent's own session
2940    /// name in it) so `spawn_subagent` persists each child's transcript
2941    /// (via [`crate::store::SessionStore::save_subagent_transcript`]) and
2942    /// lineage record (via
2943    /// [`crate::store::SessionStore::save_subagent_lineage`]) once the
2944    /// child finishes. Installing one alone changes nothing about whether
2945    /// spawning WORKS — [`Config::subagents_enabled`] is the actual gate;
2946    /// this only controls whether a completed spawn's transcript additionally
2947    /// lands on disk.
2948    pub fn set_subagent_store(
2949        &mut self,
2950        store: std::sync::Arc<crate::store::SessionStore>,
2951        session_name: impl Into<String>,
2952    ) {
2953        self.subagent_store = Some((store, session_name.into()));
2954    }
2955
2956    /// Seed the Claude runtime manifest reconstructed during resume.
2957    ///
2958    /// Installing state enables the matching Claude runtime tool schemas so
2959    /// a disk-reloaded continuation does not lose that vocabulary, but never
2960    /// starts a timer by itself. The supplied execution posture is preserved:
2961    /// an embedding scheduler may deliberately activate before installing it.
2962    pub fn set_claude_runtime_manifest(
2963        &mut self,
2964        manifest: crate::claude_runtime_state::ClaudeRuntimeManifest,
2965    ) {
2966        // A persisted manifest is itself the compatibility capability marker.
2967        // Reopening a Supercode session must not retain its timers while
2968        // silently dropping Claude's Cron*/ScheduleWakeup vocabulary.
2969        self.config.claude_runtime_tools_enabled = true;
2970        self.claude_runtime_manifest = Some(manifest);
2971    }
2972
2973    /// Reinstall project-scoped Claude named-agent definitions when a
2974    /// Supercode continuation carrying a Claude runtime manifest is reopened
2975    /// from disk. The manifest is the durable capability marker; definitions
2976    /// themselves remain authoritative in `<cwd>/.claude/agents/*.md`.
2977    pub fn restore_claude_project_agents(&mut self) -> Result<usize> {
2978        let definitions = crate::claude_compat::load_project_agents(&self.config.cwd)?;
2979        crate::claude_compat::enable_claude_subagent_compatibility(&mut self.config);
2980        for imported in &definitions {
2981            self.config.subagents_definitions.insert(
2982                imported.definition.name.clone(),
2983                imported.definition.clone(),
2984            );
2985        }
2986        Ok(definitions.len())
2987    }
2988
2989    /// Current imported Claude runtime state, including paused mutations made
2990    /// by `Cron*`/`ScheduleWakeup`, for persistence by the embedding loop.
2991    pub fn claude_runtime_manifest(
2992        &self,
2993    ) -> Option<&crate::claude_runtime_state::ClaudeRuntimeManifest> {
2994        self.claude_runtime_manifest.as_ref()
2995    }
2996
2997    /// Mutable access for an embedding scheduler driver to atomically claim
2998    /// due events and persist the resulting manifest. Merely borrowing this
2999    /// state does not start a timer; execution remains the driver's explicit
3000    /// responsibility.
3001    pub fn claude_runtime_manifest_mut(
3002        &mut self,
3003    ) -> Option<&mut crate::claude_runtime_state::ClaudeRuntimeManifest> {
3004        self.claude_runtime_manifest.as_mut()
3005    }
3006
3007    /// P5-4 (tui, closes the P5-3 §2.2 C6 deferred chain): install a
3008    /// factory this agent's `Self::run_spawn_subagent` calls (with the
3009    /// fresh child's own id and this agent's shared
3010    /// [`Self::pending_child_approvals`] queue) to build the
3011    /// `PermissionsApprovalHandler` a `background_prompts = "parent"`
3012    /// child gets, INSTEAD of the default
3013    /// [`crate::subagents::ParentQueueApprovalHandler`]. Installing one
3014    /// alone changes nothing about whether background spawning works —
3015    /// [`Config::subagents_background_prompts`] being
3016    /// [`crate::subagents::BackgroundPromptsPolicy::Parent`] is the actual
3017    /// gate that reaches this factory at all; a `Parent`-policy child
3018    /// spawned before this is installed (or on an agent that never installs
3019    /// it) still gets the immediate-deny default, unchanged.
3020    ///
3021    /// **Security note.** The factory only controls WHICH handler answers
3022    /// an `Ask`-tier request — it can never widen what gets asked in the
3023    /// first place: [`crate::permissions::approval::resolve_ask`] only
3024    /// calls a handler's `ask` when the rule engine has already resolved
3025    /// the call to `Ask` (`Deny` short-circuits before any handler is
3026    /// consulted; `Allow` never needs one), so a parent's "allow" answer
3027    /// here can only grant what the policy already routed to a prompt —
3028    /// never override a `Deny` the engine already decided.
3029    pub fn set_child_approval_handler_factory(
3030        &mut self,
3031        factory: impl Fn(
3032                String,
3033                std::sync::Arc<std::sync::Mutex<Vec<crate::subagents::QueuedApproval>>>,
3034            ) -> std::sync::Arc<dyn crate::permissions::PermissionsApprovalHandler>
3035            + Send
3036            + Sync
3037            + 'static,
3038    ) {
3039        self.child_approval_handler_factory = Some(std::sync::Arc::new(factory));
3040    }
3041
3042    /// P5-3 (§2.2 C6 "parent-surfaced queue"): every approval request a
3043    /// `background_prompts = "parent"` child has raised so far, oldest
3044    /// first — a read-only audit view, not a mutable queue the caller
3045    /// answers. Under P5-3's own default handler (no P5-4 TUI factory
3046    /// installed) every entry here WAS already resolved `Deny` (a
3047    /// background call can't wait for an answer with no handler
3048    /// installed) — but once a `crate::tui::TuiChildApprovalHandler`
3049    /// factory is installed (P5-4,
3050    /// [`Self::set_child_approval_handler_factory`]), the underlying call
3051    /// genuinely blocks and may resolve `Allow`/`AllowForSession`; this
3052    /// method still records the SAME entry for the audit trail either
3053    /// way, so "queued here" no longer implies "was denied" in general —
3054    /// see [`crate::subagents::QueuedApproval`]'s doc comment.
3055    pub fn pending_child_approvals(&self) -> Vec<crate::subagents::QueuedApproval> {
3056        self.pending_child_approvals
3057            .lock()
3058            .map(|q| q.clone())
3059            .unwrap_or_default()
3060    }
3061
3062    /// P4b: install (or replace) this agent's auto-title side-call — see
3063    /// [`crate::session_title::SessionTitler`]. Installing one alone changes
3064    /// nothing: [`Config::auto_title`] (off by default) is the actual gate a
3065    /// caller should consult before calling [`Self::auto_title`].
3066    pub fn set_session_titler(
3067        &mut self,
3068        titler: impl crate::session_title::SessionTitler + Send + Sync + 'static,
3069    ) {
3070        self.session_titler = Some(std::sync::Arc::new(titler));
3071    }
3072
3073    /// P4b: produce a title for this agent's current conversation via the
3074    /// installed [`Self::set_session_titler`] side-call. Returns `None` (never
3075    /// panics, never blocks longer than the titler itself does) if no
3076    /// titler is installed, or the side-call itself declined (see
3077    /// [`crate::session_title::auto_title`]). Does NOT consult
3078    /// [`Config::auto_title`] itself — that gate is the caller's
3079    /// responsibility, matching `Self::span_summarizer`'s precedent of
3080    /// keeping the mechanism and the policy gate separate.
3081    pub fn auto_title(&self) -> Option<String> {
3082        let titler = self.session_titler.as_deref()?;
3083        crate::session_title::auto_title(&self.history, titler)
3084    }
3085
3086    /// P4b (§1.6, catalog §4a "persisted per-turn usage records"): every
3087    /// [`crate::usage_log::UsageRecord`] this agent has accumulated so far.
3088    pub fn usage_records(&self) -> &[crate::usage_log::UsageRecord] {
3089        &self.usage_log
3090    }
3091
3092    /// P4b: persist this agent's accumulated usage log to `store` under
3093    /// `name` — a thin wrapper over [`crate::store::SessionStore::save_usage_log`]
3094    /// so callers don't need to import both types.
3095    pub fn save_usage_log(&self, store: &crate::store::SessionStore, name: &str) -> Result<()> {
3096        store.save_usage_log(name, &self.usage_log)
3097    }
3098
3099    /// BP-7 (catalog §4a "Turn/step bracketing records"): every
3100    /// [`crate::turn_record::TurnRecord`] this agent has accumulated —
3101    /// the context/usage/finish brackets of each model round-trip plus the
3102    /// retry, abort, effort and goal markers between them.
3103    pub fn turn_records(&self) -> &[crate::turn_record::TurnRecord] {
3104        &self.turn_records
3105    }
3106
3107    /// BP-7: persist the marker log to `store` under `name`
3108    /// (`<name>.events.jsonl`), the same thin-wrapper shape
3109    /// [`Self::save_usage_log`] has.
3110    pub fn save_turn_records(&self, store: &crate::store::SessionStore, name: &str) -> Result<()> {
3111        store.save_turn_records(name, &self.turn_records)
3112    }
3113
3114    /// BP-7 (catalog §4a "Per-turn cost/usage accounting"): dollars this
3115    /// agent has spent so far. `0.0` when the model is unpriceable — read
3116    /// [`Self::model_priced`] to tell "free" from "unknown".
3117    pub fn total_cost_usd(&self) -> f64 {
3118        self.total_cost_usd
3119    }
3120
3121    /// BP-7: whether this build can price this agent's model, i.e. whether
3122    /// [`Self::total_cost_usd`] is a real figure rather than a floor.
3123    pub fn model_priced(&self) -> bool {
3124        self.model_price.is_some()
3125    }
3126
3127    /// BP-7 (catalog §4a "Turn/budget caps"): tool calls this agent has
3128    /// executed so far — the counter [`Config::max_steps`] bounds.
3129    pub fn total_steps(&self) -> usize {
3130        self.total_steps
3131    }
3132
3133    /// BP-7 (catalog §4a "Interrupt/abort with state preserved"): record
3134    /// that the in-flight turn was interrupted.
3135    ///
3136    /// Called by whoever owns the cancellation (the CLI's Ctrl-C race), NOT
3137    /// by the loop itself: a cancelled `send` future is dropped mid-await,
3138    /// so the loop never runs another line. The partial work already
3139    /// appended to the transcript stands; this marker is what makes the
3140    /// interruption a persisted FACT — the residue the ledger row named —
3141    /// rather than something a reader has to infer from a dangling tool
3142    /// call on reload. Emits [`AgentEvent::TurnAborted`] as the live
3143    /// counterpart.
3144    pub fn note_abort(&mut self, source: &str) {
3145        let messages = self.history.len();
3146        self.emit(AgentEvent::TurnAborted {
3147            source: source.to_string(),
3148        });
3149        self.push_turn_marker(crate::turn_record::TurnMarker::Aborted {
3150            source: source.to_string(),
3151            messages,
3152        });
3153    }
3154
3155    // ---- BP-7: goals (catalog §4a "Goals — persistent objective across
3156    // turns"; §2 module 7 `todos`, §3.1 `capabilities.todos.goals`) ----
3157
3158    /// Set (or revise) this session's standing objective.
3159    ///
3160    /// Returns `false`, changing nothing, when `capabilities.todos.goals`
3161    /// is off — the module gate, not a silent success. A goal restates
3162    /// itself at the tail of every request until [`Self::clear_goal`], and
3163    /// each change appends a `goal` marker to the turn-record log.
3164    pub fn set_goal(&mut self, objective: impl Into<String>) -> bool {
3165        if !self.config.goals_enabled {
3166            return false;
3167        }
3168        let objective = objective.into();
3169        let now = now_ms();
3170        match &mut self.goal {
3171            Some(goal) => goal.revise(objective.clone(), now),
3172            slot @ None => *slot = Some(crate::goals::GoalRecord::new(objective.clone(), now)),
3173        }
3174        self.push_turn_marker(crate::turn_record::TurnMarker::Goal { objective });
3175        true
3176    }
3177
3178    /// This session's standing objective, if one is set.
3179    pub fn goal(&self) -> Option<&crate::goals::GoalRecord> {
3180        self.goal.as_ref()
3181    }
3182
3183    /// Drop the standing objective. `true` when there was one to drop.
3184    pub fn clear_goal(&mut self) -> bool {
3185        if self.goal.take().is_none() {
3186            return false;
3187        }
3188        self.push_turn_marker(crate::turn_record::TurnMarker::Goal {
3189            objective: String::new(),
3190        });
3191        true
3192    }
3193
3194    /// BP-7: persist (or, when cleared, remove) the standing objective
3195    /// beside the session — same thin-wrapper shape as
3196    /// [`Self::save_usage_log`].
3197    pub fn save_goal(&self, store: &crate::store::SessionStore, name: &str) -> Result<()> {
3198        match &self.goal {
3199            Some(goal) => store.save_goal(name, goal),
3200            None => store.clear_goal(name),
3201        }
3202    }
3203
3204    /// BP-7: adopt a goal loaded from the store (a resumed session picks up
3205    /// exactly where it left off). Bypasses the module gate on purpose: a
3206    /// goal already persisted is data to restore, not a new capability
3207    /// being turned on, and dropping it silently would lose session state.
3208    pub fn restore_goal(&mut self, goal: Option<crate::goals::GoalRecord>) {
3209        self.goal = goal;
3210    }
3211
3212    // ---- BP-7: extended-thinking control (catalog §4a "Extended thinking
3213    // control": "Reasoning on/off/levels mid-session") ----
3214
3215    /// The reasoning-effort level in force for the NEXT request, or `None`
3216    /// when extended thinking is off.
3217    pub fn effort(&self) -> Option<&str> {
3218        self.config.effort.as_deref()
3219    }
3220
3221    /// Change the reasoning-effort level mid-session.
3222    ///
3223    /// `Some(level)` sets the level; `None` turns extended thinking OFF —
3224    /// the on/off toggle the ledger row named as distinct from the level.
3225    /// `run_loop` reads `self.config.effort` fresh when it builds each
3226    /// `ChatRequest`, so this takes effect on the very next request with no
3227    /// other copy to update (the same contract [`Self::set_model`] has).
3228    /// The change is appended to the turn-record log as an `effort` marker,
3229    /// the extended-thinking analog of the `model_change` log.
3230    ///
3231    /// Returns the PREVIOUS setting.
3232    pub fn set_effort(&mut self, effort: Option<String>) -> Option<String> {
3233        let previous = self.config.effort.clone();
3234        if previous == effort {
3235            return previous;
3236        }
3237        self.config.effort = effort.clone();
3238        self.push_turn_marker(crate::turn_record::TurnMarker::Effort {
3239            from: previous.clone(),
3240            to: effort,
3241        });
3242        previous
3243    }
3244
3245    // ---- BP-7: review mode (catalog §4a "Review mode — dedicated
3246    // code-review flow"; §3.1 `[core.prompts]`) ----
3247
3248    /// The purpose-built review turn's prompt: the `code-review` template
3249    /// from [`Config::prompts`] with `{args}` replaced by `args`.
3250    ///
3251    /// `None` when the resolved config carries no `code-review` template —
3252    /// the preset decides whether this harness has a review mode, and the
3253    /// template IS the report format (both parity presets pin one).
3254    pub fn review_prompt(&self, args: &str) -> Option<String> {
3255        self.config
3256            .prompts
3257            .get(REVIEW_PROMPT_NAME)
3258            .map(|template| template.replace("{args}", args.trim()))
3259    }
3260
3261    /// Run the review turn: an ordinary [`Self::send`] of
3262    /// [`Self::review_prompt`], so the review's request, tools, transcript
3263    /// and records are the session's own — a purpose-built TURN, not a
3264    /// second agent.
3265    pub async fn review(&mut self, args: &str) -> Result<String> {
3266        let prompt = self.review_prompt(args).ok_or_else(|| {
3267            Error::Other(format!(
3268                "no `{REVIEW_PROMPT_NAME}` prompt template is configured for this harness"
3269            ))
3270        })?;
3271        self.send(prompt).await
3272    }
3273
3274    // ---- BP-7: side/ephemeral Q&A (catalog §4a "Side/ephemeral Q&A":
3275    // "Tool-less question over full context, never enters history") ----
3276
3277    /// Answer `question` over this session's FULL current context without
3278    /// recording anything.
3279    ///
3280    /// Three properties, all load-bearing and all asserted by this build's
3281    /// tests: the request carries the whole conversation as the next turn
3282    /// would see it; it advertises NO tools, so the model can only answer;
3283    /// and neither `history`, the sidecar recorder, the usage log nor the
3284    /// turn-record log is touched — `&self`, not `&mut self`, is the type
3285    /// system saying so. cc's `/btw` and cx's `/side`.
3286    pub async fn side_question(&self, question: &str) -> Result<String> {
3287        let mut messages = self.history.clone();
3288        if let Some(goal) = &self.goal {
3289            messages.push(ChatMessage::system(goal.reminder()));
3290        }
3291        messages.push(ChatMessage::user(format!(
3292            "{SIDE_QUESTION_PREAMBLE}
3293
3294{question}"
3295        )));
3296        let mut req = ChatRequest {
3297            model: self.config.model.clone(),
3298            messages,
3299            tools: Vec::new(),
3300            temperature: self.config.temperature,
3301            max_tokens: self.config.max_tokens,
3302            effort: self.config.effort.clone(),
3303            response_format: None,
3304            service_tier: None,
3305            thinking_budget: None,
3306            extra_body: self.config.extra_body.clone(),
3307        };
3308        // BP-13: a side question is still a request to THIS model, so it
3309        // carries the same routing decisions the loop's own requests do.
3310        self.apply_routing(&mut req);
3311        let (assistant, _usage) = self.provider.complete(&req, &|_: &str| {}).await?;
3312        Ok(assistant.content.unwrap_or_default())
3313    }
3314
3315    /// BP-7: append one marker against the NEXT round-trip's index — the
3316    /// right frame for a marker written between turns (a goal change, an
3317    /// effort change, an abort).
3318    fn push_turn_marker(&mut self, marker: crate::turn_record::TurnMarker) {
3319        self.push_turn_marker_at(self.turn_index, marker);
3320    }
3321
3322    /// BP-7: append one marker against an explicit round-trip index — used
3323    /// inside `Self::run_loop`, where markers are written on both sides of
3324    /// the `turn_index` advance and must all carry the round-trip they
3325    /// describe.
3326    fn push_turn_marker_at(&mut self, turn: usize, marker: crate::turn_record::TurnMarker) {
3327        self.turn_records.push(crate::turn_record::TurnRecord::new(
3328            turn,
3329            &self.config.model,
3330            now_ms(),
3331            marker,
3332        ));
3333    }
3334
3335    /// P4b (§1.7, pi§3 semantics): queue a mid-turn steering message —
3336    /// delivered "after current tool calls" (pi's phrasing): at the top of
3337    /// `Self::run_loop`'s NEXT iteration, before the next model request is
3338    /// built, regardless of whether this turn is still mid-flight with
3339    /// pending tool calls. Drained per [`Config::steering_mode`].
3340    pub fn queue_steer(&self, message: impl Into<String>) {
3341        let message = message.into();
3342        // BP-8 (catalog:154 "Queued-prompt persistence"): the input is
3343        // recorded BEFORE it is queued, so the window in which a crash
3344        // could lose it is zero. A no-op when `core.session.queue_persist`
3345        // is off (cx-parity: stock Codex has no queue-operation records).
3346        if self.config.session_queue_persist {
3347            self.journal_op(crate::session_journal::JournalOp::Enqueue {
3348                queue: crate::session_journal::QueueKind::Steer,
3349                text: message.clone(),
3350            });
3351        }
3352        self.steer_queue
3353            .lock()
3354            .unwrap_or_else(std::sync::PoisonError::into_inner)
3355            .queue_unchecked(message);
3356    }
3357
3358    /// Crate-internal shared steering handle used by the canonical SDK
3359    /// runtime. It remains writable while an active turn holds `&mut Agent`,
3360    /// allowing local and remote frontends to steer without owning the loop.
3361    pub(crate) fn steer_queue_handle(&self) -> std::sync::Arc<std::sync::Mutex<SteerInbox>> {
3362        self.steer_queue.clone()
3363    }
3364
3365    /// P4b: queue a follow-up message — delivered "at idle" (pi's phrasing):
3366    /// only once `Self::run_loop` would otherwise return a final answer
3367    /// (no more tool calls pending). Drained per [`Config::follow_up_mode`].
3368    pub fn queue_follow_up(&mut self, message: impl Into<String>) {
3369        let message = message.into();
3370        // BP-8 (catalog:154): same record-then-queue order as
3371        // [`Self::queue_steer`].
3372        if self.config.session_queue_persist {
3373            self.journal_op(crate::session_journal::JournalOp::Enqueue {
3374                queue: crate::session_journal::QueueKind::FollowUp,
3375                text: message.clone(),
3376            });
3377        }
3378        self.follow_up_queue.push_back(message);
3379    }
3380
3381    /// P4b: how many steering messages are currently queued (mid-turn +
3382    /// follow-up combined) — mostly for tests/diagnostics.
3383    pub fn queued_steer_count(&self) -> usize {
3384        self.steer_queue
3385            .lock()
3386            .unwrap_or_else(std::sync::PoisonError::into_inner)
3387            .len()
3388            + self.follow_up_queue.len()
3389    }
3390
3391    /// The accumulating reduction log (A5) — every reduction applied to any
3392    /// projected request view so far. Combined with a full-fidelity sidecar
3393    /// Session, this is enough to `reduce::invert` any projected view back to
3394    /// the exact original.
3395    pub fn reduction_log(&self) -> &ReductionLog {
3396        &self.reduction_log
3397    }
3398
3399    /// PARITY-18 D4 — arm the per-send context guard: `Self::run_loop`
3400    /// will refuse (via [`Error::ContextLimitExceeded`]) to build and issue
3401    /// ANY request — the first or any later turn — whose
3402    /// [`supercode_runtime::context_guard`] verdict is "does not fit" against
3403    /// `limit`. Call this once the target model's context-window size is
3404    /// known (`resume --reduced`'s preflight already computes it). Leaving
3405    /// this unset (the default) is a no-op: no guard runs, exactly today's
3406    /// pre-PARITY-18 behavior.
3407    pub fn set_context_limit(&mut self, limit: u64) {
3408        self.context_limit = Some(limit);
3409    }
3410
3411    /// This agent's armed context limit, if [`Self::set_context_limit`] has
3412    /// been called.
3413    pub fn context_limit(&self) -> Option<u64> {
3414        self.context_limit
3415    }
3416
3417    /// The model identifier this agent sends on its next request
3418    /// ([`Config::model`], as of construction/resume or the last
3419    /// [`Self::set_model`] call).
3420    pub fn model(&self) -> &str {
3421        &self.config.model
3422    }
3423
3424    /// UX-30 dev/02 — switch the model this agent sends, starting with the
3425    /// NEXT request it builds (and every one after, until changed again).
3426    /// `Self::run_loop` reads `self.config.model` fresh on every request
3427    /// (see its `ChatRequest` construction), so this alone is enough —
3428    /// there is no cached/baked-in copy anywhere else to also update.
3429    /// Takes effect immediately; safe to call only between turns (the
3430    /// REPL's `/model` picker runs at the prompt, never mid-turn). Touches
3431    /// nothing else: history, the sidecar, and reduction state are exactly
3432    /// as untouched as [`Self::set_schema_tier`] leaves them for a
3433    /// mid-session tier change.
3434    ///
3435    /// P4c-review note: this is the LOW-LEVEL primitive — it swaps
3436    /// [`Config::model`] and nothing else. It does NOT run dep 8's
3437    /// reasoning-artifact filter
3438    /// ([`reduce::rehydrate::filter_reasoning_artifacts`]) and does NOT
3439    /// create a [`crate::model_change::ModelChangeRecord`], so calling it
3440    /// directly for a mid-session handoff between two DIFFERENT models
3441    /// leaves model-A's reasoning artifacts in `history` for model-B to
3442    /// inherit. [`Self::switch_model`] is the safe superset — gated by
3443    /// [`Config::model_switch_allow_switch`], it filters and records the
3444    /// switch before delegating to this method — and is what callers
3445    /// performing a governed mid-session model switch should use instead.
3446    pub fn set_model(&mut self, model: impl Into<String>) {
3447        self.config.model = model.into();
3448        // BP-5 (catalog D2 "Per-model-family base-prompt selection"): the
3449        // family's base prompt follows the model. Codex re-selects
3450        // `base_instructions` when the model changes; leaving model-A's
3451        // base prompt in front of model-B is exactly the mismatch the row
3452        // exists to prevent. Same locate-and-replace mechanism
3453        // `refresh_env_context` uses, and a no-op whenever the selection
3454        // did not actually change (always, for a config with no family
3455        // table).
3456        self.refresh_base_prompt();
3457        // BP-7: the price follows the model, or the per-turn cost figure
3458        // would keep billing the OLD model's rates after a switch.
3459        self.model_price = crate::pricing::resolve(
3460            &self.config.model,
3461            self.config.price_input_per_mtok,
3462            self.config.price_output_per_mtok,
3463        );
3464    }
3465
3466    /// P4c (§1.10/§3.1 `core.model_switch.allow_switch`, D9 row, dep 8,
3467    /// design's "core NEW-significant" item): the mid-session model
3468    /// switch — a superset of [`Self::set_model`] gated by
3469    /// [`Config::model_switch_allow_switch`].
3470    ///
3471    /// **`allow_switch = false` (the default): EXACTLY [`Self::set_model`]**
3472    /// — same single field write, nothing else touched, no
3473    /// [`crate::model_change::ModelChangeRecord`] created. Byte-identical to
3474    /// calling `set_model` directly.
3475    ///
3476    /// **`allow_switch = true`:** additionally, before the swap takes
3477    /// effect, runs [`reduce::rehydrate::filter_reasoning_artifacts`] over
3478    /// [`Self::history`] — model-A's reasoning/thinking artifacts (any
3479    /// [`supercode_interchange::ChatMessage::metadata`] key in
3480    /// [`reduce::rehydrate::REASONING_METADATA_KEYS`], any `content_parts`
3481    /// block whose `"type"` is in
3482    /// [`reduce::rehydrate::REASONING_CONTENT_PART_TYPES`]) are stripped
3483    /// BEFORE model-B ever builds a request from this history — then
3484    /// appends a typed, translatable [`crate::model_change::ModelChangeRecord`]
3485    /// to [`Self::model_change_records`] (persist it via
3486    /// [`Self::save_model_change_log`]). A switch TO the current model
3487    /// (`model == Self::model()`) is treated as a no-op — still exactly
3488    /// `set_model`'s mechanics, no record for a switch that didn't actually
3489    /// change anything (and nothing to filter FOR, since there was no
3490    /// handoff).
3491    pub fn switch_model(&mut self, model: impl Into<String>) {
3492        let to = model.into();
3493        if !self.config.model_switch_allow_switch || self.config.model == to {
3494            self.set_model(to);
3495            return;
3496        }
3497        let from = self.config.model.clone();
3498        self.record_model_change(&from, &to, None);
3499    }
3500
3501    /// BP-13 — the ONE place a mid-session model change is performed and
3502    /// recorded, shared by [`Self::switch_model`] (a user asked) and the
3503    /// run loop's fallback pass (a provider failed).
3504    ///
3505    /// It does four things, in this order, and nothing else: strips model-A
3506    /// reasoning artifacts out of the live history (dep 8 — model B must
3507    /// never inherit them), moves [`Config::model`], appends the typed
3508    /// [`crate::model_change::ModelChangeRecord`], and writes that same
3509    /// record into the append-only session journal (BP-8) — which is where
3510    /// every persisted routing record lives; there is no second file. The
3511    /// change is also EMITTED, so a surface that renders events shows the
3512    /// switch instead of silently answering as a different model.
3513    pub fn record_model_change(&mut self, from: &str, to: &str, reason: Option<&str>) {
3514        if from == to {
3515            return;
3516        }
3517        let touched = reduce::rehydrate::filter_reasoning_artifacts(&mut self.history);
3518        self.set_model(to.to_string());
3519        let record = crate::model_change::ModelChangeRecord::new(
3520            self.turn_index,
3521            from,
3522            to,
3523            true,
3524            touched,
3525            now_ms(),
3526        )
3527        .with_reason(reason.map(str::to_string));
3528        self.journal_model_change(&record);
3529        self.model_change_log.push(record);
3530        self.emit(AgentEvent::ModelChanged {
3531            from: from.to_string(),
3532            to: to.to_string(),
3533            reason: reason.map(str::to_string),
3534        });
3535        // A switch also re-injects the switch NOTICE when the config asks
3536        // for one (Codex's own mid-session behavior: the conversation is
3537        // told the model changed, so the new model reads the handoff rather
3538        // than inferring it from a style break).
3539        if self.config.model_switch_notice {
3540            let notice = ChatMessage::user(format!(
3541                "[model changed: {from} -> {to}{}]",
3542                match reason {
3543                    Some(r) => format!(" ({r})"),
3544                    None => String::new(),
3545                }
3546            ));
3547            let _ = self.record(&notice);
3548            self.history.push(notice);
3549        }
3550    }
3551
3552    /// BP-13 (catalog D9 "Fast mode / service tiers"): set or clear the
3553    /// session-level service-tier override. `Some(tier)` WINS over the
3554    /// `[capabilities.model_catalog] service_tier` rule for every
3555    /// subsequent request (it is the live toggle the user just pulled);
3556    /// `None` puts the configured rule back in charge. Takes effect on the
3557    /// next request the loop builds, like [`Self::set_model`].
3558    pub fn set_service_tier(&mut self, tier: Option<String>) {
3559        self.config.service_tier = tier;
3560    }
3561
3562    /// BP-13 — apply the routing table to a request that already names its
3563    /// model: effort LEVEL (per-model override of `[core] effort`, clamped
3564    /// by whichever effort cap applies), thinking-token BUDGET, and service
3565    /// TIER (the live `/fast` override winning over the configured rule).
3566    /// Called for every request the loop builds AND again for every
3567    /// fallback hop, so a hop to a different model gets that model's
3568    /// routing rather than the previous model's.
3569    fn apply_routing(&self, req: &mut ChatRequest) {
3570        let routing = &self.config.model_routing;
3571        let rules = routing.rules_for(&req.model);
3572        // Plan mode's own effort tier, when the mode is live, is the
3573        // session level for this request — Codex's `/plan` is effort
3574        // steering (cx§6), so planning need not think at the executing
3575        // level. It is still clamped by whatever effort cap applies,
3576        // because `effective_effort` does the clamping, not this line.
3577        let session_effort = match (
3578            self.ctx.plan_mode.is_active(),
3579            self.config.plan_mode_effort.as_deref(),
3580        ) {
3581            (true, Some(effort)) => Some(effort),
3582            _ => self.config.effort.as_deref(),
3583        };
3584        req.effort = routing.effective_effort(&req.model, session_effort);
3585        req.thinking_budget = rules.thinking_budget;
3586        req.service_tier = self.config.service_tier.clone().or(rules.service_tier);
3587    }
3588
3589    /// BP-13 — send `req`, walking [`Config::model_fallback`] when the
3590    /// failure is one another model could plausibly answer.
3591    ///
3592    /// Returns the final outcome plus the hops actually taken, so the
3593    /// caller (which owns `&mut self`) can record each one. Each hop
3594    /// re-applies routing for the new model and strips model-A reasoning
3595    /// artifacts from the request's own message copy before model B sees
3596    /// them — the same dep-8 guarantee [`Self::record_model_change`] gives
3597    /// the live history.
3598    async fn complete_with_fallback(
3599        &self,
3600        req: &mut ChatRequest,
3601        on_delta: &(dyn for<'a> Fn(&'a str) + Send + Sync),
3602    ) -> (Result<(ChatMessage, provider::Usage)>, Vec<FallbackHop>) {
3603        let mut hops = Vec::new();
3604        let mut result = self.provider.complete(req, on_delta).await;
3605        for next in &self.config.model_fallback {
3606            let Err(error) = &result else {
3607                break;
3608            };
3609            if !is_failover_worthy(error) {
3610                break;
3611            }
3612            if next.is_empty() || next == &req.model {
3613                continue;
3614            }
3615            let reason = error.to_string();
3616            let from = std::mem::replace(&mut req.model, next.clone());
3617            reduce::rehydrate::filter_reasoning_artifacts(&mut req.messages);
3618            self.apply_routing(req);
3619            hops.push(FallbackHop {
3620                from,
3621                to: next.clone(),
3622                reason,
3623            });
3624            result = self.provider.complete(req, on_delta).await;
3625        }
3626        (result, hops)
3627    }
3628
3629    /// P4c: every [`crate::model_change::ModelChangeRecord`] this agent has
3630    /// accumulated so far (via [`Self::switch_model`] with `allow_switch`
3631    /// on). Empty when the knob is off or no switch has happened yet.
3632    pub fn model_change_records(&self) -> &[crate::model_change::ModelChangeRecord] {
3633        &self.model_change_log
3634    }
3635
3636    /// P4c: persist this agent's accumulated model-change log to `store`
3637    /// under `name` — the [`crate::model_change::ModelChangeRecord`] analog
3638    /// of [`Self::save_usage_log`].
3639    pub fn save_model_change_log(
3640        &self,
3641        store: &crate::store::SessionStore,
3642        name: &str,
3643    ) -> Result<()> {
3644        store.save_model_change_log(name, &self.model_change_log)
3645    }
3646
3647    /// P4e (§1.6/§3.1 `core.session.git_metadata`, catalog:331): this
3648    /// agent's captured git provenance, if [`Config::session_git_metadata`]
3649    /// was on at construction and the best-effort probe found a repo.
3650    pub fn git_metadata(&self) -> Option<&crate::git_metadata::GitMetadataRecord> {
3651        self.git_metadata.as_ref()
3652    }
3653
3654    /// P4e: persist this agent's captured git metadata to `store` under
3655    /// `name` — a thin wrapper over
3656    /// [`crate::store::SessionStore::save_git_metadata`], the
3657    /// [`crate::git_metadata::GitMetadataRecord`] analog of
3658    /// [`Self::save_usage_log`]. A no-op (`Ok(())`, nothing written) when
3659    /// [`Self::git_metadata`] is `None`.
3660    pub fn save_git_metadata(&self, store: &crate::store::SessionStore, name: &str) -> Result<()> {
3661        match &self.git_metadata {
3662            Some(record) => store.save_git_metadata(name, record),
3663            None => Ok(()),
3664        }
3665    }
3666
3667    /// P4e DEFECT-FIX (independent Fable-5 review of P4e: `core.session.persist`
3668    /// had a `Config` field and CLI plumbing at `ConfigProfile` → `Config` but
3669    /// no consumer at all): whether a CLI caller's session-store save sites
3670    /// (`persist_session`, `persist_full_view`) should actually write to
3671    /// disk. `true` (the default) is byte-identical to pre-fix behavior —
3672    /// every session persists. `false` makes a session ephemeral: it runs
3673    /// exactly as before, but no `<name>.jsonl`/sidecar family is ever
3674    /// written for it. A plain getter, same posture as [`Self::model`] —
3675    /// this crate itself never reads or enforces it; the CLI's save sites do.
3676    pub fn session_persist(&self) -> bool {
3677        self.config.session_persist
3678    }
3679
3680    /// P4e DEFECT-FIX (independent Fable-5 review of P4e: `core.session.name`
3681    /// had a `Config` field and CLI plumbing but no consumer): the
3682    /// caller-configured session name, if `[core.session] name` was set.
3683    /// `None` (the default) leaves session naming exactly as before —
3684    /// `mint_session_name`'s auto-generated `<tag>-<adjective>-<noun>` shape.
3685    /// A plain getter, same posture as [`Self::session_persist`].
3686    pub fn session_name(&self) -> Option<&str> {
3687        self.config.session_name.as_deref()
3688    }
3689
3690    /// PARITY-18 D3 — whether this agent has actually issued at least one
3691    /// live request to its [`Provider`] so far (set the instant
3692    /// `Self::run_loop` reaches its real send site, regardless of whether
3693    /// that call then succeeds or fails). Callers should report
3694    /// "request sent" from THIS, never from having merely passed the
3695    /// context guard or having called [`Self::send`] — either of those can
3696    /// happen with zero requests actually issued (a guard refusal, an
3697    /// interactive session quit before any turn completes).
3698    pub fn request_issued(&self) -> bool {
3699        self.requests_issued
3700    }
3701
3702    /// P5-2 (§2.2 C2): whether this agent currently considers its
3703    /// [`CachePlan::ImportedPrefix`] cache entry warm — mirrors
3704    /// [`Self::request_issued`]'s read-only-observability precedent, so a
3705    /// caller (or a test) can confirm [`Self::register_tool`]'s C2
3706    /// invalidation actually took effect without reaching into private
3707    /// state.
3708    pub fn cache_established(&self) -> bool {
3709        self.cache_established
3710    }
3711
3712    /// B7: length of the imported-prefix protected by [`CachePlan::ImportedPrefix`]
3713    /// (this agent's own system message plus every message of a
3714    /// previously-imported session), set by [`Self::load_session`]. `None`
3715    /// until a session has been loaded.
3716    pub fn imported_prefix_len(&self) -> Option<usize> {
3717        self.imported_prefix_len
3718    }
3719
3720    /// Replace this agent's accumulating reduction log (C4: `/expand`/`/reduce`
3721    /// mutate the log directly via `reduce::invert_one`/`reduce::project_messages`
3722    /// and must feed the result back here so the *next* request build or
3723    /// persist sees the updated state instead of silently recomputing from an
3724    /// empty log). Also lets a caller (`resume_cmd`, C1) seed the log with the
3725    /// initial projection it already computed for the entry banner, so
3726    /// `reduction_log()` reflects reality even before this agent's first
3727    /// `send()` (which is otherwise the only place `build_request_messages`
3728    /// populates it).
3729    pub fn set_reduction_log(&mut self, log: ReductionLog) {
3730        self.reduction_log = log;
3731    }
3732
3733    /// Replace the conversation with a loaded session, keeping this agent's own
3734    /// system prompt at the front. The session's own system/developer turns are
3735    /// preserved after it for context.
3736    pub fn load_session(&mut self, session: Session) {
3737        let system = self.history.first().cloned();
3738        self.history.clear();
3739        if let Some(sys) = system {
3740            self.history.push(sys);
3741        }
3742        self.history.extend(session.messages);
3743        // B7: the whole of `history` at this point — this agent's own system
3744        // message plus every imported message — is the stable prefix a
3745        // resumed session resends byte-identically every turn.
3746        self.imported_prefix_len = Some(self.history.len());
3747        // UX-26 (B7-warn): a freshly loaded prefix has no established cache
3748        // entry of THIS agent's own making yet (even if this agent was
3749        // resumed once before — that earlier prefix is gone). Seed the
3750        // activity clock from the loaded session's own last message
3751        // timestamp (walking backward past any trailing message that
3752        // carries none), so a session that's been sitting idle since
3753        // Claude Code/Codex/a prior supercode run last touched it is
3754        // correctly treated as already-cold on its very first turn here —
3755        // `None` (no timestamp anywhere in the loaded messages) leaves the
3756        // TTL check disarmed rather than guessing.
3757        self.cache_established = false;
3758        self.last_cache_activity_ms = self
3759            .history
3760            .iter()
3761            .rev()
3762            .find_map(|m| m.metadata.get("timestamp"))
3763            .and_then(|ts| supercode_interchange::sidecar::rfc3339_to_ms(ts));
3764    }
3765
3766    /// Append `msg` to the sidecar recorder (A3), if one is installed — a
3767    /// no-op, at zero cost, when `recorder` is `None` (today's behavior).
3768    fn record(&mut self, msg: &ChatMessage) -> Result<()> {
3769        if let Some(recorder) = self.recorder.as_mut() {
3770            recorder.append(msg)?;
3771        }
3772        // BP-8 (catalog:150): the append-only half — written and FLUSHED
3773        // here, at the moment the message exists, not at the end of the
3774        // turn. A journal failure is logged, never fatal: durability
3775        // bookkeeping must not be able to fail a turn.
3776        if let Some(journal) = &self.journal {
3777            let mut guard = journal
3778                .lock()
3779                .unwrap_or_else(std::sync::PoisonError::into_inner);
3780            if let Err(error) = guard.append_message(msg) {
3781                tracing::warn!("failed to journal a message: {error}");
3782            }
3783        }
3784        // BP-8 (catalog:151): the same message becomes a tree node, so the
3785        // tree and the linear history never disagree about what was said.
3786        if let Some(tree) = self.session_tree.as_mut() {
3787            tree.append_message(msg.clone(), now_ms());
3788        }
3789        Ok(())
3790    }
3791
3792    /// Persist the live conversation to `path` as JSONL (one [`ChatMessage`]
3793    /// per line) so the session can be resumed later — supercode's own sessions
3794    /// become first-class, resumable artifacts.
3795    pub fn save_transcript(&self, path: impl AsRef<std::path::Path>) -> Result<()> {
3796        let mut out = String::new();
3797        for m in &self.history {
3798            out.push_str(&serde_json::to_string(m).map_err(Error::Decode)?);
3799            out.push('\n');
3800        }
3801        std::fs::write(path, out)?;
3802        Ok(())
3803    }
3804
3805    /// Restore a conversation previously written with [`Self::save_transcript`],
3806    /// replacing the current history.
3807    pub fn load_transcript(&mut self, path: impl AsRef<std::path::Path>) -> Result<()> {
3808        let text = std::fs::read_to_string(path)?;
3809        let mut history = Vec::new();
3810        for line in text.lines().map(str::trim).filter(|l| !l.is_empty()) {
3811            history.push(serde_json::from_str::<ChatMessage>(line).map_err(Error::Decode)?);
3812        }
3813        self.history = history;
3814        Ok(())
3815    }
3816
3817    /// Take a checkpoint of the current conversation position. Pass it to
3818    /// [`Self::rewind_to`] to discard everything sent since (the rewind/undo
3819    /// analog of `fork`/checkpoint).
3820    pub fn checkpoint(&self) -> usize {
3821        self.history.len()
3822    }
3823
3824    /// Rewind the conversation to a [`Self::checkpoint`], discarding later turns.
3825    pub fn rewind_to(&mut self, checkpoint: usize) {
3826        self.history.truncate(checkpoint.min(self.history.len()));
3827    }
3828
3829    /// Send a message with file inputs attached — the `--file` / `-i` analog.
3830    /// Each file's contents are injected into the prompt: UTF-8 text inline,
3831    /// binary (e.g. images) noted with a size marker. (Native image *vision*
3832    /// would additionally require multimodal content parts.)
3833    pub async fn send_with_files(
3834        &mut self,
3835        text: impl Into<String>,
3836        files: &[std::path::PathBuf],
3837    ) -> Result<String> {
3838        let mut prompt = text.into();
3839        for path in files {
3840            let block = match std::fs::read(path) {
3841                Ok(bytes) => match String::from_utf8(bytes.clone()) {
3842                    Ok(s) => format!("\n\n[file: {}]\n{}", path.display(), s),
3843                    Err(_) => format!(
3844                        "\n\n[file: {} — {} bytes, binary content omitted]",
3845                        path.display(),
3846                        bytes.len()
3847                    ),
3848                },
3849                Err(e) => format!("\n\n[file: {} — could not read: {e}]", path.display()),
3850            };
3851            prompt.push_str(&block);
3852        }
3853        let expanded = self.expand_prompt_async(&prompt).await;
3854        let msg = ChatMessage::user(expanded);
3855        self.guard_candidate_message(&msg)?;
3856        self.record(&msg)?;
3857        self.history.push(msg);
3858        self.run_loop().await
3859    }
3860
3861    /// Send a message with image inputs to a vision model — the `-i/--image`
3862    /// analog. `image_urls` may be `https://…` links or `data:image/…;base64,…`
3863    /// URLs; they're attached as multimodal `image_url` content parts.
3864    pub async fn send_with_images(
3865        &mut self,
3866        text: impl Into<String>,
3867        image_urls: &[String],
3868    ) -> Result<String> {
3869        let expanded = self.expand_prompt_async(&text.into()).await;
3870        let msg = ChatMessage::user_with_images(expanded, image_urls);
3871        self.guard_candidate_message(&msg)?;
3872        self.record(&msg)?;
3873        self.history.push(msg);
3874        self.run_loop().await
3875    }
3876
3877    /// Expand a `/<name> <args>` slash command against the registered prompt
3878    /// templates (`{args}` is replaced with the trailing text). Non-matching
3879    /// input is returned unchanged.
3880    /// BP-6 additionally resolves SKILL.md invocations here, after the
3881    /// template table misses: `/skill:name args` (pi§2 "Skill commands"),
3882    /// `/name args` when the config follows Claude Code (cc§7: "a `SKILL.md`
3883    /// in a directory = a `/name` command"), and `$slug` mentions (cx§7).
3884    /// `$ARGUMENTS` in the body is replaced with the trailing text.
3885    pub fn expand_prompt(&self, input: &str) -> String {
3886        // BP-5 (catalog D2 "@-file mentions / attachments"): `@path`
3887        // expansion happens FIRST, so a mention works in a bare message, in
3888        // a slash-command's arguments, and in the text a `$slug` mention
3889        // appends to — one rule, every prompt shape.
3890        let input = &self.expand_file_mentions(input);
3891        let trimmed = input.trim_start();
3892        let Some(rest) = trimmed.strip_prefix('/') else {
3893            return self.expand_skill_mentions(input);
3894        };
3895        let (name, args) = match rest.split_once(char::is_whitespace) {
3896            Some((n, a)) => (n, a.trim()),
3897            None => (rest, ""),
3898        };
3899        match self.config.prompts.get(name) {
3900            Some(template) => template.replace("{args}", args),
3901            None => match self.expand_skill_command(name, args) {
3902                Some(expanded) => expanded,
3903                None => self.expand_skill_mentions(input),
3904            },
3905        }
3906    }
3907
3908    /// BP-5 (catalog D2 "@-file mentions / attachments"; cc§2 "`@` in the
3909    /// prompt triggers file-path autocomplete and injects file context …
3910    /// Read deny rules best-effort apply to `@file` mentions"; cx§2
3911    /// "`@`-mentions (files)"): replace each `@path` token in `input` with
3912    /// that file's contents.
3913    ///
3914    /// **Deny-rule aware, through the one permissions engine.** Each
3915    /// mention is resolved with
3916    /// [`crate::permissions::evaluate_path_safe`] — the same
3917    /// traversal/symlink-resolving check a `read_file` tool call goes
3918    /// through — against this config's own rules and protected-path floor.
3919    /// Anything short of `Allow` inlines the refusal instead of the file, so
3920    /// `@.env` under a preset whose protected paths cover it says so rather
3921    /// than quietly leaking it.
3922    ///
3923    /// A token that names nothing readable is left exactly as the user typed
3924    /// it: an email address, a decorator, or a `@`-prefixed word in prose is
3925    /// not a file mention, and must survive untouched.
3926    /// Off by default (`[core.file_mentions]`).
3927    fn expand_file_mentions(&self, input: &str) -> String {
3928        if !self.config.file_mentions || !input.contains('@') {
3929            return input.to_string();
3930        }
3931        let mut attachments = String::new();
3932        let mut seen: Vec<String> = Vec::new();
3933        for token in input.split_whitespace() {
3934            let Some(rel) = token.strip_prefix('@') else {
3935                continue;
3936            };
3937            let rel = rel.trim_end_matches([',', ';', ':', '.', ')', ']', '"', '\'']);
3938            if rel.is_empty() || seen.iter().any(|s| s == rel) {
3939                continue;
3940            }
3941            let path = if std::path::Path::new(rel).is_absolute() {
3942                std::path::PathBuf::from(rel)
3943            } else {
3944                self.config.cwd.join(rel)
3945            };
3946            if !path.is_file() {
3947                continue;
3948            }
3949            seen.push(rel.to_string());
3950            attachments.push_str(&self.render_mention(rel, &path));
3951            if seen.len() >= MAX_FILE_MENTIONS_PER_MESSAGE {
3952                break;
3953            }
3954        }
3955        if attachments.is_empty() {
3956            return input.to_string();
3957        }
3958        format!("{input}{attachments}")
3959    }
3960
3961    /// One mention's block: the permission verdict first, then the bytes.
3962    /// Text is inlined; a binary file is named with its size, the same
3963    /// shape [`Self::send_with_files`] already uses for an explicit
3964    /// attachment, so a mention and a `--file` read the same way.
3965    fn render_mention(&self, shown: &str, path: &std::path::Path) -> String {
3966        use crate::permissions::{Decision, PathKind};
3967        let rules = crate::permissions::rules_for_config(&self.config);
3968        // BP-10's multi-root form: a mention is checked against every
3969        // granted root (cwd + `additional_dirs`), folded to the strictest —
3970        // the same call the tool-dispatch gate makes for a `read_file`
3971        // path, so a mention can never reach a file a read could not.
3972        let mut roots = vec![self.config.cwd.clone()];
3973        roots.extend(self.config.additional_dirs.iter().cloned());
3974        let decision = crate::permissions::evaluate_path_safe_roots(
3975            &rules,
3976            PathKind::Read,
3977            &roots,
3978            &path.to_string_lossy(),
3979            Decision::Allow,
3980        );
3981        if decision != Decision::Allow {
3982            return format!(
3983                "\n\n[file: {shown} — not attached; the permission rules for this session \
3984                 resolve reading it to {decision:?}]"
3985            );
3986        }
3987        match std::fs::read(path) {
3988            Ok(bytes) => match String::from_utf8(bytes) {
3989                Ok(text) => {
3990                    let mut text = text;
3991                    if text.len() > MAX_FILE_MENTION_BYTES {
3992                        let mut cut = MAX_FILE_MENTION_BYTES;
3993                        while cut > 0 && !text.is_char_boundary(cut) {
3994                            cut -= 1;
3995                        }
3996                        text.truncate(cut);
3997                        text.push_str("\n[file truncated]");
3998                    }
3999                    format!("\n\n[file: {shown}]\n{text}")
4000                }
4001                Err(e) => format!(
4002                    "\n\n[file: {shown} — {} bytes, binary content omitted]",
4003                    e.into_bytes().len()
4004                ),
4005            },
4006            Err(e) => format!("\n\n[file: {shown} — could not read: {e}]"),
4007        }
4008    }
4009
4010    /// The SKILL.md packages this agent discovered (frontmatter only) — the
4011    /// exact set its prompt index lists and its `skill` tool can load.
4012    pub fn skills(&self) -> &[crate::skills::LoopSkill] {
4013        &self.skills
4014    }
4015
4016    /// BP-6: resolve a slash command against the discovered skills.
4017    ///
4018    /// `/skill:<name>` is pi's own form and is accepted under every config
4019    /// (it can never collide with a template name, which cannot contain a
4020    /// colon-prefixed `skill` segment by construction). The BARE `/<name>`
4021    /// form is Claude Code's — there, a skill IS a slash command — so it is
4022    /// honored only when the config reads Claude Code's roots; under
4023    /// `cx-parity`, where Codex has no skill slash commands, `/deploy` stays
4024    /// the literal text the user typed.
4025    fn expand_skill_command(&self, name: &str, args: &str) -> Option<String> {
4026        if self.skills.is_empty() {
4027            return None;
4028        }
4029        let bare = match name.strip_prefix("skill:") {
4030            Some(rest) => rest,
4031            None if self.config.skills_harness.as_deref()
4032                == Some(crate::HarnessId::CLAUDE_CODE) =>
4033            {
4034                name
4035            }
4036            None => return None,
4037        };
4038        let skill = self.find_skill(bare)?;
4039        skill
4040            .body_with_shell(args, &self.shell_injection)
4041            .ok()
4042            .map(|body| crate::skills::render_skill(skill, &body))
4043    }
4044
4045    /// BP-6: `$slug` mentions (cx§7 `TOOL_MENTION_SIGIL = '$'`) and — only
4046    /// under `[core.skills] implicit_match` — a description match.
4047    ///
4048    /// The user's own text is never replaced: a loaded body is APPENDED, the
4049    /// way Codex splices a skill into the turn. Mentions are only honored
4050    /// for a config that reads Codex's roots; `$WORD` is ordinary shell text
4051    /// everywhere else.
4052    fn expand_skill_mentions(&self, input: &str) -> String {
4053        if self.skills.is_empty() {
4054            return input.to_string();
4055        }
4056        let mut loaded: Vec<String> = Vec::new();
4057        let mut names: Vec<String> = Vec::new();
4058        if self.config.skills_harness.as_deref() == Some(crate::HarnessId::CODEX) {
4059            for token in input.split_whitespace() {
4060                let Some(slug) = token.strip_prefix('$') else {
4061                    continue;
4062                };
4063                let slug =
4064                    slug.trim_matches(|c: char| !c.is_alphanumeric() && c != '-' && c != ':');
4065                if slug.is_empty() {
4066                    continue;
4067                }
4068                let Some(skill) = self.find_skill(slug) else {
4069                    continue;
4070                };
4071                if names.contains(&skill.name) || loaded.len() >= MAX_SKILL_LOADS_PER_MESSAGE {
4072                    continue;
4073                }
4074                if let Ok(body) = skill.body_with_shell("", &self.shell_injection) {
4075                    names.push(skill.name.clone());
4076                    loaded.push(crate::skills::render_skill(skill, &body));
4077                }
4078            }
4079        }
4080        if loaded.is_empty() && self.config.skills_implicit_match {
4081            if let Some(skill) = crate::skills::implicit_skill_match(&self.skills, input) {
4082                if let Ok(body) = skill.body_with_shell("", &self.shell_injection) {
4083                    loaded.push(crate::skills::render_skill(skill, &body));
4084                }
4085            }
4086        }
4087        if loaded.is_empty() {
4088            return input.to_string();
4089        }
4090        format!("{input}\n\n{}", loaded.join("\n\n"))
4091    }
4092
4093    /// Resolve one invocation name against the discovered set — the same
4094    /// resolver the `skill` tool uses, so every door agrees on what a name
4095    /// means.
4096    fn find_skill(&self, name: &str) -> Option<&crate::skills::LoopSkill> {
4097        crate::skills::find_skill(&self.skills, name)
4098    }
4099
4100    /// P5-2 (§2 module 15 D7 row 4 "prompts-as-commands"): like
4101    /// [`Self::expand_prompt`], but also consults MCP-server-sourced
4102    /// prompts registered via [`Self::register_mcp_prompt`] when the local
4103    /// `Config::prompts` table has no match — a live `prompts/get`
4104    /// round-trip, which is why this is async and [`Self::expand_prompt`]
4105    /// itself stays synchronous (its public sync signature is unchanged,
4106    /// for every existing caller that doesn't need MCP prompts).
4107    ///
4108    /// **Argument mapping (a scope decision, not a protocol requirement —
4109    /// the MCP spec leaves "how does free CLI text become named prompt
4110    /// arguments" to the client):** a prompt with zero or one declared
4111    /// arguments gets the whole trailing text (empty string if the prompt
4112    /// takes no arguments and none was given); a prompt with two or more
4113    /// declared arguments expects `key=value` pairs, whitespace-separated
4114    /// (`/mcp__server__prompt lang=rust topic=async`) — an unparseable pair
4115    /// (no `=`) is simply skipped, never a hard error (matches this
4116    /// method's "non-matching input passes through" fail-open posture for
4117    /// the LOCAL-prompt case above).
4118    pub async fn expand_prompt_async(&self, input: &str) -> String {
4119        let local = self.expand_prompt(input);
4120        if local != input {
4121            return local; // a local `Config::prompts` template matched
4122        }
4123        let trimmed = input.trim_start();
4124        let Some(rest) = trimmed.strip_prefix('/') else {
4125            return input.to_string();
4126        };
4127        let (name, args) = match rest.split_once(char::is_whitespace) {
4128            Some((n, a)) => (n, a.trim()),
4129            None => (rest, ""),
4130        };
4131        let Some(source) = self.mcp_prompts.get(name) else {
4132            return input.to_string();
4133        };
4134        let arg_map = match source.arg_names() {
4135            [] => std::collections::BTreeMap::new(),
4136            [single] => {
4137                let mut m = std::collections::BTreeMap::new();
4138                if !args.is_empty() {
4139                    m.insert(single.clone(), args.to_string());
4140                }
4141                m
4142            }
4143            _ => args
4144                .split_whitespace()
4145                .filter_map(|pair| pair.split_once('='))
4146                .map(|(k, v)| (k.to_string(), v.to_string()))
4147                .collect(),
4148        };
4149        match source.render(arg_map).await {
4150            Ok(rendered) => rendered,
4151            Err(e) => format!("Error: mcp prompt `{name}` failed: {e}"),
4152        }
4153    }
4154
4155    /// P5-2 (§2 module 15 D7 row 4): register an MCP server's prompt as a
4156    /// slash-command source — `command_name` MUST already be the
4157    /// namespaced `mcp__<server>__<prompt>` form
4158    /// ([`crate::mcp::McpServerHandle::prompts`] produces exactly that
4159    /// shape); this method does not re-namespace or validate it, so a
4160    /// caller that hands it a bare name defeats the collision protection
4161    /// [`crate::mcp::McpPromptSource`]'s doc comment describes. Overwrites
4162    /// any prior registration under the same command name (re-attaching
4163    /// the same server replaces its own earlier prompt list; this can
4164    /// never touch a NON-`mcp__`-prefixed key, i.e. never a local
4165    /// `Config::prompts` entry).
4166    pub fn register_mcp_prompt(
4167        &mut self,
4168        command_name: impl Into<String>,
4169        source: impl crate::sdk::SdkPromptSource + 'static,
4170    ) {
4171        self.mcp_prompts
4172            .insert(command_name.into(), Box::new(source));
4173    }
4174
4175    /// P5-2 (§2 module 15 D7 row 5 "instructions"): fold an MCP server's
4176    /// `initialize`-time instructions (or any other free-text note) into
4177    /// this agent's system message — the context-assembly site every other
4178    /// `core.*`/`capabilities.*` prompt-section append already uses
4179    /// (`Self::with_parts`), except this one fires AFTER construction
4180    /// (attaching MCP servers happens once the agent already exists — see
4181    /// `crates/cli/src/main.rs`'s `attach_mcp`). A no-op if `history` is
4182    /// somehow empty or its first message isn't a system message (never
4183    /// true for an `Agent` built via `Self::new`/`Self::with_parts`, but
4184    /// checked rather than assumed).
4185    pub fn append_system_note(&mut self, text: &str) {
4186        if let Some(system) = self.history.first_mut() {
4187            if system.role == Role::System {
4188                system
4189                    .content
4190                    .get_or_insert_with(String::new)
4191                    .push_str(text);
4192            }
4193        }
4194    }
4195
4196    /// BP-4 (catalog:90, cx§2 `<environment_context>` "re-emitted on
4197    /// change"): re-derive the `# Environment` block and, if anything in it
4198    /// moved — cwd, the approval/sandbox policy, the git branch or its
4199    /// dirty state, the date — replace the stale copy in the system message
4200    /// with the fresh one. Returns whether the block changed.
4201    ///
4202    /// A no-op (and free — no git subprocess) when `core.env_context` is
4203    /// off, which is the default and every non-parity config. Replacing in
4204    /// place rather than appending a second block is deliberate: two
4205    /// `# Environment` sections disagreeing about cwd is worse context than
4206    /// one stale one, and the system message is re-sent on every request,
4207    /// so the rewrite IS the re-emission the model sees.
4208    pub fn refresh_env_context(&mut self) -> bool {
4209        if !self.config.env_context {
4210            return false;
4211        }
4212        let fresh = env_context_block(&self.config);
4213        let Some(stale) = self.env_context_live.clone() else {
4214            // Nothing was spliced at construction (e.g. `with_provider_arc`);
4215            // splice it now rather than silently never emitting one.
4216            self.append_system_note(&fresh);
4217            self.env_context_live = Some(fresh);
4218            return true;
4219        };
4220        if stale == fresh {
4221            return false;
4222        }
4223        if let Some(system) = self.history.first_mut() {
4224            if system.role == Role::System {
4225                if let Some(content) = system.content.as_mut() {
4226                    if let Some(at) = content.find(&stale) {
4227                        content.replace_range(at..at + stale.len(), &fresh);
4228                        self.env_context_live = Some(fresh);
4229                        return true;
4230                    }
4231                }
4232            }
4233        }
4234        false
4235    }
4236
4237    /// BP-5 (catalog D2 "Per-model-family base-prompt selection"): re-select
4238    /// the base prompt for the model now in force and replace the stale one
4239    /// in place. Returns whether the system message changed.
4240    ///
4241    /// A no-op — not even a string search — when the selection is unchanged,
4242    /// which is every config that sets no `base_prompts` table.
4243    fn refresh_base_prompt(&mut self) -> bool {
4244        let fresh = base_prompt_for_config(&self.config);
4245        if fresh == self.base_prompt_live {
4246            return false;
4247        }
4248        let stale = std::mem::replace(&mut self.base_prompt_live, fresh.clone());
4249        if stale.is_empty() {
4250            return false;
4251        }
4252        if let Some(system) = self.history.first_mut() {
4253            if system.role == Role::System {
4254                if let Some(content) = system.content.as_mut() {
4255                    if let Some(at) = content.find(&stale) {
4256                        content.replace_range(at..at + stale.len(), &fresh);
4257                        return true;
4258                    }
4259                }
4260            }
4261        }
4262        false
4263    }
4264
4265    /// BP-5: assemble the request from an already-built message list and
4266    /// tool-schema list. The ONE place a [`ChatRequest`] is constructed from
4267    /// this agent's config, so the request `Self::run_loop` issues and the
4268    /// request [`Self::model_input`] renders cannot drift apart.
4269    fn chat_request(&self, messages: Vec<ChatMessage>, tools: Vec<ToolSchema>) -> ChatRequest {
4270        let mut req = ChatRequest {
4271            model: self.config.model.clone(),
4272            messages,
4273            tools,
4274            temperature: self.config.temperature,
4275            max_tokens: self.config.max_tokens,
4276            effort: self.config.effort.clone(),
4277            response_format: self.config.response_format.clone(),
4278            service_tier: None,
4279            thinking_budget: None,
4280            extra_body: self.config.extra_body.clone(),
4281        };
4282        // BP-13 (catalog Domain 9): the per-request routing decisions —
4283        // effort LEVEL, thinking-token BUDGET and service TIER — all come
4284        // out of `Config::model_routing` keyed by the model this request is
4285        // actually going to. Applied HERE so `model_input`'s rendering and
4286        // the loop's own send can never disagree about what would be sent,
4287        // and so a mid-session switch re-decides all three for the new
4288        // model on the next pass.
4289        self.apply_routing(&mut req);
4290        req
4291    }
4292
4293    /// BP-5 (catalog D2 "Prompt-input debugging": *render the exact
4294    /// model-visible input for inspection*; cx§2 `codex debug prompt-input`,
4295    /// which "renders the exact model-visible input list as JSON"): the
4296    /// request this agent would send next.
4297    ///
4298    /// Built by the SAME two calls the loop makes
4299    /// ([`Self::build_request_messages`], [`Self::tool_schemas`]) and
4300    /// assembled by the SAME [`Self::chat_request`] — it is the real
4301    /// request, not a reconstruction of one. `&mut self` because
4302    /// `build_request_messages` is: rendering the input is exactly as
4303    /// stateful as building it for a send.
4304    pub fn model_input(&mut self) -> ChatRequest {
4305        let tools = self.tool_schemas();
4306        let messages = self.build_request_messages();
4307        self.chat_request(messages, tools)
4308    }
4309
4310    /// BP-5: [`Self::model_input`] for a turn that has not been sent —
4311    /// `prompt` is expanded exactly as [`Self::send`] would expand it
4312    /// (slash templates, skills, `@path` mentions, MCP prompts) and appended
4313    /// to the conversation IN MEMORY, then the request is rendered.
4314    ///
4315    /// Deliberately not recorded: this door inspects an input, it does not
4316    /// take a turn. Nothing is written to the session store, no journal
4317    /// entry is made, and no request is issued.
4318    pub async fn model_input_for(&mut self, prompt: &str) -> ChatRequest {
4319        let expanded = self.expand_prompt_async(prompt).await;
4320        self.history.push(ChatMessage::user(expanded));
4321        self.model_input()
4322    }
4323
4324    /// BP-5: a [`ChatRequest`] as the JSON a human (or `jq`) inspects — the
4325    /// system prompt, every message in order, and every advertised tool
4326    /// schema, plus the sampling controls that travel with them.
4327    pub fn render_model_input(req: &ChatRequest) -> serde_json::Value {
4328        serde_json::json!({
4329            "model": req.model,
4330            "temperature": req.temperature,
4331            "max_tokens": req.max_tokens,
4332            "effort": req.effort,
4333            "response_format": req.response_format,
4334            // Serialized through `ChatMessage`'s OWN wire serializer and
4335            // `ToolSchema`'s own — i.e. the exact bytes the provider is
4336            // handed, not a second rendering of them.
4337            "messages": serde_json::to_value(&req.messages).unwrap_or(serde_json::Value::Null),
4338            "tools": serde_json::to_value(&req.tools).unwrap_or(serde_json::Value::Null),
4339        })
4340    }
4341
4342    /// BP-5 (catalog D2 "Per-model-family base-prompt selection"): the base
4343    /// system prompt currently in force for this agent's model.
4344    pub fn base_prompt(&self) -> &str {
4345        &self.base_prompt_live
4346    }
4347
4348    /// BP-4 (catalog:91 "Synthetic context-injection blocks"): splice one
4349    /// named ambient block into the live context — the seam a hook's
4350    /// `additionalContext`, a frontend nudge or an orchestrator's brief
4351    /// enters through, mid-session, after construction.
4352    ///
4353    /// Requires `core.context_injections` (returns `false` otherwise): the
4354    /// gate governs the whole registry, not just its startup half. The
4355    /// block is appended to the system message and remembered, so it is
4356    /// carried by every later request and re-rendered by
4357    /// [`crate::context_injection::assemble`] wherever the prompt is
4358    /// rebuilt.
4359    pub fn inject_context_block(
4360        &mut self,
4361        name: impl Into<String>,
4362        content: impl Into<String>,
4363    ) -> bool {
4364        if !self.config.context_injections {
4365            return false;
4366        }
4367        let block = crate::config::ContextInjectionBlock::new(name, content);
4368        let rendered = crate::context_injection::render(std::slice::from_ref(&block));
4369        self.spliced_context_blocks.push(block);
4370        self.append_system_note(&rendered);
4371        true
4372    }
4373
4374    /// The blocks spliced in since construction — see
4375    /// [`Self::inject_context_block`].
4376    pub fn spliced_context_blocks(&self) -> &[crate::config::ContextInjectionBlock] {
4377        &self.spliced_context_blocks
4378    }
4379
4380    /// Compact the conversation if it has grown past the configured
4381    /// threshold.
4382    ///
4383    /// **Re-founded (A10):** with a [`ReductionPolicy`] installed
4384    /// ([`Self::set_reduction_policy`]), this no longer touches `self.history`
4385    /// at all. It derives `policy.clear_turns_older_than` from
4386    /// `compact_after_messages` so the *next* projected request view
4387    /// (`reduce::project_messages`, built in `Self::run_loop`) collapses the
4388    /// old turns into one reversible `TurnsCleared` stub instead —
4389    /// `history()` and the sidecar keep every message forever; only the view
4390    /// shrinks. Returns whether the live (unreduced) history currently
4391    /// exceeds the threshold, i.e. whether a clearing will actually be
4392    /// visible in the next projected view.
4393    ///
4394    /// **Legacy path (no policy) — LOSSY, kept only for byte-identical
4395    /// backward compatibility (D6):** destructively rewrites `self.history`,
4396    /// permanently discarding the dropped middle turns (replaced by a single
4397    /// non-reversible summary marker that becomes their SOLE remaining copy —
4398    /// exactly the lossy compaction this reduction layer differentiates
4399    /// against). Once a sidecar/recorder or a [`ReductionPolicy`] is in play,
4400    /// prefer installing a policy so this method takes the re-founded path
4401    /// above instead.
4402    pub fn maybe_compact(&mut self) -> bool {
4403        // P4e (§1.5/§3.1 `core.compaction.enabled`, "no master gate exists
4404        // yet"): checked FIRST, before either trigger — `false` disables
4405        // every auto-compaction trigger unconditionally (message-count AND
4406        // pressure), composing with them rather than replacing their own
4407        // logic. `true` (the default, matching today's pre-P4e behavior,
4408        // where nothing ever gated compaction) falls straight through to
4409        // the existing trigger checks below, unchanged.
4410        if !self.config.compaction_enabled {
4411            return false;
4412        }
4413        let threshold = self.config.compact_after_messages;
4414        // P4b (§1.5/§3.1 `core.compaction.reserve_tokens`, pi§2 shape): a
4415        // SECOND, independent trigger — context-window pressure — alongside
4416        // (not instead of) the message-count one above. `None` (the
4417        // default) is byte-identical to today's message-count-only
4418        // behavior; this whole block is a no-op then.
4419        let message_trigger = threshold.is_some_and(|t| self.history.len() > t);
4420        let pressure_trigger = self.compaction_pressure_triggered();
4421        if threshold.is_none() && self.config.compaction_reserve_tokens.is_none() {
4422            return false;
4423        }
4424        if !message_trigger && !pressure_trigger {
4425            return false;
4426        }
4427        if let Some(policy) = self.reduction_policy.as_mut() {
4428            if let Some(t) = threshold {
4429                policy.clear_turns_older_than = Some(t);
4430            }
4431            // P4b scope note: the token-PRESSURE trigger's "how much to
4432            // clear" derivation (below, for the legacy in-place path) has no
4433            // `ReductionPolicy`/A10 analog yet — that mechanism decides its
4434            // own clearing window once `clear_turns_older_than` is set, so
4435            // pressure firing alone (no message threshold configured) has
4436            // nothing new to hand it in this pass. Report the message-count
4437            // verdict only, matching today's pre-P4b behavior exactly when
4438            // only `threshold` is set.
4439            return message_trigger;
4440        }
4441        // Legacy in-place compaction (no `ReductionPolicy` installed) below.
4442        // `keep_recent`: the message-count trigger's own `threshold / 2`
4443        // shape when it's what fired (or both fired); otherwise (pressure
4444        // fired alone) a token-budget-derived count.
4445        let keep_recent = if message_trigger {
4446            (threshold.unwrap() / 2).max(2)
4447        } else {
4448            self.keep_recent_count_by_tokens()
4449        };
4450        self.compact_in_place(keep_recent, None)
4451    }
4452
4453    /// BP-4 (catalog:98 "Manual compact with focus instructions", cc§2 /
4454    /// cx§2 `/compact [instructions]`): compact NOW, regardless of whether
4455    /// either automatic trigger has fired — the mechanism behind the REPL's
4456    /// `/compact [focus]`.
4457    ///
4458    /// `focus` is this invocation's steering text: it overrides the standing
4459    /// `core.compaction.focus_instructions` for this compaction only, is
4460    /// carried into the SUMMARIZER's input (so the model-written summary
4461    /// preserves what the user asked for), and is stated on the marker. An
4462    /// empty/whitespace `focus` falls back to the configured standing value,
4463    /// which is what a bare `/compact` means.
4464    ///
4465    /// Returns whether anything was compacted (`false` when the history is
4466    /// already at or below the keep-window, or when a [`ReductionPolicy`] is
4467    /// installed — under a policy the reversible A10 path owns clearing, and
4468    /// a manual compact would be the lossy one).
4469    pub fn compact_now(&mut self, focus: Option<&str>) -> bool {
4470        self.compacting_manually = true;
4471        let compacted = self.compact_now_inner(focus);
4472        self.compacting_manually = false;
4473        compacted
4474    }
4475
4476    fn compact_now_inner(&mut self, focus: Option<&str>) -> bool {
4477        if self.reduction_policy.is_some() {
4478            return false;
4479        }
4480        // A manual compact must actually compact. The token budget alone
4481        // (`core.compaction.keep_recent_tokens`, 20k) keeps EVERYTHING on
4482        // any ordinary conversation, which is right for the pressure
4483        // trigger (it fires only when the window is nearly full) and wrong
4484        // for `/compact`, whose whole point is compacting before the
4485        // pressure arrives. So the keep-window is the tighter of the two:
4486        // the token budget, and the message-count trigger's own established
4487        // "keep the most recent half" shape (`maybe_compact`'s
4488        // `threshold / 2`, floor 2).
4489        let keep_recent = self
4490            .keep_recent_count_by_tokens()
4491            .min((self.history.len() / 2).max(2));
4492        let focus = focus.map(str::trim).filter(|f| !f.is_empty());
4493        self.compact_in_place(keep_recent, focus)
4494    }
4495
4496    /// The legacy (no-[`ReductionPolicy`]) in-place compaction both
4497    /// [`Self::maybe_compact`] and [`Self::compact_now`] run: collapse
4498    /// `history[first..cut)` into one marker, keeping the newest
4499    /// `keep_recent` messages.
4500    fn compact_in_place(&mut self, keep_recent: usize, focus_override: Option<&str>) -> bool {
4501        if self.history.len() <= keep_recent {
4502            return false;
4503        }
4504        // Indices: 0 is the system prompt; collapse [first .. len-keep_recent).
4505        // `first` is 1 (only the system prompt is ever auto-preserved) unless
4506        // B7's coordination clamp widens it.
4507        let mut first = 1usize;
4508        // B7 coordination clamp: this legacy (no-`ReductionPolicy`) path
4509        // mutates `self.history` directly, so — unlike the re-founded A10
4510        // path (clamped inside `reduce::project_messages`, threaded from
4511        // `build_request_messages`) — it must clamp itself. Widening `first`
4512        // (not `cut`) is what actually protects the imported prefix: the
4513        // drop range is `[first, cut)`, so raising `cut` alone would only
4514        // drop MORE messages, not fewer. `imported_prefix_len` is already an
4515        // absolute `history` index count (it protects `history[0..len]`), so
4516        // no offset conversion is needed here.
4517        if matches!(self.config.cache_plan, CachePlan::ImportedPrefix) {
4518            if let Some(protected) = self.imported_prefix_len {
4519                first = first.max(protected);
4520            }
4521        }
4522        let mut cut = self.history.len() - keep_recent;
4523        if cut <= first {
4524            return false;
4525        }
4526        // Never begin the kept window on a tool result: its originating
4527        // assistant turn (with the matching `tool_calls`) is about to be
4528        // dropped, which would orphan the tool message and make the replayed
4529        // conversation invalid. Advance past any leading tool results.
4530        while cut < self.history.len() && self.history[cut].role == Role::Tool {
4531            cut += 1;
4532        }
4533        if cut >= self.history.len() {
4534            return false;
4535        }
4536        let dropped = cut - first;
4537        // BP-11: the compaction is decided from here on — the one point both
4538        // the automatic triggers and `/compact` pass through — so this is
4539        // where `pre_compact` observers hear about it.
4540        self.fire_lifecycle(&crate::config::LifecycleEvent::PreCompact {
4541            messages: self.history.len(),
4542            dropped,
4543            manual: focus_override.is_some() || self.compacting_manually,
4544        });
4545        // P4b (§1.5/§3.1 `core.compaction.focus_instructions`, catalog D2
4546        // "no instruction steering" gap): appended to the marker whenever
4547        // set, regardless of which trigger fired. `None` (the default)
4548        // leaves this byte-identical to the pre-P4b marker text.
4549        //
4550        // BP-4: `focus_override` — the per-invocation `/compact <focus>`
4551        // text — wins over the standing config value for THIS compaction,
4552        // which is what "`/compact [instructions]` steers what's preserved"
4553        // means. Neither is required.
4554        let focus: Option<String> = focus_override.map(str::to_string).or_else(|| {
4555            self.config
4556                .compaction_focus_instructions
4557                .clone()
4558                .filter(|f| !f.is_empty())
4559        });
4560        // BP-1 (§1.5/§3.1 `core.compaction.summarize`): the verb the marker
4561        // uses is now the config's to state. `true` (the default, and what
4562        // every preset sets) keeps the historical "summarized" text
4563        // byte-identical; `false` says only what actually happened to the
4564        // span, so a config that turns summarization off does not leave a
4565        // marker claiming a summary exists.
4566        let verb = if self.config.compaction_summarize {
4567            "summarized"
4568        } else {
4569            "cleared"
4570        };
4571        // BP-4 (catalog:107 "LLM summaries of cleared spans", design §1.5:
4572        // obligation 5 is "auto-compaction … + a persisted marker + AN LLM
4573        // SUMMARY OF THE COMPACTED SPAN", knob `[core.compaction] summarize`
4574        // — "the summary side-call depends on a utility model … core falls
4575        // back to the main model"). The side-call is therefore CORE, not a
4576        // reduction-module privilege: when `core.compaction.summarize` is on
4577        // and a summarizer is installed, the span is summarized by the model
4578        // and the marker carries that summary instead of only a count.
4579        //
4580        // Every failure mode degrades to the count-only marker: no
4581        // summarizer installed, an `Err` from the side-call, or an empty
4582        // reply. It never blocks or fails compaction — the same contract
4583        // TR-7's own side-call site keeps.
4584        let summary_body = if self.config.compaction_summarize {
4585            self.summarize_span(first..cut, focus.as_deref())
4586        } else {
4587            None
4588        };
4589        // BP-4 (catalog:99 "Compaction markers persisted in transcript"):
4590        // the marker states where the originals went, which is the whole
4591        // point of a boundary record — a reader must be able to tell a
4592        // reversible compaction from a lossy one without knowing which
4593        // modules were on.
4594        let retention = if self.recorder.is_some() {
4595            "The compacted messages remain in this session's transcript sidecar."
4596        } else {
4597            "No transcript sidecar is attached, so this marker is the only remaining record of them."
4598        };
4599        let mut summary_text = format!(
4600            "[earlier conversation compacted: {dropped} message(s) {verb} to save context]\n{retention}"
4601        );
4602        if let Some(focus) = &focus {
4603            summary_text.push_str(&format!("\n\nFocus: {focus}"));
4604        }
4605        if let Some(body) = &summary_body {
4606            summary_text.push_str(&format!("\n\nSummary of the compacted span:\n{body}"));
4607        }
4608        let summary = ChatMessage::system(summary_text);
4609        // BP-4 (catalog:99): PERSIST the boundary. Before this the legacy
4610        // path rewrote `self.history` and never called `record`, so the
4611        // marker existed only in the live window and a resumed session had
4612        // no on-disk trace that a compaction ever happened. A recorder
4613        // failure is logged, never fatal — losing the boundary record must
4614        // not lose the compaction.
4615        if let Err(error) = self.record(&summary) {
4616            tracing::warn!("failed to persist the compaction marker: {error}");
4617        }
4618        let mut new_history = Vec::with_capacity(first + keep_recent + 2);
4619        new_history.extend(self.history[..first].iter().cloned());
4620        new_history.push(summary);
4621        new_history.extend(self.history.split_off(cut));
4622        self.history = new_history;
4623        self.fire_lifecycle(&crate::config::LifecycleEvent::PostCompact {
4624            messages: self.history.len(),
4625            dropped,
4626        });
4627        // BP-8 (catalog:150): compaction RESHAPES the live view rather than
4628        // appending to it, so the journal's "everything since the last
4629        // checkpoint is unpersisted" accounting has to be re-based here —
4630        // otherwise a crash-recovery replay would re-append messages this
4631        // compaction deliberately set aside. The set-aside messages' own
4632        // bytes stay in the log above, untouched.
4633        self.journal_checkpoint(self.history.len());
4634        true
4635    }
4636
4637    /// BP-4 (catalog:107): run the installed [`reduce::summarize::SpanSummarizer`]
4638    /// over `history[span]`, with `focus` (the `/compact <focus>` text)
4639    /// carried into the summarizer's INPUT so the model-written summary
4640    /// preserves what the user asked to keep.
4641    ///
4642    /// `None` — never an error — whenever no summarizer is installed, the
4643    /// span renders empty, the side-call fails, or it returns nothing. The
4644    /// caller falls back to the count-only marker.
4645    fn summarize_span(&self, span: std::ops::Range<usize>, focus: Option<&str>) -> Option<String> {
4646        let summarizer = self.span_summarizer.as_deref()?;
4647        let mut span_text = String::new();
4648        // The focus rides at the head of the span text (the trait's one
4649        // input) as an explicit, labeled line rather than a silent prompt
4650        // mutation: the fixed prompt's "do not state anything not present
4651        // in the span" still holds, because the focus IS present in it.
4652        if let Some(focus) = focus {
4653            span_text.push_str(&format!("[compaction focus requested: {focus}]\n\n"));
4654        }
4655        for msg in self.history.get(span)? {
4656            let role = match msg.role {
4657                Role::System => "system",
4658                Role::User => "user",
4659                Role::Assistant => "assistant",
4660                Role::Tool => "tool",
4661            };
4662            span_text.push_str(role);
4663            span_text.push_str(": ");
4664            span_text.push_str(msg.content.as_deref().unwrap_or(""));
4665            span_text.push('\n');
4666        }
4667        match summarizer.summarize(&span_text) {
4668            Ok(text) if !text.trim().is_empty() => Some(text.trim().to_string()),
4669            Ok(_) => None,
4670            Err(error) => {
4671                tracing::warn!("compaction span summarizer failed: {error}");
4672                None
4673            }
4674        }
4675    }
4676
4677    /// BP-4 (catalog:106 "Handoff (fresh objective + curated keep-set)",
4678    /// cx§1 `new_context`): reset the live working view to a fresh
4679    /// objective plus a curated keep-set, in-session.
4680    ///
4681    /// The new view is: the system prompt (plus any imported prefix a
4682    /// `CachePlan::ImportedPrefix` config protects — same clamp compaction
4683    /// uses), then a handoff marker stating the objective and what was set
4684    /// aside, then the most recent `keep_recent` messages (`None` = the
4685    /// token-budget-derived count `core.compaction.keep_recent_tokens`
4686    /// already governs, so the keep-set is curated by the same budget the
4687    /// rest of the compaction machinery uses, not by a magic number). The
4688    /// keep-set never begins on a tool result, so no tool message is left
4689    /// orphaned from its originating assistant turn.
4690    ///
4691    /// Returns how many messages were set aside. Like compaction, the
4692    /// marker is PERSISTED through the recorder, so a resumed session can
4693    /// see where the handoff happened; and like compaction, the set-aside
4694    /// messages remain in the transcript sidecar whenever one is attached.
4695    ///
4696    /// Scope note: this is the in-session `new_context` mechanism, NOT
4697    /// `Config::handoff_enabled`'s reversible ReductionLog snapshot (the
4698    /// offline `supercode handoff` projection) — that one is the reduction
4699    /// module's, and stays there.
4700    pub fn new_context(&mut self, objective: &str, keep_recent: Option<usize>) -> usize {
4701        let keep_recent = keep_recent.unwrap_or_else(|| self.keep_recent_count_by_tokens());
4702        let mut first = 1usize;
4703        if matches!(self.config.cache_plan, CachePlan::ImportedPrefix) {
4704            if let Some(protected) = self.imported_prefix_len {
4705                first = first.max(protected);
4706            }
4707        }
4708        let mut cut = self.history.len().saturating_sub(keep_recent).max(first);
4709        while cut < self.history.len() && self.history[cut].role == Role::Tool {
4710            cut += 1;
4711        }
4712        let dropped = cut.saturating_sub(first);
4713        let objective = objective.trim();
4714        let retention = if self.recorder.is_some() {
4715            "They remain in this session's transcript sidecar."
4716        } else {
4717            "No transcript sidecar is attached, so they are not retained."
4718        };
4719        let marker = ChatMessage::system(format!(
4720            "[handoff: a fresh working context starts here]\nObjective: {objective}\n\
4721             {dropped} earlier message(s) were set aside; the most recent {kept} were kept. \
4722             {retention}",
4723            kept = self.history.len() - cut,
4724        ));
4725        if let Err(error) = self.record(&marker) {
4726            tracing::warn!("failed to persist the handoff marker: {error}");
4727        }
4728        let mut new_history = Vec::with_capacity(first + keep_recent + 2);
4729        new_history.extend(self.history[..first].iter().cloned());
4730        new_history.push(marker);
4731        new_history.extend(self.history.split_off(cut));
4732        self.history = new_history;
4733        // BP-8: same re-basing as `maybe_compact` — see its comment.
4734        self.journal_checkpoint(self.history.len());
4735        dropped
4736    }
4737
4738    /// P4b (§1.5/§3.1 `core.compaction.reserve_tokens`, pi§2 shape:
4739    /// `contextTokens > contextWindow - reserveTokens`): whether the
4740    /// estimated token size of the live history is within `reserve_tokens`
4741    /// of the model's context window. `false` when
4742    /// [`Config::compaction_reserve_tokens`] is unset (the default).
4743    fn compaction_pressure_triggered(&self) -> bool {
4744        let Some(reserve) = self.config.compaction_reserve_tokens else {
4745            return false;
4746        };
4747        let limit = provider::model_context_limit(&self.config.model)
4748            .unwrap_or(provider::UNKNOWN_MODEL_CONTEXT_FLOOR);
4749        let used = supercode_runtime::estimate_view_tokens(&self.history);
4750        used.saturating_add(reserve) > limit
4751    }
4752
4753    /// P4b (§1.5/§3.1 `core.compaction.keep_recent_tokens`): how many of the
4754    /// most recent messages (walking backward from the end of `self.history`,
4755    /// skipping the system prompt) fit within the configured token budget
4756    /// (default 20,000, pi§6 precedent). Always keeps at least 2 messages,
4757    /// matching the message-count trigger's own floor.
4758    fn keep_recent_count_by_tokens(&self) -> usize {
4759        let budget = self.config.compaction_keep_recent_tokens.unwrap_or(20_000);
4760        let mut used = 0u64;
4761        let mut count = 0usize;
4762        for msg in self.history.iter().skip(1).rev() {
4763            let t = supercode_runtime::estimate_view_tokens(std::slice::from_ref(msg));
4764            if used.saturating_add(t) > budget && count > 0 {
4765                break;
4766            }
4767            used = used.saturating_add(t);
4768            count += 1;
4769        }
4770        count.max(2)
4771    }
4772
4773    /// Register an additional tool (e.g. your own capability).
4774    ///
4775    /// P5-2 (§2.2 C2 "connect invalidates cache prefix"): registering a
4776    /// tool AFTER this agent has already issued a request
4777    /// ([`Self::request_issued`]) changes the tools schema every
4778    /// subsequent request carries — the exact prefix-churn shape C2
4779    /// describes, MCP-sourced or not. Resets [`Self::cache_established`] so
4780    /// the next cache-warmth check (`provider::cache_cold_reason`) doesn't
4781    /// wrongly assume the entry is still warm. A no-op call before the
4782    /// first request (the common case: `attach_mcp` registers tools once at
4783    /// startup, before any turn runs) changes nothing — byte-identical to
4784    /// today.
4785    pub fn register_tool(&mut self, tool: impl crate::tools::Tool + 'static) {
4786        self.registry.register(tool);
4787        if self.requests_issued {
4788            self.cache_established = false;
4789        }
4790    }
4791
4792    /// The current conversation, including the system prompt.
4793    pub fn history(&self) -> &[ChatMessage] {
4794        &self.history
4795    }
4796
4797    /// Send a user message and run the loop until the model produces a final
4798    /// answer (text with no tool calls) or the iteration budget is exhausted.
4799    pub async fn send(&mut self, user_input: impl Into<String>) -> Result<String> {
4800        let expanded = self.expand_prompt_async(&user_input.into()).await;
4801        let msg = ChatMessage::user(expanded);
4802        self.guard_candidate_message(&msg)?;
4803        self.record(&msg)?;
4804        self.history.push(msg);
4805        self.run_loop().await
4806    }
4807
4808    /// BP-4 (catalog:109 "Context-usage introspection", cc§2 `/context`
4809    /// grid, cx§8 `/status` + `get_context_remaining`): the LIVE
4810    /// context-window accounting for this session — the same numbers
4811    /// `resume --dry-run`'s preflight already computes
4812    /// (`tokens::estimate_request_tokens` / `tokens::context_guard`), read
4813    /// out mid-session instead of only before one.
4814    ///
4815    /// Pure: it projects the request view exactly as
4816    /// [`Self::guard_candidate_message`] does (reduction stubs included,
4817    /// cache annotation included) without mutating the reduction log, so
4818    /// asking "how full am I?" can never change what the next request
4819    /// carries.
4820    pub fn context_usage(&self) -> ContextUsage {
4821        let messages = self.projected_view(None);
4822        let tools = self.tool_schemas();
4823        let message_tokens = supercode_runtime::estimate_view_tokens(&messages);
4824        let request_tokens = supercode_runtime::estimate_request_tokens(&messages, &tools);
4825        let limit = self.context_limit.or_else(|| {
4826            crate::provider::model_context_limit(&self.config.model)
4827                .or(Some(crate::provider::UNKNOWN_MODEL_CONTEXT_FLOOR))
4828        });
4829        let projected_tokens = supercode_runtime::with_guard_margin(request_tokens);
4830        let reserve = supercode_runtime::CONTEXT_RESPONSE_RESERVE_TOKENS;
4831        let (fits, remaining_tokens, used_pct) = match limit {
4832            Some(limit) => (
4833                projected_tokens.saturating_add(reserve) <= limit,
4834                limit
4835                    .saturating_sub(reserve)
4836                    .saturating_sub(projected_tokens),
4837                if limit == 0 {
4838                    0
4839                } else {
4840                    (projected_tokens as f64 / limit as f64 * 100.0).round() as u32
4841                },
4842            ),
4843            None => (true, 0, 0),
4844        };
4845        ContextUsage {
4846            model: self.config.model.clone(),
4847            messages: messages.len(),
4848            message_tokens,
4849            tool_count: tools.len(),
4850            tool_schema_tokens: request_tokens.saturating_sub(message_tokens),
4851            request_tokens,
4852            projected_tokens,
4853            response_reserve_tokens: reserve,
4854            context_limit: limit,
4855            remaining_tokens,
4856            used_pct,
4857            fits,
4858        }
4859    }
4860
4861    /// The messages a request would carry right now — the read-only half of
4862    /// [`Self::guard_candidate_message`]/[`Self::build_request_messages`],
4863    /// with `candidate` optionally appended as a not-yet-committed turn.
4864    /// Never mutates `self`.
4865    fn projected_view(&self, candidate: Option<&ChatMessage>) -> Vec<ChatMessage> {
4866        let messages = match &self.reduction_policy {
4867            None => {
4868                let mut messages = self.history.clone();
4869                if let Some(candidate) = candidate {
4870                    messages.push(candidate.clone());
4871                }
4872                messages
4873            }
4874            Some(policy) => {
4875                let has_system = self.history.first().is_some_and(|m| m.role == Role::System);
4876                let mut reducible = self.history[usize::from(has_system)..].to_vec();
4877                if let Some(candidate) = candidate {
4878                    reducible.push(candidate.clone());
4879                }
4880                let mut prepared = policy.clone();
4881                reduce::prepare_read_freshness(&mut prepared, &reducible);
4882                let (view, _) =
4883                    reduce::project_messages(&reducible, &prepared, &self.reduction_log);
4884                let mut messages = Vec::with_capacity(view.len() + usize::from(has_system));
4885                if has_system {
4886                    messages.push(self.history[0].clone());
4887                }
4888                messages.extend(view);
4889                messages
4890            }
4891        };
4892        provider::apply_cache_plan(&messages, self.config.cache_plan, self.imported_prefix_len)
4893    }
4894
4895    /// Refuse an oversized new user turn before it mutates canonical history
4896    /// or an attached sidecar. The in-loop guard remains authoritative for
4897    /// every actual request; this preflight closes the first-request seam
4898    /// where `send*` used to record/push the message before that guard ran.
4899    fn guard_candidate_message(&self, msg: &ChatMessage) -> Result<()> {
4900        let Some(limit) = self.context_limit else {
4901            return Ok(());
4902        };
4903
4904        let messages = self.projected_view(Some(msg));
4905        let tools = self.tool_schemas();
4906        let (fits, projected_tokens) = supercode_runtime::context_guard(&messages, &tools, limit);
4907        if !fits {
4908            return Err(Error::ContextLimitExceeded {
4909                projected_tokens,
4910                reserve_tokens: supercode_runtime::CONTEXT_RESPONSE_RESERVE_TOKENS,
4911                context_limit: limit,
4912                model: self.config.model.clone(),
4913            });
4914        }
4915        Ok(())
4916    }
4917
4918    /// The messages a provider request should carry for the CURRENT turn
4919    /// (A5/A7/A8/A10): with no [`ReductionPolicy`] installed, exactly
4920    /// `self.history.clone()` — byte-identical to every version of this
4921    /// method before reduction landed. With a policy installed, `history[0]`
4922    /// (this agent's own system prompt, never a reduction target) followed by
4923    /// [`reduce::project_messages`]'s projected view of `history[1..]`, fed
4924    /// with `self.reduction_log` so already-applied reductions reproduce
4925    /// verbatim across turns (prefix stability, A5) — the updated log is
4926    /// stored back onto `self` so the NEXT call (this turn, next turn, or a
4927    /// later `send`) sees the same accumulating state. `self.history` itself
4928    /// is never read back into or mutated by this: it stays the full
4929    /// canonical view, in lockstep with the sidecar (A3).
4930    ///
4931    /// When `policy.elide_stale_reads` is set, this re-runs
4932    /// [`reduce::probe_read_freshness`] (the one place A8's disk I/O happens)
4933    /// against `history[1..]` before projecting, so every request sees
4934    /// up-to-date freshness verdicts — `project_messages` itself stays pure.
4935    ///
4936    /// Finally, B7's [`provider::apply_cache_plan`] runs over the assembled
4937    /// view (regardless of whether a [`ReductionPolicy`] is installed) — a
4938    /// pure, cloning annotation step, so this method's `&mut self` mutations
4939    /// above (`self.reduction_log`) are already committed before it runs and
4940    /// its own output is never written back onto `self.history` or the log:
4941    /// purity for B7's cache breakpoints holds independently of A5's.
4942    fn build_request_messages(&mut self) -> Vec<ChatMessage> {
4943        let messages = match self.reduction_policy.clone() {
4944            None => self.history.clone(),
4945            Some(mut policy) => {
4946                reduce::prepare_read_freshness(&mut policy, &self.history[1..]);
4947                // B7 coordination clamp: while `CachePlan::ImportedPrefix` is
4948                // active, A10 turn-clearing must never establish a range
4949                // that dips into the imported prefix (protects the cache
4950                // breakpoint the request build will place there below).
4951                // `imported_prefix_len` counts `history[0]` (this agent's own
4952                // system message) plus the imported messages, but
4953                // `project_messages` only ever sees `history[1..]` — hence
4954                // the `- 1`.
4955                if matches!(self.config.cache_plan, CachePlan::ImportedPrefix) {
4956                    policy.protect_imported_prefix =
4957                        self.imported_prefix_len.map(|n| n.saturating_sub(1));
4958                }
4959                // TR-7 (T20): the one side-call site, run BEFORE
4960                // `project_messages` (which stays pure/I-O-free) — mirrors
4961                // `elide_stale_reads`/`probe_read_freshness` immediately
4962                // above. Only ever does anything when both the policy gate
4963                // AND a summarizer are present; either being absent means
4964                // `cleared_turns_summary` stays `None` and `project_messages`
4965                // renders the deterministic stub, same as before TR-7
4966                // existed.
4967                if policy.summarize_cleared_turns {
4968                    if let Some(summarizer) = self.span_summarizer.as_deref() {
4969                        policy.cleared_turns_summary = reduce::prepare_cleared_turns_summary(
4970                            &self.history[1..],
4971                            &policy,
4972                            &self.reduction_log,
4973                            summarizer,
4974                        );
4975                    }
4976                }
4977                let (view, log) =
4978                    reduce::project_messages(&self.history[1..], &policy, &self.reduction_log);
4979                self.reduction_log = log;
4980                let mut messages = Vec::with_capacity(view.len() + 1);
4981                messages.push(self.history[0].clone());
4982                messages.extend(view);
4983                messages
4984            }
4985        };
4986        // TR-8 (T5): a tool-schema tier change since the last request is a
4987        // cache-bust event under `CachePlan::ImportedPrefix` — the `tools`
4988        // array is part of the cache key alongside `messages`, so flag it by
4989        // skipping this one request's cache annotation rather than claiming
4990        // a prefix hit that won't actually land. Recorded unconditionally
4991        // (even under `CachePlan::Off`) so the signature stays current
4992        // regardless of which plan is active.
4993        let tier_sig = self.schema_tier_signature();
4994        let busted =
4995            provider::tier_change_is_cache_bust(self.last_tool_schema_tier_signature, tier_sig);
4996        self.last_tool_schema_tier_signature = Some(tier_sig);
4997        let effective_cache_plan = if busted {
4998            CachePlan::Off
4999        } else {
5000            self.config.cache_plan
5001        };
5002        // UX-26 (B7-warn): mirror `apply_cache_plan`'s own placement gate
5003        // (`ImportedPrefix` AND a non-zero prefix) to know whether THIS
5004        // request will actually carry a `cache_control` annotation. `busted`
5005        // requests (schema-tier change) and `CachePlan::Off` never annotate,
5006        // so `provider::cache_cold_reason` can never flag them — there was
5007        // nothing to reuse, by construction. `idle_secs` is computed
5008        // whenever a signal exists at all (even before this agent's first
5009        // annotated send — see `Self::last_cache_activity_ms`'s doc comment
5010        // on why the pre-establishment case matters); `cache_established`
5011        // additionally gates the usage-ratio check specifically (see
5012        // `provider::cache_cold_reason`'s doc comment for why those two
5013        // checks need independent gates).
5014        let will_annotate = matches!(effective_cache_plan, CachePlan::ImportedPrefix)
5015            && self.imported_prefix_len.is_some_and(|n| n > 0);
5016        let idle_secs = self
5017            .last_cache_activity_ms
5018            .map(|last| (now_ms() - last).max(0) / 1000);
5019        self.pending_cache_turn = (will_annotate, self.cache_established, idle_secs);
5020        let mut messages =
5021            provider::apply_cache_plan(&messages, effective_cache_plan, self.imported_prefix_len);
5022        // BP-7 (catalog §4a "Goals"): the standing objective, restated at
5023        // the TAIL of the request — after the cache annotation, which sits
5024        // on the PREFIX, so a goal that changes mid-session never busts the
5025        // cached prefix. Request-view only: `history` is untouched, so the
5026        // persisted transcript is exactly the conversation and a translator
5027        // never has to invent a message for a harness-tracked goal.
5028        if let Some(goal) = &self.goal {
5029            messages.push(ChatMessage::system(goal.reminder()));
5030        }
5031        messages
5032    }
5033
5034    /// Run the model/tool loop over the current history until a final answer or
5035    /// the iteration budget is exhausted. (Shared by `send`, `send_with_files`,
5036    /// and `send_with_images`.)
5037    /// P4b (§1.7, pi§3 semantics): pop the next message(s) to deliver from
5038    /// `queue` per `mode` — `All` drains everything and joins it with a
5039    /// blank line, `OneAtATime` pops exactly one. `None` when `queue` is
5040    /// empty (the default state, at zero cost).
5041    fn drain_steer_queue(
5042        queue: &mut std::collections::VecDeque<String>,
5043        mode: SteeringMode,
5044    ) -> Option<String> {
5045        if queue.is_empty() {
5046            return None;
5047        }
5048        match mode {
5049            SteeringMode::All => Some(queue.drain(..).collect::<Vec<_>>().join("\n\n")),
5050            SteeringMode::OneAtATime => queue.pop_front(),
5051        }
5052    }
5053
5054    async fn run_loop(&mut self) -> Result<String> {
5055        let _steer_turn = SteerTurnGuard::new(self.steer_queue.clone());
5056        let mut output_tokens_used: u64 = 0;
5057
5058        // BP-7 (catalog §4a "Turn/budget caps"): the SPEND cap, checked
5059        // before this `send` can issue anything. Unlike
5060        // `max_total_output_tokens` (a per-`send` allowance, unchanged),
5061        // spend accumulates over the agent's whole lifetime — a dollar
5062        // budget that resets on every prompt is not a budget. A cap reached
5063        // MID-loop ends that loop cleanly with a `spend_budget` finish
5064        // marker (below); a cap already exhausted at entry is an error,
5065        // because there is nothing to return.
5066        if let Some(budget) = self.config.max_budget_usd.filter(|b| *b > 0.0) {
5067            if self.total_cost_usd >= budget {
5068                return Err(Error::BudgetExhausted {
5069                    spent_usd: self.total_cost_usd,
5070                    budget_usd: budget,
5071                });
5072            }
5073        }
5074
5075        // P5-9 (§2 module 20, cc's "per-prompt file-history-snapshot"):
5076        // open a fresh checkpoint for THIS turn — `run_loop` is called
5077        // exactly once per `send`/`send_with_files`/`send_with_images`
5078        // call (never recursively for the same turn), so this fires once
5079        // per user prompt, matching the design's per-prompt granularity.
5080        // `self.history.last()` is the user message that call just pushed.
5081        // `None` (`checkpoint_observer` unset, the default) is a no-op —
5082        // zero cost, no disk touched.
5083        if let Some(cp) = &self.checkpoint_observer {
5084            let label = self
5085                .history
5086                .last()
5087                .and_then(|m| m.content.as_deref())
5088                .unwrap_or("")
5089                .to_string();
5090            cp.begin_turn(&label);
5091        }
5092
5093        // BP-4 (catalog:90, cx§2 "re-emitted on change"): once per user
5094        // turn — not per loop iteration — re-derive the environment block
5095        // so a cwd change, an approval/sandbox policy change or a branch
5096        // switch since the last turn reaches the model instead of leaving
5097        // it reading the startup snapshot. A no-op, with no subprocess, for
5098        // every config that doesn't set `core.env_context`.
5099        self.refresh_env_context();
5100
5101        for _ in 0..self.config.max_iterations {
5102            // BP-8 (catalog:156): flush a plan `update_plan` wrote during
5103            // the previous iteration's tool calls. A no-op when
5104            // `todos.persist` is off or the plan did not change.
5105            self.journal_plan_if_changed();
5106            // BP-7: the index of the round-trip this iteration is about to
5107            // make. Captured here because `self.turn_index` advances the
5108            // moment the usage record is written, and every marker in this
5109            // iteration — including the ones written after that point —
5110            // must carry the SAME index, or the marker log would not join
5111            // to the usage log on `turn`.
5112            let round_trip = self.turn_index;
5113            self.maybe_compact();
5114            // P4b (§1.7, pi§3 "steer = after current tool calls"): drain any
5115            // queued mid-turn steering message(s) BEFORE building the next
5116            // request — the top of every loop iteration is exactly "after
5117            // whatever tool calls the previous iteration just ran" (or, on
5118            // the very first iteration, before anything has happened yet,
5119            // which is an equally valid "deliver immediately" reading).
5120            // Empty queue (today's default state) is a no-op.
5121            let (steer_msg, steer_taken) = {
5122                let mut inbox = self
5123                    .steer_queue
5124                    .lock()
5125                    .unwrap_or_else(std::sync::PoisonError::into_inner);
5126                let before = inbox.len();
5127                let drained = inbox.drain(self.config.steering_mode);
5128                let taken = before - inbox.len();
5129                (drained, taken)
5130            };
5131            if let Some(steer_msg) = steer_msg {
5132                // BP-8 (catalog:154): the queue record's other half —
5133                // without it a replayed journal would keep re-delivering an
5134                // input the conversation already consumed.
5135                self.journal_queue_drain(crate::session_journal::QueueKind::Steer, steer_taken);
5136                let msg = ChatMessage::user(steer_msg);
5137                self.record(&msg)?;
5138                self.history.push(msg);
5139            }
5140            // Recomputed every iteration (not hoisted): under `Deferred`
5141            // advertising, a `tool_search` call earlier in this same loop
5142            // activates tools that must be advertised starting with the very
5143            // next request (B6).
5144            let tools = self.tool_schemas();
5145            let messages = self.build_request_messages();
5146
5147            // PARITY-18 D4 — re-check the context guard before EVERY
5148            // request this loop builds, not just the caller's one-shot
5149            // preflight: interactive turns 2+, `/expand all`, and any
5150            // mid-loop tool round-trip that grows `messages` can push a
5151            // barely-passing session over the limit between sends. Only
5152            // armed when a caller has opted in via `set_context_limit`.
5153            // Uses the exact same `tokens::context_guard`
5154            // formula the CLI preflight uses, so the two can never disagree.
5155            if let Some(limit) = self.context_limit {
5156                let (fits, projected_tokens) =
5157                    supercode_runtime::context_guard(&messages, &tools, limit);
5158                if !fits {
5159                    return Err(Error::ContextLimitExceeded {
5160                        projected_tokens,
5161                        reserve_tokens: supercode_runtime::CONTEXT_RESPONSE_RESERVE_TOKENS,
5162                        context_limit: limit,
5163                        model: self.config.model.clone(),
5164                    });
5165                }
5166            }
5167
5168            // BP-7 (catalog §4a "Turn/step bracketing records"): the
5169            // OPENING bracket, written before the request is issued so it
5170            // survives a request that never returns (a cancelled turn keeps
5171            // its `context` marker with no `usage`/`finish` after it).
5172            // Uses `tokens::estimate_request_tokens` — the same estimator
5173            // the context guard above uses, so the two can never disagree.
5174            self.push_turn_marker_at(
5175                round_trip,
5176                crate::turn_record::TurnMarker::Context {
5177                    messages: messages.len(),
5178                    tools: tools.len(),
5179                    estimated_tokens: supercode_runtime::estimate_request_tokens(&messages, &tools),
5180                },
5181            );
5182
5183            let mut req = self.chat_request(messages, tools);
5184
5185            let fallback_hops: Vec<FallbackHop>;
5186            let completion = {
5187                let sink = self.config.event_sink.as_ref();
5188                let on_delta = move |s: &str| {
5189                    if let Some(sink) = sink {
5190                        sink(AgentEvent::TextDelta(s.to_string()));
5191                    }
5192                };
5193                // PARITY-18 D3 — the real send site: flip the flag
5194                // immediately before issuing the request, regardless of
5195                // whether `complete` then succeeds or fails, so
5196                // `request_issued()` truthfully reflects "a live request
5197                // was attempted" rather than "the run reached this line and
5198                // later succeeded."
5199                self.requests_issued = true;
5200                // P4b (§1.1/§3.1 `core.retry`, pi§3 shape): retry-with-
5201                // backoff already lives at the TRANSPORT layer
5202                // (`provider::OpenAiProvider::send_with_retry`, pre-existing
5203                // — connection failures and 5xx responses are retried
5204                // there); `Config.retry_*` (see `Agent::new`) makes that
5205                // EXISTING mechanism config-file-settable instead of
5206                // duplicating a second retry loop here, which would nest
5207                // retries confusingly on top of the transport's own.
5208                // BP-13 (D9 "Failure fallback model chains"): the chain is
5209                // EXECUTED here, not merely resolved. On a failure another
5210                // model could plausibly answer (overload / rate limit /
5211                // unavailability — `is_failover_worthy`), the request is
5212                // re-sent against the next entry of
5213                // `Config::model_fallback`, with routing re-applied for
5214                // that model. A 4xx that is not a rate limit is the caller's
5215                // problem, not the model's, and is never retried elsewhere.
5216                // The transport-level retry above has already run and given
5217                // up by the time a hop is considered.
5218                let (result, hops) = self.complete_with_fallback(&mut req, &on_delta).await;
5219                fallback_hops = hops;
5220                result
5221            };
5222            // BP-7: drained whether the request succeeded or failed, and
5223            // BEFORE the `?` — a request that exhausted its retries and
5224            // then errored is exactly the case a retry record exists for.
5225            for notice in self.retry_log.drain() {
5226                self.emit(AgentEvent::ProviderRetry {
5227                    attempt: notice.attempt,
5228                    delay_ms: notice.delay_ms,
5229                    reason: notice.reason.clone(),
5230                });
5231                self.push_turn_marker_at(
5232                    round_trip,
5233                    crate::turn_record::TurnMarker::Retry {
5234                        attempt: notice.attempt,
5235                        delay_ms: notice.delay_ms,
5236                        reason: notice.reason,
5237                    },
5238                );
5239            }
5240            // The switch the fallback pass performed is a real mid-session
5241            // model change: it moves `Config::model` for every subsequent
5242            // request and is recorded exactly like a user-driven `/model`
5243            // switch (typed record + journal line), never as a silent retry.
5244            for hop in fallback_hops {
5245                self.record_model_change(&hop.from, &hop.to, Some(hop.reason.as_str()));
5246            }
5247            let (mut assistant, usage) = completion?;
5248            // Persist the actual generating model on the message itself.
5249            // A resumed foreign session keeps its original model in
5250            // `SessionMeta`; using only that session-level value on export
5251            // misattributes every Supercode continuation turn to the source
5252            // harness model. Per-message provenance lets native exporters
5253            // preserve the boundary accurately (for example, Claude history
5254            // followed by a GLM continuation).
5255            assistant
5256                .metadata
5257                .insert("model".to_string(), self.config.model.clone());
5258            output_tokens_used += usage.completion_tokens;
5259            self.total_output_tokens += usage.completion_tokens;
5260
5261            // UX-26 (B7-warn): consult the verdict computed at build time
5262            // (before this request was sent) now that `usage` — the only
5263            // piece that couldn't be known pre-send — is in hand. Gated on
5264            // `Config::cache_warnings` (default on; `--no-cache-warnings` /
5265            // `SUPERCODE_CACHE_WARNINGS=0` at the CLI layer, dev/03) so this
5266            // stays a zero-behavior-change no-op for every caller that
5267            // hasn't opted into `CachePlan::ImportedPrefix` in the first
5268            // place (`pending_cache_turn.0` is `false` whenever
5269            // `CachePlan::Off`, so the predicate always returns `None` then
5270            // regardless of this flag).
5271            let (will_annotate, cache_established, idle_secs) = self.pending_cache_turn;
5272            // UX-26 T2 (accuracy fold-in): `CacheColdReason::message` asserts
5273            // Anthropic-specific facts (a fixed 5-minute ephemeral TTL, and
5274            // cache-read-ratio semantics that assume Anthropic's exact-count
5275            // billing) that are only true for Anthropic-family models. This
5276            // is a WARNING-only gate, deliberately not folded into
5277            // `will_annotate`/the breakpoint-placement gate above: whether a
5278            // `cache_control` breakpoint is safe/inert to send to a
5279            // non-Anthropic model through OpenRouter is a separate cache-
5280            // behavior question this ticket doesn't touch (see
5281            // `.volter/tracker/markdown/UX-26.md`'s T2 note) — narrowing only
5282            // the warning keeps this fix scoped to warning ACCURACY, with
5283            // zero change to what gets sent on the wire.
5284            let warning_applies_to_this_model =
5285                provider::is_anthropic_family_model(&self.config.model);
5286            if self.config.cache_warnings && warning_applies_to_this_model {
5287                if let Some(reason) =
5288                    provider::cache_cold_reason(will_annotate, cache_established, idle_secs, &usage)
5289                {
5290                    self.emit(AgentEvent::CacheWarning {
5291                        message: reason.message(),
5292                    });
5293                }
5294            }
5295            // Refresh the activity clock / establish-once flag for the NEXT
5296            // turn's comparison, but only when THIS request actually carried
5297            // the annotation — an unannotated (busted/Off) request neither
5298            // warms nor cools a cache entry it never touched.
5299            if will_annotate {
5300                self.last_cache_activity_ms = Some(now_ms());
5301                self.cache_established = true;
5302            }
5303
5304            // UX-23: emitted before `TurnCompleted` so a `--trace`/
5305            // `stream-json` consumer sees "this round-trip cost N tokens"
5306            // land right alongside the round-trip it describes, rather than
5307            // needing to correlate it with a later event.
5308            self.emit(AgentEvent::Usage(usage.clone()));
5309            self.emit(AgentEvent::TurnCompleted);
5310            // P4b (§1.6, catalog §4a "persisted per-turn usage records"):
5311            // EventSink already streamed `Usage` above — this durably
5312            // accumulates the same data as a typed record (see
5313            // `Self::usage_records`/`Self::save_usage_log`), never a lossy
5314            // display-only channel.
5315            // BP-7 (catalog §4a "Per-turn cost/usage accounting"): the
5316            // record now carries the round-trip's DOLLAR cost too — the
5317            // half the row's semantics name alongside tokens — whenever
5318            // this build can price the model.
5319            // BP-13 (D9 "Model-served-vs-requested provenance"): the record
5320            // now carries BOTH sides — the model this agent asked for and,
5321            // when the provider reported one, the model that actually
5322            // answered. They can genuinely differ (a gateway aliasing a
5323            // name to a dated snapshot, a fallback hop, a routed tier), and
5324            // a record that can only ever state the request cannot show it.
5325            let served = assistant
5326                .metadata
5327                .get(crate::provider::SERVED_MODEL_KEY)
5328                .cloned();
5329            let record = crate::usage_log::UsageRecord::from_usage(
5330                self.turn_index,
5331                &self.config.model,
5332                &usage,
5333                now_ms(),
5334            )
5335            .priced(self.model_price)
5336            .with_served_model(served);
5337            self.total_cost_usd += record.cost_usd.unwrap_or(0.0);
5338            // BP-7: the round-trip's usage bracket, written from the same
5339            // point as the usage record so the two logs never disagree.
5340            self.push_turn_marker_at(
5341                round_trip,
5342                crate::turn_record::TurnMarker::Usage {
5343                    prompt_tokens: record.prompt_tokens,
5344                    completion_tokens: record.completion_tokens,
5345                    total_tokens: record.total_tokens,
5346                    cached_tokens: record.cached_tokens,
5347                    cost_usd: record.cost_usd,
5348                },
5349            );
5350            self.journal_usage(&record);
5351            self.usage_log.push(record);
5352            self.turn_index += 1;
5353            self.record(&assistant)?;
5354            self.history.push(assistant.clone());
5355
5356            let calls = assistant.tool_calls().to_vec();
5357            if calls.is_empty() {
5358                // Close steering acceptance under the same lock as the last
5359                // drain. A message accepted before this boundary extends the
5360                // current turn; anything later is rejected by the SDK and
5361                // can never leak into a future turn.
5362                let (steer_msg, steer_taken) = {
5363                    let mut inbox = self
5364                        .steer_queue
5365                        .lock()
5366                        .unwrap_or_else(std::sync::PoisonError::into_inner);
5367                    let before = inbox.len();
5368                    let drained = inbox.drain_or_close(self.config.steering_mode);
5369                    let taken = before - inbox.len();
5370                    (drained, taken)
5371                };
5372                if let Some(steer_msg) = steer_msg {
5373                    self.journal_queue_drain(crate::session_journal::QueueKind::Steer, steer_taken);
5374                    let msg = ChatMessage::user(steer_msg);
5375                    self.record(&msg)?;
5376                    self.history.push(msg);
5377                    continue;
5378                }
5379                // P4b (§1.7, pi§3 "follow-up = at idle"): a queued follow-up
5380                // message takes priority over the stop-gate — it's more
5381                // input to answer, not a veto of an answer already given.
5382                let follow_up_before = self.follow_up_queue.len();
5383                if let Some(follow_up_msg) =
5384                    Self::drain_steer_queue(&mut self.follow_up_queue, self.config.follow_up_mode)
5385                {
5386                    self.journal_queue_drain(
5387                        crate::session_journal::QueueKind::FollowUp,
5388                        follow_up_before - self.follow_up_queue.len(),
5389                    );
5390                    let msg = ChatMessage::user(follow_up_msg);
5391                    self.record(&msg)?;
5392                    self.history.push(msg);
5393                    continue;
5394                }
5395                // P4b (§1.9/§3.1 `[core] stop_gate`, D3 "stop/completion
5396                // gating"): consulted exactly once per iteration that would
5397                // otherwise return — computed into an owned `Option<String>`
5398                // so the immutable borrow of `self.config.stop_gate` ends
5399                // before the `self.record`/`self.history.push` calls below
5400                // need `&mut self`.
5401                let final_content = assistant.content.clone().unwrap_or_default();
5402                let veto_reason: Option<String> = self
5403                    .config
5404                    .stop_gate
5405                    .as_ref()
5406                    .and_then(|gate| gate(&final_content));
5407                if let Some(reason) = veto_reason {
5408                    let msg = ChatMessage::user(reason);
5409                    self.record(&msg)?;
5410                    self.history.push(msg);
5411                    continue;
5412                }
5413                // BP-8 (catalog:156): the last iteration's tool calls are
5414                // the ones the top-of-loop flush above never sees.
5415                self.journal_plan_if_changed();
5416                self.push_turn_marker_at(
5417                    round_trip,
5418                    crate::turn_record::TurnMarker::Finish {
5419                        reason: crate::turn_record::FinishReason::EndTurn,
5420                    },
5421                );
5422                return Ok(assistant.content.unwrap_or_default());
5423            }
5424
5425            // BP-7: this round-trip ended by asking for tool calls; the
5426            // loop continues. The budget arms below mark the LOOP's end
5427            // separately when one of them stops it here.
5428            self.push_turn_marker_at(
5429                round_trip,
5430                crate::turn_record::TurnMarker::Finish {
5431                    reason: crate::turn_record::FinishReason::ToolCalls,
5432                },
5433            );
5434
5435            // Output-token budget (output only — input tokens are not counted,
5436            // so this does not bound cost): stop spawning further model turns
5437            // once the cumulative output-token budget for this `send` is
5438            // exhausted.
5439            if let Some(budget) = self.config.max_total_output_tokens {
5440                if output_tokens_used >= budget {
5441                    // The assistant turn we just pushed carries unanswered
5442                    // tool_calls. Leaving them dangling yields an invalid
5443                    // history (assistant tool_calls with no tool results) that
5444                    // the provider rejects on the next `send`/resume. Emit
5445                    // synthetic results so the transcript stays well-formed.
5446                    for call in &calls {
5447                        let msg = ChatMessage::tool_result(
5448                            call.id.clone(),
5449                            call.function.name.clone(),
5450                            "[skipped: output token budget reached]".to_string(),
5451                        );
5452                        self.record(&msg)?;
5453                        self.history.push(msg);
5454                    }
5455                    self.push_turn_marker_at(
5456                        round_trip,
5457                        crate::turn_record::TurnMarker::Finish {
5458                            reason: crate::turn_record::FinishReason::OutputTokenBudget,
5459                        },
5460                    );
5461                    return Ok(assistant.content.clone().unwrap_or_default());
5462                }
5463            }
5464
5465            // BP-7 (catalog §4a "Turn/budget caps"): the SPEND and STEP
5466            // caps, at the same point and with the same shape as the
5467            // output-token cap above — checked before this turn's tool
5468            // calls run, with synthetic results so the transcript stays
5469            // well-formed for a resume.
5470            let spend_exhausted = self
5471                .config
5472                .max_budget_usd
5473                .is_some_and(|b| b > 0.0 && self.total_cost_usd >= b);
5474            let steps_exhausted = self
5475                .config
5476                .max_steps
5477                .is_some_and(|n| n > 0 && self.total_steps + calls.len() > n);
5478            if spend_exhausted || steps_exhausted {
5479                let (label, reason) = if spend_exhausted {
5480                    (
5481                        "[skipped: spend budget reached]",
5482                        crate::turn_record::FinishReason::SpendBudget,
5483                    )
5484                } else {
5485                    (
5486                        "[skipped: step budget reached]",
5487                        crate::turn_record::FinishReason::StepBudget,
5488                    )
5489                };
5490                for call in &calls {
5491                    let msg = ChatMessage::tool_result(
5492                        call.id.clone(),
5493                        call.function.name.clone(),
5494                        label.to_string(),
5495                    );
5496                    self.record(&msg)?;
5497                    self.history.push(msg);
5498                }
5499                self.push_turn_marker_at(
5500                    round_trip,
5501                    crate::turn_record::TurnMarker::Finish { reason },
5502                );
5503                return Ok(assistant.content.clone().unwrap_or_default());
5504            }
5505            self.total_steps += calls.len();
5506
5507            // P4e (§3.1 `core.parallel_tool_calls`, catalog:59): off (the
5508            // default) or a single call takes the EXACT pre-P4e sequential
5509            // path below, byte-identical. Only `true` with 2+ calls in this
5510            // turn takes `Self::run_tools_concurrently` — see its doc
5511            // comment for exactly what does and doesn't run concurrently.
5512            if self.config.parallel_tool_calls && calls.len() > 1 {
5513                for call in &calls {
5514                    self.emit(AgentEvent::tool_started(call));
5515                }
5516                let results = self.run_tools_concurrently(&calls).await;
5517                for (call, (output, is_error)) in calls.iter().zip(results) {
5518                    self.emit(AgentEvent::ToolCallCompleted {
5519                        id: call.id.clone(),
5520                        name: call.function.name.clone(),
5521                        output: output.clone(),
5522                        is_error,
5523                    });
5524                    self.apply_tool_result(call, output, is_error)?;
5525                }
5526            } else {
5527                for call in &calls {
5528                    self.emit(AgentEvent::tool_started(call));
5529                    let (output, is_error) = self.run_tool(call).await;
5530                    self.emit(AgentEvent::ToolCallCompleted {
5531                        id: call.id.clone(),
5532                        name: call.function.name.clone(),
5533                        output: output.clone(),
5534                        is_error,
5535                    });
5536                    self.apply_tool_result(call, output, is_error)?;
5537                }
5538            }
5539            // BP-3 (catalog row "Context-budget tools"): a `new_context`
5540            // call parks its request on the shared budget; this is where
5541            // the agent — the one owner of `history` — applies it, so the
5542            // NEXT request built by this loop is already the fresh window.
5543            // No parked request (every session that never calls the tool)
5544            // is a single `Option` check.
5545            self.apply_pending_new_context();
5546        }
5547
5548        self.push_turn_marker_at(
5549            self.turn_index.saturating_sub(1),
5550            crate::turn_record::TurnMarker::Finish {
5551                reason: crate::turn_record::FinishReason::MaxIterations,
5552            },
5553        );
5554        Err(Error::MaxIterations(self.config.max_iterations))
5555    }
5556
5557    /// BP-3: apply a parked [`crate::tools::NewContextRequest`], if any.
5558    ///
5559    /// The rewrite itself is BP-4's [`Self::new_context`] — the SAME
5560    /// mechanism the operator's `/handoff` runs, so the model's door and the
5561    /// human's door can never drift into two different notions of "a fresh
5562    /// window". This function is only the hand-off point between the tool
5563    /// that asked and the agent that owns `history`.
5564    fn apply_pending_new_context(&mut self) {
5565        let Some(request) = self.ctx.context_budget.take_new_context() else {
5566            return;
5567        };
5568        self.new_context(&request.objective, request.keep_recent);
5569    }
5570
5571    /// The exact post-execution handling every tool result gets, regardless
5572    /// of whether it was produced by the sequential loop or
5573    /// [`Self::run_tools_concurrently`] — factored out of `Self::run_loop`'s
5574    /// tool-dispatch section (P4e) so both paths share one copy: multimodal
5575    /// image-marker detection, A7 output capping (gated exactly as before),
5576    /// TR-10 error stamping, and the `record`/`history` append. Always
5577    /// called in ORIGINAL call order, one call at a time, so the lossless
5578    /// sidecar's append-order invariant (S1.13) holds regardless of which
5579    /// dispatch path produced the result.
5580    fn apply_tool_result(
5581        &mut self,
5582        call: &supercode_interchange::ToolCall,
5583        output: String,
5584        is_error: bool,
5585    ) -> Result<()> {
5586        // P4c (§1.2 `core.tools.read_file.multimodal` / `view_image`):
5587        // a successful tool result carrying the image-data-URL
5588        // marker becomes a `content_parts` image block instead of
5589        // plain text — checked BEFORE `cap_tool_output` (a data URL
5590        // is not meaningfully "capped" by a byte-length text notice)
5591        // and recorded identically on both the full and history
5592        // copies, mirroring `ImageRedacted`'s "images are their own
5593        // axis, orthogonal to A7 text truncation" treatment
5594        // (reduce.rs). An ERRORED call never carries the marker (a
5595        // tool only emits it on success), so `is_error` is not
5596        // re-checked here.
5597        if let Some(data_url) = output.strip_prefix(crate::tools::MULTIMODAL_IMAGE_MARKER) {
5598            let notice = format!("[{}: image content attached below]", call.function.name);
5599            let full_result = ChatMessage::tool_result_with_image(
5600                call.id.clone(),
5601                call.function.name.clone(),
5602                notice.clone(),
5603                data_url.to_string(),
5604            );
5605            let hist_result = ChatMessage::tool_result_with_image(
5606                call.id.clone(),
5607                call.function.name.clone(),
5608                notice,
5609                data_url.to_string(),
5610            );
5611            self.record(&full_result)?;
5612            self.history.push(hist_result);
5613            return Ok(());
5614        }
5615        // Record the FULL output before capping (A3): what the
5616        // sidecar keeps must never be the already-lossy, truncated
5617        // copy (#8/#40) — `history` alone governs what shrinks.
5618        let mut full_result =
5619            ChatMessage::tool_result(call.id.clone(), call.function.name.clone(), output.clone());
5620        // D6/A7 supersession gate (TR-12 land-blocker fix): `history`
5621        // is the exact slice `reduce::project_messages` mints A7/A10
5622        // reduction hashes from (`Self::build_request_messages`
5623        // below). Capping it here — as this unconditionally used to
5624        // do — would silently shrink the bytes those hashes cover, so
5625        // a hash minted now could never recompute the same way once
5626        // the sidecar is reloaded from disk later (`verify_log`/
5627        // `invert`, offline). Gate `cap_tool_output` off in exactly
5628        // the combination where reductions can be minted over
5629        // `history` AND the full bytes are durably retained: a
5630        // recorder AND a `ReductionPolicy` both installed. A7 then
5631        // owns tool-output bounding, reversibly, at projection time
5632        // (SPEC.md D6/A7) — `history`/the sidecar keep everything,
5633        // only the request view shrinks. With a policy but no
5634        // recorder (constructible via `set_reduction_policy` alone),
5635        // nothing durable backs the full bytes, so capping stays on —
5636        // the same honest-labeling spirit as `cap_tool_output`'s own
5637        // retention branch below, just applied at the gate instead of
5638        // the notice text. With no policy at all, this is untouched:
5639        // today's byte-identical legacy cap.
5640        let for_history = if self.recorder.is_some() && self.reduction_policy.is_some() {
5641            output
5642        } else {
5643            self.cap_tool_output(output)
5644        };
5645        let mut hist_result =
5646            ChatMessage::tool_result(call.id.clone(), call.function.name.clone(), for_history);
5647        if is_error {
5648            // TR-10: the reduction layer's success/failure boundary
5649            // (`ReductionKind::ToolInputElided` must never target an
5650            // errored call — TR-6's territory) has no other
5651            // structural signal on `ChatMessage`; stamp both the
5652            // recorded copy (so it survives a sidecar round-trip via
5653            // `NativeTurn`) and the live-history copy (so an
5654            // in-process `project_messages` sees it immediately).
5655            reduce::mark_tool_error(&mut full_result);
5656            reduce::mark_tool_error(&mut hist_result);
5657        }
5658        self.record(&full_result)?;
5659        self.history.push(hist_result);
5660        Ok(())
5661    }
5662
5663    /// Truncate an oversized tool result so a single runaway command can't blow
5664    /// up the context window. Cuts on a char boundary and appends a notice.
5665    fn cap_tool_output(&self, output: String) -> String {
5666        let Some(max) = self.config.max_tool_output_bytes else {
5667            return output;
5668        };
5669        if max == 0 || output.len() <= max {
5670            return output;
5671        }
5672        // Find the largest char boundary <= max.
5673        let mut end = max;
5674        while end > 0 && !output.is_char_boundary(end) {
5675            end -= 1;
5676        }
5677        let total = output.len();
5678        let mut s = output[..end].to_string();
5679        // BP-2 (catalog:58, `core.tool_output_spill`): write the full bytes
5680        // to a per-session file the model can read back. Off (the default)
5681        // leaves the notice byte-identical to before.
5682        let spill = if self.config.tool_output_spill {
5683            self.spill_tool_output(&output)
5684        } else {
5685            None
5686        };
5687        // Honest retention labeling (D6, B10-AC4): only claim the sidecar has
5688        // the full output when a recorder is actually installed — or, BP-2,
5689        // that the spill file has it when one was actually written.
5690        let retention = if self.recorder.is_some() {
5691            "full output in session sidecar"
5692        } else if spill.is_some() {
5693            "full output on disk"
5694        } else {
5695            "full output not retained"
5696        };
5697        let recovery = match &spill {
5698            // The door is named in the notice, so it works WITHOUT
5699            // `capabilities.reduction`: under a preset with a read tool
5700            // that is `read_file`; under a shell-only preset (cx-parity,
5701            // whose whole read pathway is the shell) it is `cat`.
5702            Some(path) => {
5703                let door = if self.registry.get("read_file").is_some() {
5704                    "read it with `read_file`"
5705                } else {
5706                    "read it with `cat`"
5707                };
5708                format!("; full output spilled to {} — {door}", path.display())
5709            }
5710            None => String::new(),
5711        };
5712        s.push_str(&format!(
5713            "{CAP_NOTICE_MARKER}{total} bytes total, showing first {end}; {retention}{recovery}]"
5714        ));
5715        s
5716    }
5717
5718    /// BP-2 (catalog:58 "Oversized output truncated; full content kept
5719    /// reachable"): write `full` to this session's spill directory and
5720    /// return the path, or `None` if it could not be written (a spill is a
5721    /// recovery convenience — it must never fail the tool call).
5722    ///
5723    /// The file is named by content hash, so the same output spilled twice
5724    /// costs one file and a re-run of an identical command reuses it.
5725    fn spill_tool_output(&self, full: &str) -> Option<std::path::PathBuf> {
5726        let dir = self.spill_dir();
5727        std::fs::create_dir_all(&dir).ok()?;
5728        let digest = blake3::hash(full.as_bytes()).to_hex();
5729        let path = dir.join(format!("tool-output-{}.txt", &digest[..16]));
5730        if !path.exists() {
5731            std::fs::write(&path, full).ok()?;
5732        }
5733        Some(path)
5734    }
5735
5736    /// BP-2: where this agent's spilled outputs live — beside the session
5737    /// sidecar when one is recording (per-SESSION, the same identity the
5738    /// sidecar has), else a per-PROCESS temp directory, which is as
5739    /// specific as an agent with no sidecar can honestly be.
5740    fn spill_dir(&self) -> std::path::PathBuf {
5741        if let Some(recorder) = &self.recorder {
5742            let path = recorder.path();
5743            if let (Some(parent), Some(stem)) = (path.parent(), path.file_stem()) {
5744                return parent.join(format!("{}.spill", stem.to_string_lossy()));
5745            }
5746        }
5747        std::env::temp_dir().join(format!("supercode-spill-{}", std::process::id()))
5748    }
5749
5750    /// P5-3 note on the signature: written as a plain fn returning an
5751    /// explicitly boxed future (`Pin<Box<dyn Future + Send>>`) rather than
5752    /// as `async fn`. `spawn_subagent` makes this function genuinely
5753    /// recursive at the TYPE level: `run_tool` -> `run_spawn_subagent` ->
5754    /// (a child) `Agent::send` -> `run_loop` -> `run_tool` again — an
5755    /// `async fn`'s return type is an anonymous, compiler-inferred
5756    /// self-referential state machine, and inferring one that embeds
5757    /// itself (even indirectly, through several other functions) is a
5758    /// compile error (an infinitely-sized/cyclic opaque type). Declaring
5759    /// `run_tool`'s return type EXPLICITLY as a boxed trait object breaks
5760    /// the cycle: every other function on the call graph now embeds a
5761    /// concrete, already-known type here instead of one the compiler would
5762    /// otherwise need to (cyclically) infer. Callers are unaffected —
5763    /// `self.run_tool(call).await` reads identically either way.
5764    fn run_tool<'a>(
5765        &'a mut self,
5766        call: &'a supercode_interchange::ToolCall,
5767    ) -> std::pin::Pin<Box<dyn std::future::Future<Output = (String, bool)> + Send + 'a>> {
5768        Box::pin(async move {
5769            let translated_builtin = if self.config.claude_runtime_tools_enabled {
5770                match self.translate_claude_builtin_call(call) {
5771                    Ok(translated) => translated,
5772                    Err(error) => return (format!("Error: {error}"), true),
5773                }
5774            } else {
5775                None
5776            };
5777            let call = translated_builtin.as_ref().unwrap_or(call);
5778            if self.config.claude_runtime_tools_enabled
5779                && matches!(
5780                    call.function.name.as_str(),
5781                    CLAUDE_CRON_CREATE
5782                        | CLAUDE_CRON_DELETE
5783                        | CLAUDE_CRON_LIST
5784                        | CLAUDE_SCHEDULE_WAKEUP
5785                )
5786            {
5787                return self.run_claude_runtime_tool(call);
5788            }
5789            // P5-3: `spawn_subagent`/`subagent_status` need full async
5790            // `&mut self` access (running a child agent's loop, or
5791            // awaiting an already-finished background `JoinHandle`) —
5792            // `prepare_tool_call` is purely synchronous, so these are
5793            // intercepted HERE, one level above it, rather than inside it
5794            // like `TOOL_SEARCH`/`EXPAND_REDUCTION`/`SIDECAR_SEARCH`.
5795            if call.function.name == CLAUDE_AGENT && self.config.subagents_claude_agent_alias {
5796                return match self.translate_claude_agent_call(call) {
5797                    Ok(translated) => self.run_spawn_subagent(&translated).await,
5798                    Err(error) => (format!("Error: {error}"), true),
5799                };
5800            }
5801            if call.function.name == SPAWN_SUBAGENT {
5802                return self.run_spawn_subagent(call).await;
5803            }
5804            // P5-3 safety-hardening fix (Fable-5 review, LOW "wrong error
5805            // when disabled"): gated on `subagents_enabled`, matching
5806            // `run_spawn_subagent`'s own already-correct disabled behavior
5807            // (that one gates INTERNALLY, at its own top; this one gates
5808            // HERE, at the interception point, because unlike
5809            // `spawn_subagent` it has no other reason to run any logic at
5810            // all when subagents are off). When disabled, a hallucinated
5811            // `subagent_status` call must NOT be intercepted — it falls
5812            // through to `prepare_tool_call`'s normal unknown-tool path
5813            // below, which returns `Error::UnknownTool("subagent_status")`,
5814            // byte-identical to the pre-P5-3 (and disabled-spawn_subagent)
5815            // error text — never `Error::SubagentNotFound`'s "unknown
5816            // subagent id" text, which would wrongly imply subagents are on
5817            // but this particular id is bogus.
5818            if call.function.name == SUBAGENT_STATUS && self.config.subagents_enabled {
5819                return self.run_subagent_status(call).await;
5820            }
5821            // BP-7: same interception shape and same `subagents_enabled`
5822            // gate as `SUBAGENT_STATUS` above — when the module is off a
5823            // hallucinated call falls through to the ordinary unknown-tool
5824            // error rather than a misleading "unknown subagent id".
5825            if call.function.name == SEND_MESSAGE && self.config.subagents_enabled {
5826                return self.run_send_message(call).await;
5827            }
5828            if call.function.name == SUBAGENT_RESUME && self.config.subagents_enabled {
5829                return self.run_subagent_resume(call).await;
5830            }
5831            match self.prepare_tool_call(call) {
5832                PreparedCall::Done(result) => result,
5833                PreparedCall::Ready { name, args } => {
5834                    // `prepare_tool_call` already confirmed the registry has
5835                    // this tool.
5836                    let tool = self.registry.get(&name).expect("prepared as Ready");
5837                    let (output, is_error) = match tool.execute(args, &self.ctx).await {
5838                        Ok(out) => (out, false),
5839                        Err(e) => (format!("Error: {e}"), true),
5840                    };
5841                    if let Some(hook) = &self.config.post_tool_hook {
5842                        hook(&name, &output, is_error);
5843                    }
5844                    (output, is_error)
5845                }
5846            }
5847        })
5848    }
5849
5850    /// P4e (§3.1 `core.parallel_tool_calls`, catalog:59): the SYNCHRONOUS
5851    /// half of dispatching one tool call — everything `Self::run_tool` did
5852    /// BEFORE its single `tool.execute(...).await`, factored out so
5853    /// [`Self::run_tools_concurrently`] can run these cheap, stateful,
5854    /// `&mut self` checks (agent intrinsics, unknown-tool, approval,
5855    /// doom-loop, pre-tool-hook) SEQUENTIALLY and in ORIGINAL call order —
5856    /// exactly as `run_tool` always has — before handing the remaining
5857    /// calls' `execute()` futures to `join_all`. `Self::run_tool` itself is
5858    /// now a thin wrapper over this (a pure refactor: byte-identical
5859    /// observable behavior, verified by the existing test suite).
5860    fn prepare_tool_call(&mut self, call: &supercode_interchange::ToolCall) -> PreparedCall {
5861        let name = &call.function.name;
5862        // BP-3 (catalog row "Context-budget tools"): hand the model's
5863        // `get_context_remaining` the agent's OWN accounting — BP-4's
5864        // [`Self::context_usage`], the same struct `/context` prints and
5865        // the same estimates the context guard enforces, so what the model
5866        // reads and what refuses an oversized turn can never disagree.
5867        // Computed at the moment the question is asked (the freshest
5868        // possible view) and ONLY then: `context_usage` projects the whole
5869        // request view, which is not a cost to pay on unrelated calls.
5870        if name == crate::tools::GET_CONTEXT_REMAINING {
5871            if let Ok(usage) = serde_json::to_value(self.context_usage()) {
5872                self.ctx.context_budget.publish(usage);
5873            }
5874        }
5875        if name == TOOL_SEARCH {
5876            // Agent intrinsic (B6): intercepted before registry lookup, since
5877            // `Tool::execute` has no access to the registry or `activated_tools`.
5878            return PreparedCall::Done(self.run_tool_search(call));
5879        }
5880        if name == EXPAND_REDUCTION {
5881            // Agent intrinsic (T12/TR-1): intercepted before registry lookup,
5882            // same reason — resolves against `self.reduction_log`/`self.history`,
5883            // which `Tool::execute` has no access to.
5884            return PreparedCall::Done(self.run_expand_reduction(call));
5885        }
5886        if name == SIDECAR_SEARCH {
5887            return PreparedCall::Done(self.run_sidecar_search(call));
5888        }
5889        // P5-6 (§2 module 4 `tools.background`): gated at the interception
5890        // point itself (not internally, at each method's own top) —
5891        // mirroring `SUBAGENT_STATUS`'s own fix (Fable-5 review, LOW "wrong
5892        // error when disabled"): a hallucinated call when the module is off
5893        // must fall through to the plain `Error::UnknownTool` path below,
5894        // never a background-specific error that would wrongly imply the
5895        // module is on. Unlike `SPAWN_SUBAGENT`/`SUBAGENT_STATUS`, none of
5896        // these four need async `&mut self` access (spawning a process,
5897        // `Child::try_wait`, and `Child::start_kill` are all synchronous),
5898        // so they're intercepted here in `prepare_tool_call` rather than in
5899        // `Self::run_tool`.
5900        if self.config.tools_background_enabled {
5901            if name == BACKGROUND_EXEC {
5902                return PreparedCall::Done(self.run_background_exec(call));
5903            }
5904            if name == BACKGROUND_STATUS {
5905                return PreparedCall::Done(self.run_background_status(call));
5906            }
5907            if name == BACKGROUND_LIST {
5908                return PreparedCall::Done(self.run_background_list(call));
5909            }
5910            if name == BACKGROUND_KILL {
5911                return PreparedCall::Done(self.run_background_kill(call));
5912            }
5913        }
5914        if self.registry.get(name).is_none() {
5915            let err = Error::UnknownTool(name.clone());
5916            return PreparedCall::Done((format!("Error: {err}"), true));
5917        }
5918
5919        // P5-1 (§2 modules 10-11, integration point named in
5920        // COMPOSABLE-HARNESS-DESIGN.md's activation set): the permissions
5921        // ENGINE governs the gate when `capabilities.permissions.enabled`
5922        // is on; every other config resolves this to `false`
5923        // (`Config::default`), which takes the `else` branch below —
5924        // the EXACT pre-P5-1 code, untouched, so the default posture
5925        // (approval=never/sandbox=none) and every existing test's observed
5926        // behavior is byte-for-byte unchanged.
5927        if self.config.permissions_enabled {
5928            // The engine needs the command/path TEXT the legacy tool-name-
5929            // only gate below never looked at, so args must be parsed
5930            // BEFORE the gate here (not after, like the legacy branch).
5931            let args = match call.function.parsed_arguments() {
5932                Ok(v) => v,
5933                Err(e) => {
5934                    let err = Error::InvalidArguments {
5935                        tool: name.clone(),
5936                        message: e.to_string(),
5937                    };
5938                    return PreparedCall::Done((format!("Error: {err}"), true));
5939                }
5940            };
5941            // P4c doom-loop breaker, unchanged, still before any gate.
5942            if let Some(reason) = self.check_doom_loop(name, &args) {
5943                return PreparedCall::Done((format!("Error: {reason}"), true));
5944            }
5945            // BP-10: the hook runs BEFORE the engine on this path, so its
5946            // rewrite is what the rules see and its allow/ask/deny is a
5947            // tier inside them — see `run_pre_tool_hook`.
5948            let (args, hook_decision) = match self.run_pre_tool_hook(name, args) {
5949                Ok(pair) => pair,
5950                Err(done) => return done,
5951            };
5952            if let Some(reason) = self.permissions_gate_denial(name, &args, hook_decision) {
5953                return PreparedCall::Done((format!("Error: {reason}"), true));
5954            }
5955            PreparedCall::Ready {
5956                name: name.clone(),
5957                args,
5958            }
5959        } else {
5960            // ---- pre-P5-1 gate, byte-for-byte unchanged ----
5961            // Approval gate: if the policy requires it, consult the handler
5962            // (absent handler denies, so an OnRequest/Untrusted policy is
5963            // fail-closed).
5964            if self.config.needs_approval(name) {
5965                let approved = self
5966                    .config
5967                    .approval_handler
5968                    .as_ref()
5969                    .map(|h| h(call))
5970                    .unwrap_or(false);
5971                if !approved {
5972                    return PreparedCall::Done((
5973                        format!("Error: tool `{name}` was not approved for execution"),
5974                        true,
5975                    ));
5976                }
5977            }
5978            let args = match call.function.parsed_arguments() {
5979                Ok(v) => v,
5980                Err(e) => {
5981                    let err = Error::InvalidArguments {
5982                        tool: name.clone(),
5983                        message: e.to_string(),
5984                    };
5985                    return PreparedCall::Done((format!("Error: {err}"), true));
5986                }
5987            };
5988            self.finish_prepare(name.clone(), args)
5989        }
5990    }
5991
5992    /// P5-1: the shared tail of [`Self::prepare_tool_call`] — doom-loop
5993    /// check, pre-tool hook, `Ready` construction — factored out so both
5994    /// the legacy gate and the new permissions-engine gate run the exact
5995    /// same downstream checks in the exact same order (§5.3 risk 1: the
5996    /// permissions engine changes WHO gets to run, never what happens once
5997    /// they're approved).
5998    fn finish_prepare(&mut self, name: String, args: serde_json::Value) -> PreparedCall {
5999        // P4c (§5.2 P4 "doom-loop breaker", oc UNIQUE `doom_loop` row,
6000        // catalog D3): a default, always-available veto point distinct from
6001        // `Config.pre_tool_hook` (a single user-installable slot — the
6002        // breaker must coexist with a caller's own hook, not compete for the
6003        // one slot). `None`/`Some(0|1)` is a no-op — byte-identical to
6004        // today (no repetition tracking, no call is ever refused on this
6005        // basis).
6006        if let Some(reason) = self.check_doom_loop(&name, &args) {
6007            return PreparedCall::Done((format!("Error: {reason}"), true));
6008        }
6009        // Pre-tool hook may block the call. BP-10: the pre-P5-1 gate has
6010        // no permissions engine for a hook's `Allow`/`Ask` to be a tier
6011        // OF, so only the deny half can mean anything here — an
6012        // `updated_args` rewrite still applies (it is a property of the
6013        // call, not of any gate), and `Allow`/`Ask` are no-ops, exactly
6014        // as `None` was before BP-10.
6015        let mut args = args;
6016        if let Some(hook) = &self.config.pre_tool_hook {
6017            let outcome = hook(&name, &args);
6018            if let Some(rewritten) = outcome.updated_args {
6019                args = rewritten;
6020            }
6021            if outcome.decision == crate::config::HookDecision::Deny {
6022                let reason = outcome.reason.unwrap_or_else(|| "denied".to_string());
6023                return PreparedCall::Done((
6024                    format!("Error: blocked by pre-tool hook: {reason}"),
6025                    true,
6026                ));
6027            }
6028        }
6029        PreparedCall::Ready { name, args }
6030    }
6031
6032    /// BP-10 (catalog row "Hook/plugin permission veto"): fire the pre-tool
6033    /// hook for the permissions-engine path, where it runs BEFORE the gate
6034    /// (CC's own order: a `PreToolUse` hook answers the permission question
6035    /// rather than being asked after it). Returns the possibly-REWRITTEN
6036    /// arguments plus the [`crate::config::HookDecision`] the engine folds
6037    /// in, or the finished denial when the hook refused outright.
6038    ///
6039    /// The rewrite lands BEFORE the gate deliberately: the engine must
6040    /// evaluate what will actually run, so a hook cannot launder a denied
6041    /// command by rewriting it past the rules.
6042    #[allow(clippy::type_complexity)]
6043    fn run_pre_tool_hook(
6044        &self,
6045        name: &str,
6046        args: serde_json::Value,
6047    ) -> std::result::Result<(serde_json::Value, crate::config::HookDecision), PreparedCall> {
6048        let Some(hook) = &self.config.pre_tool_hook else {
6049            return Ok((args, crate::config::HookDecision::Pass));
6050        };
6051        let outcome = hook(name, &args);
6052        let args = outcome.updated_args.unwrap_or(args);
6053        if outcome.decision == crate::config::HookDecision::Deny {
6054            let reason = outcome.reason.unwrap_or_else(|| "denied".to_string());
6055            return Err(PreparedCall::Done((
6056                format!("Error: blocked by pre-tool hook: {reason}"),
6057                true,
6058            )));
6059        }
6060        Ok((args, outcome.decision))
6061    }
6062
6063    /// D-2 (Fable-5 delta review — LOW-MEDIUM, "over-grant residual"): the
6064    /// built-in tools whose `command`/`path`/`patch` arg IS semantically
6065    /// the whole call — the ONLY tools [`Self::permissions_gate_denial`]
6066    /// is allowed to turn into an [`permissions::ApprovalRequest::subject`]
6067    /// (see that method's own doc comment on the `subject` line for the
6068    /// full story). Bash-family (`bash`, and `shell` —
6069    /// [`crate::tools::builtins::PersistentShellTool`]'s registered name,
6070    /// what the review's "persistent-shell" refers to), the file tools
6071    /// (`read_file`/`write_file`/`edit_file`/`view_image`, whose `path` IS
6072    /// the subject), and `apply_patch` (whose `patch` envelope is handled
6073    /// separately but is unconditionally this tool only, see the `patch`
6074    /// local a few lines below). Deliberately NOT `list_dir`/`glob`/
6075    /// `search` — this crate's F2 fix (`ApprovalCache::key_for_request`)
6076    /// already falls back to a full-args digest for anything not on this
6077    /// list, which is a strictly SAFER (if slightly less cache-granular)
6078    /// default than guessing at more built-ins that weren't part of this
6079    /// finding.
6080    const SUBJECT_BEARING_BUILTIN_TOOLS: &'static [&'static str] = &[
6081        "bash",
6082        "shell",
6083        "read_file",
6084        "write_file",
6085        "edit_file",
6086        "view_image",
6087        "apply_patch",
6088    ];
6089
6090    /// P5-1: evaluate `name`'s call (with parsed `args`) against the
6091    /// permissions engine (`crate::permissions`) — builds the
6092    /// [`crate::permissions::RuleSet`] from `Config`'s deny/ask/allow
6093    /// pattern lists (folding [`Config::permissions_protected_paths`] into
6094    /// the `deny` tier, module 13), picks a command-, path-, or name-only
6095    /// evaluation depending on what `args` carries, resolves an `Ask`
6096    /// decision via the session cache + THIS agent's own installed
6097    /// [`crate::permissions::PermissionsApprovalHandler`], and returns
6098    /// `Some(reason)` when the call is refused (`None` = proceed). Thin
6099    /// wrapper over [`Self::permissions_gate_denial_impl`] — see that
6100    /// method's doc comment for why the handler is a parameter there.
6101    fn permissions_gate_denial(
6102        &self,
6103        name: &str,
6104        args: &serde_json::Value,
6105        hook: crate::config::HookDecision,
6106    ) -> Option<String> {
6107        self.permissions_gate_denial_impl(
6108            name,
6109            args,
6110            self.permissions_approval_handler.as_deref(),
6111            hook,
6112        )
6113    }
6114
6115    /// P5-6 (§2.2 C6, build brief "wire to the P5-1 engine's non-
6116    /// interactive fail-closed path"): the SAME rule-evaluation body as
6117    /// [`Self::permissions_gate_denial`], but the approval `handler` is a
6118    /// PARAMETER instead of always reading `self.permissions_approval_handler`
6119    /// — `Agent::background_permission_denial` calls this with `handler:
6120    /// None` (or a [`crate::subagents::ParentQueueApprovalHandler`]) so a
6121    /// `background_exec` call's `Ask`-tier decisions resolve exactly like a
6122    /// P5-3 background child's do (`crate::permissions::resolve_ask`'s
6123    /// pre-existing "no handler ⇒ deny" contract), REGARDLESS of whether
6124    /// this agent itself has an interactive handler installed for its own
6125    /// foreground calls — a background job must never block on a prompt it
6126    /// has no way to answer, even if the agent hosting it could otherwise
6127    /// answer one. The rule SET and default-policy baseline are otherwise
6128    /// identical to a foreground call's — only how an `Ask` decision
6129    /// resolves ever differs, and only in the strictly-narrower direction
6130    /// (never escalates past what a foreground call of the same command is
6131    /// allowed).
6132    fn permissions_gate_denial_impl(
6133        &self,
6134        name: &str,
6135        args: &serde_json::Value,
6136        handler: Option<&dyn crate::permissions::PermissionsApprovalHandler>,
6137        hook: crate::config::HookDecision,
6138    ) -> Option<String> {
6139        use crate::permissions::{self, Decision, PathKind};
6140
6141        // `Config.tool_deny_patterns`/`tool_allow_patterns` (P4a) ARE the
6142        // engine's deny/allow tiers — the same `capabilities.permissions.
6143        // rules.deny`/`.allow` keys, one source of truth, no duplication.
6144        // Protected paths (module 13) are an unconditional deny floor,
6145        // folded in here rather than checked separately, so they benefit
6146        // from the SAME first-match deny-wins priority every other deny
6147        // rule gets.
6148        // BP-5: the rule set itself is `permissions::rules_for_config` —
6149        // ONE construction shared with every other surface that has to ask
6150        // this engine a question (see that function's doc comment). Plan
6151        // mode's narrowing is layered on top of it here, because it is
6152        // this agent's live state, not the config's.
6153        let mut rules = permissions::rules_for_config(&self.config);
6154        // BP-3 (§2 module 8 `plan_mode`, dependency edge `plan_mode →
6155        // permissions.rules|sandbox`): the read-only research phase IS a
6156        // narrowing of this rule set — while the mode is active the write
6157        // and execution tools join the deny tier, and get the same
6158        // first-match, never-overridable treatment every other deny rule
6159        // gets. Inactive (the default, and the only state a config without
6160        // the module can reach) contributes NOTHING, so the rule set is
6161        // byte-identical to before.
6162        rules
6163            .deny
6164            .extend(crate::tools::plan_mode::deny_rules(&self.ctx.plan_mode));
6165
6166        // The baseline decision when NO rule matches at all — derived from
6167        // `ApprovalPolicy`, the same per-policy shape
6168        // `Config::needs_approval` uses for the legacy gate (see
6169        // `ApprovalPolicy::ModelRequested`'s doc comment for why this
6170        // richer gate approximates Codex's real "mostly silent" posture
6171        // instead of that method's conservative OnRequest-alike treatment
6172        // — explicit deny/ask rules still apply on top regardless).
6173        let default = permissions::default_decision(&self.config, name);
6174
6175        // BP-10: path rules are evaluated relative to EVERY granted root
6176        // (cwd + `core.additional_dirs`/`--add-dir`), folded to the
6177        // strictest — see `permissions::evaluate_path_safe_roots`'s doc
6178        // comment for why a grant must not also remove the root-relative
6179        // protected-path floor inside the granted directory. With no extra
6180        // dirs (the default) this is the single `cwd` list, byte-identical
6181        // to before.
6182        let mut roots = vec![self.config.cwd.clone()];
6183        roots.extend(self.config.additional_dirs.iter().cloned());
6184
6185        let command = args.get("command").and_then(|v| v.as_str());
6186        let path = args.get("path").and_then(|v| v.as_str());
6187        // F4 (Fable-5 adversarial review): `apply_patch`'s args carry a
6188        // patch ENVELOPE body (`args["patch"]`), not a `command` or a
6189        // `path` — the two branches above never fire for it, which is
6190        // exactly how a patch touching a protected path bypassed
6191        // `protected_paths` entirely. Only consulted for the `apply_patch`
6192        // tool specifically (a `patch`-shaped arg on some other tool is not
6193        // this envelope format and isn't given this treatment).
6194        let patch = (name == "apply_patch")
6195            .then(|| args.get("patch").and_then(|v| v.as_str()))
6196            .flatten();
6197        let decision = if let Some(command) = command {
6198            permissions::evaluate_command(&rules, name, command, default)
6199        } else if let Some(patch) = patch {
6200            // Same dual-check shape as the path branch below (pseudo-tool
6201            // `write(...)` rules from `protected_paths`, AND a rule
6202            // authored against the real `apply_patch` tool name), applied
6203            // to EVERY path the envelope's ops touch (`Add`/`Delete`/
6204            // `Update`'s `path`, plus `*** Move to:`). A patch that fails
6205            // to parse can't be proven to avoid a protected path — fail
6206            // closed to at least `Ask`, the same floor an unparseable bash
6207            // command gets in `permissions::evaluate_command`, rather than
6208            // silently let it through on `default`.
6209            let mut d = rules.evaluate(name, None).unwrap_or(default);
6210            match crate::tools::patch_target_paths(patch) {
6211                Ok(paths) => {
6212                    for p in &paths {
6213                        // SECURITY (CRITICAL fix): route both checks through
6214                        // the safe-path-resolving variants — a patch target
6215                        // like `x/../.git/config` must be caught exactly
6216                        // like a `write_file`/`edit_file` `path` argument
6217                        // would be (see `evaluate_path_safe`'s doc comment).
6218                        let pseudo = permissions::evaluate_path_safe_roots(
6219                            &rules,
6220                            PathKind::Write,
6221                            &roots,
6222                            p,
6223                            Decision::Allow,
6224                        );
6225                        let real_tool = permissions::evaluate_path_subject_safe_roots(
6226                            &rules,
6227                            name,
6228                            &roots,
6229                            p,
6230                            Decision::Allow,
6231                        );
6232                        d = d.stricter(pseudo).stricter(real_tool);
6233                    }
6234                }
6235                Err(_) => {
6236                    d = d.stricter(Decision::Ask);
6237                }
6238            }
6239            d
6240        } else if let Some(path) = path {
6241            let kind = if matches!(name, "write_file" | "edit_file") {
6242                PathKind::Write
6243            } else {
6244                PathKind::Read
6245            };
6246            // TWO independent sources of path-shaped rules can apply to the
6247            // same call, and BOTH must be checked:
6248            // (a) the `read(...)`/`write(...)` pseudo-tool (tool-agnostic —
6249            //     applies no matter WHICH tool touches the path; this is
6250            //     what `Config::permissions_protected_paths`/module 13
6251            //     expands into, via `protected_path_deny_rules`);
6252            // (b) a rule authored against the REAL tool name with the path
6253            //     as its subject — design §4.4's own oc-parity worked
6254            //     example writes exactly this shape (`"read_file(*.env)"`,
6255            //     not a pseudo-tool), matching how `bash(cmdglob)` rules
6256            //     are authored. `RuleSet::evaluate`'s bare-tool-name-glob
6257            //     branch (no parens) ALSO fires here regardless of
6258            //     `subject`, so this one call additionally covers a
6259            //     blanket "deny this tool entirely" rule — no separate
6260            //     `rules.evaluate(name, None)` call is needed.
6261            // SECURITY (CRITICAL fix, guarantor audit): both checks now
6262            // route through the safe-path-resolving variants (see
6263            // `evaluate_path_safe`'s doc comment) instead of glob-matching
6264            // the raw model-supplied `path` string directly — this is what
6265            // closes the traversal bypass (`write_file
6266            // path="x/../.git/config"`) and the analogous symlink escape.
6267            let pseudo_decision =
6268                permissions::evaluate_path_safe_roots(&rules, kind, &roots, path, default);
6269            let real_tool_decision =
6270                permissions::evaluate_path_subject_safe_roots(&rules, name, &roots, path, default);
6271            pseudo_decision.stricter(real_tool_decision)
6272        } else {
6273            rules.evaluate(name, None).unwrap_or(default)
6274        };
6275
6276        // BP-10 (catalog row "Hook/plugin permission veto"): the hook's
6277        // verdict is a TIER of this engine, folded in the one direction
6278        // that is always safe — `Ask` tightens (`stricter` never loosens),
6279        // `Deny` is a floor. `Allow` deliberately does NOT change the
6280        // decision here: it answers the `Ask` tier below (the
6281        // `PermissionRequest`-class reply CC's hooks give on the user's
6282        // behalf), so a `deny` rule still refuses the call outright — a
6283        // hook may skip a prompt, never a floor.
6284        let decision = match hook {
6285            crate::config::HookDecision::Deny => Decision::Deny,
6286            crate::config::HookDecision::Ask => decision.stricter(Decision::Ask),
6287            crate::config::HookDecision::Allow | crate::config::HookDecision::Pass => decision,
6288        };
6289
6290        // BP-10 (catalog row "Sandbox-escalation path", cx§4
6291        // `sandbox_permissions: "require_escalated"` + justification): the
6292        // model's channel to ASK for an unsandboxed run is a rule inside
6293        // this one engine, not a switch beside it. A call carrying
6294        // `with_escalated_permissions: true` is forced to at least the
6295        // `Ask` tier — never below whatever the rules already decided, so
6296        // a denied command cannot escalate its way out (`stricter` only
6297        // tightens), and never silently allowed under
6298        // `ApprovalPolicy::ModelRequested`/`Never`, whose `Allow` baseline
6299        // is exactly what made "the model requests escalation" a no-op
6300        // before. The `justification` rides in `raw_args` below, so the
6301        // approval door shows the user the model's own reason.
6302        let decision = if args
6303            .get("with_escalated_permissions")
6304            .and_then(|v| v.as_bool())
6305            .unwrap_or(false)
6306        {
6307            decision.stricter(Decision::Ask)
6308        } else {
6309            decision
6310        };
6311
6312        // D-2 (Fable-5 delta review — LOW-MEDIUM): `command`/`path`/`patch`
6313        // above are extracted (and used to DRIVE the decision above) for
6314        // ANY tool that happens to carry one of those arg names — that
6315        // part is unchanged and correct (a rule authored against, say, an
6316        // MCP tool's own name legitimately wants to glob-match its
6317        // `command`-shaped arg too). But the narrower single-field
6318        // `subject` handed to the cache/handler below must NOT do the
6319        // same for a non-built-in tool: an MCP (or other) tool's
6320        // `command`/`path` is just one field among potentially several
6321        // that together define what the call actually does — collapsing
6322        // an `AllowForSession` grant down to that one field would silently
6323        // auto-allow a later call with the SAME `command` but different
6324        // OTHER args (e.g. `{"command":"sync","target":"staging"}`
6325        // auto-allowing `{"command":"sync","target":"production"}`).
6326        // Restricting this to the known built-ins whose `subject` really
6327        // IS the whole call leaves every other tool with `subject: None`,
6328        // which routes it through `ApprovalCache::key_for_request`'s
6329        // full-args-digest fallback (F2) instead.
6330        let subject = Self::SUBJECT_BEARING_BUILTIN_TOOLS
6331            .contains(&name)
6332            .then(|| command.or(path).or(patch))
6333            .flatten();
6334        let req = permissions::ApprovalRequest {
6335            tool: name,
6336            subject,
6337            raw_args: args,
6338        };
6339        let approved = permissions::decision_to_approved(decision, || {
6340            // BP-10: a hook `Allow` answers this ask without a prompt (and
6341            // without a cache entry — the hook is consulted on every call,
6342            // so caching its answer would be a second, staler copy of the
6343            // same decision).
6344            if hook == crate::config::HookDecision::Allow {
6345                return true;
6346            }
6347            permissions::resolve_ask(&self.permissions_approval_cache, handler, &req)
6348        });
6349        if approved {
6350            None
6351        } else {
6352            Some(format!(
6353                "tool `{name}` was not approved for execution (permissions engine: {decision:?})"
6354            ))
6355        }
6356    }
6357
6358    /// P5-6 (§2.2 C6, build brief "a bg `rm -rf` subject to the same deny
6359    /// rules... must never escalate past what a foreground exec of the
6360    /// same command is allowed"): the permission gate `background_exec`
6361    /// runs BEFORE spawning anything. Evaluated against the tool name
6362    /// `"bash"` (not `"background_exec"`) deliberately — so any
6363    /// `bash(...)`-authored deny/ask/allow rule (or protected-path floor)
6364    /// applies to a background command byte-for-byte identically to a
6365    /// foreground `bash` call, the SAME rule set + default baseline
6366    /// [`Self::permissions_gate_denial`] would use for one.
6367    ///
6368    /// The one deliberate difference (C6 itself): an `Ask`-tier decision
6369    /// NEVER reaches an interactive handler here — a background job has no
6370    /// way to block on a prompt it can't answer. When
6371    /// [`Config::subagents_background_prompts`] is
6372    /// [`crate::subagents::BackgroundPromptsPolicy::Parent`], the denied
6373    /// request is additionally queued onto [`Self::pending_child_approvals`]
6374    /// (via [`crate::subagents::ParentQueueApprovalHandler`], reused
6375    /// verbatim — the SAME "parent-surfaced queue" §2.2 C6 names for
6376    /// `subagents.background`, with the job id standing in for a child
6377    /// agent id) for later inspection; any other configuration (including
6378    /// no `background_prompts` set at all) resolves via `handler: None` —
6379    /// [`crate::permissions::resolve_ask`]'s pre-existing "no handler ⇒
6380    /// deny" fail-closed default, identical to `subagents`'s own
6381    /// `AutoPolicy` reading. Either way, `Ask` always denies; only `Allow`
6382    /// (from the rule engine itself, or a PRIOR interactively-granted
6383    /// `AllowForSession` cache entry) ever lets a background command run —
6384    /// so this can only ever be as-or-more restrictive than a foreground
6385    /// call, never looser, regardless of configuration.
6386    ///
6387    /// Covers BOTH gate generations: when [`Config::permissions_enabled`]
6388    /// is on, the P5-1 engine (above) is used; otherwise the legacy
6389    /// [`Config::needs_approval`] gate is consulted but its
6390    /// `approval_handler` closure is NEVER invoked (that closure could
6391    /// itself block, e.g. a real interactive prompt) — an approval-required
6392    /// legacy policy simply denies a background command outright, the same
6393    /// never-hang guarantee under the older gate.
6394    fn background_permission_denial(
6395        &self,
6396        command: &str,
6397        job_id: &str,
6398        hook: crate::config::HookDecision,
6399    ) -> Option<String> {
6400        let args = serde_json::json!({ "command": command });
6401        if self.config.permissions_enabled {
6402            if let Some(crate::subagents::BackgroundPromptsPolicy::Parent) =
6403                self.config.subagents_background_prompts
6404            {
6405                let handler = crate::subagents::ParentQueueApprovalHandler {
6406                    child_agent_id: format!("bg:{job_id}"),
6407                    queue: self.pending_child_approvals.clone(),
6408                };
6409                self.permissions_gate_denial_impl("bash", &args, Some(&handler), hook)
6410            } else {
6411                self.permissions_gate_denial_impl("bash", &args, None, hook)
6412            }
6413        } else if self.config.needs_approval("bash") {
6414            Some(
6415                "tool `bash` requires approval, which a background job cannot request \
6416                 interactively (§2.2 C6: auto-policy denies)"
6417                    .to_string(),
6418            )
6419        } else {
6420            None
6421        }
6422    }
6423
6424    /// P4e (§3.1 `core.parallel_tool_calls`, catalog:59 "Independent
6425    /// sibling calls run concurrently"): runs `calls`' `Tool::execute()`
6426    /// futures CONCURRENTLY via `futures::future::join_all`, for whichever
6427    /// calls [`Self::prepare_tool_call`] resolves to [`PreparedCall::Ready`]
6428    /// — i.e. every plain (non-intrinsic) registry-tool call that passes
6429    /// its synchronous approval/doom-loop/pre-tool-hook checks. A call that
6430    /// resolves to [`PreparedCall::Done`] (an intrinsic, an unknown tool, a
6431    /// denied/blocked call) is NOT parallelized — its result is already in
6432    /// hand from the synchronous prepare pass. Every prepare check still
6433    /// runs sequentially, in original call order, before ANY `execute()`
6434    /// future starts (only the actual tool I/O overlaps) — so doom-loop
6435    /// bookkeeping and pre-tool-hook vetoes see the exact same call order
6436    /// they would under the sequential path. Returns results in the SAME
6437    /// order as `calls`, so callers can always `zip` the two. Post-tool
6438    /// hooks fire per call, in original order, once every result is in
6439    /// hand — a caller-visible timing difference from the sequential path
6440    /// ONLY when this method runs at all (i.e. only when
6441    /// `Config::parallel_tool_calls` is on): hooks see "this batch
6442    /// finished" ordering rather than "this one call finished" ordering.
6443    /// Documented, not a bug.
6444    async fn run_tools_concurrently(
6445        &mut self,
6446        calls: &[supercode_interchange::ToolCall],
6447    ) -> Vec<(String, bool)> {
6448        // P5-3: `spawn_subagent`/`subagent_status` need sequential `&mut
6449        // self` access `prepare_tool_call`'s synchronous-only signature
6450        // can't give them (see `Self::run_tool`'s identical interception).
6451        // A batch that includes one falls back to dispatching the WHOLE
6452        // batch sequentially via `Self::run_tool` — a documented, narrow
6453        // simplification (not a partial-parallelization attempt) rather
6454        // than restructuring `PreparedCall` to carry a future; a batch with
6455        // no subagent intrinsic is completely unaffected and still
6456        // parallelizes exactly as before.
6457        if calls.iter().any(|c| {
6458            c.function.name == SPAWN_SUBAGENT
6459                || c.function.name == SUBAGENT_STATUS
6460                || c.function.name == SEND_MESSAGE
6461                || c.function.name == SUBAGENT_RESUME
6462                || (self.config.claude_runtime_tools_enabled
6463                    && matches!(
6464                        c.function.name.as_str(),
6465                        CLAUDE_CRON_CREATE
6466                            | CLAUDE_CRON_DELETE
6467                            | CLAUDE_CRON_LIST
6468                            | CLAUDE_SCHEDULE_WAKEUP
6469                    ))
6470        }) {
6471            let mut out = Vec::with_capacity(calls.len());
6472            for call in calls {
6473                out.push(self.run_tool(call).await);
6474            }
6475            return out;
6476        }
6477        let prepared: Vec<PreparedCall> = calls.iter().map(|c| self.prepare_tool_call(c)).collect();
6478        let mut slots: Vec<Option<(String, bool)>> = prepared
6479            .iter()
6480            .map(|p| match p {
6481                PreparedCall::Done(r) => Some(r.clone()),
6482                PreparedCall::Ready { .. } => None,
6483            })
6484            .collect();
6485
6486        let ready_idxs: Vec<usize> = prepared
6487            .iter()
6488            .enumerate()
6489            .filter(|(_, p)| matches!(p, PreparedCall::Ready { .. }))
6490            .map(|(i, _)| i)
6491            .collect();
6492
6493        if !ready_idxs.is_empty() {
6494            let futs = ready_idxs.iter().map(|&i| {
6495                let PreparedCall::Ready { name, args } = &prepared[i] else {
6496                    unreachable!("filtered to Ready above")
6497                };
6498                // `self.registry.get` borrows `self.registry` immutably;
6499                // `self.ctx` is `Clone` (P4c precedent) so each future owns
6500                // its own copy rather than borrowing `self` across the
6501                // `.await` inside `join_all`.
6502                let tool = self.registry.get(name).expect("prepared as Ready");
6503                let args = args.clone();
6504                let ctx = self.ctx.clone();
6505                async move {
6506                    match tool.execute(args, &ctx).await {
6507                        Ok(out) => (out, false),
6508                        Err(e) => (format!("Error: {e}"), true),
6509                    }
6510                }
6511            });
6512            let results = futures::future::join_all(futs).await;
6513            for (idx, result) in ready_idxs.iter().zip(results) {
6514                slots[*idx] = Some(result);
6515            }
6516        }
6517
6518        let out: Vec<(String, bool)> = slots
6519            .into_iter()
6520            .map(|s| s.expect("every call resolved to Some above"))
6521            .collect();
6522        // Post-tool hook, in original order — only for calls that actually
6523        // reached `execute()` (matches `run_tool`'s existing behavior: an
6524        // intrinsic/denied/blocked call never fires the post-tool hook).
6525        let ready_set: std::collections::HashSet<usize> = ready_idxs.into_iter().collect();
6526        for (i, call) in calls.iter().enumerate() {
6527            if !ready_set.contains(&i) {
6528                continue;
6529            }
6530            let (output, is_error) = &out[i];
6531            if let Some(hook) = &self.config.post_tool_hook {
6532                hook(&call.function.name, output, *is_error);
6533            }
6534        }
6535        out
6536    }
6537
6538    /// P4c (§5.2 P4 "doom-loop breaker", §3.1 `core.doom_loop_threshold`):
6539    /// update the consecutive-identical-call streak for `(name, args)` and
6540    /// return `Some(reason)` the moment the streak reaches
6541    /// `Config.doom_loop_threshold` (a call whose name AND JSON-canonical
6542    /// arguments are byte-identical to the immediately preceding call
6543    /// extends the streak; anything else resets it to 1). `None`
6544    /// (`Config.doom_loop_threshold` unset, or `Some(n)` with `n < 2` — a
6545    /// threshold below 2 can never fire since the FIRST call already
6546    /// "repeats zero times") never touches the streak fields at all.
6547    fn check_doom_loop(&mut self, name: &str, args: &serde_json::Value) -> Option<String> {
6548        let threshold = self.config.doom_loop_threshold?;
6549        if threshold < 2 {
6550            return None;
6551        }
6552        // `serde_json::Value::Object` is a `BTreeMap` in this workspace (no
6553        // `preserve_order` feature), so `to_string()` is already
6554        // key-order-canonical — two calls that differ only in argument key
6555        // order are still treated as identical.
6556        let key = (name.to_string(), args.to_string());
6557        if self.doom_loop_last_call.as_ref() == Some(&key) {
6558            self.doom_loop_streak += 1;
6559        } else {
6560            self.doom_loop_last_call = Some(key);
6561            self.doom_loop_streak = 1;
6562        }
6563        if self.doom_loop_streak >= threshold {
6564            Some(format!(
6565                "doom-loop breaker: `{name}` called with identical arguments {} times in a row \
6566                 — try a different approach instead of repeating the same call",
6567                self.doom_loop_streak
6568            ))
6569        } else {
6570            None
6571        }
6572    }
6573
6574    /// Whether `name` is in the eagerly-advertised "core" set for the current
6575    /// [`ToolAdvertising`] mode: every enabled tool under `Full`, or the
6576    /// explicit `core` allowlist under `Deferred`.
6577    fn is_core_tool(&self, name: &str) -> bool {
6578        match &self.config.tool_advertising {
6579            ToolAdvertising::Full => true,
6580            ToolAdvertising::Deferred { core } => core.iter().any(|c| c == name),
6581        }
6582    }
6583
6584    /// The schema advertised on the wire for `t`: the raw (as-shipped)
6585    /// schema with TR-8/T5's per-tool schema tier applied. This is what
6586    /// [`Self::tool_schemas`] sends every request.
6587    fn schema_for(&self, t: &dyn crate::tools::Tool) -> ToolSchema {
6588        let raw = self.raw_schema_for(t);
6589        let tier = self.config.schema_tier_for(t.name());
6590        let (description, parameters) =
6591            crate::tools::tiers::minify(&raw.description, &raw.parameters, tier);
6592        ToolSchema {
6593            name: raw.name,
6594            description,
6595            parameters,
6596        }
6597    }
6598
6599    /// The ORIGINAL, as-shipped schema for `t` — never tier-minified. This is
6600    /// the full contract [`Self::run_tool_search`] hands back on activation
6601    /// (TR-8/T5 dev/03: the B6 fetch path is the invert of tiering, so a
6602    /// model that fetched a tool via `tool_search` always sees the complete
6603    /// schema, byte-equal to `t.description()`/`t.parameters()` — modulo the
6604    /// pre-existing [`crate::Config::tool_description`] override, which is
6605    /// orthogonal to tiering).
6606    fn raw_schema_for(&self, t: &dyn crate::tools::Tool) -> ToolSchema {
6607        ToolSchema {
6608            name: t.name().to_string(),
6609            description: self
6610                .config
6611                .tool_description(t.name(), t.description())
6612                .to_string(),
6613            parameters: t.parameters(),
6614        }
6615    }
6616
6617    /// The synthetic `tool_search` schema advertised under `Deferred` (B6).
6618    fn tool_search_schema() -> ToolSchema {
6619        ToolSchema {
6620            name: TOOL_SEARCH.to_string(),
6621            description: "Search for additional tools not currently advertised (the deferred \
6622                MCP surface and any other non-core tools). Matches keywords case-insensitively \
6623                against each tool's name and description. Matched tools become callable starting \
6624                with your NEXT message, not this one."
6625                .to_string(),
6626            parameters: serde_json::json!({
6627                "type": "object",
6628                "properties": {
6629                    "query": {
6630                        "type": "string",
6631                        "description": "Keyword(s) to search for in tool names and descriptions."
6632                    },
6633                    "max_results": {
6634                        "type": "integer",
6635                        "description": "Maximum number of matching tools to return."
6636                    }
6637                },
6638                "required": ["query"],
6639                "additionalProperties": false
6640            }),
6641        }
6642    }
6643
6644    /// The tool-schema array this agent would advertise on its NEXT
6645    /// request, exactly as `Self::run_loop` computes it. Public
6646    /// (PARITY-18 D1) so a caller can measure the real request-token cost
6647    /// of an agent's tool surface — including the current
6648    /// [`crate::config::ToolAdvertising`] mode's core/deferred split and
6649    /// the synthetic `tool_search`/`expand_reduction`/`sidecar_search`
6650    /// schemas — BEFORE ever calling [`Self::send`], e.g. for a preflight
6651    /// context-guard check.
6652    pub fn tool_schemas(&self) -> Vec<ToolSchema> {
6653        let mut out = match &self.config.tool_advertising {
6654            ToolAdvertising::Full => self
6655                .registry
6656                .iter()
6657                .filter(|t| self.config.tool_enabled(t.name()))
6658                .map(|t| self.schema_for(t))
6659                .collect(),
6660            ToolAdvertising::Deferred { .. } => {
6661                let mut out: Vec<ToolSchema> = self
6662                    .registry
6663                    .iter()
6664                    .filter(|t| self.config.tool_enabled(t.name()))
6665                    .filter(|t| {
6666                        self.is_core_tool(t.name()) || self.activated_tools.contains(t.name())
6667                    })
6668                    .map(|t| self.schema_for(t))
6669                    .collect();
6670                out.push(Self::tool_search_schema());
6671                out
6672            }
6673        };
6674        // T12/TR-1: `expand_reduction`/`sidecar_search` are orthogonal to
6675        // `tool_advertising` (which governs the ordinary tool surface) —
6676        // advertised whenever a `ReductionPolicy` is installed, regardless of
6677        // Full/Deferred, since only a reduced session ever has anything to
6678        // expand or search (SPEC.md TR-1 dev/01).
6679        if self.reduction_policy.is_some() {
6680            out.push(Self::expand_reduction_schema());
6681            out.push(Self::sidecar_search_schema());
6682        }
6683        // P5-3 (§2 module 9): `spawn_subagent`/`subagent_status` are
6684        // orthogonal to `tool_advertising` too, same reasoning as
6685        // `expand_reduction`/`sidecar_search` above — advertised whenever
6686        // `Config::subagents_enabled` is on, Full or Deferred alike.
6687        // `false` (the default) never appends either, so a config that
6688        // never turns the module on gets byte-identical tool schemas to
6689        // today.
6690        if self.config.subagents_enabled {
6691            out.push(self.spawn_subagent_schema());
6692            if self.config.subagents_claude_agent_alias {
6693                out.push(self.claude_agent_schema());
6694            }
6695            if self.config.subagents_background {
6696                out.push(Self::subagent_status_schema());
6697                // BP-7 (catalog §4a "Background subagents + resume"): the
6698                // two halves the row named as missing — a mailbox into a
6699                // still-running child, and a resume of a finished one with
6700                // its context intact.
6701                out.push(Self::send_message_schema());
6702                out.push(Self::subagent_resume_schema());
6703            }
6704        }
6705        if self.config.claude_runtime_tools_enabled {
6706            out.extend(self.claude_builtin_tool_schemas());
6707            out.push(Self::claude_cron_create_schema());
6708            out.push(Self::claude_cron_delete_schema());
6709            out.push(Self::claude_cron_list_schema());
6710            out.push(Self::claude_schedule_wakeup_schema());
6711        }
6712        // P5-6 (§2 module 4 `tools.background`): same orthogonal-to-
6713        // `tool_advertising` treatment, advertised whenever
6714        // `Config::tools_background_enabled` is on. `false` (the default)
6715        // never appends any of the four, so a config that never turns the
6716        // module on gets byte-identical tool schemas to today.
6717        if self.config.tools_background_enabled {
6718            out.push(Self::background_exec_schema());
6719            out.push(Self::background_status_schema());
6720            out.push(Self::background_list_schema());
6721            out.push(Self::background_kill_schema());
6722        }
6723        // BP-10 (catalog row "Tool hiding via policy", cc§4 "bare-name
6724        // deny"): a policy deny does not merely REFUSE the call at
6725        // dispatch — it removes the tool from the model's view. Applied
6726        // once, here, over the finished array, so every family appended
6727        // above (`spawn_subagent`, `background_*`, the Claude aliases,
6728        // `expand_reduction`, …) is hidden by the same one rule, not by a
6729        // per-family repeat of it. See [`Self::policy_hides_tool`] for
6730        // which deny tier is consulted and why.
6731        out.retain(|schema| !self.policy_hides_tool(&schema.name));
6732        out
6733    }
6734
6735    /// BP-10: whether the CONFIG-DECLARED deny tier hides `name` from the
6736    /// model's tool surface entirely (cc§4: CC's bare-name deny "removes
6737    /// the tool from the model's view", where an ordinary rule only
6738    /// refuses the call).
6739    ///
6740    /// The ONE engine decides: this is
6741    /// [`crate::permissions::RuleSet::evaluate`] with `subject: None`, so
6742    /// exactly the patterns that can be satisfied by a tool NAME ALONE
6743    /// (`"bash"`, `"mcp_*"`, `"*"`) hide; a rule that names a
6744    /// command/path constraint (`"bash(rm -rf*)"`, `"write(.git/**)"`) is
6745    /// not satisfiable without a subject and therefore never hides a tool
6746    /// — the same `rule_matches` contract the dispatch gate uses.
6747    ///
6748    /// **Which deny tier.** `Config::tool_deny_patterns` — the
6749    /// `capabilities.permissions.rules.deny` array — and NOT the two
6750    /// runtime narrowings the dispatch gate folds in beside it:
6751    /// `protected_paths` expands to `read(...)`/`write(...)` patterns that
6752    /// carry a subject by construction (so they could never match here
6753    /// anyway), and `plan_mode::deny_rules` is a MODE, not a policy — CC's
6754    /// plan mode refuses a write, it does not make Write disappear and
6755    /// reappear as the mode toggles mid-session. Hiding is a property of
6756    /// the configured policy, which is fixed for the run.
6757    ///
6758    /// Gated on [`Config::permissions_enabled`]: a config that never turns
6759    /// the module on gets byte-identical schemas to before this existed.
6760    fn policy_hides_tool(&self, name: &str) -> bool {
6761        if !self.config.permissions_enabled || self.config.tool_deny_patterns.is_empty() {
6762            return false;
6763        }
6764        let rules = crate::permissions::RuleSet {
6765            deny: self.config.tool_deny_patterns.clone(),
6766            ..Default::default()
6767        };
6768        rules.evaluate(name, None) == Some(crate::permissions::Decision::Deny)
6769    }
6770
6771    fn claude_builtin_tool_schemas(&self) -> Vec<ToolSchema> {
6772        let mut schemas = Vec::new();
6773        let mut push = |alias: &str, native: &str, description: &str, parameters| {
6774            if self.registry.get(native).is_some() && self.config.tool_enabled(native) {
6775                schemas.push(ToolSchema {
6776                    name: alias.to_string(),
6777                    description: description.to_string(),
6778                    parameters,
6779                });
6780            }
6781        };
6782        push(
6783            CLAUDE_BASH,
6784            "bash",
6785            "Claude Code-compatible shell command execution.",
6786            serde_json::json!({
6787                "type": "object",
6788                "properties": {
6789                    "command": {"type": "string"},
6790                    "timeout": {"type": "integer", "description": "Timeout in milliseconds."},
6791                    "description": {"type": "string"}
6792                },
6793                "required": ["command"],
6794                "additionalProperties": true
6795            }),
6796        );
6797        push(
6798            CLAUDE_READ,
6799            "read_file",
6800            "Claude Code-compatible file reader.",
6801            serde_json::json!({
6802                "type": "object",
6803                "properties": {
6804                    "file_path": {"type": "string"},
6805                    "offset": {"type": "integer"},
6806                    "limit": {"type": "integer"}
6807                },
6808                "required": ["file_path"],
6809                "additionalProperties": false
6810            }),
6811        );
6812        push(
6813            CLAUDE_WRITE,
6814            "write_file",
6815            "Claude Code-compatible file writer.",
6816            serde_json::json!({
6817                "type": "object",
6818                "properties": {"file_path": {"type": "string"}, "content": {"type": "string"}},
6819                "required": ["file_path", "content"],
6820                "additionalProperties": false
6821            }),
6822        );
6823        push(
6824            CLAUDE_EDIT,
6825            "edit_file",
6826            "Claude Code-compatible exact file edit.",
6827            serde_json::json!({
6828                "type": "object",
6829                "properties": {
6830                    "file_path": {"type": "string"},
6831                    "old_string": {"type": "string"},
6832                    "new_string": {"type": "string"},
6833                    "replace_all": {"type": "boolean"}
6834                },
6835                "required": ["file_path", "old_string", "new_string"],
6836                "additionalProperties": false
6837            }),
6838        );
6839        push(
6840            CLAUDE_GLOB,
6841            "glob",
6842            "Claude Code-compatible file glob.",
6843            serde_json::json!({
6844                "type": "object",
6845                "properties": {"pattern": {"type": "string"}, "path": {"type": "string"}},
6846                "required": ["pattern"],
6847                "additionalProperties": false
6848            }),
6849        );
6850        push(
6851            CLAUDE_GREP,
6852            "search",
6853            "Claude Code-compatible content search.",
6854            serde_json::json!({
6855                "type": "object",
6856                "properties": {"pattern": {"type": "string"}, "path": {"type": "string"}},
6857                "required": ["pattern"],
6858                "additionalProperties": true
6859            }),
6860        );
6861        schemas
6862    }
6863
6864    fn translate_claude_builtin_call(
6865        &self,
6866        call: &supercode_interchange::ToolCall,
6867    ) -> Result<Option<supercode_interchange::ToolCall>> {
6868        let native = match call.function.name.as_str() {
6869            CLAUDE_BASH => "bash",
6870            CLAUDE_READ => "read_file",
6871            CLAUDE_WRITE => "write_file",
6872            CLAUDE_EDIT => "edit_file",
6873            CLAUDE_GLOB => "glob",
6874            CLAUDE_GREP => "search",
6875            _ => return Ok(None),
6876        };
6877        let mut args = call.function.parsed_arguments()?;
6878        let object = args
6879            .as_object_mut()
6880            .ok_or_else(|| Error::InvalidArguments {
6881                tool: call.function.name.clone(),
6882                message: "expected a JSON object".to_string(),
6883            })?;
6884        if let Some(path) = object.remove("file_path") {
6885            object.entry("path".to_string()).or_insert(path);
6886        }
6887        if call.function.name == CLAUDE_BASH {
6888            if let Some(timeout) = object.remove("timeout") {
6889                object.entry("timeout_ms".to_string()).or_insert(timeout);
6890            }
6891        }
6892        if call.function.name == CLAUDE_GLOB {
6893            if let Some(path) = object
6894                .remove("path")
6895                .and_then(|value| value.as_str().map(str::to_owned))
6896            {
6897                if let Some(pattern) = object.get_mut("pattern") {
6898                    if let Some(value) = pattern.as_str() {
6899                        if !std::path::Path::new(value).is_absolute() {
6900                            *pattern = serde_json::Value::String(
6901                                std::path::Path::new(&path)
6902                                    .join(value)
6903                                    .to_string_lossy()
6904                                    .into_owned(),
6905                            );
6906                        }
6907                    }
6908                }
6909            }
6910        }
6911        let mut translated = call.clone();
6912        translated.function.name = native.to_string();
6913        translated.function.arguments = serde_json::to_string(&args)?;
6914        Ok(Some(translated))
6915    }
6916
6917    fn claude_cron_create_schema() -> ToolSchema {
6918        ToolSchema {
6919            name: CLAUDE_CRON_CREATE.to_string(),
6920            description: "Record a Claude-compatible cron job in the imported runtime manifest. \
6921                The job inherits the manifest's ACTIVE or PAUSED posture; an embedding scheduler, \
6922                not this agent loop, owns execution."
6923                .to_string(),
6924            parameters: serde_json::json!({
6925                "type": "object",
6926                "properties": {
6927                    "cron": {"type": "string", "description": "Cron expression to preserve."},
6928                    "prompt": {"type": "string", "description": "Prompt associated with the job."},
6929                    "recurring": {"type": "boolean", "default": false},
6930                    "durable": {"type": "boolean", "default": false}
6931                },
6932                "required": ["cron", "prompt"],
6933                "additionalProperties": false
6934            }),
6935        }
6936    }
6937
6938    fn claude_cron_delete_schema() -> ToolSchema {
6939        ToolSchema {
6940            name: CLAUDE_CRON_DELETE.to_string(),
6941            description: "Delete a Claude-compatible cron job from the imported manifest. \
6942                This updates state only; an embedding scheduler owns execution."
6943                .to_string(),
6944            parameters: serde_json::json!({
6945                "type": "object",
6946                "properties": {"id": {"type": "string"}},
6947                "required": ["id"],
6948                "additionalProperties": false
6949            }),
6950        }
6951    }
6952
6953    fn claude_cron_list_schema() -> ToolSchema {
6954        ToolSchema {
6955            name: CLAUDE_CRON_LIST.to_string(),
6956            description: "List imported Claude cron jobs and their explicit ACTIVE or PAUSED \
6957                manifest posture. This agent loop itself does not run a scheduler."
6958                .to_string(),
6959            parameters: serde_json::json!({
6960                "type": "object",
6961                "properties": {},
6962                "additionalProperties": false
6963            }),
6964        }
6965    }
6966
6967    fn claude_schedule_wakeup_schema() -> ToolSchema {
6968        ToolSchema {
6969            name: CLAUDE_SCHEDULE_WAKEUP.to_string(),
6970            description: "Replace the one-shot wakeup stored in the imported Claude manifest. \
6971                The wakeup inherits the manifest's ACTIVE or PAUSED posture; an embedding scheduler \
6972                owns timer execution."
6973                .to_string(),
6974            parameters: serde_json::json!({
6975                "type": "object",
6976                "properties": {
6977                    "delaySeconds": {"type": "integer", "minimum": 0},
6978                    "reason": {"type": "string"},
6979                    "prompt": {"type": "string"}
6980                },
6981                "required": ["delaySeconds"],
6982                "additionalProperties": false
6983            }),
6984        }
6985    }
6986
6987    /// The `background_exec` schema (P5-6, §2 module 4, D1 "background
6988    /// exec").
6989    fn background_exec_schema() -> ToolSchema {
6990        ToolSchema {
6991            name: BACKGROUND_EXEC.to_string(),
6992            description: "Run a shell command in the BACKGROUND: spawns it as a detached \
6993                process and returns a `job_id` IMMEDIATELY, before the command finishes — this \
6994                call never returns the command's output. Poll `background_status` with the \
6995                `job_id` to check progress and retrieve captured output; use `background_kill` \
6996                to cancel it early. The command goes through the exact same sandbox/permission \
6997                checks as a foreground `bash` call, and any check that would need an \
6998                interactive approval is denied automatically (a background job cannot wait for \
6999                one)."
7000                .to_string(),
7001            parameters: serde_json::json!({
7002                "type": "object",
7003                "properties": {
7004                    "command": {
7005                        "type": "string",
7006                        "description": "Shell command to run in the background via `sh -c`."
7007                    }
7008                },
7009                "required": ["command"],
7010                "additionalProperties": false
7011            }),
7012        }
7013    }
7014
7015    /// The `background_status` schema (P5-6, D1 "monitor/event feed").
7016    fn background_status_schema() -> ToolSchema {
7017        ToolSchema {
7018            name: BACKGROUND_STATUS.to_string(),
7019            description: "Check on a background job spawned via background_exec: its \
7020                running/exited/killed status, exit code (once known), and the command's \
7021                captured stdout/stderr so far (bounded — very large output is truncated with a \
7022                marker). Once the job has exited or been killed, this call also reaps it (it \
7023                will no longer appear in background_list or accept further status polls)."
7024                .to_string(),
7025            parameters: serde_json::json!({
7026                "type": "object",
7027                "properties": {
7028                    "job_id": {
7029                        "type": "string",
7030                        "description": "The id `background_exec` returned when this job was \
7031                            started."
7032                    }
7033                },
7034                "required": ["job_id"],
7035                "additionalProperties": false
7036            }),
7037        }
7038    }
7039
7040    /// The `background_list` schema (P5-6, D10 "bg-manager").
7041    fn background_list_schema() -> ToolSchema {
7042        ToolSchema {
7043            name: BACKGROUND_LIST.to_string(),
7044            description: "List every background job currently tracked (running, or finished \
7045                but not yet polled via background_status) — job id, command, status, pid, and \
7046                start time for each. Does not retrieve output or reap anything."
7047                .to_string(),
7048            parameters: serde_json::json!({
7049                "type": "object",
7050                "properties": {},
7051                "additionalProperties": false
7052            }),
7053        }
7054    }
7055
7056    /// The `background_kill` schema (P5-6, D10 "bg-manager").
7057    fn background_kill_schema() -> ToolSchema {
7058        ToolSchema {
7059            name: BACKGROUND_KILL.to_string(),
7060            description: "Kill a background job's real process immediately (a no-op, not an \
7061                error, if it already exited on its own) and reap it."
7062                .to_string(),
7063            parameters: serde_json::json!({
7064                "type": "object",
7065                "properties": {
7066                    "job_id": {
7067                        "type": "string",
7068                        "description": "The id `background_exec` returned when this job was \
7069                            started."
7070                    }
7071                },
7072                "required": ["job_id"],
7073                "additionalProperties": false
7074            }),
7075        }
7076    }
7077
7078    /// The `spawn_subagent` schema (P5-3, §2 module 9 D1 "spawn tool").
7079    /// Lists every configured `agent_type` name so the model knows what's
7080    /// available, but `agent_type` stays optional — an ad-hoc spawn with an
7081    /// inline `system_prompt` is always allowed too.
7082    fn spawn_subagent_schema(&self) -> ToolSchema {
7083        let mut names: Vec<&str> = self
7084            .config
7085            .subagents_definitions
7086            .keys()
7087            .map(String::as_str)
7088            .collect();
7089        names.sort_unstable();
7090        let agent_type_desc = if names.is_empty() {
7091            "Optional named subagent type to run (none configured — omit this and pass \
7092             `system_prompt` instead)."
7093                .to_string()
7094        } else {
7095            format!(
7096                "Optional named subagent type to run: {}. Omit to run an ad-hoc subagent with \
7097                 your own `system_prompt` instead.",
7098                names.join(", ")
7099            )
7100        };
7101        let background_desc = if self.config.subagents_background {
7102            "Run this subagent in the background instead of waiting for it — this call \
7103             returns immediately with a `subagent_id`; poll `subagent_status` with that id for \
7104             the result."
7105        } else {
7106            "Background subagents are disabled for this agent — this must be omitted or false."
7107        };
7108        ToolSchema {
7109            name: SPAWN_SUBAGENT.to_string(),
7110            description: "Spawn a subagent to work on a self-contained task and (by default) \
7111                wait for its final answer, which is returned as this call's result. The \
7112                subagent runs its own independent reasoning/tool loop; it does not see your \
7113                conversation except for the `task` text you give it here."
7114                .to_string(),
7115            parameters: serde_json::json!({
7116                "type": "object",
7117                "properties": {
7118                    "task": {
7119                        "type": "string",
7120                        "description": "The self-contained task/prompt for the subagent."
7121                    },
7122                    "agent_type": {
7123                        "type": "string",
7124                        "description": agent_type_desc
7125                    },
7126                    "system_prompt": {
7127                        "type": "string",
7128                        "description": "Inline system prompt for an ad-hoc subagent (ignored \
7129                            if `agent_type` is given — the named type's own prompt is used \
7130                            instead)."
7131                    },
7132                    "background": {
7133                        "type": "boolean",
7134                        "description": background_desc
7135                    }
7136                },
7137                "required": ["task"],
7138                "additionalProperties": false
7139            }),
7140        }
7141    }
7142
7143    /// Claude Code-compatible alias for [`Self::spawn_subagent_schema`].
7144    fn claude_agent_schema(&self) -> ToolSchema {
7145        let mut names: Vec<String> = self.config.subagents_definitions.keys().cloned().collect();
7146        names.push("general-purpose".into());
7147        names.sort_unstable();
7148        names.dedup();
7149        ToolSchema {
7150            name: CLAUDE_AGENT.to_string(),
7151            description: "Claude Code-compatible subagent dispatcher. Runs a named or ad-hoc \
7152                child agent; children default to background execution in this compatibility mode."
7153                .to_string(),
7154            parameters: serde_json::json!({
7155                "type": "object",
7156                "properties": {
7157                    "prompt": {"type": "string", "description": "Self-contained child task."},
7158                    "subagent_type": {
7159                        "type": "string",
7160                        "description": format!("Named agent type. Available: {}", names.join(", "))
7161                    },
7162                    "description": {
7163                        "type": "string",
7164                        "description": "Short human-facing task label; preserved as descriptive input."
7165                    },
7166                    "model": {
7167                        "type": "string",
7168                        "description": "Optional model alias or full provider slug for this child."
7169                    },
7170                    "run_in_background": {
7171                        "type": "boolean",
7172                        "description": "Whether to return immediately with a child id (default true)."
7173                    }
7174                },
7175                "required": ["prompt"],
7176                "additionalProperties": false
7177            }),
7178        }
7179    }
7180
7181    /// Translate Claude's `Agent` arguments to the native subagent intrinsic.
7182    fn translate_claude_agent_call(
7183        &self,
7184        call: &supercode_interchange::ToolCall,
7185    ) -> Result<supercode_interchange::ToolCall> {
7186        let args = call
7187            .function
7188            .parsed_arguments()
7189            .map_err(|error| Error::InvalidArguments {
7190                tool: CLAUDE_AGENT.to_string(),
7191                message: error.to_string(),
7192            })?;
7193        let object = args.as_object().ok_or_else(|| Error::InvalidArguments {
7194            tool: CLAUDE_AGENT.to_string(),
7195            message: "arguments must be an object".to_string(),
7196        })?;
7197        let mut translated = serde_json::Map::new();
7198        if let Some(value) = object.get("prompt") {
7199            translated.insert("task".to_string(), value.clone());
7200        }
7201        if let Some(value) = object.get("subagent_type") {
7202            // `general-purpose` is a built-in Claude agent, not a project
7203            // definition file. Supercode's equivalent is an ad-hoc child
7204            // using the inherited default system prompt, represented by an
7205            // omitted `agent_type`.
7206            if value.as_str() != Some("general-purpose") {
7207                translated.insert("agent_type".to_string(), value.clone());
7208            }
7209        }
7210        if let Some(value) = object.get("model") {
7211            translated.insert("model".to_string(), value.clone());
7212        }
7213        translated.insert(
7214            "background".to_string(),
7215            object
7216                .get("run_in_background")
7217                .cloned()
7218                .unwrap_or(serde_json::Value::Bool(true)),
7219        );
7220        Ok(supercode_interchange::ToolCall {
7221            id: call.id.clone(),
7222            kind: call.kind.clone(),
7223            function: supercode_interchange::FunctionCall {
7224                name: SPAWN_SUBAGENT.to_string(),
7225                arguments: serde_json::Value::Object(translated).to_string(),
7226            },
7227        })
7228    }
7229
7230    /// Execute Claude's scheduling vocabulary against the imported manifest.
7231    ///
7232    /// This is intentionally a state editor, not a scheduler: it owns no
7233    /// timer/task handle, nothing downstream of it fires, and every
7234    /// successful response says so, so the model is never told a job it just
7235    /// created will run here.
7236    fn run_claude_runtime_tool(
7237        &mut self,
7238        call: &supercode_interchange::ToolCall,
7239    ) -> (String, bool) {
7240        let args = match call.function.parsed_arguments() {
7241            Ok(value) if value.is_object() => value,
7242            Ok(_) => {
7243                return (
7244                    format!("Error: {} arguments must be an object", call.function.name),
7245                    true,
7246                )
7247            }
7248            Err(error) => return (format!("Error: {error}"), true),
7249        };
7250        let object = args.as_object().expect("checked object above");
7251
7252        let Some(manifest) = self.claude_runtime_manifest.as_mut() else {
7253            return (
7254                "Error: Claude runtime compatibility was enabled without an imported runtime \
7255                 manifest; refusing to invent scheduler state"
7256                    .to_string(),
7257                true,
7258            );
7259        };
7260        // Every imported schedule is carried and inert. `state` is the single
7261        // fact the model is told about it, in the manifest's own vocabulary.
7262        let state = "paused";
7263
7264        match call.function.name.as_str() {
7265            CLAUDE_CRON_LIST => {
7266                let jobs: Vec<serde_json::Value> = manifest
7267                    .active_crons
7268                    .iter()
7269                    .map(|job| {
7270                        serde_json::json!({
7271                            "id": job.id,
7272                            "cron": job.schedule,
7273                            "prompt": job.prompt,
7274                            "recurring": job.recurring,
7275                            "durable": job.durable_requested,
7276                            "state": state
7277                        })
7278                    })
7279                    .collect();
7280                let notice = "Imported jobs are preserved but no scheduler is running.";
7281                (
7282                    serde_json::json!({
7283                        "execution_state": state,
7284                        "execution_notice": notice,
7285                        "jobs": jobs
7286                    })
7287                    .to_string(),
7288                    false,
7289                )
7290            }
7291            CLAUDE_CRON_CREATE => {
7292                let Some(schedule) = object.get("cron").and_then(serde_json::Value::as_str) else {
7293                    return ("Error: CronCreate requires string `cron`".to_string(), true);
7294                };
7295                let Some(prompt) = object.get("prompt").and_then(serde_json::Value::as_str) else {
7296                    return (
7297                        "Error: CronCreate requires string `prompt`".to_string(),
7298                        true,
7299                    );
7300                };
7301                let recurring = object
7302                    .get("recurring")
7303                    .and_then(serde_json::Value::as_bool)
7304                    .unwrap_or(false);
7305                let durable_requested = object
7306                    .get("durable")
7307                    .and_then(serde_json::Value::as_bool)
7308                    .unwrap_or(false);
7309                let mut sequence = 1_u64;
7310                let id = loop {
7311                    let candidate = format!("sc{sequence:06}");
7312                    if !manifest.active_crons.iter().any(|job| job.id == candidate) {
7313                        break candidate;
7314                    }
7315                    sequence += 1;
7316                };
7317                let kind = if recurring { "recurring " } else { "" };
7318                let result = format!(
7319                    "Scheduled {kind}job {id} ({schedule}) in PAUSED state. The job is preserved \
7320                     in the continuation manifest but no scheduler is running and it will not execute."
7321                );
7322                manifest
7323                    .active_crons
7324                    .push(crate::claude_runtime_state::ClaudeCronJob {
7325                        id: id.clone(),
7326                        tool_use_id: call.id.clone(),
7327                        schedule: schedule.to_string(),
7328                        recurring,
7329                        durable_requested,
7330                        prompt: prompt.to_string(),
7331                        // The creation instant is a fact about this
7332                        // continuation, recorded like every other manifest
7333                        // field. Nothing consults it as a due time.
7334                        created_at: Some(supercode_interchange::sidecar::ms_to_rfc3339(now_ms())),
7335                        expires_after_seconds: None,
7336                        creation_result: result.clone(),
7337                    });
7338                manifest
7339                    .active_crons
7340                    .sort_by(|left, right| left.id.cmp(&right.id));
7341                (result, false)
7342            }
7343            CLAUDE_CRON_DELETE => {
7344                let Some(id) = object.get("id").and_then(serde_json::Value::as_str) else {
7345                    return ("Error: CronDelete requires string `id`".to_string(), true);
7346                };
7347                let Some(index) = manifest.active_crons.iter().position(|job| job.id == id) else {
7348                    return (
7349                        format!("Error: unknown {state} Claude cron job `{id}`"),
7350                        true,
7351                    );
7352                };
7353                manifest.active_crons.remove(index);
7354                (
7355                    format!("Cancelled job {id}. The job was PAUSED; no execution occurred."),
7356                    false,
7357                )
7358            }
7359            CLAUDE_SCHEDULE_WAKEUP => {
7360                let Some(delay_seconds) = object
7361                    .get("delaySeconds")
7362                    .and_then(serde_json::Value::as_u64)
7363                else {
7364                    return (
7365                        "Error: ScheduleWakeup requires integer `delaySeconds`".to_string(),
7366                        true,
7367                    );
7368                };
7369                let reason = object
7370                    .get("reason")
7371                    .and_then(serde_json::Value::as_str)
7372                    .map(str::to_string);
7373                let prompt = object
7374                    .get("prompt")
7375                    .and_then(serde_json::Value::as_str)
7376                    .map(str::to_string);
7377                let now = now_ms();
7378                let created_at = Some(supercode_interchange::sidecar::ms_to_rfc3339(now));
7379                // The instant the wakeup asks for, recorded as the request
7380                // made it. No timer consults it here.
7381                let delay_ms = i64::try_from(delay_seconds)
7382                    .unwrap_or(i64::MAX)
7383                    .saturating_mul(1_000);
7384                let scheduled_for =
7385                    supercode_interchange::sidecar::ms_to_rfc3339(now.saturating_add(delay_ms));
7386                let result = format!(
7387                    "Next wakeup recorded for {scheduled_for} (in {delay_seconds}s) in PAUSED \
7388                     state. The request replaced the prior wakeup in the manifest, but no timer \
7389                     is running and it will not execute."
7390                );
7391                manifest.pending_wakeups.clear();
7392                manifest
7393                    .pending_wakeups
7394                    .push(crate::claude_runtime_state::ClaudeWakeup {
7395                        tool_use_id: call.id.clone(),
7396                        delay_seconds,
7397                        reason,
7398                        prompt,
7399                        created_at,
7400                        scheduled_for: Some(scheduled_for),
7401                        creation_result: result.clone(),
7402                    });
7403                (result, false)
7404            }
7405            _ => unreachable!("runtime tool dispatch is name-gated"),
7406        }
7407    }
7408
7409    /// The `subagent_status` schema (P5-3, D3 "background+resume").
7410    fn subagent_status_schema() -> ToolSchema {
7411        ToolSchema {
7412            name: SUBAGENT_STATUS.to_string(),
7413            description: "Check on (and, once finished, retrieve the result of) a background \
7414                subagent spawned via spawn_subagent with background=true. Pass the \
7415                `subagent_id` that spawn returned."
7416                .to_string(),
7417            parameters: serde_json::json!({
7418                "type": "object",
7419                "properties": {
7420                    "subagent_id": {
7421                        "type": "string",
7422                        "description": "The id `spawn_subagent` returned when this subagent \
7423                            was spawned."
7424                    }
7425                },
7426                "required": ["subagent_id"],
7427                "additionalProperties": false
7428            }),
7429        }
7430    }
7431
7432    /// The `send_message` schema.
7433    fn send_message_schema() -> ToolSchema {
7434        ToolSchema {
7435            name: SEND_MESSAGE.to_string(),
7436            description: "Send a message to another agent: one of your background subagents \
7437                that is STILL RUNNING (by the id `spawn_subagent` returned; it receives it at the \
7438                start of its next step), or another session of any harness, on this machine or an \
7439                enrolled one (by its name, name@machine, or sc: address; `supercode message list` \
7440                shows them). A session's reply comes back to you as a message. Send only what \
7441                asks something or carries a result; no acknowledgements."
7442                .to_string(),
7443            parameters: serde_json::json!({
7444                "type": "object",
7445                "properties": {
7446                    "to": {
7447                        "type": "string",
7448                        "description": "A running subagent's id, or a session's name, name@machine or sc: address."
7449                    },
7450                    "message": {
7451                        "type": "string",
7452                        "description": "What to tell it."
7453                    },
7454                    "notify_when_idle": {
7455                        "type": "boolean",
7456                        "description": "For a session: also get one notice when its next turn ends."
7457                    }
7458                },
7459                "required": ["to", "message"],
7460                "additionalProperties": false
7461            }),
7462        }
7463    }
7464
7465    /// BP-7: the `subagent_resume` schema.
7466    fn subagent_resume_schema() -> ToolSchema {
7467        ToolSchema {
7468            name: SUBAGENT_RESUME.to_string(),
7469            description: "Continue a subagent that has already FINISHED, with its own                 previous conversation restored, so it keeps everything it learned instead                 of being briefed again from scratch. Pass the id it was spawned with and                 the next task."
7470                .to_string(),
7471            parameters: serde_json::json!({
7472                "type": "object",
7473                "properties": {
7474                    "subagent_id": {
7475                        "type": "string",
7476                        "description": "The id of a subagent that has already finished."
7477                    },
7478                    "task": {
7479                        "type": "string",
7480                        "description": "What the resumed subagent should do next."
7481                    }
7482                },
7483                "required": ["subagent_id", "task"],
7484                "additionalProperties": false
7485            }),
7486        }
7487    }
7488
7489    /// BP-7 (catalog §4a "Background subagents + resume"): deliver a
7490    /// message into a still-running background child's mailbox.
7491    ///
7492    /// The mailbox is the child's own `SteerInbox` — the seam P4b built for
7493    /// mid-turn steering, which is writable while the child's turn holds
7494    /// `&mut Agent`. So delivery ordering is already defined: the message
7495    /// arrives at the top of the child's next loop iteration, i.e. after
7496    /// whatever tool calls it is currently running, per its
7497    /// `steering_mode`. A child that has already FINISHED is refused with
7498    /// a pointer at `subagent_resume`, which is the operation for that
7499    /// case — never silently dropped.
7500    async fn run_send_message(&mut self, call: &supercode_interchange::ToolCall) -> (String, bool) {
7501        let args = match call.function.parsed_arguments() {
7502            Ok(v) => v,
7503            Err(e) => {
7504                let err = Error::InvalidArguments {
7505                    tool: SEND_MESSAGE.to_string(),
7506                    message: e.to_string(),
7507                };
7508                return (format!("Error: {err}"), true);
7509            }
7510        };
7511        let Some(id) = args.get("to").and_then(serde_json::Value::as_str) else {
7512            let err = Error::InvalidArguments {
7513                tool: SEND_MESSAGE.to_string(),
7514                message: "`to` is required".to_string(),
7515            };
7516            return (format!("Error: {err}"), true);
7517        };
7518        let message = args
7519            .get("message")
7520            .and_then(serde_json::Value::as_str)
7521            .unwrap_or("");
7522        if message.is_empty() {
7523            let err = Error::InvalidArguments {
7524                tool: SEND_MESSAGE.to_string(),
7525                message: "`message` is required and must be non-empty".to_string(),
7526            };
7527            return (format!("Error: {err}"), true);
7528        }
7529        let Some(entry) = self.background_subagents.get(id) else {
7530            // Not one of this agent's children: another session.
7531            let notify_when_idle = args
7532                .get("notify_when_idle")
7533                .and_then(serde_json::Value::as_bool)
7534                .unwrap_or(false);
7535            return send_to_session(id, message, notify_when_idle).await;
7536        };
7537        if entry.handle.is_finished() {
7538            let out = serde_json::json!({
7539                "subagent_id": id,
7540                "status": "finished",
7541                "delivered": false,
7542                "hint": "this subagent already finished — collect it with subagent_status,                          then continue it with subagent_resume",
7543            });
7544            return (out.to_string(), false);
7545        }
7546        entry
7547            .mailbox
7548            .lock()
7549            .unwrap_or_else(std::sync::PoisonError::into_inner)
7550            .queue_unchecked(message.to_string());
7551        let out = serde_json::json!({
7552            "subagent_id": id,
7553            "status": "running",
7554            "delivered": true,
7555        });
7556        (out.to_string(), false)
7557    }
7558
7559    /// BP-7 (catalog §4a "Background subagents + resume": "resumable with
7560    /// context intact"): continue a finished child over its OWN transcript.
7561    ///
7562    /// The context comes from the reap (kept in-process) or, for a session
7563    /// that attached a subagent store, from that child's persisted
7564    /// `<parent>.subagents/<id>.sidecar.jsonl`. Either way the resumed
7565    /// child is rebuilt through `build_child_config` from the SAME named
7566    /// definition it was spawned with, so its permission posture on resume
7567    /// is the one it had originally — never a fresh, looser default.
7568    async fn run_subagent_resume(
7569        &mut self,
7570        call: &supercode_interchange::ToolCall,
7571    ) -> (String, bool) {
7572        let args = match call.function.parsed_arguments() {
7573            Ok(v) => v,
7574            Err(e) => {
7575                let err = Error::InvalidArguments {
7576                    tool: SUBAGENT_RESUME.to_string(),
7577                    message: e.to_string(),
7578                };
7579                return (format!("Error: {err}"), true);
7580            }
7581        };
7582        let Some(id) = args
7583            .get("subagent_id")
7584            .and_then(serde_json::Value::as_str)
7585            .map(String::from)
7586        else {
7587            let err = Error::InvalidArguments {
7588                tool: SUBAGENT_RESUME.to_string(),
7589                message: "`subagent_id` is required".to_string(),
7590            };
7591            return (format!("Error: {err}"), true);
7592        };
7593        let task = args
7594            .get("task")
7595            .and_then(serde_json::Value::as_str)
7596            .unwrap_or("")
7597            .to_string();
7598        if task.is_empty() {
7599            let err = Error::InvalidArguments {
7600                tool: SUBAGENT_RESUME.to_string(),
7601                message: "`task` is required and must be non-empty".to_string(),
7602            };
7603            return (format!("Error: {err}"), true);
7604        }
7605        if self
7606            .background_subagents
7607            .get(&id)
7608            .is_some_and(|e| !e.handle.is_finished())
7609        {
7610            let err = Error::tool(
7611                SUBAGENT_RESUME,
7612                format!(
7613                    "subagent `{id}` is still running — send it a message with                      send_message, or collect it with subagent_status first"
7614                ),
7615            );
7616            return (format!("Error: {err}"), true);
7617        }
7618        let lineage = self
7619            .subagent_store
7620            .as_ref()
7621            .and_then(|(store, parent)| store.load_subagent_lineage(parent, &id).ok().flatten());
7622        let Some(prior) = self.prior_subagent_transcript(&id) else {
7623            let err = Error::SubagentNotFound(id.clone());
7624            return (format!("Error: {err}"), true);
7625        };
7626
7627        let agent_type = lineage.as_ref().and_then(|l| l.agent_type.clone());
7628        let definition = agent_type
7629            .as_ref()
7630            .and_then(|name| self.config.subagents_definitions.get(name).cloned());
7631        let Some(guard) = crate::subagents::try_acquire(
7632            &self.subagent_concurrency_gauge,
7633            self.config.subagents_max_concurrent,
7634        ) else {
7635            let err = Error::SubagentConcurrencyExceeded {
7636                max_concurrent: self.config.subagents_max_concurrent,
7637            };
7638            return (format!("Error: {err}"), true);
7639        };
7640        let child_config = self.build_child_config(
7641            definition.as_ref(),
7642            None,
7643            lineage
7644                .as_ref()
7645                .map(|l| l.model.clone())
7646                .or_else(|| definition.as_ref().and_then(|d| d.model.clone())),
7647        );
7648        let mut child = Agent::with_provider_arc(child_config, self.provider.clone());
7649        child.subagent_depth = self.subagent_depth + 1;
7650        child.subagent_concurrency_gauge = self.subagent_concurrency_gauge.clone();
7651        // Context intact: the child's own prior messages, appended after
7652        // its (re-derived, identical) system prompt.
7653        child.history.extend(prior);
7654
7655        let result = child.send(task).await;
7656        let transcript = child.history()[1..].to_vec();
7657        if let Some(lineage) = &lineage {
7658            self.persist_subagent_transcript(&id, lineage, &transcript);
7659        }
7660        self.reaped_subagents.insert(id.clone(), transcript);
7661        drop(guard);
7662        match result {
7663            Ok(text) => {
7664                let out = serde_json::json!({
7665                    "subagent_id": id,
7666                    "status": "done",
7667                    "resumed": true,
7668                    "result": text,
7669                });
7670                (out.to_string(), false)
7671            }
7672            Err(e) => {
7673                let out = serde_json::json!({
7674                    "subagent_id": id,
7675                    "status": "error",
7676                    "resumed": true,
7677                    "message": e.to_string(),
7678                });
7679                (out.to_string(), true)
7680            }
7681        }
7682    }
7683
7684    /// BP-7 (catalog §4a "Named agent definitions as data"): the child
7685    /// `Config` a `spawn_subagent` of `agent_type` would build — the
7686    /// resolved posture a named definition actually produces, including
7687    /// its [`crate::subagents::AgentPermissions`] bundle applied through
7688    /// the tightening-only rules. `None` when no definition of that name
7689    /// is registered or discovered.
7690    ///
7691    /// Exposed so a caller (and this build's tests) can ask what a named
7692    /// agent WOULD run as without spawning it and paying for a turn.
7693    pub fn child_config_for_agent_type(&self, agent_type: &str) -> Option<Config> {
7694        let definition = self.config.subagents_definitions.get(agent_type)?.clone();
7695        Some(self.build_child_config(Some(&definition), None, definition.model.clone()))
7696    }
7697
7698    /// BP-7 (catalog §4a "Background subagents + resume"): the ids of
7699    /// children that have finished and been reaped, and can therefore be
7700    /// continued with [`SUBAGENT_RESUME`].
7701    pub fn reaped_subagent_ids(&self) -> Vec<String> {
7702        let mut ids: Vec<String> = self.reaped_subagents.keys().cloned().collect();
7703        ids.sort();
7704        ids
7705    }
7706
7707    /// BP-7: a finished child's own messages — from the in-process reap
7708    /// cache first, then this session's subagent store.
7709    fn prior_subagent_transcript(&self, id: &str) -> Option<Vec<ChatMessage>> {
7710        if let Some(messages) = self.reaped_subagents.get(id) {
7711            return Some(messages.clone());
7712        }
7713        let (store, parent) = self.subagent_store.as_ref()?;
7714        let jsonl = store.load_subagent_transcript(parent, id).ok()??;
7715        let session = supercode_interchange::session::Session::from_sidecar_str(&jsonl).ok()?;
7716        Some(
7717            session
7718                .messages
7719                .into_iter()
7720                .filter(|m| m.role != supercode_interchange::Role::System)
7721                .collect(),
7722        )
7723    }
7724
7725    /// Build the CHILD `Config` a `spawn_subagent` call constructs its
7726    /// [`Agent`] from. The whole point of this method (§5.3-style
7727    /// "monotonic posture", build-brief "a subagent inherits or narrows —
7728    /// never widens — the parent's permission posture"): every field that
7729    /// governs what the child is ALLOWED to do (sandbox, approval,
7730    /// tool_overrides, deny/allow patterns, protected paths, the subagents
7731    /// caps themselves) is copied VERBATIM from `self.config` — never
7732    /// loosened — and the only NARROWING lever is `definition.tools`
7733    /// (intersected with whatever the parent already had enabled, never
7734    /// unioned in anything new).
7735    ///
7736    /// P5-3 safety hardening (Fable-5 review, LOW-MEDIUM "child safety-limit
7737    /// inheritance"): the monotonic-posture guarantee above was, before this
7738    /// fix, scoped to PERMISSION fields only — a child could still silently
7739    /// get a LOOSER safety BUDGET/BREAKER than its parent, because
7740    /// `max_total_output_tokens`/`max_tool_output_bytes`/`max_tokens`/
7741    /// `doom_loop_threshold`/`edit_file_require_read_before_edit` were never
7742    /// copied and so fell back to `Config::default()`'s (looser/uncapped)
7743    /// values on every spawn regardless of what the parent had configured.
7744    /// These are now copied verbatim alongside the permission-posture
7745    /// fields — a parent that capped its own output/tool-output/doom-loop
7746    /// exposure, or required read-before-edit, gets a child that is bound
7747    /// by the exact same ceiling, never a wider one.
7748    ///
7749    /// **Full field-by-field accounting** (every [`Config`] field, so this
7750    /// doc comment stays the single place that answers "did we forget
7751    /// one?"): fields already copied above/below this note (permission
7752    /// posture: `sandbox`/`approval`/`tool_overrides`/`auto_approved_tools`/
7753    /// `tool_deny_patterns`/`tool_allow_patterns`/`permissions_enabled`/
7754    /// `permissions_ask_patterns`/`permissions_protected_paths`/
7755    /// `network_policy`/`core_tools_enabled`/`module_registry`/
7756    /// `module_activation`/every `subagents_*` field; safety limits:
7757    /// `max_iterations`/`max_total_output_tokens`/`max_tool_output_bytes`/
7758    /// `max_tokens`/`doom_loop_threshold`/`edit_file_require_read_before_edit`;
7759    /// identity/transport: `model`/`system_prompt`/`cwd`/`base_url`/
7760    /// `api_key`/`api_key_env`/`api_key_cmd`) are the ones that gate
7761    /// harm/spend/hazard exposure. Every OTHER field is deliberately left at
7762    /// `Config::default()` because none of them is a safety ceiling the
7763    /// child could "loosen" by missing it:
7764    /// - `temperature`/`effort`/`response_format`/`extra_body`/`extra_headers`/
7765    ///   `tool_advertising`/`tool_schema_tier`/`cache_plan`/`cache_warnings`/
7766    ///   `reduction_policy`/
7767    ///   `session_*`/`small_model`/`model_fallback`/`env_context`/
7768    ///   `project_root_markers`/`project_doc_max_bytes`/`instruction_imports`/
7769    ///   `retry_*`/`compaction_*`/`auto_title`/`steering_mode`/
7770    ///   `follow_up_mode`/`read_file_multimodal`/`edit_file_notebook_aware`/
7771    ///   `shell_env_snapshot`/`nested_instructions`/`model_switch_allow_switch`/
7772    ///   `context_injections`/`context_injection_blocks`/`parallel_tool_calls`
7773    ///   are behavior/cost-shaping or presentation knobs, not hard guards —
7774    ///   a child defaulting on any of these can do LESS (e.g. no multimodal
7775    ///   read, no notebook-aware edits, no proactive compaction) or the same,
7776    ///   never something the parent hadn't already exposed it to. Several
7777    ///   default to their OFF/conservative state (`false`/`None`), which is
7778    ///   the tight direction, not the loose one.
7779    /// - `additional_dirs`: governs which extra roots are reachable at all
7780    ///   (`presets.rs`'s `[core] additional_dirs` note) — a child that
7781    ///   doesn't inherit it has FEWER reachable roots than its parent, i.e.
7782    ///   strictly tighter, never looser.
7783    /// - `load_project_context`: whether instruction files are auto-loaded
7784    ///   into the system prompt — a read-time convenience, not an access
7785    ///   grant (`sandbox`/`permissions_protected_paths` already gate actual
7786    ///   file access).
7787    /// - `prompts`: named `/slash` command templates for THIS agent's own
7788    ///   user-facing input surface, not something the model can invoke
7789    ///   against the child's tool surface.
7790    /// - `stop_gate`/`post_tool_hook`/`approval_handler`/`event_sink`:
7791    ///   code-only `Box<dyn Fn>` callbacks (see the `pre_tool_hook` note
7792    ///   immediately below — same non-`Clone` shape) that are observational
7793    ///   or terminate-only, not a call-time veto over what a tool is allowed
7794    ///   to do; `approval_handler` specifically is ALREADY documented at
7795    ///   this method's call site (`Self::run_spawn_subagent`) as
7796    ///   intentionally never set here — a foreground child gets no handler
7797    ///   by design, an embedder installs its own after spawn if it wants
7798    ///   one.
7799    ///
7800    /// **`pre_tool_hook` cannot propagate, and this is deliberate + named,
7801    /// not a silent gap**: `Config::pre_tool_hook` is a `Box<dyn Fn(&str,
7802    /// &serde_json::Value) -> Option<String> + Send + Sync>` — an
7803    /// embedder's own call-time veto over every tool call. `Box<dyn Fn>` is
7804    /// not `Clone` (there is no generic way to duplicate an opaque closure),
7805    /// so it genuinely CANNOT be copied into a child `Config` the way every
7806    /// `Clone`-able field above is — there is no fix that makes this one
7807    /// "verbatim copy" like the others. An embedder relying on a
7808    /// `pre_tool_hook` veto reaching spawned children as well as the parent
7809    /// MUST re-install one on the child explicitly (e.g. via a
7810    /// `spawn_subagent`-adjacent hook of their own, or by not relying on
7811    /// `pre_tool_hook` alone for anything safety-critical across a spawn
7812    /// boundary) — named here so this is a documented contract, not a gap
7813    /// an embedder discovers by a child silently misbehaving.
7814    fn build_child_config(
7815        &self,
7816        definition: Option<&crate::subagents::NamedAgentDefinition>,
7817        inline_system_prompt: Option<String>,
7818        model_override: Option<String>,
7819    ) -> Config {
7820        let system_prompt = definition
7821            .map(|d| d.system_prompt.clone())
7822            .filter(|s| !s.is_empty())
7823            .or(inline_system_prompt)
7824            .unwrap_or_else(|| self.config.system_prompt.clone());
7825        let model = model_override.unwrap_or_else(|| self.config.model.clone());
7826
7827        let mut child = Config::builder()
7828            .model(model)
7829            .system_prompt(system_prompt)
7830            .cwd(self.config.cwd.clone())
7831            // Monotonic: verbatim, never loosened.
7832            .sandbox(self.config.sandbox)
7833            .approval(self.config.approval)
7834            .max_iterations(self.config.max_iterations)
7835            .build();
7836        child.base_url = self.config.base_url.clone();
7837        child.api_key = self.config.api_key.clone();
7838        child.api_key_env = self.config.api_key_env.clone();
7839        child.api_key_cmd = self.config.api_key_cmd.clone();
7840        // P5-3 safety hardening (Fable-5 review, LOW-MEDIUM "child
7841        // safety-limit inheritance"): the monotonic-posture spirit extends
7842        // to safety BUDGETS/BREAKERS, not just permissions — a child must
7843        // not get a looser cap/breaker than its parent by simply falling
7844        // back to `Config::default()`'s (looser) values. See this method's
7845        // doc comment for the full field-by-field accounting.
7846        child.max_total_output_tokens = self.config.max_total_output_tokens;
7847        child.max_tool_output_bytes = self.config.max_tool_output_bytes;
7848        child.max_tokens = self.config.max_tokens;
7849        child.doom_loop_threshold = self.config.doom_loop_threshold;
7850        child.edit_file_require_read_before_edit = self.config.edit_file_require_read_before_edit;
7851        // Monotonic tool posture: start from the PARENT's own overrides
7852        // (so anything the parent already disabled stays disabled), then
7853        // narrow further if a named definition restricts the tool set.
7854        child.tool_overrides = self.config.tool_overrides.clone();
7855        child.auto_approved_tools = self.config.auto_approved_tools.clone();
7856        child.tool_deny_patterns = self.config.tool_deny_patterns.clone();
7857        child.tool_allow_patterns = self.config.tool_allow_patterns.clone();
7858        child.permissions_enabled = self.config.permissions_enabled;
7859        child.permissions_ask_patterns = self.config.permissions_ask_patterns.clone();
7860        child.permissions_protected_paths = self.config.permissions_protected_paths.clone();
7861        child.network_policy = self.config.network_policy.clone();
7862        // P5-10 (§2 module 12): same monotonic-posture treatment as
7863        // `sandbox`/`approval` above — a subagent must inherit its
7864        // parent's OS-sandbox posture verbatim, never a looser
7865        // `Config::default()` fallback (`sandbox_os_enabled: None`,
7866        // `escalation: Deny`, `env_policy: Inherit` would otherwise be
7867        // right back to "confine only when the tier itself says so" for a
7868        // child whose parent explicitly forced the backstop on/off).
7869        child.sandbox_os_enabled = self.config.sandbox_os_enabled;
7870        child.sandbox_escalation = self.config.sandbox_escalation;
7871        child.sandbox_env_policy = self.config.sandbox_env_policy;
7872        if let Some(def) = definition {
7873            if let Some(allowed) = &def.tools {
7874                for name in &self.config.core_tools_enabled {
7875                    if !allowed.iter().any(|t| t == name) {
7876                        child
7877                            .tool_overrides
7878                            .entry(name.clone())
7879                            .or_default()
7880                            .enabled = Some(false);
7881                    }
7882                }
7883            }
7884            // BP-7 (catalog §4a "Named agent definitions as data": the
7885            // `permissions` component of `prompt+model+tools+permissions`).
7886            // Every arm below can only TIGHTEN — the two policy values go
7887            // through the SAME strictness ranks `configfile::
7888            // clamp_project_permissions` uses for the untrusted project
7889            // layer (a looser value is ignored, never honored), the
7890            // auto-approve list is INTERSECTED with the parent's, and the
7891            // deny list is a union. A definition may come from a
7892            // `.claude/agents/*.md` file in the repo, so it sits at the
7893            // project trust tier and must never be an escalation door.
7894            if let Some(perms) = &def.permissions {
7895                if let Some(approval) = perms.approval {
7896                    if crate::configfile::approval_rank(approval)
7897                        < crate::configfile::approval_rank(child.approval)
7898                    {
7899                        child.approval = approval;
7900                    }
7901                }
7902                if let Some(sandbox) = perms.sandbox {
7903                    if crate::configfile::sandbox_rank(sandbox)
7904                        < crate::configfile::sandbox_rank(child.sandbox)
7905                    {
7906                        child.sandbox = sandbox;
7907                    }
7908                }
7909                if let Some(allowed) = &perms.auto_approved_tools {
7910                    child
7911                        .auto_approved_tools
7912                        .retain(|tool| allowed.iter().any(|a| a == tool));
7913                }
7914                for pattern in &perms.deny {
7915                    if !child.tool_deny_patterns.iter().any(|p| p == pattern) {
7916                        child.tool_deny_patterns.push(pattern.clone());
7917                    }
7918                }
7919            }
7920        }
7921        child.core_tools_enabled = self.config.core_tools_enabled.clone();
7922        child.module_registry = self.config.module_registry;
7923        child.module_activation = self.config.module_activation.clone();
7924        // The subagents module itself never widens either: a child spawned
7925        // at depth d+1 inherits the SAME caps (never a looser depth/
7926        // concurrency/background posture than its own parent).
7927        child.subagents_enabled = self.config.subagents_enabled;
7928        child.subagents_max_depth = self.config.subagents_max_depth;
7929        child.subagents_max_concurrent = self.config.subagents_max_concurrent;
7930        child.subagents_background = self.config.subagents_background;
7931        child.subagents_background_prompts = self.config.subagents_background_prompts;
7932        child.subagents_claude_agent_alias = self.config.subagents_claude_agent_alias;
7933        child.subagents_definitions = self.config.subagents_definitions.clone();
7934        child.subagent_depth = self.subagent_depth + 1;
7935        child
7936    }
7937
7938    /// Execute the `spawn_subagent` intrinsic (P5-3, §2 module 9). See
7939    /// `Self::build_child_config` for the monotonic-posture guarantee and
7940    /// `crate::subagents` for the depth/concurrency resource bounds and the
7941    /// §2.2 C6 background-policy enforcement.
7942    async fn run_spawn_subagent(
7943        &mut self,
7944        call: &supercode_interchange::ToolCall,
7945    ) -> (String, bool) {
7946        // BP-11: `subagent_start`/`subagent_stop` bracket a call that passed
7947        // the same validation the runner applies (subagents on, non-empty
7948        // task); a refused call fires neither.
7949        let task = if self.config.subagents_enabled {
7950            call.function
7951                .parsed_arguments()
7952                .ok()
7953                .and_then(|v| {
7954                    v.get("task")
7955                        .and_then(serde_json::Value::as_str)
7956                        .map(str::to_string)
7957                })
7958                .filter(|t| !t.is_empty())
7959        } else {
7960            None
7961        };
7962        if let Some(task) = &task {
7963            self.fire_lifecycle(&crate::config::LifecycleEvent::SubagentStart {
7964                task: task.clone(),
7965            });
7966        }
7967        let (output, is_error) = self.run_spawn_subagent_inner(call).await;
7968        if let Some(task) = task {
7969            self.fire_lifecycle(&crate::config::LifecycleEvent::SubagentStop {
7970                task,
7971                is_error,
7972                output_len: output.len(),
7973            });
7974        }
7975        (output, is_error)
7976    }
7977
7978    /// Hands a lifecycle moment to the installed observer, if any (BP-11).
7979    fn fire_lifecycle(&self, event: &crate::config::LifecycleEvent) {
7980        if let Some(hook) = self.config.lifecycle_hook.as_ref() {
7981            hook(event);
7982        }
7983    }
7984
7985    /// Installs the lifecycle observer (compaction and subagent boundaries).
7986    pub fn set_lifecycle_hook(&mut self, hook: crate::config::LifecycleHook) {
7987        self.config.lifecycle_hook = Some(hook);
7988    }
7989
7990    async fn run_spawn_subagent_inner(
7991        &mut self,
7992        call: &supercode_interchange::ToolCall,
7993    ) -> (String, bool) {
7994        if !self.config.subagents_enabled {
7995            let err = Error::UnknownTool(SPAWN_SUBAGENT.to_string());
7996            return (format!("Error: {err}"), true);
7997        }
7998        let args = match call.function.parsed_arguments() {
7999            Ok(v) => v,
8000            Err(e) => {
8001                let err = Error::InvalidArguments {
8002                    tool: SPAWN_SUBAGENT.to_string(),
8003                    message: e.to_string(),
8004                };
8005                return (format!("Error: {err}"), true);
8006            }
8007        };
8008        let task = args
8009            .get("task")
8010            .and_then(serde_json::Value::as_str)
8011            .unwrap_or("")
8012            .to_string();
8013        if task.is_empty() {
8014            let err = Error::InvalidArguments {
8015                tool: SPAWN_SUBAGENT.to_string(),
8016                message: "`task` is required and must be non-empty".to_string(),
8017            };
8018            return (format!("Error: {err}"), true);
8019        }
8020        let agent_type = args
8021            .get("agent_type")
8022            .and_then(serde_json::Value::as_str)
8023            .map(String::from);
8024        let inline_system_prompt = args
8025            .get("system_prompt")
8026            .and_then(serde_json::Value::as_str)
8027            .map(String::from);
8028        let background = args
8029            .get("background")
8030            .and_then(serde_json::Value::as_bool)
8031            .unwrap_or(false);
8032        let requested_model = args
8033            .get("model")
8034            .and_then(serde_json::Value::as_str)
8035            .map(|model| crate::model_catalog::resolve_alias(model));
8036
8037        let definition = match &agent_type {
8038            Some(name) => match self.config.subagents_definitions.get(name) {
8039                Some(d) => Some(d.clone()),
8040                None => {
8041                    let err = Error::SubagentDefinitionNotFound(name.clone());
8042                    return (format!("Error: {err}"), true);
8043                }
8044            },
8045            None => None,
8046        };
8047
8048        if background {
8049            if !self.config.subagents_background {
8050                let err = Error::tool(
8051                    SPAWN_SUBAGENT,
8052                    "background=true requires capabilities.subagents.background = true",
8053                );
8054                return (format!("Error: {err}"), true);
8055            }
8056            // §2.2 C6, defensive re-check (belt-and-suspenders — see
8057            // `Error::SubagentBackgroundPolicyMissing`'s doc comment for why
8058            // this can't just trust the resolver already checked it).
8059            if self.config.subagents_background_prompts.is_none() {
8060                let err = Error::SubagentBackgroundPolicyMissing;
8061                return (format!("Error: {err}"), true);
8062            }
8063        }
8064
8065        // Resource bounds (fail-closed): depth first (cheap, no side
8066        // effect on failure), THEN concurrency (holds a slot — must be the
8067        // LAST check before actually spawning, so a refused spawn never
8068        // leaves a stray slot held).
8069        if let Err(e) =
8070            crate::subagents::check_depth(self.subagent_depth, self.config.subagents_max_depth)
8071        {
8072            return (format!("Error: {e}"), true);
8073        }
8074        let Some(guard) = crate::subagents::try_acquire(
8075            &self.subagent_concurrency_gauge,
8076            self.config.subagents_max_concurrent,
8077        ) else {
8078            let err = Error::SubagentConcurrencyExceeded {
8079                max_concurrent: self.config.subagents_max_concurrent,
8080            };
8081            return (format!("Error: {err}"), true);
8082        };
8083
8084        let child_id = next_subagent_id();
8085        let child_config = self.build_child_config(
8086            definition.as_ref(),
8087            inline_system_prompt,
8088            requested_model.or_else(|| definition.as_ref().and_then(|d| d.model.clone())),
8089        );
8090        let child_model = child_config.model.clone();
8091        let mut child = Agent::with_provider_arc(child_config, self.provider.clone());
8092        child.subagent_depth = self.subagent_depth + 1;
8093        child.subagent_concurrency_gauge = self.subagent_concurrency_gauge.clone();
8094
8095        // §2.2 C6: a background child NEVER gets a BLOCKING-BY-DEFAULT
8096        // interactive approval handler — either no handler at all
8097        // (`AutoPolicy`: the engine's pre-existing "no handler ⇒ deny"
8098        // fail-closed default), or (`Parent`) the never-blocking
8099        // `ParentQueueApprovalHandler`, UNLESS a `tui` embedder has
8100        // installed [`Self::child_approval_handler_factory`] (P5-4), in
8101        // which case THAT builds the handler instead — see
8102        // [`Self::set_child_approval_handler_factory`]'s doc comment for
8103        // why this can't escalate past what the rule engine already routed
8104        // to `Ask`. A foreground child also gets no handler here (today's
8105        // existing default posture; an embedder that wants an interactive
8106        // child installs its own via `set_permissions_approval_handler`
8107        // after this call returns, out of this method's scope).
8108        if background {
8109            if let Some(crate::subagents::BackgroundPromptsPolicy::Parent) =
8110                self.config.subagents_background_prompts
8111            {
8112                let handler: std::sync::Arc<dyn crate::permissions::PermissionsApprovalHandler> =
8113                    match &self.child_approval_handler_factory {
8114                        Some(factory) => {
8115                            factory(child_id.clone(), self.pending_child_approvals.clone())
8116                        }
8117                        None => std::sync::Arc::new(crate::subagents::ParentQueueApprovalHandler {
8118                            child_agent_id: child_id.clone(),
8119                            queue: self.pending_child_approvals.clone(),
8120                        }),
8121                    };
8122                child.ctx.sandbox_approval_handler =
8123                    Some(crate::sandbox::SandboxApprovalHandler(handler.clone()));
8124                child.permissions_approval_handler = Some(handler);
8125            }
8126        }
8127
8128        let lineage = crate::subagents::SubagentLineage {
8129            child_agent_id: child_id.clone(),
8130            parent_session_id: self.subagent_store.as_ref().map(|(_, name)| name.clone()),
8131            parent_tool_use_id: call.id.clone(),
8132            depth: self.subagent_depth + 1,
8133            agent_type: agent_type.clone(),
8134            task: task.clone(),
8135            background,
8136            spawned_at_ms: now_ms(),
8137            model: child_model,
8138        };
8139        if let Some((store, parent_name)) = &self.subagent_store {
8140            let _ = store.save_subagent_lineage(parent_name, &child_id, &lineage);
8141        }
8142
8143        if background {
8144            let spawned_task_text = task.clone();
8145            // BP-7: captured BEFORE `child` moves into the task — this is
8146            // the handle `send_message` writes into.
8147            let mailbox = child.steer_queue_handle();
8148            self.background_subagents.insert(
8149                child_id.clone(),
8150                BackgroundSubagent {
8151                    handle: tokio::spawn(async move {
8152                        // The concurrency slot lives for exactly as long as
8153                        // this future runs — moved in here, dropped when the
8154                        // child's `send` (and this future) finishes.
8155                        let _guard = guard;
8156                        let result = child.send(spawned_task_text).await;
8157                        let transcript = child.history()[1..].to_vec();
8158                        (child_id, result, transcript)
8159                    }),
8160                    task,
8161                    agent_type,
8162                    started_at_ms: lineage.spawned_at_ms,
8163                    mailbox,
8164                },
8165            );
8166            let out = serde_json::json!({
8167                "subagent_id": lineage.child_agent_id,
8168                "status": "spawned",
8169                "background": true,
8170            });
8171            return (out.to_string(), false);
8172        }
8173
8174        // Foreground: run to completion now, guard held until this
8175        // function returns (then drops, freeing the slot).
8176        let result = child.send(task).await;
8177        let transcript = child.history()[1..].to_vec();
8178        self.persist_subagent_transcript(&child_id, &lineage, &transcript);
8179        // BP-7: kept in-process so `subagent_resume` can restore this
8180        // child's context even with no session store attached.
8181        self.reaped_subagents
8182            .insert(child_id.clone(), transcript.clone());
8183        drop(guard);
8184        match result {
8185            Ok(text) => (text, false),
8186            Err(e) => (format!("Error: subagent `{child_id}` failed: {e}"), true),
8187        }
8188    }
8189
8190    /// Execute the `subagent_status` intrinsic (P5-3, D3
8191    /// "background+resume"): poll a background child; once its `JoinHandle`
8192    /// is finished, reap it (removing it from `Self::background_subagents`
8193    /// and persisting its transcript, same as the foreground path).
8194    async fn run_subagent_status(
8195        &mut self,
8196        call: &supercode_interchange::ToolCall,
8197    ) -> (String, bool) {
8198        let args = match call.function.parsed_arguments() {
8199            Ok(v) => v,
8200            Err(e) => {
8201                let err = Error::InvalidArguments {
8202                    tool: SUBAGENT_STATUS.to_string(),
8203                    message: e.to_string(),
8204                };
8205                return (format!("Error: {err}"), true);
8206            }
8207        };
8208        let Some(id) = args.get("subagent_id").and_then(serde_json::Value::as_str) else {
8209            let err = Error::InvalidArguments {
8210                tool: SUBAGENT_STATUS.to_string(),
8211                message: "`subagent_id` is required".to_string(),
8212            };
8213            return (format!("Error: {err}"), true);
8214        };
8215        let Some(entry) = self.background_subagents.get(id) else {
8216            let err = Error::SubagentNotFound(id.to_string());
8217            return (format!("Error: {err}"), true);
8218        };
8219        if !entry.handle.is_finished() {
8220            let out = serde_json::json!({
8221                "subagent_id": id,
8222                "status": "pending",
8223                "task": entry.task,
8224                "agent_type": entry.agent_type,
8225                "started_at_ms": entry.started_at_ms,
8226            });
8227            return (out.to_string(), false);
8228        }
8229        // Finished — reap it. `.await` on an already-finished handle
8230        // resolves immediately (never actually blocks).
8231        let entry = self
8232            .background_subagents
8233            .remove(id)
8234            .expect("checked Some above");
8235        let (child_id, result, transcript) = match entry.handle.await {
8236            Ok(v) => v,
8237            Err(join_err) => {
8238                let err = Error::tool(
8239                    SUBAGENT_STATUS,
8240                    format!("subagent `{id}` task panicked: {join_err}"),
8241                );
8242                return (format!("Error: {err}"), true);
8243            }
8244        };
8245        // Re-derive the lineage record for persistence (cheap; the fields
8246        // are all still in hand) — mirrors the foreground path's single
8247        // `persist_subagent_transcript` call site.
8248        if let Some((store, parent_name)) = self.subagent_store.clone() {
8249            if let Ok(Some(lineage)) = store.load_subagent_lineage(&parent_name, &child_id) {
8250                self.persist_subagent_transcript(&child_id, &lineage, &transcript);
8251            }
8252        }
8253        // BP-7: see the foreground path's identical line.
8254        self.reaped_subagents
8255            .insert(child_id.clone(), transcript.clone());
8256        match result {
8257            Ok(text) => {
8258                let out = serde_json::json!({
8259                    "subagent_id": child_id,
8260                    "status": "done",
8261                    "result": text,
8262                });
8263                (out.to_string(), false)
8264            }
8265            Err(e) => {
8266                let out = serde_json::json!({
8267                    "subagent_id": child_id,
8268                    "status": "error",
8269                    "message": e.to_string(),
8270                });
8271                (out.to_string(), true)
8272            }
8273        }
8274    }
8275
8276    /// Execute the `background_exec` intrinsic (P5-6, §2 module 4, D1
8277    /// "background exec"): spawn `args.command` as a detached OS process
8278    /// via `crate::tools::build_sandboxed_sh` — the SAME sandboxed-spawn
8279    /// path [`crate::tools::BashTool::execute`] uses — and return its job
8280    /// id IMMEDIATELY, never the command's output. Gated by the same
8281    /// permission check a foreground `bash` call gets
8282    /// ([`Self::background_permission_denial`]), then a fail-closed
8283    /// concurrency cap ([`Config::tools_background_max_concurrent`]), THEN
8284    /// the actual spawn — in that order, so a refused call never holds a
8285    /// concurrency slot and never touches the process table.
8286    fn run_background_exec(&mut self, call: &supercode_interchange::ToolCall) -> (String, bool) {
8287        let args = match call.function.parsed_arguments() {
8288            Ok(v) => v,
8289            Err(e) => {
8290                let err = Error::InvalidArguments {
8291                    tool: BACKGROUND_EXEC.to_string(),
8292                    message: e.to_string(),
8293                };
8294                return (format!("Error: {err}"), true);
8295            }
8296        };
8297        let command = args
8298            .get("command")
8299            .and_then(serde_json::Value::as_str)
8300            .unwrap_or("")
8301            .to_string();
8302        if command.is_empty() {
8303            let err = Error::InvalidArguments {
8304                tool: BACKGROUND_EXEC.to_string(),
8305                message: "`command` is required and must be non-empty".to_string(),
8306            };
8307            return (format!("Error: {err}"), true);
8308        }
8309
8310        // A job id up front (before spawning) — used both as the audit
8311        // handle for a §2.2 C6 `Parent`-policy queued denial (this call may
8312        // never actually reach the spawn below) and, if the call proceeds,
8313        // as `Self::background_jobs`'s real key.
8314        let job_id = supercode_runtime::background::next_job_id(now_ms());
8315
8316        // Fable-5 review (LOW, "pre_tool_hook + doom-loop don't cover
8317        // background_exec"): this intrinsic is intercepted in
8318        // `Self::prepare_tool_call` and returns before `Self::finish_prepare`
8319        // ever runs, so — unlike a foreground `bash` call — it was reaching
8320        // this real spawn below WITHOUT ever offering `Config.pre_tool_hook`
8321        // a chance to veto it. `background_exec` runs a REAL command (unlike
8322        // the purely in-process meta-intrinsics `tool_search`/
8323        // `expand_reduction`/`sidecar_search`, which have no such gap to
8324        // close), so it belongs behind the same security-relevant veto a
8325        // foreground call gets. Scoped to this one call site — the other
8326        // meta-intrinsics are unchanged. The doom-loop counter
8327        // (`Self::check_doom_loop`) is deliberately NOT wired here: it is a
8328        // foreground repetition breaker keyed on `(self.doom_loop_last_call,
8329        // self.doom_loop_streak)`, a single piece of state shared with the
8330        // ordinary tool-call loop — folding background jobs into that same
8331        // streak would make an interleaved foreground/background pattern
8332        // trip (or fail to trip) the breaker in ways that have nothing to
8333        // do with the foreground loop actually repeating itself; the
8334        // pre_tool_hook veto below is the security-relevant half of this
8335        // fix, the doom-loop breaker is not.
8336        // BP-10: the hook now runs BEFORE this path's permissions gate, the
8337        // same order the foreground path uses — so a rewrite is what the
8338        // rules evaluate and what actually runs, and the hook's
8339        // `Allow`/`Ask` are tiers inside the engine rather than a second
8340        // verdict beside it.
8341        let mut command = command;
8342        let mut hook_decision = crate::config::HookDecision::Pass;
8343        if let Some(hook) = &self.config.pre_tool_hook {
8344            let outcome = hook(BACKGROUND_EXEC, &args);
8345            if outcome.decision == crate::config::HookDecision::Deny {
8346                let reason = outcome.reason.unwrap_or_else(|| "denied".to_string());
8347                return (format!("Error: blocked by pre-tool hook: {reason}"), true);
8348            }
8349            if let Some(rewritten) = outcome.updated_args {
8350                command = rewritten
8351                    .get("command")
8352                    .and_then(|v| v.as_str())
8353                    .unwrap_or(&command)
8354                    .to_string();
8355            }
8356            hook_decision = outcome.decision;
8357        }
8358
8359        if let Some(reason) = self.background_permission_denial(&command, &job_id, hook_decision) {
8360            return (format!("Error: {reason}"), true);
8361        }
8362
8363        let Some(guard) = crate::subagents::try_acquire(
8364            &self.background_concurrency_gauge,
8365            self.config.tools_background_max_concurrent,
8366        ) else {
8367            let err = Error::BackgroundJobConcurrencyExceeded {
8368                max_concurrent: self.config.tools_background_max_concurrent,
8369            };
8370            return (format!("Error: {err}"), true);
8371        };
8372
8373        let mut cmd = match crate::tools::build_sandboxed_sh(&command, &self.ctx) {
8374            Ok(cmd) => cmd,
8375            Err(e) => return (format!("Error: {e}"), true),
8376        };
8377        cmd.current_dir(&self.ctx.cwd)
8378            .stdin(std::process::Stdio::null())
8379            .stdout(std::process::Stdio::piped())
8380            .stderr(std::process::Stdio::piped())
8381            // Defense-in-depth for the "must be killed on drop" guarantee —
8382            // see `impl Drop for Agent`'s doc comment; the EXPLICIT
8383            // `start_kill()` loop there is what makes the guarantee
8384            // provable, this is a second, independent line of defense for
8385            // the same outcome.
8386            .kill_on_drop(true);
8387        // Fable-5 review (HIGH, "grandchildren orphaned on kill AND
8388        // agent-drop"): `Child::start_kill` only signals the DIRECT child.
8389        // A background command that spawns a surviving subprocess (a `&`
8390        // job, a pipeline, a double-forking daemon — or, on macOS, the
8391        // `sandbox-exec` wrapper itself in `build_sandboxed_sh`, whose real
8392        // `sh` and ITS children are all grandchildren of the tracked pid)
8393        // leaves those processes running, reparented to init, after the
8394        // tracked job is "killed". Putting this job in its OWN new process
8395        // group (`pgid == its own pid`, since every descendant inherits the
8396        // group unless it explicitly opts out) lets `kill_job_process_group`
8397        // below signal the WHOLE tree at kill/drop time, not just the one
8398        // pid we happen to be tracking. No portable equivalent on Windows —
8399        // see `kill_job_process_group`'s `#[cfg(not(unix))]` fallback.
8400        #[cfg(unix)]
8401        cmd.process_group(0);
8402        // P4c (`core.shell_env_snapshot`)/P5-10 (`env_policy`):
8403        // `build_sandboxed_sh` (above) already applied both via its own
8404        // `apply_sandbox_env_policy` last step — no separate `ctx.shell_env`
8405        // application here (that would re-add a secret `Filtered`/`None`
8406        // just stripped, on top of the already-`env_clear`'d command).
8407
8408        let mut child = match cmd.spawn() {
8409            Ok(c) => c,
8410            Err(e) => {
8411                drop(guard);
8412                let err = Error::tool(
8413                    BACKGROUND_EXEC,
8414                    format!("failed to spawn background command: {e}"),
8415                );
8416                return (format!("Error: {err}"), true);
8417            }
8418        };
8419        let pid = child.id();
8420        let output = std::sync::Arc::new(supercode_runtime::background::CapturedOutput::new());
8421        let cap = self.config.tools_background_max_output_bytes;
8422        // Fire-and-forget: the reader tasks outlive this method call and
8423        // exit on their own at pipe EOF — see `spawn_output_reader`'s doc
8424        // comment. Bound to named (not `_`) locals only to keep clippy's
8425        // `let_underscore_future` lint quiet; neither handle is awaited or
8426        // aborted anywhere.
8427        if let Some(stdout) = child.stdout.take() {
8428            let _stdout_reader = spawn_output_reader(stdout, output.clone(), cap);
8429        }
8430        if let Some(stderr) = child.stderr.take() {
8431            let _stderr_reader = spawn_output_reader(stderr, output.clone(), cap);
8432        }
8433
8434        let started_at_ms = now_ms();
8435        self.background_jobs.insert(
8436            job_id.clone(),
8437            BackgroundJob {
8438                child,
8439                command: command.clone(),
8440                pid,
8441                output,
8442                started_at_ms,
8443                killed: false,
8444                _guard: guard,
8445            },
8446        );
8447
8448        let out = serde_json::json!({
8449            "job_id": job_id,
8450            "status": "running",
8451            "pid": pid,
8452            "command": command,
8453        });
8454        (out.to_string(), false)
8455    }
8456
8457    /// Execute the `background_status` intrinsic (P5-6, D1 "monitor/event
8458    /// feed"): non-blocking poll of one job's run status (via
8459    /// `Child::try_wait`), drain its output captured since the LAST poll
8460    /// and emit it as an [`AgentEvent::BackgroundOutput`] event (the
8461    /// "event feed" — a real `EventSink` consumer sees each poll's new
8462    /// output live), and return the full captured output (bounded, per
8463    /// [`Config::tools_background_max_output_bytes`]) so far either way.
8464    /// Once the job is terminal (exited or killed), this reaps it — removes
8465    /// it from [`Self::background_jobs`], freeing its concurrency slot —
8466    /// same "poll once more to reap" contract [`Self::run_subagent_status`]
8467    /// already established for background subagents.
8468    fn run_background_status(&mut self, call: &supercode_interchange::ToolCall) -> (String, bool) {
8469        let args = match call.function.parsed_arguments() {
8470            Ok(v) => v,
8471            Err(e) => {
8472                let err = Error::InvalidArguments {
8473                    tool: BACKGROUND_STATUS.to_string(),
8474                    message: e.to_string(),
8475                };
8476                return (format!("Error: {err}"), true);
8477            }
8478        };
8479        let Some(job_id) = args.get("job_id").and_then(serde_json::Value::as_str) else {
8480            let err = Error::InvalidArguments {
8481                tool: BACKGROUND_STATUS.to_string(),
8482                message: "`job_id` is required".to_string(),
8483            };
8484            return (format!("Error: {err}"), true);
8485        };
8486        let job_id = job_id.to_string();
8487
8488        // Scoped so the mutable borrow of `self.background_jobs` ends
8489        // before `self.emit(...)`/`self.background_jobs.remove(...)` below
8490        // need their own (mutable) access to `self`.
8491        let (command, pid, started_at_ms, status, output_so_far, truncated, delta) = {
8492            let Some(job) = self.background_jobs.get_mut(&job_id) else {
8493                let err = Error::BackgroundJobNotFound(job_id);
8494                return (format!("Error: {err}"), true);
8495            };
8496            let status = background_job_status(job);
8497            let (output_so_far, truncated) = job.output.snapshot();
8498            let delta = job.output.drain_new();
8499            (
8500                job.command.clone(),
8501                job.pid,
8502                job.started_at_ms,
8503                status,
8504                output_so_far,
8505                truncated,
8506                delta,
8507            )
8508        };
8509
8510        if !delta.is_empty() {
8511            self.emit(AgentEvent::BackgroundOutput {
8512                job_id: job_id.clone(),
8513                chunk: delta,
8514                truncated,
8515            });
8516        }
8517
8518        let exit_code = match status {
8519            supercode_runtime::background::JobStatus::Exited(code) => code,
8520            _ => None,
8521        };
8522        let out = serde_json::json!({
8523            "job_id": job_id,
8524            "command": command,
8525            "status": status.as_str(),
8526            "exit_code": exit_code,
8527            "pid": pid,
8528            "started_at_ms": started_at_ms,
8529            "output": output_so_far,
8530            "output_truncated": truncated,
8531        });
8532        if !matches!(status, supercode_runtime::background::JobStatus::Running) {
8533            self.background_jobs.remove(&job_id);
8534        }
8535        (out.to_string(), false)
8536    }
8537
8538    /// Execute the `background_list` intrinsic (P5-6, D10 "bg-manager"):
8539    /// list every background job this agent is currently tracking, without
8540    /// draining output or reaping anything (a read-only listing —
8541    /// `background_status` is the reaping poll).
8542    fn run_background_list(&mut self, _call: &supercode_interchange::ToolCall) -> (String, bool) {
8543        let mut jobs = Vec::new();
8544        for (job_id, job) in self.background_jobs.iter_mut() {
8545            let status = background_job_status(job);
8546            jobs.push(serde_json::json!({
8547                "job_id": job_id,
8548                "command": job.command,
8549                "status": status.as_str(),
8550                "pid": job.pid,
8551                "started_at_ms": job.started_at_ms,
8552            }));
8553        }
8554        let out = serde_json::json!({ "jobs": jobs });
8555        (out.to_string(), false)
8556    }
8557
8558    /// Execute the `background_kill` intrinsic (P5-6, D10 "bg-manager",
8559    /// build brief "kill/cancel a job"): request REAL termination of a
8560    /// background job's OS process AND its whole process group (see
8561    /// [`kill_job_process_group`] — Fable-5 review, HIGH, "grandchildren
8562    /// orphaned on kill"; a documented no-op if the process already
8563    /// exited) and reap it immediately, freeing its concurrency slot.
8564    fn run_background_kill(&mut self, call: &supercode_interchange::ToolCall) -> (String, bool) {
8565        let args = match call.function.parsed_arguments() {
8566            Ok(v) => v,
8567            Err(e) => {
8568                let err = Error::InvalidArguments {
8569                    tool: BACKGROUND_KILL.to_string(),
8570                    message: e.to_string(),
8571                };
8572                return (format!("Error: {err}"), true);
8573            }
8574        };
8575        let Some(job_id) = args.get("job_id").and_then(serde_json::Value::as_str) else {
8576            let err = Error::InvalidArguments {
8577                tool: BACKGROUND_KILL.to_string(),
8578                message: "`job_id` is required".to_string(),
8579            };
8580            return (format!("Error: {err}"), true);
8581        };
8582        let job_id = job_id.to_string();
8583        let Some(mut job) = self.background_jobs.remove(&job_id) else {
8584            let err = Error::BackgroundJobNotFound(job_id);
8585            return (format!("Error: {err}"), true);
8586        };
8587        kill_job_process_group(&mut job);
8588        job.killed = true;
8589        let out = serde_json::json!({
8590            "job_id": job_id,
8591            "status": "killed",
8592            "pid": job.pid,
8593        });
8594        // `job` (and its `ConcurrencyGuard`) drops here, freeing the slot.
8595        (out.to_string(), false)
8596    }
8597
8598    /// P5-3 (D5 "subagent transcripts… persisted + linked"): write a
8599    /// finished child's transcript to `Self::subagent_store`, if one is
8600    /// installed — a no-op otherwise (see that field's doc comment). Builds
8601    /// the child's `Session` the same way `to_native_jsonl_v2`'s doc
8602    /// comment describes (an empty imported prefix + `transcript` as
8603    /// `appended` `NativeTurn`s), with `meta.agent_id`/`parent_tool_use_id`/
8604    /// `lineage` populated from `lineage` so the native-v2 header carries
8605    /// the full lineage record on disk (see `Session::to_native_jsonl_v2`'s
8606    /// P5-3 doc note).
8607    ///
8608    /// P5-3 safety-hardening fix (Fable-5 review, LOW "translation-fidelity
8609    /// cosmetic"): `Session::from_claude_code_str("")` is used ONLY to get
8610    /// a blank `raw`/`messages` skeleton cheaply (an empty string parses
8611    /// identically under any loader) — it is NOT claiming this child's
8612    /// session actually came from Claude Code. Before this fix, that
8613    /// borrowed constructor's `meta.source` (`SessionSource::ClaudeCode`)
8614    /// leaked straight through to the persisted sidecar's `source` header,
8615    /// mislabeling a native `spawn_subagent` child as an imported CC
8616    /// session. Corrected to `SessionSource::Native` immediately after —
8617    /// see that variant's doc comment.
8618    fn persist_subagent_transcript(
8619        &self,
8620        child_id: &str,
8621        lineage: &crate::subagents::SubagentLineage,
8622        transcript: &[ChatMessage],
8623    ) {
8624        let Some((store, parent_name)) = &self.subagent_store else {
8625            return;
8626        };
8627        let mut session = match Session::from_claude_code_str("") {
8628            Ok(s) => s,
8629            Err(_) => return,
8630        };
8631        session.meta.source = supercode_interchange::session::SessionSource::Native;
8632        session.meta.agent_id = Some(lineage.child_agent_id.clone());
8633        session.meta.parent_tool_use_id = Some(lineage.parent_tool_use_id.clone());
8634        session.meta.lineage = lineage.to_lineage_map();
8635        let sidecar_jsonl = session.to_native_jsonl_v2(transcript);
8636        let _ = store.save_subagent_transcript(parent_name, child_id, &sidecar_jsonl);
8637        let _ = store.save_subagent_lineage(parent_name, child_id, lineage);
8638    }
8639
8640    /// Execute the `tool_search` intrinsic (B6): case-insensitive keyword
8641    /// match over `name` + `description` of every registered, enabled,
8642    /// non-core, not-yet-activated tool (builtin and `mcp__*` alike). Matches
8643    /// are activated (advertised starting with the next request) and
8644    /// returned as a JSON array of their full [`ToolSchema`]s.
8645    fn run_tool_search(&mut self, call: &supercode_interchange::ToolCall) -> (String, bool) {
8646        let args = match call.function.parsed_arguments() {
8647            Ok(v) => v,
8648            Err(e) => {
8649                let err = Error::InvalidArguments {
8650                    tool: TOOL_SEARCH.to_string(),
8651                    message: e.to_string(),
8652                };
8653                return (format!("Error: {err}"), true);
8654            }
8655        };
8656        let query = args
8657            .get("query")
8658            .and_then(serde_json::Value::as_str)
8659            .unwrap_or("")
8660            .to_lowercase();
8661        let max_results = args
8662            .get("max_results")
8663            .and_then(serde_json::Value::as_u64)
8664            .map(|n| n as usize);
8665
8666        let mut matches: Vec<ToolSchema> = self
8667            .registry
8668            .iter()
8669            .filter(|t| self.config.tool_enabled(t.name()))
8670            .filter(|t| !self.is_core_tool(t.name()))
8671            .filter(|t| !self.activated_tools.contains(t.name()))
8672            .filter(|t| {
8673                query.is_empty()
8674                    || t.name().to_lowercase().contains(&query)
8675                    || self
8676                        .config
8677                        .tool_description(t.name(), t.description())
8678                        .to_lowercase()
8679                        .contains(&query)
8680            })
8681            // TR-8/T5 dev/03: the on-demand fetch always returns the ORIGINAL
8682            // full schema, never the tier-minified one — that's the invert.
8683            .map(|t| self.raw_schema_for(t))
8684            .collect();
8685
8686        if let Some(max) = max_results {
8687            matches.truncate(max);
8688        }
8689
8690        for m in &matches {
8691            self.activated_tools.insert(m.name.clone());
8692        }
8693
8694        let result = serde_json::to_string(&matches).unwrap_or_else(|_| "[]".to_string());
8695        (result, false)
8696    }
8697
8698    /// The `expand_reduction` schema (T12/TR-1), advertised whenever a
8699    /// [`ReductionPolicy`] is installed.
8700    ///
8701    /// The description deliberately never spells the literal stub sentinel
8702    /// prefix: A11's export leak guard is unconditional, so an assistant
8703    /// turn that quoted a stub line verbatim (which teaching the syntax
8704    /// invites) would permanently fail export for that session. Stubs are
8705    /// described abstractly and the model is told to pass ids only.
8706    fn expand_reduction_schema() -> ToolSchema {
8707        ToolSchema {
8708            name: EXPAND_REDUCTION.to_string(),
8709            description: "Fetch back the original content hidden behind a reduction stub in \
8710                your current view — a truncated tool output, cleared old turns, or an elided \
8711                file read that was hidden to save context. Each stub line names a reduction id \
8712                like r0042-9f3c: pass ONLY that id here, and never quote or repeat a stub line \
8713                itself in your replies. The original is durably kept in the session sidecar. \
8714                Pass `byte_range` to fetch a slice of a large one at a time instead of all of \
8715                it at once; ranged results are prefixed with a `bytes start..end of total` \
8716                header so you can plan the next slice."
8717                .to_string(),
8718            parameters: serde_json::json!({
8719                "type": "object",
8720                "properties": {
8721                    "reduction_id": {
8722                        "type": "string",
8723                        "description": "The reduction id named in the stub line, e.g. \
8724                            \"r0042-9f3c\". Pass the id alone."
8725                    },
8726                    "byte_range": {
8727                        "type": "array",
8728                        "items": {"type": "integer"},
8729                        "minItems": 2,
8730                        "maxItems": 2,
8731                        "description": "Optional [start, end) byte offsets within the original \
8732                            content to fetch instead of all of it. Exactly two non-negative \
8733                            integers with start <= end."
8734                    }
8735                },
8736                "required": ["reduction_id"],
8737                "additionalProperties": false
8738            }),
8739        }
8740    }
8741
8742    /// The `sidecar_search` schema (T12/TR-1), advertised whenever a
8743    /// [`ReductionPolicy`] is installed. Same no-literal-sentinel rule as
8744    /// [`Self::expand_reduction_schema`].
8745    fn sidecar_search_schema() -> ToolSchema {
8746        ToolSchema {
8747            name: SIDECAR_SEARCH.to_string(),
8748            description: "Search content currently hidden from your view by reduction stubs \
8749                (large tool outputs, cleared old turns, elided file reads) for a substring or \
8750                regex. Only hidden content is searched, never what you can already see. \
8751                Returns match snippets with each match's reduction_id for use with \
8752                expand_reduction; refer to results by their reduction id rather than quoting \
8753                stub lines. Results are capped — if `truncated` is true, narrow the query."
8754                .to_string(),
8755            parameters: serde_json::json!({
8756                "type": "object",
8757                "properties": {
8758                    "query": {
8759                        "type": "string",
8760                        "description": "Non-empty substring or regex to search for \
8761                            (case-insensitive)."
8762                    }
8763                },
8764                "required": ["query"],
8765                "additionalProperties": false
8766            }),
8767        }
8768    }
8769
8770    /// Reload the recorder's full recorded messages from disk (TR-1's
8771    /// `recorded` resolution source). Since TR-12's D6/A7 supersession gate
8772    /// (`Self::run_loop`), a `expand_reduction`/`sidecar_search` call only
8773    /// ever exists alongside an active [`ReductionPolicy`] (see
8774    /// [`EXPAND_REDUCTION`]'s doc), and pairing one with a recorder — as the
8775    /// CLI's reduced mode always does — means the gate is already on and
8776    /// `history[1..]` holds the same full bytes as this reload: this upgrade
8777    /// is then a dormant no-op (`reduce::rehydrate::prefer_recorded` sees
8778    /// `recorded == minted` and keeps `minted`). It stops being a no-op —
8779    /// defense in depth, not the common path — for a **legacy** sidecar
8780    /// recorded before this gate existed, or for a policy-without-recorder
8781    /// agent (gate off, so `history[1..]` still carries
8782    /// [`Self::cap_tool_output`]-capped copies): only there can `history[1..]`
8783    /// diverge from the sidecar, and only there does consulting this reload
8784    /// actually recover bytes `history[1..]` alone couldn't. `Ok(None)` when
8785    /// no recorder is attached (rehydration then resolves from history alone,
8786    /// whose capped copies — if any — carry their own honest cap notice). A
8787    /// disk-level reload is fine here regardless: these intrinsic calls are
8788    /// rare, model-initiated events, not per-request work.
8789    fn recorded_messages(&self) -> std::result::Result<Option<Vec<ChatMessage>>, String> {
8790        let Some(recorder) = &self.recorder else {
8791            return Ok(None);
8792        };
8793        let raw = std::fs::read_to_string(recorder.path())
8794            .map_err(|e| format!("failed to read the session sidecar: {e}"))?;
8795        let session = Session::from_sidecar_str(&raw)
8796            .map_err(|e| format!("failed to parse the session sidecar: {e}"))?;
8797        Ok(Some(session.messages))
8798    }
8799
8800    /// Execute the `expand_reduction` intrinsic (T12/TR-1): resolves against
8801    /// `self.reduction_log` + `self.history[1..]` (the hash-minting source),
8802    /// upgraded to the recorder's full recorded bytes for cap-diverged
8803    /// content ([`Self::recorded_messages`]; the two-source contract is
8804    /// documented on `reduce::rehydrate`). `byte_range` is validated
8805    /// strictly — any malformed shape is a model-recoverable error naming
8806    /// the expected form and the original's true size, never a silent
8807    /// whole-content (or empty) return.
8808    fn run_expand_reduction(&mut self, call: &supercode_interchange::ToolCall) -> (String, bool) {
8809        let args = match call.function.parsed_arguments() {
8810            Ok(v) => v,
8811            Err(e) => {
8812                let err = Error::InvalidArguments {
8813                    tool: EXPAND_REDUCTION.to_string(),
8814                    message: e.to_string(),
8815                };
8816                return (format!("Error: {err}"), true);
8817            }
8818        };
8819        let Some(id) = args.get("reduction_id").and_then(serde_json::Value::as_str) else {
8820            return (
8821                "Error: expand_reduction requires a `reduction_id` string argument".to_string(),
8822                true,
8823            );
8824        };
8825        let recorded = match self.recorded_messages() {
8826            Ok(r) => r,
8827            Err(e) => return (format!("Error: expand_reduction: {e}"), true),
8828        };
8829        let recorded = recorded.as_deref();
8830
8831        // B3: strict shape validation — exactly two non-negative integers.
8832        // Anything else errors (with the true total when resolvable) rather
8833        // than silently degrading to a whole-content expand.
8834        let byte_range = match args.get("byte_range") {
8835            None | Some(serde_json::Value::Null) => None,
8836            Some(v) => {
8837                let parsed = v
8838                    .as_array()
8839                    .filter(|a| a.len() == 2)
8840                    .and_then(|a| Some((a[0].as_u64()? as usize, a[1].as_u64()? as usize)));
8841                match parsed {
8842                    Some(range) => Some(range),
8843                    None => {
8844                        let total = reduce::rehydrate::reduction_total_bytes(
8845                            &self.reduction_log,
8846                            &self.history[1..],
8847                            recorded,
8848                            id,
8849                        )
8850                        .map(|n| format!("; the original is {n} bytes"))
8851                        .unwrap_or_default();
8852                        return (
8853                            format!(
8854                                "Error: expand_reduction: malformed byte_range {v} — expected \
8855                                 [start, end): exactly two non-negative integers with \
8856                                 start <= end{total}"
8857                            ),
8858                            true,
8859                        );
8860                    }
8861                }
8862            }
8863        };
8864        match reduce::rehydrate::expand_reduction(
8865            &self.reduction_log,
8866            &self.history[1..],
8867            recorded,
8868            id,
8869            byte_range,
8870        ) {
8871            // A ranged result carries a provenance header naming the slice
8872            // and the true total, so the model can plan its next slice; a
8873            // whole-content expand stays byte-exact (TR-1 dev/01).
8874            Ok(outcome) => match outcome.range {
8875                Some((start, end)) => (
8876                    format!(
8877                        "[{id}: bytes {start}..{end} of {total}]\n{content}",
8878                        total = outcome.total_bytes,
8879                        content = outcome.content
8880                    ),
8881                    false,
8882                ),
8883                None => (outcome.content, false),
8884            },
8885            Err(e) => (format!("Error: {e}"), true),
8886        }
8887    }
8888
8889    /// Execute the `sidecar_search` intrinsic (T12/TR-1); same two-source
8890    /// resolution as [`Self::run_expand_reduction`]. The result is bounded
8891    /// by construction (`reduce::rehydrate::SidecarSearchResult`'s caps), so
8892    /// a broad query can never re-inflate the context or bloat the sidecar
8893    /// the recorder appends this result to.
8894    fn run_sidecar_search(&mut self, call: &supercode_interchange::ToolCall) -> (String, bool) {
8895        let args = match call.function.parsed_arguments() {
8896            Ok(v) => v,
8897            Err(e) => {
8898                let err = Error::InvalidArguments {
8899                    tool: SIDECAR_SEARCH.to_string(),
8900                    message: e.to_string(),
8901                };
8902                return (format!("Error: {err}"), true);
8903            }
8904        };
8905        let query = args
8906            .get("query")
8907            .and_then(serde_json::Value::as_str)
8908            .unwrap_or("");
8909        if query.trim().is_empty() {
8910            return (
8911                "Error: sidecar_search requires a non-empty `query` string argument".to_string(),
8912                true,
8913            );
8914        }
8915        let recorded = match self.recorded_messages() {
8916            Ok(r) => r,
8917            Err(e) => return (format!("Error: sidecar_search: {e}"), true),
8918        };
8919        match reduce::rehydrate::sidecar_search(
8920            &self.reduction_log,
8921            &self.history[1..],
8922            recorded.as_deref(),
8923            query,
8924        ) {
8925            Ok(result) => (
8926                serde_json::to_string(&result).unwrap_or_else(|_| "{}".to_string()),
8927                false,
8928            ),
8929            Err(e) => (format!("Error: {e}"), true),
8930        }
8931    }
8932
8933    fn emit(&self, event: AgentEvent) {
8934        if let Some(sink) = &self.config.event_sink {
8935            sink(event);
8936        }
8937    }
8938
8939    /// Number of non-system messages exchanged so far.
8940    pub fn turn_count(&self) -> usize {
8941        self.history
8942            .iter()
8943            .filter(|m| m.role != Role::System)
8944            .count()
8945    }
8946
8947    /// Cumulative output (completion) tokens reported by the provider across
8948    /// every `send` on this agent. Zero if the provider reports no usage.
8949    pub fn total_output_tokens(&self) -> u64 {
8950        self.total_output_tokens
8951    }
8952}
8953
8954/// Deliver an agent's `send_message` to another session through the one
8955/// send every sender uses. The sender is this process's own session (a
8956/// runtime supercode hosts, found from its ancestry), so replies can come
8957/// back.
8958#[cfg(feature = "adapter-api")]
8959async fn send_to_session(to: &str, message: &str, notify_when_idle: bool) -> (String, bool) {
8960    let homes = crate::HarnessHomes::default();
8961    let caller =
8962        match crate::mail_route::resolve_caller(&homes, &crate::mail_route::process_ancestry()) {
8963            Ok(caller) => caller,
8964            Err(_) => {
8965                return (
8966                    format!(
8967                    "Not sent: `{to}` is not one of your running subagents, and this session has \
8968                     no address other sessions can reply to (supercode does not host it), so it \
8969                     can message only its own subagents."
8970                ),
8971                    true,
8972                )
8973            }
8974        };
8975    let options = crate::mail_send::SendOptions {
8976        notify_when_idle,
8977        ..Default::default()
8978    };
8979    match crate::mail_send::send(&homes, &caller, to, message, options).await {
8980        Ok(outcome) => (outcome.text, outcome.code != 0),
8981        Err(error) => (format!("Not sent to {to}: {error}"), true),
8982    }
8983}
8984
8985#[cfg(not(feature = "adapter-api"))]
8986async fn send_to_session(to: &str, _message: &str, _notify_when_idle: bool) -> (String, bool) {
8987    (
8988        format!("Not sent: `{to}` is not one of your running subagents, and this build has no cross-session messaging."),
8989        true,
8990    )
8991}
8992
8993#[cfg(test)]
8994mod bp2_spill_tests {
8995    //! BP-2 (`.volter/tracker/markdown/BP-2.md`, catalog:58): under the
8996    //! parity presets a capped tool output stays RECOVERABLE by the model —
8997    //! without `capabilities.reduction`, which both presets leave off.
8998
8999    use super::*;
9000    use crate::configfile::{resolve, ResolveOptions};
9001
9002    /// Never called — these tests drive `cap_tool_output` directly.
9003    #[derive(Debug)]
9004    struct NeverCalledProvider;
9005
9006    #[async_trait::async_trait]
9007    impl Provider for NeverCalledProvider {
9008        async fn complete(
9009            &self,
9010            _req: &ChatRequest,
9011            _on_delta: &(dyn for<'a> Fn(&'a str) + Send + Sync),
9012        ) -> crate::Result<(ChatMessage, crate::provider::Usage)> {
9013            unreachable!("BP-2 spill tests never issue a request")
9014        }
9015    }
9016
9017    fn resolved(preset: &str) -> crate::configfile::Resolved {
9018        let toml = crate::presets::lookup(preset).unwrap();
9019        resolve(toml, None, &ResolveOptions { strict: true })
9020            .unwrap_or_else(|e| panic!("{preset} resolves: {e}"))
9021    }
9022
9023    /// The residue this closes: the cap notice said the full output was
9024    /// "not retained" / "in session sidecar" and the only door to it —
9025    /// `expand_reduction` — is advertised solely when a `ReductionPolicy`
9026    /// is installed, which `capabilities.reduction = false` never does. So
9027    /// under both parity presets a truncated output was simply lost.
9028    ///
9029    /// Now the notice NAMES a spill file, and the door is the preset's own
9030    /// read pathway: `read_file` under cc-parity, the shell under
9031    /// cx-parity (which registers no file tools at all).
9032    #[tokio::test]
9033    async fn parity_presets_spill_capped_output_and_name_a_door_the_preset_has() {
9034        for (preset, expected_door) in [
9035            ("cc-parity", "read it with `read_file`"),
9036            ("cx-parity", "read it with `cat`"),
9037        ] {
9038            let r = resolved(preset);
9039            assert!(
9040                r.config.tool_output_spill,
9041                "{preset} must set `core.tool_output_spill`"
9042            );
9043            assert_eq!(
9044                r.modules.get("reduction"),
9045                Some(&false),
9046                "{preset} leaves `capabilities.reduction` off — the spill must not depend on it"
9047            );
9048            let mut config = resolved(preset).config;
9049            config.max_tool_output_bytes = Some(1024);
9050            let registry = crate::tools::ToolRegistry::from_config(&config);
9051            let agent = Agent::with_parts(config, Box::new(NeverCalledProvider), registry);
9052            assert!(
9053                agent.reduction_policy.is_none(),
9054                "no reduction policy is installed under {preset}"
9055            );
9056
9057            let full = "R".repeat(50_000);
9058            let capped = agent.cap_tool_output(full.clone());
9059            assert!(capped.len() < full.len(), "{preset}: output must be capped");
9060            assert!(capped.contains(expected_door), "{preset}: {capped:?}");
9061
9062            // The path in the notice must actually hold the full bytes.
9063            let marker = capped.split("spilled to ").nth(1).unwrap_or_default();
9064            let path = marker.split(" — ").next().unwrap_or_default();
9065            assert!(!path.is_empty(), "{preset}: no spill path in {capped:?}");
9066            assert_eq!(
9067                std::fs::read_to_string(path).unwrap(),
9068                full,
9069                "{preset}: the spill file must hold the FULL output"
9070            );
9071
9072            // And the model can actually walk through that door: the
9073            // preset's own read pathway returns the spilled content.
9074            let ctx = build_tool_context(agent.config()).0;
9075            let recovered = match registry_read_tool(&agent) {
9076                Some(("read_file", tool)) => tool
9077                    .execute(serde_json::json!({"path": path}), &ctx)
9078                    .await
9079                    .unwrap(),
9080                Some(("bash", tool)) => tool
9081                    .execute(serde_json::json!({"command": format!("cat {path}")}), &ctx)
9082                    .await
9083                    .unwrap(),
9084                _ => panic!("{preset}: no read door registered"),
9085            };
9086            assert!(
9087                recovered.contains(&"R".repeat(2000)),
9088                "{preset}: the door must return the spilled output"
9089            );
9090            let _ = std::fs::remove_file(path);
9091        }
9092    }
9093
9094    /// The preset's read pathway: `read_file` where it exists, else the
9095    /// shell — the same choice the cap notice's wording makes.
9096    fn registry_read_tool<'a>(
9097        agent: &'a Agent,
9098    ) -> Option<(&'static str, &'a dyn crate::tools::Tool)> {
9099        if let Some(tool) = agent.registry.get("read_file") {
9100            return Some(("read_file", tool));
9101        }
9102        agent.registry.get("bash").map(|tool| ("bash", tool))
9103    }
9104
9105    /// Off (the default, every non-parity preset and every SDK embedder):
9106    /// no spill file, and the notice is byte-identical to before BP-2.
9107    #[test]
9108    fn spill_off_leaves_the_notice_unchanged_and_writes_nothing() {
9109        let config = Config::builder().max_tool_output_bytes(1024).build();
9110        assert!(!config.tool_output_spill);
9111        let agent = Agent::with_provider(config, Box::new(NeverCalledProvider));
9112        let capped = agent.cap_tool_output("S".repeat(50_000));
9113        assert!(capped.contains("full output not retained"), "{capped:?}");
9114        assert!(!capped.contains("spilled to"), "{capped:?}");
9115    }
9116}
9117
9118#[cfg(test)]
9119mod bp3_new_core_tool_tests {
9120    //! BP-3 (`.volter/tracker/markdown/BP-3.md`): the two behaviours the
9121    //! new tools can only have INSIDE the agent — plan mode narrowing the
9122    //! permissions engine, and `new_context` re-founding the request view
9123    //! through `reduce`'s handoff projection — driven over the RESOLVED
9124    //! parity presets.
9125
9126    use super::*;
9127    use crate::configfile::{resolve, ResolveOptions};
9128
9129    #[derive(Debug)]
9130    struct NeverCalledProvider;
9131
9132    #[async_trait::async_trait]
9133    impl Provider for NeverCalledProvider {
9134        async fn complete(
9135            &self,
9136            _req: &ChatRequest,
9137            _on_delta: &(dyn for<'a> Fn(&'a str) + Send + Sync),
9138        ) -> crate::Result<(ChatMessage, crate::provider::Usage)> {
9139            unreachable!("BP-3 tests never issue a request")
9140        }
9141    }
9142
9143    fn resolved(preset: &str) -> Config {
9144        let toml = crate::presets::lookup(preset).unwrap();
9145        resolve(toml, None, &ResolveOptions { strict: true })
9146            .unwrap_or_else(|e| panic!("{preset} resolves: {e}"))
9147            .config
9148    }
9149
9150    fn agent_for(preset: &str) -> Agent {
9151        Agent::with_provider(resolved(preset), Box::new(NeverCalledProvider))
9152    }
9153
9154    /// An approval door that says yes to everything — so a denial in these
9155    /// tests can only come from the DENY tier, never from cc-parity's
9156    /// `approval = "untrusted"` ask default.
9157    struct AlwaysAllow;
9158    impl crate::permissions::PermissionsApprovalHandler for AlwaysAllow {
9159        fn ask(
9160            &self,
9161            _req: &crate::permissions::ApprovalRequest,
9162        ) -> crate::permissions::ApprovalOutcome {
9163            crate::permissions::ApprovalOutcome::Allow
9164        }
9165    }
9166
9167    fn call(name: &str, args: serde_json::Value) -> supercode_interchange::ToolCall {
9168        supercode_interchange::ToolCall {
9169            id: format!("call-{name}"),
9170            kind: "function".to_string(),
9171            function: supercode_interchange::FunctionCall {
9172                name: name.to_string(),
9173                arguments: args.to_string(),
9174            },
9175        }
9176    }
9177
9178    /// The row `plan-mode-read-only-research-phase` claims: under the
9179    /// resolved `cc-parity` preset, entering plan mode makes the
9180    /// permissions engine REFUSE write and execution tools — and the
9181    /// refusal survives an approval door that allows everything, because
9182    /// the mode contributes DENY rules, the tier no approval can override.
9183    #[test]
9184    fn cc_parity_plan_mode_denies_writes_through_the_permissions_engine() {
9185        let mut agent = agent_for("cc-parity");
9186        assert!(
9187            agent.config.permissions_enabled,
9188            "cc-parity runs the permissions engine; plan mode narrows it"
9189        );
9190        agent.set_permissions_approval_handler(AlwaysAllow);
9191
9192        let write = serde_json::json!({"path": "notes.txt", "content": "x"});
9193        let bash = serde_json::json!({"command": "echo hi"});
9194        assert!(
9195            agent
9196                .permissions_gate_denial("write_file", &write, crate::config::HookDecision::Pass)
9197                .is_none(),
9198            "outside plan mode an allowed write must pass"
9199        );
9200        assert!(agent
9201            .permissions_gate_denial("bash", &bash, crate::config::HookDecision::Pass)
9202            .is_none());
9203
9204        agent.plan_mode().enter(Some("research first"));
9205
9206        let denial = agent
9207            .permissions_gate_denial("write_file", &write, crate::config::HookDecision::Pass)
9208            .expect("plan mode must refuse a write");
9209        assert!(denial.contains("Deny"), "{denial}");
9210        assert!(agent
9211            .permissions_gate_denial("bash", &bash, crate::config::HookDecision::Pass)
9212            .is_some());
9213        assert!(agent
9214            .permissions_gate_denial(
9215                "apply_patch",
9216                &serde_json::json!({"patch": "*** Begin Patch\n*** End Patch"}),
9217                crate::config::HookDecision::Pass
9218            )
9219            .is_some());
9220
9221        // The research surface, and the way out, stay open.
9222        for (tool, args) in [
9223            ("read_file", serde_json::json!({"path": "notes.txt"})),
9224            ("glob", serde_json::json!({"pattern": "*.rs"})),
9225            ("exit_plan_mode", serde_json::json!({"plan": "the plan"})),
9226            ("ask_user", serde_json::json!({"questions": []})),
9227        ] {
9228            assert!(
9229                agent
9230                    .permissions_gate_denial(tool, &args, crate::config::HookDecision::Pass)
9231                    .is_none(),
9232                "plan mode must leave `{tool}` reachable"
9233            );
9234        }
9235
9236        agent.plan_mode().exit();
9237        assert!(
9238            agent
9239                .permissions_gate_denial("write_file", &write, crate::config::HookDecision::Pass)
9240                .is_none(),
9241            "leaving plan mode restores the write surface"
9242        );
9243    }
9244
9245    /// The `context-budget-tools` row's read half: the figure
9246    /// `get_context_remaining` reports is the agent's OWN accounting,
9247    /// computed at the moment the tool asks for it.
9248    #[test]
9249    fn cx_parity_publishes_its_context_accounting_when_the_budget_tool_runs() {
9250        let mut agent = agent_for("cx-parity");
9251        assert!(agent.ctx.context_budget.snapshot().is_none(), "nothing yet");
9252        agent
9253            .history
9254            .push(ChatMessage::user("x".repeat(4000).to_string()));
9255        let _ = agent.prepare_tool_call(&call("current_time", serde_json::json!({})));
9256        assert!(
9257            agent.ctx.context_budget.snapshot().is_none(),
9258            "an unrelated tool call must not pay for the accounting"
9259        );
9260
9261        let _ = agent.prepare_tool_call(&call("get_context_remaining", serde_json::json!({})));
9262        let published = agent
9263            .ctx
9264            .context_budget
9265            .snapshot()
9266            .expect("the budget tool's own call publishes it");
9267        // What the model reads IS `Agent::context_usage()` — the same
9268        // struct `/context` prints and the guard enforces, not a second
9269        // estimate that could disagree with it.
9270        assert_eq!(
9271            published,
9272            serde_json::to_value(agent.context_usage()).unwrap()
9273        );
9274        assert!(published["context_limit"].as_u64().unwrap() > 0);
9275        assert!(published["remaining_tokens"].as_u64().unwrap() > 0);
9276    }
9277
9278    /// The `context-budget-tools` row's write half, over the resolved
9279    /// `cx-parity` preset: a parked `new_context` request re-founds the
9280    /// request view through `reduce`'s handoff projection — objective in
9281    /// the leading system message, the tail kept, the rest covered by
9282    /// `TurnsCleared` spans that land in this agent's own reduction log.
9283    #[test]
9284    fn cx_parity_new_context_rebuilds_the_window_through_the_same_handoff_the_operator_gets() {
9285        let mut agent = agent_for("cx-parity");
9286        agent.history.push(ChatMessage::system("system"));
9287        for i in 0..12 {
9288            agent.history.push(ChatMessage::user(format!("turn {i}")));
9289        }
9290        let before = agent.history.clone();
9291
9292        agent
9293            .ctx
9294            .context_budget
9295            .request_new_context(crate::tools::NewContextRequest {
9296                objective: "finish the parser".to_string(),
9297                keep_recent: Some(2),
9298            });
9299        agent.apply_pending_new_context();
9300
9301        // Exactly what `/handoff` produces — the model's door and the
9302        // operator's door run one mechanism, so this compares against it.
9303        let mut expected =
9304            Agent::with_provider(resolved("cx-parity"), Box::new(NeverCalledProvider));
9305        expected.history = before.clone();
9306        expected.new_context("finish the parser", Some(2));
9307        assert_eq!(agent.history, expected.history);
9308
9309        assert!(
9310            agent.history.len() < before.len(),
9311            "the window must actually shrink: {} -> {}",
9312            before.len(),
9313            agent.history.len()
9314        );
9315        assert_eq!(agent.history[0], before[0], "the system prompt survives");
9316        let marker = agent.history[1].content.clone().unwrap_or_default();
9317        assert!(marker.contains("fresh working context"), "{marker}");
9318        assert!(marker.contains("finish the parser"), "{marker}");
9319        assert_eq!(
9320            agent.history[agent.history.len() - 2..],
9321            before[before.len() - 2..],
9322            "the requested tail is kept verbatim"
9323        );
9324        assert!(
9325            agent.ctx.context_budget.take_new_context().is_none(),
9326            "the request is consumed exactly once"
9327        );
9328    }
9329
9330    /// `new_context` never trades recoverability for a smaller window: with
9331    /// no sidecar recorder the request is refused, the reason is handed
9332    /// back to the model, and the transcript keeps every turn.
9333    #[test]
9334    fn cx_parity_new_context_states_the_retention_it_actually_has() {
9335        let mut agent = agent_for("cx-parity");
9336        agent.history.push(ChatMessage::system("system"));
9337        for i in 0..12 {
9338            agent.history.push(ChatMessage::user(format!("turn {i}")));
9339        }
9340        agent
9341            .ctx
9342            .context_budget
9343            .request_new_context(crate::tools::NewContextRequest {
9344                objective: "finish the parser".to_string(),
9345                keep_recent: Some(2),
9346            });
9347        agent.apply_pending_new_context();
9348
9349        // No recorder is installed here, and the marker says so rather than
9350        // implying the set-aside turns are still somewhere.
9351        let marker = agent.history[1].content.clone().unwrap_or_default();
9352        assert!(
9353            marker.contains("No transcript sidecar is attached"),
9354            "the marker must not overstate retention: {marker}"
9355        );
9356    }
9357}
9358
9359/// BP-8 (catalog:152): what [`Agent::rewind_conversation`] did.
9360#[derive(Debug, Clone, PartialEq, Eq)]
9361pub struct RewindOutcome {
9362    /// Messages remaining, including the system message at index 0.
9363    pub kept: usize,
9364    /// Messages removed from the live conversation (still on disk, in the
9365    /// journal, and still in the tree under `preserved_branch`).
9366    pub removed: usize,
9367    /// When the tree module is on and the rewind actually moved the leaf:
9368    /// the sibling branch the old leaf was preserved under, so the rewound
9369    /// path stays independently addressable.
9370    pub preserved_branch: Option<String>,
9371}
9372
9373#[cfg(test)]
9374mod bp1_compaction_tests {
9375    //! BP-1 (`.volter/tracker/markdown/BP-1.md` AC2): `cx-parity` fires
9376    //! [`Agent::maybe_compact`] at its own trigger, and
9377    //! `core.compaction.summarize` reaches [`Config`] and is what the
9378    //! compaction marker says.
9379
9380    use super::*;
9381    use crate::configfile::{resolve, ResolveOptions};
9382
9383    /// Never called — these tests drive `maybe_compact` directly, which
9384    /// makes no request.
9385    #[derive(Debug)]
9386    struct NeverCalledProvider;
9387
9388    #[async_trait::async_trait]
9389    impl Provider for NeverCalledProvider {
9390        async fn complete(
9391            &self,
9392            _req: &ChatRequest,
9393            _on_delta: &(dyn for<'a> Fn(&'a str) + Send + Sync),
9394        ) -> crate::Result<(ChatMessage, crate::provider::Usage)> {
9395            unreachable!("BP-1 compaction tests never issue a request")
9396        }
9397    }
9398
9399    fn resolved_cx_parity() -> Config {
9400        let toml = crate::presets::lookup("cx-parity").unwrap();
9401        resolve(toml, None, &ResolveOptions { strict: true })
9402            .expect("cx-parity resolves")
9403            .config
9404    }
9405
9406    /// Enough history to sit inside `reserve_tokens` of ANY model context
9407    /// window (`estimate_view_tokens` is size-proportional, and the largest
9408    /// window in the catalog is far below this).
9409    fn stuff_history(agent: &mut Agent) {
9410        agent.history.push(ChatMessage::system("system"));
9411        for i in 0..400 {
9412            agent
9413                .history
9414                .push(ChatMessage::user(format!("turn {i}: {}", "x".repeat(8000))));
9415        }
9416    }
9417
9418    /// The defect: `cx-parity` armed NEITHER compaction trigger, so
9419    /// `maybe_compact` returned `false` on its
9420    /// `threshold.is_none() && compaction_reserve_tokens.is_none()` guard
9421    /// no matter how large the conversation grew.
9422    #[test]
9423    fn cx_parity_fires_maybe_compact_at_its_pressure_trigger() {
9424        let config = resolved_cx_parity();
9425        assert!(config.compaction_enabled);
9426        assert_eq!(config.compaction_reserve_tokens, Some(16384));
9427        assert!(config.compaction_summarize);
9428
9429        let mut agent = Agent::with_provider(config, Box::new(NeverCalledProvider));
9430        stuff_history(&mut agent);
9431        let before = agent.history.len();
9432        assert!(
9433            agent.maybe_compact(),
9434            "cx-parity must compact under context pressure"
9435        );
9436        assert!(agent.history.len() < before, "history must actually shrink");
9437        let marker = agent
9438            .history
9439            .iter()
9440            .find(|m| {
9441                m.content
9442                    .as_deref()
9443                    .is_some_and(|c| c.contains("earlier conversation compacted"))
9444            })
9445            .expect("a compaction marker must be present");
9446        assert!(
9447            marker
9448                .content
9449                .as_deref()
9450                .unwrap()
9451                .contains("summarized to save context"),
9452            "cx-parity sets `core.compaction.summarize = true`"
9453        );
9454    }
9455
9456    /// `core.compaction.summarize = false` reaches `Config` too, and the
9457    /// marker then states only what actually happened to the span.
9458    #[test]
9459    fn compaction_summarize_false_reaches_config_and_changes_the_marker() {
9460        let toml = "extends = \"cx-parity\"\n[core.compaction]\nsummarize = false\n";
9461        let config = resolve(toml, None, &ResolveOptions::default())
9462            .expect("resolves")
9463            .config;
9464        assert!(!config.compaction_summarize);
9465
9466        let mut agent = Agent::with_provider(config, Box::new(NeverCalledProvider));
9467        stuff_history(&mut agent);
9468        assert!(agent.maybe_compact());
9469        let marker = agent
9470            .history
9471            .iter()
9472            .find(|m| {
9473                m.content
9474                    .as_deref()
9475                    .is_some_and(|c| c.contains("earlier conversation compacted"))
9476            })
9477            .expect("a compaction marker must be present");
9478        let text = marker.content.as_deref().unwrap();
9479        assert!(text.contains("cleared to save context"), "got: {text}");
9480        assert!(!text.contains("summarized"));
9481    }
9482}
9483
9484#[cfg(test)]
9485mod api_key_cmd_tests {
9486    //! P4 (design §5.2, §1.8 D6 row): `api_key_cmd` credential-helper
9487    //! resolution. `Agent::new` never makes a network call, so these tests
9488    //! exercise the real resolution chain end-to-end without mocking.
9489
9490    use super::*;
9491
9492    /// Default-off: with no `api_key`/`api_key_cmd` set and an env var that
9493    /// isn't set either, resolution fails exactly as it always has —
9494    /// `api_key_cmd` being a brand-new field changes nothing when unset.
9495    #[test]
9496    fn default_none_falls_through_to_missing_api_key_error() {
9497        let config = Config::builder()
9498            .api_key_env("SUPERCODE_TEST_UNSET_VAR_API_KEY_CMD")
9499            .build();
9500        assert!(config.api_key.is_none());
9501        assert!(config.api_key_cmd.is_none());
9502        let err = Agent::new(config).err().expect("no key source configured");
9503        assert!(matches!(err, Error::MissingApiKey(_)));
9504    }
9505
9506    /// Happy path: `api_key_cmd` alone (no `api_key`, no matching env var)
9507    /// is enough for `Agent::new` to succeed — the helper's stdout is
9508    /// resolved and used.
9509    #[test]
9510    fn api_key_cmd_alone_resolves_successfully() {
9511        let config = Config::builder()
9512            .api_key_cmd("echo sk-test-from-helper")
9513            .api_key_env("SUPERCODE_TEST_UNSET_VAR_API_KEY_CMD_2")
9514            .build();
9515        assert!(Agent::new(config).is_ok());
9516    }
9517
9518    /// A failing helper command (non-zero exit, or empty stdout) falls
9519    /// through to `api_key_env` rather than propagating the helper's own
9520    /// failure — same "try the next source" posture as every other layer.
9521    #[test]
9522    fn api_key_cmd_failure_falls_through_to_env() {
9523        std::env::set_var(
9524            "SUPERCODE_TEST_API_KEY_CMD_FALLBACK",
9525            "sk-from-env-fallback",
9526        );
9527        let config = Config::builder()
9528            .api_key_cmd("exit 1")
9529            .api_key_env("SUPERCODE_TEST_API_KEY_CMD_FALLBACK")
9530            .build();
9531        assert!(Agent::new(config).is_ok());
9532        std::env::remove_var("SUPERCODE_TEST_API_KEY_CMD_FALLBACK");
9533    }
9534
9535    /// A failing helper AND no fallback env var still produces the same
9536    /// `MissingApiKey` error today's no-key path always produced — the new
9537    /// source never turns a hard failure into a silent empty key.
9538    #[test]
9539    fn api_key_cmd_failure_with_no_fallback_still_errors() {
9540        let config = Config::builder()
9541            .api_key_cmd("exit 1")
9542            .api_key_env("SUPERCODE_TEST_UNSET_VAR_API_KEY_CMD_3")
9543            .build();
9544        let err = Agent::new(config)
9545            .err()
9546            .expect("helper failed, no env fallback");
9547        assert!(matches!(err, Error::MissingApiKey(_)));
9548    }
9549
9550    /// `run_api_key_cmd` directly: happy path trims trailing whitespace/
9551    /// newline from the command's stdout.
9552    #[test]
9553    fn run_api_key_cmd_trims_output() {
9554        assert_eq!(run_api_key_cmd("echo '  sk-abc123  '"), "sk-abc123");
9555    }
9556
9557    /// `run_api_key_cmd` directly: a nonexistent binary fails to spawn and
9558    /// returns an empty string rather than panicking.
9559    #[test]
9560    fn run_api_key_cmd_spawn_failure_returns_empty() {
9561        // `sh -c` itself always spawns; feed it a command that can't run.
9562        assert_eq!(
9563            run_api_key_cmd("/no/such/binary/at/all --flag"),
9564            String::new()
9565        );
9566    }
9567}
9568
9569#[cfg(test)]
9570mod bp4_prompt_context_tests {
9571    //! BP-4 (`.volter/tracker/markdown/BP-4.md`): the prompt/context knobs
9572    //! the parity presets never set, proved over the RESOLVED `cc-parity` /
9573    //! `cx-parity` configs (not over hand-built `Config`s — a preset that
9574    //! doesn't arm the knob would pass that weaker test).
9575
9576    use super::*;
9577    use crate::configfile::{resolve, ResolveOptions};
9578
9579    fn resolved(preset: &str) -> Config {
9580        let toml = crate::presets::lookup(preset).unwrap();
9581        resolve(toml, None, &ResolveOptions { strict: true })
9582            .unwrap_or_else(|e| panic!("{preset} resolves: {e}"))
9583            .config
9584    }
9585
9586    /// Never called — these tests assemble prompts and drive
9587    /// `refresh_env_context`/`inject_context_block`, none of which issue a
9588    /// request.
9589    #[derive(Debug)]
9590    struct NoProvider;
9591
9592    #[async_trait::async_trait]
9593    impl Provider for NoProvider {
9594        async fn complete(
9595            &self,
9596            _req: &ChatRequest,
9597            _on_delta: &(dyn for<'a> Fn(&'a str) + Send + Sync),
9598        ) -> crate::Result<(ChatMessage, crate::provider::Usage)> {
9599            unreachable!("BP-4 prompt/context tests never issue a request")
9600        }
9601    }
9602
9603    fn scratch(tag: &str) -> std::path::PathBuf {
9604        let dir = std::env::temp_dir().join(format!(
9605            "supercode-bp4-{tag}-{}-{}",
9606            std::process::id(),
9607            std::time::SystemTime::now()
9608                .duration_since(std::time::UNIX_EPOCH)
9609                .unwrap()
9610                .as_nanos()
9611        ));
9612        std::fs::create_dir_all(&dir).unwrap();
9613        dir
9614    }
9615
9616    /// Row `project-instruction-files-w-directory-walk`: a CLAUDE.md above
9617    /// the working directory is discovered, ordered root→cwd (nearest wins
9618    /// by appearing last), and the climb STOPS at the `.git` root — the
9619    /// directory above it is never read.
9620    #[test]
9621    fn presets_walk_ancestors_up_to_the_git_root_nearest_last() {
9622        for preset in ["cc-parity", "cx-parity"] {
9623            let base = scratch("walk");
9624            let above = base.join("above");
9625            let root = above.join("repo");
9626            let deep = root.join("crates").join("thing");
9627            std::fs::create_dir_all(&deep).unwrap();
9628            std::fs::create_dir_all(root.join(".git")).unwrap();
9629            std::fs::write(above.join("CLAUDE.md"), "ABOVE-THE-ROOT-MARKER").unwrap();
9630            std::fs::write(above.join("AGENTS.md"), "ABOVE-THE-ROOT-MARKER").unwrap();
9631            std::fs::write(root.join("CLAUDE.md"), "REPO-ROOT-MARKER").unwrap();
9632            std::fs::write(root.join("AGENTS.md"), "REPO-ROOT-MARKER").unwrap();
9633            std::fs::write(deep.join("CLAUDE.md"), "NEAREST-DIR-MARKER").unwrap();
9634            std::fs::write(deep.join("AGENTS.md"), "NEAREST-DIR-MARKER").unwrap();
9635
9636            let mut config = resolved(preset);
9637            config.cwd = deep.clone();
9638            let blob = assemble_project_instructions(&config);
9639
9640            let root_at = blob
9641                .find("REPO-ROOT-MARKER")
9642                .unwrap_or_else(|| panic!("{preset}: the ancestor repo root was not walked"));
9643            let near_at = blob
9644                .find("NEAREST-DIR-MARKER")
9645                .unwrap_or_else(|| panic!("{preset}: cwd's own file was not loaded"));
9646            assert!(
9647                root_at < near_at,
9648                "{preset}: nearest-to-cwd must win by appearing LAST (root→cwd)"
9649            );
9650            assert!(
9651                !blob.contains("ABOVE-THE-ROOT-MARKER"),
9652                "{preset}: the walk must stop at the `.git` root"
9653            );
9654            let _ = std::fs::remove_dir_all(&base);
9655        }
9656    }
9657
9658    /// The walk is bounded even with no root marker anywhere: it terminates
9659    /// at the filesystem root instead of looping.
9660    #[test]
9661    fn walk_terminates_without_a_root_marker() {
9662        let base = scratch("nomarker");
9663        let deep = base.join("a").join("b").join("c");
9664        std::fs::create_dir_all(&deep).unwrap();
9665        let mut config = resolved("cx-parity");
9666        config.cwd = deep.clone();
9667        let roots = instruction_walk_roots(&config);
9668        assert!(roots.len() <= MAX_INSTRUCTION_WALK_DEPTH);
9669        assert_eq!(roots.last().unwrap(), &deep, "cwd is the LAST root");
9670        let _ = std::fs::remove_dir_all(&base);
9671    }
9672
9673    /// Row `instruction-file-hygiene-controls`, cx half: `cx-parity` sets
9674    /// the documented 32 KiB `project_doc_max_bytes`, and it binds per file
9675    /// AND over the aggregate, each with its own notice.
9676    #[test]
9677    fn cx_parity_enforces_the_documented_instruction_byte_cap() {
9678        let config = resolved("cx-parity");
9679        assert_eq!(
9680            config.project_doc_max_bytes,
9681            Some(32_768),
9682            "cx-parity must arm cx§2's documented 32 KiB cap"
9683        );
9684
9685        let base = scratch("cap");
9686        std::fs::create_dir_all(base.join(".git")).unwrap();
9687        std::fs::write(base.join("CLAUDE.md"), "x".repeat(40_000)).unwrap();
9688        std::fs::write(base.join("AGENTS.md"), "y".repeat(40_000)).unwrap();
9689        let mut config = config;
9690        config.cwd = base.clone();
9691        let blob = assemble_project_instructions(&config);
9692        assert!(
9693            blob.contains("[supercode: file truncated at core.project_doc_max_bytes]"),
9694            "the per-file cap must fire with a notice"
9695        );
9696        assert!(
9697            blob.contains(
9698                "[supercode: instruction content truncated at core.project_doc_max_bytes]"
9699            ),
9700            "the aggregate cap must fire with a notice"
9701        );
9702        assert!(blob.len() < 33_200, "aggregate blob stayed over the cap");
9703        let _ = std::fs::remove_dir_all(&base);
9704    }
9705
9706    /// Row `instruction-file-hygiene-controls`, cc half: `cc-parity` arms
9707    /// CC's own two levers — HTML-comment stripping and `claudeMdExcludes`
9708    /// — and arms NO byte cap, because CC documents none.
9709    #[test]
9710    fn cc_parity_strips_html_comments_and_honours_excludes() {
9711        let mut config = resolved("cc-parity");
9712        assert!(config.project_doc_strip_comments, "cc strips `<!-- … -->`");
9713        assert_eq!(
9714            config.project_doc_max_bytes, None,
9715            "`project_doc_max_bytes = 0` is §3.1's spelling for uncapped"
9716        );
9717
9718        let base = scratch("hygiene");
9719        std::fs::create_dir_all(base.join(".git")).unwrap();
9720        std::fs::write(
9721            base.join("CLAUDE.md"),
9722            "KEEP-THIS<!-- MAINTAINER-NOTE -->AND-THIS",
9723        )
9724        .unwrap();
9725        std::fs::write(base.join("AGENTS.md"), "EXCLUDED-FILE-MARKER").unwrap();
9726        config.cwd = base.clone();
9727        config.project_doc_excludes = vec!["AGENTS.md".to_string()];
9728        let blob = assemble_project_instructions(&config);
9729        assert!(blob.contains("KEEP-THIS") && blob.contains("AND-THIS"));
9730        assert!(
9731            !blob.contains("MAINTAINER-NOTE"),
9732            "block HTML comments must be stripped before injection"
9733        );
9734        assert!(
9735            !blob.contains("EXCLUDED-FILE-MARKER"),
9736            "an excluded instruction file must never be read into the prompt"
9737        );
9738        let _ = std::fs::remove_dir_all(&base);
9739    }
9740
9741    /// A project layer must not be able to suppress the user's own global
9742    /// instruction files by adding an exclude pattern (§3.3 trust boundary).
9743    #[test]
9744    fn project_layer_cannot_set_instruction_excludes() {
9745        let hc = crate::configfile::HarnessConfig::from_toml_str(
9746            "schema_version = 1\n[core]\nproject_doc_excludes = [\"CLAUDE.md\"]\n",
9747        )
9748        .unwrap();
9749        let (sanitized, dropped) = crate::configfile::sanitize_for_project(&hc);
9750        assert!(sanitized.core.project_doc_excludes.is_none());
9751        assert!(dropped.iter().any(|d| d == "core.project_doc_excludes"));
9752    }
9753
9754    /// Row `environment-context-block`: both presets emit the policy line
9755    /// the row's semantics name, and the block is RE-EMITTED when the thing
9756    /// it describes moves (cx§2 "re-emitted on change").
9757    #[test]
9758    fn env_context_block_carries_policy_and_re_emits_on_change() {
9759        for preset in ["cc-parity", "cx-parity"] {
9760            let base = scratch("env");
9761            // A SIBLING, not a child: `contains` assertions below must not
9762            // be satisfiable by a path prefix.
9763            let here = base.join("here");
9764            let other = base.join("elsewhere");
9765            std::fs::create_dir_all(&here).unwrap();
9766            std::fs::create_dir_all(&other).unwrap();
9767
9768            let mut config = resolved(preset);
9769            config.cwd = here.clone();
9770            assert!(config.env_context, "{preset} must set core.env_context");
9771            let expected_policy = format!(
9772                "approval policy: {} · sandbox: {}",
9773                approval_policy_label(config.approval),
9774                sandbox_policy_label(config.sandbox),
9775            );
9776
9777            let mut agent = Agent::with_provider(config, Box::new(NoProvider));
9778            let system = agent.history[0].content.clone().unwrap_or_default();
9779            assert!(
9780                system.contains(&expected_policy),
9781                "{preset}: the environment block must state the approval/sandbox policy — {system}"
9782            );
9783            assert!(system.contains(&format!("cwd: {}", here.display())));
9784
9785            // Nothing moved ⇒ no churn (the prompt cache is not busted for
9786            // free).
9787            assert!(!agent.refresh_env_context(), "{preset}: spurious re-emit");
9788
9789            // cwd + policy move mid-session.
9790            agent.config.cwd = other.clone();
9791            agent.config.approval = crate::config::ApprovalPolicy::Untrusted;
9792            assert!(
9793                agent.refresh_env_context(),
9794                "{preset}: change not re-emitted"
9795            );
9796            let system = agent.history[0].content.clone().unwrap_or_default();
9797            assert!(
9798                system.contains(&format!("cwd: {}", other.display())),
9799                "{preset}: the fresh cwd must reach the model"
9800            );
9801            assert!(system.contains("approval policy: untrusted"));
9802            assert!(
9803                !system.contains(&format!("cwd: {}", here.display())),
9804                "{preset}: the stale block must be REPLACED, not duplicated"
9805            );
9806            assert_eq!(
9807                system.matches("# Environment").count(),
9808                1,
9809                "{preset}: exactly one environment block"
9810            );
9811            let _ = std::fs::remove_dir_all(&base);
9812        }
9813    }
9814
9815    /// Row `synthetic-context-injection-blocks`: both presets arm the
9816    /// registry, the built-in blocks reach the assembled system prompt, and
9817    /// a block spliced mid-session reaches it too.
9818    #[test]
9819    fn presets_splice_builtin_and_runtime_context_blocks() {
9820        for preset in ["cc-parity", "cx-parity"] {
9821            let config = resolved(preset);
9822            assert!(
9823                config.context_injections,
9824                "{preset} must set core.context_injections"
9825            );
9826            let mut agent = Agent::with_provider(config, Box::new(NoProvider));
9827            let system = agent.history[0].content.clone().unwrap_or_default();
9828            assert!(
9829                system.contains("# Task list"),
9830                "{preset}: a built-in ambient block must reach the prompt"
9831            );
9832
9833            assert!(agent.inject_context_block("Mid session", "SPLICED-BODY-MARKER"));
9834            let system = agent.history[0].content.clone().unwrap_or_default();
9835            assert!(
9836                system.contains("# Mid session") && system.contains("SPLICED-BODY-MARKER"),
9837                "{preset}: a runtime splice must reach the prompt"
9838            );
9839            assert_eq!(agent.spliced_context_blocks().len(), 1);
9840        }
9841    }
9842
9843    /// A deterministic stand-in for the CLI's real provider-backed
9844    /// summarizer: it records exactly what it was asked to summarize, so the
9845    /// test can prove the `/compact <focus>` text reached the summarizer's
9846    /// INPUT and not only the marker.
9847    #[derive(Debug, Default)]
9848    struct RecordingSummarizer {
9849        seen: std::sync::Mutex<Vec<String>>,
9850    }
9851
9852    impl reduce::summarize::SpanSummarizer for RecordingSummarizer {
9853        fn summarize(&self, span_text: &str) -> reduce::Result<String> {
9854            self.seen
9855                .lock()
9856                .unwrap_or_else(std::sync::PoisonError::into_inner)
9857                .push(span_text.to_string());
9858            Ok("MODEL-WRITTEN-SUMMARY".to_string())
9859        }
9860
9861        fn model_id(&self) -> &str {
9862            "test-summarizer"
9863        }
9864    }
9865
9866    fn stuffed_agent(preset: &str) -> Agent {
9867        let config = resolved(preset);
9868        let mut agent = Agent::with_provider(config, Box::new(NoProvider));
9869        for i in 0..40 {
9870            agent.history.push(ChatMessage::user(format!("turn {i}")));
9871            agent
9872                .history
9873                .push(ChatMessage::assistant(format!("reply {i}")));
9874        }
9875        agent
9876    }
9877
9878    /// Row `manual-compact-with-focus-instructions`: `/compact <focus>`
9879    /// compacts on demand (no trigger needed) and the focus text lands in
9880    /// BOTH the summarizer's input and the marker.
9881    #[test]
9882    fn presets_manual_compact_carries_focus_into_the_summarizer_and_the_marker() {
9883        for preset in ["cc-parity", "cx-parity"] {
9884            let config = resolved(preset);
9885            assert!(
9886                config.compaction_focus_instructions.is_some(),
9887                "{preset} must state core.compaction.focus_instructions"
9888            );
9889            let mut agent = stuffed_agent(preset);
9890            let summarizer = std::sync::Arc::new(RecordingSummarizer::default());
9891            agent.set_span_summarizer_arc(summarizer.clone());
9892
9893            let before = agent.history().len();
9894            assert!(
9895                agent.compact_now(Some("keep the migration steps")),
9896                "{preset}: /compact must compact on demand"
9897            );
9898            assert!(agent.history().len() < before, "{preset}: nothing dropped");
9899
9900            let marker = agent
9901                .history()
9902                .iter()
9903                .find_map(|m| m.content.as_deref())
9904                .filter(|c| c.contains("earlier conversation compacted"))
9905                .or_else(|| {
9906                    agent
9907                        .history()
9908                        .iter()
9909                        .filter_map(|m| m.content.as_deref())
9910                        .find(|c| c.contains("earlier conversation compacted"))
9911                })
9912                .unwrap_or_else(|| panic!("{preset}: no compaction marker"))
9913                .to_string();
9914            assert!(
9915                marker.contains("Focus: keep the migration steps"),
9916                "{preset}: {marker}"
9917            );
9918
9919            let seen = summarizer
9920                .seen
9921                .lock()
9922                .unwrap_or_else(std::sync::PoisonError::into_inner);
9923            assert_eq!(seen.len(), 1, "{preset}: exactly one side-call");
9924            assert!(
9925                seen[0].contains("keep the migration steps"),
9926                "{preset}: the focus must reach the summarizer INPUT — {}",
9927                &seen[0][..seen[0].len().min(200)]
9928            );
9929        }
9930    }
9931
9932    /// Row `llm-summaries-of-cleared-spans`: the model-written summary is
9933    /// produced under the presets WITHOUT `capabilities.reduction` — design
9934    /// §1.5 puts "an LLM summary of the compacted span" in core obligation
9935    /// 5, knob `[core.compaction] summarize`.
9936    #[test]
9937    fn presets_summarize_the_cleared_span_without_the_reduction_module() {
9938        for preset in ["cc-parity", "cx-parity"] {
9939            let toml = crate::presets::lookup(preset).unwrap();
9940            let r = resolve(toml, None, &ResolveOptions { strict: true }).unwrap();
9941            assert_eq!(
9942                r.modules.get("reduction"),
9943                Some(&false),
9944                "{preset}: this row must hold with the reduction module OFF"
9945            );
9946            assert!(r.config.compaction_summarize);
9947
9948            let mut agent = stuffed_agent(preset);
9949            agent.set_span_summarizer_arc(std::sync::Arc::new(RecordingSummarizer::default()));
9950            assert!(agent.compact_now(None));
9951            let marker = agent
9952                .history()
9953                .iter()
9954                .filter_map(|m| m.content.as_deref())
9955                .find(|c| c.contains("earlier conversation compacted"))
9956                .unwrap_or_else(|| panic!("{preset}: no compaction marker"));
9957            assert!(
9958                marker.contains("MODEL-WRITTEN-SUMMARY"),
9959                "{preset}: the marker must carry the model-written summary — {marker}"
9960            );
9961        }
9962    }
9963
9964    /// BP-11 (catalog "Lifecycle hooks, config-registered"): the compaction
9965    /// boundary is observable — `pre_compact` fires once the compaction is
9966    /// decided (with the manual/auto trigger named) and `post_compact` once
9967    /// the window has been rewritten, under both parity presets.
9968    #[test]
9969    fn compaction_fires_pre_and_post_lifecycle_events_under_both_presets() {
9970        use crate::config::LifecycleEvent;
9971        for preset in ["cc-parity", "cx-parity"] {
9972            let seen = std::sync::Arc::new(std::sync::Mutex::new(Vec::new()));
9973            let mut agent = stuffed_agent(preset);
9974            let sink = seen.clone();
9975            agent.set_lifecycle_hook(Box::new(move |event| {
9976                sink.lock().unwrap().push(event.clone());
9977            }));
9978            let before = agent.history().len();
9979            assert!(agent.compact_now(Some("keep the plan")), "{preset}");
9980            let after = agent.history().len();
9981            let seen = seen.lock().unwrap();
9982            assert_eq!(seen.len(), 2, "{preset}: exactly pre + post — {seen:?}");
9983            match &seen[0] {
9984                LifecycleEvent::PreCompact {
9985                    messages,
9986                    dropped,
9987                    manual,
9988                } => {
9989                    assert_eq!(*messages, before, "{preset}");
9990                    assert!(*dropped > 0, "{preset}");
9991                    assert!(*manual, "{preset}: /compact is the manual trigger");
9992                }
9993                other => panic!("{preset}: first event must be PreCompact, got {other:?}"),
9994            }
9995            match &seen[1] {
9996                LifecycleEvent::PostCompact { messages, dropped } => {
9997                    assert_eq!(*messages, after, "{preset}");
9998                    assert_eq!(
9999                        *dropped,
10000                        before - after + 1,
10001                        "{preset}: dropped span + 1 marker"
10002                    );
10003                }
10004                other => panic!("{preset}: second event must be PostCompact, got {other:?}"),
10005            }
10006        }
10007    }
10008
10009    /// The automatic trigger reports itself as such, and a window too small
10010    /// to compact fires nothing at all (no pre without a post).
10011    #[test]
10012    fn automatic_compaction_reports_the_auto_trigger_and_a_no_op_fires_nothing() {
10013        use crate::config::LifecycleEvent;
10014        let seen = std::sync::Arc::new(std::sync::Mutex::new(Vec::new()));
10015        let mut agent = stuffed_agent("cc-parity");
10016        let sink = seen.clone();
10017        agent.set_lifecycle_hook(Box::new(move |event| {
10018            sink.lock().unwrap().push(event.clone());
10019        }));
10020        agent.config.compact_after_messages = Some(10);
10021        assert!(agent.maybe_compact());
10022        assert!(matches!(
10023            seen.lock().unwrap()[0],
10024            LifecycleEvent::PreCompact { manual: false, .. }
10025        ));
10026        seen.lock().unwrap().clear();
10027        let mut small = Agent::with_provider(resolved("cc-parity"), Box::new(NoProvider));
10028        let sink = seen.clone();
10029        small.set_lifecycle_hook(Box::new(move |event| {
10030            sink.lock().unwrap().push(event.clone());
10031        }));
10032        assert!(!small.compact_now(None));
10033        assert!(seen.lock().unwrap().is_empty());
10034    }
10035
10036    /// With no summarizer installed the marker degrades to the count-only
10037    /// form — the side-call never blocks or fails compaction.
10038    #[test]
10039    fn compaction_without_a_summarizer_keeps_the_count_only_marker() {
10040        let mut agent = stuffed_agent("cc-parity");
10041        assert!(agent.compact_now(None));
10042        let marker = agent
10043            .history()
10044            .iter()
10045            .filter_map(|m| m.content.as_deref())
10046            .find(|c| c.contains("earlier conversation compacted"))
10047            .unwrap();
10048        assert!(!marker.contains("Summary of the compacted span"));
10049    }
10050
10051    /// Row `compaction-markers-persisted-in-transcript`: under the presets
10052    /// (reduction OFF) the boundary marker is written to the session
10053    /// sidecar, and says where the originals went.
10054    #[test]
10055    fn presets_persist_the_compaction_marker_to_the_transcript() {
10056        for preset in ["cc-parity", "cx-parity"] {
10057            let dir = scratch("marker");
10058            let path = dir.join("session.jsonl");
10059            let empty = supercode_interchange::session::Session::from_claude_code_str("").unwrap();
10060            let writer =
10061                supercode_interchange::sidecar::SidecarWriter::create(&path, &empty).unwrap();
10062            let mut agent = stuffed_agent(preset);
10063            agent.set_recorder(writer);
10064            assert!(agent.reduction_policy().is_none(), "{preset}");
10065
10066            assert!(agent.compact_now(None));
10067            let on_disk = std::fs::read_to_string(&path).unwrap();
10068            assert!(
10069                on_disk.contains("earlier conversation compacted"),
10070                "{preset}: the marker must reach the transcript on disk"
10071            );
10072            assert!(
10073                on_disk.contains("remain in this session's transcript sidecar"),
10074                "{preset}: the marker must say where the originals went"
10075            );
10076            let _ = std::fs::remove_dir_all(&dir);
10077        }
10078    }
10079
10080    /// Row `handoff-fresh-objective-curated-keep-set`: an in-session
10081    /// `new_context` — fresh objective, curated recent keep-set, persisted
10082    /// marker — under cx-parity, where `capabilities.reduction` is off.
10083    #[test]
10084    fn cx_parity_handoff_seeds_a_fresh_objective_with_a_curated_keep_set() {
10085        let mut agent = stuffed_agent("cx-parity");
10086        assert!(!agent.config().handoff_enabled, "reduction handoff is off");
10087        agent.history.push(ChatMessage::user("LAST-USER-TURN"));
10088        let before = agent.history().len();
10089
10090        let dropped = agent.new_context("ship the migration", Some(3));
10091        assert!(dropped > 0, "messages must be set aside");
10092        assert!(agent.history().len() < before);
10093        let system_prompt = agent.history()[0].content.clone().unwrap_or_default();
10094        assert!(
10095            system_prompt.contains("supercode") || !system_prompt.is_empty(),
10096            "the system prompt survives a handoff"
10097        );
10098        let marker = agent
10099            .history()
10100            .iter()
10101            .filter_map(|m| m.content.as_deref())
10102            .find(|c| c.contains("[handoff:"))
10103            .expect("handoff marker");
10104        assert!(marker.contains("Objective: ship the migration"));
10105        assert!(
10106            agent
10107                .history()
10108                .iter()
10109                .any(|m| m.content.as_deref() == Some("LAST-USER-TURN")),
10110            "the curated keep-set must carry the most recent turns"
10111        );
10112    }
10113
10114    /// Row `context-usage-introspection`: a live breakdown, from the same
10115    /// estimator the context guard enforces, without sending anything.
10116    #[test]
10117    fn presets_report_live_context_usage() {
10118        for preset in ["cc-parity", "cx-parity"] {
10119            let agent = stuffed_agent(preset);
10120            let usage = agent.context_usage();
10121            assert_eq!(usage.messages, agent.history().len(), "{preset}");
10122            assert!(usage.message_tokens > 0, "{preset}");
10123            assert_eq!(
10124                usage.request_tokens,
10125                usage.message_tokens + usage.tool_schema_tokens,
10126                "{preset}: the breakdown must add up"
10127            );
10128            assert!(usage.projected_tokens >= usage.request_tokens, "{preset}");
10129            assert!(usage.context_limit.is_some(), "{preset}: window known");
10130            assert!(usage.fits, "{preset}");
10131            let line = usage.summary_line();
10132            assert!(line.contains('%') && line.contains(&usage.model), "{line}");
10133
10134            // Pure: asking must not change what the next request carries.
10135            let again = agent.context_usage();
10136            assert_eq!(usage, again, "{preset}");
10137        }
10138    }
10139}
10140
10141#[cfg(test)]
10142#[path = "agent_bp10_tests.rs"]
10143mod bp10_permissions_tests;