supercode_harness/agent.rs
1//! The agent loop.
2
3use std::collections::HashSet;
4
5use crate::config::{CachePlan, Config, SteeringMode, ToolAdvertising};
6use crate::error::{Error, Result};
7use crate::provider::{self, ChatRequest, OpenAiProvider, Provider, ToolSchema};
8use crate::reduce::rehydrate::CAP_NOTICE_MARKER;
9use crate::reduce::{self, ReductionLog, ReductionPolicy};
10use crate::tools::{ToolContext, ToolRegistry};
11use supercode_interchange::session::Session;
12use supercode_interchange::sidecar::SidecarWriter;
13use supercode_interchange::{ChatMessage, Role};
14use supercode_runtime::AgentEvent;
15
16/// BP-6: how many skill bodies one user message may pull in through `$slug`
17/// mentions — Claude Code caps skill chaining at six per message (cc§7
18/// "Skill chaining"); the same ceiling bounds the mention path here.
19const MAX_SKILL_LOADS_PER_MESSAGE: usize = 6;
20
21/// BP-5: how many `@path` mentions one message may attach. A prompt is not
22/// a bulk loader; past this the user means `--file`.
23const MAX_FILE_MENTIONS_PER_MESSAGE: usize = 10;
24
25/// BP-5: ceiling on the bytes one `@path` mention contributes.
26const MAX_FILE_MENTION_BYTES: usize = 64 * 1024;
27
28/// Tool name of the `tool_search` agent intrinsic (B6). Never a registered
29/// [`crate::tools::Tool`] — intercepted in [`Agent::run_tool`] before registry
30/// lookup, so it works under any [`ToolAdvertising`] mode.
31const TOOL_SEARCH: &str = "tool_search";
32
33/// Tool name of the `expand_reduction` agent intrinsic (T12/TR-1) — the
34/// model-invocable rehydration counterpart to `tool_search`, same
35/// interception pattern. Advertised whenever a [`ReductionPolicy`] is
36/// installed, regardless of [`ToolAdvertising`] mode (see [`Self::tool_schemas`]).
37const EXPAND_REDUCTION: &str = "expand_reduction";
38
39/// Tool name of the `sidecar_search` agent intrinsic (T12/TR-1).
40const SIDECAR_SEARCH: &str = "sidecar_search";
41
42/// Tool name of the `spawn_subagent` agent intrinsic (P5-3, §2 module 9 D1
43/// "spawn tool"). Same interception pattern as [`TOOL_SEARCH`] — never a
44/// registered [`crate::tools::Tool`], intercepted in [`Agent::run_tool`]
45/// before registry lookup — but ALSO needs full `&mut self` async access
46/// (running a whole child agent loop, or `tokio::spawn`-ing one), which
47/// [`Agent::prepare_tool_call`]'s purely-synchronous intrinsics don't, so
48/// the interception point is `Self::run_tool`'s top, not
49/// `prepare_tool_call`.
50const SPAWN_SUBAGENT: &str = "spawn_subagent";
51
52/// BP-7 (catalog §4a "Review mode"): the `[core.prompts]` key the review
53/// turn's template lives under. One name for both presets — cc spells the
54/// command `/code-review`, cx spells it `/review`, and both resolve to this
55/// template, so the row's evidence is one config key, not two.
56pub const REVIEW_PROMPT_NAME: &str = "code-review";
57
58/// BP-7 (catalog §4a "Side/ephemeral Q&A"): the instruction prefixed to a
59/// side question, so the model knows it is answering ABOUT the session
60/// rather than continuing it. The exchange never enters history either way;
61/// this keeps the answer from reading like the next assistant turn.
62const SIDE_QUESTION_PREAMBLE: &str = "[side question — answer from the conversation above; this exchange is not part of the conversation and you have no tools for it]";
63
64/// Claude Code's native name for [`SPAWN_SUBAGENT`]. It is exposed only when
65/// `Config::subagents_claude_agent_alias` is enabled for a Claude import.
66const CLAUDE_AGENT: &str = "Agent";
67
68/// Claude Code spellings for core filesystem/shell tools. Imported Claude
69/// context frequently continues to call these names even when another model
70/// is driving the turn, so emulation must translate execution as well as
71/// preserve the original call/result names in the transcript.
72const CLAUDE_BASH: &str = "Bash";
73const CLAUDE_READ: &str = "Read";
74const CLAUDE_WRITE: &str = "Write";
75const CLAUDE_EDIT: &str = "Edit";
76const CLAUDE_GLOB: &str = "Glob";
77const CLAUDE_GREP: &str = "Grep";
78
79/// Claude Code scheduler compatibility intrinsics. They edit an imported
80/// [`crate::ClaudeRuntimeManifest`]; actual timer execution belongs to an
81/// embedding scheduler driver, never this agent loop.
82const CLAUDE_CRON_CREATE: &str = "CronCreate";
83const CLAUDE_CRON_DELETE: &str = "CronDelete";
84const CLAUDE_CRON_LIST: &str = "CronList";
85const CLAUDE_SCHEDULE_WAKEUP: &str = "ScheduleWakeup";
86
87/// Shared SDK steering mailbox. `accepting` and `queue` share one lock so a
88/// turn's final boundary can close acceptance atomically with its last drain;
89/// a steer can therefore never be acknowledged into the following turn.
90#[derive(Default)]
91pub(crate) struct SteerInbox {
92 queue: std::collections::VecDeque<QueuedSteer>,
93 accepting: bool,
94}
95
96struct QueuedSteer {
97 message: String,
98 sdk_bound: bool,
99}
100
101impl SteerInbox {
102 pub(crate) fn open(&mut self) {
103 self.queue.clear();
104 self.accepting = true;
105 }
106
107 pub(crate) fn enqueue(&mut self, message: String) -> bool {
108 if !self.accepting {
109 return false;
110 }
111 self.queue.push_back(QueuedSteer {
112 message,
113 sdk_bound: true,
114 });
115 true
116 }
117
118 pub(crate) fn close(&mut self) {
119 self.accepting = false;
120 self.queue.retain(|queued| !queued.sdk_bound);
121 }
122
123 fn drain(&mut self, mode: SteeringMode) -> Option<String> {
124 if self.queue.is_empty() {
125 return None;
126 }
127 match mode {
128 SteeringMode::All => Some(
129 self.queue
130 .drain(..)
131 .map(|queued| queued.message)
132 .collect::<Vec<_>>()
133 .join("\n\n"),
134 ),
135 SteeringMode::OneAtATime => self.queue.pop_front().map(|queued| queued.message),
136 }
137 }
138
139 fn drain_or_close(&mut self, mode: SteeringMode) -> Option<String> {
140 if !self.queue.iter().any(|queued| queued.sdk_bound) {
141 self.accepting = false;
142 return None;
143 }
144 self.drain(mode)
145 }
146
147 fn queue_unchecked(&mut self, message: String) {
148 self.queue.push_back(QueuedSteer {
149 message,
150 sdk_bound: false,
151 });
152 }
153
154 fn len(&self) -> usize {
155 self.queue.len()
156 }
157}
158
159struct SteerTurnGuard {
160 inbox: std::sync::Arc<std::sync::Mutex<SteerInbox>>,
161}
162
163impl SteerTurnGuard {
164 fn new(inbox: std::sync::Arc<std::sync::Mutex<SteerInbox>>) -> Self {
165 inbox
166 .lock()
167 .unwrap_or_else(std::sync::PoisonError::into_inner)
168 .accepting = true;
169 Self { inbox }
170 }
171}
172
173impl Drop for SteerTurnGuard {
174 fn drop(&mut self) {
175 self.inbox
176 .lock()
177 .unwrap_or_else(std::sync::PoisonError::into_inner)
178 .close();
179 }
180}
181
182/// Tool name of the `subagent_status` agent intrinsic (P5-3, D3
183/// "background+resume"): poll (and reap, once finished) a background child
184/// spawned via [`SPAWN_SUBAGENT`]. Only advertised when
185/// `Config::subagents_background` is on (see [`Agent::tool_schemas`]).
186const SUBAGENT_STATUS: &str = "subagent_status";
187
188/// The agent's one message verb (cc's `SendMessage`, cx's v2 `send_message`):
189/// `to` is either a STILL-RUNNING background child (delivered into its own
190/// inbox) or another session of any harness, which goes through the one send
191/// every sender uses ([`crate::mail_send`]). Only advertised when
192/// `Config::subagents_background` is on.
193const SEND_MESSAGE: &str = "send_message";
194
195/// BP-7 (catalog §4a "Background subagents + resume": "resumable with
196/// context intact"): continue a FINISHED child with its own transcript
197/// restored, rather than starting a fresh one that has to be re-briefed.
198const SUBAGENT_RESUME: &str = "subagent_resume";
199
200/// Tool name of the `background_exec` agent intrinsic (P5-6, §2 module 4
201/// `tools.background` D1 "background exec"). Unlike [`SPAWN_SUBAGENT`], this
202/// needs no async child-agent loop — spawning a process
203/// (`tokio::process::Command::spawn`) is itself synchronous — so, like
204/// [`TOOL_SEARCH`], it is intercepted in [`Agent::prepare_tool_call`], not
205/// [`Agent::run_tool`].
206const BACKGROUND_EXEC: &str = "background_exec";
207
208/// Tool name of the `background_status` agent intrinsic (P5-6, D1 "monitor/
209/// event feed"): poll a background job's run status, drain its newly
210/// captured output as an [`AgentEvent::BackgroundOutput`] event, and reap it
211/// (remove it from [`Agent::background_jobs`]) once it has exited or been
212/// killed.
213const BACKGROUND_STATUS: &str = "background_status";
214
215/// Tool name of the `background_list` agent intrinsic (P5-6, D10
216/// "bg-manager"): list every background job this agent is currently
217/// tracking (running or finished-but-unreaped), without draining output or
218/// reaping anything.
219const BACKGROUND_LIST: &str = "background_list";
220
221/// Tool name of the `background_kill` agent intrinsic (P5-6, D10
222/// "bg-manager"): kill a background job's real OS process
223/// (`tokio::process::Child::start_kill`) and reap it immediately.
224const BACKGROUND_KILL: &str = "background_kill";
225
226/// P4e (§3.1 `core.parallel_tool_calls`): the synchronous outcome of
227/// [`Agent::prepare_tool_call`] — either a result already in hand (an
228/// intrinsic, or a call refused before it ever reached `Tool::execute`), or
229/// a plain registry-tool call ready for the (possibly concurrent) async
230/// `execute()` step.
231enum PreparedCall {
232 /// A final `(output, is_error)` result — no `Tool::execute` call is
233 /// coming for this one.
234 Done((String, bool)),
235 /// Passed every synchronous check; `execute(args, &ctx)` on the named
236 /// registry tool is the only remaining step.
237 Ready {
238 name: String,
239 args: serde_json::Value,
240 },
241}
242
243/// Marker prefix of the notice [`Agent::cap_tool_output`] appends to an
244/// oversized tool result kept in `history` (the recorder receives the full
245/// UX-26 (B7-warn): current wall-clock time as unix milliseconds, the same
246/// unit [`supercode_interchange::sidecar::rfc3339_to_ms`] parses session timestamps into —
247/// lets [`Agent::build_request_messages`] compare "now" against a
248/// cross-process signal (a loaded session's last message timestamp) on
249/// equal footing with an in-process one (this agent's own last annotated
250/// send). Saturates to 0 on a pre-epoch clock rather than panicking (never
251/// happens on real hardware, but `duration_since` can theoretically error).
252fn now_ms() -> i64 {
253 std::time::SystemTime::now()
254 .duration_since(std::time::UNIX_EPOCH)
255 .map(|d| d.as_millis() as i64)
256 .unwrap_or(0)
257}
258
259/// P5-3: process-wide sequence number backing [`next_subagent_id`] —
260/// disambiguates two spawns landing in the same millisecond (which
261/// `now_ms()` alone cannot).
262static SUBAGENT_ID_SEQ: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(0);
263
264/// P5-3: a fresh, process-unique child agent id (`"agent-<hex-ts>-<hex-seq>"`
265/// — the native analog of Claude Code's `agent-<id>` naming, see
266/// `supercode_interchange::session::SessionMeta::agent_id`'s doc comment).
267fn next_subagent_id() -> String {
268 let seq = SUBAGENT_ID_SEQ.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
269 format!("agent-{:x}-{:x}", now_ms(), seq)
270}
271
272/// P5-4: the shape [`Agent::child_approval_handler_factory`]/
273/// [`Agent::set_child_approval_handler_factory`] share — factored into its
274/// own alias (clippy `type_complexity`) rather than spelled out inline at
275/// both use sites.
276type ChildApprovalHandlerFactory = dyn Fn(
277 String,
278 std::sync::Arc<std::sync::Mutex<Vec<crate::subagents::QueuedApproval>>>,
279 ) -> std::sync::Arc<dyn crate::permissions::PermissionsApprovalHandler>
280 + Send
281 + Sync;
282
283/// BP-4 (catalog:109 "Context-usage introspection"): the live
284/// context-window accounting [`Agent::context_usage`] reports — cc's
285/// `/context` grid and cx's `/status` + `get_context_remaining` in one
286/// shape, over the numbers `resume --dry-run`'s preflight already computes.
287///
288/// Every token figure is the SAME estimate the context guard enforces
289/// (`supercode_runtime`), so what this reports and what refuses an oversized
290/// turn can never disagree.
291#[derive(Debug, Clone, PartialEq, Eq, serde::Serialize, serde::Deserialize)]
292pub struct ContextUsage {
293 /// The model the accounting is against.
294 pub model: String,
295 /// Messages in the projected request view (reduction stubs included).
296 pub messages: usize,
297 /// Estimated tokens for those messages.
298 pub message_tokens: u64,
299 /// Tools advertised on the next request.
300 pub tool_count: usize,
301 /// Estimated tokens for the serialized tool-schema array — a real part
302 /// of the wire request, and the half a message-only count misses.
303 pub tool_schema_tokens: u64,
304 /// `message_tokens + tool_schema_tokens`.
305 pub request_tokens: u64,
306 /// `request_tokens` with the guard's safety margin applied — the figure
307 /// the context guard actually compares.
308 pub projected_tokens: u64,
309 /// Headroom the guard reserves for the model's own reply.
310 pub response_reserve_tokens: u64,
311 /// The model's context window, when known.
312 pub context_limit: Option<u64>,
313 /// Tokens still available after the reply reserve, `0` when unknown.
314 pub remaining_tokens: u64,
315 /// `projected_tokens` as a whole percentage of the window (rounded),
316 /// `0` when the window is unknown. An integer so this whole struct
317 /// stays `Eq`-comparable on the frontend wire.
318 pub used_pct: u32,
319 /// Whether the next request would pass the context guard.
320 pub fits: bool,
321}
322
323impl ContextUsage {
324 /// One human line, the shape a `/context` command prints.
325 pub fn summary_line(&self) -> String {
326 match self.context_limit {
327 Some(limit) => format!(
328 "{} · {}% of {} used ({} projected, {} left) · {} messages {} · {} tool schemas {}",
329 self.model,
330 self.used_pct,
331 supercode_runtime::fmt_approx_tokens(limit),
332 supercode_runtime::fmt_approx_tokens(self.projected_tokens),
333 supercode_runtime::fmt_approx_tokens(self.remaining_tokens),
334 self.messages,
335 supercode_runtime::fmt_approx_tokens(self.message_tokens),
336 self.tool_count,
337 supercode_runtime::fmt_approx_tokens(self.tool_schema_tokens),
338 ),
339 None => format!(
340 "{} · context window unknown · {} projected · {} messages {} · {} tool schemas {}",
341 self.model,
342 supercode_runtime::fmt_approx_tokens(self.projected_tokens),
343 self.messages,
344 supercode_runtime::fmt_approx_tokens(self.message_tokens),
345 self.tool_count,
346 supercode_runtime::fmt_approx_tokens(self.tool_schema_tokens),
347 ),
348 }
349 }
350}
351
352/// A stateful agent: configuration, a model transport, a tool set, and the
353/// running conversation. Drive it with [`Agent::send`].
354/// BP-13 — one hop the run loop's failure-fallback pass performed: the
355/// model it was on, the model it moved to, and the provider failure that
356/// made it move.
357#[derive(Debug, Clone, PartialEq, Eq)]
358pub struct FallbackHop {
359 /// The model that failed.
360 pub from: String,
361 /// The next chain entry, which the request was re-sent against.
362 pub to: String,
363 /// The failure, rendered — the record's `reason`.
364 pub reason: String,
365}
366
367/// BP-13 — whether `error` is the kind of failure ANOTHER MODEL could
368/// plausibly answer, i.e. one the fallback chain exists for.
369///
370/// Deliberately narrow: rate limiting (429) and server-side failures (5xx,
371/// which is where "overloaded" lives) are properties of the model/endpoint
372/// that was asked, so asking a different one is a real remedy. Everything
373/// else — a bad request, a refused key, a decode failure, a tool error —
374/// is the CALLER's problem and would fail identically against every entry
375/// in the chain, so walking it would only multiply the same error by three.
376/// The transport's own retry (`OpenAiProvider::send_with_retry`) has
377/// already run and given up by the time this is consulted.
378pub fn is_failover_worthy(error: &Error) -> bool {
379 matches!(error, Error::Provider { status, .. } if *status == 429 || *status >= 500)
380}
381
382pub struct Agent {
383 config: Config,
384 provider: std::sync::Arc<dyn Provider>,
385 registry: ToolRegistry,
386 history: Vec<ChatMessage>,
387 ctx: ToolContext,
388 /// Cumulative output (completion) tokens across every `send` on this agent.
389 total_output_tokens: u64,
390 /// Names of non-core tools discovered via `tool_search` (B6): advertised
391 /// starting with the *next* request once populated.
392 activated_tools: HashSet<String>,
393 /// The live sidecar writer (A3), if this agent is recording. `None` is
394 /// today's behavior, at zero cost: every append point becomes a no-op.
395 recorder: Option<SidecarWriter>,
396 /// BP-8 (catalog:150 "Append-only durable transcript"): the live
397 /// append-only journal, if one is installed
398 /// ([`Self::set_journal`], armed by the caller when
399 /// [`Config::session_append_only`] is on). Behind an `Arc<Mutex<_>>`
400 /// rather than owned outright because the queue doors
401 /// ([`Self::queue_steer`]) take `&self` — a pending input has to be
402 /// recorded from a shared handle while a turn holds `&mut Agent`.
403 /// `None` (the default) is a no-op at every append point: today's
404 /// behavior, no file created.
405 journal: Option<std::sync::Arc<std::sync::Mutex<crate::session_journal::SessionJournal>>>,
406 /// BP-8 (catalog:151 "In-place conversation tree"): the live
407 /// `SessionTree` for this session, materialized when
408 /// [`Config::session_tree_enabled`] is on. Every recorded message
409 /// becomes a node, and [`Self::rewind_conversation`] moves the active
410 /// branch's leaf — the tree is what makes a rewind lossless (the old
411 /// leaf is preserved under a sibling branch) rather than a truncation.
412 /// `None` (the default, and every preset that leaves the module off) is
413 /// zero cost: nothing is built and nothing is persisted.
414 session_tree: Option<supercode_interchange::session_tree::SessionTree>,
415 /// BP-8 (catalog:152 "Rewind/rollback conversation"): tails removed by
416 /// rewinds that have not been undone, newest last. Restored from the
417 /// journal on resume, so "undo the rewind" survives a restart.
418 rewind_undo: Vec<Vec<ChatMessage>>,
419 /// BP-8 (catalog:156): the plan as last written to the journal —
420 /// compared against `ctx.plan` so an unchanged plan is not re-journaled
421 /// on every loop iteration.
422 journaled_plan: Vec<crate::session_journal::PlanEntry>,
423 /// Reversible reduction policy (A5/A7/A10). `None` is today's behavior,
424 /// at zero cost: every provider request is built from `self.history`
425 /// verbatim, exactly as before this landed.
426 reduction_policy: Option<ReductionPolicy>,
427 /// The accumulating reduction log (A5): fed back into
428 /// [`reduce::project_messages`] on every request-build so already-applied
429 /// reductions reproduce verbatim across turns and `send` calls (prefix
430 /// stability). `history` itself is never touched by this — see
431 /// `Self::run_loop`.
432 reduction_log: ReductionLog,
433 /// B7: length of the stable, byte-identical-across-turns prefix at the
434 /// front of [`Self::history`] — this agent's own system message plus
435 /// every message of a previously-imported session — set by
436 /// [`Self::load_session`]. `None` (the default) means no session has been
437 /// loaded, so [`crate::provider::apply_cache_plan`] has nothing to
438 /// annotate even under [`CachePlan::ImportedPrefix`].
439 imported_prefix_len: Option<usize>,
440 /// BP-11: set for the duration of a `/compact` so the `pre_compact`
441 /// observer can tell a manual compaction from an automatic trigger.
442 compacting_manually: bool,
443 /// BP-4 (catalog:90 "Environment context block", cx§2 "re-emitted on
444 /// change"): the `# Environment` block currently spliced into
445 /// `history[0]`, verbatim — `None` when `core.env_context` is off (or
446 /// on a construction path that assembles no prompt). Kept so
447 /// [`Self::refresh_env_context`] can locate and replace exactly this
448 /// text when cwd, approval/sandbox policy or the git branch moves
449 /// mid-session, instead of leaving the model reading a block that
450 /// stopped being true.
451 env_context_live: Option<String>,
452 /// BP-5 (catalog D2 "Per-model-family base-prompt selection"): the base
453 /// system prompt currently spliced at the head of `history[0]`,
454 /// verbatim. Kept for the same reason [`Self::env_context_live`] is:
455 /// when the model changes ([`Self::set_model`]) the family's prompt
456 /// changes with it, and the stale text has to be located and REPLACED
457 /// rather than left in front of the new model.
458 base_prompt_live: String,
459 /// BP-5 (catalog D2 "Shell-output injection in templates/skills"): the
460 /// permission-engine authorization every `` !`cmd` `` in a skill or
461 /// command body runs under. Inert (executes nothing) unless
462 /// `[core.skills] shell_injection` is on.
463 shell_injection: crate::skills::ShellInjection,
464 /// BP-4 (catalog:91): blocks spliced into the context AFTER
465 /// construction — see [`Self::inject_context_block`] and
466 /// [`crate::context_injection`]. Empty by default, at zero cost.
467 spliced_context_blocks: Vec<crate::config::ContextInjectionBlock>,
468 /// TR-7 (T20): the injectable side-call ([`reduce::summarize::SpanSummarizer`])
469 /// used to summarize an A10 `TurnsCleared` span, if one is installed
470 /// ([`Self::set_span_summarizer`]). `None` is today's behavior, at zero
471 /// cost: `Self::build_request_messages` never calls
472 /// [`reduce::prepare_cleared_turns_summary`] without one, so
473 /// `policy.summarize_cleared_turns` being on with no summarizer
474 /// installed behaves exactly like it being off (deterministic stub only)
475 /// — never a panic, never a blocked request.
476 span_summarizer: Option<std::sync::Arc<dyn reduce::summarize::SpanSummarizer + Send + Sync>>,
477 /// TR-8 (T5): the tool-schema tier signature (global knob + per-tool
478 /// overrides) as of the last request this agent built, or `None` before
479 /// the first request. Compared against the CURRENT signature at the top
480 /// of every `Self::build_request_messages` call so a tier change made
481 /// mid-session (via [`Self::set_schema_tier`] /
482 /// [`Self::set_tool_schema_tier`]) is detected and flagged to the B7
483 /// cache planner as a cache-bust event (`provider::tier_change_is_cache_bust`).
484 last_tool_schema_tier_signature: Option<u64>,
485 /// PARITY-18 D4 — the target model's context-window size, if the caller
486 /// has armed the guard via [`Self::set_context_limit`]. `None` (the
487 /// default) means no guard: every request is sent unconditionally.
488 /// CLI entry points arm it for their resolved model; direct SDK callers
489 /// retain explicit control through [`Self::set_context_limit`].
490 /// Once set, `Self::run_loop` re-checks
491 /// [`supercode_runtime::context_guard`] before EVERY request it builds —
492 /// not just the first — so "never sends an over-context request" holds
493 /// for the whole session, not only a one-shot preflight.
494 context_limit: Option<u64>,
495 /// PARITY-18 D3 — becomes `true` the first time `Self::run_loop`
496 /// actually reaches its real send site (immediately before
497 /// [`Provider::complete`]). Exposed via [`Self::request_issued`] so a
498 /// caller can report "request sent" truthfully — never asserted ahead
499 /// of time, so a pre-delivery failure (guard refusal, a build error) or
500 /// an interactive session that quits before any turn completes is
501 /// reported honestly as "not sent".
502 requests_issued: bool,
503 /// UX-26 (B7-warn): unix-ms wall-clock time this agent last knew the
504 /// active [`CachePlan::ImportedPrefix`] breakpoint to be warm. Seeded by
505 /// [`Self::load_session`] from the just-loaded session's OWN last
506 /// message timestamp (`metadata["timestamp"]`, parsed via
507 /// [`supercode_interchange::sidecar::rfc3339_to_ms`]) — a cross-process signal: how long
508 /// the resumed conversation has sat idle since ANY tool last touched it,
509 /// which is exactly when Anthropic's server-side cache entry (if one
510 /// ever existed) was last capable of being warm. Refreshed to "now"
511 /// every time `Self::run_loop` actually sends a cache-annotated
512 /// request (an in-process signal: idle time between this agent's own
513 /// turns). `None` when no imported prefix exists yet, or the loaded
514 /// session's last message carries no parseable timestamp — never
515 /// guessed, so the TTL check in [`provider::cache_cold_reason`] simply
516 /// doesn't fire rather than risk a false positive.
517 last_cache_activity_ms: Option<i64>,
518 /// UX-26: whether a PRIOR request already carried a cache_control
519 /// annotation for the current [`Self::imported_prefix_len`] — i.e.
520 /// whether reuse is genuinely "expected" on the NEXT annotated request.
521 /// `false` until the first annotated request goes out (that one is
522 /// establishing the cache entry, a legitimate write, never a "miss") and
523 /// reset to `false` by [`Self::load_session`] whenever the imported
524 /// prefix itself changes.
525 cache_established: bool,
526 /// UX-26 scratch: this turn's cache-warmth context, computed once at the
527 /// top of `Self::build_request_messages` (before the request is sent,
528 /// while `effective_cache_plan`/`busted` are in scope) and consumed once
529 /// in `Self::run_loop` right after `usage` comes back — never read
530 /// across turns, so a stale value can't leak. `(will_annotate,
531 /// cache_established, idle_secs)` — see [`provider::cache_cold_reason`]
532 /// for what each of the first two independently gates.
533 pending_cache_turn: (bool, bool, Option<i64>),
534 /// P4b: the injectable auto-title side-call ([`Self::set_session_titler`]),
535 /// mirroring `Self::span_summarizer`'s "installing one alone changes
536 /// nothing" contract — `Config::auto_title` is the actual gate a caller
537 /// consults before invoking [`Self::auto_title`].
538 session_titler: Option<std::sync::Arc<dyn crate::session_title::SessionTitler + Send + Sync>>,
539 /// P4b (§1.6, catalog §4a "persisted per-turn usage records"): every
540 /// [`crate::usage_log::UsageRecord`] recorded so far this agent's
541 /// lifetime. Always accumulated (cheap, small) regardless of whether a
542 /// caller ever persists it — see [`Self::usage_records`]/
543 /// [`Self::save_usage_log`].
544 usage_log: Vec<crate::usage_log::UsageRecord>,
545 /// P4b: 0-based index of the NEXT model round-trip, for
546 /// [`crate::usage_log::UsageRecord::turn`].
547 turn_index: usize,
548 /// BP-7 (catalog §4a "Turn/step bracketing records"): the per-round-trip
549 /// marker log — context/usage/finish brackets plus the retry, abort,
550 /// effort and goal markers. Persisted beside the session as
551 /// `<name>.events.jsonl` (see [`Self::save_turn_records`]).
552 turn_records: Vec<crate::turn_record::TurnRecord>,
553 /// BP-7: retries the transport reported, drained after every
554 /// `complete()` so each notice attaches to the round-trip that produced
555 /// it. Only the HTTP provider built by [`Self::new`] writes into this;
556 /// an injected provider simply never records anything.
557 retry_log: std::sync::Arc<crate::provider::RetryLog>,
558 /// BP-7 (catalog §4a "Per-turn cost/usage accounting", "Turn/budget
559 /// caps"): the price to bill this agent's model at, resolved at
560 /// construction from [`Config::price_input_per_mtok`]/
561 /// [`Config::price_output_per_mtok`] or [`crate::pricing`]'s table, and
562 /// re-resolved by [`Self::set_model`]. `None` = unpriceable, so no cost
563 /// is recorded (never a guess).
564 model_price: Option<crate::pricing::ModelPrice>,
565 /// BP-7: dollars this agent has spent across its whole lifetime — the
566 /// counter [`Config::max_budget_usd`] is measured against.
567 total_cost_usd: f64,
568 /// BP-7: tool calls this agent has executed across its whole lifetime —
569 /// the counter [`Config::max_steps`] is measured against.
570 total_steps: usize,
571 /// BP-7 (catalog §4a "Background subagents + resume"): a finished
572 /// child's post-system-prompt transcript, kept after the reap so
573 /// `subagent_resume` can restore its context in-process. A session
574 /// with a subagent store attached also has it on disk; this makes
575 /// resume work for an embedder that never attached one.
576 reaped_subagents: std::collections::HashMap<String, Vec<ChatMessage>>,
577 /// BP-7 (catalog §4a "Goals"): the session's standing objective, when
578 /// `capabilities.todos.goals` is on and one has been set. Restated at
579 /// the TAIL of every request while it stands (see
580 /// [`crate::goals::GoalRecord::reminder`]) and persisted as
581 /// `<session>.goal.json` — never written into `history`, so the
582 /// transcript stays exactly what the conversation was.
583 goal: Option<crate::goals::GoalRecord>,
584 /// P4b (§1.7/§3.1 `core.steering`, pi§3 semantics): queued mid-turn
585 /// steering messages — drained at the top of `Self::run_loop`'s next
586 /// iteration (pi's "steer = after current tool calls"). Empty by
587 /// default, at zero cost: `Self::run_loop` skips the drain entirely
588 /// when empty.
589 steer_queue: std::sync::Arc<std::sync::Mutex<SteerInbox>>,
590 /// P4b: queued follow-up messages — drained only once the loop is
591 /// otherwise idle (pi's "follow-up = at idle"), i.e. exactly the point
592 /// `Self::run_loop` would otherwise return a final answer.
593 follow_up_queue: std::collections::VecDeque<String>,
594 /// P4c (§5.2 P4 "doom-loop breaker", §3.1 `core.doom_loop_threshold`):
595 /// `(tool name, canonical JSON args)` of the most recent tool call, if
596 /// [`Config::doom_loop_threshold`] is armed — `None` before the first
597 /// call this agent has run. See [`Self::check_doom_loop`].
598 doom_loop_last_call: Option<(String, String)>,
599 /// P4c: how many times [`Self::doom_loop_last_call`] has repeated
600 /// consecutively so far (starts at 1 on the call that SET it).
601 doom_loop_streak: u32,
602 /// P4c (§1.10/§3.1 `core.model_switch.allow_switch`): every
603 /// [`crate::model_change::ModelChangeRecord`] [`Self::switch_model`] has
604 /// created so far this agent's lifetime. Always empty when
605 /// `Config::model_switch_allow_switch` is off (the default) or no
606 /// switch has happened yet.
607 model_change_log: Vec<crate::model_change::ModelChangeRecord>,
608 /// P4e (§1.6/§3.1 `core.session.git_metadata`, catalog:331): captured
609 /// once at construction when [`Config::session_git_metadata`] is on;
610 /// `None` when the gate is off (the default) or the best-effort git
611 /// probe found nothing (not a repo, `git` missing). See
612 /// [`Self::git_metadata`]/[`Self::save_git_metadata`].
613 git_metadata: Option<crate::git_metadata::GitMetadataRecord>,
614 /// P5-1 (§2.10, session-scoped "approve for session" cache): populated
615 /// only when a [`crate::permissions::PermissionsApprovalHandler`]
616 /// returns [`crate::permissions::ApprovalOutcome::AllowForSession`] —
617 /// see [`Self::prepare_tool_call`]'s `Config::permissions_enabled`
618 /// branch. Always constructed (cheap, empty) regardless of whether the
619 /// engine is ever active — the same "zero cost when off" posture as
620 /// [`Self::doom_loop_last_call`].
621 permissions_approval_cache: crate::permissions::ApprovalCache,
622 /// P5-1: the non-interactive decision seam a caller installs via
623 /// [`Self::set_permissions_approval_handler`] — mirrors
624 /// `Self::span_summarizer`/[`Self::session_titler`]'s "installing one
625 /// alone changes nothing, `Config::permissions_enabled` is the actual
626 /// gate" pattern. `None` (the default) means every `Ask`-tier decision
627 /// is denied (fail-closed — see
628 /// `crate::permissions::approval::PermissionsApprovalHandler`'s doc
629 /// comment).
630 permissions_approval_handler:
631 Option<std::sync::Arc<dyn crate::permissions::PermissionsApprovalHandler>>,
632 /// P5-2 (§2 module 15 D7 row 4 "prompts-as-commands"): MCP server
633 /// prompts registered via [`Self::register_mcp_prompt`], keyed by their
634 /// ALREADY-NAMESPACED command name (`mcp__<server>__<prompt>` — see
635 /// [`crate::mcp::McpPromptSource`]'s doc comment for why that namespace
636 /// is what keeps an untrusted server's prompt from ever colliding with
637 /// a trusted `Config::prompts` entry). Empty by default, at zero cost:
638 /// [`Self::expand_prompt_async`] only consults this after
639 /// `Config::prompts` finds no match.
640 mcp_prompts: std::collections::HashMap<String, Box<dyn crate::sdk::SdkPromptSource>>,
641 /// BP-6 (catalog D2 "Skills (progressive-disclosure packages)", D7
642 /// "Skill discovery from multiple roots"): the SKILL.md packages
643 /// discovered for this config, frontmatter only — name, description,
644 /// version and the manifest path. Never a body: a body is read from
645 /// disk on invocation (`/name`, `/skill:name`, a `$slug` mention, or
646 /// the `skill` tool) and nowhere else. Empty unless `[core.skills]` is
647 /// on AND names a harness whose roots to read.
648 skills: Vec<crate::skills::LoopSkill>,
649 /// P5-3 (§2 module 9): how deep in the spawn tree THIS agent is — `0`
650 /// for a top-level agent. Set from [`Config::subagent_depth`] at
651 /// construction; `Self::run_spawn_subagent` builds a child `Config`
652 /// with `subagent_depth = self.subagent_depth + 1` and ALSO overwrites
653 /// the freshly-built child `Agent`'s own field to match (belt-and-
654 /// suspenders — the child never has to trust its own `Config` alone).
655 subagent_depth: usize,
656 /// P5-3 (resource bound, "must not fork-bomb"): the shared, tree-wide
657 /// concurrency gauge every spawn (this agent's own, and every
658 /// descendant's) increments/decrements against
659 /// (`crate::subagents::try_acquire`/`ConcurrencyGuard`). A TOP-level
660 /// agent gets a fresh `Arc::new(AtomicUsize::new(0))` at construction;
661 /// `Self::run_spawn_subagent` clones this SAME `Arc` into every child it
662 /// spawns (never a fresh one), so a cap of N holds across the WHOLE
663 /// tree regardless of its branching shape — a parent with 3 children
664 /// each spawning 3 more shares one counter, not nine independent ones.
665 subagent_concurrency_gauge: std::sync::Arc<std::sync::atomic::AtomicUsize>,
666 /// P5-3 (D3 "background+resume"): background subagents this agent has
667 /// spawned and not yet reaped via `subagent_status`, keyed by their
668 /// `child_agent_id`. Each entry's `JoinHandle` moves its own
669 /// [`crate::subagents::ConcurrencyGuard`] into the spawned task, so the
670 /// concurrency slot is held for exactly as long as the child is
671 /// actually running, independent of whether/when the parent polls.
672 background_subagents: std::collections::HashMap<String, BackgroundSubagent>,
673 /// P5-3 (§2.2 C6 "parent-surfaced queue"): approval requests a
674 /// `background_prompts = "parent"` child raised, queued here rather
675 /// than blocking (see [`crate::subagents::QueuedApproval`]'s doc
676 /// comment — each is already resolved `Deny` by the time it lands
677 /// here). Exposed read-only via [`Self::pending_child_approvals`].
678 /// Always constructed (cheap, empty) regardless of whether background
679 /// spawning is ever used, same "zero cost when off" posture as
680 /// [`Self::permissions_approval_cache`].
681 pending_child_approvals:
682 std::sync::Arc<std::sync::Mutex<Vec<crate::subagents::QueuedApproval>>>,
683 /// P5-4 (tui, closes the P5-3 §2.2 C6 deferred chain — see
684 /// [`crate::subagents::ParentQueueApprovalHandler`]'s doc comment for
685 /// the "never blocks" contract this OVERRIDES only when a factory is
686 /// installed): when `Some`, `Self::run_spawn_subagent` uses THIS
687 /// factory — instead of constructing the default never-blocking
688 /// [`crate::subagents::ParentQueueApprovalHandler`] — to build the
689 /// `PermissionsApprovalHandler` a `background_prompts = "parent"`
690 /// child gets. Installed via
691 /// [`Self::set_child_approval_handler_factory`] by a `tui` embedder
692 /// that wants queued child approvals to be genuinely ANSWERABLE
693 /// (blocks the child's tool call until the parent resolves it, or
694 /// denies if the factory's handler's channel is ever dropped/closed —
695 /// still fail-closed, never a hang past process lifetime). `None` (the
696 /// default) preserves P5-3's shipped behavior byte-for-byte: every
697 /// `background_prompts = "parent"` child still gets the immediate-deny
698 /// `ParentQueueApprovalHandler`, and [`Self::pending_child_approvals`]
699 /// stays exactly the read-only audit view it already is.
700 child_approval_handler_factory: Option<std::sync::Arc<ChildApprovalHandlerFactory>>,
701 /// P5-3 (D5 "subagent transcripts… persisted + linked"): an optional
702 /// `(store, this agent's own session name)` pair installed via
703 /// [`Self::set_subagent_store`] — mirrors [`Self::set_recorder`]/
704 /// [`Self::set_span_summarizer`]'s "installing one alone changes
705 /// nothing" pattern. `None` (the default) means a spawned child's
706 /// transcript/lineage is still joined back into THIS agent's context
707 /// (the foreground/background mechanics work either way) but nothing
708 /// is written to a [`crate::store::SessionStore`] — no behavior change
709 /// for any caller that never installs one (e.g. every pre-P5-3 caller).
710 subagent_store: Option<(std::sync::Arc<crate::store::SessionStore>, String)>,
711 /// Imported Claude runtime state. The manifest can be paused or active,
712 /// but this Agent contains no scheduler or timer handle; an embedding
713 /// driver owns execution and persistence.
714 claude_runtime_manifest: Option<crate::claude_runtime_state::ClaudeRuntimeManifest>,
715 /// P5-6 (§2 module 4 `tools.background`, D10 "bg-manager"): background
716 /// OS processes spawned via `background_exec`, keyed by job id, tracked
717 /// until reaped (a terminal `background_status` poll, or an explicit
718 /// `background_kill`) — see [`BackgroundJob`]'s doc comment. Always
719 /// constructed (cheap, empty), same "zero cost when off" posture as
720 /// [`Self::background_subagents`].
721 background_jobs: std::collections::HashMap<String, BackgroundJob>,
722 /// P5-6 (resource bound, mirroring [`Self::subagent_concurrency_gauge`]'s
723 /// own precedent): the shared concurrency gauge every `background_exec`
724 /// call on this agent increments/decrements against
725 /// (`crate::subagents::try_acquire`/`ConcurrencyGuard` — reused
726 /// verbatim, a second independent gauge instance scoped to background
727 /// JOBS rather than subagent SPAWNS).
728 background_concurrency_gauge: std::sync::Arc<std::sync::atomic::AtomicUsize>,
729 /// P5-9 (§2 module 20 `checkpoint`): the write-path-interception
730 /// observer installed on [`Self::ctx`]'s `write_observer` (as a
731 /// `dyn WriteObserver`), held here ADDITIONALLY as its concrete type so
732 /// `Self::run_loop` can call
733 /// [`crate::checkpoint::CheckpointObserver::begin_turn`] once per turn.
734 /// `None` when `Config::checkpoint_enabled` is `false` (the default) or
735 /// the shadow store failed to open — see
736 /// [`crate::checkpoint::observer_for_config`].
737 checkpoint_observer: Option<std::sync::Arc<crate::checkpoint::CheckpointObserver>>,
738 /// P5-11 (§2 module 28 `lsp`): the LSP server registry installed (via
739 /// `crate::lsp::LspDiagnosticsObserver`) on `Self::ctx`'s
740 /// `write_observer` chain, held here ADDITIONALLY as its concrete type
741 /// so `impl Drop for Agent` can reach
742 /// [`crate::lsp::LspManager::kill_all_sync`] (no orphaned language-
743 /// server processes) and a clean-exit caller can reach
744 /// [`crate::lsp::LspManager::shutdown_all`] for a graceful handshake.
745 /// `None` when `Config::lsp_enabled` is `false` (the default).
746 lsp_manager: Option<std::sync::Arc<crate::lsp::LspManager>>,
747}
748
749/// P5-6 (§2 module 4 `tools.background`): one background-spawned OS process
750/// this agent is tracking, awaiting a `background_status`/`background_list`
751/// poll (or `background_kill`/agent drop) to reap or terminate it.
752///
753/// **Real process, not a child agent.** Unlike [`BackgroundSubagent`] (which
754/// wraps a whole recursive child [`Agent`] loop against the SAME mock/real
755/// provider), this wraps a plain OS subprocess spawned via
756/// `crate::tools::build_sandboxed_sh` — the exact function
757/// [`crate::tools::BashTool::execute`] itself calls, so a background
758/// command gets byte-identical sandboxing/cwd/env handling to a foreground
759/// `bash` call (build brief: "reuse the bash tool's execution + sandbox
760/// path").
761struct BackgroundJob {
762 /// The live process handle — kept directly on the job (not moved into a
763 /// spawned task) so [`Agent::run_background_status`]/
764 /// [`Agent::run_background_list`] can call the SYNCHRONOUS,
765 /// non-blocking `Child::try_wait` to observe exit status, and
766 /// [`Agent::run_background_kill`]/[`impl Drop for Agent`] can call the
767 /// SYNCHRONOUS `Child::start_kill` for a REAL process kill — never just
768 /// a `tokio::task::JoinHandle::abort` (which would only cancel a Rust
769 /// future, not the OS process it spawned). `kill_on_drop(true)` was set
770 /// at spawn time as defense-in-depth: even a `BackgroundJob` dropped
771 /// through some path OTHER than the explicit kill call sites below
772 /// still kills its child (a documented tokio behavior; a no-op if the
773 /// process already exited).
774 child: tokio::process::Child,
775 /// The exact command text this job is running — the SAME text that was
776 /// already checked against the permissions engine at spawn time (see
777 /// [`Agent::background_permission_denial`]).
778 command: String,
779 /// The OS process id, captured once at spawn time (before `child` is
780 /// ever mutated) — surfaced in every status/list/kill result, and the
781 /// only thing an OUTSIDE observer (e.g. a test proving real
782 /// termination) needs to check liveness independent of this process's
783 /// own bookkeeping.
784 pid: Option<u32>,
785 /// Bounded, incrementally-appended combined stdout+stderr capture —
786 /// written to by the reader tasks [`Agent::run_background_exec`] spawns
787 /// right after `child.stdout`/`child.stderr` are taken, read by every
788 /// status/list poll. Shared via `Arc` since the reader tasks outlive
789 /// this method call.
790 output: std::sync::Arc<supercode_runtime::background::CapturedOutput>,
791 /// Unix-ms wall-clock time the spawn happened.
792 started_at_ms: i64,
793 /// Set by [`Agent::run_background_kill`] — [`Agent::run_background_status`]/
794 /// [`Agent::run_background_list`] report [`supercode_runtime::background::JobStatus::Killed`]
795 /// unconditionally once this is `true`, rather than racing
796 /// `Child::try_wait` to see whether the kill signal has landed yet.
797 killed: bool,
798 /// The concurrency-gauge slot this job holds for as long as it remains
799 /// in [`Agent::background_jobs`] — dropped (freeing the slot) when this
800 /// `BackgroundJob` is removed from the map (a terminal reap, or an
801 /// explicit kill), exactly mirroring [`BackgroundSubagent`]'s own
802 /// "guard held for as long as it's tracked, not just while the process
803 /// is alive" posture (§2 module 9 precedent, kept consistent here).
804 _guard: crate::subagents::ConcurrencyGuard,
805}
806
807/// P5-6: the non-blocking status read [`Agent::run_background_status`]/
808/// [`Agent::run_background_list`] share — `job.killed` (set by
809/// [`Agent::run_background_kill`]) always wins over a fresh `try_wait`,
810/// since a kill signal racing the OS reaping the process is otherwise
811/// indistinguishable from "still running" for one poll cycle; reporting
812/// `Killed` unconditionally once requested avoids that race entirely. A
813/// `try_wait` error (would only happen if this job's id were somehow
814/// double-reaped, which the map ownership below already prevents) is
815/// treated as "no news yet" — `Running` — rather than inventing a made-up
816/// exit code.
817fn background_job_status(job: &mut BackgroundJob) -> supercode_runtime::background::JobStatus {
818 if job.killed {
819 return supercode_runtime::background::JobStatus::Killed;
820 }
821 match job.child.try_wait() {
822 Ok(Some(status)) => supercode_runtime::background::JobStatus::Exited(status.code()),
823 Ok(None) | Err(_) => supercode_runtime::background::JobStatus::Running,
824 }
825}
826
827/// Fable-5 review (HIGH, "grandchildren orphaned on kill AND agent-drop"):
828/// the shared real-kill body for both [`Agent::run_background_kill`] and
829/// `impl Drop for Agent` — sends `SIGKILL` to `job`'s ENTIRE process group,
830/// not just the one directly-tracked pid, so a surviving `&` job, pipeline
831/// stage, or double-forking daemon spawned by the job is killed too, then
832/// reaps the group leader so it doesn't linger as a zombie.
833///
834/// Relies on the spawn site (`Agent::run_background_exec`) having put the
835/// job in its OWN new process group via `Command::process_group(0)` — which
836/// makes the leader's pgid equal to its own pid, so `job.pid` doubles as the
837/// group id here.
838#[cfg(unix)]
839fn kill_job_process_group(job: &mut BackgroundJob) {
840 if let Some(pid) = job.pid {
841 // SAFETY: `libc::kill` with a negative pid is `killpg` — it only
842 // ever sends a signal (never dereferences memory), so this is safe
843 // regardless of whether the group is still alive. A `-1`/`ESRCH`
844 // return means the leader (and thus the whole group, since a group
845 // can't outlive its leader) already exited — not an error, just
846 // "already dead", exactly like `Child::start_kill`'s own documented
847 // no-op-on-already-exited contract.
848 unsafe {
849 libc::kill(-(pid as libc::pid_t), libc::SIGKILL);
850 }
851 }
852 // Belt-and-suspenders for the leader itself — `kill_on_drop(true)` set
853 // at spawn time is the same outcome via a different (implicit) path —
854 // then reap it so the SIGKILL we just delivered doesn't leave a zombie
855 // behind.
856 let _ = job.child.start_kill();
857 let _ = job.child.try_wait();
858}
859
860/// Non-unix fallback: no portable process-group primitive is wired up here
861/// (same posture as `crate::tools::build_sandboxed_sh`'s own platform
862/// split) — falls back to the pre-fix per-child kill. A background job that
863/// spawns a surviving grandchild process on a non-Unix target is a
864/// documented residual, not silently claimed fixed by this cfg arm.
865#[cfg(not(unix))]
866fn kill_job_process_group(job: &mut BackgroundJob) {
867 let _ = job.child.start_kill();
868}
869
870/// P5-6 (D1 "monitor/event feed", "output captured incrementally +
871/// BOUNDED"): spawn a fire-and-forget reader task that continuously drains
872/// `reader` (a piped `ChildStdout`/`ChildStderr`) into `output`, bounded at
873/// `cap` bytes. Reading NEVER stops at the cap — only what's RETAINED is
874/// bounded ([`supercode_runtime::background::CapturedOutput::append`]'s own contract)
875/// — because a background job's child process would otherwise block
876/// forever writing to a full, undrained OS pipe once this stopped reading
877/// it, silently hanging real work behind an apparently-"running" job. The
878/// task exits on its own once the pipe reaches EOF (the process closed the
879/// descriptor, whether by exiting or being killed) — no explicit
880/// abort/cleanup call site is needed; a detached `tokio::spawn` this short-
881/// lived is not the kind of orphaned-task risk `impl Drop for Agent`'s own
882/// doc comment is about (that one concerns a whole recursive provider-
883/// calling child AGENT loop, not a bounded byte-copy loop that ends the
884/// instant its source pipe closes).
885fn spawn_output_reader<R>(
886 reader: R,
887 output: std::sync::Arc<supercode_runtime::background::CapturedOutput>,
888 cap: usize,
889) -> tokio::task::JoinHandle<()>
890where
891 R: tokio::io::AsyncRead + Unpin + Send + 'static,
892{
893 tokio::spawn(async move {
894 use tokio::io::AsyncReadExt;
895 let mut reader = reader;
896 let mut buf = [0u8; 8192];
897 loop {
898 match reader.read(&mut buf).await {
899 Ok(0) => break,
900 Ok(n) => {
901 let chunk = String::from_utf8_lossy(&buf[..n]);
902 output.append(&chunk, cap);
903 }
904 Err(_) => break,
905 }
906 }
907 })
908}
909
910/// BP-7 (catalog §4a "Named agent definitions as data"): merge
911/// `<cwd>/.claude/agents/*.md` into `config.subagents_definitions`.
912///
913/// Runs for every `Agent` whose `subagents` module is on, whatever preset
914/// it came from — before BP-7 the `.md` loader was reachable only from the
915/// Claude emulate/resume path, so a cc-parity or cx-parity session ignored
916/// definitions sitting right there in the repo.
917///
918/// * A no-op when the module is off (the default), and for every spawned
919/// CHILD (`subagent_depth > 0`), which already inherits its parent's
920/// resolved definitions verbatim.
921/// * A config-table entry WINS over a discovered file of the same name:
922/// `[capabilities.subagents.agents.<name>]` is explicit configuration,
923/// the file is discovery.
924/// * A malformed file is skipped with a warning, never a failed
925/// construction: `Agent::with_provider` has no error channel, and a
926/// broken agent file in some repo must not make the harness unusable
927/// there. (The emulate/resume path keeps its own strict behavior, where
928/// a definition the resumed session may depend on going missing IS worth
929/// failing over.)
930fn merge_project_agent_definitions(config: &mut Config) {
931 if !config.subagents_enabled || config.subagent_depth > 0 {
932 return;
933 }
934 match crate::claude_compat::load_project_agents(&config.cwd) {
935 Ok(agents) => {
936 for agent in agents {
937 config
938 .subagents_definitions
939 .entry(agent.definition.name.clone())
940 .or_insert(agent.definition);
941 }
942 }
943 Err(e) => {
944 tracing::debug!(
945 error = %e,
946 "skipping .claude/agents discovery: a definition file could not be parsed"
947 );
948 }
949 }
950}
951
952/// P5-3: one background-spawned child this agent is tracking, awaiting a
953/// `subagent_status` poll (or agent drop) to reap it.
954struct BackgroundSubagent {
955 /// Resolves to `(child_agent_id, child's final result, the child's own
956 /// post-system-prompt history — for D5 transcript persistence once
957 /// reaped)` — the concurrency-guard slot for this child is held INSIDE
958 /// the spawned future (moved in at spawn time), so it releases the
959 /// instant the child's own run loop finishes, not when the parent gets
960 /// around to polling.
961 handle: tokio::task::JoinHandle<(String, Result<String>, Vec<ChatMessage>)>,
962 /// The task/prompt text the child was spawned with (surfaced by a
963 /// `"pending"` status poll, since the handle alone can't answer "what
964 /// is it doing").
965 task: String,
966 /// The named `agent_type` spawned, if any.
967 agent_type: Option<String>,
968 /// Unix-ms wall-clock time the spawn happened.
969 started_at_ms: i64,
970 /// BP-7 (catalog §4a "Background subagents + resume"): the child's own
971 /// steering inbox, captured before the child moved into its task.
972 ///
973 /// This IS the mailbox. `SteerInbox` was built (P4b) to be writable
974 /// while an active turn holds `&mut Agent` — exactly the property a
975 /// message-to-a-running-child needs — so the mailbox is that existing
976 /// seam reached from outside, not a second delivery channel with its
977 /// own ordering rules. A message lands at the top of the child's next
978 /// loop iteration, per `Config::steering_mode`.
979 mailbox: std::sync::Arc<std::sync::Mutex<SteerInbox>>,
980}
981
982/// P5-3 safety hardening (Fable-5 review, MEDIUM-LOW "orphaned billed
983/// spend"): a dropped parent must not leave a detached background child
984/// running against a REAL provider. Without this, a parent dropped
985/// mid-run (the caller's own process exits the scope, panics, or simply
986/// stops polling) leaves every still-running `BackgroundSubagent::handle`
987/// as an orphaned `tokio::spawn` task: nothing had ever awaited or
988/// aborted it, so it runs to its own (`max_iterations`-bounded)
989/// completion regardless — bounded but real provider spend nobody is
990/// paying attention to.
991///
992/// `.abort()` on a [`tokio::task::JoinHandle`] is safe to call
993/// unconditionally, including on an ALREADY-finished task (a documented
994/// no-op there — see tokio's `JoinHandle::abort` docs) — so this never
995/// needs to distinguish "still running" from "already done"; a background
996/// child that already finished and is merely awaiting a `subagent_status`
997/// reap is untouched in practice (aborting a finished task changes
998/// nothing observable). For a task still mid-flight, tokio cancels it at
999/// its next `.await` point, which drops that future in place — including
1000/// the `_guard: ConcurrencyGuard` moved into it at spawn time (see
1001/// `Self::run_spawn_subagent`'s `tokio::spawn` body) — so the
1002/// concurrency-gauge slot is released exactly the same way a normal
1003/// completion releases it (`ConcurrencyGuard`'s own `Drop`, in
1004/// `crate::subagents`). No separate cleanup call site to forget.
1005///
1006/// Deliberately does NOT touch [`Self::pending_child_approvals]` or
1007/// `Self::subagent_store` — this is purely "stop burning provider
1008/// calls on behalf of a caller who's gone", not a transcript-persistence
1009/// path (a child aborted mid-flight has no finished result to persist;
1010/// see this build's named residual on abort-time transcript loss).
1011impl Drop for Agent {
1012 fn drop(&mut self) {
1013 for (child_id, bg) in self.background_subagents.drain() {
1014 // Named, not silent: a child that was still running gets its
1015 // provider calls cut off here — worth a trace even though
1016 // there's no transcript left to persist (the future is
1017 // dropped mid-flight, before it ever returns a result).
1018 if !bg.handle.is_finished() {
1019 tracing::debug!(
1020 child_id = %child_id,
1021 "parent Agent dropped: aborting still-running background subagent \
1022 to stop further provider spend"
1023 );
1024 }
1025 bg.handle.abort();
1026 }
1027 // P5-6 (§2 module 4 `tools.background`, build brief "on agent drop
1028 // / session end, jobs MUST be killed... real process kill via the
1029 // child handle's kill(), not just tokio task abort"): a REAL OS
1030 // process, not a Rust task — `Child::start_kill` (synchronous, no
1031 // `.await` needed, so callable from this non-async `Drop::drop`)
1032 // sends the actual kill signal; a no-op, per its own docs, on a
1033 // job that already exited. `kill_on_drop(true)` (set at spawn
1034 // time) is a second, independent line of defense for the same
1035 // outcome, but this explicit loop is what makes the guarantee
1036 // provable/traceable rather than relying solely on an implicit
1037 // tokio runtime behavior.
1038 for (job_id, mut job) in self.background_jobs.drain() {
1039 if !job.killed {
1040 tracing::debug!(
1041 job_id = %job_id,
1042 command = %job.command,
1043 "parent Agent dropped: killing still-tracked background job's real \
1044 OS process (and its whole process group — see \
1045 `kill_job_process_group`)"
1046 );
1047 }
1048 kill_job_process_group(&mut job);
1049 }
1050 // P5-11 (§2 module 28 `lsp`, build brief "no orphaned language-
1051 // server processes"): a REAL OS process, same rationale as the
1052 // background-job loop just above — `kill_all_sync` is
1053 // synchronous (`Child::start_kill`, no `.await` needed, so
1054 // callable from this non-async `Drop::drop`), SIGKILLs each
1055 // server's WHOLE process group (unix — same `kill_job_process_group`
1056 // mechanism as the background-job loop above, so worker
1057 // grandchildren like rust-analyzer's proc-macro server or
1058 // typescript-language-server's `tsserver` are killed too, not just
1059 // the one directly-tracked pid), and is provable/traceable rather
1060 // than relying solely on `kill_on_drop(true)`'s implicit tokio
1061 // runtime behavior (which remains a second, independent line of
1062 // defense on every spawned `LspClient`).
1063 if let Some(lsp) = &self.lsp_manager {
1064 lsp.kill_all_sync();
1065 }
1066 }
1067}
1068
1069/// P4 (§1.8 credential-helper indirection, D6 row): run an `api_key_cmd`
1070/// through the shell and return its trimmed stdout. Runs via `sh -c` (POSIX
1071/// shell, matching pi's `!command` precedent) so the configured string can
1072/// use pipes/substitution, e.g. `pass show api-key`. Never panics or
1073/// propagates an error: a spawn failure or non-zero exit is reported via
1074/// `tracing::warn!` and returns an empty `String`, which
1075/// `Agent::new`'s resolution chain treats exactly like an unset helper —
1076/// falling through to `Config::api_key_env`.
1077fn run_api_key_cmd(cmd: &str) -> String {
1078 match std::process::Command::new("sh").arg("-c").arg(cmd).output() {
1079 Ok(out) if out.status.success() => String::from_utf8_lossy(&out.stdout).trim().to_string(),
1080 Ok(out) => {
1081 tracing::warn!(
1082 "api_key_cmd exited with status {:?}; falling back to api_key_env",
1083 out.status.code()
1084 );
1085 String::new()
1086 }
1087 Err(e) => {
1088 tracing::warn!("api_key_cmd failed to run ({e}); falling back to api_key_env");
1089 String::new()
1090 }
1091 }
1092}
1093
1094/// BP-9 (§3.1 `core.api_key_command`, D6 row "Credential helpers /
1095/// keyring"): run an ARGV credential helper and return its trimmed stdout.
1096/// No shell is involved — `argv[0]` is exec'd with the rest as arguments —
1097/// so a helper path with spaces, or an argument containing `$`/`;`, means
1098/// what it says. Same never-panics, fall-through-on-failure contract as
1099/// [`run_api_key_cmd`]: an empty result is treated as "no helper".
1100pub(crate) fn run_api_key_command(argv: &[String]) -> String {
1101 let Some((program, args)) = argv.split_first() else {
1102 return String::new();
1103 };
1104 match std::process::Command::new(program).args(args).output() {
1105 Ok(out) if out.status.success() => String::from_utf8_lossy(&out.stdout).trim().to_string(),
1106 Ok(out) => {
1107 tracing::warn!(
1108 "api_key_command exited with status {:?}; trying the next credential source",
1109 out.status.code()
1110 );
1111 String::new()
1112 }
1113 Err(e) => {
1114 tracing::warn!(
1115 "api_key_command failed to run ({e}); trying the next credential source"
1116 );
1117 String::new()
1118 }
1119 }
1120}
1121
1122/// P4c (§1.2/§3.1 `core.shell_env_snapshot`, SPLIT CC+CX row, catalog:338):
1123/// capture the user's interactive login-shell environment ONCE, best-effort.
1124/// Runs `$SHELL -lc env` (falling back to `sh -lc env` when `$SHELL` is
1125/// unset) — a LOGIN shell (`-l`) sources the user's rc files, which is
1126/// exactly the sourcing `bash` calls should no longer need to repeat once
1127/// this snapshot is in hand. Never panics: any failure (spawn error,
1128/// non-zero exit, unparseable output) returns an empty map, which
1129/// `ToolContext::shell_env`'s "no-op when `None`/empty" contract already
1130/// treats as harmless.
1131fn capture_shell_env() -> std::collections::HashMap<String, String> {
1132 let shell = std::env::var("SHELL").unwrap_or_else(|_| "sh".to_string());
1133 let out = match std::process::Command::new(&shell)
1134 .arg("-lc")
1135 .arg("env")
1136 .output()
1137 {
1138 Ok(o) if o.status.success() => o.stdout,
1139 Ok(o) => {
1140 tracing::warn!(
1141 "shell_env_snapshot: `{shell} -lc env` exited with status {:?}; snapshot is empty",
1142 o.status.code()
1143 );
1144 return std::collections::HashMap::new();
1145 }
1146 Err(e) => {
1147 tracing::warn!(
1148 "shell_env_snapshot: failed to run `{shell} -lc env` ({e}); snapshot is empty"
1149 );
1150 return std::collections::HashMap::new();
1151 }
1152 };
1153 let text = String::from_utf8_lossy(&out);
1154 let mut map = std::collections::HashMap::new();
1155 for line in text.lines() {
1156 if let Some((k, v)) = line.split_once('=') {
1157 if !k.is_empty() {
1158 map.insert(k.to_string(), v.to_string());
1159 }
1160 }
1161 }
1162 map
1163}
1164
1165/// Build the [`ToolContext`] an [`Agent`] hands to every tool call, folding
1166/// in every P4c per-tool config knob (§1.2) alongside the pre-existing
1167/// `cwd`/`sandbox` — shared by [`Agent::with_parts`]/[`Agent::with_provider_arc`]
1168/// so the two construction paths can never drift apart on which config
1169/// fields reach the context. Also builds (P5-9) the
1170/// [`crate::checkpoint::CheckpointObserver`], if `config.checkpoint_enabled`
1171/// — installed on the returned context's `write_observer` AND returned
1172/// separately (as the concrete type) so `Agent::run_loop` can call
1173/// [`crate::checkpoint::CheckpointObserver::begin_turn`] once per turn.
1174/// `None`/no-op end to end when the module is off — see
1175/// [`crate::checkpoint::observer_for_config`]'s own doc comment for the
1176/// default-off byte-identity guarantee.
1177///
1178/// P5-11 (§2 modules 28/29, D-5 "shared write-path interception seam"):
1179/// `crate::formatters::observer_for_config`/`crate::lsp::manager_for_config`
1180/// are folded into the SAME `write_observer` slot via
1181/// [`crate::tools::WriteObserverChain`], in the design's required order —
1182/// `checkpoint -> formatters -> lsp` (checkpoint's pre-image capture must
1183/// see the file before ANY mutation; lsp's diagnostics must see the file
1184/// AFTER formatting, never before). When 0 or 1 of the three modules is
1185/// active, this degrades to exactly what P5-9 shipped (`None`, or the
1186/// single concrete observer installed directly) — no chain wrapper is
1187/// introduced unless there is actually more than one observer to order,
1188/// keeping every single-module (or all-off) configuration byte-identical
1189/// to before this function grew multi-observer support. The `lsp` manager
1190/// is ALSO returned separately (like `checkpoint_observer`), so
1191/// `Agent`'s `Drop` impl can reach `crate::lsp::LspManager::kill_all_sync`
1192/// regardless of how the chain is shaped.
1193/// BP-2: `pub(crate)` so a parity test can build the SAME `ToolContext` an
1194/// `Agent` would from a resolved preset's `Config` and drive a registry
1195/// tool through it — a tool's behavior under a preset is exactly the
1196/// composition of the two, and a test that hand-assembled a context would
1197/// be proving an unwired function.
1198pub(crate) fn build_tool_context(
1199 config: &Config,
1200) -> (
1201 ToolContext,
1202 Option<std::sync::Arc<crate::checkpoint::CheckpointObserver>>,
1203 Option<std::sync::Arc<crate::lsp::LspManager>>,
1204) {
1205 let shell_env = if config.shell_env_snapshot {
1206 Some(std::sync::Arc::new(capture_shell_env()))
1207 } else {
1208 None
1209 };
1210 let checkpoint_observer = crate::checkpoint::observer_for_config(config);
1211 let format_observer = crate::formatters::observer_for_config(config);
1212 let lsp_manager = crate::lsp::manager_for_config(config);
1213 let lsp_observer = lsp_manager
1214 .clone()
1215 .map(|m| std::sync::Arc::new(crate::lsp::LspDiagnosticsObserver::new(m)));
1216 let mut observers: Vec<std::sync::Arc<dyn crate::tools::WriteObserver>> = Vec::new();
1217 if let Some(cp) = &checkpoint_observer {
1218 observers.push(cp.clone() as std::sync::Arc<dyn crate::tools::WriteObserver>);
1219 }
1220 if let Some(f) = &format_observer {
1221 observers.push(f.clone() as std::sync::Arc<dyn crate::tools::WriteObserver>);
1222 }
1223 if let Some(l) = &lsp_observer {
1224 observers.push(l.clone() as std::sync::Arc<dyn crate::tools::WriteObserver>);
1225 }
1226 let write_observer: Option<std::sync::Arc<dyn crate::tools::WriteObserver>> =
1227 match observers.len() {
1228 0 => None,
1229 1 => observers.into_iter().next(),
1230 _ => Some(std::sync::Arc::new(crate::tools::WriteObserverChain::new(
1231 observers,
1232 ))),
1233 };
1234 let ctx = ToolContext {
1235 cwd: config.cwd.clone(),
1236 // BP-10 (catalog row "Additional working directories"): the
1237 // `--add-dir`/`core.additional_dirs` roots reach the TOOLS now,
1238 // not just the project-context walk — `ToolContext::check_write`,
1239 // the OS backstop's writable set, and the permissions engine's
1240 // path rules all read them. Empty (the default) is byte-identical
1241 // to confining everything to `cwd`.
1242 extra_roots: config.additional_dirs.clone(),
1243 sandbox: config.sandbox,
1244 multimodal_read: config.read_file_multimodal,
1245 // BP-2 (§1.2 `core.tools.read_file.line_numbers`, catalog:26).
1246 read_line_numbers: config.read_file_line_numbers,
1247 require_read_before_edit: config.edit_file_require_read_before_edit,
1248 // BP-2: path → content hash at read time (`ToolContext::read_state`).
1249 read_paths: std::sync::Arc::new(std::sync::Mutex::new(std::collections::HashMap::new())),
1250 notebook_aware: config.edit_file_notebook_aware,
1251 shell_env,
1252 nested_instructions: config.nested_instructions,
1253 injected_instruction_dirs: std::sync::Arc::new(std::sync::Mutex::new(HashSet::new())),
1254 // BP-5: filled in by `Agent::with_parts` (the one construction path
1255 // that assembles a prompt, and therefore the one that knows which
1256 // rules were held back); empty everywhere else.
1257 path_rules: std::sync::Arc::new(Vec::new()),
1258 injected_rule_files: std::sync::Arc::new(std::sync::Mutex::new(HashSet::new())),
1259 // P5-1 (§2 module 12 carry-forward): now sourced from real config
1260 // (`capabilities.permissions.sandbox.network.*`, wired by
1261 // `configfile::materialize_config`) instead of always `None`. `None`
1262 // (the default, unchanged when the config never sets it) is still
1263 // byte-identical to today's behavior.
1264 network_policy: config.network_policy.clone(),
1265 // BP-10 (catalog row "Allow/ask/deny rule language", the DOMAIN
1266 // subject): the config's own rule arrays reach the network surface
1267 // too, so a `domain(...)` rule is evaluated by the SAME engine that
1268 // evaluates `bash(...)`/`write(...)` at the dispatch gate — not by
1269 // a second matcher over a second list. `None` when the permissions
1270 // module is off, which is byte-identical to before.
1271 permission_rules: config.permissions_enabled.then(|| {
1272 std::sync::Arc::new(crate::permissions::RuleSet {
1273 deny: config.tool_deny_patterns.clone(),
1274 ask: config.permissions_ask_patterns.clone(),
1275 allow: config.tool_allow_patterns.clone(),
1276 })
1277 }),
1278 // P4e (S3.1 `core.tools.bash.timeout_secs`, S14): folds the `bash`
1279 // `ToolOverride`'s `timeout_secs`, if set, into the context every
1280 // `BashTool::execute` call receives -- `None` (no override
1281 // configured) is byte-identical to today's behavior.
1282 bash_timeout_secs: config
1283 .tool_overrides
1284 .get("bash")
1285 .and_then(|o| o.timeout_secs),
1286 write_observer,
1287 // P5-10 (§2 module 12): sourced from real config
1288 // (`capabilities.permissions.sandbox.{enabled,escalation,env_policy}`,
1289 // wired by `configfile::materialize_config`). `sandbox_approval_handler`
1290 // starts `None` here (no handler is installed yet at `Agent`
1291 // construction time) and is kept in sync by
1292 // `Agent::set_permissions_approval_handler` — see that method's doc
1293 // comment.
1294 sandbox_os_enabled: config.sandbox_os_enabled,
1295 sandbox_escalation: config.sandbox_escalation,
1296 sandbox_env_policy: config.sandbox_env_policy,
1297 sandbox_approval_handler: None,
1298 // BP-3: both handler seams start `None` (nothing is installed at
1299 // construction time) and are filled by
1300 // `Agent::set_permissions_approval_handler` /
1301 // `Agent::set_user_question_handler`, exactly like
1302 // `sandbox_approval_handler` above. The two shared states are
1303 // always present but inert: plan mode starts off (contributing no
1304 // rules), and the budget starts unpublished.
1305 question_handler: None,
1306 approval_handler: None,
1307 plan_mode: std::sync::Arc::new(crate::tools::PlanModeState::new()),
1308 context_budget: std::sync::Arc::new(crate::tools::ContextBudget::new()),
1309 // BP-8 (catalog:156): the shared plan the agent journals and
1310 // persists. Always present, empty and inert until `update_plan`
1311 // writes one.
1312 plan: std::sync::Arc::new(std::sync::Mutex::new(Vec::new())),
1313 };
1314 (ctx, checkpoint_observer, lsp_manager)
1315}
1316
1317/// P4b (§1.4/§3.1, catalog §4a "Global/user-level instruction file tier"):
1318/// where the user/global instruction tier lives — `$SUPERCODE_HOME`, else
1319/// `$XDG_CONFIG_HOME/supercode`, else `~/.config/supercode`. Deliberately
1320/// duplicates `crates/cli/src/userconfig.rs::config_home`'s exact precedence
1321/// rather than depending on the `cli` crate from `core` (wrong dependency
1322/// direction — `cli` depends on `core`, never the reverse). `pub(crate)`:
1323/// also the DEFAULT shadow-store root `crate::checkpoint::observer_for_config`
1324/// (P5-9) derives from when `Config::checkpoint_dir` is unset — one
1325/// `$SUPERCODE_HOME` resolver, not a second hand-rolled one.
1326pub(crate) fn global_instructions_dir() -> std::path::PathBuf {
1327 if let Ok(h) = std::env::var("SUPERCODE_HOME") {
1328 if !h.is_empty() {
1329 return std::path::PathBuf::from(h);
1330 }
1331 }
1332 if let Ok(xdg) = std::env::var("XDG_CONFIG_HOME") {
1333 if !xdg.is_empty() {
1334 return std::path::PathBuf::from(xdg).join("supercode");
1335 }
1336 }
1337 let home = supercode_interchange::user_home()
1338 .map(|home| home.to_string_lossy().into_owned())
1339 .ok_or(std::env::VarError::NotPresent)
1340 .unwrap_or_else(|_| ".".into());
1341 std::path::PathBuf::from(home)
1342 .join(".config")
1343 .join("supercode")
1344}
1345
1346/// P4b (§1.4, catalog §4a "Instruction imports"): is `rel` (an `@`-import
1347/// target found inside a PROJECT-sourced instruction file) LEXICALLY safe to
1348/// resolve? Mirrors `configfile::is_safe_project_dir`'s posture (LOW-1
1349/// precedent): rejects absolute paths, `~`-relative paths, and any `..`
1350/// component — an untrusted repo's own CLAUDE.md/AGENTS.md must not be able
1351/// to `@import` its way to an arbitrary file on disk (e.g. `@/etc/passwd`,
1352/// `@../../.ssh/id_rsa`). Global-tier files (the user's own machine, same
1353/// trust level as the user's shell) are NOT run through this check.
1354///
1355/// This is a cheap PRE-FILTER only — it operates on the literal token text
1356/// and cannot see through a symlink committed in the repo whose *target*
1357/// escapes the root while the *link itself* has a clean, traversal-free
1358/// relative name (e.g. `@link.md` where `link.md -> /etc/passwd`). See
1359/// [`import_target_is_contained`] for the canonicalizing check that closes
1360/// that gap; the project-scoped resolution path runs both.
1361fn import_path_is_safe(rel: &str) -> bool {
1362 if rel.is_empty() || rel.contains('\0') {
1363 return false;
1364 }
1365 let path = std::path::Path::new(rel);
1366 if path.is_absolute() || rel.starts_with('~') {
1367 return false;
1368 }
1369 !path
1370 .components()
1371 .any(|c| matches!(c, std::path::Component::ParentDir))
1372}
1373
1374/// P4b security fix (Fable-5 review, MEDIUM: symlink bypass of the
1375/// project-scoped `@`-import boundary): does `candidate` — after resolving
1376/// symlinks — stay inside `root` — also after resolving symlinks? This is
1377/// what actually enforces [`import_path_is_safe`]'s doc-comment guarantee
1378/// ("must not be able to `@import` its way to an arbitrary file on disk"):
1379/// the lexical check alone rejects `@/etc/passwd` and `@../../secret`, but a
1380/// repo can commit a symlink (e.g. `link.md -> /etc/passwd`) whose own
1381/// relative name is perfectly clean, defeating a purely lexical check.
1382///
1383/// Both sides are canonicalized before the comparison — not just
1384/// `candidate` — because `root` itself can legitimately be a symlink (a
1385/// tempdir under macOS's `/tmp` -> `/private/tmp`, or any other symlinked
1386/// project checkout); comparing a canonicalized candidate against a
1387/// non-canonicalized root would falsely reject genuinely-in-root files.
1388///
1389/// Fails CLOSED: a `canonicalize()` failure (broken symlink, a target that
1390/// doesn't exist, a permission error) returns `false` — never inlined,
1391/// mirroring [`expand_instruction_imports`]'s existing "unreadable file ⇒
1392/// left as literal text" posture rather than panicking or defaulting open.
1393pub(crate) fn import_target_is_contained(
1394 candidate: &std::path::Path,
1395 root: &std::path::Path,
1396) -> bool {
1397 let (Ok(real_root), Ok(real_candidate)) = (
1398 std::fs::canonicalize(root),
1399 std::fs::canonicalize(candidate),
1400 ) else {
1401 return false;
1402 };
1403 real_candidate.starts_with(&real_root)
1404}
1405
1406/// P4b (§1.4/§3.1 `core.instruction_imports`, catalog:85): inline `@path`
1407/// import tokens found in `text` with the referenced file's own (trimmed)
1408/// content, resolved relative to `dir` (the directory the CONTAINING file
1409/// lives in — so a chain of imports each resolves relative to its own
1410/// location, not the original file's). `depth` bounds recursion (CC's own
1411/// default of 4, cited in the cc-parity preset) so a cyclical or
1412/// deeply-nested import chain can't blow the stack or loop forever.
1413/// `project_scoped` gates [`import_path_is_safe`] AND
1414/// [`import_target_is_contained`] — see their doc comments; `root` is the
1415/// containment boundary those checks canonicalize against (the SAME root
1416/// for every level of a nested import chain, even though `dir` itself walks
1417/// deeper with each level — an import three levels deep must still resolve
1418/// under the original project root, not merely under its own immediate
1419/// parent). Ignored when `!project_scoped` (the global/user tier, trusted,
1420/// unrestricted — see [`append_instruction_file`]'s doc comment).
1421/// Any token that isn't `@`-prefixed, doesn't resolve to a readable file, or
1422/// (project-scoped) fails the safety/containment check is left as literal
1423/// text — an import is best-effort, never a hard error that could make
1424/// instruction loading fail outright.
1425fn expand_instruction_imports(
1426 text: &str,
1427 dir: &std::path::Path,
1428 root: &std::path::Path,
1429 project_scoped: bool,
1430 depth: u8,
1431) -> String {
1432 if depth >= 4 {
1433 return text.to_string();
1434 }
1435 let mut out = String::with_capacity(text.len());
1436 for token in split_preserving_whitespace(text) {
1437 if let Some(rel) = token.strip_prefix('@') {
1438 if !rel.is_empty()
1439 && !rel.contains(char::is_whitespace)
1440 && (!project_scoped || import_path_is_safe(rel))
1441 {
1442 let candidate = dir.join(rel);
1443 if !project_scoped || import_target_is_contained(&candidate, root) {
1444 if let Ok(imported) = std::fs::read_to_string(&candidate) {
1445 let imported = imported.trim();
1446 if !imported.is_empty() {
1447 let imported_dir = candidate.parent().unwrap_or(dir);
1448 out.push_str(&expand_instruction_imports(
1449 imported,
1450 imported_dir,
1451 root,
1452 project_scoped,
1453 depth + 1,
1454 ));
1455 continue;
1456 }
1457 }
1458 }
1459 }
1460 }
1461 out.push_str(token);
1462 }
1463 out
1464}
1465
1466/// Split `text` into tokens that, concatenated, reproduce it exactly —
1467/// alternating runs of non-whitespace and whitespace. Used by
1468/// [`expand_instruction_imports`] so `@import` tokens can be located and
1469/// replaced without disturbing surrounding formatting/whitespace.
1470fn split_preserving_whitespace(text: &str) -> Vec<&str> {
1471 let mut out = Vec::new();
1472 let mut start = 0;
1473 let mut in_ws = None;
1474 for (i, c) in text.char_indices() {
1475 let ws = c.is_whitespace();
1476 match in_ws {
1477 None => in_ws = Some(ws),
1478 Some(prev) if prev != ws => {
1479 out.push(&text[start..i]);
1480 start = i;
1481 in_ws = Some(ws);
1482 }
1483 _ => {}
1484 }
1485 }
1486 if start < text.len() {
1487 out.push(&text[start..]);
1488 }
1489 out
1490}
1491
1492/// BP-4 (catalog:87 "Instruction-file hygiene controls", cc§2
1493/// `claudeMdExcludes`): whether `path` is excluded from instruction loading
1494/// by [`Config::project_doc_excludes`]. A pattern matches when it
1495/// [`crate::config::glob_match`]es the file's NAME (`CLAUDE.md`), its full
1496/// path, or its path relative to `root` — the three spellings cc's own
1497/// "glob/absolute-path list" accepts. Empty (the default) excludes nothing.
1498fn instruction_file_excluded(
1499 config: &Config,
1500 path: &std::path::Path,
1501 root: &std::path::Path,
1502) -> bool {
1503 if config.project_doc_excludes.is_empty() {
1504 return false;
1505 }
1506 let full = path.to_string_lossy().to_string();
1507 let name = path
1508 .file_name()
1509 .map(|n| n.to_string_lossy().to_string())
1510 .unwrap_or_default();
1511 let rel = path
1512 .strip_prefix(root)
1513 .ok()
1514 .map(|p| p.to_string_lossy().to_string());
1515 config.project_doc_excludes.iter().any(|pat| {
1516 crate::config::glob_match(pat, &full)
1517 || crate::config::glob_match(pat, &name)
1518 || rel
1519 .as_deref()
1520 .is_some_and(|r| crate::config::glob_match(pat, r))
1521 })
1522}
1523
1524/// BP-4 (catalog:87, cc§2 "HTML comment stripping"): drop block-level
1525/// `<!-- … -->` spans from an instruction file's text so maintainer notes
1526/// cost no tokens, exactly as cc does before injection. Unterminated
1527/// openers drop the remainder (the same reading a markdown renderer takes).
1528/// Off by default ([`Config::project_doc_strip_comments`]) — cx does NOT
1529/// strip, so this is a per-preset hygiene lever, not a universal one.
1530fn strip_html_comments(text: &str) -> String {
1531 let mut out = String::with_capacity(text.len());
1532 let mut rest = text;
1533 while let Some(open) = rest.find("<!--") {
1534 out.push_str(&rest[..open]);
1535 match rest[open..].find("-->") {
1536 Some(close) => rest = &rest[open + close + 3..],
1537 None => return out,
1538 }
1539 }
1540 out.push_str(rest);
1541 out
1542}
1543
1544/// P4b (§1.4): append one instruction file's (trimmed, import-expanded)
1545/// content to `blob` as a labeled section, exactly like the pre-P4b inline
1546/// loop did — a no-op when `path` doesn't exist or is empty (the common
1547/// case). `project_scoped` distinguishes the project tier (imports bounded
1548/// to `root`, canonicalized-and-contained — see
1549/// [`import_target_is_contained`]) from the global tier (imports
1550/// unrestricted, same trust level as the user's own machine — `root` is
1551/// unused in that case). `root` is normally `path`'s own parent (the tier
1552/// root `path` was discovered under, e.g. an ancestor of `cwd` or an
1553/// `additional_dirs` entry) — see [`assemble_project_instructions`]'s call
1554/// sites.
1555///
1556/// BP-4 adds the hygiene controls (catalog:87): the exclude list
1557/// ([`instruction_file_excluded`]), HTML-comment stripping
1558/// ([`strip_html_comments`]) and the [`InstructionBudget`] —
1559/// [`Config::project_doc_max_bytes`] spent INCREMENTALLY as files are
1560/// concatenated root→cwd, which is how cx's own cap works on its root-down
1561/// concat, rather than one chop at the end (that chop would silently eat
1562/// the trailing per-file notices it had just written).
1563fn append_instruction_file(
1564 blob: &mut String,
1565 config: &Config,
1566 path: &std::path::Path,
1567 root: &std::path::Path,
1568 label: &str,
1569 project_scoped: bool,
1570 budget: &mut InstructionBudget,
1571) {
1572 if budget.exhausted() || instruction_file_excluded(config, path, root) {
1573 return;
1574 }
1575 let Ok(text) = std::fs::read_to_string(path) else {
1576 return;
1577 };
1578 let stripped;
1579 let text = if config.project_doc_strip_comments {
1580 stripped = strip_html_comments(&text);
1581 stripped.trim()
1582 } else {
1583 text.trim()
1584 };
1585 if text.is_empty() {
1586 return;
1587 }
1588 let dir = path.parent().unwrap_or(std::path::Path::new("."));
1589 let mut content = if config.instruction_imports {
1590 expand_instruction_imports(text, dir, root, project_scoped, 0)
1591 } else {
1592 text.to_string()
1593 };
1594 if !budget.take(&mut content) {
1595 return;
1596 }
1597 blob.push_str(&format!("\n\n# {label}\n{content}"));
1598}
1599
1600/// Truncate `s` to at most `max` BYTES, backing off to the nearest char
1601/// boundary — shared by the per-file and aggregate instruction caps.
1602fn truncate_at_char_boundary(s: &mut String, max: usize) {
1603 let mut end = max;
1604 while end > 0 && !s.is_char_boundary(end) {
1605 end -= 1;
1606 }
1607 s.truncate(end);
1608}
1609
1610/// BP-4 (catalog:87 "Instruction-file hygiene controls", cx§2
1611/// `project_doc_max_bytes`): the instruction-content byte budget, spent as
1612/// files are concatenated root→cwd.
1613///
1614/// `None` (the default, and cc-parity's explicit `= 0`) is uncapped, so
1615/// [`Self::take`] is a no-op and assembly is byte-identical to a config
1616/// that never heard of the cap. With a cap set, each file is truncated to
1617/// whatever budget REMAINS (per-file notice), and once the budget is gone
1618/// the remaining files are skipped entirely (aggregate notice, emitted once
1619/// by [`Self::aggregate_notice`]) — the total instruction CONTENT can
1620/// therefore never exceed the cap, and the notices survive because nothing
1621/// chops the assembled blob afterwards.
1622struct InstructionBudget {
1623 remaining: Option<usize>,
1624 hit: bool,
1625}
1626
1627impl InstructionBudget {
1628 fn new(config: &Config) -> Self {
1629 InstructionBudget {
1630 remaining: config.project_doc_max_bytes,
1631 hit: false,
1632 }
1633 }
1634
1635 /// True once the cap has consumed the whole budget — later files are
1636 /// skipped rather than partially appended.
1637 fn exhausted(&self) -> bool {
1638 self.remaining == Some(0)
1639 }
1640
1641 /// Charge `content` against the budget, truncating it (and appending a
1642 /// per-file notice) when it doesn't fit. Returns whether anything is
1643 /// left to append.
1644 fn take(&mut self, content: &mut String) -> bool {
1645 let Some(remaining) = self.remaining else {
1646 return true;
1647 };
1648 if content.len() <= remaining {
1649 self.remaining = Some(remaining - content.len());
1650 return true;
1651 }
1652 self.hit = true;
1653 self.remaining = Some(0);
1654 if remaining == 0 {
1655 return false;
1656 }
1657 truncate_at_char_boundary(content, remaining);
1658 content.push_str("\n[supercode: file truncated at core.project_doc_max_bytes]");
1659 true
1660 }
1661
1662 /// The one aggregate notice, appended after assembly when the cap bound
1663 /// anywhere — the statement that the assembled block is not the whole
1664 /// instruction set.
1665 fn aggregate_notice(&self) -> &'static str {
1666 if self.hit {
1667 "\n\n[supercode: instruction content truncated at core.project_doc_max_bytes]"
1668 } else {
1669 ""
1670 }
1671 }
1672}
1673
1674/// BP-4 (catalog:81 "Project instruction files w/ directory walk"; cc§2
1675/// "Directory-walk loading", cx§2 "walk project root (git root) down to
1676/// cwd"): the ancestor chain instruction files are discovered on, ordered
1677/// OUTERMOST FIRST so the nearest directory wins precedence by appearing
1678/// last in the concatenated blob (the root→cwd ordering both inventories
1679/// document).
1680///
1681/// The walk starts at [`Config::cwd`] and climbs until it has included the
1682/// project root [`crate::config::project_root_for`] identifies (`.git` by
1683/// default — cx's `project_root_markers`, §3.1), or until the filesystem
1684/// root, whichever comes first. [`MAX_INSTRUCTION_WALK_DEPTH`] bounds it
1685/// unconditionally, so a marker-less path deep under `/` can never turn
1686/// prompt assembly into an unbounded stat storm.
1687pub(crate) fn instruction_walk_roots(config: &Config) -> Vec<std::path::PathBuf> {
1688 // BP-9's shared answer to "where does the project stop?" — the same
1689 // walk the `env_context` git probe and the CLI's `.supercode.toml`
1690 // discovery use, so one `project_root_markers` value cannot mean three
1691 // different things. `None` (no marker anywhere, or an empty list) means
1692 // no root was found, and the climb below then stops at the filesystem
1693 // root under `MAX_INSTRUCTION_WALK_DEPTH`.
1694 let root = crate::config::project_root_for(&config.cwd, &config.project_root_markers);
1695 let mut chain: Vec<std::path::PathBuf> = Vec::new();
1696 let mut dir = config.cwd.clone();
1697 loop {
1698 let at_root = root.as_deref() == Some(dir.as_path());
1699 chain.push(dir.clone());
1700 if at_root || chain.len() >= MAX_INSTRUCTION_WALK_DEPTH {
1701 break;
1702 }
1703 match dir.parent() {
1704 Some(parent) if parent != dir => dir = parent.to_path_buf(),
1705 _ => break,
1706 }
1707 }
1708 chain.reverse();
1709 chain
1710}
1711
1712/// Hard bound on [`instruction_walk_roots`]'s ancestor climb.
1713const MAX_INSTRUCTION_WALK_DEPTH: usize = 64;
1714
1715/// P4b (§1.4, obligation 4 assembly site): the full instruction-file blob —
1716/// global/user tier (catalog §4a "Global/user-level instruction file tier")
1717/// FIRST, then the project tier — capped by
1718/// [`Config::project_doc_max_bytes`] if set (catalog §4a "hygiene caps
1719/// (`project_doc_max_bytes` analog)").
1720///
1721/// BP-4 (catalog:81): the project tier is no longer `cwd` alone. It is the
1722/// ANCESTOR WALK [`instruction_walk_roots`] returns (cwd's chain up to the
1723/// git root, outermost first) followed by `additional_dirs` — root-first
1724/// ordering throughout, so the nearest directory wins by appearing later,
1725/// which is exactly how both cc§2 ("concatenated root→cwd, closest read
1726/// last") and cx§2 ("nearer-to-cwd wins by appearing later") describe their
1727/// own walks. `cwd` is the last element of the walk chain, so a config
1728/// whose cwd IS the project root assembles byte-identically to the pre-BP-4
1729/// loop.
1730/// BP-5 (catalog D2 "Per-model-family base-prompt selection", cx§2
1731/// "Per-model base instructions": "the system prompt is selected per model
1732/// family from bundled markdown … the active `base_instructions` are
1733/// persisted verbatim into the rollout `session_meta`"): the base system
1734/// prompt for the model this config runs.
1735///
1736/// `[capabilities.model_catalog] base_prompts` maps a model-id glob to that
1737/// family's prompt; the most specific match wins
1738/// ([`crate::model_catalog::base_prompt_for`]). No table and no match both
1739/// give [`Config::system_prompt`] verbatim, so this is a no-op for every
1740/// config that does not set the table.
1741fn base_prompt_for_config(config: &Config) -> String {
1742 crate::model_catalog::base_prompt_for(&config.model_family_prompts, &config.model)
1743 .map(str::to_string)
1744 .unwrap_or_else(|| config.system_prompt.clone())
1745}
1746
1747fn assemble_project_instructions(config: &Config) -> String {
1748 let mut blob = String::new();
1749 let mut budget = InstructionBudget::new(config);
1750 let global_dir = global_instructions_dir();
1751 for name in ["CLAUDE.md", "AGENTS.md"] {
1752 append_instruction_file(
1753 &mut blob,
1754 config,
1755 &global_dir.join(name),
1756 // Global tier is trusted/unrestricted (project_scoped=false
1757 // below) — `root` is never consulted, but pass `global_dir`
1758 // rather than a bogus value for clarity.
1759 &global_dir,
1760 name,
1761 false,
1762 &mut budget,
1763 );
1764 }
1765 // BP-10 (catalog row "Project/workspace trust gate", cc§4/cx§4:
1766 // "Prompt before loading project-local config/code"): the PROJECT tier
1767 // is trust-gated. The global/user tier above is not — it is the user's
1768 // own machine, the same trust level as their shell, exactly as
1769 // `append_instruction_file`'s `project_scoped = false` argument
1770 // already says.
1771 //
1772 // `crate::trust::is_trusted` asks the `Config::trust_handler` door
1773 // once per project and records the answer; with no door installed
1774 // `TrustSurface::Instructions` resolves to LOADED, which is both the
1775 // pre-BP-10 behavior and what a headless run of either upstream
1776 // harness does — see `crate::trust`'s doc comment for why the
1777 // undecided answer differs between text and code.
1778 if !crate::trust::is_trusted(config, crate::trust::TrustSurface::Instructions) {
1779 blob.push_str(
1780 "\n[project instruction files were not loaded: this workspace is not trusted (capabilities.trust)]\n",
1781 );
1782 blob.push_str(budget.aggregate_notice());
1783 return blob;
1784 }
1785 let walk = instruction_walk_roots(config);
1786 for root in walk.iter().chain(config.additional_dirs.iter()) {
1787 for name in ["CLAUDE.md", "AGENTS.md"] {
1788 append_instruction_file(
1789 &mut blob,
1790 config,
1791 &root.join(name),
1792 // Project tier: `@`-imports from THIS file must stay under
1793 // THIS root (canonicalized) — see
1794 // `import_target_is_contained`.
1795 root,
1796 name,
1797 true,
1798 &mut budget,
1799 );
1800 }
1801 // A repository-native agent package is an additional project
1802 // instruction tier. It is subject to the same `project_context`
1803 // switch, import containment, and aggregate byte cap as root
1804 // AGENTS.md/CLAUDE.md; loading it never executes package code.
1805 for path in crate::agent_package::workspace_package_instruction_files(root) {
1806 append_instruction_file(
1807 &mut blob,
1808 config,
1809 &path,
1810 root,
1811 "Volter Harness agent package instructions",
1812 true,
1813 &mut budget,
1814 );
1815 }
1816 }
1817 blob.push_str(budget.aggregate_notice());
1818 blob
1819}
1820
1821/// P4b (§1.4/§3.1 `core.env_context`, catalog §4a "Environment context block
1822/// injection"): cwd, platform, date, and a best-effort git branch/dirty
1823/// status (silently absent when `cwd` isn't a git repo or `git` isn't on
1824/// `PATH` — never blocks agent construction).
1825///
1826/// BP-4 (catalog:90): plus the APPROVAL/SANDBOX POLICY line the row's own
1827/// semantics name ("cwd/git/platform/date/**policy**") and cx's
1828/// `<environment_context>` supplies — the model is told which approval mode
1829/// and which filesystem confinement it is operating under, which is what
1830/// makes "ask before you do X" instructions legible to it. The block is
1831/// re-derivable at any moment from `config` alone, which is what lets
1832/// [`Agent::refresh_env_context`] re-emit it mid-session on change.
1833/// BP-6 (catalog D2 "Skills (progressive-disclosure packages)", §1.4
1834/// obligation 4): the `# Skills` prompt section — the discovered SKILL.md
1835/// packages' names and descriptions, plus the `[core.prompts]` template
1836/// names, and NOTHING else. A skill's body is deliberately absent: it costs
1837/// its tokens only when something actually invokes it (`docs:skills`
1838/// "body loads only when used"; cx§7; pi§2 "progressive disclosure").
1839///
1840/// A skill whose frontmatter hides it from the model (`enabled: false`,
1841/// `disable-model-invocation: true`) is left OUT of the index while staying
1842/// user-invocable — cc§7 "Invocation control", pi§2.
1843///
1844/// Empty string when there is nothing to list, so an agent with neither
1845/// skills nor templates keeps the prompt it had before this existed.
1846fn skills_prompt_section(config: &Config, skills: &[crate::skills::LoopSkill]) -> String {
1847 let listed: Vec<&crate::skills::LoopSkill> = skills
1848 .iter()
1849 .filter(|skill| skill.model_invocable)
1850 .collect();
1851 let mut templates: Vec<&str> = config.prompts.keys().map(String::as_str).collect();
1852 templates.sort_unstable();
1853 if listed.is_empty() && templates.is_empty() {
1854 return String::new();
1855 }
1856 let mut out = String::from("\n\n# Skills\n");
1857 if !listed.is_empty() {
1858 out.push_str(
1859 "Installed skill packages. Only each skill's name and description are listed \
1860 here; call the `skill` tool with a name below to load that skill's full \
1861 instructions when it applies, then follow them.\n",
1862 );
1863 for skill in listed {
1864 out.push_str(&skill.index_line());
1865 out.push('\n');
1866 }
1867 }
1868 if !templates.is_empty() {
1869 if !out.ends_with("# Skills\n") {
1870 out.push('\n');
1871 }
1872 out.push_str("Prompt templates (invoke via `/name args`):\n");
1873 for name in templates {
1874 out.push_str(&format!("- {name}\n"));
1875 }
1876 }
1877 out
1878}
1879
1880fn env_context_block(config: &Config) -> String {
1881 let mut lines = vec![
1882 format!("cwd: {}", config.cwd.display()),
1883 format!("platform: {}", std::env::consts::OS),
1884 format!(
1885 "date: {}",
1886 supercode_interchange::sidecar::now_rfc3339()
1887 .get(..10)
1888 .unwrap_or("")
1889 ),
1890 format!(
1891 "approval policy: {} · sandbox: {}",
1892 approval_policy_label(config.approval),
1893 sandbox_policy_label(config.sandbox),
1894 ),
1895 ];
1896 // BP-9 (§3.1 `core.project_root_markers`, catalog:232): the git probe
1897 // runs at the PROJECT ROOT the markers define, not at whatever
1898 // subdirectory the process happens to sit in — the marker knob's whole
1899 // job is deciding where "the project" starts. Falls back to `cwd` when
1900 // no ancestor carries a marker (or the list is empty), which is
1901 // byte-identical to the pre-BP-9 behavior.
1902 let root = crate::config::project_root_for(&config.cwd, &config.project_root_markers)
1903 .unwrap_or_else(|| config.cwd.clone());
1904 if let Some(status) = env_context_git_status(&root) {
1905 lines.push(status);
1906 }
1907 format!("\n\n# Environment\n{}", lines.join("\n"))
1908}
1909
1910/// The `[capabilities.permissions] approval` spelling of a policy — the same
1911/// token the config schema accepts (`configfile::parse_approval_str`), so
1912/// the block reports the policy in the vocabulary the user configured it in.
1913fn approval_policy_label(policy: crate::config::ApprovalPolicy) -> &'static str {
1914 match policy {
1915 crate::config::ApprovalPolicy::Never => "never",
1916 crate::config::ApprovalPolicy::OnRequest => "on-request",
1917 crate::config::ApprovalPolicy::Untrusted => "untrusted",
1918 crate::config::ApprovalPolicy::ModelRequested => "model-requested",
1919 }
1920}
1921
1922/// The `[capabilities.permissions] sandbox` spelling of a tier — see
1923/// [`approval_policy_label`].
1924fn sandbox_policy_label(policy: crate::tools::SandboxPolicy) -> &'static str {
1925 match policy {
1926 crate::tools::SandboxPolicy::ReadOnly => "read-only",
1927 crate::tools::SandboxPolicy::WorkspaceWrite => "workspace-write",
1928 crate::tools::SandboxPolicy::DangerFullAccess => "danger-full-access",
1929 }
1930}
1931
1932/// Best-effort `git branch (dirty|clean)` for [`env_context_block`]. `None`
1933/// on anything short of a clean success (not a repo, `git` missing, a
1934/// detached/errored state) — this is informational context, never worth
1935/// failing agent construction over.
1936fn env_context_git_status(cwd: &std::path::Path) -> Option<String> {
1937 let branch_out = std::process::Command::new("git")
1938 .args(["rev-parse", "--abbrev-ref", "HEAD"])
1939 .current_dir(cwd)
1940 .output()
1941 .ok()?;
1942 if !branch_out.status.success() {
1943 return None;
1944 }
1945 let branch = String::from_utf8_lossy(&branch_out.stdout)
1946 .trim()
1947 .to_string();
1948 if branch.is_empty() {
1949 return None;
1950 }
1951 let dirty = std::process::Command::new("git")
1952 .args(["status", "--porcelain"])
1953 .current_dir(cwd)
1954 .output()
1955 .ok()
1956 .map(|o| !o.stdout.is_empty())
1957 .unwrap_or(false);
1958 Some(format!(
1959 "git branch: {branch} ({})",
1960 if dirty { "dirty" } else { "clean" }
1961 ))
1962}
1963
1964impl Agent {
1965 /// Build an agent backed by an OpenAI-compatible endpoint (OpenRouter by
1966 /// default). The API key is taken from [`Config::api_key`], then
1967 /// [`Config::api_key_cmd`] (P4: a credential-helper command, run via the
1968 /// shell — see `run_api_key_cmd`), then the configured environment
1969 /// variable ([`Config::api_key_env`]).
1970 pub fn new(config: Config) -> Result<Self> {
1971 let api_key = match &config.api_key {
1972 Some(k) if !k.is_empty() => k.clone(),
1973 // BP-9: the ARGV helper (`core.api_key_command`) is consulted
1974 // first — it is the form with no shell in the path, so a config
1975 // that sets both gets the one with fewer ways to surprise its
1976 // author. Empty/failed → fall through, same as `api_key_cmd`.
1977 _ => match config
1978 .api_key_command
1979 .as_deref()
1980 .filter(|argv| !argv.is_empty())
1981 .map(run_api_key_command)
1982 .filter(|k| !k.is_empty())
1983 .or_else(|| {
1984 config
1985 .api_key_cmd
1986 .as_deref()
1987 .filter(|c| !c.is_empty())
1988 .map(run_api_key_cmd)
1989 }) {
1990 // P4 (§1.8 credential-helper indirection, D6 row): the
1991 // helper ran and produced a non-empty key — use it. A
1992 // failed/empty helper falls through to `api_key_env` rather
1993 // than erroring outright, same "try the next source"
1994 // posture as every other layer in this resolution chain.
1995 Some(k) if !k.is_empty() => k,
1996 _ => std::env::var(&config.api_key_env)
1997 .ok()
1998 .filter(|k| !k.is_empty())
1999 .ok_or_else(|| Error::MissingApiKey(config.api_key_env.clone()))?,
2000 },
2001 };
2002 // P4b (§1.1/§3.1 `core.retry`, pi§3 shape): `Config.retry_*` now
2003 // reaches the pre-existing transport-layer retry mechanism (see
2004 // `provider::HttpOptions::from_retry_config`'s doc comment for the
2005 // exact "byte-identical when unset" contract).
2006 let http_options = provider::HttpOptions::from_retry_config(
2007 config.retry_enabled,
2008 config.retry_max_retries,
2009 config.retry_base_delay_ms,
2010 );
2011 // BP-7 (catalog §4a "Turn/budget caps"): a spend cap armed against
2012 // a model this build cannot price is refused HERE rather than
2013 // accepted and silently never enforced. See
2014 // `Config::max_budget_usd`.
2015 if config.max_budget_usd.is_some_and(|b| b > 0.0)
2016 && crate::pricing::resolve(
2017 &config.model,
2018 config.price_input_per_mtok,
2019 config.price_output_per_mtok,
2020 )
2021 .is_none()
2022 {
2023 return Err(Error::UnpriceableBudget {
2024 model: config.model.clone(),
2025 });
2026 }
2027 // BP-7 (catalog §4a "Auto-retry on transient provider errors"): the
2028 // shared log the transport's retry loop reports into and
2029 // `Self::run_loop` drains after every completion.
2030 let retry_log = std::sync::Arc::new(crate::provider::RetryLog::default());
2031 let provider = OpenAiProvider::new_with_options(
2032 config.base_url.clone(),
2033 api_key,
2034 config.extra_headers.clone(),
2035 http_options,
2036 )
2037 .with_retry_log(retry_log.clone());
2038 // P3 (design §5.2): `ToolRegistry::from_config` replaces the
2039 // unconditional `with_builtins()` call — a no-op when
2040 // `config.module_registry` is off (the default, §5.3 risk 2).
2041 let registry = ToolRegistry::from_config(&config);
2042 let mut agent = Self::with_parts(config, Box::new(provider), registry);
2043 agent.retry_log = retry_log;
2044 Ok(agent)
2045 }
2046
2047 /// Build an agent with an explicit provider and the built-in tools. Handy
2048 /// for tests (inject a mock provider) or custom transports.
2049 pub fn with_provider(config: Config, provider: Box<dyn Provider>) -> Self {
2050 let registry = ToolRegistry::from_config(&config);
2051 Self::with_parts(config, provider, registry)
2052 }
2053
2054 /// Build an agent from all three parts.
2055 pub fn with_parts(
2056 mut config: Config,
2057 provider: Box<dyn Provider>,
2058 mut registry: ToolRegistry,
2059 ) -> Self {
2060 // P5-12 (§2 module 18 `plugins`, D-10): register every trusted,
2061 // loaded plugin's declared tools — the same "unconditional, config-
2062 // gated" wiring `build_tool_context` just below gives
2063 // checkpoint/formatters/lsp. `crate::plugins::register_into` is a
2064 // true no-op (no filesystem read, no subprocess) whenever
2065 // `config.plugins_enabled` is `false` (the default) — byte-identical
2066 // to before this module existed. Runs here (the one tail every
2067 // `Agent` construction path funnels through — `new`/`with_provider`
2068 // both call this) rather than in `ToolRegistry::from_config`, so it
2069 // is NOT entangled with that function's unrelated `module_registry`
2070 // experimental gate.
2071 crate::plugins::register_into(&config, &mut registry);
2072 let (mut ctx, checkpoint_observer, lsp_manager) = build_tool_context(&config);
2073 // Auto-load project context files (CLAUDE.md / AGENTS.md) from the
2074 // working directory (and any extra roots), appending them to the system
2075 // prompt — the analog of how Claude Code / Codex discover them.
2076 // P4b (§1.4): also the global/user tier + instruction imports + the
2077 // `project_doc_max_bytes` hygiene cap — see `assemble_project_instructions`.
2078 // BP-5 (catalog D2 "Per-model-family base-prompt selection", cx§2
2079 // "Per-model base instructions"): the base prompt is chosen for the
2080 // model in force, not fixed before the model is known — the exact
2081 // residue the ledger row named. `base_prompt_for_config` is
2082 // `config.system_prompt` verbatim for every config that sets no
2083 // family table, so this is a no-op by default.
2084 let base_prompt_live = base_prompt_for_config(&config);
2085 // BP-5 (catalog D2 "Output style / personality module"): a custom
2086 // style may REPLACE the base coding instructions rather than append
2087 // to them (cc§7 `keep-coding-instructions`); every other style is
2088 // appended at the end of the assembled prompt, below.
2089 let output_style = crate::output_style::resolve(&config);
2090 let mut system = match output_style.as_ref() {
2091 Some(style) if style.replaces_base => style.text.clone(),
2092 _ => base_prompt_live.clone(),
2093 };
2094 if config.load_project_context {
2095 system.push_str(&assemble_project_instructions(&config));
2096 }
2097 // BP-5 (catalog D2 "Path-scoped rules", cc§2 `.claude/rules`): the
2098 // UNSCOPED rules join the instruction blob here. A rule carrying a
2099 // `paths:` selector deliberately does not — it waits for a tool to
2100 // touch a matching file (`tools::builtins::path_rules_notice`).
2101 let path_rules = crate::path_rules::load(&config);
2102 system.push_str(&crate::path_rules::always_on_text(&path_rules));
2103 // P4b (§1.4/§3.1 `core.env_context`, catalog §4a "Environment
2104 // context block injection"): `false` (the default) is a no-op —
2105 // byte-identical to today's behavior. BP-4 keeps the rendered block
2106 // on the agent (`env_context_live`) so `refresh_env_context` can
2107 // find and REPLACE exactly this text when cwd/policy/branch move,
2108 // rather than leaving a stale block in the prompt forever.
2109 let env_context_live = if config.env_context {
2110 let block = env_context_block(&config);
2111 system.push_str(&block);
2112 Some(block)
2113 } else {
2114 None
2115 };
2116 // P4e (§1.4/§3.1 `core.context_injections`, catalog:91 "Synthetic
2117 // context-injection blocks"): same assembly site, right after
2118 // `env_context`. `false` (the default) is a no-op — byte-identical
2119 // to today's behavior. BP-4 routes it through
2120 // `crate::context_injection`, so the gate now delivers the built-in
2121 // ambient blocks the row is about (and stays extensible at runtime
2122 // through `Self::inject_context_block`) instead of only whatever
2123 // static list an embedder happened to populate.
2124 system.push_str(&crate::context_injection::assemble(&config, &[]));
2125 // P3 (design §5.2, §1.4 obligation 4, D-7): the skills prompt
2126 // section is a MODULE-GATED prompt section, the design's own
2127 // illustration of "a disabled module contributes no prompt
2128 // sections" — only assembled at all under
2129 // `[experimental] module_registry = true` (§5.3 risk 2: flag-off is
2130 // byte-for-byte today's behavior, and today's behavior never emits
2131 // this section, since it doesn't exist pre-P3). Gated further by
2132 // D-7 itself: `core.skills` requires a read pathway (`read_file` or
2133 // `bash`) — absent either, no section is appended, matching the
2134 // hard-dependency shape `configfile::validate_modules` enforces at
2135 // resolve time.
2136 //
2137 // BP-6 (catalog D2 "Skills (progressive-disclosure packages)"): the
2138 // section is now the discovered SKILL.md INDEX — each package's
2139 // frontmatter `name` and `description`, nothing else. A body is
2140 // never assembled here; it is read on invocation only, which is
2141 // what "progressive disclosure" means. The `[core.prompts]`
2142 // template names keep their own sub-list below it.
2143 let skills = crate::skills::load_for_config(&config);
2144 if config.module_registry && config.skills_enabled {
2145 let has_read_pathway = config
2146 .core_tools_enabled
2147 .iter()
2148 .any(|t| t == "read_file" || t == "bash");
2149 if has_read_pathway {
2150 system.push_str(&skills_prompt_section(&config, &skills));
2151 }
2152 }
2153 // BP-5: the style layer lands LAST, where cc puts it ("output styles
2154 // append custom instructions to the END of the system prompt").
2155 // Empty for a neutral style (`default`/`none`) and for a style that
2156 // already replaced the base above.
2157 if let Some(style) = output_style.as_ref().filter(|s| !s.replaces_base) {
2158 system.push_str(&style.section());
2159 }
2160 // BP-5: the SCOPED rules travel with the tool context, which is
2161 // where a "a tool touched a matching file" event can see them.
2162 ctx.path_rules = std::sync::Arc::new(path_rules);
2163 let history = vec![ChatMessage::system(system)];
2164 // P4e (§1.6/§3.1 `core.session.git_metadata`, catalog:331): captured
2165 // once here, alongside `env_context`'s own git probe — `false` (the
2166 // default) is a no-op, byte-identical to today's behavior.
2167 let git_metadata = if config.session_git_metadata {
2168 crate::git_metadata::capture(&config.cwd, now_ms())
2169 } else {
2170 None
2171 };
2172 // BP-7 (catalog §4a "Named agent definitions as data"): discover
2173 // `<cwd>/.claude/agents/*.md` for EVERY harness that turns the
2174 // subagents module on, not only the Claude emulate/resume path —
2175 // that restriction was the second half of the ledger row's
2176 // residue.
2177 merge_project_agent_definitions(&mut config);
2178 // P5-3: captured before `config` moves into the literal below (a
2179 // `usize` field READ, not a move, but it must happen before the
2180 // `config` shorthand field consumes the binding).
2181 let subagent_depth = config.subagent_depth;
2182 // BP-5 (catalog D2 "Shell-output injection in templates/skills"):
2183 // the authorization every `` !`cmd` `` in a skill/command body is
2184 // evaluated under — this config's own permission rules, resolved
2185 // once. Disabled unless `[core.skills] shell_injection` is on.
2186 let shell_injection = crate::skills::ShellInjection::from_config(&config);
2187 // BP-8 (catalog:151): `[capabilities.session_tree] enabled` finally
2188 // has a reader. An armed tree starts empty and grows one node per
2189 // recorded message — the degenerate single-path case, byte-for-byte
2190 // the same conversation, until a rewind or branch actually forks it.
2191 let session_tree = if config.session_tree_enabled {
2192 Some(supercode_interchange::session_tree::SessionTree::new())
2193 } else {
2194 None
2195 };
2196 // BP-7: resolved once here so the request path never re-does the
2197 // lookup, and so `Self::model_price` is `None` exactly when this
2198 // build cannot price the model.
2199 let model_price = crate::pricing::resolve(
2200 &config.model,
2201 config.price_input_per_mtok,
2202 config.price_output_per_mtok,
2203 );
2204 // BP-10: same reason — built before `config` moves into the
2205 // literal. `Config::permissions_approvals_persist` off (the
2206 // default) makes this the pre-BP-10 in-memory cache and touches no
2207 // filesystem.
2208 let permissions_approval_cache = crate::permissions::cache_for_config(&config);
2209 Agent {
2210 config,
2211 provider: std::sync::Arc::from(provider),
2212 registry,
2213 history,
2214 ctx,
2215 total_output_tokens: 0,
2216 activated_tools: HashSet::new(),
2217 recorder: None,
2218 journal: None,
2219 session_tree,
2220 rewind_undo: Vec::new(),
2221 journaled_plan: Vec::new(),
2222 reduction_policy: None,
2223 reduction_log: ReductionLog::default(),
2224 imported_prefix_len: None,
2225 compacting_manually: false,
2226 env_context_live,
2227 base_prompt_live,
2228 shell_injection,
2229 spliced_context_blocks: Vec::new(),
2230 span_summarizer: None,
2231 last_tool_schema_tier_signature: None,
2232 context_limit: None,
2233 requests_issued: false,
2234 last_cache_activity_ms: None,
2235 cache_established: false,
2236 pending_cache_turn: (false, false, None),
2237 session_titler: None,
2238 usage_log: Vec::new(),
2239 turn_index: 0,
2240 turn_records: Vec::new(),
2241 retry_log: std::sync::Arc::new(crate::provider::RetryLog::default()),
2242 model_price,
2243 total_cost_usd: 0.0,
2244 total_steps: 0,
2245 reaped_subagents: std::collections::HashMap::new(),
2246 goal: None,
2247 steer_queue: std::sync::Arc::new(std::sync::Mutex::new(SteerInbox::default())),
2248 follow_up_queue: std::collections::VecDeque::new(),
2249 doom_loop_last_call: None,
2250 doom_loop_streak: 0,
2251 model_change_log: Vec::new(),
2252 git_metadata,
2253 permissions_approval_cache,
2254 permissions_approval_handler: None,
2255 mcp_prompts: std::collections::HashMap::new(),
2256 skills,
2257 subagent_depth,
2258 subagent_concurrency_gauge: std::sync::Arc::new(std::sync::atomic::AtomicUsize::new(0)),
2259 background_subagents: std::collections::HashMap::new(),
2260 pending_child_approvals: std::sync::Arc::new(std::sync::Mutex::new(Vec::new())),
2261 child_approval_handler_factory: None,
2262 subagent_store: None,
2263 claude_runtime_manifest: None,
2264 background_jobs: std::collections::HashMap::new(),
2265 background_concurrency_gauge: std::sync::Arc::new(std::sync::atomic::AtomicUsize::new(
2266 0,
2267 )),
2268 checkpoint_observer,
2269 lsp_manager,
2270 }
2271 }
2272
2273 /// A handle to this agent's model transport, for sharing with subagents.
2274 pub fn provider_arc(&self) -> std::sync::Arc<dyn Provider> {
2275 self.provider.clone()
2276 }
2277
2278 /// Read-only access to this agent's resolved [`Config`] — e.g. so a
2279 /// caller (`crates/cli`'s `attach_mcp`) can consult
2280 /// [`Config::module_registry`]/[`Config::module_activation`] AFTER
2281 /// construction without having to separately thread the config through
2282 /// every call site that builds an `Agent` and later needs it again.
2283 /// Same trust boundary as every other already-public `Agent` accessor
2284 /// (`history`, `provider_arc`) — the caller is the same process that
2285 /// built this `Config` in the first place, not a new exposure surface.
2286 pub fn config(&self) -> &Config {
2287 &self.config
2288 }
2289
2290 /// P5-9 (§2 module 20 `checkpoint`): this agent's checkpoint engine, if
2291 /// `Config::checkpoint_enabled` is `true` and the shadow store opened
2292 /// successfully — `None` otherwise (the default-off case, or a
2293 /// graceful-degrade after an I/O failure). A caller (CLI/TUI/embedder)
2294 /// uses this to `list`/`turn_diff`/`restore` WITHOUT re-deriving the
2295 /// shadow-store root itself. Deliberately named `checkpoint_observer`,
2296 /// not `checkpoint` — [`Self::checkpoint`] already names the unrelated
2297 /// in-memory conversation-position marker (see that method's doc
2298 /// comment).
2299 pub fn checkpoint_observer(&self) -> Option<&crate::checkpoint::CheckpointObserver> {
2300 self.checkpoint_observer.as_deref()
2301 }
2302
2303 /// P5-11 (§2 module 28 `lsp`): this agent's LSP server registry, if
2304 /// `Config::lsp_enabled` is `true` — `None` otherwise (the default-off
2305 /// case). `impl Drop for Agent` already covers production teardown via
2306 /// [`crate::lsp::LspManager::kill_all_sync`] (a real, group-killing OS
2307 /// process kill — see `crate::lsp`'s module doc). This accessor exists
2308 /// for an OPTIONAL caller (CLI/TUI/embedder) that manages its own
2309 /// `Agent` lifecycle and additionally wants to reach
2310 /// [`crate::lsp::LspManager::shutdown_all`] for a graceful LSP
2311 /// `shutdown`/`exit` handshake BEFORE dropping the agent — nothing
2312 /// calls `shutdown_all` automatically today.
2313 pub fn lsp_manager(&self) -> Option<&crate::lsp::LspManager> {
2314 self.lsp_manager.as_deref()
2315 }
2316
2317 /// Spawn a subagent that shares this agent's model transport, runs `task`
2318 /// to completion with its own fresh conversation (seeded with `system`), and
2319 /// returns its final answer. The analog of `Agent` / `spawn_agent`.
2320 pub async fn run_subagent(
2321 &self,
2322 system: impl Into<String>,
2323 task: impl Into<String>,
2324 ) -> Result<String> {
2325 let mut sub_config = Config::builder()
2326 .model(self.config.model.clone())
2327 .system_prompt(system)
2328 .cwd(self.config.cwd.clone())
2329 .sandbox(self.config.sandbox)
2330 .max_iterations(self.config.max_iterations)
2331 .build();
2332 sub_config.base_url = self.config.base_url.clone();
2333 let mut sub = Agent::with_provider_arc(sub_config, self.provider.clone());
2334 sub.send(task).await
2335 }
2336
2337 /// Like [`Self::with_provider`] but sharing an existing transport handle.
2338 pub fn with_provider_arc(mut config: Config, provider: std::sync::Arc<dyn Provider>) -> Self {
2339 let (ctx, checkpoint_observer, lsp_manager) = build_tool_context(&config);
2340 let history = vec![ChatMessage::system(config.system_prompt.clone())];
2341 // P3 (design §5.2): see the `Self::new` doc note — a no-op when
2342 // `config.module_registry` is off (the default).
2343 let mut registry = ToolRegistry::from_config(&config);
2344 // P5-12: see `Self::with_parts`'s identical call — a no-op when
2345 // `config.plugins_enabled` is `false` (the default).
2346 crate::plugins::register_into(&config, &mut registry);
2347 // P4e: see `Self::with_parts`'s identical capture.
2348 let git_metadata = if config.session_git_metadata {
2349 crate::git_metadata::capture(&config.cwd, now_ms())
2350 } else {
2351 None
2352 };
2353 // BP-7 (catalog §4a "Named agent definitions as data"): discover
2354 // `<cwd>/.claude/agents/*.md` for EVERY harness that turns the
2355 // subagents module on, not only the Claude emulate/resume path —
2356 // that restriction was the second half of the ledger row's
2357 // residue.
2358 merge_project_agent_definitions(&mut config);
2359 // P5-3: captured before `config` moves into the literal below (a
2360 // `usize` field READ, not a move, but it must happen before the
2361 // `config` shorthand field consumes the binding).
2362 let subagent_depth = config.subagent_depth;
2363 // BP-6: this constructor assembles no prompt sections at all (it
2364 // takes `config.system_prompt` verbatim), so there is no skills
2365 // INDEX here — but the discovered set still rides along, so an
2366 // explicit invocation (`/name`, `$slug`, the `skill` tool) resolves
2367 // the same packages the registry's own `skill` tool holds.
2368 let skills = crate::skills::load_for_config(&config);
2369 let base_prompt_live = config.system_prompt.clone();
2370 let shell_injection = crate::skills::ShellInjection::from_config(&config);
2371 // BP-8 (catalog:151): `[capabilities.session_tree] enabled` finally
2372 // has a reader. An armed tree starts empty and grows one node per
2373 // recorded message — the degenerate single-path case, byte-for-byte
2374 // the same conversation, until a rewind or branch actually forks it.
2375 let session_tree = if config.session_tree_enabled {
2376 Some(supercode_interchange::session_tree::SessionTree::new())
2377 } else {
2378 None
2379 };
2380 // BP-7: resolved once here so the request path never re-does the
2381 // lookup, and so `Self::model_price` is `None` exactly when this
2382 // build cannot price the model.
2383 let model_price = crate::pricing::resolve(
2384 &config.model,
2385 config.price_input_per_mtok,
2386 config.price_output_per_mtok,
2387 );
2388 // BP-10: see the sibling constructor — built before `config` moves.
2389 let permissions_approval_cache = crate::permissions::cache_for_config(&config);
2390 Agent {
2391 config,
2392 provider,
2393 registry,
2394 history,
2395 ctx,
2396 total_output_tokens: 0,
2397 activated_tools: HashSet::new(),
2398 recorder: None,
2399 journal: None,
2400 session_tree,
2401 rewind_undo: Vec::new(),
2402 journaled_plan: Vec::new(),
2403 reduction_policy: None,
2404 reduction_log: ReductionLog::default(),
2405 imported_prefix_len: None,
2406 compacting_manually: false,
2407 env_context_live: None,
2408 // BP-5: this constructor assembles no prompt sections (see the
2409 // skills note above) — `config.system_prompt` IS the whole
2410 // system message, so that is what a later `set_model` would
2411 // have to replace.
2412 base_prompt_live,
2413 shell_injection,
2414 spliced_context_blocks: Vec::new(),
2415 span_summarizer: None,
2416 last_tool_schema_tier_signature: None,
2417 context_limit: None,
2418 requests_issued: false,
2419 last_cache_activity_ms: None,
2420 cache_established: false,
2421 pending_cache_turn: (false, false, None),
2422 session_titler: None,
2423 usage_log: Vec::new(),
2424 turn_index: 0,
2425 turn_records: Vec::new(),
2426 retry_log: std::sync::Arc::new(crate::provider::RetryLog::default()),
2427 model_price,
2428 total_cost_usd: 0.0,
2429 total_steps: 0,
2430 reaped_subagents: std::collections::HashMap::new(),
2431 goal: None,
2432 steer_queue: std::sync::Arc::new(std::sync::Mutex::new(SteerInbox::default())),
2433 follow_up_queue: std::collections::VecDeque::new(),
2434 doom_loop_last_call: None,
2435 doom_loop_streak: 0,
2436 model_change_log: Vec::new(),
2437 git_metadata,
2438 permissions_approval_cache,
2439 permissions_approval_handler: None,
2440 mcp_prompts: std::collections::HashMap::new(),
2441 skills,
2442 subagent_depth,
2443 subagent_concurrency_gauge: std::sync::Arc::new(std::sync::atomic::AtomicUsize::new(0)),
2444 background_subagents: std::collections::HashMap::new(),
2445 pending_child_approvals: std::sync::Arc::new(std::sync::Mutex::new(Vec::new())),
2446 child_approval_handler_factory: None,
2447 subagent_store: None,
2448 claude_runtime_manifest: None,
2449 background_jobs: std::collections::HashMap::new(),
2450 background_concurrency_gauge: std::sync::Arc::new(std::sync::atomic::AtomicUsize::new(
2451 0,
2452 )),
2453 checkpoint_observer,
2454 lsp_manager,
2455 }
2456 }
2457
2458 /// Run a prompt on a background task, returning a handle that resolves to
2459 /// the final answer (and the agent, so the caller can continue it). The
2460 /// analog of background/async agent runs.
2461 pub fn run_in_background(
2462 mut self,
2463 prompt: impl Into<String>,
2464 ) -> tokio::task::JoinHandle<(Self, Result<String>)>
2465 where
2466 Self: Send + 'static,
2467 {
2468 let prompt = prompt.into();
2469 tokio::spawn(async move {
2470 let result = self.send(prompt).await;
2471 (self, result)
2472 })
2473 }
2474
2475 /// Build an agent and seed it with a previously-recorded [`Session`] so it
2476 /// can continue where Claude Code or Codex left off.
2477 pub fn resume(config: Config, session: Session) -> Result<Self> {
2478 let mut agent = Agent::new(config)?;
2479 agent.load_session(session);
2480 Ok(agent)
2481 }
2482
2483 /// Like [`Self::resume`], but also begins recording (A2/A3): a fresh
2484 /// native-v2 sidecar is created at `sidecar_path` from `session` (header +
2485 /// `session.raw` verbatim — the imported prefix's own fidelity), and every
2486 /// subsequent turn this agent produces is appended to it at full fidelity,
2487 /// independent of whatever `cap_tool_output`/`maybe_compact` (D6) do to
2488 /// `history`.
2489 ///
2490 /// Invariant this establishes ONLY once [`Self::set_reduction_policy`] is
2491 /// also called (the D6/A7 supersession gate, `Self::run_loop`): at any
2492 /// instant, `Session::from_native_str(sidecar).messages` equals
2493 /// `session.messages` (the imported prefix) followed by every message
2494 /// appended since — i.e. `self.history()[1..]` (`history[0]` is this
2495 /// agent's own system prompt, per [`Self::load_session`]; it is never
2496 /// part of `session` and is never written to the sidecar). Recording
2497 /// alone (no policy) leaves the gate off: `cap_tool_output` still runs on
2498 /// oversized tool results, and `history` can diverge from the sidecar for
2499 /// them — honestly, via the notice's "full output in session sidecar"
2500 /// label, never silently.
2501 pub fn resume_recorded(
2502 config: Config,
2503 session: Session,
2504 sidecar_path: &std::path::Path,
2505 ) -> Result<Self> {
2506 let mut agent = Agent::new(config)?;
2507 let recorder = SidecarWriter::create(sidecar_path, &session)?;
2508 agent.load_session(session);
2509 agent.recorder = Some(recorder);
2510 Ok(agent)
2511 }
2512
2513 /// Install (or replace) this agent's sidecar recorder (A3).
2514 pub fn set_recorder(&mut self, w: SidecarWriter) {
2515 self.recorder = Some(w);
2516 }
2517
2518 // ---- BP-8: the append-only journal (catalog:150/152/154/156) --------
2519
2520 /// Install (or replace) this agent's append-only session journal — the
2521 /// durable, flush-per-record log of every message it produces plus
2522 /// every queue/rewind/plan operation performed on it. Installing one
2523 /// alone changes nothing about the conversation; it only makes the
2524 /// session survive a crash mid-turn.
2525 pub fn set_journal(&mut self, journal: crate::session_journal::SessionJournal) {
2526 self.journal = Some(std::sync::Arc::new(std::sync::Mutex::new(journal)));
2527 }
2528
2529 /// Whether an append-only journal is installed.
2530 pub fn has_journal(&self) -> bool {
2531 self.journal.is_some()
2532 }
2533
2534 /// Append one operation to the journal, if installed. Best-effort by
2535 /// design: losing a durability record must never fail the turn it
2536 /// describes, so the failure is logged and the loop continues — the
2537 /// same contract the compaction-marker `record` call keeps.
2538 fn journal_op(&self, op: crate::session_journal::JournalOp) {
2539 let Some(journal) = &self.journal else { return };
2540 let mut guard = journal
2541 .lock()
2542 .unwrap_or_else(std::sync::PoisonError::into_inner);
2543 if let Err(error) = guard.append(op) {
2544 tracing::warn!("failed to append a session-journal record: {error}");
2545 }
2546 }
2547
2548 /// Declare the durable view caught up: `<name>.jsonl` now holds
2549 /// `messages` messages and every journal record before this point is
2550 /// already in it. Everything journaled AFTER the last such record is
2551 /// exactly what a crash would have lost — see
2552 /// [`crate::session_journal::JournalState::unpersisted`].
2553 pub fn journal_checkpoint(&self, messages: usize) {
2554 self.journal_op(crate::session_journal::JournalOp::Checkpoint { messages });
2555 }
2556
2557 /// BP-13: record one per-turn usage entry in the append-only journal.
2558 pub fn journal_usage(&self, record: &crate::usage_log::UsageRecord) {
2559 self.journal_op(crate::session_journal::JournalOp::Usage {
2560 record: record.clone(),
2561 });
2562 }
2563
2564 /// BP-13: record one mid-session model change in the append-only
2565 /// journal — the ONE persisted home for a routing record (BP-8's
2566 /// journal), never a second file.
2567 pub fn journal_model_change(&self, record: &crate::model_change::ModelChangeRecord) {
2568 self.journal_op(crate::session_journal::JournalOp::ModelChange {
2569 record: record.clone(),
2570 });
2571 }
2572
2573 /// BP-8 (catalog:151): this session's conversation tree, when the
2574 /// module is on.
2575 pub fn session_tree(&self) -> Option<&supercode_interchange::session_tree::SessionTree> {
2576 self.session_tree.as_ref()
2577 }
2578
2579 /// Install a tree loaded from the store (a resume), replacing whatever
2580 /// this agent built. A no-op when the module is off — a session whose
2581 /// preset does not enable `session_tree` must not acquire one through
2582 /// the back door of an old sidecar.
2583 pub fn set_session_tree(&mut self, tree: supercode_interchange::session_tree::SessionTree) {
2584 if self.config.session_tree_enabled {
2585 self.session_tree = Some(tree);
2586 }
2587 }
2588
2589 /// BP-8: rebuild the tree from the current linear history — used after
2590 /// a resume that loaded a transcript but had no `.tree.json` to restore
2591 /// (every session recorded before the module was on).
2592 pub fn rebuild_session_tree_from_history(&mut self) {
2593 if !self.config.session_tree_enabled {
2594 return;
2595 }
2596 let linear: Vec<ChatMessage> = self.history.iter().skip(1).cloned().collect();
2597 self.session_tree =
2598 Some(supercode_interchange::session_tree::SessionTree::from_linear(&linear, now_ms()));
2599 }
2600
2601 /// BP-8 (catalog:152 "Rewind/rollback conversation"): move THIS
2602 /// conversation back to an earlier point — the whole row, not the
2603 /// last-exchange special case [`Self::rewind_to`] serves and not
2604 /// `sessions fork --at`, which makes a different session.
2605 ///
2606 /// `keep` is a message count (index into `history`), so `keep = 1`
2607 /// leaves only the system message. Three things happen, in this order:
2608 ///
2609 /// 1. the removed tail is pushed onto an undo stack, so
2610 /// [`Self::undo_rewind`] can put it back;
2611 /// 2. a [`crate::session_journal::JournalOp::Rewind`] record is
2612 /// APPENDED — nothing is deleted from disk, so the rewound-away
2613 /// messages remain recoverable from the log;
2614 /// 3. when the tree module is on, the active branch's leaf moves to the
2615 /// node at `keep`, and the old leaf is preserved under a fresh
2616 /// sibling branch — the next message appended forks there rather
2617 /// than overwriting.
2618 ///
2619 /// Returns what it did. Rewinding to a point at or past the end is a
2620 /// no-op with `removed = 0`, never an error.
2621 pub fn rewind_conversation(&mut self, keep: usize) -> RewindOutcome {
2622 let keep = keep.max(1).min(self.history.len());
2623 let removed: Vec<ChatMessage> = self.history.split_off(keep);
2624 if removed.is_empty() {
2625 return RewindOutcome {
2626 kept: self.history.len(),
2627 removed: 0,
2628 preserved_branch: None,
2629 };
2630 }
2631 let removed_count = removed.len();
2632 self.rewind_undo.push(removed);
2633 // `keep` counts the system message; the journal records only
2634 // `history[1..]`, so its own view is one shorter.
2635 self.journal_op(crate::session_journal::JournalOp::Rewind { to: keep - 1 });
2636 let preserved_branch = self.session_tree.as_mut().and_then(|tree| {
2637 let path = tree.active_path().unwrap_or_default();
2638 // `keep - 1` messages remain after the system message, so the
2639 // new leaf is the node at index `keep - 2`.
2640 match keep.checked_sub(2).and_then(|i| path.get(i).cloned()) {
2641 Some(node) => tree.rewind(&node, now_ms()).ok().flatten(),
2642 None => None,
2643 }
2644 });
2645 RewindOutcome {
2646 kept: self.history.len(),
2647 removed: removed_count,
2648 preserved_branch,
2649 }
2650 }
2651
2652 /// BP-8: invert the most recent [`Self::rewind_conversation`] — the
2653 /// messages come back, and the inversion is itself an appended journal
2654 /// record. `false` when there is nothing to undo.
2655 pub fn undo_rewind(&mut self) -> bool {
2656 let Some(mut tail) = self.rewind_undo.pop() else {
2657 return false;
2658 };
2659 self.history.append(&mut tail);
2660 self.journal_op(crate::session_journal::JournalOp::Unrewind);
2661 if self.config.session_tree_enabled {
2662 self.rebuild_session_tree_from_history();
2663 }
2664 true
2665 }
2666
2667 /// BP-8 (catalog:150): append messages recovered from the journal
2668 /// after a crash — they were already recorded, so this deliberately
2669 /// does NOT re-journal them; it puts the live conversation back where
2670 /// the interrupted process left it.
2671 pub fn append_recovered_messages(&mut self, messages: &[ChatMessage]) {
2672 for msg in messages {
2673 if let Some(tree) = self.session_tree.as_mut() {
2674 tree.append_message(msg.clone(), now_ms());
2675 }
2676 self.history.push(msg.clone());
2677 }
2678 }
2679
2680 /// BP-8: how many rewinds are currently undoable.
2681 pub fn undoable_rewinds(&self) -> usize {
2682 self.rewind_undo.len()
2683 }
2684
2685 /// BP-8: restore the undo stack a previous process left in the journal,
2686 /// so `/rewind undo` works across a restart.
2687 pub fn restore_rewind_undo(&mut self, stack: Vec<Vec<ChatMessage>>) {
2688 self.rewind_undo = stack;
2689 }
2690
2691 /// BP-8 (catalog:154 "Queued-prompt persistence"): re-queue pending
2692 /// inputs recovered from the journal WITHOUT re-recording them — they
2693 /// are already in the log, and journaling them again would double them
2694 /// on the next restart.
2695 pub fn restore_queues(&mut self, steer: &[String], follow_up: &[String]) {
2696 for message in steer {
2697 self.steer_queue
2698 .lock()
2699 .unwrap_or_else(std::sync::PoisonError::into_inner)
2700 .queue_unchecked(message.clone());
2701 }
2702 for message in follow_up {
2703 self.follow_up_queue.push_back(message.clone());
2704 }
2705 }
2706
2707 /// BP-8 (catalog:156 "Todos/plan persisted per session"): the session's
2708 /// current `update_plan` checklist.
2709 pub fn plan(&self) -> Vec<crate::session_journal::PlanEntry> {
2710 self.ctx.plan_snapshot()
2711 }
2712
2713 /// BP-8: restore a plan read back from the store on resume. Marked as
2714 /// already-journaled, so a resume that changes nothing writes nothing.
2715 pub fn set_plan(&mut self, steps: Vec<crate::session_journal::PlanEntry>) {
2716 self.ctx.set_plan(steps.clone());
2717 self.journaled_plan = steps;
2718 }
2719
2720 /// BP-8 (catalog:154): record that `count` pending inputs left `queue`
2721 /// and became conversation. A no-op when nothing was taken, or when
2722 /// queue persistence is off.
2723 fn journal_queue_drain(&self, queue: crate::session_journal::QueueKind, count: usize) {
2724 if count == 0 || !self.config.session_queue_persist {
2725 return;
2726 }
2727 self.journal_op(crate::session_journal::JournalOp::Dequeue { queue, count });
2728 }
2729
2730 /// BP-8: journal the plan if `update_plan` changed it since the last
2731 /// time this ran. Called at every loop boundary — a plan that a crash
2732 /// would otherwise strand in the tool's memory is on disk within one
2733 /// iteration of being written.
2734 fn journal_plan_if_changed(&mut self) {
2735 if !self.config.todos_persist {
2736 return;
2737 }
2738 let current = self.ctx.plan_snapshot();
2739 if current == self.journaled_plan {
2740 return;
2741 }
2742 self.journaled_plan.clone_from(¤t);
2743 self.journal_op(crate::session_journal::JournalOp::Plan { steps: current });
2744 }
2745
2746 /// Install (or replace) this agent's reduction policy (A5/A7/A10). Once
2747 /// set, every provider request is built from a *projected* view of
2748 /// `history[1..]` (`reduce::project_messages`) rather than `history`
2749 /// verbatim — `history` itself is never shrunk or mutated by this; only
2750 /// the request view does.
2751 pub fn set_reduction_policy(&mut self, policy: ReductionPolicy) {
2752 self.reduction_policy = Some(policy);
2753 }
2754
2755 /// This agent's reduction policy, if one is installed.
2756 pub fn reduction_policy(&self) -> Option<&ReductionPolicy> {
2757 self.reduction_policy.as_ref()
2758 }
2759
2760 /// Change the global tool-schema tier (TR-8/T5) mid-session. Takes effect
2761 /// starting with the NEXT request this agent builds. Under
2762 /// [`CachePlan::ImportedPrefix`], the first request built after a change
2763 /// is flagged as a cache-bust event and its cache-control annotation is
2764 /// skipped for that one request (see [`provider::tier_change_is_cache_bust`],
2765 /// consulted in `Self::build_request_messages`) — normal annotation
2766 /// resumes on the next request if the tier doesn't change again.
2767 pub fn set_schema_tier(&mut self, tier: crate::tools::SchemaTier) {
2768 self.config.tool_schema_tier = tier;
2769 }
2770
2771 /// Override the schema tier for a single tool (TR-8/T5) mid-session, same
2772 /// cache-bust interaction as [`Self::set_schema_tier`].
2773 pub fn set_tool_schema_tier(
2774 &mut self,
2775 name: impl Into<String>,
2776 tier: crate::tools::SchemaTier,
2777 ) {
2778 self.config
2779 .tool_overrides
2780 .entry(name.into())
2781 .or_default()
2782 .schema_tier = Some(tier);
2783 }
2784
2785 /// A deterministic fingerprint of the current tool-schema tier
2786 /// configuration (global knob + every per-tool override), used to detect
2787 /// a mid-session tier change (TR-8/T5, dev/05). Order-independent over
2788 /// `tool_overrides` (sorted by name before hashing) so insertion order
2789 /// never spuriously changes the signature.
2790 fn schema_tier_signature(&self) -> u64 {
2791 use std::hash::{Hash, Hasher};
2792 let mut hasher = std::collections::hash_map::DefaultHasher::new();
2793 self.config.tool_schema_tier.hash(&mut hasher);
2794 let mut overrides: Vec<(&str, crate::tools::SchemaTier)> = self
2795 .config
2796 .tool_overrides
2797 .iter()
2798 .filter_map(|(name, o)| o.schema_tier.map(|t| (name.as_str(), t)))
2799 .collect();
2800 overrides.sort_by_key(|(name, _)| *name);
2801 for (name, tier) in overrides {
2802 name.hash(&mut hasher);
2803 tier.hash(&mut hasher);
2804 }
2805 hasher.finish()
2806 }
2807
2808 /// Install (or replace) this agent's TR-7 span summarizer — the
2809 /// injectable side-call `Self::build_request_messages` uses to turn an
2810 /// A10 `TurnsCleared` span into an LLM-written summary paragraph when
2811 /// `policy.summarize_cleared_turns` is on. Installing one alone changes
2812 /// nothing: [`ReductionPolicy::summarize_cleared_turns`] (off by
2813 /// default) is the actual gate, so tests/callers that want the
2814 /// deterministic stub can simply never call this.
2815 pub fn set_span_summarizer(
2816 &mut self,
2817 summarizer: impl reduce::summarize::SpanSummarizer + Send + Sync + 'static,
2818 ) {
2819 self.span_summarizer = Some(std::sync::Arc::new(summarizer));
2820 }
2821
2822 /// Install an already-shared summarizer — same seam as
2823 /// [`Self::set_span_summarizer`], for callers (and tests) that need to
2824 /// keep their own handle on it.
2825 pub fn set_span_summarizer_arc(
2826 &mut self,
2827 summarizer: std::sync::Arc<dyn reduce::summarize::SpanSummarizer + Send + Sync>,
2828 ) {
2829 self.span_summarizer = Some(summarizer);
2830 }
2831
2832 /// Prepare TR-7 metadata with this agent's installed summarizer for a
2833 /// projection performed by an outer driver before session history/log
2834 /// are loaded (the CLI foreign-resume preflight). `None` preserves the
2835 /// deterministic fallback when the gate is off, no summarizer exists,
2836 /// the span is below the cost floor, or the side-call fails.
2837 pub fn prepare_cleared_turns_summary(
2838 &self,
2839 msgs: &[ChatMessage],
2840 policy: &ReductionPolicy,
2841 prior: &ReductionLog,
2842 ) -> Option<reduce::PreparedClearSummary> {
2843 let summarizer = self.span_summarizer.as_deref()?;
2844 reduce::prepare_cleared_turns_summary(msgs, policy, prior, summarizer)
2845 }
2846
2847 /// P5-4: install (or replace) this agent's [`crate::EventSink`] AFTER
2848 /// construction — `Config::event_sink` is otherwise only set at
2849 /// `Config`-build time (before `Agent::new`), which is too early for a
2850 /// `tui` embedder that only knows it's activating (and needs to
2851 /// replace whatever print-mode/REPL sink was already installed with
2852 /// one that feeds its own render loop instead of writing straight to
2853 /// stdout) once it already holds a live `Agent`. Mirrors [`Self::
2854 /// set_permissions_approval_handler`]'s "installing one alone changes
2855 /// nothing beyond what already consults `Config::event_sink`" pattern
2856 /// — this is a plain replacement, not a new activation gate.
2857 pub fn set_event_sink(&mut self, sink: crate::EventSink) {
2858 self.config.event_sink = Some(sink);
2859 }
2860
2861 /// P5-1: install (or replace) this agent's permissions-engine approval
2862 /// handler — see [`crate::permissions::PermissionsApprovalHandler`].
2863 /// This is the non-interactive decision seam a CLI/TUI/SDK embedder
2864 /// implements for the `Ask`-tier prompt; the TUI's actual interactive
2865 /// UI is a separate module (P5 row 4), not built here. Installing one
2866 /// alone changes nothing: [`Config::permissions_enabled`] (off by
2867 /// default) is the actual gate — with no handler installed, every
2868 /// `Ask`-tier decision denies (fail-closed, see that trait's doc
2869 /// comment).
2870 ///
2871 /// P5-10 (§2 module 12, `escalation = "ask"`): the SAME handler also
2872 /// backs a sandbox-unenforceable `ask` decision
2873 /// (`crate::sandbox::decide_fs`'s `approval` parameter) — one installed
2874 /// seam serves both `permissions.rules`' `Ask` tier and
2875 /// `permissions.sandbox`'s `escalation = "ask"`, rather than requiring
2876 /// an embedder to install two near-identical handlers. Kept in sync on
2877 /// `self.ctx` (not just `self.permissions_approval_handler`) because
2878 /// `BashTool::execute`/`PersistentShellTool::execute` only ever see
2879 /// `&ToolContext`, never `&Agent` — see `ToolContext::
2880 /// sandbox_approval_handler`'s doc comment.
2881 /// BP-10 (catalog row "Session approval caching"): this agent's
2882 /// approval cache — the door an embedder/TUI uses to inspect or REVOKE
2883 /// remembered grants (`ApprovalCache::clear` forgets every one, in
2884 /// memory and on disk, and the next matching call asks again). Also
2885 /// how a test proves a grant really did survive the process:
2886 /// `store_path()` names the file a second agent reads back.
2887 pub fn permissions_approval_cache(&self) -> &crate::permissions::ApprovalCache {
2888 &self.permissions_approval_cache
2889 }
2890
2891 pub fn set_permissions_approval_handler(
2892 &mut self,
2893 handler: impl crate::permissions::PermissionsApprovalHandler + 'static,
2894 ) {
2895 let handler: std::sync::Arc<dyn crate::permissions::PermissionsApprovalHandler> =
2896 std::sync::Arc::new(handler);
2897 self.permissions_approval_handler = Some(handler.clone());
2898 self.ctx.sandbox_approval_handler =
2899 Some(crate::sandbox::SandboxApprovalHandler(handler.clone()));
2900 // BP-3 (§2 module 8): `exit_plan_mode` presents its plan on this
2901 // same door — one approval seam for the session, not a second one
2902 // the operator would have to answer separately.
2903 self.ctx.approval_handler = Some(crate::tools::ToolApprovalHandler(handler));
2904 }
2905
2906 /// BP-3 (§2 module 6 `tools.question`): install the door `ask_user`
2907 /// asks the human through — the `elicitation/create` handler the design
2908 /// names as the module's protocol side. SDK-owned frontends pass the
2909 /// broker-backed handler (`crate::server::FrontendRequestBridge::
2910 /// elicitation_handler`), which is what makes the question a real
2911 /// frontend request the turn waits on. `None` (the default, nothing
2912 /// installed) leaves the tool deny-default: it reports that nobody can
2913 /// be asked instead of blocking.
2914 ///
2915 /// Installing one alone changes nothing about whether the tool EXISTS —
2916 /// `[capabilities.tools_question]` is that gate, applied by
2917 /// `crate::tools::ToolRegistry::from_config`.
2918 pub fn set_user_question_handler(
2919 &mut self,
2920 handler: std::sync::Arc<dyn crate::mcp::McpElicitationHandler>,
2921 ) {
2922 self.ctx.question_handler = Some(crate::tools::UserQuestionHandler(handler));
2923 }
2924
2925 /// BP-3 (§2 module 8): the shared plan-mode state, so a frontend (the
2926 /// REPL's `/plan`, a TUI toggle) can enter or leave the read-only
2927 /// research phase the same tools and permission gate see.
2928 pub fn plan_mode(&self) -> &std::sync::Arc<crate::tools::PlanModeState> {
2929 &self.ctx.plan_mode
2930 }
2931
2932 /// Install the compatibility approval seam used when the composable
2933 /// permissions engine is disabled. SDK-owned interactive frontends call
2934 /// this alongside [`Self::set_permissions_approval_handler`] so the same
2935 /// authenticated request channel works under either policy engine; the
2936 /// selected engine remains entirely a configuration decision.
2937 pub fn set_legacy_approval_handler(&mut self, handler: crate::config::ApprovalHandler) {
2938 self.config.approval_handler = Some(handler);
2939 }
2940
2941 /// P5-3 (§2 module 9 D5 "subagent transcripts… persisted + linked"):
2942 /// install a [`crate::store::SessionStore`] (+ this agent's own session
2943 /// name in it) so `spawn_subagent` persists each child's transcript
2944 /// (via [`crate::store::SessionStore::save_subagent_transcript`]) and
2945 /// lineage record (via
2946 /// [`crate::store::SessionStore::save_subagent_lineage`]) once the
2947 /// child finishes. Installing one alone changes nothing about whether
2948 /// spawning WORKS — [`Config::subagents_enabled`] is the actual gate;
2949 /// this only controls whether a completed spawn's transcript additionally
2950 /// lands on disk.
2951 pub fn set_subagent_store(
2952 &mut self,
2953 store: std::sync::Arc<crate::store::SessionStore>,
2954 session_name: impl Into<String>,
2955 ) {
2956 self.subagent_store = Some((store, session_name.into()));
2957 }
2958
2959 /// Seed the Claude runtime manifest reconstructed during resume.
2960 ///
2961 /// Installing state enables the matching Claude runtime tool schemas so
2962 /// a disk-reloaded continuation does not lose that vocabulary, but never
2963 /// starts a timer by itself. The supplied execution posture is preserved:
2964 /// an embedding scheduler may deliberately activate before installing it.
2965 pub fn set_claude_runtime_manifest(
2966 &mut self,
2967 manifest: crate::claude_runtime_state::ClaudeRuntimeManifest,
2968 ) {
2969 // A persisted manifest is itself the compatibility capability marker.
2970 // Reopening a Supercode session must not retain its timers while
2971 // silently dropping Claude's Cron*/ScheduleWakeup vocabulary.
2972 self.config.claude_runtime_tools_enabled = true;
2973 self.claude_runtime_manifest = Some(manifest);
2974 }
2975
2976 /// Reinstall project-scoped Claude named-agent definitions when a
2977 /// Supercode continuation carrying a Claude runtime manifest is reopened
2978 /// from disk. The manifest is the durable capability marker; definitions
2979 /// themselves remain authoritative in `<cwd>/.claude/agents/*.md`.
2980 pub fn restore_claude_project_agents(&mut self) -> Result<usize> {
2981 let definitions = crate::claude_compat::load_project_agents(&self.config.cwd)?;
2982 crate::claude_compat::enable_claude_subagent_compatibility(&mut self.config);
2983 for imported in &definitions {
2984 self.config.subagents_definitions.insert(
2985 imported.definition.name.clone(),
2986 imported.definition.clone(),
2987 );
2988 }
2989 Ok(definitions.len())
2990 }
2991
2992 /// Current imported Claude runtime state, including paused mutations made
2993 /// by `Cron*`/`ScheduleWakeup`, for persistence by the embedding loop.
2994 pub fn claude_runtime_manifest(
2995 &self,
2996 ) -> Option<&crate::claude_runtime_state::ClaudeRuntimeManifest> {
2997 self.claude_runtime_manifest.as_ref()
2998 }
2999
3000 /// Mutable access for an embedding scheduler driver to atomically claim
3001 /// due events and persist the resulting manifest. Merely borrowing this
3002 /// state does not start a timer; execution remains the driver's explicit
3003 /// responsibility.
3004 pub fn claude_runtime_manifest_mut(
3005 &mut self,
3006 ) -> Option<&mut crate::claude_runtime_state::ClaudeRuntimeManifest> {
3007 self.claude_runtime_manifest.as_mut()
3008 }
3009
3010 /// P5-4 (tui, closes the P5-3 §2.2 C6 deferred chain): install a
3011 /// factory this agent's `Self::run_spawn_subagent` calls (with the
3012 /// fresh child's own id and this agent's shared
3013 /// [`Self::pending_child_approvals`] queue) to build the
3014 /// `PermissionsApprovalHandler` a `background_prompts = "parent"`
3015 /// child gets, INSTEAD of the default
3016 /// [`crate::subagents::ParentQueueApprovalHandler`]. Installing one
3017 /// alone changes nothing about whether background spawning works —
3018 /// [`Config::subagents_background_prompts`] being
3019 /// [`crate::subagents::BackgroundPromptsPolicy::Parent`] is the actual
3020 /// gate that reaches this factory at all; a `Parent`-policy child
3021 /// spawned before this is installed (or on an agent that never installs
3022 /// it) still gets the immediate-deny default, unchanged.
3023 ///
3024 /// **Security note.** The factory only controls WHICH handler answers
3025 /// an `Ask`-tier request — it can never widen what gets asked in the
3026 /// first place: [`crate::permissions::approval::resolve_ask`] only
3027 /// calls a handler's `ask` when the rule engine has already resolved
3028 /// the call to `Ask` (`Deny` short-circuits before any handler is
3029 /// consulted; `Allow` never needs one), so a parent's "allow" answer
3030 /// here can only grant what the policy already routed to a prompt —
3031 /// never override a `Deny` the engine already decided.
3032 pub fn set_child_approval_handler_factory(
3033 &mut self,
3034 factory: impl Fn(
3035 String,
3036 std::sync::Arc<std::sync::Mutex<Vec<crate::subagents::QueuedApproval>>>,
3037 ) -> std::sync::Arc<dyn crate::permissions::PermissionsApprovalHandler>
3038 + Send
3039 + Sync
3040 + 'static,
3041 ) {
3042 self.child_approval_handler_factory = Some(std::sync::Arc::new(factory));
3043 }
3044
3045 /// P5-3 (§2.2 C6 "parent-surfaced queue"): every approval request a
3046 /// `background_prompts = "parent"` child has raised so far, oldest
3047 /// first — a read-only audit view, not a mutable queue the caller
3048 /// answers. Under P5-3's own default handler (no P5-4 TUI factory
3049 /// installed) every entry here WAS already resolved `Deny` (a
3050 /// background call can't wait for an answer with no handler
3051 /// installed) — but once a `crate::tui::TuiChildApprovalHandler`
3052 /// factory is installed (P5-4,
3053 /// [`Self::set_child_approval_handler_factory`]), the underlying call
3054 /// genuinely blocks and may resolve `Allow`/`AllowForSession`; this
3055 /// method still records the SAME entry for the audit trail either
3056 /// way, so "queued here" no longer implies "was denied" in general —
3057 /// see [`crate::subagents::QueuedApproval`]'s doc comment.
3058 pub fn pending_child_approvals(&self) -> Vec<crate::subagents::QueuedApproval> {
3059 self.pending_child_approvals
3060 .lock()
3061 .map(|q| q.clone())
3062 .unwrap_or_default()
3063 }
3064
3065 /// P4b: install (or replace) this agent's auto-title side-call — see
3066 /// [`crate::session_title::SessionTitler`]. Installing one alone changes
3067 /// nothing: [`Config::auto_title`] (off by default) is the actual gate a
3068 /// caller should consult before calling [`Self::auto_title`].
3069 pub fn set_session_titler(
3070 &mut self,
3071 titler: impl crate::session_title::SessionTitler + Send + Sync + 'static,
3072 ) {
3073 self.session_titler = Some(std::sync::Arc::new(titler));
3074 }
3075
3076 /// P4b: produce a title for this agent's current conversation via the
3077 /// installed [`Self::set_session_titler`] side-call. Returns `None` (never
3078 /// panics, never blocks longer than the titler itself does) if no
3079 /// titler is installed, or the side-call itself declined (see
3080 /// [`crate::session_title::auto_title`]). Does NOT consult
3081 /// [`Config::auto_title`] itself — that gate is the caller's
3082 /// responsibility, matching `Self::span_summarizer`'s precedent of
3083 /// keeping the mechanism and the policy gate separate.
3084 pub fn auto_title(&self) -> Option<String> {
3085 let titler = self.session_titler.as_deref()?;
3086 crate::session_title::auto_title(&self.history, titler)
3087 }
3088
3089 /// P4b (§1.6, catalog §4a "persisted per-turn usage records"): every
3090 /// [`crate::usage_log::UsageRecord`] this agent has accumulated so far.
3091 pub fn usage_records(&self) -> &[crate::usage_log::UsageRecord] {
3092 &self.usage_log
3093 }
3094
3095 /// P4b: persist this agent's accumulated usage log to `store` under
3096 /// `name` — a thin wrapper over [`crate::store::SessionStore::save_usage_log`]
3097 /// so callers don't need to import both types.
3098 pub fn save_usage_log(&self, store: &crate::store::SessionStore, name: &str) -> Result<()> {
3099 store.save_usage_log(name, &self.usage_log)
3100 }
3101
3102 /// BP-7 (catalog §4a "Turn/step bracketing records"): every
3103 /// [`crate::turn_record::TurnRecord`] this agent has accumulated —
3104 /// the context/usage/finish brackets of each model round-trip plus the
3105 /// retry, abort, effort and goal markers between them.
3106 pub fn turn_records(&self) -> &[crate::turn_record::TurnRecord] {
3107 &self.turn_records
3108 }
3109
3110 /// BP-7: persist the marker log to `store` under `name`
3111 /// (`<name>.events.jsonl`), the same thin-wrapper shape
3112 /// [`Self::save_usage_log`] has.
3113 pub fn save_turn_records(&self, store: &crate::store::SessionStore, name: &str) -> Result<()> {
3114 store.save_turn_records(name, &self.turn_records)
3115 }
3116
3117 /// BP-7 (catalog §4a "Per-turn cost/usage accounting"): dollars this
3118 /// agent has spent so far. `0.0` when the model is unpriceable — read
3119 /// [`Self::model_priced`] to tell "free" from "unknown".
3120 pub fn total_cost_usd(&self) -> f64 {
3121 self.total_cost_usd
3122 }
3123
3124 /// BP-7: whether this build can price this agent's model, i.e. whether
3125 /// [`Self::total_cost_usd`] is a real figure rather than a floor.
3126 pub fn model_priced(&self) -> bool {
3127 self.model_price.is_some()
3128 }
3129
3130 /// BP-7 (catalog §4a "Turn/budget caps"): tool calls this agent has
3131 /// executed so far — the counter [`Config::max_steps`] bounds.
3132 pub fn total_steps(&self) -> usize {
3133 self.total_steps
3134 }
3135
3136 /// BP-7 (catalog §4a "Interrupt/abort with state preserved"): record
3137 /// that the in-flight turn was interrupted.
3138 ///
3139 /// Called by whoever owns the cancellation (the CLI's Ctrl-C race), NOT
3140 /// by the loop itself: a cancelled `send` future is dropped mid-await,
3141 /// so the loop never runs another line. The partial work already
3142 /// appended to the transcript stands; this marker is what makes the
3143 /// interruption a persisted FACT — the residue the ledger row named —
3144 /// rather than something a reader has to infer from a dangling tool
3145 /// call on reload. Emits [`AgentEvent::TurnAborted`] as the live
3146 /// counterpart.
3147 pub fn note_abort(&mut self, source: &str) {
3148 let messages = self.history.len();
3149 self.emit(AgentEvent::TurnAborted {
3150 source: source.to_string(),
3151 });
3152 self.push_turn_marker(crate::turn_record::TurnMarker::Aborted {
3153 source: source.to_string(),
3154 messages,
3155 });
3156 }
3157
3158 // ---- BP-7: goals (catalog §4a "Goals — persistent objective across
3159 // turns"; §2 module 7 `todos`, §3.1 `capabilities.todos.goals`) ----
3160
3161 /// Set (or revise) this session's standing objective.
3162 ///
3163 /// Returns `false`, changing nothing, when `capabilities.todos.goals`
3164 /// is off — the module gate, not a silent success. A goal restates
3165 /// itself at the tail of every request until [`Self::clear_goal`], and
3166 /// each change appends a `goal` marker to the turn-record log.
3167 pub fn set_goal(&mut self, objective: impl Into<String>) -> bool {
3168 if !self.config.goals_enabled {
3169 return false;
3170 }
3171 let objective = objective.into();
3172 let now = now_ms();
3173 match &mut self.goal {
3174 Some(goal) => goal.revise(objective.clone(), now),
3175 slot @ None => *slot = Some(crate::goals::GoalRecord::new(objective.clone(), now)),
3176 }
3177 self.push_turn_marker(crate::turn_record::TurnMarker::Goal { objective });
3178 true
3179 }
3180
3181 /// This session's standing objective, if one is set.
3182 pub fn goal(&self) -> Option<&crate::goals::GoalRecord> {
3183 self.goal.as_ref()
3184 }
3185
3186 /// Drop the standing objective. `true` when there was one to drop.
3187 pub fn clear_goal(&mut self) -> bool {
3188 if self.goal.take().is_none() {
3189 return false;
3190 }
3191 self.push_turn_marker(crate::turn_record::TurnMarker::Goal {
3192 objective: String::new(),
3193 });
3194 true
3195 }
3196
3197 /// BP-7: persist (or, when cleared, remove) the standing objective
3198 /// beside the session — same thin-wrapper shape as
3199 /// [`Self::save_usage_log`].
3200 pub fn save_goal(&self, store: &crate::store::SessionStore, name: &str) -> Result<()> {
3201 match &self.goal {
3202 Some(goal) => store.save_goal(name, goal),
3203 None => store.clear_goal(name),
3204 }
3205 }
3206
3207 /// BP-7: adopt a goal loaded from the store (a resumed session picks up
3208 /// exactly where it left off). Bypasses the module gate on purpose: a
3209 /// goal already persisted is data to restore, not a new capability
3210 /// being turned on, and dropping it silently would lose session state.
3211 pub fn restore_goal(&mut self, goal: Option<crate::goals::GoalRecord>) {
3212 self.goal = goal;
3213 }
3214
3215 // ---- BP-7: extended-thinking control (catalog §4a "Extended thinking
3216 // control": "Reasoning on/off/levels mid-session") ----
3217
3218 /// The reasoning-effort level in force for the NEXT request, or `None`
3219 /// when extended thinking is off.
3220 pub fn effort(&self) -> Option<&str> {
3221 self.config.effort.as_deref()
3222 }
3223
3224 /// Change the reasoning-effort level mid-session.
3225 ///
3226 /// `Some(level)` sets the level; `None` turns extended thinking OFF —
3227 /// the on/off toggle the ledger row named as distinct from the level.
3228 /// `run_loop` reads `self.config.effort` fresh when it builds each
3229 /// `ChatRequest`, so this takes effect on the very next request with no
3230 /// other copy to update (the same contract [`Self::set_model`] has).
3231 /// The change is appended to the turn-record log as an `effort` marker,
3232 /// the extended-thinking analog of the `model_change` log.
3233 ///
3234 /// Returns the PREVIOUS setting.
3235 pub fn set_effort(&mut self, effort: Option<String>) -> Option<String> {
3236 let previous = self.config.effort.clone();
3237 if previous == effort {
3238 return previous;
3239 }
3240 self.config.effort = effort.clone();
3241 self.push_turn_marker(crate::turn_record::TurnMarker::Effort {
3242 from: previous.clone(),
3243 to: effort,
3244 });
3245 previous
3246 }
3247
3248 // ---- BP-7: review mode (catalog §4a "Review mode — dedicated
3249 // code-review flow"; §3.1 `[core.prompts]`) ----
3250
3251 /// The purpose-built review turn's prompt: the `code-review` template
3252 /// from [`Config::prompts`] with `{args}` replaced by `args`.
3253 ///
3254 /// `None` when the resolved config carries no `code-review` template —
3255 /// the preset decides whether this harness has a review mode, and the
3256 /// template IS the report format (both parity presets pin one).
3257 pub fn review_prompt(&self, args: &str) -> Option<String> {
3258 self.config
3259 .prompts
3260 .get(REVIEW_PROMPT_NAME)
3261 .map(|template| template.replace("{args}", args.trim()))
3262 }
3263
3264 /// Run the review turn: an ordinary [`Self::send`] of
3265 /// [`Self::review_prompt`], so the review's request, tools, transcript
3266 /// and records are the session's own — a purpose-built TURN, not a
3267 /// second agent.
3268 pub async fn review(&mut self, args: &str) -> Result<String> {
3269 let prompt = self.review_prompt(args).ok_or_else(|| {
3270 Error::Other(format!(
3271 "no `{REVIEW_PROMPT_NAME}` prompt template is configured for this harness"
3272 ))
3273 })?;
3274 self.send(prompt).await
3275 }
3276
3277 // ---- BP-7: side/ephemeral Q&A (catalog §4a "Side/ephemeral Q&A":
3278 // "Tool-less question over full context, never enters history") ----
3279
3280 /// Answer `question` over this session's FULL current context without
3281 /// recording anything.
3282 ///
3283 /// Three properties, all load-bearing and all asserted by this build's
3284 /// tests: the request carries the whole conversation as the next turn
3285 /// would see it; it advertises NO tools, so the model can only answer;
3286 /// and neither `history`, the sidecar recorder, the usage log nor the
3287 /// turn-record log is touched — `&self`, not `&mut self`, is the type
3288 /// system saying so. cc's `/btw` and cx's `/side`.
3289 pub async fn side_question(&self, question: &str) -> Result<String> {
3290 let mut messages = self.history.clone();
3291 if let Some(goal) = &self.goal {
3292 messages.push(ChatMessage::system(goal.reminder()));
3293 }
3294 messages.push(ChatMessage::user(format!(
3295 "{SIDE_QUESTION_PREAMBLE}
3296
3297{question}"
3298 )));
3299 let mut req = ChatRequest {
3300 model: self.config.model.clone(),
3301 messages,
3302 tools: Vec::new(),
3303 temperature: self.config.temperature,
3304 max_tokens: self.config.max_tokens,
3305 effort: self.config.effort.clone(),
3306 response_format: None,
3307 service_tier: None,
3308 thinking_budget: None,
3309 extra_body: self.config.extra_body.clone(),
3310 };
3311 // BP-13: a side question is still a request to THIS model, so it
3312 // carries the same routing decisions the loop's own requests do.
3313 self.apply_routing(&mut req);
3314 let (assistant, _usage) = self.provider.complete(&req, &|_: &str| {}).await?;
3315 Ok(assistant.content.unwrap_or_default())
3316 }
3317
3318 /// BP-7: append one marker against the NEXT round-trip's index — the
3319 /// right frame for a marker written between turns (a goal change, an
3320 /// effort change, an abort).
3321 fn push_turn_marker(&mut self, marker: crate::turn_record::TurnMarker) {
3322 self.push_turn_marker_at(self.turn_index, marker);
3323 }
3324
3325 /// BP-7: append one marker against an explicit round-trip index — used
3326 /// inside `Self::run_loop`, where markers are written on both sides of
3327 /// the `turn_index` advance and must all carry the round-trip they
3328 /// describe.
3329 fn push_turn_marker_at(&mut self, turn: usize, marker: crate::turn_record::TurnMarker) {
3330 self.turn_records.push(crate::turn_record::TurnRecord::new(
3331 turn,
3332 &self.config.model,
3333 now_ms(),
3334 marker,
3335 ));
3336 }
3337
3338 /// P4b (§1.7, pi§3 semantics): queue a mid-turn steering message —
3339 /// delivered "after current tool calls" (pi's phrasing): at the top of
3340 /// `Self::run_loop`'s NEXT iteration, before the next model request is
3341 /// built, regardless of whether this turn is still mid-flight with
3342 /// pending tool calls. Drained per [`Config::steering_mode`].
3343 pub fn queue_steer(&self, message: impl Into<String>) {
3344 let message = message.into();
3345 // BP-8 (catalog:154 "Queued-prompt persistence"): the input is
3346 // recorded BEFORE it is queued, so the window in which a crash
3347 // could lose it is zero. A no-op when `core.session.queue_persist`
3348 // is off (cx-parity: stock Codex has no queue-operation records).
3349 if self.config.session_queue_persist {
3350 self.journal_op(crate::session_journal::JournalOp::Enqueue {
3351 queue: crate::session_journal::QueueKind::Steer,
3352 text: message.clone(),
3353 });
3354 }
3355 self.steer_queue
3356 .lock()
3357 .unwrap_or_else(std::sync::PoisonError::into_inner)
3358 .queue_unchecked(message);
3359 }
3360
3361 /// Crate-internal shared steering handle used by the canonical SDK
3362 /// runtime. It remains writable while an active turn holds `&mut Agent`,
3363 /// allowing local and remote frontends to steer without owning the loop.
3364 pub(crate) fn steer_queue_handle(&self) -> std::sync::Arc<std::sync::Mutex<SteerInbox>> {
3365 self.steer_queue.clone()
3366 }
3367
3368 /// P4b: queue a follow-up message — delivered "at idle" (pi's phrasing):
3369 /// only once `Self::run_loop` would otherwise return a final answer
3370 /// (no more tool calls pending). Drained per [`Config::follow_up_mode`].
3371 pub fn queue_follow_up(&mut self, message: impl Into<String>) {
3372 let message = message.into();
3373 // BP-8 (catalog:154): same record-then-queue order as
3374 // [`Self::queue_steer`].
3375 if self.config.session_queue_persist {
3376 self.journal_op(crate::session_journal::JournalOp::Enqueue {
3377 queue: crate::session_journal::QueueKind::FollowUp,
3378 text: message.clone(),
3379 });
3380 }
3381 self.follow_up_queue.push_back(message);
3382 }
3383
3384 /// P4b: how many steering messages are currently queued (mid-turn +
3385 /// follow-up combined) — mostly for tests/diagnostics.
3386 pub fn queued_steer_count(&self) -> usize {
3387 self.steer_queue
3388 .lock()
3389 .unwrap_or_else(std::sync::PoisonError::into_inner)
3390 .len()
3391 + self.follow_up_queue.len()
3392 }
3393
3394 /// The accumulating reduction log (A5) — every reduction applied to any
3395 /// projected request view so far. Combined with a full-fidelity sidecar
3396 /// Session, this is enough to `reduce::invert` any projected view back to
3397 /// the exact original.
3398 pub fn reduction_log(&self) -> &ReductionLog {
3399 &self.reduction_log
3400 }
3401
3402 /// PARITY-18 D4 — arm the per-send context guard: `Self::run_loop`
3403 /// will refuse (via [`Error::ContextLimitExceeded`]) to build and issue
3404 /// ANY request — the first or any later turn — whose
3405 /// [`supercode_runtime::context_guard`] verdict is "does not fit" against
3406 /// `limit`. Call this once the target model's context-window size is
3407 /// known (`resume --reduced`'s preflight already computes it). Leaving
3408 /// this unset (the default) is a no-op: no guard runs, exactly today's
3409 /// pre-PARITY-18 behavior.
3410 pub fn set_context_limit(&mut self, limit: u64) {
3411 self.context_limit = Some(limit);
3412 }
3413
3414 /// This agent's armed context limit, if [`Self::set_context_limit`] has
3415 /// been called.
3416 pub fn context_limit(&self) -> Option<u64> {
3417 self.context_limit
3418 }
3419
3420 /// The model identifier this agent sends on its next request
3421 /// ([`Config::model`], as of construction/resume or the last
3422 /// [`Self::set_model`] call).
3423 pub fn model(&self) -> &str {
3424 &self.config.model
3425 }
3426
3427 /// UX-30 dev/02 — switch the model this agent sends, starting with the
3428 /// NEXT request it builds (and every one after, until changed again).
3429 /// `Self::run_loop` reads `self.config.model` fresh on every request
3430 /// (see its `ChatRequest` construction), so this alone is enough —
3431 /// there is no cached/baked-in copy anywhere else to also update.
3432 /// Takes effect immediately; safe to call only between turns (the
3433 /// REPL's `/model` picker runs at the prompt, never mid-turn). Touches
3434 /// nothing else: history, the sidecar, and reduction state are exactly
3435 /// as untouched as [`Self::set_schema_tier`] leaves them for a
3436 /// mid-session tier change.
3437 ///
3438 /// P4c-review note: this is the LOW-LEVEL primitive — it swaps
3439 /// [`Config::model`] and nothing else. It does NOT run dep 8's
3440 /// reasoning-artifact filter
3441 /// ([`reduce::rehydrate::filter_reasoning_artifacts`]) and does NOT
3442 /// create a [`crate::model_change::ModelChangeRecord`], so calling it
3443 /// directly for a mid-session handoff between two DIFFERENT models
3444 /// leaves model-A's reasoning artifacts in `history` for model-B to
3445 /// inherit. [`Self::switch_model`] is the safe superset — gated by
3446 /// [`Config::model_switch_allow_switch`], it filters and records the
3447 /// switch before delegating to this method — and is what callers
3448 /// performing a governed mid-session model switch should use instead.
3449 pub fn set_model(&mut self, model: impl Into<String>) {
3450 self.config.model = model.into();
3451 // BP-5 (catalog D2 "Per-model-family base-prompt selection"): the
3452 // family's base prompt follows the model. Codex re-selects
3453 // `base_instructions` when the model changes; leaving model-A's
3454 // base prompt in front of model-B is exactly the mismatch the row
3455 // exists to prevent. Same locate-and-replace mechanism
3456 // `refresh_env_context` uses, and a no-op whenever the selection
3457 // did not actually change (always, for a config with no family
3458 // table).
3459 self.refresh_base_prompt();
3460 // BP-7: the price follows the model, or the per-turn cost figure
3461 // would keep billing the OLD model's rates after a switch.
3462 self.model_price = crate::pricing::resolve(
3463 &self.config.model,
3464 self.config.price_input_per_mtok,
3465 self.config.price_output_per_mtok,
3466 );
3467 }
3468
3469 /// P4c (§1.10/§3.1 `core.model_switch.allow_switch`, D9 row, dep 8,
3470 /// design's "core NEW-significant" item): the mid-session model
3471 /// switch — a superset of [`Self::set_model`] gated by
3472 /// [`Config::model_switch_allow_switch`].
3473 ///
3474 /// **`allow_switch = false` (the default): EXACTLY [`Self::set_model`]**
3475 /// — same single field write, nothing else touched, no
3476 /// [`crate::model_change::ModelChangeRecord`] created. Byte-identical to
3477 /// calling `set_model` directly.
3478 ///
3479 /// **`allow_switch = true`:** additionally, before the swap takes
3480 /// effect, runs [`reduce::rehydrate::filter_reasoning_artifacts`] over
3481 /// [`Self::history`] — model-A's reasoning/thinking artifacts (any
3482 /// [`supercode_interchange::ChatMessage::metadata`] key in
3483 /// [`reduce::rehydrate::REASONING_METADATA_KEYS`], any `content_parts`
3484 /// block whose `"type"` is in
3485 /// [`reduce::rehydrate::REASONING_CONTENT_PART_TYPES`]) are stripped
3486 /// BEFORE model-B ever builds a request from this history — then
3487 /// appends a typed, translatable [`crate::model_change::ModelChangeRecord`]
3488 /// to [`Self::model_change_records`] (persist it via
3489 /// [`Self::save_model_change_log`]). A switch TO the current model
3490 /// (`model == Self::model()`) is treated as a no-op — still exactly
3491 /// `set_model`'s mechanics, no record for a switch that didn't actually
3492 /// change anything (and nothing to filter FOR, since there was no
3493 /// handoff).
3494 pub fn switch_model(&mut self, model: impl Into<String>) {
3495 let to = model.into();
3496 if !self.config.model_switch_allow_switch || self.config.model == to {
3497 self.set_model(to);
3498 return;
3499 }
3500 let from = self.config.model.clone();
3501 self.record_model_change(&from, &to, None);
3502 }
3503
3504 /// BP-13 — the ONE place a mid-session model change is performed and
3505 /// recorded, shared by [`Self::switch_model`] (a user asked) and the
3506 /// run loop's fallback pass (a provider failed).
3507 ///
3508 /// It does four things, in this order, and nothing else: strips model-A
3509 /// reasoning artifacts out of the live history (dep 8 — model B must
3510 /// never inherit them), moves [`Config::model`], appends the typed
3511 /// [`crate::model_change::ModelChangeRecord`], and writes that same
3512 /// record into the append-only session journal (BP-8) — which is where
3513 /// every persisted routing record lives; there is no second file. The
3514 /// change is also EMITTED, so a surface that renders events shows the
3515 /// switch instead of silently answering as a different model.
3516 pub fn record_model_change(&mut self, from: &str, to: &str, reason: Option<&str>) {
3517 if from == to {
3518 return;
3519 }
3520 let touched = reduce::rehydrate::filter_reasoning_artifacts(&mut self.history);
3521 self.set_model(to.to_string());
3522 let record = crate::model_change::ModelChangeRecord::new(
3523 self.turn_index,
3524 from,
3525 to,
3526 true,
3527 touched,
3528 now_ms(),
3529 )
3530 .with_reason(reason.map(str::to_string));
3531 self.journal_model_change(&record);
3532 self.model_change_log.push(record);
3533 self.emit(AgentEvent::ModelChanged {
3534 from: from.to_string(),
3535 to: to.to_string(),
3536 reason: reason.map(str::to_string),
3537 });
3538 // A switch also re-injects the switch NOTICE when the config asks
3539 // for one (Codex's own mid-session behavior: the conversation is
3540 // told the model changed, so the new model reads the handoff rather
3541 // than inferring it from a style break).
3542 if self.config.model_switch_notice {
3543 let notice = ChatMessage::user(format!(
3544 "[model changed: {from} -> {to}{}]",
3545 match reason {
3546 Some(r) => format!(" ({r})"),
3547 None => String::new(),
3548 }
3549 ));
3550 let _ = self.record(¬ice);
3551 self.history.push(notice);
3552 }
3553 }
3554
3555 /// BP-13 (catalog D9 "Fast mode / service tiers"): set or clear the
3556 /// session-level service-tier override. `Some(tier)` WINS over the
3557 /// `[capabilities.model_catalog] service_tier` rule for every
3558 /// subsequent request (it is the live toggle the user just pulled);
3559 /// `None` puts the configured rule back in charge. Takes effect on the
3560 /// next request the loop builds, like [`Self::set_model`].
3561 pub fn set_service_tier(&mut self, tier: Option<String>) {
3562 self.config.service_tier = tier;
3563 }
3564
3565 /// BP-13 — apply the routing table to a request that already names its
3566 /// model: effort LEVEL (per-model override of `[core] effort`, clamped
3567 /// by whichever effort cap applies), thinking-token BUDGET, and service
3568 /// TIER (the live `/fast` override winning over the configured rule).
3569 /// Called for every request the loop builds AND again for every
3570 /// fallback hop, so a hop to a different model gets that model's
3571 /// routing rather than the previous model's.
3572 fn apply_routing(&self, req: &mut ChatRequest) {
3573 let routing = &self.config.model_routing;
3574 let rules = routing.rules_for(&req.model);
3575 // Plan mode's own effort tier, when the mode is live, is the
3576 // session level for this request — Codex's `/plan` is effort
3577 // steering (cx§6), so planning need not think at the executing
3578 // level. It is still clamped by whatever effort cap applies,
3579 // because `effective_effort` does the clamping, not this line.
3580 let session_effort = match (
3581 self.ctx.plan_mode.is_active(),
3582 self.config.plan_mode_effort.as_deref(),
3583 ) {
3584 (true, Some(effort)) => Some(effort),
3585 _ => self.config.effort.as_deref(),
3586 };
3587 req.effort = routing.effective_effort(&req.model, session_effort);
3588 req.thinking_budget = rules.thinking_budget;
3589 req.service_tier = self.config.service_tier.clone().or(rules.service_tier);
3590 }
3591
3592 /// BP-13 — send `req`, walking [`Config::model_fallback`] when the
3593 /// failure is one another model could plausibly answer.
3594 ///
3595 /// Returns the final outcome plus the hops actually taken, so the
3596 /// caller (which owns `&mut self`) can record each one. Each hop
3597 /// re-applies routing for the new model and strips model-A reasoning
3598 /// artifacts from the request's own message copy before model B sees
3599 /// them — the same dep-8 guarantee [`Self::record_model_change`] gives
3600 /// the live history.
3601 async fn complete_with_fallback(
3602 &self,
3603 req: &mut ChatRequest,
3604 on_delta: &(dyn for<'a> Fn(&'a str) + Send + Sync),
3605 ) -> (Result<(ChatMessage, provider::Usage)>, Vec<FallbackHop>) {
3606 let mut hops = Vec::new();
3607 let mut result = self.provider.complete(req, on_delta).await;
3608 for next in &self.config.model_fallback {
3609 let Err(error) = &result else {
3610 break;
3611 };
3612 if !is_failover_worthy(error) {
3613 break;
3614 }
3615 if next.is_empty() || next == &req.model {
3616 continue;
3617 }
3618 let reason = error.to_string();
3619 let from = std::mem::replace(&mut req.model, next.clone());
3620 reduce::rehydrate::filter_reasoning_artifacts(&mut req.messages);
3621 self.apply_routing(req);
3622 hops.push(FallbackHop {
3623 from,
3624 to: next.clone(),
3625 reason,
3626 });
3627 result = self.provider.complete(req, on_delta).await;
3628 }
3629 (result, hops)
3630 }
3631
3632 /// P4c: every [`crate::model_change::ModelChangeRecord`] this agent has
3633 /// accumulated so far (via [`Self::switch_model`] with `allow_switch`
3634 /// on). Empty when the knob is off or no switch has happened yet.
3635 pub fn model_change_records(&self) -> &[crate::model_change::ModelChangeRecord] {
3636 &self.model_change_log
3637 }
3638
3639 /// P4c: persist this agent's accumulated model-change log to `store`
3640 /// under `name` — the [`crate::model_change::ModelChangeRecord`] analog
3641 /// of [`Self::save_usage_log`].
3642 pub fn save_model_change_log(
3643 &self,
3644 store: &crate::store::SessionStore,
3645 name: &str,
3646 ) -> Result<()> {
3647 store.save_model_change_log(name, &self.model_change_log)
3648 }
3649
3650 /// P4e (§1.6/§3.1 `core.session.git_metadata`, catalog:331): this
3651 /// agent's captured git provenance, if [`Config::session_git_metadata`]
3652 /// was on at construction and the best-effort probe found a repo.
3653 pub fn git_metadata(&self) -> Option<&crate::git_metadata::GitMetadataRecord> {
3654 self.git_metadata.as_ref()
3655 }
3656
3657 /// P4e: persist this agent's captured git metadata to `store` under
3658 /// `name` — a thin wrapper over
3659 /// [`crate::store::SessionStore::save_git_metadata`], the
3660 /// [`crate::git_metadata::GitMetadataRecord`] analog of
3661 /// [`Self::save_usage_log`]. A no-op (`Ok(())`, nothing written) when
3662 /// [`Self::git_metadata`] is `None`.
3663 pub fn save_git_metadata(&self, store: &crate::store::SessionStore, name: &str) -> Result<()> {
3664 match &self.git_metadata {
3665 Some(record) => store.save_git_metadata(name, record),
3666 None => Ok(()),
3667 }
3668 }
3669
3670 /// P4e DEFECT-FIX (independent Fable-5 review of P4e: `core.session.persist`
3671 /// had a `Config` field and CLI plumbing at `ConfigProfile` → `Config` but
3672 /// no consumer at all): whether a CLI caller's session-store save sites
3673 /// (`persist_session`, `persist_full_view`) should actually write to
3674 /// disk. `true` (the default) is byte-identical to pre-fix behavior —
3675 /// every session persists. `false` makes a session ephemeral: it runs
3676 /// exactly as before, but no `<name>.jsonl`/sidecar family is ever
3677 /// written for it. A plain getter, same posture as [`Self::model`] —
3678 /// this crate itself never reads or enforces it; the CLI's save sites do.
3679 pub fn session_persist(&self) -> bool {
3680 self.config.session_persist
3681 }
3682
3683 /// P4e DEFECT-FIX (independent Fable-5 review of P4e: `core.session.name`
3684 /// had a `Config` field and CLI plumbing but no consumer): the
3685 /// caller-configured session name, if `[core.session] name` was set.
3686 /// `None` (the default) leaves session naming exactly as before —
3687 /// `mint_session_name`'s auto-generated `<tag>-<adjective>-<noun>` shape.
3688 /// A plain getter, same posture as [`Self::session_persist`].
3689 pub fn session_name(&self) -> Option<&str> {
3690 self.config.session_name.as_deref()
3691 }
3692
3693 /// PARITY-18 D3 — whether this agent has actually issued at least one
3694 /// live request to its [`Provider`] so far (set the instant
3695 /// `Self::run_loop` reaches its real send site, regardless of whether
3696 /// that call then succeeds or fails). Callers should report
3697 /// "request sent" from THIS, never from having merely passed the
3698 /// context guard or having called [`Self::send`] — either of those can
3699 /// happen with zero requests actually issued (a guard refusal, an
3700 /// interactive session quit before any turn completes).
3701 pub fn request_issued(&self) -> bool {
3702 self.requests_issued
3703 }
3704
3705 /// P5-2 (§2.2 C2): whether this agent currently considers its
3706 /// [`CachePlan::ImportedPrefix`] cache entry warm — mirrors
3707 /// [`Self::request_issued`]'s read-only-observability precedent, so a
3708 /// caller (or a test) can confirm [`Self::register_tool`]'s C2
3709 /// invalidation actually took effect without reaching into private
3710 /// state.
3711 pub fn cache_established(&self) -> bool {
3712 self.cache_established
3713 }
3714
3715 /// B7: length of the imported-prefix protected by [`CachePlan::ImportedPrefix`]
3716 /// (this agent's own system message plus every message of a
3717 /// previously-imported session), set by [`Self::load_session`]. `None`
3718 /// until a session has been loaded.
3719 pub fn imported_prefix_len(&self) -> Option<usize> {
3720 self.imported_prefix_len
3721 }
3722
3723 /// Replace this agent's accumulating reduction log (C4: `/expand`/`/reduce`
3724 /// mutate the log directly via `reduce::invert_one`/`reduce::project_messages`
3725 /// and must feed the result back here so the *next* request build or
3726 /// persist sees the updated state instead of silently recomputing from an
3727 /// empty log). Also lets a caller (`resume_cmd`, C1) seed the log with the
3728 /// initial projection it already computed for the entry banner, so
3729 /// `reduction_log()` reflects reality even before this agent's first
3730 /// `send()` (which is otherwise the only place `build_request_messages`
3731 /// populates it).
3732 pub fn set_reduction_log(&mut self, log: ReductionLog) {
3733 self.reduction_log = log;
3734 }
3735
3736 /// Replace the conversation with a loaded session, keeping this agent's own
3737 /// system prompt at the front. The session's own system/developer turns are
3738 /// preserved after it for context.
3739 pub fn load_session(&mut self, session: Session) {
3740 let system = self.history.first().cloned();
3741 self.history.clear();
3742 if let Some(sys) = system {
3743 self.history.push(sys);
3744 }
3745 self.history.extend(session.messages);
3746 // B7: the whole of `history` at this point — this agent's own system
3747 // message plus every imported message — is the stable prefix a
3748 // resumed session resends byte-identically every turn.
3749 self.imported_prefix_len = Some(self.history.len());
3750 // UX-26 (B7-warn): a freshly loaded prefix has no established cache
3751 // entry of THIS agent's own making yet (even if this agent was
3752 // resumed once before — that earlier prefix is gone). Seed the
3753 // activity clock from the loaded session's own last message
3754 // timestamp (walking backward past any trailing message that
3755 // carries none), so a session that's been sitting idle since
3756 // Claude Code/Codex/a prior supercode run last touched it is
3757 // correctly treated as already-cold on its very first turn here —
3758 // `None` (no timestamp anywhere in the loaded messages) leaves the
3759 // TTL check disarmed rather than guessing.
3760 self.cache_established = false;
3761 self.last_cache_activity_ms = self
3762 .history
3763 .iter()
3764 .rev()
3765 .find_map(|m| m.metadata.get("timestamp"))
3766 .and_then(|ts| supercode_interchange::sidecar::rfc3339_to_ms(ts));
3767 }
3768
3769 /// Append `msg` to the sidecar recorder (A3), if one is installed — a
3770 /// no-op, at zero cost, when `recorder` is `None` (today's behavior).
3771 fn record(&mut self, msg: &ChatMessage) -> Result<()> {
3772 if let Some(recorder) = self.recorder.as_mut() {
3773 recorder.append(msg)?;
3774 }
3775 // BP-8 (catalog:150): the append-only half — written and FLUSHED
3776 // here, at the moment the message exists, not at the end of the
3777 // turn. A journal failure is logged, never fatal: durability
3778 // bookkeeping must not be able to fail a turn.
3779 if let Some(journal) = &self.journal {
3780 let mut guard = journal
3781 .lock()
3782 .unwrap_or_else(std::sync::PoisonError::into_inner);
3783 if let Err(error) = guard.append_message(msg) {
3784 tracing::warn!("failed to journal a message: {error}");
3785 }
3786 }
3787 // BP-8 (catalog:151): the same message becomes a tree node, so the
3788 // tree and the linear history never disagree about what was said.
3789 if let Some(tree) = self.session_tree.as_mut() {
3790 tree.append_message(msg.clone(), now_ms());
3791 }
3792 Ok(())
3793 }
3794
3795 /// Persist the live conversation to `path` as JSONL (one [`ChatMessage`]
3796 /// per line) so the session can be resumed later — supercode's own sessions
3797 /// become first-class, resumable artifacts.
3798 pub fn save_transcript(&self, path: impl AsRef<std::path::Path>) -> Result<()> {
3799 let mut out = String::new();
3800 for m in &self.history {
3801 out.push_str(&serde_json::to_string(m).map_err(Error::Decode)?);
3802 out.push('\n');
3803 }
3804 std::fs::write(path, out)?;
3805 Ok(())
3806 }
3807
3808 /// Restore a conversation previously written with [`Self::save_transcript`],
3809 /// replacing the current history.
3810 pub fn load_transcript(&mut self, path: impl AsRef<std::path::Path>) -> Result<()> {
3811 let text = std::fs::read_to_string(path)?;
3812 let mut history = Vec::new();
3813 for line in text.lines().map(str::trim).filter(|l| !l.is_empty()) {
3814 history.push(serde_json::from_str::<ChatMessage>(line).map_err(Error::Decode)?);
3815 }
3816 self.history = history;
3817 Ok(())
3818 }
3819
3820 /// Take a checkpoint of the current conversation position. Pass it to
3821 /// [`Self::rewind_to`] to discard everything sent since (the rewind/undo
3822 /// analog of `fork`/checkpoint).
3823 pub fn checkpoint(&self) -> usize {
3824 self.history.len()
3825 }
3826
3827 /// Rewind the conversation to a [`Self::checkpoint`], discarding later turns.
3828 pub fn rewind_to(&mut self, checkpoint: usize) {
3829 self.history.truncate(checkpoint.min(self.history.len()));
3830 }
3831
3832 /// Send a message with file inputs attached — the `--file` / `-i` analog.
3833 /// Each file's contents are injected into the prompt: UTF-8 text inline,
3834 /// binary (e.g. images) noted with a size marker. (Native image *vision*
3835 /// would additionally require multimodal content parts.)
3836 pub async fn send_with_files(
3837 &mut self,
3838 text: impl Into<String>,
3839 files: &[std::path::PathBuf],
3840 ) -> Result<String> {
3841 let mut prompt = text.into();
3842 for path in files {
3843 let block = match std::fs::read(path) {
3844 Ok(bytes) => match String::from_utf8(bytes.clone()) {
3845 Ok(s) => format!("\n\n[file: {}]\n{}", path.display(), s),
3846 Err(_) => format!(
3847 "\n\n[file: {} — {} bytes, binary content omitted]",
3848 path.display(),
3849 bytes.len()
3850 ),
3851 },
3852 Err(e) => format!("\n\n[file: {} — could not read: {e}]", path.display()),
3853 };
3854 prompt.push_str(&block);
3855 }
3856 let expanded = self.expand_prompt_async(&prompt).await;
3857 let msg = ChatMessage::user(expanded);
3858 self.guard_candidate_message(&msg)?;
3859 self.record(&msg)?;
3860 self.history.push(msg);
3861 self.run_loop().await
3862 }
3863
3864 /// Send a message with image inputs to a vision model — the `-i/--image`
3865 /// analog. `image_urls` may be `https://…` links or `data:image/…;base64,…`
3866 /// URLs; they're attached as multimodal `image_url` content parts.
3867 pub async fn send_with_images(
3868 &mut self,
3869 text: impl Into<String>,
3870 image_urls: &[String],
3871 ) -> Result<String> {
3872 let expanded = self.expand_prompt_async(&text.into()).await;
3873 let msg = ChatMessage::user_with_images(expanded, image_urls);
3874 self.guard_candidate_message(&msg)?;
3875 self.record(&msg)?;
3876 self.history.push(msg);
3877 self.run_loop().await
3878 }
3879
3880 /// Expand a `/<name> <args>` slash command against the registered prompt
3881 /// templates (`{args}` is replaced with the trailing text). Non-matching
3882 /// input is returned unchanged.
3883 /// BP-6 additionally resolves SKILL.md invocations here, after the
3884 /// template table misses: `/skill:name args` (pi§2 "Skill commands"),
3885 /// `/name args` when the config follows Claude Code (cc§7: "a `SKILL.md`
3886 /// in a directory = a `/name` command"), and `$slug` mentions (cx§7).
3887 /// `$ARGUMENTS` in the body is replaced with the trailing text.
3888 pub fn expand_prompt(&self, input: &str) -> String {
3889 // BP-5 (catalog D2 "@-file mentions / attachments"): `@path`
3890 // expansion happens FIRST, so a mention works in a bare message, in
3891 // a slash-command's arguments, and in the text a `$slug` mention
3892 // appends to — one rule, every prompt shape.
3893 let input = &self.expand_file_mentions(input);
3894 let trimmed = input.trim_start();
3895 let Some(rest) = trimmed.strip_prefix('/') else {
3896 return self.expand_skill_mentions(input);
3897 };
3898 let (name, args) = match rest.split_once(char::is_whitespace) {
3899 Some((n, a)) => (n, a.trim()),
3900 None => (rest, ""),
3901 };
3902 match self.config.prompts.get(name) {
3903 Some(template) => template.replace("{args}", args),
3904 None => match self.expand_skill_command(name, args) {
3905 Some(expanded) => expanded,
3906 None => self.expand_skill_mentions(input),
3907 },
3908 }
3909 }
3910
3911 /// BP-5 (catalog D2 "@-file mentions / attachments"; cc§2 "`@` in the
3912 /// prompt triggers file-path autocomplete and injects file context …
3913 /// Read deny rules best-effort apply to `@file` mentions"; cx§2
3914 /// "`@`-mentions (files)"): replace each `@path` token in `input` with
3915 /// that file's contents.
3916 ///
3917 /// **Deny-rule aware, through the one permissions engine.** Each
3918 /// mention is resolved with
3919 /// [`crate::permissions::evaluate_path_safe`] — the same
3920 /// traversal/symlink-resolving check a `read_file` tool call goes
3921 /// through — against this config's own rules and protected-path floor.
3922 /// Anything short of `Allow` inlines the refusal instead of the file, so
3923 /// `@.env` under a preset whose protected paths cover it says so rather
3924 /// than quietly leaking it.
3925 ///
3926 /// A token that names nothing readable is left exactly as the user typed
3927 /// it: an email address, a decorator, or a `@`-prefixed word in prose is
3928 /// not a file mention, and must survive untouched.
3929 /// Off by default (`[core.file_mentions]`).
3930 fn expand_file_mentions(&self, input: &str) -> String {
3931 if !self.config.file_mentions || !input.contains('@') {
3932 return input.to_string();
3933 }
3934 let mut attachments = String::new();
3935 let mut seen: Vec<String> = Vec::new();
3936 for token in input.split_whitespace() {
3937 let Some(rel) = token.strip_prefix('@') else {
3938 continue;
3939 };
3940 let rel = rel.trim_end_matches([',', ';', ':', '.', ')', ']', '"', '\'']);
3941 if rel.is_empty() || seen.iter().any(|s| s == rel) {
3942 continue;
3943 }
3944 let path = if std::path::Path::new(rel).is_absolute() {
3945 std::path::PathBuf::from(rel)
3946 } else {
3947 self.config.cwd.join(rel)
3948 };
3949 if !path.is_file() {
3950 continue;
3951 }
3952 seen.push(rel.to_string());
3953 attachments.push_str(&self.render_mention(rel, &path));
3954 if seen.len() >= MAX_FILE_MENTIONS_PER_MESSAGE {
3955 break;
3956 }
3957 }
3958 if attachments.is_empty() {
3959 return input.to_string();
3960 }
3961 format!("{input}{attachments}")
3962 }
3963
3964 /// One mention's block: the permission verdict first, then the bytes.
3965 /// Text is inlined; a binary file is named with its size, the same
3966 /// shape [`Self::send_with_files`] already uses for an explicit
3967 /// attachment, so a mention and a `--file` read the same way.
3968 fn render_mention(&self, shown: &str, path: &std::path::Path) -> String {
3969 use crate::permissions::{Decision, PathKind};
3970 let rules = crate::permissions::rules_for_config(&self.config);
3971 // BP-10's multi-root form: a mention is checked against every
3972 // granted root (cwd + `additional_dirs`), folded to the strictest —
3973 // the same call the tool-dispatch gate makes for a `read_file`
3974 // path, so a mention can never reach a file a read could not.
3975 let mut roots = vec![self.config.cwd.clone()];
3976 roots.extend(self.config.additional_dirs.iter().cloned());
3977 let decision = crate::permissions::evaluate_path_safe_roots(
3978 &rules,
3979 PathKind::Read,
3980 &roots,
3981 &path.to_string_lossy(),
3982 Decision::Allow,
3983 );
3984 if decision != Decision::Allow {
3985 return format!(
3986 "\n\n[file: {shown} — not attached; the permission rules for this session \
3987 resolve reading it to {decision:?}]"
3988 );
3989 }
3990 match std::fs::read(path) {
3991 Ok(bytes) => match String::from_utf8(bytes) {
3992 Ok(text) => {
3993 let mut text = text;
3994 if text.len() > MAX_FILE_MENTION_BYTES {
3995 let mut cut = MAX_FILE_MENTION_BYTES;
3996 while cut > 0 && !text.is_char_boundary(cut) {
3997 cut -= 1;
3998 }
3999 text.truncate(cut);
4000 text.push_str("\n[file truncated]");
4001 }
4002 format!("\n\n[file: {shown}]\n{text}")
4003 }
4004 Err(e) => format!(
4005 "\n\n[file: {shown} — {} bytes, binary content omitted]",
4006 e.into_bytes().len()
4007 ),
4008 },
4009 Err(e) => format!("\n\n[file: {shown} — could not read: {e}]"),
4010 }
4011 }
4012
4013 /// The SKILL.md packages this agent discovered (frontmatter only) — the
4014 /// exact set its prompt index lists and its `skill` tool can load.
4015 pub fn skills(&self) -> &[crate::skills::LoopSkill] {
4016 &self.skills
4017 }
4018
4019 /// BP-6: resolve a slash command against the discovered skills.
4020 ///
4021 /// `/skill:<name>` is pi's own form and is accepted under every config
4022 /// (it can never collide with a template name, which cannot contain a
4023 /// colon-prefixed `skill` segment by construction). The BARE `/<name>`
4024 /// form is Claude Code's — there, a skill IS a slash command — so it is
4025 /// honored only when the config reads Claude Code's roots; under
4026 /// `cx-parity`, where Codex has no skill slash commands, `/deploy` stays
4027 /// the literal text the user typed.
4028 fn expand_skill_command(&self, name: &str, args: &str) -> Option<String> {
4029 if self.skills.is_empty() {
4030 return None;
4031 }
4032 let bare = match name.strip_prefix("skill:") {
4033 Some(rest) => rest,
4034 None if self.config.skills_harness.as_deref()
4035 == Some(crate::HarnessId::CLAUDE_CODE) =>
4036 {
4037 name
4038 }
4039 None => return None,
4040 };
4041 let skill = self.find_skill(bare)?;
4042 skill
4043 .body_with_shell(args, &self.shell_injection)
4044 .ok()
4045 .map(|body| crate::skills::render_skill(skill, &body))
4046 }
4047
4048 /// BP-6: `$slug` mentions (cx§7 `TOOL_MENTION_SIGIL = '$'`) and — only
4049 /// under `[core.skills] implicit_match` — a description match.
4050 ///
4051 /// The user's own text is never replaced: a loaded body is APPENDED, the
4052 /// way Codex splices a skill into the turn. Mentions are only honored
4053 /// for a config that reads Codex's roots; `$WORD` is ordinary shell text
4054 /// everywhere else.
4055 fn expand_skill_mentions(&self, input: &str) -> String {
4056 if self.skills.is_empty() {
4057 return input.to_string();
4058 }
4059 let mut loaded: Vec<String> = Vec::new();
4060 let mut names: Vec<String> = Vec::new();
4061 if self.config.skills_harness.as_deref() == Some(crate::HarnessId::CODEX) {
4062 for token in input.split_whitespace() {
4063 let Some(slug) = token.strip_prefix('$') else {
4064 continue;
4065 };
4066 let slug =
4067 slug.trim_matches(|c: char| !c.is_alphanumeric() && c != '-' && c != ':');
4068 if slug.is_empty() {
4069 continue;
4070 }
4071 let Some(skill) = self.find_skill(slug) else {
4072 continue;
4073 };
4074 if names.contains(&skill.name) || loaded.len() >= MAX_SKILL_LOADS_PER_MESSAGE {
4075 continue;
4076 }
4077 if let Ok(body) = skill.body_with_shell("", &self.shell_injection) {
4078 names.push(skill.name.clone());
4079 loaded.push(crate::skills::render_skill(skill, &body));
4080 }
4081 }
4082 }
4083 if loaded.is_empty() && self.config.skills_implicit_match {
4084 if let Some(skill) = crate::skills::implicit_skill_match(&self.skills, input) {
4085 if let Ok(body) = skill.body_with_shell("", &self.shell_injection) {
4086 loaded.push(crate::skills::render_skill(skill, &body));
4087 }
4088 }
4089 }
4090 if loaded.is_empty() {
4091 return input.to_string();
4092 }
4093 format!("{input}\n\n{}", loaded.join("\n\n"))
4094 }
4095
4096 /// Resolve one invocation name against the discovered set — the same
4097 /// resolver the `skill` tool uses, so every door agrees on what a name
4098 /// means.
4099 fn find_skill(&self, name: &str) -> Option<&crate::skills::LoopSkill> {
4100 crate::skills::find_skill(&self.skills, name)
4101 }
4102
4103 /// P5-2 (§2 module 15 D7 row 4 "prompts-as-commands"): like
4104 /// [`Self::expand_prompt`], but also consults MCP-server-sourced
4105 /// prompts registered via [`Self::register_mcp_prompt`] when the local
4106 /// `Config::prompts` table has no match — a live `prompts/get`
4107 /// round-trip, which is why this is async and [`Self::expand_prompt`]
4108 /// itself stays synchronous (its public sync signature is unchanged,
4109 /// for every existing caller that doesn't need MCP prompts).
4110 ///
4111 /// **Argument mapping (a scope decision, not a protocol requirement —
4112 /// the MCP spec leaves "how does free CLI text become named prompt
4113 /// arguments" to the client):** a prompt with zero or one declared
4114 /// arguments gets the whole trailing text (empty string if the prompt
4115 /// takes no arguments and none was given); a prompt with two or more
4116 /// declared arguments expects `key=value` pairs, whitespace-separated
4117 /// (`/mcp__server__prompt lang=rust topic=async`) — an unparseable pair
4118 /// (no `=`) is simply skipped, never a hard error (matches this
4119 /// method's "non-matching input passes through" fail-open posture for
4120 /// the LOCAL-prompt case above).
4121 pub async fn expand_prompt_async(&self, input: &str) -> String {
4122 let local = self.expand_prompt(input);
4123 if local != input {
4124 return local; // a local `Config::prompts` template matched
4125 }
4126 let trimmed = input.trim_start();
4127 let Some(rest) = trimmed.strip_prefix('/') else {
4128 return input.to_string();
4129 };
4130 let (name, args) = match rest.split_once(char::is_whitespace) {
4131 Some((n, a)) => (n, a.trim()),
4132 None => (rest, ""),
4133 };
4134 let Some(source) = self.mcp_prompts.get(name) else {
4135 return input.to_string();
4136 };
4137 let arg_map = match source.arg_names() {
4138 [] => std::collections::BTreeMap::new(),
4139 [single] => {
4140 let mut m = std::collections::BTreeMap::new();
4141 if !args.is_empty() {
4142 m.insert(single.clone(), args.to_string());
4143 }
4144 m
4145 }
4146 _ => args
4147 .split_whitespace()
4148 .filter_map(|pair| pair.split_once('='))
4149 .map(|(k, v)| (k.to_string(), v.to_string()))
4150 .collect(),
4151 };
4152 match source.render(arg_map).await {
4153 Ok(rendered) => rendered,
4154 Err(e) => format!("Error: mcp prompt `{name}` failed: {e}"),
4155 }
4156 }
4157
4158 /// P5-2 (§2 module 15 D7 row 4): register an MCP server's prompt as a
4159 /// slash-command source — `command_name` MUST already be the
4160 /// namespaced `mcp__<server>__<prompt>` form
4161 /// ([`crate::mcp::McpServerHandle::prompts`] produces exactly that
4162 /// shape); this method does not re-namespace or validate it, so a
4163 /// caller that hands it a bare name defeats the collision protection
4164 /// [`crate::mcp::McpPromptSource`]'s doc comment describes. Overwrites
4165 /// any prior registration under the same command name (re-attaching
4166 /// the same server replaces its own earlier prompt list; this can
4167 /// never touch a NON-`mcp__`-prefixed key, i.e. never a local
4168 /// `Config::prompts` entry).
4169 pub fn register_mcp_prompt(
4170 &mut self,
4171 command_name: impl Into<String>,
4172 source: impl crate::sdk::SdkPromptSource + 'static,
4173 ) {
4174 self.mcp_prompts
4175 .insert(command_name.into(), Box::new(source));
4176 }
4177
4178 /// P5-2 (§2 module 15 D7 row 5 "instructions"): fold an MCP server's
4179 /// `initialize`-time instructions (or any other free-text note) into
4180 /// this agent's system message — the context-assembly site every other
4181 /// `core.*`/`capabilities.*` prompt-section append already uses
4182 /// (`Self::with_parts`), except this one fires AFTER construction
4183 /// (attaching MCP servers happens once the agent already exists — see
4184 /// `crates/cli/src/main.rs`'s `attach_mcp`). A no-op if `history` is
4185 /// somehow empty or its first message isn't a system message (never
4186 /// true for an `Agent` built via `Self::new`/`Self::with_parts`, but
4187 /// checked rather than assumed).
4188 pub fn append_system_note(&mut self, text: &str) {
4189 if let Some(system) = self.history.first_mut() {
4190 if system.role == Role::System {
4191 system
4192 .content
4193 .get_or_insert_with(String::new)
4194 .push_str(text);
4195 }
4196 }
4197 }
4198
4199 /// BP-4 (catalog:90, cx§2 `<environment_context>` "re-emitted on
4200 /// change"): re-derive the `# Environment` block and, if anything in it
4201 /// moved — cwd, the approval/sandbox policy, the git branch or its
4202 /// dirty state, the date — replace the stale copy in the system message
4203 /// with the fresh one. Returns whether the block changed.
4204 ///
4205 /// A no-op (and free — no git subprocess) when `core.env_context` is
4206 /// off, which is the default and every non-parity config. Replacing in
4207 /// place rather than appending a second block is deliberate: two
4208 /// `# Environment` sections disagreeing about cwd is worse context than
4209 /// one stale one, and the system message is re-sent on every request,
4210 /// so the rewrite IS the re-emission the model sees.
4211 pub fn refresh_env_context(&mut self) -> bool {
4212 if !self.config.env_context {
4213 return false;
4214 }
4215 let fresh = env_context_block(&self.config);
4216 let Some(stale) = self.env_context_live.clone() else {
4217 // Nothing was spliced at construction (e.g. `with_provider_arc`);
4218 // splice it now rather than silently never emitting one.
4219 self.append_system_note(&fresh);
4220 self.env_context_live = Some(fresh);
4221 return true;
4222 };
4223 if stale == fresh {
4224 return false;
4225 }
4226 if let Some(system) = self.history.first_mut() {
4227 if system.role == Role::System {
4228 if let Some(content) = system.content.as_mut() {
4229 if let Some(at) = content.find(&stale) {
4230 content.replace_range(at..at + stale.len(), &fresh);
4231 self.env_context_live = Some(fresh);
4232 return true;
4233 }
4234 }
4235 }
4236 }
4237 false
4238 }
4239
4240 /// BP-5 (catalog D2 "Per-model-family base-prompt selection"): re-select
4241 /// the base prompt for the model now in force and replace the stale one
4242 /// in place. Returns whether the system message changed.
4243 ///
4244 /// A no-op — not even a string search — when the selection is unchanged,
4245 /// which is every config that sets no `base_prompts` table.
4246 fn refresh_base_prompt(&mut self) -> bool {
4247 let fresh = base_prompt_for_config(&self.config);
4248 if fresh == self.base_prompt_live {
4249 return false;
4250 }
4251 let stale = std::mem::replace(&mut self.base_prompt_live, fresh.clone());
4252 if stale.is_empty() {
4253 return false;
4254 }
4255 if let Some(system) = self.history.first_mut() {
4256 if system.role == Role::System {
4257 if let Some(content) = system.content.as_mut() {
4258 if let Some(at) = content.find(&stale) {
4259 content.replace_range(at..at + stale.len(), &fresh);
4260 return true;
4261 }
4262 }
4263 }
4264 }
4265 false
4266 }
4267
4268 /// BP-5: assemble the request from an already-built message list and
4269 /// tool-schema list. The ONE place a [`ChatRequest`] is constructed from
4270 /// this agent's config, so the request `Self::run_loop` issues and the
4271 /// request [`Self::model_input`] renders cannot drift apart.
4272 fn chat_request(&self, messages: Vec<ChatMessage>, tools: Vec<ToolSchema>) -> ChatRequest {
4273 let mut req = ChatRequest {
4274 model: self.config.model.clone(),
4275 messages,
4276 tools,
4277 temperature: self.config.temperature,
4278 max_tokens: self.config.max_tokens,
4279 effort: self.config.effort.clone(),
4280 response_format: self.config.response_format.clone(),
4281 service_tier: None,
4282 thinking_budget: None,
4283 extra_body: self.config.extra_body.clone(),
4284 };
4285 // BP-13 (catalog Domain 9): the per-request routing decisions —
4286 // effort LEVEL, thinking-token BUDGET and service TIER — all come
4287 // out of `Config::model_routing` keyed by the model this request is
4288 // actually going to. Applied HERE so `model_input`'s rendering and
4289 // the loop's own send can never disagree about what would be sent,
4290 // and so a mid-session switch re-decides all three for the new
4291 // model on the next pass.
4292 self.apply_routing(&mut req);
4293 req
4294 }
4295
4296 /// BP-5 (catalog D2 "Prompt-input debugging": *render the exact
4297 /// model-visible input for inspection*; cx§2 `codex debug prompt-input`,
4298 /// which "renders the exact model-visible input list as JSON"): the
4299 /// request this agent would send next.
4300 ///
4301 /// Built by the SAME two calls the loop makes
4302 /// ([`Self::build_request_messages`], [`Self::tool_schemas`]) and
4303 /// assembled by the SAME [`Self::chat_request`] — it is the real
4304 /// request, not a reconstruction of one. `&mut self` because
4305 /// `build_request_messages` is: rendering the input is exactly as
4306 /// stateful as building it for a send.
4307 pub fn model_input(&mut self) -> ChatRequest {
4308 let tools = self.tool_schemas();
4309 let messages = self.build_request_messages();
4310 self.chat_request(messages, tools)
4311 }
4312
4313 /// BP-5: [`Self::model_input`] for a turn that has not been sent —
4314 /// `prompt` is expanded exactly as [`Self::send`] would expand it
4315 /// (slash templates, skills, `@path` mentions, MCP prompts) and appended
4316 /// to the conversation IN MEMORY, then the request is rendered.
4317 ///
4318 /// Deliberately not recorded: this door inspects an input, it does not
4319 /// take a turn. Nothing is written to the session store, no journal
4320 /// entry is made, and no request is issued.
4321 pub async fn model_input_for(&mut self, prompt: &str) -> ChatRequest {
4322 let expanded = self.expand_prompt_async(prompt).await;
4323 self.history.push(ChatMessage::user(expanded));
4324 self.model_input()
4325 }
4326
4327 /// BP-5: a [`ChatRequest`] as the JSON a human (or `jq`) inspects — the
4328 /// system prompt, every message in order, and every advertised tool
4329 /// schema, plus the sampling controls that travel with them.
4330 pub fn render_model_input(req: &ChatRequest) -> serde_json::Value {
4331 serde_json::json!({
4332 "model": req.model,
4333 "temperature": req.temperature,
4334 "max_tokens": req.max_tokens,
4335 "effort": req.effort,
4336 "response_format": req.response_format,
4337 // Serialized through `ChatMessage`'s OWN wire serializer and
4338 // `ToolSchema`'s own — i.e. the exact bytes the provider is
4339 // handed, not a second rendering of them.
4340 "messages": serde_json::to_value(&req.messages).unwrap_or(serde_json::Value::Null),
4341 "tools": serde_json::to_value(&req.tools).unwrap_or(serde_json::Value::Null),
4342 })
4343 }
4344
4345 /// BP-5 (catalog D2 "Per-model-family base-prompt selection"): the base
4346 /// system prompt currently in force for this agent's model.
4347 pub fn base_prompt(&self) -> &str {
4348 &self.base_prompt_live
4349 }
4350
4351 /// BP-4 (catalog:91 "Synthetic context-injection blocks"): splice one
4352 /// named ambient block into the live context — the seam a hook's
4353 /// `additionalContext`, a frontend nudge or an orchestrator's brief
4354 /// enters through, mid-session, after construction.
4355 ///
4356 /// Requires `core.context_injections` (returns `false` otherwise): the
4357 /// gate governs the whole registry, not just its startup half. The
4358 /// block is appended to the system message and remembered, so it is
4359 /// carried by every later request and re-rendered by
4360 /// [`crate::context_injection::assemble`] wherever the prompt is
4361 /// rebuilt.
4362 pub fn inject_context_block(
4363 &mut self,
4364 name: impl Into<String>,
4365 content: impl Into<String>,
4366 ) -> bool {
4367 if !self.config.context_injections {
4368 return false;
4369 }
4370 let block = crate::config::ContextInjectionBlock::new(name, content);
4371 let rendered = crate::context_injection::render(std::slice::from_ref(&block));
4372 self.spliced_context_blocks.push(block);
4373 self.append_system_note(&rendered);
4374 true
4375 }
4376
4377 /// The blocks spliced in since construction — see
4378 /// [`Self::inject_context_block`].
4379 pub fn spliced_context_blocks(&self) -> &[crate::config::ContextInjectionBlock] {
4380 &self.spliced_context_blocks
4381 }
4382
4383 /// Compact the conversation if it has grown past the configured
4384 /// threshold.
4385 ///
4386 /// **Re-founded (A10):** with a [`ReductionPolicy`] installed
4387 /// ([`Self::set_reduction_policy`]), this no longer touches `self.history`
4388 /// at all. It derives `policy.clear_turns_older_than` from
4389 /// `compact_after_messages` so the *next* projected request view
4390 /// (`reduce::project_messages`, built in `Self::run_loop`) collapses the
4391 /// old turns into one reversible `TurnsCleared` stub instead —
4392 /// `history()` and the sidecar keep every message forever; only the view
4393 /// shrinks. Returns whether the live (unreduced) history currently
4394 /// exceeds the threshold, i.e. whether a clearing will actually be
4395 /// visible in the next projected view.
4396 ///
4397 /// **Legacy path (no policy) — LOSSY, kept only for byte-identical
4398 /// backward compatibility (D6):** destructively rewrites `self.history`,
4399 /// permanently discarding the dropped middle turns (replaced by a single
4400 /// non-reversible summary marker that becomes their SOLE remaining copy —
4401 /// exactly the lossy compaction this reduction layer differentiates
4402 /// against). Once a sidecar/recorder or a [`ReductionPolicy`] is in play,
4403 /// prefer installing a policy so this method takes the re-founded path
4404 /// above instead.
4405 pub fn maybe_compact(&mut self) -> bool {
4406 // P4e (§1.5/§3.1 `core.compaction.enabled`, "no master gate exists
4407 // yet"): checked FIRST, before either trigger — `false` disables
4408 // every auto-compaction trigger unconditionally (message-count AND
4409 // pressure), composing with them rather than replacing their own
4410 // logic. `true` (the default, matching today's pre-P4e behavior,
4411 // where nothing ever gated compaction) falls straight through to
4412 // the existing trigger checks below, unchanged.
4413 if !self.config.compaction_enabled {
4414 return false;
4415 }
4416 let threshold = self.config.compact_after_messages;
4417 // P4b (§1.5/§3.1 `core.compaction.reserve_tokens`, pi§2 shape): a
4418 // SECOND, independent trigger — context-window pressure — alongside
4419 // (not instead of) the message-count one above. `None` (the
4420 // default) is byte-identical to today's message-count-only
4421 // behavior; this whole block is a no-op then.
4422 let message_trigger = threshold.is_some_and(|t| self.history.len() > t);
4423 let pressure_trigger = self.compaction_pressure_triggered();
4424 if threshold.is_none() && self.config.compaction_reserve_tokens.is_none() {
4425 return false;
4426 }
4427 if !message_trigger && !pressure_trigger {
4428 return false;
4429 }
4430 if let Some(policy) = self.reduction_policy.as_mut() {
4431 if let Some(t) = threshold {
4432 policy.clear_turns_older_than = Some(t);
4433 }
4434 // P4b scope note: the token-PRESSURE trigger's "how much to
4435 // clear" derivation (below, for the legacy in-place path) has no
4436 // `ReductionPolicy`/A10 analog yet — that mechanism decides its
4437 // own clearing window once `clear_turns_older_than` is set, so
4438 // pressure firing alone (no message threshold configured) has
4439 // nothing new to hand it in this pass. Report the message-count
4440 // verdict only, matching today's pre-P4b behavior exactly when
4441 // only `threshold` is set.
4442 return message_trigger;
4443 }
4444 // Legacy in-place compaction (no `ReductionPolicy` installed) below.
4445 // `keep_recent`: the message-count trigger's own `threshold / 2`
4446 // shape when it's what fired (or both fired); otherwise (pressure
4447 // fired alone) a token-budget-derived count.
4448 let keep_recent = if message_trigger {
4449 (threshold.unwrap() / 2).max(2)
4450 } else {
4451 self.keep_recent_count_by_tokens()
4452 };
4453 self.compact_in_place(keep_recent, None)
4454 }
4455
4456 /// BP-4 (catalog:98 "Manual compact with focus instructions", cc§2 /
4457 /// cx§2 `/compact [instructions]`): compact NOW, regardless of whether
4458 /// either automatic trigger has fired — the mechanism behind the REPL's
4459 /// `/compact [focus]`.
4460 ///
4461 /// `focus` is this invocation's steering text: it overrides the standing
4462 /// `core.compaction.focus_instructions` for this compaction only, is
4463 /// carried into the SUMMARIZER's input (so the model-written summary
4464 /// preserves what the user asked for), and is stated on the marker. An
4465 /// empty/whitespace `focus` falls back to the configured standing value,
4466 /// which is what a bare `/compact` means.
4467 ///
4468 /// Returns whether anything was compacted (`false` when the history is
4469 /// already at or below the keep-window, or when a [`ReductionPolicy`] is
4470 /// installed — under a policy the reversible A10 path owns clearing, and
4471 /// a manual compact would be the lossy one).
4472 pub fn compact_now(&mut self, focus: Option<&str>) -> bool {
4473 self.compacting_manually = true;
4474 let compacted = self.compact_now_inner(focus);
4475 self.compacting_manually = false;
4476 compacted
4477 }
4478
4479 fn compact_now_inner(&mut self, focus: Option<&str>) -> bool {
4480 if self.reduction_policy.is_some() {
4481 return false;
4482 }
4483 // A manual compact must actually compact. The token budget alone
4484 // (`core.compaction.keep_recent_tokens`, 20k) keeps EVERYTHING on
4485 // any ordinary conversation, which is right for the pressure
4486 // trigger (it fires only when the window is nearly full) and wrong
4487 // for `/compact`, whose whole point is compacting before the
4488 // pressure arrives. So the keep-window is the tighter of the two:
4489 // the token budget, and the message-count trigger's own established
4490 // "keep the most recent half" shape (`maybe_compact`'s
4491 // `threshold / 2`, floor 2).
4492 let keep_recent = self
4493 .keep_recent_count_by_tokens()
4494 .min((self.history.len() / 2).max(2));
4495 let focus = focus.map(str::trim).filter(|f| !f.is_empty());
4496 self.compact_in_place(keep_recent, focus)
4497 }
4498
4499 /// The legacy (no-[`ReductionPolicy`]) in-place compaction both
4500 /// [`Self::maybe_compact`] and [`Self::compact_now`] run: collapse
4501 /// `history[first..cut)` into one marker, keeping the newest
4502 /// `keep_recent` messages.
4503 fn compact_in_place(&mut self, keep_recent: usize, focus_override: Option<&str>) -> bool {
4504 if self.history.len() <= keep_recent {
4505 return false;
4506 }
4507 // Indices: 0 is the system prompt; collapse [first .. len-keep_recent).
4508 // `first` is 1 (only the system prompt is ever auto-preserved) unless
4509 // B7's coordination clamp widens it.
4510 let mut first = 1usize;
4511 // B7 coordination clamp: this legacy (no-`ReductionPolicy`) path
4512 // mutates `self.history` directly, so — unlike the re-founded A10
4513 // path (clamped inside `reduce::project_messages`, threaded from
4514 // `build_request_messages`) — it must clamp itself. Widening `first`
4515 // (not `cut`) is what actually protects the imported prefix: the
4516 // drop range is `[first, cut)`, so raising `cut` alone would only
4517 // drop MORE messages, not fewer. `imported_prefix_len` is already an
4518 // absolute `history` index count (it protects `history[0..len]`), so
4519 // no offset conversion is needed here.
4520 if matches!(self.config.cache_plan, CachePlan::ImportedPrefix) {
4521 if let Some(protected) = self.imported_prefix_len {
4522 first = first.max(protected);
4523 }
4524 }
4525 let mut cut = self.history.len() - keep_recent;
4526 if cut <= first {
4527 return false;
4528 }
4529 // Never begin the kept window on a tool result: its originating
4530 // assistant turn (with the matching `tool_calls`) is about to be
4531 // dropped, which would orphan the tool message and make the replayed
4532 // conversation invalid. Advance past any leading tool results.
4533 while cut < self.history.len() && self.history[cut].role == Role::Tool {
4534 cut += 1;
4535 }
4536 if cut >= self.history.len() {
4537 return false;
4538 }
4539 let dropped = cut - first;
4540 // BP-11: the compaction is decided from here on — the one point both
4541 // the automatic triggers and `/compact` pass through — so this is
4542 // where `pre_compact` observers hear about it.
4543 self.fire_lifecycle(&crate::config::LifecycleEvent::PreCompact {
4544 messages: self.history.len(),
4545 dropped,
4546 manual: focus_override.is_some() || self.compacting_manually,
4547 });
4548 // P4b (§1.5/§3.1 `core.compaction.focus_instructions`, catalog D2
4549 // "no instruction steering" gap): appended to the marker whenever
4550 // set, regardless of which trigger fired. `None` (the default)
4551 // leaves this byte-identical to the pre-P4b marker text.
4552 //
4553 // BP-4: `focus_override` — the per-invocation `/compact <focus>`
4554 // text — wins over the standing config value for THIS compaction,
4555 // which is what "`/compact [instructions]` steers what's preserved"
4556 // means. Neither is required.
4557 let focus: Option<String> = focus_override.map(str::to_string).or_else(|| {
4558 self.config
4559 .compaction_focus_instructions
4560 .clone()
4561 .filter(|f| !f.is_empty())
4562 });
4563 // BP-1 (§1.5/§3.1 `core.compaction.summarize`): the verb the marker
4564 // uses is now the config's to state. `true` (the default, and what
4565 // every preset sets) keeps the historical "summarized" text
4566 // byte-identical; `false` says only what actually happened to the
4567 // span, so a config that turns summarization off does not leave a
4568 // marker claiming a summary exists.
4569 let verb = if self.config.compaction_summarize {
4570 "summarized"
4571 } else {
4572 "cleared"
4573 };
4574 // BP-4 (catalog:107 "LLM summaries of cleared spans", design §1.5:
4575 // obligation 5 is "auto-compaction … + a persisted marker + AN LLM
4576 // SUMMARY OF THE COMPACTED SPAN", knob `[core.compaction] summarize`
4577 // — "the summary side-call depends on a utility model … core falls
4578 // back to the main model"). The side-call is therefore CORE, not a
4579 // reduction-module privilege: when `core.compaction.summarize` is on
4580 // and a summarizer is installed, the span is summarized by the model
4581 // and the marker carries that summary instead of only a count.
4582 //
4583 // Every failure mode degrades to the count-only marker: no
4584 // summarizer installed, an `Err` from the side-call, or an empty
4585 // reply. It never blocks or fails compaction — the same contract
4586 // TR-7's own side-call site keeps.
4587 let summary_body = if self.config.compaction_summarize {
4588 self.summarize_span(first..cut, focus.as_deref())
4589 } else {
4590 None
4591 };
4592 // BP-4 (catalog:99 "Compaction markers persisted in transcript"):
4593 // the marker states where the originals went, which is the whole
4594 // point of a boundary record — a reader must be able to tell a
4595 // reversible compaction from a lossy one without knowing which
4596 // modules were on.
4597 let retention = if self.recorder.is_some() {
4598 "The compacted messages remain in this session's transcript sidecar."
4599 } else {
4600 "No transcript sidecar is attached, so this marker is the only remaining record of them."
4601 };
4602 let mut summary_text = format!(
4603 "[earlier conversation compacted: {dropped} message(s) {verb} to save context]\n{retention}"
4604 );
4605 if let Some(focus) = &focus {
4606 summary_text.push_str(&format!("\n\nFocus: {focus}"));
4607 }
4608 if let Some(body) = &summary_body {
4609 summary_text.push_str(&format!("\n\nSummary of the compacted span:\n{body}"));
4610 }
4611 let summary = ChatMessage::system(summary_text);
4612 // BP-4 (catalog:99): PERSIST the boundary. Before this the legacy
4613 // path rewrote `self.history` and never called `record`, so the
4614 // marker existed only in the live window and a resumed session had
4615 // no on-disk trace that a compaction ever happened. A recorder
4616 // failure is logged, never fatal — losing the boundary record must
4617 // not lose the compaction.
4618 if let Err(error) = self.record(&summary) {
4619 tracing::warn!("failed to persist the compaction marker: {error}");
4620 }
4621 let mut new_history = Vec::with_capacity(first + keep_recent + 2);
4622 new_history.extend(self.history[..first].iter().cloned());
4623 new_history.push(summary);
4624 new_history.extend(self.history.split_off(cut));
4625 self.history = new_history;
4626 self.fire_lifecycle(&crate::config::LifecycleEvent::PostCompact {
4627 messages: self.history.len(),
4628 dropped,
4629 });
4630 // BP-8 (catalog:150): compaction RESHAPES the live view rather than
4631 // appending to it, so the journal's "everything since the last
4632 // checkpoint is unpersisted" accounting has to be re-based here —
4633 // otherwise a crash-recovery replay would re-append messages this
4634 // compaction deliberately set aside. The set-aside messages' own
4635 // bytes stay in the log above, untouched.
4636 self.journal_checkpoint(self.history.len());
4637 true
4638 }
4639
4640 /// BP-4 (catalog:107): run the installed [`reduce::summarize::SpanSummarizer`]
4641 /// over `history[span]`, with `focus` (the `/compact <focus>` text)
4642 /// carried into the summarizer's INPUT so the model-written summary
4643 /// preserves what the user asked to keep.
4644 ///
4645 /// `None` — never an error — whenever no summarizer is installed, the
4646 /// span renders empty, the side-call fails, or it returns nothing. The
4647 /// caller falls back to the count-only marker.
4648 fn summarize_span(&self, span: std::ops::Range<usize>, focus: Option<&str>) -> Option<String> {
4649 let summarizer = self.span_summarizer.as_deref()?;
4650 let mut span_text = String::new();
4651 // The focus rides at the head of the span text (the trait's one
4652 // input) as an explicit, labeled line rather than a silent prompt
4653 // mutation: the fixed prompt's "do not state anything not present
4654 // in the span" still holds, because the focus IS present in it.
4655 if let Some(focus) = focus {
4656 span_text.push_str(&format!("[compaction focus requested: {focus}]\n\n"));
4657 }
4658 for msg in self.history.get(span)? {
4659 let role = match msg.role {
4660 Role::System => "system",
4661 Role::User => "user",
4662 Role::Assistant => "assistant",
4663 Role::Tool => "tool",
4664 };
4665 span_text.push_str(role);
4666 span_text.push_str(": ");
4667 span_text.push_str(msg.content.as_deref().unwrap_or(""));
4668 span_text.push('\n');
4669 }
4670 match summarizer.summarize(&span_text) {
4671 Ok(text) if !text.trim().is_empty() => Some(text.trim().to_string()),
4672 Ok(_) => None,
4673 Err(error) => {
4674 tracing::warn!("compaction span summarizer failed: {error}");
4675 None
4676 }
4677 }
4678 }
4679
4680 /// BP-4 (catalog:106 "Handoff (fresh objective + curated keep-set)",
4681 /// cx§1 `new_context`): reset the live working view to a fresh
4682 /// objective plus a curated keep-set, in-session.
4683 ///
4684 /// The new view is: the system prompt (plus any imported prefix a
4685 /// `CachePlan::ImportedPrefix` config protects — same clamp compaction
4686 /// uses), then a handoff marker stating the objective and what was set
4687 /// aside, then the most recent `keep_recent` messages (`None` = the
4688 /// token-budget-derived count `core.compaction.keep_recent_tokens`
4689 /// already governs, so the keep-set is curated by the same budget the
4690 /// rest of the compaction machinery uses, not by a magic number). The
4691 /// keep-set never begins on a tool result, so no tool message is left
4692 /// orphaned from its originating assistant turn.
4693 ///
4694 /// Returns how many messages were set aside. Like compaction, the
4695 /// marker is PERSISTED through the recorder, so a resumed session can
4696 /// see where the handoff happened; and like compaction, the set-aside
4697 /// messages remain in the transcript sidecar whenever one is attached.
4698 ///
4699 /// Scope note: this is the in-session `new_context` mechanism, NOT
4700 /// `Config::handoff_enabled`'s reversible ReductionLog snapshot (the
4701 /// offline `supercode handoff` projection) — that one is the reduction
4702 /// module's, and stays there.
4703 pub fn new_context(&mut self, objective: &str, keep_recent: Option<usize>) -> usize {
4704 let keep_recent = keep_recent.unwrap_or_else(|| self.keep_recent_count_by_tokens());
4705 let mut first = 1usize;
4706 if matches!(self.config.cache_plan, CachePlan::ImportedPrefix) {
4707 if let Some(protected) = self.imported_prefix_len {
4708 first = first.max(protected);
4709 }
4710 }
4711 let mut cut = self.history.len().saturating_sub(keep_recent).max(first);
4712 while cut < self.history.len() && self.history[cut].role == Role::Tool {
4713 cut += 1;
4714 }
4715 let dropped = cut.saturating_sub(first);
4716 let objective = objective.trim();
4717 let retention = if self.recorder.is_some() {
4718 "They remain in this session's transcript sidecar."
4719 } else {
4720 "No transcript sidecar is attached, so they are not retained."
4721 };
4722 let marker = ChatMessage::system(format!(
4723 "[handoff: a fresh working context starts here]\nObjective: {objective}\n\
4724 {dropped} earlier message(s) were set aside; the most recent {kept} were kept. \
4725 {retention}",
4726 kept = self.history.len() - cut,
4727 ));
4728 if let Err(error) = self.record(&marker) {
4729 tracing::warn!("failed to persist the handoff marker: {error}");
4730 }
4731 let mut new_history = Vec::with_capacity(first + keep_recent + 2);
4732 new_history.extend(self.history[..first].iter().cloned());
4733 new_history.push(marker);
4734 new_history.extend(self.history.split_off(cut));
4735 self.history = new_history;
4736 // BP-8: same re-basing as `maybe_compact` — see its comment.
4737 self.journal_checkpoint(self.history.len());
4738 dropped
4739 }
4740
4741 /// P4b (§1.5/§3.1 `core.compaction.reserve_tokens`, pi§2 shape:
4742 /// `contextTokens > contextWindow - reserveTokens`): whether the
4743 /// estimated token size of the live history is within `reserve_tokens`
4744 /// of the model's context window. `false` when
4745 /// [`Config::compaction_reserve_tokens`] is unset (the default).
4746 fn compaction_pressure_triggered(&self) -> bool {
4747 let Some(reserve) = self.config.compaction_reserve_tokens else {
4748 return false;
4749 };
4750 let limit = provider::model_context_limit(&self.config.model)
4751 .unwrap_or(provider::UNKNOWN_MODEL_CONTEXT_FLOOR);
4752 let used = supercode_runtime::estimate_view_tokens(&self.history);
4753 used.saturating_add(reserve) > limit
4754 }
4755
4756 /// P4b (§1.5/§3.1 `core.compaction.keep_recent_tokens`): how many of the
4757 /// most recent messages (walking backward from the end of `self.history`,
4758 /// skipping the system prompt) fit within the configured token budget
4759 /// (default 20,000, pi§6 precedent). Always keeps at least 2 messages,
4760 /// matching the message-count trigger's own floor.
4761 fn keep_recent_count_by_tokens(&self) -> usize {
4762 let budget = self.config.compaction_keep_recent_tokens.unwrap_or(20_000);
4763 let mut used = 0u64;
4764 let mut count = 0usize;
4765 for msg in self.history.iter().skip(1).rev() {
4766 let t = supercode_runtime::estimate_view_tokens(std::slice::from_ref(msg));
4767 if used.saturating_add(t) > budget && count > 0 {
4768 break;
4769 }
4770 used = used.saturating_add(t);
4771 count += 1;
4772 }
4773 count.max(2)
4774 }
4775
4776 /// Register an additional tool (e.g. your own capability).
4777 ///
4778 /// P5-2 (§2.2 C2 "connect invalidates cache prefix"): registering a
4779 /// tool AFTER this agent has already issued a request
4780 /// ([`Self::request_issued`]) changes the tools schema every
4781 /// subsequent request carries — the exact prefix-churn shape C2
4782 /// describes, MCP-sourced or not. Resets [`Self::cache_established`] so
4783 /// the next cache-warmth check (`provider::cache_cold_reason`) doesn't
4784 /// wrongly assume the entry is still warm. A no-op call before the
4785 /// first request (the common case: `attach_mcp` registers tools once at
4786 /// startup, before any turn runs) changes nothing — byte-identical to
4787 /// today.
4788 pub fn register_tool(&mut self, tool: impl crate::tools::Tool + 'static) {
4789 self.registry.register(tool);
4790 if self.requests_issued {
4791 self.cache_established = false;
4792 }
4793 }
4794
4795 /// The current conversation, including the system prompt.
4796 pub fn history(&self) -> &[ChatMessage] {
4797 &self.history
4798 }
4799
4800 /// Send a user message and run the loop until the model produces a final
4801 /// answer (text with no tool calls) or the iteration budget is exhausted.
4802 pub async fn send(&mut self, user_input: impl Into<String>) -> Result<String> {
4803 let expanded = self.expand_prompt_async(&user_input.into()).await;
4804 let msg = ChatMessage::user(expanded);
4805 self.guard_candidate_message(&msg)?;
4806 self.record(&msg)?;
4807 self.history.push(msg);
4808 self.run_loop().await
4809 }
4810
4811 /// BP-4 (catalog:109 "Context-usage introspection", cc§2 `/context`
4812 /// grid, cx§8 `/status` + `get_context_remaining`): the LIVE
4813 /// context-window accounting for this session — the same numbers
4814 /// `resume --dry-run`'s preflight already computes
4815 /// (`tokens::estimate_request_tokens` / `tokens::context_guard`), read
4816 /// out mid-session instead of only before one.
4817 ///
4818 /// Pure: it projects the request view exactly as
4819 /// [`Self::guard_candidate_message`] does (reduction stubs included,
4820 /// cache annotation included) without mutating the reduction log, so
4821 /// asking "how full am I?" can never change what the next request
4822 /// carries.
4823 pub fn context_usage(&self) -> ContextUsage {
4824 let messages = self.projected_view(None);
4825 let tools = self.tool_schemas();
4826 let message_tokens = supercode_runtime::estimate_view_tokens(&messages);
4827 let request_tokens = supercode_runtime::estimate_request_tokens(&messages, &tools);
4828 let limit = self.context_limit.or_else(|| {
4829 crate::provider::model_context_limit(&self.config.model)
4830 .or(Some(crate::provider::UNKNOWN_MODEL_CONTEXT_FLOOR))
4831 });
4832 let projected_tokens = supercode_runtime::with_guard_margin(request_tokens);
4833 let reserve = supercode_runtime::CONTEXT_RESPONSE_RESERVE_TOKENS;
4834 let (fits, remaining_tokens, used_pct) = match limit {
4835 Some(limit) => (
4836 projected_tokens.saturating_add(reserve) <= limit,
4837 limit
4838 .saturating_sub(reserve)
4839 .saturating_sub(projected_tokens),
4840 if limit == 0 {
4841 0
4842 } else {
4843 (projected_tokens as f64 / limit as f64 * 100.0).round() as u32
4844 },
4845 ),
4846 None => (true, 0, 0),
4847 };
4848 ContextUsage {
4849 model: self.config.model.clone(),
4850 messages: messages.len(),
4851 message_tokens,
4852 tool_count: tools.len(),
4853 tool_schema_tokens: request_tokens.saturating_sub(message_tokens),
4854 request_tokens,
4855 projected_tokens,
4856 response_reserve_tokens: reserve,
4857 context_limit: limit,
4858 remaining_tokens,
4859 used_pct,
4860 fits,
4861 }
4862 }
4863
4864 /// The messages a request would carry right now — the read-only half of
4865 /// [`Self::guard_candidate_message`]/[`Self::build_request_messages`],
4866 /// with `candidate` optionally appended as a not-yet-committed turn.
4867 /// Never mutates `self`.
4868 fn projected_view(&self, candidate: Option<&ChatMessage>) -> Vec<ChatMessage> {
4869 let messages = match &self.reduction_policy {
4870 None => {
4871 let mut messages = self.history.clone();
4872 if let Some(candidate) = candidate {
4873 messages.push(candidate.clone());
4874 }
4875 messages
4876 }
4877 Some(policy) => {
4878 let has_system = self.history.first().is_some_and(|m| m.role == Role::System);
4879 let mut reducible = self.history[usize::from(has_system)..].to_vec();
4880 if let Some(candidate) = candidate {
4881 reducible.push(candidate.clone());
4882 }
4883 let mut prepared = policy.clone();
4884 reduce::prepare_read_freshness(&mut prepared, &reducible);
4885 let (view, _) =
4886 reduce::project_messages(&reducible, &prepared, &self.reduction_log);
4887 let mut messages = Vec::with_capacity(view.len() + usize::from(has_system));
4888 if has_system {
4889 messages.push(self.history[0].clone());
4890 }
4891 messages.extend(view);
4892 messages
4893 }
4894 };
4895 provider::apply_cache_plan(&messages, self.config.cache_plan, self.imported_prefix_len)
4896 }
4897
4898 /// Refuse an oversized new user turn before it mutates canonical history
4899 /// or an attached sidecar. The in-loop guard remains authoritative for
4900 /// every actual request; this preflight closes the first-request seam
4901 /// where `send*` used to record/push the message before that guard ran.
4902 fn guard_candidate_message(&self, msg: &ChatMessage) -> Result<()> {
4903 let Some(limit) = self.context_limit else {
4904 return Ok(());
4905 };
4906
4907 let messages = self.projected_view(Some(msg));
4908 let tools = self.tool_schemas();
4909 let (fits, projected_tokens) = supercode_runtime::context_guard(&messages, &tools, limit);
4910 if !fits {
4911 return Err(Error::ContextLimitExceeded {
4912 projected_tokens,
4913 reserve_tokens: supercode_runtime::CONTEXT_RESPONSE_RESERVE_TOKENS,
4914 context_limit: limit,
4915 model: self.config.model.clone(),
4916 });
4917 }
4918 Ok(())
4919 }
4920
4921 /// The messages a provider request should carry for the CURRENT turn
4922 /// (A5/A7/A8/A10): with no [`ReductionPolicy`] installed, exactly
4923 /// `self.history.clone()` — byte-identical to every version of this
4924 /// method before reduction landed. With a policy installed, `history[0]`
4925 /// (this agent's own system prompt, never a reduction target) followed by
4926 /// [`reduce::project_messages`]'s projected view of `history[1..]`, fed
4927 /// with `self.reduction_log` so already-applied reductions reproduce
4928 /// verbatim across turns (prefix stability, A5) — the updated log is
4929 /// stored back onto `self` so the NEXT call (this turn, next turn, or a
4930 /// later `send`) sees the same accumulating state. `self.history` itself
4931 /// is never read back into or mutated by this: it stays the full
4932 /// canonical view, in lockstep with the sidecar (A3).
4933 ///
4934 /// When `policy.elide_stale_reads` is set, this re-runs
4935 /// [`reduce::probe_read_freshness`] (the one place A8's disk I/O happens)
4936 /// against `history[1..]` before projecting, so every request sees
4937 /// up-to-date freshness verdicts — `project_messages` itself stays pure.
4938 ///
4939 /// Finally, B7's [`provider::apply_cache_plan`] runs over the assembled
4940 /// view (regardless of whether a [`ReductionPolicy`] is installed) — a
4941 /// pure, cloning annotation step, so this method's `&mut self` mutations
4942 /// above (`self.reduction_log`) are already committed before it runs and
4943 /// its own output is never written back onto `self.history` or the log:
4944 /// purity for B7's cache breakpoints holds independently of A5's.
4945 fn build_request_messages(&mut self) -> Vec<ChatMessage> {
4946 let messages = match self.reduction_policy.clone() {
4947 None => self.history.clone(),
4948 Some(mut policy) => {
4949 reduce::prepare_read_freshness(&mut policy, &self.history[1..]);
4950 // B7 coordination clamp: while `CachePlan::ImportedPrefix` is
4951 // active, A10 turn-clearing must never establish a range
4952 // that dips into the imported prefix (protects the cache
4953 // breakpoint the request build will place there below).
4954 // `imported_prefix_len` counts `history[0]` (this agent's own
4955 // system message) plus the imported messages, but
4956 // `project_messages` only ever sees `history[1..]` — hence
4957 // the `- 1`.
4958 if matches!(self.config.cache_plan, CachePlan::ImportedPrefix) {
4959 policy.protect_imported_prefix =
4960 self.imported_prefix_len.map(|n| n.saturating_sub(1));
4961 }
4962 // TR-7 (T20): the one side-call site, run BEFORE
4963 // `project_messages` (which stays pure/I-O-free) — mirrors
4964 // `elide_stale_reads`/`probe_read_freshness` immediately
4965 // above. Only ever does anything when both the policy gate
4966 // AND a summarizer are present; either being absent means
4967 // `cleared_turns_summary` stays `None` and `project_messages`
4968 // renders the deterministic stub, same as before TR-7
4969 // existed.
4970 if policy.summarize_cleared_turns {
4971 if let Some(summarizer) = self.span_summarizer.as_deref() {
4972 policy.cleared_turns_summary = reduce::prepare_cleared_turns_summary(
4973 &self.history[1..],
4974 &policy,
4975 &self.reduction_log,
4976 summarizer,
4977 );
4978 }
4979 }
4980 let (view, log) =
4981 reduce::project_messages(&self.history[1..], &policy, &self.reduction_log);
4982 self.reduction_log = log;
4983 let mut messages = Vec::with_capacity(view.len() + 1);
4984 messages.push(self.history[0].clone());
4985 messages.extend(view);
4986 messages
4987 }
4988 };
4989 // TR-8 (T5): a tool-schema tier change since the last request is a
4990 // cache-bust event under `CachePlan::ImportedPrefix` — the `tools`
4991 // array is part of the cache key alongside `messages`, so flag it by
4992 // skipping this one request's cache annotation rather than claiming
4993 // a prefix hit that won't actually land. Recorded unconditionally
4994 // (even under `CachePlan::Off`) so the signature stays current
4995 // regardless of which plan is active.
4996 let tier_sig = self.schema_tier_signature();
4997 let busted =
4998 provider::tier_change_is_cache_bust(self.last_tool_schema_tier_signature, tier_sig);
4999 self.last_tool_schema_tier_signature = Some(tier_sig);
5000 let effective_cache_plan = if busted {
5001 CachePlan::Off
5002 } else {
5003 self.config.cache_plan
5004 };
5005 // UX-26 (B7-warn): mirror `apply_cache_plan`'s own placement gate
5006 // (`ImportedPrefix` AND a non-zero prefix) to know whether THIS
5007 // request will actually carry a `cache_control` annotation. `busted`
5008 // requests (schema-tier change) and `CachePlan::Off` never annotate,
5009 // so `provider::cache_cold_reason` can never flag them — there was
5010 // nothing to reuse, by construction. `idle_secs` is computed
5011 // whenever a signal exists at all (even before this agent's first
5012 // annotated send — see `Self::last_cache_activity_ms`'s doc comment
5013 // on why the pre-establishment case matters); `cache_established`
5014 // additionally gates the usage-ratio check specifically (see
5015 // `provider::cache_cold_reason`'s doc comment for why those two
5016 // checks need independent gates).
5017 let will_annotate = matches!(effective_cache_plan, CachePlan::ImportedPrefix)
5018 && self.imported_prefix_len.is_some_and(|n| n > 0);
5019 let idle_secs = self
5020 .last_cache_activity_ms
5021 .map(|last| (now_ms() - last).max(0) / 1000);
5022 self.pending_cache_turn = (will_annotate, self.cache_established, idle_secs);
5023 let mut messages =
5024 provider::apply_cache_plan(&messages, effective_cache_plan, self.imported_prefix_len);
5025 // BP-7 (catalog §4a "Goals"): the standing objective, restated at
5026 // the TAIL of the request — after the cache annotation, which sits
5027 // on the PREFIX, so a goal that changes mid-session never busts the
5028 // cached prefix. Request-view only: `history` is untouched, so the
5029 // persisted transcript is exactly the conversation and a translator
5030 // never has to invent a message for a harness-tracked goal.
5031 if let Some(goal) = &self.goal {
5032 messages.push(ChatMessage::system(goal.reminder()));
5033 }
5034 messages
5035 }
5036
5037 /// Run the model/tool loop over the current history until a final answer or
5038 /// the iteration budget is exhausted. (Shared by `send`, `send_with_files`,
5039 /// and `send_with_images`.)
5040 /// P4b (§1.7, pi§3 semantics): pop the next message(s) to deliver from
5041 /// `queue` per `mode` — `All` drains everything and joins it with a
5042 /// blank line, `OneAtATime` pops exactly one. `None` when `queue` is
5043 /// empty (the default state, at zero cost).
5044 fn drain_steer_queue(
5045 queue: &mut std::collections::VecDeque<String>,
5046 mode: SteeringMode,
5047 ) -> Option<String> {
5048 if queue.is_empty() {
5049 return None;
5050 }
5051 match mode {
5052 SteeringMode::All => Some(queue.drain(..).collect::<Vec<_>>().join("\n\n")),
5053 SteeringMode::OneAtATime => queue.pop_front(),
5054 }
5055 }
5056
5057 async fn run_loop(&mut self) -> Result<String> {
5058 let _steer_turn = SteerTurnGuard::new(self.steer_queue.clone());
5059 let mut output_tokens_used: u64 = 0;
5060
5061 // BP-7 (catalog §4a "Turn/budget caps"): the SPEND cap, checked
5062 // before this `send` can issue anything. Unlike
5063 // `max_total_output_tokens` (a per-`send` allowance, unchanged),
5064 // spend accumulates over the agent's whole lifetime — a dollar
5065 // budget that resets on every prompt is not a budget. A cap reached
5066 // MID-loop ends that loop cleanly with a `spend_budget` finish
5067 // marker (below); a cap already exhausted at entry is an error,
5068 // because there is nothing to return.
5069 if let Some(budget) = self.config.max_budget_usd.filter(|b| *b > 0.0) {
5070 if self.total_cost_usd >= budget {
5071 return Err(Error::BudgetExhausted {
5072 spent_usd: self.total_cost_usd,
5073 budget_usd: budget,
5074 });
5075 }
5076 }
5077
5078 // P5-9 (§2 module 20, cc's "per-prompt file-history-snapshot"):
5079 // open a fresh checkpoint for THIS turn — `run_loop` is called
5080 // exactly once per `send`/`send_with_files`/`send_with_images`
5081 // call (never recursively for the same turn), so this fires once
5082 // per user prompt, matching the design's per-prompt granularity.
5083 // `self.history.last()` is the user message that call just pushed.
5084 // `None` (`checkpoint_observer` unset, the default) is a no-op —
5085 // zero cost, no disk touched.
5086 if let Some(cp) = &self.checkpoint_observer {
5087 let label = self
5088 .history
5089 .last()
5090 .and_then(|m| m.content.as_deref())
5091 .unwrap_or("")
5092 .to_string();
5093 cp.begin_turn(&label);
5094 }
5095
5096 // BP-4 (catalog:90, cx§2 "re-emitted on change"): once per user
5097 // turn — not per loop iteration — re-derive the environment block
5098 // so a cwd change, an approval/sandbox policy change or a branch
5099 // switch since the last turn reaches the model instead of leaving
5100 // it reading the startup snapshot. A no-op, with no subprocess, for
5101 // every config that doesn't set `core.env_context`.
5102 self.refresh_env_context();
5103
5104 for _ in 0..self.config.max_iterations {
5105 // BP-8 (catalog:156): flush a plan `update_plan` wrote during
5106 // the previous iteration's tool calls. A no-op when
5107 // `todos.persist` is off or the plan did not change.
5108 self.journal_plan_if_changed();
5109 // BP-7: the index of the round-trip this iteration is about to
5110 // make. Captured here because `self.turn_index` advances the
5111 // moment the usage record is written, and every marker in this
5112 // iteration — including the ones written after that point —
5113 // must carry the SAME index, or the marker log would not join
5114 // to the usage log on `turn`.
5115 let round_trip = self.turn_index;
5116 self.maybe_compact();
5117 // P4b (§1.7, pi§3 "steer = after current tool calls"): drain any
5118 // queued mid-turn steering message(s) BEFORE building the next
5119 // request — the top of every loop iteration is exactly "after
5120 // whatever tool calls the previous iteration just ran" (or, on
5121 // the very first iteration, before anything has happened yet,
5122 // which is an equally valid "deliver immediately" reading).
5123 // Empty queue (today's default state) is a no-op.
5124 let (steer_msg, steer_taken) = {
5125 let mut inbox = self
5126 .steer_queue
5127 .lock()
5128 .unwrap_or_else(std::sync::PoisonError::into_inner);
5129 let before = inbox.len();
5130 let drained = inbox.drain(self.config.steering_mode);
5131 let taken = before - inbox.len();
5132 (drained, taken)
5133 };
5134 if let Some(steer_msg) = steer_msg {
5135 // BP-8 (catalog:154): the queue record's other half —
5136 // without it a replayed journal would keep re-delivering an
5137 // input the conversation already consumed.
5138 self.journal_queue_drain(crate::session_journal::QueueKind::Steer, steer_taken);
5139 let msg = ChatMessage::user(steer_msg);
5140 self.record(&msg)?;
5141 self.history.push(msg);
5142 }
5143 // Recomputed every iteration (not hoisted): under `Deferred`
5144 // advertising, a `tool_search` call earlier in this same loop
5145 // activates tools that must be advertised starting with the very
5146 // next request (B6).
5147 let tools = self.tool_schemas();
5148 let messages = self.build_request_messages();
5149
5150 // PARITY-18 D4 — re-check the context guard before EVERY
5151 // request this loop builds, not just the caller's one-shot
5152 // preflight: interactive turns 2+, `/expand all`, and any
5153 // mid-loop tool round-trip that grows `messages` can push a
5154 // barely-passing session over the limit between sends. Only
5155 // armed when a caller has opted in via `set_context_limit`.
5156 // Uses the exact same `tokens::context_guard`
5157 // formula the CLI preflight uses, so the two can never disagree.
5158 if let Some(limit) = self.context_limit {
5159 let (fits, projected_tokens) =
5160 supercode_runtime::context_guard(&messages, &tools, limit);
5161 if !fits {
5162 return Err(Error::ContextLimitExceeded {
5163 projected_tokens,
5164 reserve_tokens: supercode_runtime::CONTEXT_RESPONSE_RESERVE_TOKENS,
5165 context_limit: limit,
5166 model: self.config.model.clone(),
5167 });
5168 }
5169 }
5170
5171 // BP-7 (catalog §4a "Turn/step bracketing records"): the
5172 // OPENING bracket, written before the request is issued so it
5173 // survives a request that never returns (a cancelled turn keeps
5174 // its `context` marker with no `usage`/`finish` after it).
5175 // Uses `tokens::estimate_request_tokens` — the same estimator
5176 // the context guard above uses, so the two can never disagree.
5177 self.push_turn_marker_at(
5178 round_trip,
5179 crate::turn_record::TurnMarker::Context {
5180 messages: messages.len(),
5181 tools: tools.len(),
5182 estimated_tokens: supercode_runtime::estimate_request_tokens(&messages, &tools),
5183 },
5184 );
5185
5186 let mut req = self.chat_request(messages, tools);
5187
5188 let fallback_hops: Vec<FallbackHop>;
5189 let completion = {
5190 let sink = self.config.event_sink.as_ref();
5191 let on_delta = move |s: &str| {
5192 if let Some(sink) = sink {
5193 sink(AgentEvent::TextDelta(s.to_string()));
5194 }
5195 };
5196 // PARITY-18 D3 — the real send site: flip the flag
5197 // immediately before issuing the request, regardless of
5198 // whether `complete` then succeeds or fails, so
5199 // `request_issued()` truthfully reflects "a live request
5200 // was attempted" rather than "the run reached this line and
5201 // later succeeded."
5202 self.requests_issued = true;
5203 // P4b (§1.1/§3.1 `core.retry`, pi§3 shape): retry-with-
5204 // backoff already lives at the TRANSPORT layer
5205 // (`provider::OpenAiProvider::send_with_retry`, pre-existing
5206 // — connection failures and 5xx responses are retried
5207 // there); `Config.retry_*` (see `Agent::new`) makes that
5208 // EXISTING mechanism config-file-settable instead of
5209 // duplicating a second retry loop here, which would nest
5210 // retries confusingly on top of the transport's own.
5211 // BP-13 (D9 "Failure fallback model chains"): the chain is
5212 // EXECUTED here, not merely resolved. On a failure another
5213 // model could plausibly answer (overload / rate limit /
5214 // unavailability — `is_failover_worthy`), the request is
5215 // re-sent against the next entry of
5216 // `Config::model_fallback`, with routing re-applied for
5217 // that model. A 4xx that is not a rate limit is the caller's
5218 // problem, not the model's, and is never retried elsewhere.
5219 // The transport-level retry above has already run and given
5220 // up by the time a hop is considered.
5221 let (result, hops) = self.complete_with_fallback(&mut req, &on_delta).await;
5222 fallback_hops = hops;
5223 result
5224 };
5225 // BP-7: drained whether the request succeeded or failed, and
5226 // BEFORE the `?` — a request that exhausted its retries and
5227 // then errored is exactly the case a retry record exists for.
5228 for notice in self.retry_log.drain() {
5229 self.emit(AgentEvent::ProviderRetry {
5230 attempt: notice.attempt,
5231 delay_ms: notice.delay_ms,
5232 reason: notice.reason.clone(),
5233 });
5234 self.push_turn_marker_at(
5235 round_trip,
5236 crate::turn_record::TurnMarker::Retry {
5237 attempt: notice.attempt,
5238 delay_ms: notice.delay_ms,
5239 reason: notice.reason,
5240 },
5241 );
5242 }
5243 // The switch the fallback pass performed is a real mid-session
5244 // model change: it moves `Config::model` for every subsequent
5245 // request and is recorded exactly like a user-driven `/model`
5246 // switch (typed record + journal line), never as a silent retry.
5247 for hop in fallback_hops {
5248 self.record_model_change(&hop.from, &hop.to, Some(hop.reason.as_str()));
5249 }
5250 let (mut assistant, usage) = completion?;
5251 // Persist the actual generating model on the message itself.
5252 // A resumed foreign session keeps its original model in
5253 // `SessionMeta`; using only that session-level value on export
5254 // misattributes every Supercode continuation turn to the source
5255 // harness model. Per-message provenance lets native exporters
5256 // preserve the boundary accurately (for example, Claude history
5257 // followed by a GLM continuation).
5258 assistant
5259 .metadata
5260 .insert("model".to_string(), self.config.model.clone());
5261 output_tokens_used += usage.completion_tokens;
5262 self.total_output_tokens += usage.completion_tokens;
5263
5264 // UX-26 (B7-warn): consult the verdict computed at build time
5265 // (before this request was sent) now that `usage` — the only
5266 // piece that couldn't be known pre-send — is in hand. Gated on
5267 // `Config::cache_warnings` (default on; `--no-cache-warnings` /
5268 // `SUPERCODE_CACHE_WARNINGS=0` at the CLI layer, dev/03) so this
5269 // stays a zero-behavior-change no-op for every caller that
5270 // hasn't opted into `CachePlan::ImportedPrefix` in the first
5271 // place (`pending_cache_turn.0` is `false` whenever
5272 // `CachePlan::Off`, so the predicate always returns `None` then
5273 // regardless of this flag).
5274 let (will_annotate, cache_established, idle_secs) = self.pending_cache_turn;
5275 // UX-26 T2 (accuracy fold-in): `CacheColdReason::message` asserts
5276 // Anthropic-specific facts (a fixed 5-minute ephemeral TTL, and
5277 // cache-read-ratio semantics that assume Anthropic's exact-count
5278 // billing) that are only true for Anthropic-family models. This
5279 // is a WARNING-only gate, deliberately not folded into
5280 // `will_annotate`/the breakpoint-placement gate above: whether a
5281 // `cache_control` breakpoint is safe/inert to send to a
5282 // non-Anthropic model through OpenRouter is a separate cache-
5283 // behavior question this ticket doesn't touch (see
5284 // `.volter/tracker/markdown/UX-26.md`'s T2 note) — narrowing only
5285 // the warning keeps this fix scoped to warning ACCURACY, with
5286 // zero change to what gets sent on the wire.
5287 let warning_applies_to_this_model =
5288 provider::is_anthropic_family_model(&self.config.model);
5289 if self.config.cache_warnings && warning_applies_to_this_model {
5290 if let Some(reason) =
5291 provider::cache_cold_reason(will_annotate, cache_established, idle_secs, &usage)
5292 {
5293 self.emit(AgentEvent::CacheWarning {
5294 message: reason.message(),
5295 });
5296 }
5297 }
5298 // Refresh the activity clock / establish-once flag for the NEXT
5299 // turn's comparison, but only when THIS request actually carried
5300 // the annotation — an unannotated (busted/Off) request neither
5301 // warms nor cools a cache entry it never touched.
5302 if will_annotate {
5303 self.last_cache_activity_ms = Some(now_ms());
5304 self.cache_established = true;
5305 }
5306
5307 // UX-23: emitted before `TurnCompleted` so a `--trace`/
5308 // `stream-json` consumer sees "this round-trip cost N tokens"
5309 // land right alongside the round-trip it describes, rather than
5310 // needing to correlate it with a later event.
5311 self.emit(AgentEvent::Usage(usage.clone()));
5312 self.emit(AgentEvent::TurnCompleted);
5313 // P4b (§1.6, catalog §4a "persisted per-turn usage records"):
5314 // EventSink already streamed `Usage` above — this durably
5315 // accumulates the same data as a typed record (see
5316 // `Self::usage_records`/`Self::save_usage_log`), never a lossy
5317 // display-only channel.
5318 // BP-7 (catalog §4a "Per-turn cost/usage accounting"): the
5319 // record now carries the round-trip's DOLLAR cost too — the
5320 // half the row's semantics name alongside tokens — whenever
5321 // this build can price the model.
5322 // BP-13 (D9 "Model-served-vs-requested provenance"): the record
5323 // now carries BOTH sides — the model this agent asked for and,
5324 // when the provider reported one, the model that actually
5325 // answered. They can genuinely differ (a gateway aliasing a
5326 // name to a dated snapshot, a fallback hop, a routed tier), and
5327 // a record that can only ever state the request cannot show it.
5328 let served = assistant
5329 .metadata
5330 .get(crate::provider::SERVED_MODEL_KEY)
5331 .cloned();
5332 let record = crate::usage_log::UsageRecord::from_usage(
5333 self.turn_index,
5334 &self.config.model,
5335 &usage,
5336 now_ms(),
5337 )
5338 .priced(self.model_price)
5339 .with_served_model(served);
5340 self.total_cost_usd += record.cost_usd.unwrap_or(0.0);
5341 // BP-7: the round-trip's usage bracket, written from the same
5342 // point as the usage record so the two logs never disagree.
5343 self.push_turn_marker_at(
5344 round_trip,
5345 crate::turn_record::TurnMarker::Usage {
5346 prompt_tokens: record.prompt_tokens,
5347 completion_tokens: record.completion_tokens,
5348 total_tokens: record.total_tokens,
5349 cached_tokens: record.cached_tokens,
5350 cost_usd: record.cost_usd,
5351 },
5352 );
5353 self.journal_usage(&record);
5354 self.usage_log.push(record);
5355 self.turn_index += 1;
5356 self.record(&assistant)?;
5357 self.history.push(assistant.clone());
5358
5359 let calls = assistant.tool_calls().to_vec();
5360 if calls.is_empty() {
5361 // Close steering acceptance under the same lock as the last
5362 // drain. A message accepted before this boundary extends the
5363 // current turn; anything later is rejected by the SDK and
5364 // can never leak into a future turn.
5365 let (steer_msg, steer_taken) = {
5366 let mut inbox = self
5367 .steer_queue
5368 .lock()
5369 .unwrap_or_else(std::sync::PoisonError::into_inner);
5370 let before = inbox.len();
5371 let drained = inbox.drain_or_close(self.config.steering_mode);
5372 let taken = before - inbox.len();
5373 (drained, taken)
5374 };
5375 if let Some(steer_msg) = steer_msg {
5376 self.journal_queue_drain(crate::session_journal::QueueKind::Steer, steer_taken);
5377 let msg = ChatMessage::user(steer_msg);
5378 self.record(&msg)?;
5379 self.history.push(msg);
5380 continue;
5381 }
5382 // P4b (§1.7, pi§3 "follow-up = at idle"): a queued follow-up
5383 // message takes priority over the stop-gate — it's more
5384 // input to answer, not a veto of an answer already given.
5385 let follow_up_before = self.follow_up_queue.len();
5386 if let Some(follow_up_msg) =
5387 Self::drain_steer_queue(&mut self.follow_up_queue, self.config.follow_up_mode)
5388 {
5389 self.journal_queue_drain(
5390 crate::session_journal::QueueKind::FollowUp,
5391 follow_up_before - self.follow_up_queue.len(),
5392 );
5393 let msg = ChatMessage::user(follow_up_msg);
5394 self.record(&msg)?;
5395 self.history.push(msg);
5396 continue;
5397 }
5398 // P4b (§1.9/§3.1 `[core] stop_gate`, D3 "stop/completion
5399 // gating"): consulted exactly once per iteration that would
5400 // otherwise return — computed into an owned `Option<String>`
5401 // so the immutable borrow of `self.config.stop_gate` ends
5402 // before the `self.record`/`self.history.push` calls below
5403 // need `&mut self`.
5404 let final_content = assistant.content.clone().unwrap_or_default();
5405 let veto_reason: Option<String> = self
5406 .config
5407 .stop_gate
5408 .as_ref()
5409 .and_then(|gate| gate(&final_content));
5410 if let Some(reason) = veto_reason {
5411 let msg = ChatMessage::user(reason);
5412 self.record(&msg)?;
5413 self.history.push(msg);
5414 continue;
5415 }
5416 // BP-8 (catalog:156): the last iteration's tool calls are
5417 // the ones the top-of-loop flush above never sees.
5418 self.journal_plan_if_changed();
5419 self.push_turn_marker_at(
5420 round_trip,
5421 crate::turn_record::TurnMarker::Finish {
5422 reason: crate::turn_record::FinishReason::EndTurn,
5423 },
5424 );
5425 return Ok(assistant.content.unwrap_or_default());
5426 }
5427
5428 // BP-7: this round-trip ended by asking for tool calls; the
5429 // loop continues. The budget arms below mark the LOOP's end
5430 // separately when one of them stops it here.
5431 self.push_turn_marker_at(
5432 round_trip,
5433 crate::turn_record::TurnMarker::Finish {
5434 reason: crate::turn_record::FinishReason::ToolCalls,
5435 },
5436 );
5437
5438 // Output-token budget (output only — input tokens are not counted,
5439 // so this does not bound cost): stop spawning further model turns
5440 // once the cumulative output-token budget for this `send` is
5441 // exhausted.
5442 if let Some(budget) = self.config.max_total_output_tokens {
5443 if output_tokens_used >= budget {
5444 // The assistant turn we just pushed carries unanswered
5445 // tool_calls. Leaving them dangling yields an invalid
5446 // history (assistant tool_calls with no tool results) that
5447 // the provider rejects on the next `send`/resume. Emit
5448 // synthetic results so the transcript stays well-formed.
5449 for call in &calls {
5450 let msg = ChatMessage::tool_result(
5451 call.id.clone(),
5452 call.function.name.clone(),
5453 "[skipped: output token budget reached]".to_string(),
5454 );
5455 self.record(&msg)?;
5456 self.history.push(msg);
5457 }
5458 self.push_turn_marker_at(
5459 round_trip,
5460 crate::turn_record::TurnMarker::Finish {
5461 reason: crate::turn_record::FinishReason::OutputTokenBudget,
5462 },
5463 );
5464 return Ok(assistant.content.clone().unwrap_or_default());
5465 }
5466 }
5467
5468 // BP-7 (catalog §4a "Turn/budget caps"): the SPEND and STEP
5469 // caps, at the same point and with the same shape as the
5470 // output-token cap above — checked before this turn's tool
5471 // calls run, with synthetic results so the transcript stays
5472 // well-formed for a resume.
5473 let spend_exhausted = self
5474 .config
5475 .max_budget_usd
5476 .is_some_and(|b| b > 0.0 && self.total_cost_usd >= b);
5477 let steps_exhausted = self
5478 .config
5479 .max_steps
5480 .is_some_and(|n| n > 0 && self.total_steps + calls.len() > n);
5481 if spend_exhausted || steps_exhausted {
5482 let (label, reason) = if spend_exhausted {
5483 (
5484 "[skipped: spend budget reached]",
5485 crate::turn_record::FinishReason::SpendBudget,
5486 )
5487 } else {
5488 (
5489 "[skipped: step budget reached]",
5490 crate::turn_record::FinishReason::StepBudget,
5491 )
5492 };
5493 for call in &calls {
5494 let msg = ChatMessage::tool_result(
5495 call.id.clone(),
5496 call.function.name.clone(),
5497 label.to_string(),
5498 );
5499 self.record(&msg)?;
5500 self.history.push(msg);
5501 }
5502 self.push_turn_marker_at(
5503 round_trip,
5504 crate::turn_record::TurnMarker::Finish { reason },
5505 );
5506 return Ok(assistant.content.clone().unwrap_or_default());
5507 }
5508 self.total_steps += calls.len();
5509
5510 // P4e (§3.1 `core.parallel_tool_calls`, catalog:59): off (the
5511 // default) or a single call takes the EXACT pre-P4e sequential
5512 // path below, byte-identical. Only `true` with 2+ calls in this
5513 // turn takes `Self::run_tools_concurrently` — see its doc
5514 // comment for exactly what does and doesn't run concurrently.
5515 if self.config.parallel_tool_calls && calls.len() > 1 {
5516 for call in &calls {
5517 self.emit(AgentEvent::tool_started(call));
5518 }
5519 let results = self.run_tools_concurrently(&calls).await;
5520 for (call, (output, is_error)) in calls.iter().zip(results) {
5521 self.emit(AgentEvent::ToolCallCompleted {
5522 id: call.id.clone(),
5523 name: call.function.name.clone(),
5524 output: output.clone(),
5525 is_error,
5526 });
5527 self.apply_tool_result(call, output, is_error)?;
5528 }
5529 } else {
5530 for call in &calls {
5531 self.emit(AgentEvent::tool_started(call));
5532 let (output, is_error) = self.run_tool(call).await;
5533 self.emit(AgentEvent::ToolCallCompleted {
5534 id: call.id.clone(),
5535 name: call.function.name.clone(),
5536 output: output.clone(),
5537 is_error,
5538 });
5539 self.apply_tool_result(call, output, is_error)?;
5540 }
5541 }
5542 // BP-3 (catalog row "Context-budget tools"): a `new_context`
5543 // call parks its request on the shared budget; this is where
5544 // the agent — the one owner of `history` — applies it, so the
5545 // NEXT request built by this loop is already the fresh window.
5546 // No parked request (every session that never calls the tool)
5547 // is a single `Option` check.
5548 self.apply_pending_new_context();
5549 }
5550
5551 self.push_turn_marker_at(
5552 self.turn_index.saturating_sub(1),
5553 crate::turn_record::TurnMarker::Finish {
5554 reason: crate::turn_record::FinishReason::MaxIterations,
5555 },
5556 );
5557 Err(Error::MaxIterations(self.config.max_iterations))
5558 }
5559
5560 /// BP-3: apply a parked [`crate::tools::NewContextRequest`], if any.
5561 ///
5562 /// The rewrite itself is BP-4's [`Self::new_context`] — the SAME
5563 /// mechanism the operator's `/handoff` runs, so the model's door and the
5564 /// human's door can never drift into two different notions of "a fresh
5565 /// window". This function is only the hand-off point between the tool
5566 /// that asked and the agent that owns `history`.
5567 fn apply_pending_new_context(&mut self) {
5568 let Some(request) = self.ctx.context_budget.take_new_context() else {
5569 return;
5570 };
5571 self.new_context(&request.objective, request.keep_recent);
5572 }
5573
5574 /// The exact post-execution handling every tool result gets, regardless
5575 /// of whether it was produced by the sequential loop or
5576 /// [`Self::run_tools_concurrently`] — factored out of `Self::run_loop`'s
5577 /// tool-dispatch section (P4e) so both paths share one copy: multimodal
5578 /// image-marker detection, A7 output capping (gated exactly as before),
5579 /// TR-10 error stamping, and the `record`/`history` append. Always
5580 /// called in ORIGINAL call order, one call at a time, so the lossless
5581 /// sidecar's append-order invariant (S1.13) holds regardless of which
5582 /// dispatch path produced the result.
5583 fn apply_tool_result(
5584 &mut self,
5585 call: &supercode_interchange::ToolCall,
5586 output: String,
5587 is_error: bool,
5588 ) -> Result<()> {
5589 // P4c (§1.2 `core.tools.read_file.multimodal` / `view_image`):
5590 // a successful tool result carrying the image-data-URL
5591 // marker becomes a `content_parts` image block instead of
5592 // plain text — checked BEFORE `cap_tool_output` (a data URL
5593 // is not meaningfully "capped" by a byte-length text notice)
5594 // and recorded identically on both the full and history
5595 // copies, mirroring `ImageRedacted`'s "images are their own
5596 // axis, orthogonal to A7 text truncation" treatment
5597 // (reduce.rs). An ERRORED call never carries the marker (a
5598 // tool only emits it on success), so `is_error` is not
5599 // re-checked here.
5600 if let Some(data_url) = output.strip_prefix(crate::tools::MULTIMODAL_IMAGE_MARKER) {
5601 let notice = format!("[{}: image content attached below]", call.function.name);
5602 let full_result = ChatMessage::tool_result_with_image(
5603 call.id.clone(),
5604 call.function.name.clone(),
5605 notice.clone(),
5606 data_url.to_string(),
5607 );
5608 let hist_result = ChatMessage::tool_result_with_image(
5609 call.id.clone(),
5610 call.function.name.clone(),
5611 notice,
5612 data_url.to_string(),
5613 );
5614 self.record(&full_result)?;
5615 self.history.push(hist_result);
5616 return Ok(());
5617 }
5618 // Record the FULL output before capping (A3): what the
5619 // sidecar keeps must never be the already-lossy, truncated
5620 // copy (#8/#40) — `history` alone governs what shrinks.
5621 let mut full_result =
5622 ChatMessage::tool_result(call.id.clone(), call.function.name.clone(), output.clone());
5623 // D6/A7 supersession gate (TR-12 land-blocker fix): `history`
5624 // is the exact slice `reduce::project_messages` mints A7/A10
5625 // reduction hashes from (`Self::build_request_messages`
5626 // below). Capping it here — as this unconditionally used to
5627 // do — would silently shrink the bytes those hashes cover, so
5628 // a hash minted now could never recompute the same way once
5629 // the sidecar is reloaded from disk later (`verify_log`/
5630 // `invert`, offline). Gate `cap_tool_output` off in exactly
5631 // the combination where reductions can be minted over
5632 // `history` AND the full bytes are durably retained: a
5633 // recorder AND a `ReductionPolicy` both installed. A7 then
5634 // owns tool-output bounding, reversibly, at projection time
5635 // (SPEC.md D6/A7) — `history`/the sidecar keep everything,
5636 // only the request view shrinks. With a policy but no
5637 // recorder (constructible via `set_reduction_policy` alone),
5638 // nothing durable backs the full bytes, so capping stays on —
5639 // the same honest-labeling spirit as `cap_tool_output`'s own
5640 // retention branch below, just applied at the gate instead of
5641 // the notice text. With no policy at all, this is untouched:
5642 // today's byte-identical legacy cap.
5643 let for_history = if self.recorder.is_some() && self.reduction_policy.is_some() {
5644 output
5645 } else {
5646 self.cap_tool_output(output)
5647 };
5648 let mut hist_result =
5649 ChatMessage::tool_result(call.id.clone(), call.function.name.clone(), for_history);
5650 if is_error {
5651 // TR-10: the reduction layer's success/failure boundary
5652 // (`ReductionKind::ToolInputElided` must never target an
5653 // errored call — TR-6's territory) has no other
5654 // structural signal on `ChatMessage`; stamp both the
5655 // recorded copy (so it survives a sidecar round-trip via
5656 // `NativeTurn`) and the live-history copy (so an
5657 // in-process `project_messages` sees it immediately).
5658 reduce::mark_tool_error(&mut full_result);
5659 reduce::mark_tool_error(&mut hist_result);
5660 }
5661 self.record(&full_result)?;
5662 self.history.push(hist_result);
5663 Ok(())
5664 }
5665
5666 /// Truncate an oversized tool result so a single runaway command can't blow
5667 /// up the context window. Cuts on a char boundary and appends a notice.
5668 fn cap_tool_output(&self, output: String) -> String {
5669 let Some(max) = self.config.max_tool_output_bytes else {
5670 return output;
5671 };
5672 if max == 0 || output.len() <= max {
5673 return output;
5674 }
5675 // Find the largest char boundary <= max.
5676 let mut end = max;
5677 while end > 0 && !output.is_char_boundary(end) {
5678 end -= 1;
5679 }
5680 let total = output.len();
5681 let mut s = output[..end].to_string();
5682 // BP-2 (catalog:58, `core.tool_output_spill`): write the full bytes
5683 // to a per-session file the model can read back. Off (the default)
5684 // leaves the notice byte-identical to before.
5685 let spill = if self.config.tool_output_spill {
5686 self.spill_tool_output(&output)
5687 } else {
5688 None
5689 };
5690 // Honest retention labeling (D6, B10-AC4): only claim the sidecar has
5691 // the full output when a recorder is actually installed — or, BP-2,
5692 // that the spill file has it when one was actually written.
5693 let retention = if self.recorder.is_some() {
5694 "full output in session sidecar"
5695 } else if spill.is_some() {
5696 "full output on disk"
5697 } else {
5698 "full output not retained"
5699 };
5700 let recovery = match &spill {
5701 // The door is named in the notice, so it works WITHOUT
5702 // `capabilities.reduction`: under a preset with a read tool
5703 // that is `read_file`; under a shell-only preset (cx-parity,
5704 // whose whole read pathway is the shell) it is `cat`.
5705 Some(path) => {
5706 let door = if self.registry.get("read_file").is_some() {
5707 "read it with `read_file`"
5708 } else {
5709 "read it with `cat`"
5710 };
5711 format!("; full output spilled to {} — {door}", path.display())
5712 }
5713 None => String::new(),
5714 };
5715 s.push_str(&format!(
5716 "{CAP_NOTICE_MARKER}{total} bytes total, showing first {end}; {retention}{recovery}]"
5717 ));
5718 s
5719 }
5720
5721 /// BP-2 (catalog:58 "Oversized output truncated; full content kept
5722 /// reachable"): write `full` to this session's spill directory and
5723 /// return the path, or `None` if it could not be written (a spill is a
5724 /// recovery convenience — it must never fail the tool call).
5725 ///
5726 /// The file is named by content hash, so the same output spilled twice
5727 /// costs one file and a re-run of an identical command reuses it.
5728 fn spill_tool_output(&self, full: &str) -> Option<std::path::PathBuf> {
5729 let dir = self.spill_dir();
5730 std::fs::create_dir_all(&dir).ok()?;
5731 let digest = blake3::hash(full.as_bytes()).to_hex();
5732 let path = dir.join(format!("tool-output-{}.txt", &digest[..16]));
5733 if !path.exists() {
5734 std::fs::write(&path, full).ok()?;
5735 }
5736 Some(path)
5737 }
5738
5739 /// BP-2: where this agent's spilled outputs live — beside the session
5740 /// sidecar when one is recording (per-SESSION, the same identity the
5741 /// sidecar has), else a per-PROCESS temp directory, which is as
5742 /// specific as an agent with no sidecar can honestly be.
5743 fn spill_dir(&self) -> std::path::PathBuf {
5744 if let Some(recorder) = &self.recorder {
5745 let path = recorder.path();
5746 if let (Some(parent), Some(stem)) = (path.parent(), path.file_stem()) {
5747 return parent.join(format!("{}.spill", stem.to_string_lossy()));
5748 }
5749 }
5750 std::env::temp_dir().join(format!("supercode-spill-{}", std::process::id()))
5751 }
5752
5753 /// P5-3 note on the signature: written as a plain fn returning an
5754 /// explicitly boxed future (`Pin<Box<dyn Future + Send>>`) rather than
5755 /// as `async fn`. `spawn_subagent` makes this function genuinely
5756 /// recursive at the TYPE level: `run_tool` -> `run_spawn_subagent` ->
5757 /// (a child) `Agent::send` -> `run_loop` -> `run_tool` again — an
5758 /// `async fn`'s return type is an anonymous, compiler-inferred
5759 /// self-referential state machine, and inferring one that embeds
5760 /// itself (even indirectly, through several other functions) is a
5761 /// compile error (an infinitely-sized/cyclic opaque type). Declaring
5762 /// `run_tool`'s return type EXPLICITLY as a boxed trait object breaks
5763 /// the cycle: every other function on the call graph now embeds a
5764 /// concrete, already-known type here instead of one the compiler would
5765 /// otherwise need to (cyclically) infer. Callers are unaffected —
5766 /// `self.run_tool(call).await` reads identically either way.
5767 fn run_tool<'a>(
5768 &'a mut self,
5769 call: &'a supercode_interchange::ToolCall,
5770 ) -> std::pin::Pin<Box<dyn std::future::Future<Output = (String, bool)> + Send + 'a>> {
5771 Box::pin(async move {
5772 let translated_builtin = if self.config.claude_runtime_tools_enabled {
5773 match self.translate_claude_builtin_call(call) {
5774 Ok(translated) => translated,
5775 Err(error) => return (format!("Error: {error}"), true),
5776 }
5777 } else {
5778 None
5779 };
5780 let call = translated_builtin.as_ref().unwrap_or(call);
5781 if self.config.claude_runtime_tools_enabled
5782 && matches!(
5783 call.function.name.as_str(),
5784 CLAUDE_CRON_CREATE
5785 | CLAUDE_CRON_DELETE
5786 | CLAUDE_CRON_LIST
5787 | CLAUDE_SCHEDULE_WAKEUP
5788 )
5789 {
5790 return self.run_claude_runtime_tool(call);
5791 }
5792 // P5-3: `spawn_subagent`/`subagent_status` need full async
5793 // `&mut self` access (running a child agent's loop, or
5794 // awaiting an already-finished background `JoinHandle`) —
5795 // `prepare_tool_call` is purely synchronous, so these are
5796 // intercepted HERE, one level above it, rather than inside it
5797 // like `TOOL_SEARCH`/`EXPAND_REDUCTION`/`SIDECAR_SEARCH`.
5798 if call.function.name == CLAUDE_AGENT && self.config.subagents_claude_agent_alias {
5799 return match self.translate_claude_agent_call(call) {
5800 Ok(translated) => self.run_spawn_subagent(&translated).await,
5801 Err(error) => (format!("Error: {error}"), true),
5802 };
5803 }
5804 if call.function.name == SPAWN_SUBAGENT {
5805 return self.run_spawn_subagent(call).await;
5806 }
5807 // P5-3 safety-hardening fix (Fable-5 review, LOW "wrong error
5808 // when disabled"): gated on `subagents_enabled`, matching
5809 // `run_spawn_subagent`'s own already-correct disabled behavior
5810 // (that one gates INTERNALLY, at its own top; this one gates
5811 // HERE, at the interception point, because unlike
5812 // `spawn_subagent` it has no other reason to run any logic at
5813 // all when subagents are off). When disabled, a hallucinated
5814 // `subagent_status` call must NOT be intercepted — it falls
5815 // through to `prepare_tool_call`'s normal unknown-tool path
5816 // below, which returns `Error::UnknownTool("subagent_status")`,
5817 // byte-identical to the pre-P5-3 (and disabled-spawn_subagent)
5818 // error text — never `Error::SubagentNotFound`'s "unknown
5819 // subagent id" text, which would wrongly imply subagents are on
5820 // but this particular id is bogus.
5821 if call.function.name == SUBAGENT_STATUS && self.config.subagents_enabled {
5822 return self.run_subagent_status(call).await;
5823 }
5824 // BP-7: same interception shape and same `subagents_enabled`
5825 // gate as `SUBAGENT_STATUS` above — when the module is off a
5826 // hallucinated call falls through to the ordinary unknown-tool
5827 // error rather than a misleading "unknown subagent id".
5828 if call.function.name == SEND_MESSAGE && self.config.subagents_enabled {
5829 return self.run_send_message(call).await;
5830 }
5831 if call.function.name == SUBAGENT_RESUME && self.config.subagents_enabled {
5832 return self.run_subagent_resume(call).await;
5833 }
5834 match self.prepare_tool_call(call) {
5835 PreparedCall::Done(result) => result,
5836 PreparedCall::Ready { name, args } => {
5837 // `prepare_tool_call` already confirmed the registry has
5838 // this tool.
5839 let tool = self.registry.get(&name).expect("prepared as Ready");
5840 let (output, is_error) = match tool.execute(args, &self.ctx).await {
5841 Ok(out) => (out, false),
5842 Err(e) => (format!("Error: {e}"), true),
5843 };
5844 if let Some(hook) = &self.config.post_tool_hook {
5845 hook(&name, &output, is_error);
5846 }
5847 (output, is_error)
5848 }
5849 }
5850 })
5851 }
5852
5853 /// P4e (§3.1 `core.parallel_tool_calls`, catalog:59): the SYNCHRONOUS
5854 /// half of dispatching one tool call — everything `Self::run_tool` did
5855 /// BEFORE its single `tool.execute(...).await`, factored out so
5856 /// [`Self::run_tools_concurrently`] can run these cheap, stateful,
5857 /// `&mut self` checks (agent intrinsics, unknown-tool, approval,
5858 /// doom-loop, pre-tool-hook) SEQUENTIALLY and in ORIGINAL call order —
5859 /// exactly as `run_tool` always has — before handing the remaining
5860 /// calls' `execute()` futures to `join_all`. `Self::run_tool` itself is
5861 /// now a thin wrapper over this (a pure refactor: byte-identical
5862 /// observable behavior, verified by the existing test suite).
5863 fn prepare_tool_call(&mut self, call: &supercode_interchange::ToolCall) -> PreparedCall {
5864 let name = &call.function.name;
5865 // BP-3 (catalog row "Context-budget tools"): hand the model's
5866 // `get_context_remaining` the agent's OWN accounting — BP-4's
5867 // [`Self::context_usage`], the same struct `/context` prints and
5868 // the same estimates the context guard enforces, so what the model
5869 // reads and what refuses an oversized turn can never disagree.
5870 // Computed at the moment the question is asked (the freshest
5871 // possible view) and ONLY then: `context_usage` projects the whole
5872 // request view, which is not a cost to pay on unrelated calls.
5873 if name == crate::tools::GET_CONTEXT_REMAINING {
5874 if let Ok(usage) = serde_json::to_value(self.context_usage()) {
5875 self.ctx.context_budget.publish(usage);
5876 }
5877 }
5878 if name == TOOL_SEARCH {
5879 // Agent intrinsic (B6): intercepted before registry lookup, since
5880 // `Tool::execute` has no access to the registry or `activated_tools`.
5881 return PreparedCall::Done(self.run_tool_search(call));
5882 }
5883 if name == EXPAND_REDUCTION {
5884 // Agent intrinsic (T12/TR-1): intercepted before registry lookup,
5885 // same reason — resolves against `self.reduction_log`/`self.history`,
5886 // which `Tool::execute` has no access to.
5887 return PreparedCall::Done(self.run_expand_reduction(call));
5888 }
5889 if name == SIDECAR_SEARCH {
5890 return PreparedCall::Done(self.run_sidecar_search(call));
5891 }
5892 // P5-6 (§2 module 4 `tools.background`): gated at the interception
5893 // point itself (not internally, at each method's own top) —
5894 // mirroring `SUBAGENT_STATUS`'s own fix (Fable-5 review, LOW "wrong
5895 // error when disabled"): a hallucinated call when the module is off
5896 // must fall through to the plain `Error::UnknownTool` path below,
5897 // never a background-specific error that would wrongly imply the
5898 // module is on. Unlike `SPAWN_SUBAGENT`/`SUBAGENT_STATUS`, none of
5899 // these four need async `&mut self` access (spawning a process,
5900 // `Child::try_wait`, and `Child::start_kill` are all synchronous),
5901 // so they're intercepted here in `prepare_tool_call` rather than in
5902 // `Self::run_tool`.
5903 if self.config.tools_background_enabled {
5904 if name == BACKGROUND_EXEC {
5905 return PreparedCall::Done(self.run_background_exec(call));
5906 }
5907 if name == BACKGROUND_STATUS {
5908 return PreparedCall::Done(self.run_background_status(call));
5909 }
5910 if name == BACKGROUND_LIST {
5911 return PreparedCall::Done(self.run_background_list(call));
5912 }
5913 if name == BACKGROUND_KILL {
5914 return PreparedCall::Done(self.run_background_kill(call));
5915 }
5916 }
5917 if self.registry.get(name).is_none() {
5918 let err = Error::UnknownTool(name.clone());
5919 return PreparedCall::Done((format!("Error: {err}"), true));
5920 }
5921
5922 // P5-1 (§2 modules 10-11, integration point named in
5923 // COMPOSABLE-HARNESS-DESIGN.md's activation set): the permissions
5924 // ENGINE governs the gate when `capabilities.permissions.enabled`
5925 // is on; every other config resolves this to `false`
5926 // (`Config::default`), which takes the `else` branch below —
5927 // the EXACT pre-P5-1 code, untouched, so the default posture
5928 // (approval=never/sandbox=none) and every existing test's observed
5929 // behavior is byte-for-byte unchanged.
5930 if self.config.permissions_enabled {
5931 // The engine needs the command/path TEXT the legacy tool-name-
5932 // only gate below never looked at, so args must be parsed
5933 // BEFORE the gate here (not after, like the legacy branch).
5934 let args = match call.function.parsed_arguments() {
5935 Ok(v) => v,
5936 Err(e) => {
5937 let err = Error::InvalidArguments {
5938 tool: name.clone(),
5939 message: e.to_string(),
5940 };
5941 return PreparedCall::Done((format!("Error: {err}"), true));
5942 }
5943 };
5944 // P4c doom-loop breaker, unchanged, still before any gate.
5945 if let Some(reason) = self.check_doom_loop(name, &args) {
5946 return PreparedCall::Done((format!("Error: {reason}"), true));
5947 }
5948 // BP-10: the hook runs BEFORE the engine on this path, so its
5949 // rewrite is what the rules see and its allow/ask/deny is a
5950 // tier inside them — see `run_pre_tool_hook`.
5951 let (args, hook_decision) = match self.run_pre_tool_hook(name, args) {
5952 Ok(pair) => pair,
5953 Err(done) => return done,
5954 };
5955 if let Some(reason) = self.permissions_gate_denial(name, &args, hook_decision) {
5956 return PreparedCall::Done((format!("Error: {reason}"), true));
5957 }
5958 PreparedCall::Ready {
5959 name: name.clone(),
5960 args,
5961 }
5962 } else {
5963 // ---- pre-P5-1 gate, byte-for-byte unchanged ----
5964 // Approval gate: if the policy requires it, consult the handler
5965 // (absent handler denies, so an OnRequest/Untrusted policy is
5966 // fail-closed).
5967 if self.config.needs_approval(name) {
5968 let approved = self
5969 .config
5970 .approval_handler
5971 .as_ref()
5972 .map(|h| h(call))
5973 .unwrap_or(false);
5974 if !approved {
5975 return PreparedCall::Done((
5976 format!("Error: tool `{name}` was not approved for execution"),
5977 true,
5978 ));
5979 }
5980 }
5981 let args = match call.function.parsed_arguments() {
5982 Ok(v) => v,
5983 Err(e) => {
5984 let err = Error::InvalidArguments {
5985 tool: name.clone(),
5986 message: e.to_string(),
5987 };
5988 return PreparedCall::Done((format!("Error: {err}"), true));
5989 }
5990 };
5991 self.finish_prepare(name.clone(), args)
5992 }
5993 }
5994
5995 /// P5-1: the shared tail of [`Self::prepare_tool_call`] — doom-loop
5996 /// check, pre-tool hook, `Ready` construction — factored out so both
5997 /// the legacy gate and the new permissions-engine gate run the exact
5998 /// same downstream checks in the exact same order (§5.3 risk 1: the
5999 /// permissions engine changes WHO gets to run, never what happens once
6000 /// they're approved).
6001 fn finish_prepare(&mut self, name: String, args: serde_json::Value) -> PreparedCall {
6002 // P4c (§5.2 P4 "doom-loop breaker", oc UNIQUE `doom_loop` row,
6003 // catalog D3): a default, always-available veto point distinct from
6004 // `Config.pre_tool_hook` (a single user-installable slot — the
6005 // breaker must coexist with a caller's own hook, not compete for the
6006 // one slot). `None`/`Some(0|1)` is a no-op — byte-identical to
6007 // today (no repetition tracking, no call is ever refused on this
6008 // basis).
6009 if let Some(reason) = self.check_doom_loop(&name, &args) {
6010 return PreparedCall::Done((format!("Error: {reason}"), true));
6011 }
6012 // Pre-tool hook may block the call. BP-10: the pre-P5-1 gate has
6013 // no permissions engine for a hook's `Allow`/`Ask` to be a tier
6014 // OF, so only the deny half can mean anything here — an
6015 // `updated_args` rewrite still applies (it is a property of the
6016 // call, not of any gate), and `Allow`/`Ask` are no-ops, exactly
6017 // as `None` was before BP-10.
6018 let mut args = args;
6019 if let Some(hook) = &self.config.pre_tool_hook {
6020 let outcome = hook(&name, &args);
6021 if let Some(rewritten) = outcome.updated_args {
6022 args = rewritten;
6023 }
6024 if outcome.decision == crate::config::HookDecision::Deny {
6025 let reason = outcome.reason.unwrap_or_else(|| "denied".to_string());
6026 return PreparedCall::Done((
6027 format!("Error: blocked by pre-tool hook: {reason}"),
6028 true,
6029 ));
6030 }
6031 }
6032 PreparedCall::Ready { name, args }
6033 }
6034
6035 /// BP-10 (catalog row "Hook/plugin permission veto"): fire the pre-tool
6036 /// hook for the permissions-engine path, where it runs BEFORE the gate
6037 /// (CC's own order: a `PreToolUse` hook answers the permission question
6038 /// rather than being asked after it). Returns the possibly-REWRITTEN
6039 /// arguments plus the [`crate::config::HookDecision`] the engine folds
6040 /// in, or the finished denial when the hook refused outright.
6041 ///
6042 /// The rewrite lands BEFORE the gate deliberately: the engine must
6043 /// evaluate what will actually run, so a hook cannot launder a denied
6044 /// command by rewriting it past the rules.
6045 #[allow(clippy::type_complexity)]
6046 fn run_pre_tool_hook(
6047 &self,
6048 name: &str,
6049 args: serde_json::Value,
6050 ) -> std::result::Result<(serde_json::Value, crate::config::HookDecision), PreparedCall> {
6051 let Some(hook) = &self.config.pre_tool_hook else {
6052 return Ok((args, crate::config::HookDecision::Pass));
6053 };
6054 let outcome = hook(name, &args);
6055 let args = outcome.updated_args.unwrap_or(args);
6056 if outcome.decision == crate::config::HookDecision::Deny {
6057 let reason = outcome.reason.unwrap_or_else(|| "denied".to_string());
6058 return Err(PreparedCall::Done((
6059 format!("Error: blocked by pre-tool hook: {reason}"),
6060 true,
6061 )));
6062 }
6063 Ok((args, outcome.decision))
6064 }
6065
6066 /// D-2 (Fable-5 delta review — LOW-MEDIUM, "over-grant residual"): the
6067 /// built-in tools whose `command`/`path`/`patch` arg IS semantically
6068 /// the whole call — the ONLY tools [`Self::permissions_gate_denial`]
6069 /// is allowed to turn into an [`permissions::ApprovalRequest::subject`]
6070 /// (see that method's own doc comment on the `subject` line for the
6071 /// full story). Bash-family (`bash`, and `shell` —
6072 /// [`crate::tools::builtins::PersistentShellTool`]'s registered name,
6073 /// what the review's "persistent-shell" refers to), the file tools
6074 /// (`read_file`/`write_file`/`edit_file`/`view_image`, whose `path` IS
6075 /// the subject), and `apply_patch` (whose `patch` envelope is handled
6076 /// separately but is unconditionally this tool only, see the `patch`
6077 /// local a few lines below). Deliberately NOT `list_dir`/`glob`/
6078 /// `search` — this crate's F2 fix (`ApprovalCache::key_for_request`)
6079 /// already falls back to a full-args digest for anything not on this
6080 /// list, which is a strictly SAFER (if slightly less cache-granular)
6081 /// default than guessing at more built-ins that weren't part of this
6082 /// finding.
6083 const SUBJECT_BEARING_BUILTIN_TOOLS: &'static [&'static str] = &[
6084 "bash",
6085 "shell",
6086 "read_file",
6087 "write_file",
6088 "edit_file",
6089 "view_image",
6090 "apply_patch",
6091 ];
6092
6093 /// P5-1: evaluate `name`'s call (with parsed `args`) against the
6094 /// permissions engine (`crate::permissions`) — builds the
6095 /// [`crate::permissions::RuleSet`] from `Config`'s deny/ask/allow
6096 /// pattern lists (folding [`Config::permissions_protected_paths`] into
6097 /// the `deny` tier, module 13), picks a command-, path-, or name-only
6098 /// evaluation depending on what `args` carries, resolves an `Ask`
6099 /// decision via the session cache + THIS agent's own installed
6100 /// [`crate::permissions::PermissionsApprovalHandler`], and returns
6101 /// `Some(reason)` when the call is refused (`None` = proceed). Thin
6102 /// wrapper over [`Self::permissions_gate_denial_impl`] — see that
6103 /// method's doc comment for why the handler is a parameter there.
6104 fn permissions_gate_denial(
6105 &self,
6106 name: &str,
6107 args: &serde_json::Value,
6108 hook: crate::config::HookDecision,
6109 ) -> Option<String> {
6110 self.permissions_gate_denial_impl(
6111 name,
6112 args,
6113 self.permissions_approval_handler.as_deref(),
6114 hook,
6115 )
6116 }
6117
6118 /// P5-6 (§2.2 C6, build brief "wire to the P5-1 engine's non-
6119 /// interactive fail-closed path"): the SAME rule-evaluation body as
6120 /// [`Self::permissions_gate_denial`], but the approval `handler` is a
6121 /// PARAMETER instead of always reading `self.permissions_approval_handler`
6122 /// — `Agent::background_permission_denial` calls this with `handler:
6123 /// None` (or a [`crate::subagents::ParentQueueApprovalHandler`]) so a
6124 /// `background_exec` call's `Ask`-tier decisions resolve exactly like a
6125 /// P5-3 background child's do (`crate::permissions::resolve_ask`'s
6126 /// pre-existing "no handler ⇒ deny" contract), REGARDLESS of whether
6127 /// this agent itself has an interactive handler installed for its own
6128 /// foreground calls — a background job must never block on a prompt it
6129 /// has no way to answer, even if the agent hosting it could otherwise
6130 /// answer one. The rule SET and default-policy baseline are otherwise
6131 /// identical to a foreground call's — only how an `Ask` decision
6132 /// resolves ever differs, and only in the strictly-narrower direction
6133 /// (never escalates past what a foreground call of the same command is
6134 /// allowed).
6135 fn permissions_gate_denial_impl(
6136 &self,
6137 name: &str,
6138 args: &serde_json::Value,
6139 handler: Option<&dyn crate::permissions::PermissionsApprovalHandler>,
6140 hook: crate::config::HookDecision,
6141 ) -> Option<String> {
6142 use crate::permissions::{self, Decision, PathKind};
6143
6144 // `Config.tool_deny_patterns`/`tool_allow_patterns` (P4a) ARE the
6145 // engine's deny/allow tiers — the same `capabilities.permissions.
6146 // rules.deny`/`.allow` keys, one source of truth, no duplication.
6147 // Protected paths (module 13) are an unconditional deny floor,
6148 // folded in here rather than checked separately, so they benefit
6149 // from the SAME first-match deny-wins priority every other deny
6150 // rule gets.
6151 // BP-5: the rule set itself is `permissions::rules_for_config` —
6152 // ONE construction shared with every other surface that has to ask
6153 // this engine a question (see that function's doc comment). Plan
6154 // mode's narrowing is layered on top of it here, because it is
6155 // this agent's live state, not the config's.
6156 let mut rules = permissions::rules_for_config(&self.config);
6157 // BP-3 (§2 module 8 `plan_mode`, dependency edge `plan_mode →
6158 // permissions.rules|sandbox`): the read-only research phase IS a
6159 // narrowing of this rule set — while the mode is active the write
6160 // and execution tools join the deny tier, and get the same
6161 // first-match, never-overridable treatment every other deny rule
6162 // gets. Inactive (the default, and the only state a config without
6163 // the module can reach) contributes NOTHING, so the rule set is
6164 // byte-identical to before.
6165 rules
6166 .deny
6167 .extend(crate::tools::plan_mode::deny_rules(&self.ctx.plan_mode));
6168
6169 // The baseline decision when NO rule matches at all — derived from
6170 // `ApprovalPolicy`, the same per-policy shape
6171 // `Config::needs_approval` uses for the legacy gate (see
6172 // `ApprovalPolicy::ModelRequested`'s doc comment for why this
6173 // richer gate approximates Codex's real "mostly silent" posture
6174 // instead of that method's conservative OnRequest-alike treatment
6175 // — explicit deny/ask rules still apply on top regardless).
6176 let default = permissions::default_decision(&self.config, name);
6177
6178 // BP-10: path rules are evaluated relative to EVERY granted root
6179 // (cwd + `core.additional_dirs`/`--add-dir`), folded to the
6180 // strictest — see `permissions::evaluate_path_safe_roots`'s doc
6181 // comment for why a grant must not also remove the root-relative
6182 // protected-path floor inside the granted directory. With no extra
6183 // dirs (the default) this is the single `cwd` list, byte-identical
6184 // to before.
6185 let mut roots = vec![self.config.cwd.clone()];
6186 roots.extend(self.config.additional_dirs.iter().cloned());
6187
6188 let command = args.get("command").and_then(|v| v.as_str());
6189 let path = args.get("path").and_then(|v| v.as_str());
6190 // F4 (Fable-5 adversarial review): `apply_patch`'s args carry a
6191 // patch ENVELOPE body (`args["patch"]`), not a `command` or a
6192 // `path` — the two branches above never fire for it, which is
6193 // exactly how a patch touching a protected path bypassed
6194 // `protected_paths` entirely. Only consulted for the `apply_patch`
6195 // tool specifically (a `patch`-shaped arg on some other tool is not
6196 // this envelope format and isn't given this treatment).
6197 let patch = (name == "apply_patch")
6198 .then(|| args.get("patch").and_then(|v| v.as_str()))
6199 .flatten();
6200 let decision = if let Some(command) = command {
6201 permissions::evaluate_command(&rules, name, command, default)
6202 } else if let Some(patch) = patch {
6203 // Same dual-check shape as the path branch below (pseudo-tool
6204 // `write(...)` rules from `protected_paths`, AND a rule
6205 // authored against the real `apply_patch` tool name), applied
6206 // to EVERY path the envelope's ops touch (`Add`/`Delete`/
6207 // `Update`'s `path`, plus `*** Move to:`). A patch that fails
6208 // to parse can't be proven to avoid a protected path — fail
6209 // closed to at least `Ask`, the same floor an unparseable bash
6210 // command gets in `permissions::evaluate_command`, rather than
6211 // silently let it through on `default`.
6212 let mut d = rules.evaluate(name, None).unwrap_or(default);
6213 match crate::tools::patch_target_paths(patch) {
6214 Ok(paths) => {
6215 for p in &paths {
6216 // SECURITY (CRITICAL fix): route both checks through
6217 // the safe-path-resolving variants — a patch target
6218 // like `x/../.git/config` must be caught exactly
6219 // like a `write_file`/`edit_file` `path` argument
6220 // would be (see `evaluate_path_safe`'s doc comment).
6221 let pseudo = permissions::evaluate_path_safe_roots(
6222 &rules,
6223 PathKind::Write,
6224 &roots,
6225 p,
6226 Decision::Allow,
6227 );
6228 let real_tool = permissions::evaluate_path_subject_safe_roots(
6229 &rules,
6230 name,
6231 &roots,
6232 p,
6233 Decision::Allow,
6234 );
6235 d = d.stricter(pseudo).stricter(real_tool);
6236 }
6237 }
6238 Err(_) => {
6239 d = d.stricter(Decision::Ask);
6240 }
6241 }
6242 d
6243 } else if let Some(path) = path {
6244 let kind = if matches!(name, "write_file" | "edit_file") {
6245 PathKind::Write
6246 } else {
6247 PathKind::Read
6248 };
6249 // TWO independent sources of path-shaped rules can apply to the
6250 // same call, and BOTH must be checked:
6251 // (a) the `read(...)`/`write(...)` pseudo-tool (tool-agnostic —
6252 // applies no matter WHICH tool touches the path; this is
6253 // what `Config::permissions_protected_paths`/module 13
6254 // expands into, via `protected_path_deny_rules`);
6255 // (b) a rule authored against the REAL tool name with the path
6256 // as its subject — design §4.4's own oc-parity worked
6257 // example writes exactly this shape (`"read_file(*.env)"`,
6258 // not a pseudo-tool), matching how `bash(cmdglob)` rules
6259 // are authored. `RuleSet::evaluate`'s bare-tool-name-glob
6260 // branch (no parens) ALSO fires here regardless of
6261 // `subject`, so this one call additionally covers a
6262 // blanket "deny this tool entirely" rule — no separate
6263 // `rules.evaluate(name, None)` call is needed.
6264 // SECURITY (CRITICAL fix, guarantor audit): both checks now
6265 // route through the safe-path-resolving variants (see
6266 // `evaluate_path_safe`'s doc comment) instead of glob-matching
6267 // the raw model-supplied `path` string directly — this is what
6268 // closes the traversal bypass (`write_file
6269 // path="x/../.git/config"`) and the analogous symlink escape.
6270 let pseudo_decision =
6271 permissions::evaluate_path_safe_roots(&rules, kind, &roots, path, default);
6272 let real_tool_decision =
6273 permissions::evaluate_path_subject_safe_roots(&rules, name, &roots, path, default);
6274 pseudo_decision.stricter(real_tool_decision)
6275 } else {
6276 rules.evaluate(name, None).unwrap_or(default)
6277 };
6278
6279 // BP-10 (catalog row "Hook/plugin permission veto"): the hook's
6280 // verdict is a TIER of this engine, folded in the one direction
6281 // that is always safe — `Ask` tightens (`stricter` never loosens),
6282 // `Deny` is a floor. `Allow` deliberately does NOT change the
6283 // decision here: it answers the `Ask` tier below (the
6284 // `PermissionRequest`-class reply CC's hooks give on the user's
6285 // behalf), so a `deny` rule still refuses the call outright — a
6286 // hook may skip a prompt, never a floor.
6287 let decision = match hook {
6288 crate::config::HookDecision::Deny => Decision::Deny,
6289 crate::config::HookDecision::Ask => decision.stricter(Decision::Ask),
6290 crate::config::HookDecision::Allow | crate::config::HookDecision::Pass => decision,
6291 };
6292
6293 // BP-10 (catalog row "Sandbox-escalation path", cx§4
6294 // `sandbox_permissions: "require_escalated"` + justification): the
6295 // model's channel to ASK for an unsandboxed run is a rule inside
6296 // this one engine, not a switch beside it. A call carrying
6297 // `with_escalated_permissions: true` is forced to at least the
6298 // `Ask` tier — never below whatever the rules already decided, so
6299 // a denied command cannot escalate its way out (`stricter` only
6300 // tightens), and never silently allowed under
6301 // `ApprovalPolicy::ModelRequested`/`Never`, whose `Allow` baseline
6302 // is exactly what made "the model requests escalation" a no-op
6303 // before. The `justification` rides in `raw_args` below, so the
6304 // approval door shows the user the model's own reason.
6305 let decision = if args
6306 .get("with_escalated_permissions")
6307 .and_then(|v| v.as_bool())
6308 .unwrap_or(false)
6309 {
6310 decision.stricter(Decision::Ask)
6311 } else {
6312 decision
6313 };
6314
6315 // D-2 (Fable-5 delta review — LOW-MEDIUM): `command`/`path`/`patch`
6316 // above are extracted (and used to DRIVE the decision above) for
6317 // ANY tool that happens to carry one of those arg names — that
6318 // part is unchanged and correct (a rule authored against, say, an
6319 // MCP tool's own name legitimately wants to glob-match its
6320 // `command`-shaped arg too). But the narrower single-field
6321 // `subject` handed to the cache/handler below must NOT do the
6322 // same for a non-built-in tool: an MCP (or other) tool's
6323 // `command`/`path` is just one field among potentially several
6324 // that together define what the call actually does — collapsing
6325 // an `AllowForSession` grant down to that one field would silently
6326 // auto-allow a later call with the SAME `command` but different
6327 // OTHER args (e.g. `{"command":"sync","target":"staging"}`
6328 // auto-allowing `{"command":"sync","target":"production"}`).
6329 // Restricting this to the known built-ins whose `subject` really
6330 // IS the whole call leaves every other tool with `subject: None`,
6331 // which routes it through `ApprovalCache::key_for_request`'s
6332 // full-args-digest fallback (F2) instead.
6333 let subject = Self::SUBJECT_BEARING_BUILTIN_TOOLS
6334 .contains(&name)
6335 .then(|| command.or(path).or(patch))
6336 .flatten();
6337 let req = permissions::ApprovalRequest {
6338 tool: name,
6339 subject,
6340 raw_args: args,
6341 };
6342 let approved = permissions::decision_to_approved(decision, || {
6343 // BP-10: a hook `Allow` answers this ask without a prompt (and
6344 // without a cache entry — the hook is consulted on every call,
6345 // so caching its answer would be a second, staler copy of the
6346 // same decision).
6347 if hook == crate::config::HookDecision::Allow {
6348 return true;
6349 }
6350 permissions::resolve_ask(&self.permissions_approval_cache, handler, &req)
6351 });
6352 if approved {
6353 None
6354 } else {
6355 Some(format!(
6356 "tool `{name}` was not approved for execution (permissions engine: {decision:?})"
6357 ))
6358 }
6359 }
6360
6361 /// P5-6 (§2.2 C6, build brief "a bg `rm -rf` subject to the same deny
6362 /// rules... must never escalate past what a foreground exec of the
6363 /// same command is allowed"): the permission gate `background_exec`
6364 /// runs BEFORE spawning anything. Evaluated against the tool name
6365 /// `"bash"` (not `"background_exec"`) deliberately — so any
6366 /// `bash(...)`-authored deny/ask/allow rule (or protected-path floor)
6367 /// applies to a background command byte-for-byte identically to a
6368 /// foreground `bash` call, the SAME rule set + default baseline
6369 /// [`Self::permissions_gate_denial`] would use for one.
6370 ///
6371 /// The one deliberate difference (C6 itself): an `Ask`-tier decision
6372 /// NEVER reaches an interactive handler here — a background job has no
6373 /// way to block on a prompt it can't answer. When
6374 /// [`Config::subagents_background_prompts`] is
6375 /// [`crate::subagents::BackgroundPromptsPolicy::Parent`], the denied
6376 /// request is additionally queued onto [`Self::pending_child_approvals`]
6377 /// (via [`crate::subagents::ParentQueueApprovalHandler`], reused
6378 /// verbatim — the SAME "parent-surfaced queue" §2.2 C6 names for
6379 /// `subagents.background`, with the job id standing in for a child
6380 /// agent id) for later inspection; any other configuration (including
6381 /// no `background_prompts` set at all) resolves via `handler: None` —
6382 /// [`crate::permissions::resolve_ask`]'s pre-existing "no handler ⇒
6383 /// deny" fail-closed default, identical to `subagents`'s own
6384 /// `AutoPolicy` reading. Either way, `Ask` always denies; only `Allow`
6385 /// (from the rule engine itself, or a PRIOR interactively-granted
6386 /// `AllowForSession` cache entry) ever lets a background command run —
6387 /// so this can only ever be as-or-more restrictive than a foreground
6388 /// call, never looser, regardless of configuration.
6389 ///
6390 /// Covers BOTH gate generations: when [`Config::permissions_enabled`]
6391 /// is on, the P5-1 engine (above) is used; otherwise the legacy
6392 /// [`Config::needs_approval`] gate is consulted but its
6393 /// `approval_handler` closure is NEVER invoked (that closure could
6394 /// itself block, e.g. a real interactive prompt) — an approval-required
6395 /// legacy policy simply denies a background command outright, the same
6396 /// never-hang guarantee under the older gate.
6397 fn background_permission_denial(
6398 &self,
6399 command: &str,
6400 job_id: &str,
6401 hook: crate::config::HookDecision,
6402 ) -> Option<String> {
6403 let args = serde_json::json!({ "command": command });
6404 if self.config.permissions_enabled {
6405 if let Some(crate::subagents::BackgroundPromptsPolicy::Parent) =
6406 self.config.subagents_background_prompts
6407 {
6408 let handler = crate::subagents::ParentQueueApprovalHandler {
6409 child_agent_id: format!("bg:{job_id}"),
6410 queue: self.pending_child_approvals.clone(),
6411 };
6412 self.permissions_gate_denial_impl("bash", &args, Some(&handler), hook)
6413 } else {
6414 self.permissions_gate_denial_impl("bash", &args, None, hook)
6415 }
6416 } else if self.config.needs_approval("bash") {
6417 Some(
6418 "tool `bash` requires approval, which a background job cannot request \
6419 interactively (§2.2 C6: auto-policy denies)"
6420 .to_string(),
6421 )
6422 } else {
6423 None
6424 }
6425 }
6426
6427 /// P4e (§3.1 `core.parallel_tool_calls`, catalog:59 "Independent
6428 /// sibling calls run concurrently"): runs `calls`' `Tool::execute()`
6429 /// futures CONCURRENTLY via `futures::future::join_all`, for whichever
6430 /// calls [`Self::prepare_tool_call`] resolves to [`PreparedCall::Ready`]
6431 /// — i.e. every plain (non-intrinsic) registry-tool call that passes
6432 /// its synchronous approval/doom-loop/pre-tool-hook checks. A call that
6433 /// resolves to [`PreparedCall::Done`] (an intrinsic, an unknown tool, a
6434 /// denied/blocked call) is NOT parallelized — its result is already in
6435 /// hand from the synchronous prepare pass. Every prepare check still
6436 /// runs sequentially, in original call order, before ANY `execute()`
6437 /// future starts (only the actual tool I/O overlaps) — so doom-loop
6438 /// bookkeeping and pre-tool-hook vetoes see the exact same call order
6439 /// they would under the sequential path. Returns results in the SAME
6440 /// order as `calls`, so callers can always `zip` the two. Post-tool
6441 /// hooks fire per call, in original order, once every result is in
6442 /// hand — a caller-visible timing difference from the sequential path
6443 /// ONLY when this method runs at all (i.e. only when
6444 /// `Config::parallel_tool_calls` is on): hooks see "this batch
6445 /// finished" ordering rather than "this one call finished" ordering.
6446 /// Documented, not a bug.
6447 async fn run_tools_concurrently(
6448 &mut self,
6449 calls: &[supercode_interchange::ToolCall],
6450 ) -> Vec<(String, bool)> {
6451 // P5-3: `spawn_subagent`/`subagent_status` need sequential `&mut
6452 // self` access `prepare_tool_call`'s synchronous-only signature
6453 // can't give them (see `Self::run_tool`'s identical interception).
6454 // A batch that includes one falls back to dispatching the WHOLE
6455 // batch sequentially via `Self::run_tool` — a documented, narrow
6456 // simplification (not a partial-parallelization attempt) rather
6457 // than restructuring `PreparedCall` to carry a future; a batch with
6458 // no subagent intrinsic is completely unaffected and still
6459 // parallelizes exactly as before.
6460 if calls.iter().any(|c| {
6461 c.function.name == SPAWN_SUBAGENT
6462 || c.function.name == SUBAGENT_STATUS
6463 || c.function.name == SEND_MESSAGE
6464 || c.function.name == SUBAGENT_RESUME
6465 || (self.config.claude_runtime_tools_enabled
6466 && matches!(
6467 c.function.name.as_str(),
6468 CLAUDE_CRON_CREATE
6469 | CLAUDE_CRON_DELETE
6470 | CLAUDE_CRON_LIST
6471 | CLAUDE_SCHEDULE_WAKEUP
6472 ))
6473 }) {
6474 let mut out = Vec::with_capacity(calls.len());
6475 for call in calls {
6476 out.push(self.run_tool(call).await);
6477 }
6478 return out;
6479 }
6480 let prepared: Vec<PreparedCall> = calls.iter().map(|c| self.prepare_tool_call(c)).collect();
6481 let mut slots: Vec<Option<(String, bool)>> = prepared
6482 .iter()
6483 .map(|p| match p {
6484 PreparedCall::Done(r) => Some(r.clone()),
6485 PreparedCall::Ready { .. } => None,
6486 })
6487 .collect();
6488
6489 let ready_idxs: Vec<usize> = prepared
6490 .iter()
6491 .enumerate()
6492 .filter(|(_, p)| matches!(p, PreparedCall::Ready { .. }))
6493 .map(|(i, _)| i)
6494 .collect();
6495
6496 if !ready_idxs.is_empty() {
6497 let futs = ready_idxs.iter().map(|&i| {
6498 let PreparedCall::Ready { name, args } = &prepared[i] else {
6499 unreachable!("filtered to Ready above")
6500 };
6501 // `self.registry.get` borrows `self.registry` immutably;
6502 // `self.ctx` is `Clone` (P4c precedent) so each future owns
6503 // its own copy rather than borrowing `self` across the
6504 // `.await` inside `join_all`.
6505 let tool = self.registry.get(name).expect("prepared as Ready");
6506 let args = args.clone();
6507 let ctx = self.ctx.clone();
6508 async move {
6509 match tool.execute(args, &ctx).await {
6510 Ok(out) => (out, false),
6511 Err(e) => (format!("Error: {e}"), true),
6512 }
6513 }
6514 });
6515 let results = futures::future::join_all(futs).await;
6516 for (idx, result) in ready_idxs.iter().zip(results) {
6517 slots[*idx] = Some(result);
6518 }
6519 }
6520
6521 let out: Vec<(String, bool)> = slots
6522 .into_iter()
6523 .map(|s| s.expect("every call resolved to Some above"))
6524 .collect();
6525 // Post-tool hook, in original order — only for calls that actually
6526 // reached `execute()` (matches `run_tool`'s existing behavior: an
6527 // intrinsic/denied/blocked call never fires the post-tool hook).
6528 let ready_set: std::collections::HashSet<usize> = ready_idxs.into_iter().collect();
6529 for (i, call) in calls.iter().enumerate() {
6530 if !ready_set.contains(&i) {
6531 continue;
6532 }
6533 let (output, is_error) = &out[i];
6534 if let Some(hook) = &self.config.post_tool_hook {
6535 hook(&call.function.name, output, *is_error);
6536 }
6537 }
6538 out
6539 }
6540
6541 /// P4c (§5.2 P4 "doom-loop breaker", §3.1 `core.doom_loop_threshold`):
6542 /// update the consecutive-identical-call streak for `(name, args)` and
6543 /// return `Some(reason)` the moment the streak reaches
6544 /// `Config.doom_loop_threshold` (a call whose name AND JSON-canonical
6545 /// arguments are byte-identical to the immediately preceding call
6546 /// extends the streak; anything else resets it to 1). `None`
6547 /// (`Config.doom_loop_threshold` unset, or `Some(n)` with `n < 2` — a
6548 /// threshold below 2 can never fire since the FIRST call already
6549 /// "repeats zero times") never touches the streak fields at all.
6550 fn check_doom_loop(&mut self, name: &str, args: &serde_json::Value) -> Option<String> {
6551 let threshold = self.config.doom_loop_threshold?;
6552 if threshold < 2 {
6553 return None;
6554 }
6555 // `serde_json::Value::Object` is a `BTreeMap` in this workspace (no
6556 // `preserve_order` feature), so `to_string()` is already
6557 // key-order-canonical — two calls that differ only in argument key
6558 // order are still treated as identical.
6559 let key = (name.to_string(), args.to_string());
6560 if self.doom_loop_last_call.as_ref() == Some(&key) {
6561 self.doom_loop_streak += 1;
6562 } else {
6563 self.doom_loop_last_call = Some(key);
6564 self.doom_loop_streak = 1;
6565 }
6566 if self.doom_loop_streak >= threshold {
6567 Some(format!(
6568 "doom-loop breaker: `{name}` called with identical arguments {} times in a row \
6569 — try a different approach instead of repeating the same call",
6570 self.doom_loop_streak
6571 ))
6572 } else {
6573 None
6574 }
6575 }
6576
6577 /// Whether `name` is in the eagerly-advertised "core" set for the current
6578 /// [`ToolAdvertising`] mode: every enabled tool under `Full`, or the
6579 /// explicit `core` allowlist under `Deferred`.
6580 fn is_core_tool(&self, name: &str) -> bool {
6581 match &self.config.tool_advertising {
6582 ToolAdvertising::Full => true,
6583 ToolAdvertising::Deferred { core } => core.iter().any(|c| c == name),
6584 }
6585 }
6586
6587 /// The schema advertised on the wire for `t`: the raw (as-shipped)
6588 /// schema with TR-8/T5's per-tool schema tier applied. This is what
6589 /// [`Self::tool_schemas`] sends every request.
6590 fn schema_for(&self, t: &dyn crate::tools::Tool) -> ToolSchema {
6591 let raw = self.raw_schema_for(t);
6592 let tier = self.config.schema_tier_for(t.name());
6593 let (description, parameters) =
6594 crate::tools::tiers::minify(&raw.description, &raw.parameters, tier);
6595 ToolSchema {
6596 name: raw.name,
6597 description,
6598 parameters,
6599 }
6600 }
6601
6602 /// The ORIGINAL, as-shipped schema for `t` — never tier-minified. This is
6603 /// the full contract [`Self::run_tool_search`] hands back on activation
6604 /// (TR-8/T5 dev/03: the B6 fetch path is the invert of tiering, so a
6605 /// model that fetched a tool via `tool_search` always sees the complete
6606 /// schema, byte-equal to `t.description()`/`t.parameters()` — modulo the
6607 /// pre-existing [`crate::Config::tool_description`] override, which is
6608 /// orthogonal to tiering).
6609 fn raw_schema_for(&self, t: &dyn crate::tools::Tool) -> ToolSchema {
6610 ToolSchema {
6611 name: t.name().to_string(),
6612 description: self
6613 .config
6614 .tool_description(t.name(), t.description())
6615 .to_string(),
6616 parameters: t.parameters(),
6617 }
6618 }
6619
6620 /// The synthetic `tool_search` schema advertised under `Deferred` (B6).
6621 fn tool_search_schema() -> ToolSchema {
6622 ToolSchema {
6623 name: TOOL_SEARCH.to_string(),
6624 description: "Search for additional tools not currently advertised (the deferred \
6625 MCP surface and any other non-core tools). Matches keywords case-insensitively \
6626 against each tool's name and description. Matched tools become callable starting \
6627 with your NEXT message, not this one."
6628 .to_string(),
6629 parameters: serde_json::json!({
6630 "type": "object",
6631 "properties": {
6632 "query": {
6633 "type": "string",
6634 "description": "Keyword(s) to search for in tool names and descriptions."
6635 },
6636 "max_results": {
6637 "type": "integer",
6638 "description": "Maximum number of matching tools to return."
6639 }
6640 },
6641 "required": ["query"],
6642 "additionalProperties": false
6643 }),
6644 }
6645 }
6646
6647 /// The tool-schema array this agent would advertise on its NEXT
6648 /// request, exactly as `Self::run_loop` computes it. Public
6649 /// (PARITY-18 D1) so a caller can measure the real request-token cost
6650 /// of an agent's tool surface — including the current
6651 /// [`crate::config::ToolAdvertising`] mode's core/deferred split and
6652 /// the synthetic `tool_search`/`expand_reduction`/`sidecar_search`
6653 /// schemas — BEFORE ever calling [`Self::send`], e.g. for a preflight
6654 /// context-guard check.
6655 pub fn tool_schemas(&self) -> Vec<ToolSchema> {
6656 let mut out = match &self.config.tool_advertising {
6657 ToolAdvertising::Full => self
6658 .registry
6659 .iter()
6660 .filter(|t| self.config.tool_enabled(t.name()))
6661 .map(|t| self.schema_for(t))
6662 .collect(),
6663 ToolAdvertising::Deferred { .. } => {
6664 let mut out: Vec<ToolSchema> = self
6665 .registry
6666 .iter()
6667 .filter(|t| self.config.tool_enabled(t.name()))
6668 .filter(|t| {
6669 self.is_core_tool(t.name()) || self.activated_tools.contains(t.name())
6670 })
6671 .map(|t| self.schema_for(t))
6672 .collect();
6673 out.push(Self::tool_search_schema());
6674 out
6675 }
6676 };
6677 // T12/TR-1: `expand_reduction`/`sidecar_search` are orthogonal to
6678 // `tool_advertising` (which governs the ordinary tool surface) —
6679 // advertised whenever a `ReductionPolicy` is installed, regardless of
6680 // Full/Deferred, since only a reduced session ever has anything to
6681 // expand or search (SPEC.md TR-1 dev/01).
6682 if self.reduction_policy.is_some() {
6683 out.push(Self::expand_reduction_schema());
6684 out.push(Self::sidecar_search_schema());
6685 }
6686 // P5-3 (§2 module 9): `spawn_subagent`/`subagent_status` are
6687 // orthogonal to `tool_advertising` too, same reasoning as
6688 // `expand_reduction`/`sidecar_search` above — advertised whenever
6689 // `Config::subagents_enabled` is on, Full or Deferred alike.
6690 // `false` (the default) never appends either, so a config that
6691 // never turns the module on gets byte-identical tool schemas to
6692 // today.
6693 if self.config.subagents_enabled {
6694 out.push(self.spawn_subagent_schema());
6695 if self.config.subagents_claude_agent_alias {
6696 out.push(self.claude_agent_schema());
6697 }
6698 if self.config.subagents_background {
6699 out.push(Self::subagent_status_schema());
6700 // BP-7 (catalog §4a "Background subagents + resume"): the
6701 // two halves the row named as missing — a mailbox into a
6702 // still-running child, and a resume of a finished one with
6703 // its context intact.
6704 out.push(Self::send_message_schema());
6705 out.push(Self::subagent_resume_schema());
6706 }
6707 }
6708 if self.config.claude_runtime_tools_enabled {
6709 out.extend(self.claude_builtin_tool_schemas());
6710 out.push(Self::claude_cron_create_schema());
6711 out.push(Self::claude_cron_delete_schema());
6712 out.push(Self::claude_cron_list_schema());
6713 out.push(Self::claude_schedule_wakeup_schema());
6714 }
6715 // P5-6 (§2 module 4 `tools.background`): same orthogonal-to-
6716 // `tool_advertising` treatment, advertised whenever
6717 // `Config::tools_background_enabled` is on. `false` (the default)
6718 // never appends any of the four, so a config that never turns the
6719 // module on gets byte-identical tool schemas to today.
6720 if self.config.tools_background_enabled {
6721 out.push(Self::background_exec_schema());
6722 out.push(Self::background_status_schema());
6723 out.push(Self::background_list_schema());
6724 out.push(Self::background_kill_schema());
6725 }
6726 // BP-10 (catalog row "Tool hiding via policy", cc§4 "bare-name
6727 // deny"): a policy deny does not merely REFUSE the call at
6728 // dispatch — it removes the tool from the model's view. Applied
6729 // once, here, over the finished array, so every family appended
6730 // above (`spawn_subagent`, `background_*`, the Claude aliases,
6731 // `expand_reduction`, …) is hidden by the same one rule, not by a
6732 // per-family repeat of it. See [`Self::policy_hides_tool`] for
6733 // which deny tier is consulted and why.
6734 out.retain(|schema| !self.policy_hides_tool(&schema.name));
6735 out
6736 }
6737
6738 /// BP-10: whether the CONFIG-DECLARED deny tier hides `name` from the
6739 /// model's tool surface entirely (cc§4: CC's bare-name deny "removes
6740 /// the tool from the model's view", where an ordinary rule only
6741 /// refuses the call).
6742 ///
6743 /// The ONE engine decides: this is
6744 /// [`crate::permissions::RuleSet::evaluate`] with `subject: None`, so
6745 /// exactly the patterns that can be satisfied by a tool NAME ALONE
6746 /// (`"bash"`, `"mcp_*"`, `"*"`) hide; a rule that names a
6747 /// command/path constraint (`"bash(rm -rf*)"`, `"write(.git/**)"`) is
6748 /// not satisfiable without a subject and therefore never hides a tool
6749 /// — the same `rule_matches` contract the dispatch gate uses.
6750 ///
6751 /// **Which deny tier.** `Config::tool_deny_patterns` — the
6752 /// `capabilities.permissions.rules.deny` array — and NOT the two
6753 /// runtime narrowings the dispatch gate folds in beside it:
6754 /// `protected_paths` expands to `read(...)`/`write(...)` patterns that
6755 /// carry a subject by construction (so they could never match here
6756 /// anyway), and `plan_mode::deny_rules` is a MODE, not a policy — CC's
6757 /// plan mode refuses a write, it does not make Write disappear and
6758 /// reappear as the mode toggles mid-session. Hiding is a property of
6759 /// the configured policy, which is fixed for the run.
6760 ///
6761 /// Gated on [`Config::permissions_enabled`]: a config that never turns
6762 /// the module on gets byte-identical schemas to before this existed.
6763 fn policy_hides_tool(&self, name: &str) -> bool {
6764 if !self.config.permissions_enabled || self.config.tool_deny_patterns.is_empty() {
6765 return false;
6766 }
6767 let rules = crate::permissions::RuleSet {
6768 deny: self.config.tool_deny_patterns.clone(),
6769 ..Default::default()
6770 };
6771 rules.evaluate(name, None) == Some(crate::permissions::Decision::Deny)
6772 }
6773
6774 fn claude_builtin_tool_schemas(&self) -> Vec<ToolSchema> {
6775 let mut schemas = Vec::new();
6776 let mut push = |alias: &str, native: &str, description: &str, parameters| {
6777 if self.registry.get(native).is_some() && self.config.tool_enabled(native) {
6778 schemas.push(ToolSchema {
6779 name: alias.to_string(),
6780 description: description.to_string(),
6781 parameters,
6782 });
6783 }
6784 };
6785 push(
6786 CLAUDE_BASH,
6787 "bash",
6788 "Claude Code-compatible shell command execution.",
6789 serde_json::json!({
6790 "type": "object",
6791 "properties": {
6792 "command": {"type": "string"},
6793 "timeout": {"type": "integer", "description": "Timeout in milliseconds."},
6794 "description": {"type": "string"}
6795 },
6796 "required": ["command"],
6797 "additionalProperties": true
6798 }),
6799 );
6800 push(
6801 CLAUDE_READ,
6802 "read_file",
6803 "Claude Code-compatible file reader.",
6804 serde_json::json!({
6805 "type": "object",
6806 "properties": {
6807 "file_path": {"type": "string"},
6808 "offset": {"type": "integer"},
6809 "limit": {"type": "integer"}
6810 },
6811 "required": ["file_path"],
6812 "additionalProperties": false
6813 }),
6814 );
6815 push(
6816 CLAUDE_WRITE,
6817 "write_file",
6818 "Claude Code-compatible file writer.",
6819 serde_json::json!({
6820 "type": "object",
6821 "properties": {"file_path": {"type": "string"}, "content": {"type": "string"}},
6822 "required": ["file_path", "content"],
6823 "additionalProperties": false
6824 }),
6825 );
6826 push(
6827 CLAUDE_EDIT,
6828 "edit_file",
6829 "Claude Code-compatible exact file edit.",
6830 serde_json::json!({
6831 "type": "object",
6832 "properties": {
6833 "file_path": {"type": "string"},
6834 "old_string": {"type": "string"},
6835 "new_string": {"type": "string"},
6836 "replace_all": {"type": "boolean"}
6837 },
6838 "required": ["file_path", "old_string", "new_string"],
6839 "additionalProperties": false
6840 }),
6841 );
6842 push(
6843 CLAUDE_GLOB,
6844 "glob",
6845 "Claude Code-compatible file glob.",
6846 serde_json::json!({
6847 "type": "object",
6848 "properties": {"pattern": {"type": "string"}, "path": {"type": "string"}},
6849 "required": ["pattern"],
6850 "additionalProperties": false
6851 }),
6852 );
6853 push(
6854 CLAUDE_GREP,
6855 "search",
6856 "Claude Code-compatible content search.",
6857 serde_json::json!({
6858 "type": "object",
6859 "properties": {"pattern": {"type": "string"}, "path": {"type": "string"}},
6860 "required": ["pattern"],
6861 "additionalProperties": true
6862 }),
6863 );
6864 schemas
6865 }
6866
6867 fn translate_claude_builtin_call(
6868 &self,
6869 call: &supercode_interchange::ToolCall,
6870 ) -> Result<Option<supercode_interchange::ToolCall>> {
6871 let native = match call.function.name.as_str() {
6872 CLAUDE_BASH => "bash",
6873 CLAUDE_READ => "read_file",
6874 CLAUDE_WRITE => "write_file",
6875 CLAUDE_EDIT => "edit_file",
6876 CLAUDE_GLOB => "glob",
6877 CLAUDE_GREP => "search",
6878 _ => return Ok(None),
6879 };
6880 let mut args = call.function.parsed_arguments()?;
6881 let object = args
6882 .as_object_mut()
6883 .ok_or_else(|| Error::InvalidArguments {
6884 tool: call.function.name.clone(),
6885 message: "expected a JSON object".to_string(),
6886 })?;
6887 if let Some(path) = object.remove("file_path") {
6888 object.entry("path".to_string()).or_insert(path);
6889 }
6890 if call.function.name == CLAUDE_BASH {
6891 if let Some(timeout) = object.remove("timeout") {
6892 object.entry("timeout_ms".to_string()).or_insert(timeout);
6893 }
6894 }
6895 if call.function.name == CLAUDE_GLOB {
6896 if let Some(path) = object
6897 .remove("path")
6898 .and_then(|value| value.as_str().map(str::to_owned))
6899 {
6900 if let Some(pattern) = object.get_mut("pattern") {
6901 if let Some(value) = pattern.as_str() {
6902 if !std::path::Path::new(value).is_absolute() {
6903 *pattern = serde_json::Value::String(
6904 std::path::Path::new(&path)
6905 .join(value)
6906 .to_string_lossy()
6907 .into_owned(),
6908 );
6909 }
6910 }
6911 }
6912 }
6913 }
6914 let mut translated = call.clone();
6915 translated.function.name = native.to_string();
6916 translated.function.arguments = serde_json::to_string(&args)?;
6917 Ok(Some(translated))
6918 }
6919
6920 fn claude_cron_create_schema() -> ToolSchema {
6921 ToolSchema {
6922 name: CLAUDE_CRON_CREATE.to_string(),
6923 description: "Record a Claude-compatible cron job in the imported runtime manifest. \
6924 The job inherits the manifest's ACTIVE or PAUSED posture; an embedding scheduler, \
6925 not this agent loop, owns execution."
6926 .to_string(),
6927 parameters: serde_json::json!({
6928 "type": "object",
6929 "properties": {
6930 "cron": {"type": "string", "description": "Cron expression to preserve."},
6931 "prompt": {"type": "string", "description": "Prompt associated with the job."},
6932 "recurring": {"type": "boolean", "default": false},
6933 "durable": {"type": "boolean", "default": false}
6934 },
6935 "required": ["cron", "prompt"],
6936 "additionalProperties": false
6937 }),
6938 }
6939 }
6940
6941 fn claude_cron_delete_schema() -> ToolSchema {
6942 ToolSchema {
6943 name: CLAUDE_CRON_DELETE.to_string(),
6944 description: "Delete a Claude-compatible cron job from the imported manifest. \
6945 This updates state only; an embedding scheduler owns execution."
6946 .to_string(),
6947 parameters: serde_json::json!({
6948 "type": "object",
6949 "properties": {"id": {"type": "string"}},
6950 "required": ["id"],
6951 "additionalProperties": false
6952 }),
6953 }
6954 }
6955
6956 fn claude_cron_list_schema() -> ToolSchema {
6957 ToolSchema {
6958 name: CLAUDE_CRON_LIST.to_string(),
6959 description: "List imported Claude cron jobs and their explicit ACTIVE or PAUSED \
6960 manifest posture. This agent loop itself does not run a scheduler."
6961 .to_string(),
6962 parameters: serde_json::json!({
6963 "type": "object",
6964 "properties": {},
6965 "additionalProperties": false
6966 }),
6967 }
6968 }
6969
6970 fn claude_schedule_wakeup_schema() -> ToolSchema {
6971 ToolSchema {
6972 name: CLAUDE_SCHEDULE_WAKEUP.to_string(),
6973 description: "Replace the one-shot wakeup stored in the imported Claude manifest. \
6974 The wakeup inherits the manifest's ACTIVE or PAUSED posture; an embedding scheduler \
6975 owns timer execution."
6976 .to_string(),
6977 parameters: serde_json::json!({
6978 "type": "object",
6979 "properties": {
6980 "delaySeconds": {"type": "integer", "minimum": 0},
6981 "reason": {"type": "string"},
6982 "prompt": {"type": "string"}
6983 },
6984 "required": ["delaySeconds"],
6985 "additionalProperties": false
6986 }),
6987 }
6988 }
6989
6990 /// The `background_exec` schema (P5-6, §2 module 4, D1 "background
6991 /// exec").
6992 fn background_exec_schema() -> ToolSchema {
6993 ToolSchema {
6994 name: BACKGROUND_EXEC.to_string(),
6995 description: "Run a shell command in the BACKGROUND: spawns it as a detached \
6996 process and returns a `job_id` IMMEDIATELY, before the command finishes — this \
6997 call never returns the command's output. Poll `background_status` with the \
6998 `job_id` to check progress and retrieve captured output; use `background_kill` \
6999 to cancel it early. The command goes through the exact same sandbox/permission \
7000 checks as a foreground `bash` call, and any check that would need an \
7001 interactive approval is denied automatically (a background job cannot wait for \
7002 one)."
7003 .to_string(),
7004 parameters: serde_json::json!({
7005 "type": "object",
7006 "properties": {
7007 "command": {
7008 "type": "string",
7009 "description": "Shell command to run in the background via `sh -c`."
7010 }
7011 },
7012 "required": ["command"],
7013 "additionalProperties": false
7014 }),
7015 }
7016 }
7017
7018 /// The `background_status` schema (P5-6, D1 "monitor/event feed").
7019 fn background_status_schema() -> ToolSchema {
7020 ToolSchema {
7021 name: BACKGROUND_STATUS.to_string(),
7022 description: "Check on a background job spawned via background_exec: its \
7023 running/exited/killed status, exit code (once known), and the command's \
7024 captured stdout/stderr so far (bounded — very large output is truncated with a \
7025 marker). Once the job has exited or been killed, this call also reaps it (it \
7026 will no longer appear in background_list or accept further status polls)."
7027 .to_string(),
7028 parameters: serde_json::json!({
7029 "type": "object",
7030 "properties": {
7031 "job_id": {
7032 "type": "string",
7033 "description": "The id `background_exec` returned when this job was \
7034 started."
7035 }
7036 },
7037 "required": ["job_id"],
7038 "additionalProperties": false
7039 }),
7040 }
7041 }
7042
7043 /// The `background_list` schema (P5-6, D10 "bg-manager").
7044 fn background_list_schema() -> ToolSchema {
7045 ToolSchema {
7046 name: BACKGROUND_LIST.to_string(),
7047 description: "List every background job currently tracked (running, or finished \
7048 but not yet polled via background_status) — job id, command, status, pid, and \
7049 start time for each. Does not retrieve output or reap anything."
7050 .to_string(),
7051 parameters: serde_json::json!({
7052 "type": "object",
7053 "properties": {},
7054 "additionalProperties": false
7055 }),
7056 }
7057 }
7058
7059 /// The `background_kill` schema (P5-6, D10 "bg-manager").
7060 fn background_kill_schema() -> ToolSchema {
7061 ToolSchema {
7062 name: BACKGROUND_KILL.to_string(),
7063 description: "Kill a background job's real process immediately (a no-op, not an \
7064 error, if it already exited on its own) and reap it."
7065 .to_string(),
7066 parameters: serde_json::json!({
7067 "type": "object",
7068 "properties": {
7069 "job_id": {
7070 "type": "string",
7071 "description": "The id `background_exec` returned when this job was \
7072 started."
7073 }
7074 },
7075 "required": ["job_id"],
7076 "additionalProperties": false
7077 }),
7078 }
7079 }
7080
7081 /// The `spawn_subagent` schema (P5-3, §2 module 9 D1 "spawn tool").
7082 /// Lists every configured `agent_type` name so the model knows what's
7083 /// available, but `agent_type` stays optional — an ad-hoc spawn with an
7084 /// inline `system_prompt` is always allowed too.
7085 fn spawn_subagent_schema(&self) -> ToolSchema {
7086 let mut names: Vec<&str> = self
7087 .config
7088 .subagents_definitions
7089 .keys()
7090 .map(String::as_str)
7091 .collect();
7092 names.sort_unstable();
7093 let agent_type_desc = if names.is_empty() {
7094 "Optional named subagent type to run (none configured — omit this and pass \
7095 `system_prompt` instead)."
7096 .to_string()
7097 } else {
7098 format!(
7099 "Optional named subagent type to run: {}. Omit to run an ad-hoc subagent with \
7100 your own `system_prompt` instead.",
7101 names.join(", ")
7102 )
7103 };
7104 let background_desc = if self.config.subagents_background {
7105 "Run this subagent in the background instead of waiting for it — this call \
7106 returns immediately with a `subagent_id`; poll `subagent_status` with that id for \
7107 the result."
7108 } else {
7109 "Background subagents are disabled for this agent — this must be omitted or false."
7110 };
7111 ToolSchema {
7112 name: SPAWN_SUBAGENT.to_string(),
7113 description: "Spawn a subagent to work on a self-contained task and (by default) \
7114 wait for its final answer, which is returned as this call's result. The \
7115 subagent runs its own independent reasoning/tool loop; it does not see your \
7116 conversation except for the `task` text you give it here."
7117 .to_string(),
7118 parameters: serde_json::json!({
7119 "type": "object",
7120 "properties": {
7121 "task": {
7122 "type": "string",
7123 "description": "The self-contained task/prompt for the subagent."
7124 },
7125 "agent_type": {
7126 "type": "string",
7127 "description": agent_type_desc
7128 },
7129 "system_prompt": {
7130 "type": "string",
7131 "description": "Inline system prompt for an ad-hoc subagent (ignored \
7132 if `agent_type` is given — the named type's own prompt is used \
7133 instead)."
7134 },
7135 "background": {
7136 "type": "boolean",
7137 "description": background_desc
7138 }
7139 },
7140 "required": ["task"],
7141 "additionalProperties": false
7142 }),
7143 }
7144 }
7145
7146 /// Claude Code-compatible alias for [`Self::spawn_subagent_schema`].
7147 fn claude_agent_schema(&self) -> ToolSchema {
7148 let mut names: Vec<String> = self.config.subagents_definitions.keys().cloned().collect();
7149 names.push("general-purpose".into());
7150 names.sort_unstable();
7151 names.dedup();
7152 ToolSchema {
7153 name: CLAUDE_AGENT.to_string(),
7154 description: "Claude Code-compatible subagent dispatcher. Runs a named or ad-hoc \
7155 child agent; children default to background execution in this compatibility mode."
7156 .to_string(),
7157 parameters: serde_json::json!({
7158 "type": "object",
7159 "properties": {
7160 "prompt": {"type": "string", "description": "Self-contained child task."},
7161 "subagent_type": {
7162 "type": "string",
7163 "description": format!("Named agent type. Available: {}", names.join(", "))
7164 },
7165 "description": {
7166 "type": "string",
7167 "description": "Short human-facing task label; preserved as descriptive input."
7168 },
7169 "model": {
7170 "type": "string",
7171 "description": "Optional model alias or full provider slug for this child."
7172 },
7173 "run_in_background": {
7174 "type": "boolean",
7175 "description": "Whether to return immediately with a child id (default true)."
7176 }
7177 },
7178 "required": ["prompt"],
7179 "additionalProperties": false
7180 }),
7181 }
7182 }
7183
7184 /// Translate Claude's `Agent` arguments to the native subagent intrinsic.
7185 fn translate_claude_agent_call(
7186 &self,
7187 call: &supercode_interchange::ToolCall,
7188 ) -> Result<supercode_interchange::ToolCall> {
7189 let args = call
7190 .function
7191 .parsed_arguments()
7192 .map_err(|error| Error::InvalidArguments {
7193 tool: CLAUDE_AGENT.to_string(),
7194 message: error.to_string(),
7195 })?;
7196 let object = args.as_object().ok_or_else(|| Error::InvalidArguments {
7197 tool: CLAUDE_AGENT.to_string(),
7198 message: "arguments must be an object".to_string(),
7199 })?;
7200 let mut translated = serde_json::Map::new();
7201 if let Some(value) = object.get("prompt") {
7202 translated.insert("task".to_string(), value.clone());
7203 }
7204 if let Some(value) = object.get("subagent_type") {
7205 // `general-purpose` is a built-in Claude agent, not a project
7206 // definition file. Supercode's equivalent is an ad-hoc child
7207 // using the inherited default system prompt, represented by an
7208 // omitted `agent_type`.
7209 if value.as_str() != Some("general-purpose") {
7210 translated.insert("agent_type".to_string(), value.clone());
7211 }
7212 }
7213 if let Some(value) = object.get("model") {
7214 translated.insert("model".to_string(), value.clone());
7215 }
7216 translated.insert(
7217 "background".to_string(),
7218 object
7219 .get("run_in_background")
7220 .cloned()
7221 .unwrap_or(serde_json::Value::Bool(true)),
7222 );
7223 Ok(supercode_interchange::ToolCall {
7224 id: call.id.clone(),
7225 kind: call.kind.clone(),
7226 function: supercode_interchange::FunctionCall {
7227 name: SPAWN_SUBAGENT.to_string(),
7228 arguments: serde_json::Value::Object(translated).to_string(),
7229 },
7230 })
7231 }
7232
7233 /// Execute Claude's scheduling vocabulary against the imported manifest.
7234 ///
7235 /// This is intentionally a state editor, not a scheduler: it owns no
7236 /// timer/task handle, nothing downstream of it fires, and every
7237 /// successful response says so, so the model is never told a job it just
7238 /// created will run here.
7239 fn run_claude_runtime_tool(
7240 &mut self,
7241 call: &supercode_interchange::ToolCall,
7242 ) -> (String, bool) {
7243 let args = match call.function.parsed_arguments() {
7244 Ok(value) if value.is_object() => value,
7245 Ok(_) => {
7246 return (
7247 format!("Error: {} arguments must be an object", call.function.name),
7248 true,
7249 )
7250 }
7251 Err(error) => return (format!("Error: {error}"), true),
7252 };
7253 let object = args.as_object().expect("checked object above");
7254
7255 let Some(manifest) = self.claude_runtime_manifest.as_mut() else {
7256 return (
7257 "Error: Claude runtime compatibility was enabled without an imported runtime \
7258 manifest; refusing to invent scheduler state"
7259 .to_string(),
7260 true,
7261 );
7262 };
7263 // Every imported schedule is carried and inert. `state` is the single
7264 // fact the model is told about it, in the manifest's own vocabulary.
7265 let state = "paused";
7266
7267 match call.function.name.as_str() {
7268 CLAUDE_CRON_LIST => {
7269 let jobs: Vec<serde_json::Value> = manifest
7270 .active_crons
7271 .iter()
7272 .map(|job| {
7273 serde_json::json!({
7274 "id": job.id,
7275 "cron": job.schedule,
7276 "prompt": job.prompt,
7277 "recurring": job.recurring,
7278 "durable": job.durable_requested,
7279 "state": state
7280 })
7281 })
7282 .collect();
7283 let notice = "Imported jobs are preserved but no scheduler is running.";
7284 (
7285 serde_json::json!({
7286 "execution_state": state,
7287 "execution_notice": notice,
7288 "jobs": jobs
7289 })
7290 .to_string(),
7291 false,
7292 )
7293 }
7294 CLAUDE_CRON_CREATE => {
7295 let Some(schedule) = object.get("cron").and_then(serde_json::Value::as_str) else {
7296 return ("Error: CronCreate requires string `cron`".to_string(), true);
7297 };
7298 let Some(prompt) = object.get("prompt").and_then(serde_json::Value::as_str) else {
7299 return (
7300 "Error: CronCreate requires string `prompt`".to_string(),
7301 true,
7302 );
7303 };
7304 let recurring = object
7305 .get("recurring")
7306 .and_then(serde_json::Value::as_bool)
7307 .unwrap_or(false);
7308 let durable_requested = object
7309 .get("durable")
7310 .and_then(serde_json::Value::as_bool)
7311 .unwrap_or(false);
7312 let mut sequence = 1_u64;
7313 let id = loop {
7314 let candidate = format!("sc{sequence:06}");
7315 if !manifest.active_crons.iter().any(|job| job.id == candidate) {
7316 break candidate;
7317 }
7318 sequence += 1;
7319 };
7320 let kind = if recurring { "recurring " } else { "" };
7321 let result = format!(
7322 "Scheduled {kind}job {id} ({schedule}) in PAUSED state. The job is preserved \
7323 in the continuation manifest but no scheduler is running and it will not execute."
7324 );
7325 manifest
7326 .active_crons
7327 .push(crate::claude_runtime_state::ClaudeCronJob {
7328 id: id.clone(),
7329 tool_use_id: call.id.clone(),
7330 schedule: schedule.to_string(),
7331 recurring,
7332 durable_requested,
7333 prompt: prompt.to_string(),
7334 // The creation instant is a fact about this
7335 // continuation, recorded like every other manifest
7336 // field. Nothing consults it as a due time.
7337 created_at: Some(supercode_interchange::sidecar::ms_to_rfc3339(now_ms())),
7338 expires_after_seconds: None,
7339 creation_result: result.clone(),
7340 });
7341 manifest
7342 .active_crons
7343 .sort_by(|left, right| left.id.cmp(&right.id));
7344 (result, false)
7345 }
7346 CLAUDE_CRON_DELETE => {
7347 let Some(id) = object.get("id").and_then(serde_json::Value::as_str) else {
7348 return ("Error: CronDelete requires string `id`".to_string(), true);
7349 };
7350 let Some(index) = manifest.active_crons.iter().position(|job| job.id == id) else {
7351 return (
7352 format!("Error: unknown {state} Claude cron job `{id}`"),
7353 true,
7354 );
7355 };
7356 manifest.active_crons.remove(index);
7357 (
7358 format!("Cancelled job {id}. The job was PAUSED; no execution occurred."),
7359 false,
7360 )
7361 }
7362 CLAUDE_SCHEDULE_WAKEUP => {
7363 let Some(delay_seconds) = object
7364 .get("delaySeconds")
7365 .and_then(serde_json::Value::as_u64)
7366 else {
7367 return (
7368 "Error: ScheduleWakeup requires integer `delaySeconds`".to_string(),
7369 true,
7370 );
7371 };
7372 let reason = object
7373 .get("reason")
7374 .and_then(serde_json::Value::as_str)
7375 .map(str::to_string);
7376 let prompt = object
7377 .get("prompt")
7378 .and_then(serde_json::Value::as_str)
7379 .map(str::to_string);
7380 let now = now_ms();
7381 let created_at = Some(supercode_interchange::sidecar::ms_to_rfc3339(now));
7382 // The instant the wakeup asks for, recorded as the request
7383 // made it. No timer consults it here.
7384 let delay_ms = i64::try_from(delay_seconds)
7385 .unwrap_or(i64::MAX)
7386 .saturating_mul(1_000);
7387 let scheduled_for =
7388 supercode_interchange::sidecar::ms_to_rfc3339(now.saturating_add(delay_ms));
7389 let result = format!(
7390 "Next wakeup recorded for {scheduled_for} (in {delay_seconds}s) in PAUSED \
7391 state. The request replaced the prior wakeup in the manifest, but no timer \
7392 is running and it will not execute."
7393 );
7394 manifest.pending_wakeups.clear();
7395 manifest
7396 .pending_wakeups
7397 .push(crate::claude_runtime_state::ClaudeWakeup {
7398 tool_use_id: call.id.clone(),
7399 delay_seconds,
7400 reason,
7401 prompt,
7402 created_at,
7403 scheduled_for: Some(scheduled_for),
7404 creation_result: result.clone(),
7405 });
7406 (result, false)
7407 }
7408 _ => unreachable!("runtime tool dispatch is name-gated"),
7409 }
7410 }
7411
7412 /// The `subagent_status` schema (P5-3, D3 "background+resume").
7413 fn subagent_status_schema() -> ToolSchema {
7414 ToolSchema {
7415 name: SUBAGENT_STATUS.to_string(),
7416 description: "Check on (and, once finished, retrieve the result of) a background \
7417 subagent spawned via spawn_subagent with background=true. Pass the \
7418 `subagent_id` that spawn returned."
7419 .to_string(),
7420 parameters: serde_json::json!({
7421 "type": "object",
7422 "properties": {
7423 "subagent_id": {
7424 "type": "string",
7425 "description": "The id `spawn_subagent` returned when this subagent \
7426 was spawned."
7427 }
7428 },
7429 "required": ["subagent_id"],
7430 "additionalProperties": false
7431 }),
7432 }
7433 }
7434
7435 /// The `send_message` schema.
7436 fn send_message_schema() -> ToolSchema {
7437 ToolSchema {
7438 name: SEND_MESSAGE.to_string(),
7439 description: "Send a message to another agent: one of your background subagents \
7440 that is STILL RUNNING (by the id `spawn_subagent` returned; it receives it at the \
7441 start of its next step), or another session of any harness, on this machine or an \
7442 enrolled one (by its name, name@machine, or sc: address; `supercode message list` \
7443 shows them). A session's reply comes back to you as a message. Send only what \
7444 asks something or carries a result; no acknowledgements."
7445 .to_string(),
7446 parameters: serde_json::json!({
7447 "type": "object",
7448 "properties": {
7449 "to": {
7450 "type": "string",
7451 "description": "A running subagent's id, or a session's name, name@machine or sc: address."
7452 },
7453 "message": {
7454 "type": "string",
7455 "description": "What to tell it."
7456 },
7457 "notify_when_idle": {
7458 "type": "boolean",
7459 "description": "For a session: also get one notice when its next turn ends."
7460 }
7461 },
7462 "required": ["to", "message"],
7463 "additionalProperties": false
7464 }),
7465 }
7466 }
7467
7468 /// BP-7: the `subagent_resume` schema.
7469 fn subagent_resume_schema() -> ToolSchema {
7470 ToolSchema {
7471 name: SUBAGENT_RESUME.to_string(),
7472 description: "Continue a subagent that has already FINISHED, with its own previous conversation restored, so it keeps everything it learned instead of being briefed again from scratch. Pass the id it was spawned with and the next task."
7473 .to_string(),
7474 parameters: serde_json::json!({
7475 "type": "object",
7476 "properties": {
7477 "subagent_id": {
7478 "type": "string",
7479 "description": "The id of a subagent that has already finished."
7480 },
7481 "task": {
7482 "type": "string",
7483 "description": "What the resumed subagent should do next."
7484 }
7485 },
7486 "required": ["subagent_id", "task"],
7487 "additionalProperties": false
7488 }),
7489 }
7490 }
7491
7492 /// BP-7 (catalog §4a "Background subagents + resume"): deliver a
7493 /// message into a still-running background child's mailbox.
7494 ///
7495 /// The mailbox is the child's own `SteerInbox` — the seam P4b built for
7496 /// mid-turn steering, which is writable while the child's turn holds
7497 /// `&mut Agent`. So delivery ordering is already defined: the message
7498 /// arrives at the top of the child's next loop iteration, i.e. after
7499 /// whatever tool calls it is currently running, per its
7500 /// `steering_mode`. A child that has already FINISHED is refused with
7501 /// a pointer at `subagent_resume`, which is the operation for that
7502 /// case — never silently dropped.
7503 async fn run_send_message(&mut self, call: &supercode_interchange::ToolCall) -> (String, bool) {
7504 let args = match call.function.parsed_arguments() {
7505 Ok(v) => v,
7506 Err(e) => {
7507 let err = Error::InvalidArguments {
7508 tool: SEND_MESSAGE.to_string(),
7509 message: e.to_string(),
7510 };
7511 return (format!("Error: {err}"), true);
7512 }
7513 };
7514 let Some(id) = args.get("to").and_then(serde_json::Value::as_str) else {
7515 let err = Error::InvalidArguments {
7516 tool: SEND_MESSAGE.to_string(),
7517 message: "`to` is required".to_string(),
7518 };
7519 return (format!("Error: {err}"), true);
7520 };
7521 let message = args
7522 .get("message")
7523 .and_then(serde_json::Value::as_str)
7524 .unwrap_or("");
7525 if message.is_empty() {
7526 let err = Error::InvalidArguments {
7527 tool: SEND_MESSAGE.to_string(),
7528 message: "`message` is required and must be non-empty".to_string(),
7529 };
7530 return (format!("Error: {err}"), true);
7531 }
7532 let Some(entry) = self.background_subagents.get(id) else {
7533 // Not one of this agent's children: another session.
7534 let notify_when_idle = args
7535 .get("notify_when_idle")
7536 .and_then(serde_json::Value::as_bool)
7537 .unwrap_or(false);
7538 return send_to_session(id, message, notify_when_idle).await;
7539 };
7540 if entry.handle.is_finished() {
7541 let out = serde_json::json!({
7542 "subagent_id": id,
7543 "status": "finished",
7544 "delivered": false,
7545 "hint": "this subagent already finished — collect it with subagent_status, then continue it with subagent_resume",
7546 });
7547 return (out.to_string(), false);
7548 }
7549 entry
7550 .mailbox
7551 .lock()
7552 .unwrap_or_else(std::sync::PoisonError::into_inner)
7553 .queue_unchecked(message.to_string());
7554 let out = serde_json::json!({
7555 "subagent_id": id,
7556 "status": "running",
7557 "delivered": true,
7558 });
7559 (out.to_string(), false)
7560 }
7561
7562 /// BP-7 (catalog §4a "Background subagents + resume": "resumable with
7563 /// context intact"): continue a finished child over its OWN transcript.
7564 ///
7565 /// The context comes from the reap (kept in-process) or, for a session
7566 /// that attached a subagent store, from that child's persisted
7567 /// `<parent>.subagents/<id>.sidecar.jsonl`. Either way the resumed
7568 /// child is rebuilt through `build_child_config` from the SAME named
7569 /// definition it was spawned with, so its permission posture on resume
7570 /// is the one it had originally — never a fresh, looser default.
7571 async fn run_subagent_resume(
7572 &mut self,
7573 call: &supercode_interchange::ToolCall,
7574 ) -> (String, bool) {
7575 let args = match call.function.parsed_arguments() {
7576 Ok(v) => v,
7577 Err(e) => {
7578 let err = Error::InvalidArguments {
7579 tool: SUBAGENT_RESUME.to_string(),
7580 message: e.to_string(),
7581 };
7582 return (format!("Error: {err}"), true);
7583 }
7584 };
7585 let Some(id) = args
7586 .get("subagent_id")
7587 .and_then(serde_json::Value::as_str)
7588 .map(String::from)
7589 else {
7590 let err = Error::InvalidArguments {
7591 tool: SUBAGENT_RESUME.to_string(),
7592 message: "`subagent_id` is required".to_string(),
7593 };
7594 return (format!("Error: {err}"), true);
7595 };
7596 let task = args
7597 .get("task")
7598 .and_then(serde_json::Value::as_str)
7599 .unwrap_or("")
7600 .to_string();
7601 if task.is_empty() {
7602 let err = Error::InvalidArguments {
7603 tool: SUBAGENT_RESUME.to_string(),
7604 message: "`task` is required and must be non-empty".to_string(),
7605 };
7606 return (format!("Error: {err}"), true);
7607 }
7608 if self
7609 .background_subagents
7610 .get(&id)
7611 .is_some_and(|e| !e.handle.is_finished())
7612 {
7613 let err = Error::tool(
7614 SUBAGENT_RESUME,
7615 format!(
7616 "subagent `{id}` is still running — send it a message with send_message, or collect it with subagent_status first"
7617 ),
7618 );
7619 return (format!("Error: {err}"), true);
7620 }
7621 let lineage = self
7622 .subagent_store
7623 .as_ref()
7624 .and_then(|(store, parent)| store.load_subagent_lineage(parent, &id).ok().flatten());
7625 let Some(prior) = self.prior_subagent_transcript(&id) else {
7626 let err = Error::SubagentNotFound(id.clone());
7627 return (format!("Error: {err}"), true);
7628 };
7629
7630 let agent_type = lineage.as_ref().and_then(|l| l.agent_type.clone());
7631 let definition = agent_type
7632 .as_ref()
7633 .and_then(|name| self.config.subagents_definitions.get(name).cloned());
7634 let Some(guard) = crate::subagents::try_acquire(
7635 &self.subagent_concurrency_gauge,
7636 self.config.subagents_max_concurrent,
7637 ) else {
7638 let err = Error::SubagentConcurrencyExceeded {
7639 max_concurrent: self.config.subagents_max_concurrent,
7640 };
7641 return (format!("Error: {err}"), true);
7642 };
7643 let child_config = self.build_child_config(
7644 definition.as_ref(),
7645 None,
7646 lineage
7647 .as_ref()
7648 .map(|l| l.model.clone())
7649 .or_else(|| definition.as_ref().and_then(|d| d.model.clone())),
7650 );
7651 let mut child = Agent::with_provider_arc(child_config, self.provider.clone());
7652 child.subagent_depth = self.subagent_depth + 1;
7653 child.subagent_concurrency_gauge = self.subagent_concurrency_gauge.clone();
7654 // Context intact: the child's own prior messages, appended after
7655 // its (re-derived, identical) system prompt.
7656 child.history.extend(prior);
7657
7658 let result = child.send(task).await;
7659 let transcript = child.history()[1..].to_vec();
7660 if let Some(lineage) = &lineage {
7661 self.persist_subagent_transcript(&id, lineage, &transcript);
7662 }
7663 self.reaped_subagents.insert(id.clone(), transcript);
7664 drop(guard);
7665 match result {
7666 Ok(text) => {
7667 let out = serde_json::json!({
7668 "subagent_id": id,
7669 "status": "done",
7670 "resumed": true,
7671 "result": text,
7672 });
7673 (out.to_string(), false)
7674 }
7675 Err(e) => {
7676 let out = serde_json::json!({
7677 "subagent_id": id,
7678 "status": "error",
7679 "resumed": true,
7680 "message": e.to_string(),
7681 });
7682 (out.to_string(), true)
7683 }
7684 }
7685 }
7686
7687 /// BP-7 (catalog §4a "Named agent definitions as data"): the child
7688 /// `Config` a `spawn_subagent` of `agent_type` would build — the
7689 /// resolved posture a named definition actually produces, including
7690 /// its [`crate::subagents::AgentPermissions`] bundle applied through
7691 /// the tightening-only rules. `None` when no definition of that name
7692 /// is registered or discovered.
7693 ///
7694 /// Exposed so a caller (and this build's tests) can ask what a named
7695 /// agent WOULD run as without spawning it and paying for a turn.
7696 pub fn child_config_for_agent_type(&self, agent_type: &str) -> Option<Config> {
7697 let definition = self.config.subagents_definitions.get(agent_type)?.clone();
7698 Some(self.build_child_config(Some(&definition), None, definition.model.clone()))
7699 }
7700
7701 /// BP-7 (catalog §4a "Background subagents + resume"): the ids of
7702 /// children that have finished and been reaped, and can therefore be
7703 /// continued with [`SUBAGENT_RESUME`].
7704 pub fn reaped_subagent_ids(&self) -> Vec<String> {
7705 let mut ids: Vec<String> = self.reaped_subagents.keys().cloned().collect();
7706 ids.sort();
7707 ids
7708 }
7709
7710 /// BP-7: a finished child's own messages — from the in-process reap
7711 /// cache first, then this session's subagent store.
7712 fn prior_subagent_transcript(&self, id: &str) -> Option<Vec<ChatMessage>> {
7713 if let Some(messages) = self.reaped_subagents.get(id) {
7714 return Some(messages.clone());
7715 }
7716 let (store, parent) = self.subagent_store.as_ref()?;
7717 let jsonl = store.load_subagent_transcript(parent, id).ok()??;
7718 let session = supercode_interchange::session::Session::from_sidecar_str(&jsonl).ok()?;
7719 Some(
7720 session
7721 .messages
7722 .into_iter()
7723 .filter(|m| m.role != supercode_interchange::Role::System)
7724 .collect(),
7725 )
7726 }
7727
7728 /// Build the CHILD `Config` a `spawn_subagent` call constructs its
7729 /// [`Agent`] from. The whole point of this method (§5.3-style
7730 /// "monotonic posture", build-brief "a subagent inherits or narrows —
7731 /// never widens — the parent's permission posture"): every field that
7732 /// governs what the child is ALLOWED to do (sandbox, approval,
7733 /// tool_overrides, deny/allow patterns, protected paths, the subagents
7734 /// caps themselves) is copied VERBATIM from `self.config` — never
7735 /// loosened — and the only NARROWING lever is `definition.tools`
7736 /// (intersected with whatever the parent already had enabled, never
7737 /// unioned in anything new).
7738 ///
7739 /// P5-3 safety hardening (Fable-5 review, LOW-MEDIUM "child safety-limit
7740 /// inheritance"): the monotonic-posture guarantee above was, before this
7741 /// fix, scoped to PERMISSION fields only — a child could still silently
7742 /// get a LOOSER safety BUDGET/BREAKER than its parent, because
7743 /// `max_total_output_tokens`/`max_tool_output_bytes`/`max_tokens`/
7744 /// `doom_loop_threshold`/`edit_file_require_read_before_edit` were never
7745 /// copied and so fell back to `Config::default()`'s (looser/uncapped)
7746 /// values on every spawn regardless of what the parent had configured.
7747 /// These are now copied verbatim alongside the permission-posture
7748 /// fields — a parent that capped its own output/tool-output/doom-loop
7749 /// exposure, or required read-before-edit, gets a child that is bound
7750 /// by the exact same ceiling, never a wider one.
7751 ///
7752 /// **Full field-by-field accounting** (every [`Config`] field, so this
7753 /// doc comment stays the single place that answers "did we forget
7754 /// one?"): fields already copied above/below this note (permission
7755 /// posture: `sandbox`/`approval`/`tool_overrides`/`auto_approved_tools`/
7756 /// `tool_deny_patterns`/`tool_allow_patterns`/`permissions_enabled`/
7757 /// `permissions_ask_patterns`/`permissions_protected_paths`/
7758 /// `network_policy`/`core_tools_enabled`/`module_registry`/
7759 /// `module_activation`/every `subagents_*` field; safety limits:
7760 /// `max_iterations`/`max_total_output_tokens`/`max_tool_output_bytes`/
7761 /// `max_tokens`/`doom_loop_threshold`/`edit_file_require_read_before_edit`;
7762 /// identity/transport: `model`/`system_prompt`/`cwd`/`base_url`/
7763 /// `api_key`/`api_key_env`/`api_key_cmd`) are the ones that gate
7764 /// harm/spend/hazard exposure. Every OTHER field is deliberately left at
7765 /// `Config::default()` because none of them is a safety ceiling the
7766 /// child could "loosen" by missing it:
7767 /// - `temperature`/`effort`/`response_format`/`extra_body`/`extra_headers`/
7768 /// `tool_advertising`/`tool_schema_tier`/`cache_plan`/`cache_warnings`/
7769 /// `reduction_policy`/
7770 /// `session_*`/`small_model`/`model_fallback`/`env_context`/
7771 /// `project_root_markers`/`project_doc_max_bytes`/`instruction_imports`/
7772 /// `retry_*`/`compaction_*`/`auto_title`/`steering_mode`/
7773 /// `follow_up_mode`/`read_file_multimodal`/`edit_file_notebook_aware`/
7774 /// `shell_env_snapshot`/`nested_instructions`/`model_switch_allow_switch`/
7775 /// `context_injections`/`context_injection_blocks`/`parallel_tool_calls`
7776 /// are behavior/cost-shaping or presentation knobs, not hard guards —
7777 /// a child defaulting on any of these can do LESS (e.g. no multimodal
7778 /// read, no notebook-aware edits, no proactive compaction) or the same,
7779 /// never something the parent hadn't already exposed it to. Several
7780 /// default to their OFF/conservative state (`false`/`None`), which is
7781 /// the tight direction, not the loose one.
7782 /// - `additional_dirs`: governs which extra roots are reachable at all
7783 /// (`presets.rs`'s `[core] additional_dirs` note) — a child that
7784 /// doesn't inherit it has FEWER reachable roots than its parent, i.e.
7785 /// strictly tighter, never looser.
7786 /// - `load_project_context`: whether instruction files are auto-loaded
7787 /// into the system prompt — a read-time convenience, not an access
7788 /// grant (`sandbox`/`permissions_protected_paths` already gate actual
7789 /// file access).
7790 /// - `prompts`: named `/slash` command templates for THIS agent's own
7791 /// user-facing input surface, not something the model can invoke
7792 /// against the child's tool surface.
7793 /// - `stop_gate`/`post_tool_hook`/`approval_handler`/`event_sink`:
7794 /// code-only `Box<dyn Fn>` callbacks (see the `pre_tool_hook` note
7795 /// immediately below — same non-`Clone` shape) that are observational
7796 /// or terminate-only, not a call-time veto over what a tool is allowed
7797 /// to do; `approval_handler` specifically is ALREADY documented at
7798 /// this method's call site (`Self::run_spawn_subagent`) as
7799 /// intentionally never set here — a foreground child gets no handler
7800 /// by design, an embedder installs its own after spawn if it wants
7801 /// one.
7802 ///
7803 /// **`pre_tool_hook` cannot propagate, and this is deliberate + named,
7804 /// not a silent gap**: `Config::pre_tool_hook` is a `Box<dyn Fn(&str,
7805 /// &serde_json::Value) -> Option<String> + Send + Sync>` — an
7806 /// embedder's own call-time veto over every tool call. `Box<dyn Fn>` is
7807 /// not `Clone` (there is no generic way to duplicate an opaque closure),
7808 /// so it genuinely CANNOT be copied into a child `Config` the way every
7809 /// `Clone`-able field above is — there is no fix that makes this one
7810 /// "verbatim copy" like the others. An embedder relying on a
7811 /// `pre_tool_hook` veto reaching spawned children as well as the parent
7812 /// MUST re-install one on the child explicitly (e.g. via a
7813 /// `spawn_subagent`-adjacent hook of their own, or by not relying on
7814 /// `pre_tool_hook` alone for anything safety-critical across a spawn
7815 /// boundary) — named here so this is a documented contract, not a gap
7816 /// an embedder discovers by a child silently misbehaving.
7817 fn build_child_config(
7818 &self,
7819 definition: Option<&crate::subagents::NamedAgentDefinition>,
7820 inline_system_prompt: Option<String>,
7821 model_override: Option<String>,
7822 ) -> Config {
7823 let system_prompt = definition
7824 .map(|d| d.system_prompt.clone())
7825 .filter(|s| !s.is_empty())
7826 .or(inline_system_prompt)
7827 .unwrap_or_else(|| self.config.system_prompt.clone());
7828 let model = model_override.unwrap_or_else(|| self.config.model.clone());
7829
7830 let mut child = Config::builder()
7831 .model(model)
7832 .system_prompt(system_prompt)
7833 .cwd(self.config.cwd.clone())
7834 // Monotonic: verbatim, never loosened.
7835 .sandbox(self.config.sandbox)
7836 .approval(self.config.approval)
7837 .max_iterations(self.config.max_iterations)
7838 .build();
7839 child.base_url = self.config.base_url.clone();
7840 child.api_key = self.config.api_key.clone();
7841 child.api_key_env = self.config.api_key_env.clone();
7842 child.api_key_cmd = self.config.api_key_cmd.clone();
7843 // P5-3 safety hardening (Fable-5 review, LOW-MEDIUM "child
7844 // safety-limit inheritance"): the monotonic-posture spirit extends
7845 // to safety BUDGETS/BREAKERS, not just permissions — a child must
7846 // not get a looser cap/breaker than its parent by simply falling
7847 // back to `Config::default()`'s (looser) values. See this method's
7848 // doc comment for the full field-by-field accounting.
7849 child.max_total_output_tokens = self.config.max_total_output_tokens;
7850 child.max_tool_output_bytes = self.config.max_tool_output_bytes;
7851 child.max_tokens = self.config.max_tokens;
7852 child.doom_loop_threshold = self.config.doom_loop_threshold;
7853 child.edit_file_require_read_before_edit = self.config.edit_file_require_read_before_edit;
7854 // Monotonic tool posture: start from the PARENT's own overrides
7855 // (so anything the parent already disabled stays disabled), then
7856 // narrow further if a named definition restricts the tool set.
7857 child.tool_overrides = self.config.tool_overrides.clone();
7858 child.auto_approved_tools = self.config.auto_approved_tools.clone();
7859 child.tool_deny_patterns = self.config.tool_deny_patterns.clone();
7860 child.tool_allow_patterns = self.config.tool_allow_patterns.clone();
7861 child.permissions_enabled = self.config.permissions_enabled;
7862 child.permissions_ask_patterns = self.config.permissions_ask_patterns.clone();
7863 child.permissions_protected_paths = self.config.permissions_protected_paths.clone();
7864 child.network_policy = self.config.network_policy.clone();
7865 // P5-10 (§2 module 12): same monotonic-posture treatment as
7866 // `sandbox`/`approval` above — a subagent must inherit its
7867 // parent's OS-sandbox posture verbatim, never a looser
7868 // `Config::default()` fallback (`sandbox_os_enabled: None`,
7869 // `escalation: Deny`, `env_policy: Inherit` would otherwise be
7870 // right back to "confine only when the tier itself says so" for a
7871 // child whose parent explicitly forced the backstop on/off).
7872 child.sandbox_os_enabled = self.config.sandbox_os_enabled;
7873 child.sandbox_escalation = self.config.sandbox_escalation;
7874 child.sandbox_env_policy = self.config.sandbox_env_policy;
7875 if let Some(def) = definition {
7876 if let Some(allowed) = &def.tools {
7877 for name in &self.config.core_tools_enabled {
7878 if !allowed.iter().any(|t| t == name) {
7879 child
7880 .tool_overrides
7881 .entry(name.clone())
7882 .or_default()
7883 .enabled = Some(false);
7884 }
7885 }
7886 }
7887 // BP-7 (catalog §4a "Named agent definitions as data": the
7888 // `permissions` component of `prompt+model+tools+permissions`).
7889 // Every arm below can only TIGHTEN — the two policy values go
7890 // through the SAME strictness ranks `configfile::
7891 // clamp_project_permissions` uses for the untrusted project
7892 // layer (a looser value is ignored, never honored), the
7893 // auto-approve list is INTERSECTED with the parent's, and the
7894 // deny list is a union. A definition may come from a
7895 // `.claude/agents/*.md` file in the repo, so it sits at the
7896 // project trust tier and must never be an escalation door.
7897 if let Some(perms) = &def.permissions {
7898 if let Some(approval) = perms.approval {
7899 if crate::configfile::approval_rank(approval)
7900 < crate::configfile::approval_rank(child.approval)
7901 {
7902 child.approval = approval;
7903 }
7904 }
7905 if let Some(sandbox) = perms.sandbox {
7906 if crate::configfile::sandbox_rank(sandbox)
7907 < crate::configfile::sandbox_rank(child.sandbox)
7908 {
7909 child.sandbox = sandbox;
7910 }
7911 }
7912 if let Some(allowed) = &perms.auto_approved_tools {
7913 child
7914 .auto_approved_tools
7915 .retain(|tool| allowed.iter().any(|a| a == tool));
7916 }
7917 for pattern in &perms.deny {
7918 if !child.tool_deny_patterns.iter().any(|p| p == pattern) {
7919 child.tool_deny_patterns.push(pattern.clone());
7920 }
7921 }
7922 }
7923 }
7924 child.core_tools_enabled = self.config.core_tools_enabled.clone();
7925 child.module_registry = self.config.module_registry;
7926 child.module_activation = self.config.module_activation.clone();
7927 // The subagents module itself never widens either: a child spawned
7928 // at depth d+1 inherits the SAME caps (never a looser depth/
7929 // concurrency/background posture than its own parent).
7930 child.subagents_enabled = self.config.subagents_enabled;
7931 child.subagents_max_depth = self.config.subagents_max_depth;
7932 child.subagents_max_concurrent = self.config.subagents_max_concurrent;
7933 child.subagents_background = self.config.subagents_background;
7934 child.subagents_background_prompts = self.config.subagents_background_prompts;
7935 child.subagents_claude_agent_alias = self.config.subagents_claude_agent_alias;
7936 child.subagents_definitions = self.config.subagents_definitions.clone();
7937 child.subagent_depth = self.subagent_depth + 1;
7938 child
7939 }
7940
7941 /// Execute the `spawn_subagent` intrinsic (P5-3, §2 module 9). See
7942 /// `Self::build_child_config` for the monotonic-posture guarantee and
7943 /// `crate::subagents` for the depth/concurrency resource bounds and the
7944 /// §2.2 C6 background-policy enforcement.
7945 async fn run_spawn_subagent(
7946 &mut self,
7947 call: &supercode_interchange::ToolCall,
7948 ) -> (String, bool) {
7949 // BP-11: `subagent_start`/`subagent_stop` bracket a call that passed
7950 // the same validation the runner applies (subagents on, non-empty
7951 // task); a refused call fires neither.
7952 let task = if self.config.subagents_enabled {
7953 call.function
7954 .parsed_arguments()
7955 .ok()
7956 .and_then(|v| {
7957 v.get("task")
7958 .and_then(serde_json::Value::as_str)
7959 .map(str::to_string)
7960 })
7961 .filter(|t| !t.is_empty())
7962 } else {
7963 None
7964 };
7965 if let Some(task) = &task {
7966 self.fire_lifecycle(&crate::config::LifecycleEvent::SubagentStart {
7967 task: task.clone(),
7968 });
7969 }
7970 let (output, is_error) = self.run_spawn_subagent_inner(call).await;
7971 if let Some(task) = task {
7972 self.fire_lifecycle(&crate::config::LifecycleEvent::SubagentStop {
7973 task,
7974 is_error,
7975 output_len: output.len(),
7976 });
7977 }
7978 (output, is_error)
7979 }
7980
7981 /// Hands a lifecycle moment to the installed observer, if any (BP-11).
7982 fn fire_lifecycle(&self, event: &crate::config::LifecycleEvent) {
7983 if let Some(hook) = self.config.lifecycle_hook.as_ref() {
7984 hook(event);
7985 }
7986 }
7987
7988 /// Installs the lifecycle observer (compaction and subagent boundaries).
7989 pub fn set_lifecycle_hook(&mut self, hook: crate::config::LifecycleHook) {
7990 self.config.lifecycle_hook = Some(hook);
7991 }
7992
7993 async fn run_spawn_subagent_inner(
7994 &mut self,
7995 call: &supercode_interchange::ToolCall,
7996 ) -> (String, bool) {
7997 if !self.config.subagents_enabled {
7998 let err = Error::UnknownTool(SPAWN_SUBAGENT.to_string());
7999 return (format!("Error: {err}"), true);
8000 }
8001 let args = match call.function.parsed_arguments() {
8002 Ok(v) => v,
8003 Err(e) => {
8004 let err = Error::InvalidArguments {
8005 tool: SPAWN_SUBAGENT.to_string(),
8006 message: e.to_string(),
8007 };
8008 return (format!("Error: {err}"), true);
8009 }
8010 };
8011 let task = args
8012 .get("task")
8013 .and_then(serde_json::Value::as_str)
8014 .unwrap_or("")
8015 .to_string();
8016 if task.is_empty() {
8017 let err = Error::InvalidArguments {
8018 tool: SPAWN_SUBAGENT.to_string(),
8019 message: "`task` is required and must be non-empty".to_string(),
8020 };
8021 return (format!("Error: {err}"), true);
8022 }
8023 let agent_type = args
8024 .get("agent_type")
8025 .and_then(serde_json::Value::as_str)
8026 .map(String::from);
8027 let inline_system_prompt = args
8028 .get("system_prompt")
8029 .and_then(serde_json::Value::as_str)
8030 .map(String::from);
8031 let background = args
8032 .get("background")
8033 .and_then(serde_json::Value::as_bool)
8034 .unwrap_or(false);
8035 let requested_model = args
8036 .get("model")
8037 .and_then(serde_json::Value::as_str)
8038 .map(|model| crate::model_catalog::resolve_alias(model));
8039
8040 let definition = match &agent_type {
8041 Some(name) => match self.config.subagents_definitions.get(name) {
8042 Some(d) => Some(d.clone()),
8043 None => {
8044 let err = Error::SubagentDefinitionNotFound(name.clone());
8045 return (format!("Error: {err}"), true);
8046 }
8047 },
8048 None => None,
8049 };
8050
8051 if background {
8052 if !self.config.subagents_background {
8053 let err = Error::tool(
8054 SPAWN_SUBAGENT,
8055 "background=true requires capabilities.subagents.background = true",
8056 );
8057 return (format!("Error: {err}"), true);
8058 }
8059 // §2.2 C6, defensive re-check (belt-and-suspenders — see
8060 // `Error::SubagentBackgroundPolicyMissing`'s doc comment for why
8061 // this can't just trust the resolver already checked it).
8062 if self.config.subagents_background_prompts.is_none() {
8063 let err = Error::SubagentBackgroundPolicyMissing;
8064 return (format!("Error: {err}"), true);
8065 }
8066 }
8067
8068 // Resource bounds (fail-closed): depth first (cheap, no side
8069 // effect on failure), THEN concurrency (holds a slot — must be the
8070 // LAST check before actually spawning, so a refused spawn never
8071 // leaves a stray slot held).
8072 if let Err(e) =
8073 crate::subagents::check_depth(self.subagent_depth, self.config.subagents_max_depth)
8074 {
8075 return (format!("Error: {e}"), true);
8076 }
8077 let Some(guard) = crate::subagents::try_acquire(
8078 &self.subagent_concurrency_gauge,
8079 self.config.subagents_max_concurrent,
8080 ) else {
8081 let err = Error::SubagentConcurrencyExceeded {
8082 max_concurrent: self.config.subagents_max_concurrent,
8083 };
8084 return (format!("Error: {err}"), true);
8085 };
8086
8087 let child_id = next_subagent_id();
8088 let child_config = self.build_child_config(
8089 definition.as_ref(),
8090 inline_system_prompt,
8091 requested_model.or_else(|| definition.as_ref().and_then(|d| d.model.clone())),
8092 );
8093 let child_model = child_config.model.clone();
8094 let mut child = Agent::with_provider_arc(child_config, self.provider.clone());
8095 child.subagent_depth = self.subagent_depth + 1;
8096 child.subagent_concurrency_gauge = self.subagent_concurrency_gauge.clone();
8097
8098 // §2.2 C6: a background child NEVER gets a BLOCKING-BY-DEFAULT
8099 // interactive approval handler — either no handler at all
8100 // (`AutoPolicy`: the engine's pre-existing "no handler ⇒ deny"
8101 // fail-closed default), or (`Parent`) the never-blocking
8102 // `ParentQueueApprovalHandler`, UNLESS a `tui` embedder has
8103 // installed [`Self::child_approval_handler_factory`] (P5-4), in
8104 // which case THAT builds the handler instead — see
8105 // [`Self::set_child_approval_handler_factory`]'s doc comment for
8106 // why this can't escalate past what the rule engine already routed
8107 // to `Ask`. A foreground child also gets no handler here (today's
8108 // existing default posture; an embedder that wants an interactive
8109 // child installs its own via `set_permissions_approval_handler`
8110 // after this call returns, out of this method's scope).
8111 if background {
8112 if let Some(crate::subagents::BackgroundPromptsPolicy::Parent) =
8113 self.config.subagents_background_prompts
8114 {
8115 let handler: std::sync::Arc<dyn crate::permissions::PermissionsApprovalHandler> =
8116 match &self.child_approval_handler_factory {
8117 Some(factory) => {
8118 factory(child_id.clone(), self.pending_child_approvals.clone())
8119 }
8120 None => std::sync::Arc::new(crate::subagents::ParentQueueApprovalHandler {
8121 child_agent_id: child_id.clone(),
8122 queue: self.pending_child_approvals.clone(),
8123 }),
8124 };
8125 child.ctx.sandbox_approval_handler =
8126 Some(crate::sandbox::SandboxApprovalHandler(handler.clone()));
8127 child.permissions_approval_handler = Some(handler);
8128 }
8129 }
8130
8131 let lineage = crate::subagents::SubagentLineage {
8132 child_agent_id: child_id.clone(),
8133 parent_session_id: self.subagent_store.as_ref().map(|(_, name)| name.clone()),
8134 parent_tool_use_id: call.id.clone(),
8135 depth: self.subagent_depth + 1,
8136 agent_type: agent_type.clone(),
8137 task: task.clone(),
8138 background,
8139 spawned_at_ms: now_ms(),
8140 model: child_model,
8141 };
8142 if let Some((store, parent_name)) = &self.subagent_store {
8143 let _ = store.save_subagent_lineage(parent_name, &child_id, &lineage);
8144 }
8145
8146 if background {
8147 let spawned_task_text = task.clone();
8148 // BP-7: captured BEFORE `child` moves into the task — this is
8149 // the handle `send_message` writes into.
8150 let mailbox = child.steer_queue_handle();
8151 self.background_subagents.insert(
8152 child_id.clone(),
8153 BackgroundSubagent {
8154 handle: tokio::spawn(async move {
8155 // The concurrency slot lives for exactly as long as
8156 // this future runs — moved in here, dropped when the
8157 // child's `send` (and this future) finishes.
8158 let _guard = guard;
8159 let result = child.send(spawned_task_text).await;
8160 let transcript = child.history()[1..].to_vec();
8161 (child_id, result, transcript)
8162 }),
8163 task,
8164 agent_type,
8165 started_at_ms: lineage.spawned_at_ms,
8166 mailbox,
8167 },
8168 );
8169 let out = serde_json::json!({
8170 "subagent_id": lineage.child_agent_id,
8171 "status": "spawned",
8172 "background": true,
8173 });
8174 return (out.to_string(), false);
8175 }
8176
8177 // Foreground: run to completion now, guard held until this
8178 // function returns (then drops, freeing the slot).
8179 let result = child.send(task).await;
8180 let transcript = child.history()[1..].to_vec();
8181 self.persist_subagent_transcript(&child_id, &lineage, &transcript);
8182 // BP-7: kept in-process so `subagent_resume` can restore this
8183 // child's context even with no session store attached.
8184 self.reaped_subagents
8185 .insert(child_id.clone(), transcript.clone());
8186 drop(guard);
8187 match result {
8188 Ok(text) => (text, false),
8189 Err(e) => (format!("Error: subagent `{child_id}` failed: {e}"), true),
8190 }
8191 }
8192
8193 /// Execute the `subagent_status` intrinsic (P5-3, D3
8194 /// "background+resume"): poll a background child; once its `JoinHandle`
8195 /// is finished, reap it (removing it from `Self::background_subagents`
8196 /// and persisting its transcript, same as the foreground path).
8197 async fn run_subagent_status(
8198 &mut self,
8199 call: &supercode_interchange::ToolCall,
8200 ) -> (String, bool) {
8201 let args = match call.function.parsed_arguments() {
8202 Ok(v) => v,
8203 Err(e) => {
8204 let err = Error::InvalidArguments {
8205 tool: SUBAGENT_STATUS.to_string(),
8206 message: e.to_string(),
8207 };
8208 return (format!("Error: {err}"), true);
8209 }
8210 };
8211 let Some(id) = args.get("subagent_id").and_then(serde_json::Value::as_str) else {
8212 let err = Error::InvalidArguments {
8213 tool: SUBAGENT_STATUS.to_string(),
8214 message: "`subagent_id` is required".to_string(),
8215 };
8216 return (format!("Error: {err}"), true);
8217 };
8218 let Some(entry) = self.background_subagents.get(id) else {
8219 let err = Error::SubagentNotFound(id.to_string());
8220 return (format!("Error: {err}"), true);
8221 };
8222 if !entry.handle.is_finished() {
8223 let out = serde_json::json!({
8224 "subagent_id": id,
8225 "status": "pending",
8226 "task": entry.task,
8227 "agent_type": entry.agent_type,
8228 "started_at_ms": entry.started_at_ms,
8229 });
8230 return (out.to_string(), false);
8231 }
8232 // Finished — reap it. `.await` on an already-finished handle
8233 // resolves immediately (never actually blocks).
8234 let entry = self
8235 .background_subagents
8236 .remove(id)
8237 .expect("checked Some above");
8238 let (child_id, result, transcript) = match entry.handle.await {
8239 Ok(v) => v,
8240 Err(join_err) => {
8241 let err = Error::tool(
8242 SUBAGENT_STATUS,
8243 format!("subagent `{id}` task panicked: {join_err}"),
8244 );
8245 return (format!("Error: {err}"), true);
8246 }
8247 };
8248 // Re-derive the lineage record for persistence (cheap; the fields
8249 // are all still in hand) — mirrors the foreground path's single
8250 // `persist_subagent_transcript` call site.
8251 if let Some((store, parent_name)) = self.subagent_store.clone() {
8252 if let Ok(Some(lineage)) = store.load_subagent_lineage(&parent_name, &child_id) {
8253 self.persist_subagent_transcript(&child_id, &lineage, &transcript);
8254 }
8255 }
8256 // BP-7: see the foreground path's identical line.
8257 self.reaped_subagents
8258 .insert(child_id.clone(), transcript.clone());
8259 match result {
8260 Ok(text) => {
8261 let out = serde_json::json!({
8262 "subagent_id": child_id,
8263 "status": "done",
8264 "result": text,
8265 });
8266 (out.to_string(), false)
8267 }
8268 Err(e) => {
8269 let out = serde_json::json!({
8270 "subagent_id": child_id,
8271 "status": "error",
8272 "message": e.to_string(),
8273 });
8274 (out.to_string(), true)
8275 }
8276 }
8277 }
8278
8279 /// Execute the `background_exec` intrinsic (P5-6, §2 module 4, D1
8280 /// "background exec"): spawn `args.command` as a detached OS process
8281 /// via `crate::tools::build_sandboxed_sh` — the SAME sandboxed-spawn
8282 /// path [`crate::tools::BashTool::execute`] uses — and return its job
8283 /// id IMMEDIATELY, never the command's output. Gated by the same
8284 /// permission check a foreground `bash` call gets
8285 /// ([`Self::background_permission_denial`]), then a fail-closed
8286 /// concurrency cap ([`Config::tools_background_max_concurrent`]), THEN
8287 /// the actual spawn — in that order, so a refused call never holds a
8288 /// concurrency slot and never touches the process table.
8289 fn run_background_exec(&mut self, call: &supercode_interchange::ToolCall) -> (String, bool) {
8290 let args = match call.function.parsed_arguments() {
8291 Ok(v) => v,
8292 Err(e) => {
8293 let err = Error::InvalidArguments {
8294 tool: BACKGROUND_EXEC.to_string(),
8295 message: e.to_string(),
8296 };
8297 return (format!("Error: {err}"), true);
8298 }
8299 };
8300 let command = args
8301 .get("command")
8302 .and_then(serde_json::Value::as_str)
8303 .unwrap_or("")
8304 .to_string();
8305 if command.is_empty() {
8306 let err = Error::InvalidArguments {
8307 tool: BACKGROUND_EXEC.to_string(),
8308 message: "`command` is required and must be non-empty".to_string(),
8309 };
8310 return (format!("Error: {err}"), true);
8311 }
8312
8313 // A job id up front (before spawning) — used both as the audit
8314 // handle for a §2.2 C6 `Parent`-policy queued denial (this call may
8315 // never actually reach the spawn below) and, if the call proceeds,
8316 // as `Self::background_jobs`'s real key.
8317 let job_id = supercode_runtime::background::next_job_id(now_ms());
8318
8319 // Fable-5 review (LOW, "pre_tool_hook + doom-loop don't cover
8320 // background_exec"): this intrinsic is intercepted in
8321 // `Self::prepare_tool_call` and returns before `Self::finish_prepare`
8322 // ever runs, so — unlike a foreground `bash` call — it was reaching
8323 // this real spawn below WITHOUT ever offering `Config.pre_tool_hook`
8324 // a chance to veto it. `background_exec` runs a REAL command (unlike
8325 // the purely in-process meta-intrinsics `tool_search`/
8326 // `expand_reduction`/`sidecar_search`, which have no such gap to
8327 // close), so it belongs behind the same security-relevant veto a
8328 // foreground call gets. Scoped to this one call site — the other
8329 // meta-intrinsics are unchanged. The doom-loop counter
8330 // (`Self::check_doom_loop`) is deliberately NOT wired here: it is a
8331 // foreground repetition breaker keyed on `(self.doom_loop_last_call,
8332 // self.doom_loop_streak)`, a single piece of state shared with the
8333 // ordinary tool-call loop — folding background jobs into that same
8334 // streak would make an interleaved foreground/background pattern
8335 // trip (or fail to trip) the breaker in ways that have nothing to
8336 // do with the foreground loop actually repeating itself; the
8337 // pre_tool_hook veto below is the security-relevant half of this
8338 // fix, the doom-loop breaker is not.
8339 // BP-10: the hook now runs BEFORE this path's permissions gate, the
8340 // same order the foreground path uses — so a rewrite is what the
8341 // rules evaluate and what actually runs, and the hook's
8342 // `Allow`/`Ask` are tiers inside the engine rather than a second
8343 // verdict beside it.
8344 let mut command = command;
8345 let mut hook_decision = crate::config::HookDecision::Pass;
8346 if let Some(hook) = &self.config.pre_tool_hook {
8347 let outcome = hook(BACKGROUND_EXEC, &args);
8348 if outcome.decision == crate::config::HookDecision::Deny {
8349 let reason = outcome.reason.unwrap_or_else(|| "denied".to_string());
8350 return (format!("Error: blocked by pre-tool hook: {reason}"), true);
8351 }
8352 if let Some(rewritten) = outcome.updated_args {
8353 command = rewritten
8354 .get("command")
8355 .and_then(|v| v.as_str())
8356 .unwrap_or(&command)
8357 .to_string();
8358 }
8359 hook_decision = outcome.decision;
8360 }
8361
8362 if let Some(reason) = self.background_permission_denial(&command, &job_id, hook_decision) {
8363 return (format!("Error: {reason}"), true);
8364 }
8365
8366 let Some(guard) = crate::subagents::try_acquire(
8367 &self.background_concurrency_gauge,
8368 self.config.tools_background_max_concurrent,
8369 ) else {
8370 let err = Error::BackgroundJobConcurrencyExceeded {
8371 max_concurrent: self.config.tools_background_max_concurrent,
8372 };
8373 return (format!("Error: {err}"), true);
8374 };
8375
8376 let mut cmd = match crate::tools::build_sandboxed_sh(&command, &self.ctx) {
8377 Ok(cmd) => cmd,
8378 Err(e) => return (format!("Error: {e}"), true),
8379 };
8380 cmd.current_dir(&self.ctx.cwd)
8381 .stdin(std::process::Stdio::null())
8382 .stdout(std::process::Stdio::piped())
8383 .stderr(std::process::Stdio::piped())
8384 // Defense-in-depth for the "must be killed on drop" guarantee —
8385 // see `impl Drop for Agent`'s doc comment; the EXPLICIT
8386 // `start_kill()` loop there is what makes the guarantee
8387 // provable, this is a second, independent line of defense for
8388 // the same outcome.
8389 .kill_on_drop(true);
8390 // Fable-5 review (HIGH, "grandchildren orphaned on kill AND
8391 // agent-drop"): `Child::start_kill` only signals the DIRECT child.
8392 // A background command that spawns a surviving subprocess (a `&`
8393 // job, a pipeline, a double-forking daemon — or, on macOS, the
8394 // `sandbox-exec` wrapper itself in `build_sandboxed_sh`, whose real
8395 // `sh` and ITS children are all grandchildren of the tracked pid)
8396 // leaves those processes running, reparented to init, after the
8397 // tracked job is "killed". Putting this job in its OWN new process
8398 // group (`pgid == its own pid`, since every descendant inherits the
8399 // group unless it explicitly opts out) lets `kill_job_process_group`
8400 // below signal the WHOLE tree at kill/drop time, not just the one
8401 // pid we happen to be tracking. No portable equivalent on Windows —
8402 // see `kill_job_process_group`'s `#[cfg(not(unix))]` fallback.
8403 #[cfg(unix)]
8404 cmd.process_group(0);
8405 // P4c (`core.shell_env_snapshot`)/P5-10 (`env_policy`):
8406 // `build_sandboxed_sh` (above) already applied both via its own
8407 // `apply_sandbox_env_policy` last step — no separate `ctx.shell_env`
8408 // application here (that would re-add a secret `Filtered`/`None`
8409 // just stripped, on top of the already-`env_clear`'d command).
8410
8411 let mut child = match cmd.spawn() {
8412 Ok(c) => c,
8413 Err(e) => {
8414 drop(guard);
8415 let err = Error::tool(
8416 BACKGROUND_EXEC,
8417 format!("failed to spawn background command: {e}"),
8418 );
8419 return (format!("Error: {err}"), true);
8420 }
8421 };
8422 let pid = child.id();
8423 let output = std::sync::Arc::new(supercode_runtime::background::CapturedOutput::new());
8424 let cap = self.config.tools_background_max_output_bytes;
8425 // Fire-and-forget: the reader tasks outlive this method call and
8426 // exit on their own at pipe EOF — see `spawn_output_reader`'s doc
8427 // comment. Bound to named (not `_`) locals only to keep clippy's
8428 // `let_underscore_future` lint quiet; neither handle is awaited or
8429 // aborted anywhere.
8430 if let Some(stdout) = child.stdout.take() {
8431 let _stdout_reader = spawn_output_reader(stdout, output.clone(), cap);
8432 }
8433 if let Some(stderr) = child.stderr.take() {
8434 let _stderr_reader = spawn_output_reader(stderr, output.clone(), cap);
8435 }
8436
8437 let started_at_ms = now_ms();
8438 self.background_jobs.insert(
8439 job_id.clone(),
8440 BackgroundJob {
8441 child,
8442 command: command.clone(),
8443 pid,
8444 output,
8445 started_at_ms,
8446 killed: false,
8447 _guard: guard,
8448 },
8449 );
8450
8451 let out = serde_json::json!({
8452 "job_id": job_id,
8453 "status": "running",
8454 "pid": pid,
8455 "command": command,
8456 });
8457 (out.to_string(), false)
8458 }
8459
8460 /// Execute the `background_status` intrinsic (P5-6, D1 "monitor/event
8461 /// feed"): non-blocking poll of one job's run status (via
8462 /// `Child::try_wait`), drain its output captured since the LAST poll
8463 /// and emit it as an [`AgentEvent::BackgroundOutput`] event (the
8464 /// "event feed" — a real `EventSink` consumer sees each poll's new
8465 /// output live), and return the full captured output (bounded, per
8466 /// [`Config::tools_background_max_output_bytes`]) so far either way.
8467 /// Once the job is terminal (exited or killed), this reaps it — removes
8468 /// it from [`Self::background_jobs`], freeing its concurrency slot —
8469 /// same "poll once more to reap" contract [`Self::run_subagent_status`]
8470 /// already established for background subagents.
8471 fn run_background_status(&mut self, call: &supercode_interchange::ToolCall) -> (String, bool) {
8472 let args = match call.function.parsed_arguments() {
8473 Ok(v) => v,
8474 Err(e) => {
8475 let err = Error::InvalidArguments {
8476 tool: BACKGROUND_STATUS.to_string(),
8477 message: e.to_string(),
8478 };
8479 return (format!("Error: {err}"), true);
8480 }
8481 };
8482 let Some(job_id) = args.get("job_id").and_then(serde_json::Value::as_str) else {
8483 let err = Error::InvalidArguments {
8484 tool: BACKGROUND_STATUS.to_string(),
8485 message: "`job_id` is required".to_string(),
8486 };
8487 return (format!("Error: {err}"), true);
8488 };
8489 let job_id = job_id.to_string();
8490
8491 // Scoped so the mutable borrow of `self.background_jobs` ends
8492 // before `self.emit(...)`/`self.background_jobs.remove(...)` below
8493 // need their own (mutable) access to `self`.
8494 let (command, pid, started_at_ms, status, output_so_far, truncated, delta) = {
8495 let Some(job) = self.background_jobs.get_mut(&job_id) else {
8496 let err = Error::BackgroundJobNotFound(job_id);
8497 return (format!("Error: {err}"), true);
8498 };
8499 let status = background_job_status(job);
8500 let (output_so_far, truncated) = job.output.snapshot();
8501 let delta = job.output.drain_new();
8502 (
8503 job.command.clone(),
8504 job.pid,
8505 job.started_at_ms,
8506 status,
8507 output_so_far,
8508 truncated,
8509 delta,
8510 )
8511 };
8512
8513 if !delta.is_empty() {
8514 self.emit(AgentEvent::BackgroundOutput {
8515 job_id: job_id.clone(),
8516 chunk: delta,
8517 truncated,
8518 });
8519 }
8520
8521 let exit_code = match status {
8522 supercode_runtime::background::JobStatus::Exited(code) => code,
8523 _ => None,
8524 };
8525 let out = serde_json::json!({
8526 "job_id": job_id,
8527 "command": command,
8528 "status": status.as_str(),
8529 "exit_code": exit_code,
8530 "pid": pid,
8531 "started_at_ms": started_at_ms,
8532 "output": output_so_far,
8533 "output_truncated": truncated,
8534 });
8535 if !matches!(status, supercode_runtime::background::JobStatus::Running) {
8536 self.background_jobs.remove(&job_id);
8537 }
8538 (out.to_string(), false)
8539 }
8540
8541 /// Execute the `background_list` intrinsic (P5-6, D10 "bg-manager"):
8542 /// list every background job this agent is currently tracking, without
8543 /// draining output or reaping anything (a read-only listing —
8544 /// `background_status` is the reaping poll).
8545 fn run_background_list(&mut self, _call: &supercode_interchange::ToolCall) -> (String, bool) {
8546 let mut jobs = Vec::new();
8547 for (job_id, job) in self.background_jobs.iter_mut() {
8548 let status = background_job_status(job);
8549 jobs.push(serde_json::json!({
8550 "job_id": job_id,
8551 "command": job.command,
8552 "status": status.as_str(),
8553 "pid": job.pid,
8554 "started_at_ms": job.started_at_ms,
8555 }));
8556 }
8557 let out = serde_json::json!({ "jobs": jobs });
8558 (out.to_string(), false)
8559 }
8560
8561 /// Execute the `background_kill` intrinsic (P5-6, D10 "bg-manager",
8562 /// build brief "kill/cancel a job"): request REAL termination of a
8563 /// background job's OS process AND its whole process group (see
8564 /// [`kill_job_process_group`] — Fable-5 review, HIGH, "grandchildren
8565 /// orphaned on kill"; a documented no-op if the process already
8566 /// exited) and reap it immediately, freeing its concurrency slot.
8567 fn run_background_kill(&mut self, call: &supercode_interchange::ToolCall) -> (String, bool) {
8568 let args = match call.function.parsed_arguments() {
8569 Ok(v) => v,
8570 Err(e) => {
8571 let err = Error::InvalidArguments {
8572 tool: BACKGROUND_KILL.to_string(),
8573 message: e.to_string(),
8574 };
8575 return (format!("Error: {err}"), true);
8576 }
8577 };
8578 let Some(job_id) = args.get("job_id").and_then(serde_json::Value::as_str) else {
8579 let err = Error::InvalidArguments {
8580 tool: BACKGROUND_KILL.to_string(),
8581 message: "`job_id` is required".to_string(),
8582 };
8583 return (format!("Error: {err}"), true);
8584 };
8585 let job_id = job_id.to_string();
8586 let Some(mut job) = self.background_jobs.remove(&job_id) else {
8587 let err = Error::BackgroundJobNotFound(job_id);
8588 return (format!("Error: {err}"), true);
8589 };
8590 kill_job_process_group(&mut job);
8591 job.killed = true;
8592 let out = serde_json::json!({
8593 "job_id": job_id,
8594 "status": "killed",
8595 "pid": job.pid,
8596 });
8597 // `job` (and its `ConcurrencyGuard`) drops here, freeing the slot.
8598 (out.to_string(), false)
8599 }
8600
8601 /// P5-3 (D5 "subagent transcripts… persisted + linked"): write a
8602 /// finished child's transcript to `Self::subagent_store`, if one is
8603 /// installed — a no-op otherwise (see that field's doc comment). Builds
8604 /// the child's `Session` the same way `to_native_jsonl_v2`'s doc
8605 /// comment describes (an empty imported prefix + `transcript` as
8606 /// `appended` `NativeTurn`s), with `meta.agent_id`/`parent_tool_use_id`/
8607 /// `lineage` populated from `lineage` so the native-v2 header carries
8608 /// the full lineage record on disk (see `Session::to_native_jsonl_v2`'s
8609 /// P5-3 doc note).
8610 ///
8611 /// P5-3 safety-hardening fix (Fable-5 review, LOW "translation-fidelity
8612 /// cosmetic"): `Session::from_claude_code_str("")` is used ONLY to get
8613 /// a blank `raw`/`messages` skeleton cheaply (an empty string parses
8614 /// identically under any loader) — it is NOT claiming this child's
8615 /// session actually came from Claude Code. Before this fix, that
8616 /// borrowed constructor's `meta.source` (`SessionSource::ClaudeCode`)
8617 /// leaked straight through to the persisted sidecar's `source` header,
8618 /// mislabeling a native `spawn_subagent` child as an imported CC
8619 /// session. Corrected to `SessionSource::Native` immediately after —
8620 /// see that variant's doc comment.
8621 fn persist_subagent_transcript(
8622 &self,
8623 child_id: &str,
8624 lineage: &crate::subagents::SubagentLineage,
8625 transcript: &[ChatMessage],
8626 ) {
8627 let Some((store, parent_name)) = &self.subagent_store else {
8628 return;
8629 };
8630 let mut session = match Session::from_claude_code_str("") {
8631 Ok(s) => s,
8632 Err(_) => return,
8633 };
8634 session.meta.source = supercode_interchange::session::SessionSource::Native;
8635 session.meta.agent_id = Some(lineage.child_agent_id.clone());
8636 session.meta.parent_tool_use_id = Some(lineage.parent_tool_use_id.clone());
8637 session.meta.lineage = lineage.to_lineage_map();
8638 let sidecar_jsonl = session.to_native_jsonl_v2(transcript);
8639 let _ = store.save_subagent_transcript(parent_name, child_id, &sidecar_jsonl);
8640 let _ = store.save_subagent_lineage(parent_name, child_id, lineage);
8641 }
8642
8643 /// Execute the `tool_search` intrinsic (B6): case-insensitive keyword
8644 /// match over `name` + `description` of every registered, enabled,
8645 /// non-core, not-yet-activated tool (builtin and `mcp__*` alike). Matches
8646 /// are activated (advertised starting with the next request) and
8647 /// returned as a JSON array of their full [`ToolSchema`]s.
8648 fn run_tool_search(&mut self, call: &supercode_interchange::ToolCall) -> (String, bool) {
8649 let args = match call.function.parsed_arguments() {
8650 Ok(v) => v,
8651 Err(e) => {
8652 let err = Error::InvalidArguments {
8653 tool: TOOL_SEARCH.to_string(),
8654 message: e.to_string(),
8655 };
8656 return (format!("Error: {err}"), true);
8657 }
8658 };
8659 let query = args
8660 .get("query")
8661 .and_then(serde_json::Value::as_str)
8662 .unwrap_or("")
8663 .to_lowercase();
8664 let max_results = args
8665 .get("max_results")
8666 .and_then(serde_json::Value::as_u64)
8667 .map(|n| n as usize);
8668
8669 let mut matches: Vec<ToolSchema> = self
8670 .registry
8671 .iter()
8672 .filter(|t| self.config.tool_enabled(t.name()))
8673 .filter(|t| !self.is_core_tool(t.name()))
8674 .filter(|t| !self.activated_tools.contains(t.name()))
8675 .filter(|t| {
8676 query.is_empty()
8677 || t.name().to_lowercase().contains(&query)
8678 || self
8679 .config
8680 .tool_description(t.name(), t.description())
8681 .to_lowercase()
8682 .contains(&query)
8683 })
8684 // TR-8/T5 dev/03: the on-demand fetch always returns the ORIGINAL
8685 // full schema, never the tier-minified one — that's the invert.
8686 .map(|t| self.raw_schema_for(t))
8687 .collect();
8688
8689 if let Some(max) = max_results {
8690 matches.truncate(max);
8691 }
8692
8693 for m in &matches {
8694 self.activated_tools.insert(m.name.clone());
8695 }
8696
8697 let result = serde_json::to_string(&matches).unwrap_or_else(|_| "[]".to_string());
8698 (result, false)
8699 }
8700
8701 /// The `expand_reduction` schema (T12/TR-1), advertised whenever a
8702 /// [`ReductionPolicy`] is installed.
8703 ///
8704 /// The description deliberately never spells the literal stub sentinel
8705 /// prefix: A11's export leak guard is unconditional, so an assistant
8706 /// turn that quoted a stub line verbatim (which teaching the syntax
8707 /// invites) would permanently fail export for that session. Stubs are
8708 /// described abstractly and the model is told to pass ids only.
8709 fn expand_reduction_schema() -> ToolSchema {
8710 ToolSchema {
8711 name: EXPAND_REDUCTION.to_string(),
8712 description: "Fetch back the original content hidden behind a reduction stub in \
8713 your current view — a truncated tool output, cleared old turns, or an elided \
8714 file read that was hidden to save context. Each stub line names a reduction id \
8715 like r0042-9f3c: pass ONLY that id here, and never quote or repeat a stub line \
8716 itself in your replies. The original is durably kept in the session sidecar. \
8717 Pass `byte_range` to fetch a slice of a large one at a time instead of all of \
8718 it at once; ranged results are prefixed with a `bytes start..end of total` \
8719 header so you can plan the next slice."
8720 .to_string(),
8721 parameters: serde_json::json!({
8722 "type": "object",
8723 "properties": {
8724 "reduction_id": {
8725 "type": "string",
8726 "description": "The reduction id named in the stub line, e.g. \
8727 \"r0042-9f3c\". Pass the id alone."
8728 },
8729 "byte_range": {
8730 "type": "array",
8731 "items": {"type": "integer"},
8732 "minItems": 2,
8733 "maxItems": 2,
8734 "description": "Optional [start, end) byte offsets within the original \
8735 content to fetch instead of all of it. Exactly two non-negative \
8736 integers with start <= end."
8737 }
8738 },
8739 "required": ["reduction_id"],
8740 "additionalProperties": false
8741 }),
8742 }
8743 }
8744
8745 /// The `sidecar_search` schema (T12/TR-1), advertised whenever a
8746 /// [`ReductionPolicy`] is installed. Same no-literal-sentinel rule as
8747 /// [`Self::expand_reduction_schema`].
8748 fn sidecar_search_schema() -> ToolSchema {
8749 ToolSchema {
8750 name: SIDECAR_SEARCH.to_string(),
8751 description: "Search content currently hidden from your view by reduction stubs \
8752 (large tool outputs, cleared old turns, elided file reads) for a substring or \
8753 regex. Only hidden content is searched, never what you can already see. \
8754 Returns match snippets with each match's reduction_id for use with \
8755 expand_reduction; refer to results by their reduction id rather than quoting \
8756 stub lines. Results are capped — if `truncated` is true, narrow the query."
8757 .to_string(),
8758 parameters: serde_json::json!({
8759 "type": "object",
8760 "properties": {
8761 "query": {
8762 "type": "string",
8763 "description": "Non-empty substring or regex to search for \
8764 (case-insensitive)."
8765 }
8766 },
8767 "required": ["query"],
8768 "additionalProperties": false
8769 }),
8770 }
8771 }
8772
8773 /// Reload the recorder's full recorded messages from disk (TR-1's
8774 /// `recorded` resolution source). Since TR-12's D6/A7 supersession gate
8775 /// (`Self::run_loop`), a `expand_reduction`/`sidecar_search` call only
8776 /// ever exists alongside an active [`ReductionPolicy`] (see
8777 /// [`EXPAND_REDUCTION`]'s doc), and pairing one with a recorder — as the
8778 /// CLI's reduced mode always does — means the gate is already on and
8779 /// `history[1..]` holds the same full bytes as this reload: this upgrade
8780 /// is then a dormant no-op (`reduce::rehydrate::prefer_recorded` sees
8781 /// `recorded == minted` and keeps `minted`). It stops being a no-op —
8782 /// defense in depth, not the common path — for a **legacy** sidecar
8783 /// recorded before this gate existed, or for a policy-without-recorder
8784 /// agent (gate off, so `history[1..]` still carries
8785 /// [`Self::cap_tool_output`]-capped copies): only there can `history[1..]`
8786 /// diverge from the sidecar, and only there does consulting this reload
8787 /// actually recover bytes `history[1..]` alone couldn't. `Ok(None)` when
8788 /// no recorder is attached (rehydration then resolves from history alone,
8789 /// whose capped copies — if any — carry their own honest cap notice). A
8790 /// disk-level reload is fine here regardless: these intrinsic calls are
8791 /// rare, model-initiated events, not per-request work.
8792 fn recorded_messages(&self) -> std::result::Result<Option<Vec<ChatMessage>>, String> {
8793 let Some(recorder) = &self.recorder else {
8794 return Ok(None);
8795 };
8796 let raw = std::fs::read_to_string(recorder.path())
8797 .map_err(|e| format!("failed to read the session sidecar: {e}"))?;
8798 let session = Session::from_sidecar_str(&raw)
8799 .map_err(|e| format!("failed to parse the session sidecar: {e}"))?;
8800 Ok(Some(session.messages))
8801 }
8802
8803 /// Execute the `expand_reduction` intrinsic (T12/TR-1): resolves against
8804 /// `self.reduction_log` + `self.history[1..]` (the hash-minting source),
8805 /// upgraded to the recorder's full recorded bytes for cap-diverged
8806 /// content ([`Self::recorded_messages`]; the two-source contract is
8807 /// documented on `reduce::rehydrate`). `byte_range` is validated
8808 /// strictly — any malformed shape is a model-recoverable error naming
8809 /// the expected form and the original's true size, never a silent
8810 /// whole-content (or empty) return.
8811 fn run_expand_reduction(&mut self, call: &supercode_interchange::ToolCall) -> (String, bool) {
8812 let args = match call.function.parsed_arguments() {
8813 Ok(v) => v,
8814 Err(e) => {
8815 let err = Error::InvalidArguments {
8816 tool: EXPAND_REDUCTION.to_string(),
8817 message: e.to_string(),
8818 };
8819 return (format!("Error: {err}"), true);
8820 }
8821 };
8822 let Some(id) = args.get("reduction_id").and_then(serde_json::Value::as_str) else {
8823 return (
8824 "Error: expand_reduction requires a `reduction_id` string argument".to_string(),
8825 true,
8826 );
8827 };
8828 let recorded = match self.recorded_messages() {
8829 Ok(r) => r,
8830 Err(e) => return (format!("Error: expand_reduction: {e}"), true),
8831 };
8832 let recorded = recorded.as_deref();
8833
8834 // B3: strict shape validation — exactly two non-negative integers.
8835 // Anything else errors (with the true total when resolvable) rather
8836 // than silently degrading to a whole-content expand.
8837 let byte_range = match args.get("byte_range") {
8838 None | Some(serde_json::Value::Null) => None,
8839 Some(v) => {
8840 let parsed = v
8841 .as_array()
8842 .filter(|a| a.len() == 2)
8843 .and_then(|a| Some((a[0].as_u64()? as usize, a[1].as_u64()? as usize)));
8844 match parsed {
8845 Some(range) => Some(range),
8846 None => {
8847 let total = reduce::rehydrate::reduction_total_bytes(
8848 &self.reduction_log,
8849 &self.history[1..],
8850 recorded,
8851 id,
8852 )
8853 .map(|n| format!("; the original is {n} bytes"))
8854 .unwrap_or_default();
8855 return (
8856 format!(
8857 "Error: expand_reduction: malformed byte_range {v} — expected \
8858 [start, end): exactly two non-negative integers with \
8859 start <= end{total}"
8860 ),
8861 true,
8862 );
8863 }
8864 }
8865 }
8866 };
8867 match reduce::rehydrate::expand_reduction(
8868 &self.reduction_log,
8869 &self.history[1..],
8870 recorded,
8871 id,
8872 byte_range,
8873 ) {
8874 // A ranged result carries a provenance header naming the slice
8875 // and the true total, so the model can plan its next slice; a
8876 // whole-content expand stays byte-exact (TR-1 dev/01).
8877 Ok(outcome) => match outcome.range {
8878 Some((start, end)) => (
8879 format!(
8880 "[{id}: bytes {start}..{end} of {total}]\n{content}",
8881 total = outcome.total_bytes,
8882 content = outcome.content
8883 ),
8884 false,
8885 ),
8886 None => (outcome.content, false),
8887 },
8888 Err(e) => (format!("Error: {e}"), true),
8889 }
8890 }
8891
8892 /// Execute the `sidecar_search` intrinsic (T12/TR-1); same two-source
8893 /// resolution as [`Self::run_expand_reduction`]. The result is bounded
8894 /// by construction (`reduce::rehydrate::SidecarSearchResult`'s caps), so
8895 /// a broad query can never re-inflate the context or bloat the sidecar
8896 /// the recorder appends this result to.
8897 fn run_sidecar_search(&mut self, call: &supercode_interchange::ToolCall) -> (String, bool) {
8898 let args = match call.function.parsed_arguments() {
8899 Ok(v) => v,
8900 Err(e) => {
8901 let err = Error::InvalidArguments {
8902 tool: SIDECAR_SEARCH.to_string(),
8903 message: e.to_string(),
8904 };
8905 return (format!("Error: {err}"), true);
8906 }
8907 };
8908 let query = args
8909 .get("query")
8910 .and_then(serde_json::Value::as_str)
8911 .unwrap_or("");
8912 if query.trim().is_empty() {
8913 return (
8914 "Error: sidecar_search requires a non-empty `query` string argument".to_string(),
8915 true,
8916 );
8917 }
8918 let recorded = match self.recorded_messages() {
8919 Ok(r) => r,
8920 Err(e) => return (format!("Error: sidecar_search: {e}"), true),
8921 };
8922 match reduce::rehydrate::sidecar_search(
8923 &self.reduction_log,
8924 &self.history[1..],
8925 recorded.as_deref(),
8926 query,
8927 ) {
8928 Ok(result) => (
8929 serde_json::to_string(&result).unwrap_or_else(|_| "{}".to_string()),
8930 false,
8931 ),
8932 Err(e) => (format!("Error: {e}"), true),
8933 }
8934 }
8935
8936 fn emit(&self, event: AgentEvent) {
8937 if let Some(sink) = &self.config.event_sink {
8938 sink(event);
8939 }
8940 }
8941
8942 /// Number of non-system messages exchanged so far.
8943 pub fn turn_count(&self) -> usize {
8944 self.history
8945 .iter()
8946 .filter(|m| m.role != Role::System)
8947 .count()
8948 }
8949
8950 /// Cumulative output (completion) tokens reported by the provider across
8951 /// every `send` on this agent. Zero if the provider reports no usage.
8952 pub fn total_output_tokens(&self) -> u64 {
8953 self.total_output_tokens
8954 }
8955}
8956
8957/// Deliver an agent's `send_message` to another session through the one
8958/// send every sender uses. The sender is this process's own session (a
8959/// runtime supercode hosts, found from its ancestry), so replies can come
8960/// back.
8961#[cfg(feature = "adapter-api")]
8962async fn send_to_session(to: &str, message: &str, notify_when_idle: bool) -> (String, bool) {
8963 let homes = crate::HarnessHomes::default();
8964 let caller =
8965 match crate::mail_route::resolve_caller(&homes, &crate::mail_route::process_ancestry()) {
8966 Ok(caller) => caller,
8967 Err(_) => {
8968 return (
8969 format!(
8970 "Not sent: `{to}` is not one of your running subagents, and this session has \
8971 no address other sessions can reply to (supercode does not host it), so it \
8972 can message only its own subagents."
8973 ),
8974 true,
8975 )
8976 }
8977 };
8978 let options = crate::mail_send::SendOptions {
8979 notify_when_idle,
8980 ..Default::default()
8981 };
8982 match crate::mail_send::send(&homes, &caller, to, message, options).await {
8983 Ok(outcome) => (outcome.text, outcome.code != 0),
8984 Err(error) => (format!("Not sent to {to}: {error}"), true),
8985 }
8986}
8987
8988#[cfg(not(feature = "adapter-api"))]
8989async fn send_to_session(to: &str, _message: &str, _notify_when_idle: bool) -> (String, bool) {
8990 (
8991 format!("Not sent: `{to}` is not one of your running subagents, and this build has no cross-session messaging."),
8992 true,
8993 )
8994}
8995
8996/// BP-8 (catalog:152): what [`Agent::rewind_conversation`] did.
8997#[derive(Debug, Clone, PartialEq, Eq)]
8998pub struct RewindOutcome {
8999 /// Messages remaining, including the system message at index 0.
9000 pub kept: usize,
9001 /// Messages removed from the live conversation (still on disk, in the
9002 /// journal, and still in the tree under `preserved_branch`).
9003 pub removed: usize,
9004 /// When the tree module is on and the rewind actually moved the leaf:
9005 /// the sibling branch the old leaf was preserved under, so the rewound
9006 /// path stays independently addressable.
9007 pub preserved_branch: Option<String>,
9008}