supercode_harness/agent.rs
1//! The agent loop.
2
3use std::collections::HashSet;
4
5use crate::config::{CachePlan, Config, SteeringMode, ToolAdvertising};
6use crate::error::{Error, Result};
7use crate::provider::{self, ChatRequest, OpenAiProvider, Provider, ToolSchema};
8use crate::reduce::rehydrate::CAP_NOTICE_MARKER;
9use crate::reduce::{self, ReductionLog, ReductionPolicy};
10use crate::tools::{ToolContext, ToolRegistry};
11use supercode_interchange::session::Session;
12use supercode_interchange::sidecar::SidecarWriter;
13use supercode_interchange::{ChatMessage, Role};
14use supercode_runtime::AgentEvent;
15
16/// BP-6: how many skill bodies one user message may pull in through `$slug`
17/// mentions — Claude Code caps skill chaining at six per message (cc§7
18/// "Skill chaining"); the same ceiling bounds the mention path here.
19const MAX_SKILL_LOADS_PER_MESSAGE: usize = 6;
20
21/// BP-5: how many `@path` mentions one message may attach. A prompt is not
22/// a bulk loader; past this the user means `--file`.
23const MAX_FILE_MENTIONS_PER_MESSAGE: usize = 10;
24
25/// BP-5: ceiling on the bytes one `@path` mention contributes.
26const MAX_FILE_MENTION_BYTES: usize = 64 * 1024;
27
28/// Tool name of the `tool_search` agent intrinsic (B6). Never a registered
29/// [`crate::tools::Tool`] — intercepted in [`Agent::run_tool`] before registry
30/// lookup, so it works under any [`ToolAdvertising`] mode.
31const TOOL_SEARCH: &str = "tool_search";
32
33/// Tool name of the `expand_reduction` agent intrinsic (T12/TR-1) — the
34/// model-invocable rehydration counterpart to `tool_search`, same
35/// interception pattern. Advertised whenever a [`ReductionPolicy`] is
36/// installed, regardless of [`ToolAdvertising`] mode (see [`Self::tool_schemas`]).
37const EXPAND_REDUCTION: &str = "expand_reduction";
38
39/// Tool name of the `sidecar_search` agent intrinsic (T12/TR-1).
40const SIDECAR_SEARCH: &str = "sidecar_search";
41
42/// Tool name of the `spawn_subagent` agent intrinsic (P5-3, §2 module 9 D1
43/// "spawn tool"). Same interception pattern as [`TOOL_SEARCH`] — never a
44/// registered [`crate::tools::Tool`], intercepted in [`Agent::run_tool`]
45/// before registry lookup — but ALSO needs full `&mut self` async access
46/// (running a whole child agent loop, or `tokio::spawn`-ing one), which
47/// [`Agent::prepare_tool_call`]'s purely-synchronous intrinsics don't, so
48/// the interception point is `Self::run_tool`'s top, not
49/// `prepare_tool_call`.
50const SPAWN_SUBAGENT: &str = "spawn_subagent";
51
52/// BP-7 (catalog §4a "Review mode"): the `[core.prompts]` key the review
53/// turn's template lives under. One name for both presets — cc spells the
54/// command `/code-review`, cx spells it `/review`, and both resolve to this
55/// template, so the row's evidence is one config key, not two.
56pub const REVIEW_PROMPT_NAME: &str = "code-review";
57
58/// BP-7 (catalog §4a "Side/ephemeral Q&A"): the instruction prefixed to a
59/// side question, so the model knows it is answering ABOUT the session
60/// rather than continuing it. The exchange never enters history either way;
61/// this keeps the answer from reading like the next assistant turn.
62const SIDE_QUESTION_PREAMBLE: &str = "[side question — answer from the conversation above; this exchange is not part of the conversation and you have no tools for it]";
63
64/// Claude Code's native name for [`SPAWN_SUBAGENT`]. It is exposed only when
65/// `Config::subagents_claude_agent_alias` is enabled for a Claude import.
66const CLAUDE_AGENT: &str = "Agent";
67
68/// Claude Code spellings for core filesystem/shell tools. Imported Claude
69/// context frequently continues to call these names even when another model
70/// is driving the turn, so emulation must translate execution as well as
71/// preserve the original call/result names in the transcript.
72const CLAUDE_BASH: &str = "Bash";
73const CLAUDE_READ: &str = "Read";
74const CLAUDE_WRITE: &str = "Write";
75const CLAUDE_EDIT: &str = "Edit";
76const CLAUDE_GLOB: &str = "Glob";
77const CLAUDE_GREP: &str = "Grep";
78
79/// Claude Code scheduler compatibility intrinsics. They edit an imported
80/// [`crate::ClaudeRuntimeManifest`]; actual timer execution belongs to an
81/// embedding scheduler driver, never this agent loop.
82const CLAUDE_CRON_CREATE: &str = "CronCreate";
83const CLAUDE_CRON_DELETE: &str = "CronDelete";
84const CLAUDE_CRON_LIST: &str = "CronList";
85const CLAUDE_SCHEDULE_WAKEUP: &str = "ScheduleWakeup";
86
87/// Shared SDK steering mailbox. `accepting` and `queue` share one lock so a
88/// turn's final boundary can close acceptance atomically with its last drain;
89/// a steer can therefore never be acknowledged into the following turn.
90#[derive(Default)]
91pub(crate) struct SteerInbox {
92 queue: std::collections::VecDeque<QueuedSteer>,
93 accepting: bool,
94}
95
96struct QueuedSteer {
97 message: String,
98 sdk_bound: bool,
99}
100
101impl SteerInbox {
102 pub(crate) fn open(&mut self) {
103 self.queue.clear();
104 self.accepting = true;
105 }
106
107 pub(crate) fn enqueue(&mut self, message: String) -> bool {
108 if !self.accepting {
109 return false;
110 }
111 self.queue.push_back(QueuedSteer {
112 message,
113 sdk_bound: true,
114 });
115 true
116 }
117
118 pub(crate) fn close(&mut self) {
119 self.accepting = false;
120 self.queue.retain(|queued| !queued.sdk_bound);
121 }
122
123 fn drain(&mut self, mode: SteeringMode) -> Option<String> {
124 if self.queue.is_empty() {
125 return None;
126 }
127 match mode {
128 SteeringMode::All => Some(
129 self.queue
130 .drain(..)
131 .map(|queued| queued.message)
132 .collect::<Vec<_>>()
133 .join("\n\n"),
134 ),
135 SteeringMode::OneAtATime => self.queue.pop_front().map(|queued| queued.message),
136 }
137 }
138
139 fn drain_or_close(&mut self, mode: SteeringMode) -> Option<String> {
140 if !self.queue.iter().any(|queued| queued.sdk_bound) {
141 self.accepting = false;
142 return None;
143 }
144 self.drain(mode)
145 }
146
147 fn queue_unchecked(&mut self, message: String) {
148 self.queue.push_back(QueuedSteer {
149 message,
150 sdk_bound: false,
151 });
152 }
153
154 fn len(&self) -> usize {
155 self.queue.len()
156 }
157}
158
159struct SteerTurnGuard {
160 inbox: std::sync::Arc<std::sync::Mutex<SteerInbox>>,
161}
162
163impl SteerTurnGuard {
164 fn new(inbox: std::sync::Arc<std::sync::Mutex<SteerInbox>>) -> Self {
165 inbox
166 .lock()
167 .unwrap_or_else(std::sync::PoisonError::into_inner)
168 .accepting = true;
169 Self { inbox }
170 }
171}
172
173impl Drop for SteerTurnGuard {
174 fn drop(&mut self) {
175 self.inbox
176 .lock()
177 .unwrap_or_else(std::sync::PoisonError::into_inner)
178 .close();
179 }
180}
181
182/// Tool name of the `subagent_status` agent intrinsic (P5-3, D3
183/// "background+resume"): poll (and reap, once finished) a background child
184/// spawned via [`SPAWN_SUBAGENT`]. Only advertised when
185/// `Config::subagents_background` is on (see [`Agent::tool_schemas`]).
186const SUBAGENT_STATUS: &str = "subagent_status";
187
188/// The agent's one message verb (cc's `SendMessage`, cx's v2 `send_message`):
189/// `to` is either a STILL-RUNNING background child (delivered into its own
190/// inbox) or another session of any harness, which goes through the one send
191/// every sender uses ([`crate::mail_send`]). Only advertised when
192/// `Config::subagents_background` is on.
193const SEND_MESSAGE: &str = "send_message";
194
195/// BP-7 (catalog §4a "Background subagents + resume": "resumable with
196/// context intact"): continue a FINISHED child with its own transcript
197/// restored, rather than starting a fresh one that has to be re-briefed.
198const SUBAGENT_RESUME: &str = "subagent_resume";
199
200/// Tool name of the `background_exec` agent intrinsic (P5-6, §2 module 4
201/// `tools.background` D1 "background exec"). Unlike [`SPAWN_SUBAGENT`], this
202/// needs no async child-agent loop — spawning a process
203/// (`tokio::process::Command::spawn`) is itself synchronous — so, like
204/// [`TOOL_SEARCH`], it is intercepted in [`Agent::prepare_tool_call`], not
205/// [`Agent::run_tool`].
206const BACKGROUND_EXEC: &str = "background_exec";
207
208/// Tool name of the `background_status` agent intrinsic (P5-6, D1 "monitor/
209/// event feed"): poll a background job's run status, drain its newly
210/// captured output as an [`AgentEvent::BackgroundOutput`] event, and reap it
211/// (remove it from [`Agent::background_jobs`]) once it has exited or been
212/// killed.
213const BACKGROUND_STATUS: &str = "background_status";
214
215/// Tool name of the `background_list` agent intrinsic (P5-6, D10
216/// "bg-manager"): list every background job this agent is currently
217/// tracking (running or finished-but-unreaped), without draining output or
218/// reaping anything.
219const BACKGROUND_LIST: &str = "background_list";
220
221/// Tool name of the `background_kill` agent intrinsic (P5-6, D10
222/// "bg-manager"): kill a background job's real OS process
223/// (`tokio::process::Child::start_kill`) and reap it immediately.
224const BACKGROUND_KILL: &str = "background_kill";
225
226/// P4e (§3.1 `core.parallel_tool_calls`): the synchronous outcome of
227/// [`Agent::prepare_tool_call`] — either a result already in hand (an
228/// intrinsic, or a call refused before it ever reached `Tool::execute`), or
229/// a plain registry-tool call ready for the (possibly concurrent) async
230/// `execute()` step.
231enum PreparedCall {
232 /// A final `(output, is_error)` result — no `Tool::execute` call is
233 /// coming for this one.
234 Done((String, bool)),
235 /// Passed every synchronous check; `execute(args, &ctx)` on the named
236 /// registry tool is the only remaining step.
237 Ready {
238 name: String,
239 args: serde_json::Value,
240 },
241}
242
243/// Marker prefix of the notice [`Agent::cap_tool_output`] appends to an
244/// oversized tool result kept in `history` (the recorder receives the full
245/// UX-26 (B7-warn): current wall-clock time as unix milliseconds, the same
246/// unit [`supercode_interchange::sidecar::rfc3339_to_ms`] parses session timestamps into —
247/// lets [`Agent::build_request_messages`] compare "now" against a
248/// cross-process signal (a loaded session's last message timestamp) on
249/// equal footing with an in-process one (this agent's own last annotated
250/// send). Saturates to 0 on a pre-epoch clock rather than panicking (never
251/// happens on real hardware, but `duration_since` can theoretically error).
252fn now_ms() -> i64 {
253 std::time::SystemTime::now()
254 .duration_since(std::time::UNIX_EPOCH)
255 .map(|d| d.as_millis() as i64)
256 .unwrap_or(0)
257}
258
259/// P5-3: process-wide sequence number backing [`next_subagent_id`] —
260/// disambiguates two spawns landing in the same millisecond (which
261/// `now_ms()` alone cannot).
262static SUBAGENT_ID_SEQ: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(0);
263
264/// P5-3: a fresh, process-unique child agent id (`"agent-<hex-ts>-<hex-seq>"`
265/// — the native analog of Claude Code's `agent-<id>` naming, see
266/// `supercode_interchange::session::SessionMeta::agent_id`'s doc comment).
267fn next_subagent_id() -> String {
268 let seq = SUBAGENT_ID_SEQ.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
269 format!("agent-{:x}-{:x}", now_ms(), seq)
270}
271
272/// P5-4: the shape [`Agent::child_approval_handler_factory`]/
273/// [`Agent::set_child_approval_handler_factory`] share — factored into its
274/// own alias (clippy `type_complexity`) rather than spelled out inline at
275/// both use sites.
276type ChildApprovalHandlerFactory = dyn Fn(
277 String,
278 std::sync::Arc<std::sync::Mutex<Vec<crate::subagents::QueuedApproval>>>,
279 ) -> std::sync::Arc<dyn crate::permissions::PermissionsApprovalHandler>
280 + Send
281 + Sync;
282
283/// BP-4 (catalog:109 "Context-usage introspection"): the live
284/// context-window accounting [`Agent::context_usage`] reports — cc's
285/// `/context` grid and cx's `/status` + `get_context_remaining` in one
286/// shape, over the numbers `resume --dry-run`'s preflight already computes.
287///
288/// Every token figure is the SAME estimate the context guard enforces
289/// (`supercode_runtime`), so what this reports and what refuses an oversized
290/// turn can never disagree.
291#[derive(Debug, Clone, PartialEq, Eq, serde::Serialize, serde::Deserialize)]
292pub struct ContextUsage {
293 /// The model the accounting is against.
294 pub model: String,
295 /// Messages in the projected request view (reduction stubs included).
296 pub messages: usize,
297 /// Estimated tokens for those messages.
298 pub message_tokens: u64,
299 /// Tools advertised on the next request.
300 pub tool_count: usize,
301 /// Estimated tokens for the serialized tool-schema array — a real part
302 /// of the wire request, and the half a message-only count misses.
303 pub tool_schema_tokens: u64,
304 /// `message_tokens + tool_schema_tokens`.
305 pub request_tokens: u64,
306 /// `request_tokens` with the guard's safety margin applied — the figure
307 /// the context guard actually compares.
308 pub projected_tokens: u64,
309 /// Headroom the guard reserves for the model's own reply.
310 pub response_reserve_tokens: u64,
311 /// The model's context window, when known.
312 pub context_limit: Option<u64>,
313 /// Tokens still available after the reply reserve, `0` when unknown.
314 pub remaining_tokens: u64,
315 /// `projected_tokens` as a whole percentage of the window (rounded),
316 /// `0` when the window is unknown. An integer so this whole struct
317 /// stays `Eq`-comparable on the frontend wire.
318 pub used_pct: u32,
319 /// Whether the next request would pass the context guard.
320 pub fits: bool,
321}
322
323impl ContextUsage {
324 /// One human line, the shape a `/context` command prints.
325 pub fn summary_line(&self) -> String {
326 match self.context_limit {
327 Some(limit) => format!(
328 "{} · {}% of {} used ({} projected, {} left) · {} messages {} · {} tool schemas {}",
329 self.model,
330 self.used_pct,
331 supercode_runtime::fmt_approx_tokens(limit),
332 supercode_runtime::fmt_approx_tokens(self.projected_tokens),
333 supercode_runtime::fmt_approx_tokens(self.remaining_tokens),
334 self.messages,
335 supercode_runtime::fmt_approx_tokens(self.message_tokens),
336 self.tool_count,
337 supercode_runtime::fmt_approx_tokens(self.tool_schema_tokens),
338 ),
339 None => format!(
340 "{} · context window unknown · {} projected · {} messages {} · {} tool schemas {}",
341 self.model,
342 supercode_runtime::fmt_approx_tokens(self.projected_tokens),
343 self.messages,
344 supercode_runtime::fmt_approx_tokens(self.message_tokens),
345 self.tool_count,
346 supercode_runtime::fmt_approx_tokens(self.tool_schema_tokens),
347 ),
348 }
349 }
350}
351
352/// A stateful agent: configuration, a model transport, a tool set, and the
353/// running conversation. Drive it with [`Agent::send`].
354/// BP-13 — one hop the run loop's failure-fallback pass performed: the
355/// model it was on, the model it moved to, and the provider failure that
356/// made it move.
357#[derive(Debug, Clone, PartialEq, Eq)]
358pub struct FallbackHop {
359 /// The model that failed.
360 pub from: String,
361 /// The next chain entry, which the request was re-sent against.
362 pub to: String,
363 /// The failure, rendered — the record's `reason`.
364 pub reason: String,
365}
366
367/// BP-13 — whether `error` is the kind of failure ANOTHER MODEL could
368/// plausibly answer, i.e. one the fallback chain exists for.
369///
370/// Deliberately narrow: rate limiting (429) and server-side failures (5xx,
371/// which is where "overloaded" lives) are properties of the model/endpoint
372/// that was asked, so asking a different one is a real remedy. Everything
373/// else — a bad request, a refused key, a decode failure, a tool error —
374/// is the CALLER's problem and would fail identically against every entry
375/// in the chain, so walking it would only multiply the same error by three.
376/// The transport's own retry (`OpenAiProvider::send_with_retry`) has
377/// already run and given up by the time this is consulted.
378pub fn is_failover_worthy(error: &Error) -> bool {
379 matches!(error, Error::Provider { status, .. } if *status == 429 || *status >= 500)
380}
381
382pub struct Agent {
383 config: Config,
384 provider: std::sync::Arc<dyn Provider>,
385 registry: ToolRegistry,
386 history: Vec<ChatMessage>,
387 ctx: ToolContext,
388 /// Cumulative output (completion) tokens across every `send` on this agent.
389 total_output_tokens: u64,
390 /// Names of non-core tools discovered via `tool_search` (B6): advertised
391 /// starting with the *next* request once populated.
392 activated_tools: HashSet<String>,
393 /// The live sidecar writer (A3), if this agent is recording. `None` is
394 /// today's behavior, at zero cost: every append point becomes a no-op.
395 recorder: Option<SidecarWriter>,
396 /// BP-8 (catalog:150 "Append-only durable transcript"): the live
397 /// append-only journal, if one is installed
398 /// ([`Self::set_journal`], armed by the caller when
399 /// [`Config::session_append_only`] is on). Behind an `Arc<Mutex<_>>`
400 /// rather than owned outright because the queue doors
401 /// ([`Self::queue_steer`]) take `&self` — a pending input has to be
402 /// recorded from a shared handle while a turn holds `&mut Agent`.
403 /// `None` (the default) is a no-op at every append point: today's
404 /// behavior, no file created.
405 journal: Option<std::sync::Arc<std::sync::Mutex<crate::session_journal::SessionJournal>>>,
406 /// BP-8 (catalog:151 "In-place conversation tree"): the live
407 /// `SessionTree` for this session, materialized when
408 /// [`Config::session_tree_enabled`] is on. Every recorded message
409 /// becomes a node, and [`Self::rewind_conversation`] moves the active
410 /// branch's leaf — the tree is what makes a rewind lossless (the old
411 /// leaf is preserved under a sibling branch) rather than a truncation.
412 /// `None` (the default, and every preset that leaves the module off) is
413 /// zero cost: nothing is built and nothing is persisted.
414 session_tree: Option<supercode_interchange::session_tree::SessionTree>,
415 /// BP-8 (catalog:152 "Rewind/rollback conversation"): tails removed by
416 /// rewinds that have not been undone, newest last. Restored from the
417 /// journal on resume, so "undo the rewind" survives a restart.
418 rewind_undo: Vec<Vec<ChatMessage>>,
419 /// BP-8 (catalog:156): the plan as last written to the journal —
420 /// compared against `ctx.plan` so an unchanged plan is not re-journaled
421 /// on every loop iteration.
422 journaled_plan: Vec<crate::session_journal::PlanEntry>,
423 /// Reversible reduction policy (A5/A7/A10). `None` is today's behavior,
424 /// at zero cost: every provider request is built from `self.history`
425 /// verbatim, exactly as before this landed.
426 reduction_policy: Option<ReductionPolicy>,
427 /// The accumulating reduction log (A5): fed back into
428 /// [`reduce::project_messages`] on every request-build so already-applied
429 /// reductions reproduce verbatim across turns and `send` calls (prefix
430 /// stability). `history` itself is never touched by this — see
431 /// `Self::run_loop`.
432 reduction_log: ReductionLog,
433 /// B7: length of the stable, byte-identical-across-turns prefix at the
434 /// front of [`Self::history`] — this agent's own system message plus
435 /// every message of a previously-imported session — set by
436 /// [`Self::load_session`]. `None` (the default) means no session has been
437 /// loaded, so [`crate::provider::apply_cache_plan`] has nothing to
438 /// annotate even under [`CachePlan::ImportedPrefix`].
439 imported_prefix_len: Option<usize>,
440 /// BP-11: set for the duration of a `/compact` so the `pre_compact`
441 /// observer can tell a manual compaction from an automatic trigger.
442 compacting_manually: bool,
443 /// BP-4 (catalog:90 "Environment context block", cx§2 "re-emitted on
444 /// change"): the `# Environment` block currently spliced into
445 /// `history[0]`, verbatim — `None` when `core.env_context` is off (or
446 /// on a construction path that assembles no prompt). Kept so
447 /// [`Self::refresh_env_context`] can locate and replace exactly this
448 /// text when cwd, approval/sandbox policy or the git branch moves
449 /// mid-session, instead of leaving the model reading a block that
450 /// stopped being true.
451 env_context_live: Option<String>,
452 /// BP-5 (catalog D2 "Per-model-family base-prompt selection"): the base
453 /// system prompt currently spliced at the head of `history[0]`,
454 /// verbatim. Kept for the same reason [`Self::env_context_live`] is:
455 /// when the model changes ([`Self::set_model`]) the family's prompt
456 /// changes with it, and the stale text has to be located and REPLACED
457 /// rather than left in front of the new model.
458 base_prompt_live: String,
459 /// BP-5 (catalog D2 "Shell-output injection in templates/skills"): the
460 /// permission-engine authorization every `` !`cmd` `` in a skill or
461 /// command body runs under. Inert (executes nothing) unless
462 /// `[core.skills] shell_injection` is on.
463 shell_injection: crate::skills::ShellInjection,
464 /// BP-4 (catalog:91): blocks spliced into the context AFTER
465 /// construction — see [`Self::inject_context_block`] and
466 /// [`crate::context_injection`]. Empty by default, at zero cost.
467 spliced_context_blocks: Vec<crate::config::ContextInjectionBlock>,
468 /// TR-7 (T20): the injectable side-call ([`reduce::summarize::SpanSummarizer`])
469 /// used to summarize an A10 `TurnsCleared` span, if one is installed
470 /// ([`Self::set_span_summarizer`]). `None` is today's behavior, at zero
471 /// cost: `Self::build_request_messages` never calls
472 /// [`reduce::prepare_cleared_turns_summary`] without one, so
473 /// `policy.summarize_cleared_turns` being on with no summarizer
474 /// installed behaves exactly like it being off (deterministic stub only)
475 /// — never a panic, never a blocked request.
476 span_summarizer: Option<std::sync::Arc<dyn reduce::summarize::SpanSummarizer + Send + Sync>>,
477 /// TR-8 (T5): the tool-schema tier signature (global knob + per-tool
478 /// overrides) as of the last request this agent built, or `None` before
479 /// the first request. Compared against the CURRENT signature at the top
480 /// of every `Self::build_request_messages` call so a tier change made
481 /// mid-session (via [`Self::set_schema_tier`] /
482 /// [`Self::set_tool_schema_tier`]) is detected and flagged to the B7
483 /// cache planner as a cache-bust event (`provider::tier_change_is_cache_bust`).
484 last_tool_schema_tier_signature: Option<u64>,
485 /// PARITY-18 D4 — the target model's context-window size, if the caller
486 /// has armed the guard via [`Self::set_context_limit`]. `None` (the
487 /// default) means no guard: every request is sent unconditionally.
488 /// CLI entry points arm it for their resolved model; direct SDK callers
489 /// retain explicit control through [`Self::set_context_limit`].
490 /// Once set, `Self::run_loop` re-checks
491 /// [`supercode_runtime::context_guard`] before EVERY request it builds —
492 /// not just the first — so "never sends an over-context request" holds
493 /// for the whole session, not only a one-shot preflight.
494 context_limit: Option<u64>,
495 /// PARITY-18 D3 — becomes `true` the first time `Self::run_loop`
496 /// actually reaches its real send site (immediately before
497 /// [`Provider::complete`]). Exposed via [`Self::request_issued`] so a
498 /// caller can report "request sent" truthfully — never asserted ahead
499 /// of time, so a pre-delivery failure (guard refusal, a build error) or
500 /// an interactive session that quits before any turn completes is
501 /// reported honestly as "not sent".
502 requests_issued: bool,
503 /// UX-26 (B7-warn): unix-ms wall-clock time this agent last knew the
504 /// active [`CachePlan::ImportedPrefix`] breakpoint to be warm. Seeded by
505 /// [`Self::load_session`] from the just-loaded session's OWN last
506 /// message timestamp (`metadata["timestamp"]`, parsed via
507 /// [`supercode_interchange::sidecar::rfc3339_to_ms`]) — a cross-process signal: how long
508 /// the resumed conversation has sat idle since ANY tool last touched it,
509 /// which is exactly when Anthropic's server-side cache entry (if one
510 /// ever existed) was last capable of being warm. Refreshed to "now"
511 /// every time `Self::run_loop` actually sends a cache-annotated
512 /// request (an in-process signal: idle time between this agent's own
513 /// turns). `None` when no imported prefix exists yet, or the loaded
514 /// session's last message carries no parseable timestamp — never
515 /// guessed, so the TTL check in [`provider::cache_cold_reason`] simply
516 /// doesn't fire rather than risk a false positive.
517 last_cache_activity_ms: Option<i64>,
518 /// UX-26: whether a PRIOR request already carried a cache_control
519 /// annotation for the current [`Self::imported_prefix_len`] — i.e.
520 /// whether reuse is genuinely "expected" on the NEXT annotated request.
521 /// `false` until the first annotated request goes out (that one is
522 /// establishing the cache entry, a legitimate write, never a "miss") and
523 /// reset to `false` by [`Self::load_session`] whenever the imported
524 /// prefix itself changes.
525 cache_established: bool,
526 /// UX-26 scratch: this turn's cache-warmth context, computed once at the
527 /// top of `Self::build_request_messages` (before the request is sent,
528 /// while `effective_cache_plan`/`busted` are in scope) and consumed once
529 /// in `Self::run_loop` right after `usage` comes back — never read
530 /// across turns, so a stale value can't leak. `(will_annotate,
531 /// cache_established, idle_secs)` — see [`provider::cache_cold_reason`]
532 /// for what each of the first two independently gates.
533 pending_cache_turn: (bool, bool, Option<i64>),
534 /// P4b: the injectable auto-title side-call ([`Self::set_session_titler`]),
535 /// mirroring `Self::span_summarizer`'s "installing one alone changes
536 /// nothing" contract — `Config::auto_title` is the actual gate a caller
537 /// consults before invoking [`Self::auto_title`].
538 session_titler: Option<std::sync::Arc<dyn crate::session_title::SessionTitler + Send + Sync>>,
539 /// P4b (§1.6, catalog §4a "persisted per-turn usage records"): every
540 /// [`crate::usage_log::UsageRecord`] recorded so far this agent's
541 /// lifetime. Always accumulated (cheap, small) regardless of whether a
542 /// caller ever persists it — see [`Self::usage_records`]/
543 /// [`Self::save_usage_log`].
544 usage_log: Vec<crate::usage_log::UsageRecord>,
545 /// P4b: 0-based index of the NEXT model round-trip, for
546 /// [`crate::usage_log::UsageRecord::turn`].
547 turn_index: usize,
548 /// BP-7 (catalog §4a "Turn/step bracketing records"): the per-round-trip
549 /// marker log — context/usage/finish brackets plus the retry, abort,
550 /// effort and goal markers. Persisted beside the session as
551 /// `<name>.events.jsonl` (see [`Self::save_turn_records`]).
552 turn_records: Vec<crate::turn_record::TurnRecord>,
553 /// BP-7: retries the transport reported, drained after every
554 /// `complete()` so each notice attaches to the round-trip that produced
555 /// it. Only the HTTP provider built by [`Self::new`] writes into this;
556 /// an injected provider simply never records anything.
557 retry_log: std::sync::Arc<crate::provider::RetryLog>,
558 /// BP-7 (catalog §4a "Per-turn cost/usage accounting", "Turn/budget
559 /// caps"): the price to bill this agent's model at, resolved at
560 /// construction from [`Config::price_input_per_mtok`]/
561 /// [`Config::price_output_per_mtok`] or [`crate::pricing`]'s table, and
562 /// re-resolved by [`Self::set_model`]. `None` = unpriceable, so no cost
563 /// is recorded (never a guess).
564 model_price: Option<crate::pricing::ModelPrice>,
565 /// BP-7: dollars this agent has spent across its whole lifetime — the
566 /// counter [`Config::max_budget_usd`] is measured against.
567 total_cost_usd: f64,
568 /// BP-7: tool calls this agent has executed across its whole lifetime —
569 /// the counter [`Config::max_steps`] is measured against.
570 total_steps: usize,
571 /// BP-7 (catalog §4a "Background subagents + resume"): a finished
572 /// child's post-system-prompt transcript, kept after the reap so
573 /// `subagent_resume` can restore its context in-process. A session
574 /// with a subagent store attached also has it on disk; this makes
575 /// resume work for an embedder that never attached one.
576 reaped_subagents: std::collections::HashMap<String, Vec<ChatMessage>>,
577 /// BP-7 (catalog §4a "Goals"): the session's standing objective, when
578 /// `capabilities.todos.goals` is on and one has been set. Restated at
579 /// the TAIL of every request while it stands (see
580 /// [`crate::goals::GoalRecord::reminder`]) and persisted as
581 /// `<session>.goal.json` — never written into `history`, so the
582 /// transcript stays exactly what the conversation was.
583 goal: Option<crate::goals::GoalRecord>,
584 /// P4b (§1.7/§3.1 `core.steering`, pi§3 semantics): queued mid-turn
585 /// steering messages — drained at the top of `Self::run_loop`'s next
586 /// iteration (pi's "steer = after current tool calls"). Empty by
587 /// default, at zero cost: `Self::run_loop` skips the drain entirely
588 /// when empty.
589 steer_queue: std::sync::Arc<std::sync::Mutex<SteerInbox>>,
590 /// P4b: queued follow-up messages — drained only once the loop is
591 /// otherwise idle (pi's "follow-up = at idle"), i.e. exactly the point
592 /// `Self::run_loop` would otherwise return a final answer.
593 follow_up_queue: std::collections::VecDeque<String>,
594 /// P4c (§5.2 P4 "doom-loop breaker", §3.1 `core.doom_loop_threshold`):
595 /// `(tool name, canonical JSON args)` of the most recent tool call, if
596 /// [`Config::doom_loop_threshold`] is armed — `None` before the first
597 /// call this agent has run. See [`Self::check_doom_loop`].
598 doom_loop_last_call: Option<(String, String)>,
599 /// P4c: how many times [`Self::doom_loop_last_call`] has repeated
600 /// consecutively so far (starts at 1 on the call that SET it).
601 doom_loop_streak: u32,
602 /// P4c (§1.10/§3.1 `core.model_switch.allow_switch`): every
603 /// [`crate::model_change::ModelChangeRecord`] [`Self::switch_model`] has
604 /// created so far this agent's lifetime. Always empty when
605 /// `Config::model_switch_allow_switch` is off (the default) or no
606 /// switch has happened yet.
607 model_change_log: Vec<crate::model_change::ModelChangeRecord>,
608 /// P4e (§1.6/§3.1 `core.session.git_metadata`, catalog:331): captured
609 /// once at construction when [`Config::session_git_metadata`] is on;
610 /// `None` when the gate is off (the default) or the best-effort git
611 /// probe found nothing (not a repo, `git` missing). See
612 /// [`Self::git_metadata`]/[`Self::save_git_metadata`].
613 git_metadata: Option<crate::git_metadata::GitMetadataRecord>,
614 /// P5-1 (§2.10, session-scoped "approve for session" cache): populated
615 /// only when a [`crate::permissions::PermissionsApprovalHandler`]
616 /// returns [`crate::permissions::ApprovalOutcome::AllowForSession`] —
617 /// see [`Self::prepare_tool_call`]'s `Config::permissions_enabled`
618 /// branch. Always constructed (cheap, empty) regardless of whether the
619 /// engine is ever active — the same "zero cost when off" posture as
620 /// [`Self::doom_loop_last_call`].
621 permissions_approval_cache: crate::permissions::ApprovalCache,
622 /// P5-1: the non-interactive decision seam a caller installs via
623 /// [`Self::set_permissions_approval_handler`] — mirrors
624 /// `Self::span_summarizer`/[`Self::session_titler`]'s "installing one
625 /// alone changes nothing, `Config::permissions_enabled` is the actual
626 /// gate" pattern. `None` (the default) means every `Ask`-tier decision
627 /// is denied (fail-closed — see
628 /// `crate::permissions::approval::PermissionsApprovalHandler`'s doc
629 /// comment).
630 permissions_approval_handler:
631 Option<std::sync::Arc<dyn crate::permissions::PermissionsApprovalHandler>>,
632 /// P5-2 (§2 module 15 D7 row 4 "prompts-as-commands"): MCP server
633 /// prompts registered via [`Self::register_mcp_prompt`], keyed by their
634 /// ALREADY-NAMESPACED command name (`mcp__<server>__<prompt>` — see
635 /// [`crate::mcp::McpPromptSource`]'s doc comment for why that namespace
636 /// is what keeps an untrusted server's prompt from ever colliding with
637 /// a trusted `Config::prompts` entry). Empty by default, at zero cost:
638 /// [`Self::expand_prompt_async`] only consults this after
639 /// `Config::prompts` finds no match.
640 mcp_prompts: std::collections::HashMap<String, Box<dyn crate::sdk::SdkPromptSource>>,
641 /// BP-6 (catalog D2 "Skills (progressive-disclosure packages)", D7
642 /// "Skill discovery from multiple roots"): the SKILL.md packages
643 /// discovered for this config, frontmatter only — name, description,
644 /// version and the manifest path. Never a body: a body is read from
645 /// disk on invocation (`/name`, `/skill:name`, a `$slug` mention, or
646 /// the `skill` tool) and nowhere else. Empty unless `[core.skills]` is
647 /// on AND names a harness whose roots to read.
648 skills: Vec<crate::skills::LoopSkill>,
649 /// P5-3 (§2 module 9): how deep in the spawn tree THIS agent is — `0`
650 /// for a top-level agent. Set from [`Config::subagent_depth`] at
651 /// construction; `Self::run_spawn_subagent` builds a child `Config`
652 /// with `subagent_depth = self.subagent_depth + 1` and ALSO overwrites
653 /// the freshly-built child `Agent`'s own field to match (belt-and-
654 /// suspenders — the child never has to trust its own `Config` alone).
655 subagent_depth: usize,
656 /// P5-3 (resource bound, "must not fork-bomb"): the shared, tree-wide
657 /// concurrency gauge every spawn (this agent's own, and every
658 /// descendant's) increments/decrements against
659 /// (`crate::subagents::try_acquire`/`ConcurrencyGuard`). A TOP-level
660 /// agent gets a fresh `Arc::new(AtomicUsize::new(0))` at construction;
661 /// `Self::run_spawn_subagent` clones this SAME `Arc` into every child it
662 /// spawns (never a fresh one), so a cap of N holds across the WHOLE
663 /// tree regardless of its branching shape — a parent with 3 children
664 /// each spawning 3 more shares one counter, not nine independent ones.
665 subagent_concurrency_gauge: std::sync::Arc<std::sync::atomic::AtomicUsize>,
666 /// P5-3 (D3 "background+resume"): background subagents this agent has
667 /// spawned and not yet reaped via `subagent_status`, keyed by their
668 /// `child_agent_id`. Each entry's `JoinHandle` moves its own
669 /// [`crate::subagents::ConcurrencyGuard`] into the spawned task, so the
670 /// concurrency slot is held for exactly as long as the child is
671 /// actually running, independent of whether/when the parent polls.
672 background_subagents: std::collections::HashMap<String, BackgroundSubagent>,
673 /// P5-3 (§2.2 C6 "parent-surfaced queue"): approval requests a
674 /// `background_prompts = "parent"` child raised, queued here rather
675 /// than blocking (see [`crate::subagents::QueuedApproval`]'s doc
676 /// comment — each is already resolved `Deny` by the time it lands
677 /// here). Exposed read-only via [`Self::pending_child_approvals`].
678 /// Always constructed (cheap, empty) regardless of whether background
679 /// spawning is ever used, same "zero cost when off" posture as
680 /// [`Self::permissions_approval_cache`].
681 pending_child_approvals:
682 std::sync::Arc<std::sync::Mutex<Vec<crate::subagents::QueuedApproval>>>,
683 /// P5-4 (tui, closes the P5-3 §2.2 C6 deferred chain — see
684 /// [`crate::subagents::ParentQueueApprovalHandler`]'s doc comment for
685 /// the "never blocks" contract this OVERRIDES only when a factory is
686 /// installed): when `Some`, `Self::run_spawn_subagent` uses THIS
687 /// factory — instead of constructing the default never-blocking
688 /// [`crate::subagents::ParentQueueApprovalHandler`] — to build the
689 /// `PermissionsApprovalHandler` a `background_prompts = "parent"`
690 /// child gets. Installed via
691 /// [`Self::set_child_approval_handler_factory`] by a `tui` embedder
692 /// that wants queued child approvals to be genuinely ANSWERABLE
693 /// (blocks the child's tool call until the parent resolves it, or
694 /// denies if the factory's handler's channel is ever dropped/closed —
695 /// still fail-closed, never a hang past process lifetime). `None` (the
696 /// default) preserves P5-3's shipped behavior byte-for-byte: every
697 /// `background_prompts = "parent"` child still gets the immediate-deny
698 /// `ParentQueueApprovalHandler`, and [`Self::pending_child_approvals`]
699 /// stays exactly the read-only audit view it already is.
700 child_approval_handler_factory: Option<std::sync::Arc<ChildApprovalHandlerFactory>>,
701 /// P5-3 (D5 "subagent transcripts… persisted + linked"): an optional
702 /// `(store, this agent's own session name)` pair installed via
703 /// [`Self::set_subagent_store`] — mirrors [`Self::set_recorder`]/
704 /// [`Self::set_span_summarizer`]'s "installing one alone changes
705 /// nothing" pattern. `None` (the default) means a spawned child's
706 /// transcript/lineage is still joined back into THIS agent's context
707 /// (the foreground/background mechanics work either way) but nothing
708 /// is written to a [`crate::store::SessionStore`] — no behavior change
709 /// for any caller that never installs one (e.g. every pre-P5-3 caller).
710 subagent_store: Option<(std::sync::Arc<crate::store::SessionStore>, String)>,
711 /// Imported Claude runtime state. The manifest can be paused or active,
712 /// but this Agent contains no scheduler or timer handle; an embedding
713 /// driver owns execution and persistence.
714 claude_runtime_manifest: Option<crate::claude_runtime_state::ClaudeRuntimeManifest>,
715 /// P5-6 (§2 module 4 `tools.background`, D10 "bg-manager"): background
716 /// OS processes spawned via `background_exec`, keyed by job id, tracked
717 /// until reaped (a terminal `background_status` poll, or an explicit
718 /// `background_kill`) — see [`BackgroundJob`]'s doc comment. Always
719 /// constructed (cheap, empty), same "zero cost when off" posture as
720 /// [`Self::background_subagents`].
721 background_jobs: std::collections::HashMap<String, BackgroundJob>,
722 /// P5-6 (resource bound, mirroring [`Self::subagent_concurrency_gauge`]'s
723 /// own precedent): the shared concurrency gauge every `background_exec`
724 /// call on this agent increments/decrements against
725 /// (`crate::subagents::try_acquire`/`ConcurrencyGuard` — reused
726 /// verbatim, a second independent gauge instance scoped to background
727 /// JOBS rather than subagent SPAWNS).
728 background_concurrency_gauge: std::sync::Arc<std::sync::atomic::AtomicUsize>,
729 /// P5-9 (§2 module 20 `checkpoint`): the write-path-interception
730 /// observer installed on [`Self::ctx`]'s `write_observer` (as a
731 /// `dyn WriteObserver`), held here ADDITIONALLY as its concrete type so
732 /// `Self::run_loop` can call
733 /// [`crate::checkpoint::CheckpointObserver::begin_turn`] once per turn.
734 /// `None` when `Config::checkpoint_enabled` is `false` (the default) or
735 /// the shadow store failed to open — see
736 /// [`crate::checkpoint::observer_for_config`].
737 checkpoint_observer: Option<std::sync::Arc<crate::checkpoint::CheckpointObserver>>,
738 /// P5-11 (§2 module 28 `lsp`): the LSP server registry installed (via
739 /// `crate::lsp::LspDiagnosticsObserver`) on `Self::ctx`'s
740 /// `write_observer` chain, held here ADDITIONALLY as its concrete type
741 /// so `impl Drop for Agent` can reach
742 /// [`crate::lsp::LspManager::kill_all_sync`] (no orphaned language-
743 /// server processes) and a clean-exit caller can reach
744 /// [`crate::lsp::LspManager::shutdown_all`] for a graceful handshake.
745 /// `None` when `Config::lsp_enabled` is `false` (the default).
746 lsp_manager: Option<std::sync::Arc<crate::lsp::LspManager>>,
747}
748
749/// P5-6 (§2 module 4 `tools.background`): one background-spawned OS process
750/// this agent is tracking, awaiting a `background_status`/`background_list`
751/// poll (or `background_kill`/agent drop) to reap or terminate it.
752///
753/// **Real process, not a child agent.** Unlike [`BackgroundSubagent`] (which
754/// wraps a whole recursive child [`Agent`] loop against the SAME mock/real
755/// provider), this wraps a plain OS subprocess spawned via
756/// `crate::tools::build_sandboxed_sh` — the exact function
757/// [`crate::tools::BashTool::execute`] itself calls, so a background
758/// command gets byte-identical sandboxing/cwd/env handling to a foreground
759/// `bash` call (build brief: "reuse the bash tool's execution + sandbox
760/// path").
761struct BackgroundJob {
762 /// The live process handle — kept directly on the job (not moved into a
763 /// spawned task) so [`Agent::run_background_status`]/
764 /// [`Agent::run_background_list`] can call the SYNCHRONOUS,
765 /// non-blocking `Child::try_wait` to observe exit status, and
766 /// [`Agent::run_background_kill`]/[`impl Drop for Agent`] can call the
767 /// SYNCHRONOUS `Child::start_kill` for a REAL process kill — never just
768 /// a `tokio::task::JoinHandle::abort` (which would only cancel a Rust
769 /// future, not the OS process it spawned). `kill_on_drop(true)` was set
770 /// at spawn time as defense-in-depth: even a `BackgroundJob` dropped
771 /// through some path OTHER than the explicit kill call sites below
772 /// still kills its child (a documented tokio behavior; a no-op if the
773 /// process already exited).
774 child: tokio::process::Child,
775 /// The exact command text this job is running — the SAME text that was
776 /// already checked against the permissions engine at spawn time (see
777 /// [`Agent::background_permission_denial`]).
778 command: String,
779 /// The OS process id, captured once at spawn time (before `child` is
780 /// ever mutated) — surfaced in every status/list/kill result, and the
781 /// only thing an OUTSIDE observer (e.g. a test proving real
782 /// termination) needs to check liveness independent of this process's
783 /// own bookkeeping.
784 pid: Option<u32>,
785 /// Bounded, incrementally-appended combined stdout+stderr capture —
786 /// written to by the reader tasks [`Agent::run_background_exec`] spawns
787 /// right after `child.stdout`/`child.stderr` are taken, read by every
788 /// status/list poll. Shared via `Arc` since the reader tasks outlive
789 /// this method call.
790 output: std::sync::Arc<supercode_runtime::background::CapturedOutput>,
791 /// Unix-ms wall-clock time the spawn happened.
792 started_at_ms: i64,
793 /// Set by [`Agent::run_background_kill`] — [`Agent::run_background_status`]/
794 /// [`Agent::run_background_list`] report [`supercode_runtime::background::JobStatus::Killed`]
795 /// unconditionally once this is `true`, rather than racing
796 /// `Child::try_wait` to see whether the kill signal has landed yet.
797 killed: bool,
798 /// The concurrency-gauge slot this job holds for as long as it remains
799 /// in [`Agent::background_jobs`] — dropped (freeing the slot) when this
800 /// `BackgroundJob` is removed from the map (a terminal reap, or an
801 /// explicit kill), exactly mirroring [`BackgroundSubagent`]'s own
802 /// "guard held for as long as it's tracked, not just while the process
803 /// is alive" posture (§2 module 9 precedent, kept consistent here).
804 _guard: crate::subagents::ConcurrencyGuard,
805}
806
807/// P5-6: the non-blocking status read [`Agent::run_background_status`]/
808/// [`Agent::run_background_list`] share — `job.killed` (set by
809/// [`Agent::run_background_kill`]) always wins over a fresh `try_wait`,
810/// since a kill signal racing the OS reaping the process is otherwise
811/// indistinguishable from "still running" for one poll cycle; reporting
812/// `Killed` unconditionally once requested avoids that race entirely. A
813/// `try_wait` error (would only happen if this job's id were somehow
814/// double-reaped, which the map ownership below already prevents) is
815/// treated as "no news yet" — `Running` — rather than inventing a made-up
816/// exit code.
817fn background_job_status(job: &mut BackgroundJob) -> supercode_runtime::background::JobStatus {
818 if job.killed {
819 return supercode_runtime::background::JobStatus::Killed;
820 }
821 match job.child.try_wait() {
822 Ok(Some(status)) => supercode_runtime::background::JobStatus::Exited(status.code()),
823 Ok(None) | Err(_) => supercode_runtime::background::JobStatus::Running,
824 }
825}
826
827/// Fable-5 review (HIGH, "grandchildren orphaned on kill AND agent-drop"):
828/// the shared real-kill body for both [`Agent::run_background_kill`] and
829/// `impl Drop for Agent` — sends `SIGKILL` to `job`'s ENTIRE process group,
830/// not just the one directly-tracked pid, so a surviving `&` job, pipeline
831/// stage, or double-forking daemon spawned by the job is killed too, then
832/// reaps the group leader so it doesn't linger as a zombie.
833///
834/// Relies on the spawn site (`Agent::run_background_exec`) having put the
835/// job in its OWN new process group via `Command::process_group(0)` — which
836/// makes the leader's pgid equal to its own pid, so `job.pid` doubles as the
837/// group id here.
838#[cfg(unix)]
839fn kill_job_process_group(job: &mut BackgroundJob) {
840 if let Some(pid) = job.pid {
841 // SAFETY: `libc::kill` with a negative pid is `killpg` — it only
842 // ever sends a signal (never dereferences memory), so this is safe
843 // regardless of whether the group is still alive. A `-1`/`ESRCH`
844 // return means the leader (and thus the whole group, since a group
845 // can't outlive its leader) already exited — not an error, just
846 // "already dead", exactly like `Child::start_kill`'s own documented
847 // no-op-on-already-exited contract.
848 unsafe {
849 libc::kill(-(pid as libc::pid_t), libc::SIGKILL);
850 }
851 }
852 // Belt-and-suspenders for the leader itself — `kill_on_drop(true)` set
853 // at spawn time is the same outcome via a different (implicit) path —
854 // then reap it so the SIGKILL we just delivered doesn't leave a zombie
855 // behind.
856 let _ = job.child.start_kill();
857 let _ = job.child.try_wait();
858}
859
860/// Non-unix fallback: no portable process-group primitive is wired up here
861/// (same posture as `crate::tools::build_sandboxed_sh`'s own platform
862/// split) — falls back to the pre-fix per-child kill. A background job that
863/// spawns a surviving grandchild process on a non-Unix target is a
864/// documented residual, not silently claimed fixed by this cfg arm.
865#[cfg(not(unix))]
866fn kill_job_process_group(job: &mut BackgroundJob) {
867 let _ = job.child.start_kill();
868}
869
870/// P5-6 (D1 "monitor/event feed", "output captured incrementally +
871/// BOUNDED"): spawn a fire-and-forget reader task that continuously drains
872/// `reader` (a piped `ChildStdout`/`ChildStderr`) into `output`, bounded at
873/// `cap` bytes. Reading NEVER stops at the cap — only what's RETAINED is
874/// bounded ([`supercode_runtime::background::CapturedOutput::append`]'s own contract)
875/// — because a background job's child process would otherwise block
876/// forever writing to a full, undrained OS pipe once this stopped reading
877/// it, silently hanging real work behind an apparently-"running" job. The
878/// task exits on its own once the pipe reaches EOF (the process closed the
879/// descriptor, whether by exiting or being killed) — no explicit
880/// abort/cleanup call site is needed; a detached `tokio::spawn` this short-
881/// lived is not the kind of orphaned-task risk `impl Drop for Agent`'s own
882/// doc comment is about (that one concerns a whole recursive provider-
883/// calling child AGENT loop, not a bounded byte-copy loop that ends the
884/// instant its source pipe closes).
885fn spawn_output_reader<R>(
886 reader: R,
887 output: std::sync::Arc<supercode_runtime::background::CapturedOutput>,
888 cap: usize,
889) -> tokio::task::JoinHandle<()>
890where
891 R: tokio::io::AsyncRead + Unpin + Send + 'static,
892{
893 tokio::spawn(async move {
894 use tokio::io::AsyncReadExt;
895 let mut reader = reader;
896 let mut buf = [0u8; 8192];
897 loop {
898 match reader.read(&mut buf).await {
899 Ok(0) => break,
900 Ok(n) => {
901 let chunk = String::from_utf8_lossy(&buf[..n]);
902 output.append(&chunk, cap);
903 }
904 Err(_) => break,
905 }
906 }
907 })
908}
909
910/// BP-7 (catalog §4a "Named agent definitions as data"): merge
911/// `<cwd>/.claude/agents/*.md` into `config.subagents_definitions`.
912///
913/// Runs for every `Agent` whose `subagents` module is on, whatever preset
914/// it came from — before BP-7 the `.md` loader was reachable only from the
915/// Claude emulate/resume path, so a cc-parity or cx-parity session ignored
916/// definitions sitting right there in the repo.
917///
918/// * A no-op when the module is off (the default), and for every spawned
919/// CHILD (`subagent_depth > 0`), which already inherits its parent's
920/// resolved definitions verbatim.
921/// * A config-table entry WINS over a discovered file of the same name:
922/// `[capabilities.subagents.agents.<name>]` is explicit configuration,
923/// the file is discovery.
924/// * A malformed file is skipped with a warning, never a failed
925/// construction: `Agent::with_provider` has no error channel, and a
926/// broken agent file in some repo must not make the harness unusable
927/// there. (The emulate/resume path keeps its own strict behavior, where
928/// a definition the resumed session may depend on going missing IS worth
929/// failing over.)
930fn merge_project_agent_definitions(config: &mut Config) {
931 if !config.subagents_enabled || config.subagent_depth > 0 {
932 return;
933 }
934 match crate::claude_compat::load_project_agents(&config.cwd) {
935 Ok(agents) => {
936 for agent in agents {
937 config
938 .subagents_definitions
939 .entry(agent.definition.name.clone())
940 .or_insert(agent.definition);
941 }
942 }
943 Err(e) => {
944 tracing::debug!(
945 error = %e,
946 "skipping .claude/agents discovery: a definition file could not be parsed"
947 );
948 }
949 }
950}
951
952/// P5-3: one background-spawned child this agent is tracking, awaiting a
953/// `subagent_status` poll (or agent drop) to reap it.
954struct BackgroundSubagent {
955 /// Resolves to `(child_agent_id, child's final result, the child's own
956 /// post-system-prompt history — for D5 transcript persistence once
957 /// reaped)` — the concurrency-guard slot for this child is held INSIDE
958 /// the spawned future (moved in at spawn time), so it releases the
959 /// instant the child's own run loop finishes, not when the parent gets
960 /// around to polling.
961 handle: tokio::task::JoinHandle<(String, Result<String>, Vec<ChatMessage>)>,
962 /// The task/prompt text the child was spawned with (surfaced by a
963 /// `"pending"` status poll, since the handle alone can't answer "what
964 /// is it doing").
965 task: String,
966 /// The named `agent_type` spawned, if any.
967 agent_type: Option<String>,
968 /// Unix-ms wall-clock time the spawn happened.
969 started_at_ms: i64,
970 /// BP-7 (catalog §4a "Background subagents + resume"): the child's own
971 /// steering inbox, captured before the child moved into its task.
972 ///
973 /// This IS the mailbox. `SteerInbox` was built (P4b) to be writable
974 /// while an active turn holds `&mut Agent` — exactly the property a
975 /// message-to-a-running-child needs — so the mailbox is that existing
976 /// seam reached from outside, not a second delivery channel with its
977 /// own ordering rules. A message lands at the top of the child's next
978 /// loop iteration, per `Config::steering_mode`.
979 mailbox: std::sync::Arc<std::sync::Mutex<SteerInbox>>,
980}
981
982/// P5-3 safety hardening (Fable-5 review, MEDIUM-LOW "orphaned billed
983/// spend"): a dropped parent must not leave a detached background child
984/// running against a REAL provider. Without this, a parent dropped
985/// mid-run (the caller's own process exits the scope, panics, or simply
986/// stops polling) leaves every still-running `BackgroundSubagent::handle`
987/// as an orphaned `tokio::spawn` task: nothing had ever awaited or
988/// aborted it, so it runs to its own (`max_iterations`-bounded)
989/// completion regardless — bounded but real provider spend nobody is
990/// paying attention to.
991///
992/// `.abort()` on a [`tokio::task::JoinHandle`] is safe to call
993/// unconditionally, including on an ALREADY-finished task (a documented
994/// no-op there — see tokio's `JoinHandle::abort` docs) — so this never
995/// needs to distinguish "still running" from "already done"; a background
996/// child that already finished and is merely awaiting a `subagent_status`
997/// reap is untouched in practice (aborting a finished task changes
998/// nothing observable). For a task still mid-flight, tokio cancels it at
999/// its next `.await` point, which drops that future in place — including
1000/// the `_guard: ConcurrencyGuard` moved into it at spawn time (see
1001/// `Self::run_spawn_subagent`'s `tokio::spawn` body) — so the
1002/// concurrency-gauge slot is released exactly the same way a normal
1003/// completion releases it (`ConcurrencyGuard`'s own `Drop`, in
1004/// `crate::subagents`). No separate cleanup call site to forget.
1005///
1006/// Deliberately does NOT touch [`Self::pending_child_approvals]` or
1007/// `Self::subagent_store` — this is purely "stop burning provider
1008/// calls on behalf of a caller who's gone", not a transcript-persistence
1009/// path (a child aborted mid-flight has no finished result to persist;
1010/// see this build's named residual on abort-time transcript loss).
1011impl Drop for Agent {
1012 fn drop(&mut self) {
1013 for (child_id, bg) in self.background_subagents.drain() {
1014 // Named, not silent: a child that was still running gets its
1015 // provider calls cut off here — worth a trace even though
1016 // there's no transcript left to persist (the future is
1017 // dropped mid-flight, before it ever returns a result).
1018 if !bg.handle.is_finished() {
1019 tracing::debug!(
1020 child_id = %child_id,
1021 "parent Agent dropped: aborting still-running background subagent \
1022 to stop further provider spend"
1023 );
1024 }
1025 bg.handle.abort();
1026 }
1027 // P5-6 (§2 module 4 `tools.background`, build brief "on agent drop
1028 // / session end, jobs MUST be killed... real process kill via the
1029 // child handle's kill(), not just tokio task abort"): a REAL OS
1030 // process, not a Rust task — `Child::start_kill` (synchronous, no
1031 // `.await` needed, so callable from this non-async `Drop::drop`)
1032 // sends the actual kill signal; a no-op, per its own docs, on a
1033 // job that already exited. `kill_on_drop(true)` (set at spawn
1034 // time) is a second, independent line of defense for the same
1035 // outcome, but this explicit loop is what makes the guarantee
1036 // provable/traceable rather than relying solely on an implicit
1037 // tokio runtime behavior.
1038 for (job_id, mut job) in self.background_jobs.drain() {
1039 if !job.killed {
1040 tracing::debug!(
1041 job_id = %job_id,
1042 command = %job.command,
1043 "parent Agent dropped: killing still-tracked background job's real \
1044 OS process (and its whole process group — see \
1045 `kill_job_process_group`)"
1046 );
1047 }
1048 kill_job_process_group(&mut job);
1049 }
1050 // P5-11 (§2 module 28 `lsp`, build brief "no orphaned language-
1051 // server processes"): a REAL OS process, same rationale as the
1052 // background-job loop just above — `kill_all_sync` is
1053 // synchronous (`Child::start_kill`, no `.await` needed, so
1054 // callable from this non-async `Drop::drop`), SIGKILLs each
1055 // server's WHOLE process group (unix — same `kill_job_process_group`
1056 // mechanism as the background-job loop above, so worker
1057 // grandchildren like rust-analyzer's proc-macro server or
1058 // typescript-language-server's `tsserver` are killed too, not just
1059 // the one directly-tracked pid), and is provable/traceable rather
1060 // than relying solely on `kill_on_drop(true)`'s implicit tokio
1061 // runtime behavior (which remains a second, independent line of
1062 // defense on every spawned `LspClient`).
1063 if let Some(lsp) = &self.lsp_manager {
1064 lsp.kill_all_sync();
1065 }
1066 }
1067}
1068
1069/// P4 (§1.8 credential-helper indirection, D6 row): run an `api_key_cmd`
1070/// through the shell and return its trimmed stdout. Runs via `sh -c` (POSIX
1071/// shell, matching pi's `!command` precedent) so the configured string can
1072/// use pipes/substitution, e.g. `pass show api-key`. Never panics or
1073/// propagates an error: a spawn failure or non-zero exit is reported via
1074/// `tracing::warn!` and returns an empty `String`, which
1075/// `Agent::new`'s resolution chain treats exactly like an unset helper —
1076/// falling through to `Config::api_key_env`.
1077fn run_api_key_cmd(cmd: &str) -> String {
1078 match std::process::Command::new("sh").arg("-c").arg(cmd).output() {
1079 Ok(out) if out.status.success() => String::from_utf8_lossy(&out.stdout).trim().to_string(),
1080 Ok(out) => {
1081 tracing::warn!(
1082 "api_key_cmd exited with status {:?}; falling back to api_key_env",
1083 out.status.code()
1084 );
1085 String::new()
1086 }
1087 Err(e) => {
1088 tracing::warn!("api_key_cmd failed to run ({e}); falling back to api_key_env");
1089 String::new()
1090 }
1091 }
1092}
1093
1094/// BP-9 (§3.1 `core.api_key_command`, D6 row "Credential helpers /
1095/// keyring"): run an ARGV credential helper and return its trimmed stdout.
1096/// No shell is involved — `argv[0]` is exec'd with the rest as arguments —
1097/// so a helper path with spaces, or an argument containing `$`/`;`, means
1098/// what it says. Same never-panics, fall-through-on-failure contract as
1099/// [`run_api_key_cmd`]: an empty result is treated as "no helper".
1100pub(crate) fn run_api_key_command(argv: &[String]) -> String {
1101 let Some((program, args)) = argv.split_first() else {
1102 return String::new();
1103 };
1104 match std::process::Command::new(program).args(args).output() {
1105 Ok(out) if out.status.success() => String::from_utf8_lossy(&out.stdout).trim().to_string(),
1106 Ok(out) => {
1107 tracing::warn!(
1108 "api_key_command exited with status {:?}; trying the next credential source",
1109 out.status.code()
1110 );
1111 String::new()
1112 }
1113 Err(e) => {
1114 tracing::warn!(
1115 "api_key_command failed to run ({e}); trying the next credential source"
1116 );
1117 String::new()
1118 }
1119 }
1120}
1121
1122/// P4c (§1.2/§3.1 `core.shell_env_snapshot`, SPLIT CC+CX row, catalog:338):
1123/// capture the user's interactive login-shell environment ONCE, best-effort.
1124/// Runs `$SHELL -lc env` (falling back to `sh -lc env` when `$SHELL` is
1125/// unset) — a LOGIN shell (`-l`) sources the user's rc files, which is
1126/// exactly the sourcing `bash` calls should no longer need to repeat once
1127/// this snapshot is in hand. Never panics: any failure (spawn error,
1128/// non-zero exit, unparseable output) returns an empty map, which
1129/// `ToolContext::shell_env`'s "no-op when `None`/empty" contract already
1130/// treats as harmless.
1131fn capture_shell_env() -> std::collections::HashMap<String, String> {
1132 let shell = std::env::var("SHELL").unwrap_or_else(|_| "sh".to_string());
1133 let out = match std::process::Command::new(&shell)
1134 .arg("-lc")
1135 .arg("env")
1136 .output()
1137 {
1138 Ok(o) if o.status.success() => o.stdout,
1139 Ok(o) => {
1140 tracing::warn!(
1141 "shell_env_snapshot: `{shell} -lc env` exited with status {:?}; snapshot is empty",
1142 o.status.code()
1143 );
1144 return std::collections::HashMap::new();
1145 }
1146 Err(e) => {
1147 tracing::warn!(
1148 "shell_env_snapshot: failed to run `{shell} -lc env` ({e}); snapshot is empty"
1149 );
1150 return std::collections::HashMap::new();
1151 }
1152 };
1153 let text = String::from_utf8_lossy(&out);
1154 let mut map = std::collections::HashMap::new();
1155 for line in text.lines() {
1156 if let Some((k, v)) = line.split_once('=') {
1157 if !k.is_empty() {
1158 map.insert(k.to_string(), v.to_string());
1159 }
1160 }
1161 }
1162 map
1163}
1164
1165/// Build the [`ToolContext`] an [`Agent`] hands to every tool call, folding
1166/// in every P4c per-tool config knob (§1.2) alongside the pre-existing
1167/// `cwd`/`sandbox` — shared by [`Agent::with_parts`]/[`Agent::with_provider_arc`]
1168/// so the two construction paths can never drift apart on which config
1169/// fields reach the context. Also builds (P5-9) the
1170/// [`crate::checkpoint::CheckpointObserver`], if `config.checkpoint_enabled`
1171/// — installed on the returned context's `write_observer` AND returned
1172/// separately (as the concrete type) so `Agent::run_loop` can call
1173/// [`crate::checkpoint::CheckpointObserver::begin_turn`] once per turn.
1174/// `None`/no-op end to end when the module is off — see
1175/// [`crate::checkpoint::observer_for_config`]'s own doc comment for the
1176/// default-off byte-identity guarantee.
1177///
1178/// P5-11 (§2 modules 28/29, D-5 "shared write-path interception seam"):
1179/// `crate::formatters::observer_for_config`/`crate::lsp::manager_for_config`
1180/// are folded into the SAME `write_observer` slot via
1181/// [`crate::tools::WriteObserverChain`], in the design's required order —
1182/// `checkpoint -> formatters -> lsp` (checkpoint's pre-image capture must
1183/// see the file before ANY mutation; lsp's diagnostics must see the file
1184/// AFTER formatting, never before). When 0 or 1 of the three modules is
1185/// active, this degrades to exactly what P5-9 shipped (`None`, or the
1186/// single concrete observer installed directly) — no chain wrapper is
1187/// introduced unless there is actually more than one observer to order,
1188/// keeping every single-module (or all-off) configuration byte-identical
1189/// to before this function grew multi-observer support. The `lsp` manager
1190/// is ALSO returned separately (like `checkpoint_observer`), so
1191/// `Agent`'s `Drop` impl can reach `crate::lsp::LspManager::kill_all_sync`
1192/// regardless of how the chain is shaped.
1193/// BP-2: `pub(crate)` so a parity test can build the SAME `ToolContext` an
1194/// `Agent` would from a resolved preset's `Config` and drive a registry
1195/// tool through it — a tool's behavior under a preset is exactly the
1196/// composition of the two, and a test that hand-assembled a context would
1197/// be proving an unwired function.
1198pub(crate) fn build_tool_context(
1199 config: &Config,
1200) -> (
1201 ToolContext,
1202 Option<std::sync::Arc<crate::checkpoint::CheckpointObserver>>,
1203 Option<std::sync::Arc<crate::lsp::LspManager>>,
1204) {
1205 let shell_env = if config.shell_env_snapshot {
1206 Some(std::sync::Arc::new(capture_shell_env()))
1207 } else {
1208 None
1209 };
1210 let checkpoint_observer = crate::checkpoint::observer_for_config(config);
1211 let format_observer = crate::formatters::observer_for_config(config);
1212 let lsp_manager = crate::lsp::manager_for_config(config);
1213 let lsp_observer = lsp_manager
1214 .clone()
1215 .map(|m| std::sync::Arc::new(crate::lsp::LspDiagnosticsObserver::new(m)));
1216 let mut observers: Vec<std::sync::Arc<dyn crate::tools::WriteObserver>> = Vec::new();
1217 if let Some(cp) = &checkpoint_observer {
1218 observers.push(cp.clone() as std::sync::Arc<dyn crate::tools::WriteObserver>);
1219 }
1220 if let Some(f) = &format_observer {
1221 observers.push(f.clone() as std::sync::Arc<dyn crate::tools::WriteObserver>);
1222 }
1223 if let Some(l) = &lsp_observer {
1224 observers.push(l.clone() as std::sync::Arc<dyn crate::tools::WriteObserver>);
1225 }
1226 let write_observer: Option<std::sync::Arc<dyn crate::tools::WriteObserver>> =
1227 match observers.len() {
1228 0 => None,
1229 1 => observers.into_iter().next(),
1230 _ => Some(std::sync::Arc::new(crate::tools::WriteObserverChain::new(
1231 observers,
1232 ))),
1233 };
1234 let ctx = ToolContext {
1235 cwd: config.cwd.clone(),
1236 // BP-10 (catalog row "Additional working directories"): the
1237 // `--add-dir`/`core.additional_dirs` roots reach the TOOLS now,
1238 // not just the project-context walk — `ToolContext::check_write`,
1239 // the OS backstop's writable set, and the permissions engine's
1240 // path rules all read them. Empty (the default) is byte-identical
1241 // to confining everything to `cwd`.
1242 extra_roots: config.additional_dirs.clone(),
1243 sandbox: config.sandbox,
1244 multimodal_read: config.read_file_multimodal,
1245 // BP-2 (§1.2 `core.tools.read_file.line_numbers`, catalog:26).
1246 read_line_numbers: config.read_file_line_numbers,
1247 require_read_before_edit: config.edit_file_require_read_before_edit,
1248 // BP-2: path → content hash at read time (`ToolContext::read_state`).
1249 read_paths: std::sync::Arc::new(std::sync::Mutex::new(std::collections::HashMap::new())),
1250 notebook_aware: config.edit_file_notebook_aware,
1251 shell_env,
1252 nested_instructions: config.nested_instructions,
1253 injected_instruction_dirs: std::sync::Arc::new(std::sync::Mutex::new(HashSet::new())),
1254 // BP-5: filled in by `Agent::with_parts` (the one construction path
1255 // that assembles a prompt, and therefore the one that knows which
1256 // rules were held back); empty everywhere else.
1257 path_rules: std::sync::Arc::new(Vec::new()),
1258 injected_rule_files: std::sync::Arc::new(std::sync::Mutex::new(HashSet::new())),
1259 // P5-1 (§2 module 12 carry-forward): now sourced from real config
1260 // (`capabilities.permissions.sandbox.network.*`, wired by
1261 // `configfile::materialize_config`) instead of always `None`. `None`
1262 // (the default, unchanged when the config never sets it) is still
1263 // byte-identical to today's behavior.
1264 network_policy: config.network_policy.clone(),
1265 // BP-10 (catalog row "Allow/ask/deny rule language", the DOMAIN
1266 // subject): the config's own rule arrays reach the network surface
1267 // too, so a `domain(...)` rule is evaluated by the SAME engine that
1268 // evaluates `bash(...)`/`write(...)` at the dispatch gate — not by
1269 // a second matcher over a second list. `None` when the permissions
1270 // module is off, which is byte-identical to before.
1271 permission_rules: config.permissions_enabled.then(|| {
1272 std::sync::Arc::new(crate::permissions::RuleSet {
1273 deny: config.tool_deny_patterns.clone(),
1274 ask: config.permissions_ask_patterns.clone(),
1275 allow: config.tool_allow_patterns.clone(),
1276 })
1277 }),
1278 // P4e (S3.1 `core.tools.bash.timeout_secs`, S14): folds the `bash`
1279 // `ToolOverride`'s `timeout_secs`, if set, into the context every
1280 // `BashTool::execute` call receives -- `None` (no override
1281 // configured) is byte-identical to today's behavior.
1282 bash_timeout_secs: config
1283 .tool_overrides
1284 .get("bash")
1285 .and_then(|o| o.timeout_secs),
1286 write_observer,
1287 // P5-10 (§2 module 12): sourced from real config
1288 // (`capabilities.permissions.sandbox.{enabled,escalation,env_policy}`,
1289 // wired by `configfile::materialize_config`). `sandbox_approval_handler`
1290 // starts `None` here (no handler is installed yet at `Agent`
1291 // construction time) and is kept in sync by
1292 // `Agent::set_permissions_approval_handler` — see that method's doc
1293 // comment.
1294 sandbox_os_enabled: config.sandbox_os_enabled,
1295 sandbox_escalation: config.sandbox_escalation,
1296 sandbox_env_policy: config.sandbox_env_policy,
1297 sandbox_approval_handler: None,
1298 // BP-3: both handler seams start `None` (nothing is installed at
1299 // construction time) and are filled by
1300 // `Agent::set_permissions_approval_handler` /
1301 // `Agent::set_user_question_handler`, exactly like
1302 // `sandbox_approval_handler` above. The two shared states are
1303 // always present but inert: plan mode starts off (contributing no
1304 // rules), and the budget starts unpublished.
1305 question_handler: None,
1306 approval_handler: None,
1307 plan_mode: std::sync::Arc::new(crate::tools::PlanModeState::new()),
1308 context_budget: std::sync::Arc::new(crate::tools::ContextBudget::new()),
1309 // BP-8 (catalog:156): the shared plan the agent journals and
1310 // persists. Always present, empty and inert until `update_plan`
1311 // writes one.
1312 plan: std::sync::Arc::new(std::sync::Mutex::new(Vec::new())),
1313 };
1314 (ctx, checkpoint_observer, lsp_manager)
1315}
1316
1317/// P4b (§1.4/§3.1, catalog §4a "Global/user-level instruction file tier"):
1318/// where the user/global instruction tier lives — `$SUPERCODE_HOME`, else
1319/// `$XDG_CONFIG_HOME/supercode`, else `~/.config/supercode`. Deliberately
1320/// duplicates `crates/cli/src/userconfig.rs::config_home`'s exact precedence
1321/// rather than depending on the `cli` crate from `core` (wrong dependency
1322/// direction — `cli` depends on `core`, never the reverse). `pub(crate)`:
1323/// also the DEFAULT shadow-store root `crate::checkpoint::observer_for_config`
1324/// (P5-9) derives from when `Config::checkpoint_dir` is unset — one
1325/// `$SUPERCODE_HOME` resolver, not a second hand-rolled one.
1326pub(crate) fn global_instructions_dir() -> std::path::PathBuf {
1327 if let Ok(h) = std::env::var("SUPERCODE_HOME") {
1328 if !h.is_empty() {
1329 return std::path::PathBuf::from(h);
1330 }
1331 }
1332 if let Ok(xdg) = std::env::var("XDG_CONFIG_HOME") {
1333 if !xdg.is_empty() {
1334 return std::path::PathBuf::from(xdg).join("supercode");
1335 }
1336 }
1337 let home = std::env::var("HOME").unwrap_or_else(|_| ".".into());
1338 std::path::PathBuf::from(home)
1339 .join(".config")
1340 .join("supercode")
1341}
1342
1343/// P4b (§1.4, catalog §4a "Instruction imports"): is `rel` (an `@`-import
1344/// target found inside a PROJECT-sourced instruction file) LEXICALLY safe to
1345/// resolve? Mirrors `configfile::is_safe_project_dir`'s posture (LOW-1
1346/// precedent): rejects absolute paths, `~`-relative paths, and any `..`
1347/// component — an untrusted repo's own CLAUDE.md/AGENTS.md must not be able
1348/// to `@import` its way to an arbitrary file on disk (e.g. `@/etc/passwd`,
1349/// `@../../.ssh/id_rsa`). Global-tier files (the user's own machine, same
1350/// trust level as the user's shell) are NOT run through this check.
1351///
1352/// This is a cheap PRE-FILTER only — it operates on the literal token text
1353/// and cannot see through a symlink committed in the repo whose *target*
1354/// escapes the root while the *link itself* has a clean, traversal-free
1355/// relative name (e.g. `@link.md` where `link.md -> /etc/passwd`). See
1356/// [`import_target_is_contained`] for the canonicalizing check that closes
1357/// that gap; the project-scoped resolution path runs both.
1358fn import_path_is_safe(rel: &str) -> bool {
1359 if rel.is_empty() || rel.contains('\0') {
1360 return false;
1361 }
1362 let path = std::path::Path::new(rel);
1363 if path.is_absolute() || rel.starts_with('~') {
1364 return false;
1365 }
1366 !path
1367 .components()
1368 .any(|c| matches!(c, std::path::Component::ParentDir))
1369}
1370
1371/// P4b security fix (Fable-5 review, MEDIUM: symlink bypass of the
1372/// project-scoped `@`-import boundary): does `candidate` — after resolving
1373/// symlinks — stay inside `root` — also after resolving symlinks? This is
1374/// what actually enforces [`import_path_is_safe`]'s doc-comment guarantee
1375/// ("must not be able to `@import` its way to an arbitrary file on disk"):
1376/// the lexical check alone rejects `@/etc/passwd` and `@../../secret`, but a
1377/// repo can commit a symlink (e.g. `link.md -> /etc/passwd`) whose own
1378/// relative name is perfectly clean, defeating a purely lexical check.
1379///
1380/// Both sides are canonicalized before the comparison — not just
1381/// `candidate` — because `root` itself can legitimately be a symlink (a
1382/// tempdir under macOS's `/tmp` -> `/private/tmp`, or any other symlinked
1383/// project checkout); comparing a canonicalized candidate against a
1384/// non-canonicalized root would falsely reject genuinely-in-root files.
1385///
1386/// Fails CLOSED: a `canonicalize()` failure (broken symlink, a target that
1387/// doesn't exist, a permission error) returns `false` — never inlined,
1388/// mirroring [`expand_instruction_imports`]'s existing "unreadable file ⇒
1389/// left as literal text" posture rather than panicking or defaulting open.
1390pub(crate) fn import_target_is_contained(
1391 candidate: &std::path::Path,
1392 root: &std::path::Path,
1393) -> bool {
1394 let (Ok(real_root), Ok(real_candidate)) = (
1395 std::fs::canonicalize(root),
1396 std::fs::canonicalize(candidate),
1397 ) else {
1398 return false;
1399 };
1400 real_candidate.starts_with(&real_root)
1401}
1402
1403/// P4b (§1.4/§3.1 `core.instruction_imports`, catalog:85): inline `@path`
1404/// import tokens found in `text` with the referenced file's own (trimmed)
1405/// content, resolved relative to `dir` (the directory the CONTAINING file
1406/// lives in — so a chain of imports each resolves relative to its own
1407/// location, not the original file's). `depth` bounds recursion (CC's own
1408/// default of 4, cited in the cc-parity preset) so a cyclical or
1409/// deeply-nested import chain can't blow the stack or loop forever.
1410/// `project_scoped` gates [`import_path_is_safe`] AND
1411/// [`import_target_is_contained`] — see their doc comments; `root` is the
1412/// containment boundary those checks canonicalize against (the SAME root
1413/// for every level of a nested import chain, even though `dir` itself walks
1414/// deeper with each level — an import three levels deep must still resolve
1415/// under the original project root, not merely under its own immediate
1416/// parent). Ignored when `!project_scoped` (the global/user tier, trusted,
1417/// unrestricted — see [`append_instruction_file`]'s doc comment).
1418/// Any token that isn't `@`-prefixed, doesn't resolve to a readable file, or
1419/// (project-scoped) fails the safety/containment check is left as literal
1420/// text — an import is best-effort, never a hard error that could make
1421/// instruction loading fail outright.
1422fn expand_instruction_imports(
1423 text: &str,
1424 dir: &std::path::Path,
1425 root: &std::path::Path,
1426 project_scoped: bool,
1427 depth: u8,
1428) -> String {
1429 if depth >= 4 {
1430 return text.to_string();
1431 }
1432 let mut out = String::with_capacity(text.len());
1433 for token in split_preserving_whitespace(text) {
1434 if let Some(rel) = token.strip_prefix('@') {
1435 if !rel.is_empty()
1436 && !rel.contains(char::is_whitespace)
1437 && (!project_scoped || import_path_is_safe(rel))
1438 {
1439 let candidate = dir.join(rel);
1440 if !project_scoped || import_target_is_contained(&candidate, root) {
1441 if let Ok(imported) = std::fs::read_to_string(&candidate) {
1442 let imported = imported.trim();
1443 if !imported.is_empty() {
1444 let imported_dir = candidate.parent().unwrap_or(dir);
1445 out.push_str(&expand_instruction_imports(
1446 imported,
1447 imported_dir,
1448 root,
1449 project_scoped,
1450 depth + 1,
1451 ));
1452 continue;
1453 }
1454 }
1455 }
1456 }
1457 }
1458 out.push_str(token);
1459 }
1460 out
1461}
1462
1463/// Split `text` into tokens that, concatenated, reproduce it exactly —
1464/// alternating runs of non-whitespace and whitespace. Used by
1465/// [`expand_instruction_imports`] so `@import` tokens can be located and
1466/// replaced without disturbing surrounding formatting/whitespace.
1467fn split_preserving_whitespace(text: &str) -> Vec<&str> {
1468 let mut out = Vec::new();
1469 let mut start = 0;
1470 let mut in_ws = None;
1471 for (i, c) in text.char_indices() {
1472 let ws = c.is_whitespace();
1473 match in_ws {
1474 None => in_ws = Some(ws),
1475 Some(prev) if prev != ws => {
1476 out.push(&text[start..i]);
1477 start = i;
1478 in_ws = Some(ws);
1479 }
1480 _ => {}
1481 }
1482 }
1483 if start < text.len() {
1484 out.push(&text[start..]);
1485 }
1486 out
1487}
1488
1489/// BP-4 (catalog:87 "Instruction-file hygiene controls", cc§2
1490/// `claudeMdExcludes`): whether `path` is excluded from instruction loading
1491/// by [`Config::project_doc_excludes`]. A pattern matches when it
1492/// [`crate::config::glob_match`]es the file's NAME (`CLAUDE.md`), its full
1493/// path, or its path relative to `root` — the three spellings cc's own
1494/// "glob/absolute-path list" accepts. Empty (the default) excludes nothing.
1495fn instruction_file_excluded(
1496 config: &Config,
1497 path: &std::path::Path,
1498 root: &std::path::Path,
1499) -> bool {
1500 if config.project_doc_excludes.is_empty() {
1501 return false;
1502 }
1503 let full = path.to_string_lossy().to_string();
1504 let name = path
1505 .file_name()
1506 .map(|n| n.to_string_lossy().to_string())
1507 .unwrap_or_default();
1508 let rel = path
1509 .strip_prefix(root)
1510 .ok()
1511 .map(|p| p.to_string_lossy().to_string());
1512 config.project_doc_excludes.iter().any(|pat| {
1513 crate::config::glob_match(pat, &full)
1514 || crate::config::glob_match(pat, &name)
1515 || rel
1516 .as_deref()
1517 .is_some_and(|r| crate::config::glob_match(pat, r))
1518 })
1519}
1520
1521/// BP-4 (catalog:87, cc§2 "HTML comment stripping"): drop block-level
1522/// `<!-- … -->` spans from an instruction file's text so maintainer notes
1523/// cost no tokens, exactly as cc does before injection. Unterminated
1524/// openers drop the remainder (the same reading a markdown renderer takes).
1525/// Off by default ([`Config::project_doc_strip_comments`]) — cx does NOT
1526/// strip, so this is a per-preset hygiene lever, not a universal one.
1527fn strip_html_comments(text: &str) -> String {
1528 let mut out = String::with_capacity(text.len());
1529 let mut rest = text;
1530 while let Some(open) = rest.find("<!--") {
1531 out.push_str(&rest[..open]);
1532 match rest[open..].find("-->") {
1533 Some(close) => rest = &rest[open + close + 3..],
1534 None => return out,
1535 }
1536 }
1537 out.push_str(rest);
1538 out
1539}
1540
1541/// P4b (§1.4): append one instruction file's (trimmed, import-expanded)
1542/// content to `blob` as a labeled section, exactly like the pre-P4b inline
1543/// loop did — a no-op when `path` doesn't exist or is empty (the common
1544/// case). `project_scoped` distinguishes the project tier (imports bounded
1545/// to `root`, canonicalized-and-contained — see
1546/// [`import_target_is_contained`]) from the global tier (imports
1547/// unrestricted, same trust level as the user's own machine — `root` is
1548/// unused in that case). `root` is normally `path`'s own parent (the tier
1549/// root `path` was discovered under, e.g. an ancestor of `cwd` or an
1550/// `additional_dirs` entry) — see [`assemble_project_instructions`]'s call
1551/// sites.
1552///
1553/// BP-4 adds the hygiene controls (catalog:87): the exclude list
1554/// ([`instruction_file_excluded`]), HTML-comment stripping
1555/// ([`strip_html_comments`]) and the [`InstructionBudget`] —
1556/// [`Config::project_doc_max_bytes`] spent INCREMENTALLY as files are
1557/// concatenated root→cwd, which is how cx's own cap works on its root-down
1558/// concat, rather than one chop at the end (that chop would silently eat
1559/// the trailing per-file notices it had just written).
1560fn append_instruction_file(
1561 blob: &mut String,
1562 config: &Config,
1563 path: &std::path::Path,
1564 root: &std::path::Path,
1565 label: &str,
1566 project_scoped: bool,
1567 budget: &mut InstructionBudget,
1568) {
1569 if budget.exhausted() || instruction_file_excluded(config, path, root) {
1570 return;
1571 }
1572 let Ok(text) = std::fs::read_to_string(path) else {
1573 return;
1574 };
1575 let stripped;
1576 let text = if config.project_doc_strip_comments {
1577 stripped = strip_html_comments(&text);
1578 stripped.trim()
1579 } else {
1580 text.trim()
1581 };
1582 if text.is_empty() {
1583 return;
1584 }
1585 let dir = path.parent().unwrap_or(std::path::Path::new("."));
1586 let mut content = if config.instruction_imports {
1587 expand_instruction_imports(text, dir, root, project_scoped, 0)
1588 } else {
1589 text.to_string()
1590 };
1591 if !budget.take(&mut content) {
1592 return;
1593 }
1594 blob.push_str(&format!("\n\n# {label}\n{content}"));
1595}
1596
1597/// Truncate `s` to at most `max` BYTES, backing off to the nearest char
1598/// boundary — shared by the per-file and aggregate instruction caps.
1599fn truncate_at_char_boundary(s: &mut String, max: usize) {
1600 let mut end = max;
1601 while end > 0 && !s.is_char_boundary(end) {
1602 end -= 1;
1603 }
1604 s.truncate(end);
1605}
1606
1607/// BP-4 (catalog:87 "Instruction-file hygiene controls", cx§2
1608/// `project_doc_max_bytes`): the instruction-content byte budget, spent as
1609/// files are concatenated root→cwd.
1610///
1611/// `None` (the default, and cc-parity's explicit `= 0`) is uncapped, so
1612/// [`Self::take`] is a no-op and assembly is byte-identical to a config
1613/// that never heard of the cap. With a cap set, each file is truncated to
1614/// whatever budget REMAINS (per-file notice), and once the budget is gone
1615/// the remaining files are skipped entirely (aggregate notice, emitted once
1616/// by [`Self::aggregate_notice`]) — the total instruction CONTENT can
1617/// therefore never exceed the cap, and the notices survive because nothing
1618/// chops the assembled blob afterwards.
1619struct InstructionBudget {
1620 remaining: Option<usize>,
1621 hit: bool,
1622}
1623
1624impl InstructionBudget {
1625 fn new(config: &Config) -> Self {
1626 InstructionBudget {
1627 remaining: config.project_doc_max_bytes,
1628 hit: false,
1629 }
1630 }
1631
1632 /// True once the cap has consumed the whole budget — later files are
1633 /// skipped rather than partially appended.
1634 fn exhausted(&self) -> bool {
1635 self.remaining == Some(0)
1636 }
1637
1638 /// Charge `content` against the budget, truncating it (and appending a
1639 /// per-file notice) when it doesn't fit. Returns whether anything is
1640 /// left to append.
1641 fn take(&mut self, content: &mut String) -> bool {
1642 let Some(remaining) = self.remaining else {
1643 return true;
1644 };
1645 if content.len() <= remaining {
1646 self.remaining = Some(remaining - content.len());
1647 return true;
1648 }
1649 self.hit = true;
1650 self.remaining = Some(0);
1651 if remaining == 0 {
1652 return false;
1653 }
1654 truncate_at_char_boundary(content, remaining);
1655 content.push_str("\n[supercode: file truncated at core.project_doc_max_bytes]");
1656 true
1657 }
1658
1659 /// The one aggregate notice, appended after assembly when the cap bound
1660 /// anywhere — the statement that the assembled block is not the whole
1661 /// instruction set.
1662 fn aggregate_notice(&self) -> &'static str {
1663 if self.hit {
1664 "\n\n[supercode: instruction content truncated at core.project_doc_max_bytes]"
1665 } else {
1666 ""
1667 }
1668 }
1669}
1670
1671/// BP-4 (catalog:81 "Project instruction files w/ directory walk"; cc§2
1672/// "Directory-walk loading", cx§2 "walk project root (git root) down to
1673/// cwd"): the ancestor chain instruction files are discovered on, ordered
1674/// OUTERMOST FIRST so the nearest directory wins precedence by appearing
1675/// last in the concatenated blob (the root→cwd ordering both inventories
1676/// document).
1677///
1678/// The walk starts at [`Config::cwd`] and climbs until it has included the
1679/// project root [`crate::config::project_root_for`] identifies (`.git` by
1680/// default — cx's `project_root_markers`, §3.1), or until the filesystem
1681/// root, whichever comes first. [`MAX_INSTRUCTION_WALK_DEPTH`] bounds it
1682/// unconditionally, so a marker-less path deep under `/` can never turn
1683/// prompt assembly into an unbounded stat storm.
1684pub(crate) fn instruction_walk_roots(config: &Config) -> Vec<std::path::PathBuf> {
1685 // BP-9's shared answer to "where does the project stop?" — the same
1686 // walk the `env_context` git probe and the CLI's `.supercode.toml`
1687 // discovery use, so one `project_root_markers` value cannot mean three
1688 // different things. `None` (no marker anywhere, or an empty list) means
1689 // no root was found, and the climb below then stops at the filesystem
1690 // root under `MAX_INSTRUCTION_WALK_DEPTH`.
1691 let root = crate::config::project_root_for(&config.cwd, &config.project_root_markers);
1692 let mut chain: Vec<std::path::PathBuf> = Vec::new();
1693 let mut dir = config.cwd.clone();
1694 loop {
1695 let at_root = root.as_deref() == Some(dir.as_path());
1696 chain.push(dir.clone());
1697 if at_root || chain.len() >= MAX_INSTRUCTION_WALK_DEPTH {
1698 break;
1699 }
1700 match dir.parent() {
1701 Some(parent) if parent != dir => dir = parent.to_path_buf(),
1702 _ => break,
1703 }
1704 }
1705 chain.reverse();
1706 chain
1707}
1708
1709/// Hard bound on [`instruction_walk_roots`]'s ancestor climb.
1710const MAX_INSTRUCTION_WALK_DEPTH: usize = 64;
1711
1712/// P4b (§1.4, obligation 4 assembly site): the full instruction-file blob —
1713/// global/user tier (catalog §4a "Global/user-level instruction file tier")
1714/// FIRST, then the project tier — capped by
1715/// [`Config::project_doc_max_bytes`] if set (catalog §4a "hygiene caps
1716/// (`project_doc_max_bytes` analog)").
1717///
1718/// BP-4 (catalog:81): the project tier is no longer `cwd` alone. It is the
1719/// ANCESTOR WALK [`instruction_walk_roots`] returns (cwd's chain up to the
1720/// git root, outermost first) followed by `additional_dirs` — root-first
1721/// ordering throughout, so the nearest directory wins by appearing later,
1722/// which is exactly how both cc§2 ("concatenated root→cwd, closest read
1723/// last") and cx§2 ("nearer-to-cwd wins by appearing later") describe their
1724/// own walks. `cwd` is the last element of the walk chain, so a config
1725/// whose cwd IS the project root assembles byte-identically to the pre-BP-4
1726/// loop.
1727/// BP-5 (catalog D2 "Per-model-family base-prompt selection", cx§2
1728/// "Per-model base instructions": "the system prompt is selected per model
1729/// family from bundled markdown … the active `base_instructions` are
1730/// persisted verbatim into the rollout `session_meta`"): the base system
1731/// prompt for the model this config runs.
1732///
1733/// `[capabilities.model_catalog] base_prompts` maps a model-id glob to that
1734/// family's prompt; the most specific match wins
1735/// ([`crate::model_catalog::base_prompt_for`]). No table and no match both
1736/// give [`Config::system_prompt`] verbatim, so this is a no-op for every
1737/// config that does not set the table.
1738fn base_prompt_for_config(config: &Config) -> String {
1739 crate::model_catalog::base_prompt_for(&config.model_family_prompts, &config.model)
1740 .map(str::to_string)
1741 .unwrap_or_else(|| config.system_prompt.clone())
1742}
1743
1744fn assemble_project_instructions(config: &Config) -> String {
1745 let mut blob = String::new();
1746 let mut budget = InstructionBudget::new(config);
1747 let global_dir = global_instructions_dir();
1748 for name in ["CLAUDE.md", "AGENTS.md"] {
1749 append_instruction_file(
1750 &mut blob,
1751 config,
1752 &global_dir.join(name),
1753 // Global tier is trusted/unrestricted (project_scoped=false
1754 // below) — `root` is never consulted, but pass `global_dir`
1755 // rather than a bogus value for clarity.
1756 &global_dir,
1757 name,
1758 false,
1759 &mut budget,
1760 );
1761 }
1762 // BP-10 (catalog row "Project/workspace trust gate", cc§4/cx§4:
1763 // "Prompt before loading project-local config/code"): the PROJECT tier
1764 // is trust-gated. The global/user tier above is not — it is the user's
1765 // own machine, the same trust level as their shell, exactly as
1766 // `append_instruction_file`'s `project_scoped = false` argument
1767 // already says.
1768 //
1769 // `crate::trust::is_trusted` asks the `Config::trust_handler` door
1770 // once per project and records the answer; with no door installed
1771 // `TrustSurface::Instructions` resolves to LOADED, which is both the
1772 // pre-BP-10 behavior and what a headless run of either upstream
1773 // harness does — see `crate::trust`'s doc comment for why the
1774 // undecided answer differs between text and code.
1775 if !crate::trust::is_trusted(config, crate::trust::TrustSurface::Instructions) {
1776 blob.push_str(
1777 "\n[project instruction files were not loaded: this workspace is not trusted (capabilities.trust)]\n",
1778 );
1779 blob.push_str(budget.aggregate_notice());
1780 return blob;
1781 }
1782 let walk = instruction_walk_roots(config);
1783 for root in walk.iter().chain(config.additional_dirs.iter()) {
1784 for name in ["CLAUDE.md", "AGENTS.md"] {
1785 append_instruction_file(
1786 &mut blob,
1787 config,
1788 &root.join(name),
1789 // Project tier: `@`-imports from THIS file must stay under
1790 // THIS root (canonicalized) — see
1791 // `import_target_is_contained`.
1792 root,
1793 name,
1794 true,
1795 &mut budget,
1796 );
1797 }
1798 // A repository-native agent package is an additional project
1799 // instruction tier. It is subject to the same `project_context`
1800 // switch, import containment, and aggregate byte cap as root
1801 // AGENTS.md/CLAUDE.md; loading it never executes package code.
1802 for path in crate::agent_package::workspace_package_instruction_files(root) {
1803 append_instruction_file(
1804 &mut blob,
1805 config,
1806 &path,
1807 root,
1808 "Volter Harness agent package instructions",
1809 true,
1810 &mut budget,
1811 );
1812 }
1813 }
1814 blob.push_str(budget.aggregate_notice());
1815 blob
1816}
1817
1818/// P4b (§1.4/§3.1 `core.env_context`, catalog §4a "Environment context block
1819/// injection"): cwd, platform, date, and a best-effort git branch/dirty
1820/// status (silently absent when `cwd` isn't a git repo or `git` isn't on
1821/// `PATH` — never blocks agent construction).
1822///
1823/// BP-4 (catalog:90): plus the APPROVAL/SANDBOX POLICY line the row's own
1824/// semantics name ("cwd/git/platform/date/**policy**") and cx's
1825/// `<environment_context>` supplies — the model is told which approval mode
1826/// and which filesystem confinement it is operating under, which is what
1827/// makes "ask before you do X" instructions legible to it. The block is
1828/// re-derivable at any moment from `config` alone, which is what lets
1829/// [`Agent::refresh_env_context`] re-emit it mid-session on change.
1830/// BP-6 (catalog D2 "Skills (progressive-disclosure packages)", §1.4
1831/// obligation 4): the `# Skills` prompt section — the discovered SKILL.md
1832/// packages' names and descriptions, plus the `[core.prompts]` template
1833/// names, and NOTHING else. A skill's body is deliberately absent: it costs
1834/// its tokens only when something actually invokes it (`docs:skills`
1835/// "body loads only when used"; cx§7; pi§2 "progressive disclosure").
1836///
1837/// A skill whose frontmatter hides it from the model (`enabled: false`,
1838/// `disable-model-invocation: true`) is left OUT of the index while staying
1839/// user-invocable — cc§7 "Invocation control", pi§2.
1840///
1841/// Empty string when there is nothing to list, so an agent with neither
1842/// skills nor templates keeps the prompt it had before this existed.
1843fn skills_prompt_section(config: &Config, skills: &[crate::skills::LoopSkill]) -> String {
1844 let listed: Vec<&crate::skills::LoopSkill> = skills
1845 .iter()
1846 .filter(|skill| skill.model_invocable)
1847 .collect();
1848 let mut templates: Vec<&str> = config.prompts.keys().map(String::as_str).collect();
1849 templates.sort_unstable();
1850 if listed.is_empty() && templates.is_empty() {
1851 return String::new();
1852 }
1853 let mut out = String::from("\n\n# Skills\n");
1854 if !listed.is_empty() {
1855 out.push_str(
1856 "Installed skill packages. Only each skill's name and description are listed \
1857 here; call the `skill` tool with a name below to load that skill's full \
1858 instructions when it applies, then follow them.\n",
1859 );
1860 for skill in listed {
1861 out.push_str(&skill.index_line());
1862 out.push('\n');
1863 }
1864 }
1865 if !templates.is_empty() {
1866 if !out.ends_with("# Skills\n") {
1867 out.push('\n');
1868 }
1869 out.push_str("Prompt templates (invoke via `/name args`):\n");
1870 for name in templates {
1871 out.push_str(&format!("- {name}\n"));
1872 }
1873 }
1874 out
1875}
1876
1877fn env_context_block(config: &Config) -> String {
1878 let mut lines = vec![
1879 format!("cwd: {}", config.cwd.display()),
1880 format!("platform: {}", std::env::consts::OS),
1881 format!(
1882 "date: {}",
1883 supercode_interchange::sidecar::now_rfc3339()
1884 .get(..10)
1885 .unwrap_or("")
1886 ),
1887 format!(
1888 "approval policy: {} · sandbox: {}",
1889 approval_policy_label(config.approval),
1890 sandbox_policy_label(config.sandbox),
1891 ),
1892 ];
1893 // BP-9 (§3.1 `core.project_root_markers`, catalog:232): the git probe
1894 // runs at the PROJECT ROOT the markers define, not at whatever
1895 // subdirectory the process happens to sit in — the marker knob's whole
1896 // job is deciding where "the project" starts. Falls back to `cwd` when
1897 // no ancestor carries a marker (or the list is empty), which is
1898 // byte-identical to the pre-BP-9 behavior.
1899 let root = crate::config::project_root_for(&config.cwd, &config.project_root_markers)
1900 .unwrap_or_else(|| config.cwd.clone());
1901 if let Some(status) = env_context_git_status(&root) {
1902 lines.push(status);
1903 }
1904 format!("\n\n# Environment\n{}", lines.join("\n"))
1905}
1906
1907/// The `[capabilities.permissions] approval` spelling of a policy — the same
1908/// token the config schema accepts (`configfile::parse_approval_str`), so
1909/// the block reports the policy in the vocabulary the user configured it in.
1910fn approval_policy_label(policy: crate::config::ApprovalPolicy) -> &'static str {
1911 match policy {
1912 crate::config::ApprovalPolicy::Never => "never",
1913 crate::config::ApprovalPolicy::OnRequest => "on-request",
1914 crate::config::ApprovalPolicy::Untrusted => "untrusted",
1915 crate::config::ApprovalPolicy::ModelRequested => "model-requested",
1916 }
1917}
1918
1919/// The `[capabilities.permissions] sandbox` spelling of a tier — see
1920/// [`approval_policy_label`].
1921fn sandbox_policy_label(policy: crate::tools::SandboxPolicy) -> &'static str {
1922 match policy {
1923 crate::tools::SandboxPolicy::ReadOnly => "read-only",
1924 crate::tools::SandboxPolicy::WorkspaceWrite => "workspace-write",
1925 crate::tools::SandboxPolicy::DangerFullAccess => "danger-full-access",
1926 }
1927}
1928
1929/// Best-effort `git branch (dirty|clean)` for [`env_context_block`]. `None`
1930/// on anything short of a clean success (not a repo, `git` missing, a
1931/// detached/errored state) — this is informational context, never worth
1932/// failing agent construction over.
1933fn env_context_git_status(cwd: &std::path::Path) -> Option<String> {
1934 let branch_out = std::process::Command::new("git")
1935 .args(["rev-parse", "--abbrev-ref", "HEAD"])
1936 .current_dir(cwd)
1937 .output()
1938 .ok()?;
1939 if !branch_out.status.success() {
1940 return None;
1941 }
1942 let branch = String::from_utf8_lossy(&branch_out.stdout)
1943 .trim()
1944 .to_string();
1945 if branch.is_empty() {
1946 return None;
1947 }
1948 let dirty = std::process::Command::new("git")
1949 .args(["status", "--porcelain"])
1950 .current_dir(cwd)
1951 .output()
1952 .ok()
1953 .map(|o| !o.stdout.is_empty())
1954 .unwrap_or(false);
1955 Some(format!(
1956 "git branch: {branch} ({})",
1957 if dirty { "dirty" } else { "clean" }
1958 ))
1959}
1960
1961impl Agent {
1962 /// Build an agent backed by an OpenAI-compatible endpoint (OpenRouter by
1963 /// default). The API key is taken from [`Config::api_key`], then
1964 /// [`Config::api_key_cmd`] (P4: a credential-helper command, run via the
1965 /// shell — see `run_api_key_cmd`), then the configured environment
1966 /// variable ([`Config::api_key_env`]).
1967 pub fn new(config: Config) -> Result<Self> {
1968 let api_key = match &config.api_key {
1969 Some(k) if !k.is_empty() => k.clone(),
1970 // BP-9: the ARGV helper (`core.api_key_command`) is consulted
1971 // first — it is the form with no shell in the path, so a config
1972 // that sets both gets the one with fewer ways to surprise its
1973 // author. Empty/failed → fall through, same as `api_key_cmd`.
1974 _ => match config
1975 .api_key_command
1976 .as_deref()
1977 .filter(|argv| !argv.is_empty())
1978 .map(run_api_key_command)
1979 .filter(|k| !k.is_empty())
1980 .or_else(|| {
1981 config
1982 .api_key_cmd
1983 .as_deref()
1984 .filter(|c| !c.is_empty())
1985 .map(run_api_key_cmd)
1986 }) {
1987 // P4 (§1.8 credential-helper indirection, D6 row): the
1988 // helper ran and produced a non-empty key — use it. A
1989 // failed/empty helper falls through to `api_key_env` rather
1990 // than erroring outright, same "try the next source"
1991 // posture as every other layer in this resolution chain.
1992 Some(k) if !k.is_empty() => k,
1993 _ => std::env::var(&config.api_key_env)
1994 .ok()
1995 .filter(|k| !k.is_empty())
1996 .ok_or_else(|| Error::MissingApiKey(config.api_key_env.clone()))?,
1997 },
1998 };
1999 // P4b (§1.1/§3.1 `core.retry`, pi§3 shape): `Config.retry_*` now
2000 // reaches the pre-existing transport-layer retry mechanism (see
2001 // `provider::HttpOptions::from_retry_config`'s doc comment for the
2002 // exact "byte-identical when unset" contract).
2003 let http_options = provider::HttpOptions::from_retry_config(
2004 config.retry_enabled,
2005 config.retry_max_retries,
2006 config.retry_base_delay_ms,
2007 );
2008 // BP-7 (catalog §4a "Turn/budget caps"): a spend cap armed against
2009 // a model this build cannot price is refused HERE rather than
2010 // accepted and silently never enforced. See
2011 // `Config::max_budget_usd`.
2012 if config.max_budget_usd.is_some_and(|b| b > 0.0)
2013 && crate::pricing::resolve(
2014 &config.model,
2015 config.price_input_per_mtok,
2016 config.price_output_per_mtok,
2017 )
2018 .is_none()
2019 {
2020 return Err(Error::UnpriceableBudget {
2021 model: config.model.clone(),
2022 });
2023 }
2024 // BP-7 (catalog §4a "Auto-retry on transient provider errors"): the
2025 // shared log the transport's retry loop reports into and
2026 // `Self::run_loop` drains after every completion.
2027 let retry_log = std::sync::Arc::new(crate::provider::RetryLog::default());
2028 let provider = OpenAiProvider::new_with_options(
2029 config.base_url.clone(),
2030 api_key,
2031 config.extra_headers.clone(),
2032 http_options,
2033 )
2034 .with_retry_log(retry_log.clone());
2035 // P3 (design §5.2): `ToolRegistry::from_config` replaces the
2036 // unconditional `with_builtins()` call — a no-op when
2037 // `config.module_registry` is off (the default, §5.3 risk 2).
2038 let registry = ToolRegistry::from_config(&config);
2039 let mut agent = Self::with_parts(config, Box::new(provider), registry);
2040 agent.retry_log = retry_log;
2041 Ok(agent)
2042 }
2043
2044 /// Build an agent with an explicit provider and the built-in tools. Handy
2045 /// for tests (inject a mock provider) or custom transports.
2046 pub fn with_provider(config: Config, provider: Box<dyn Provider>) -> Self {
2047 let registry = ToolRegistry::from_config(&config);
2048 Self::with_parts(config, provider, registry)
2049 }
2050
2051 /// Build an agent from all three parts.
2052 pub fn with_parts(
2053 mut config: Config,
2054 provider: Box<dyn Provider>,
2055 mut registry: ToolRegistry,
2056 ) -> Self {
2057 // P5-12 (§2 module 18 `plugins`, D-10): register every trusted,
2058 // loaded plugin's declared tools — the same "unconditional, config-
2059 // gated" wiring `build_tool_context` just below gives
2060 // checkpoint/formatters/lsp. `crate::plugins::register_into` is a
2061 // true no-op (no filesystem read, no subprocess) whenever
2062 // `config.plugins_enabled` is `false` (the default) — byte-identical
2063 // to before this module existed. Runs here (the one tail every
2064 // `Agent` construction path funnels through — `new`/`with_provider`
2065 // both call this) rather than in `ToolRegistry::from_config`, so it
2066 // is NOT entangled with that function's unrelated `module_registry`
2067 // experimental gate.
2068 crate::plugins::register_into(&config, &mut registry);
2069 let (mut ctx, checkpoint_observer, lsp_manager) = build_tool_context(&config);
2070 // Auto-load project context files (CLAUDE.md / AGENTS.md) from the
2071 // working directory (and any extra roots), appending them to the system
2072 // prompt — the analog of how Claude Code / Codex discover them.
2073 // P4b (§1.4): also the global/user tier + instruction imports + the
2074 // `project_doc_max_bytes` hygiene cap — see `assemble_project_instructions`.
2075 // BP-5 (catalog D2 "Per-model-family base-prompt selection", cx§2
2076 // "Per-model base instructions"): the base prompt is chosen for the
2077 // model in force, not fixed before the model is known — the exact
2078 // residue the ledger row named. `base_prompt_for_config` is
2079 // `config.system_prompt` verbatim for every config that sets no
2080 // family table, so this is a no-op by default.
2081 let base_prompt_live = base_prompt_for_config(&config);
2082 // BP-5 (catalog D2 "Output style / personality module"): a custom
2083 // style may REPLACE the base coding instructions rather than append
2084 // to them (cc§7 `keep-coding-instructions`); every other style is
2085 // appended at the end of the assembled prompt, below.
2086 let output_style = crate::output_style::resolve(&config);
2087 let mut system = match output_style.as_ref() {
2088 Some(style) if style.replaces_base => style.text.clone(),
2089 _ => base_prompt_live.clone(),
2090 };
2091 if config.load_project_context {
2092 system.push_str(&assemble_project_instructions(&config));
2093 }
2094 // BP-5 (catalog D2 "Path-scoped rules", cc§2 `.claude/rules`): the
2095 // UNSCOPED rules join the instruction blob here. A rule carrying a
2096 // `paths:` selector deliberately does not — it waits for a tool to
2097 // touch a matching file (`tools::builtins::path_rules_notice`).
2098 let path_rules = crate::path_rules::load(&config);
2099 system.push_str(&crate::path_rules::always_on_text(&path_rules));
2100 // P4b (§1.4/§3.1 `core.env_context`, catalog §4a "Environment
2101 // context block injection"): `false` (the default) is a no-op —
2102 // byte-identical to today's behavior. BP-4 keeps the rendered block
2103 // on the agent (`env_context_live`) so `refresh_env_context` can
2104 // find and REPLACE exactly this text when cwd/policy/branch move,
2105 // rather than leaving a stale block in the prompt forever.
2106 let env_context_live = if config.env_context {
2107 let block = env_context_block(&config);
2108 system.push_str(&block);
2109 Some(block)
2110 } else {
2111 None
2112 };
2113 // P4e (§1.4/§3.1 `core.context_injections`, catalog:91 "Synthetic
2114 // context-injection blocks"): same assembly site, right after
2115 // `env_context`. `false` (the default) is a no-op — byte-identical
2116 // to today's behavior. BP-4 routes it through
2117 // `crate::context_injection`, so the gate now delivers the built-in
2118 // ambient blocks the row is about (and stays extensible at runtime
2119 // through `Self::inject_context_block`) instead of only whatever
2120 // static list an embedder happened to populate.
2121 system.push_str(&crate::context_injection::assemble(&config, &[]));
2122 // P3 (design §5.2, §1.4 obligation 4, D-7): the skills prompt
2123 // section is a MODULE-GATED prompt section, the design's own
2124 // illustration of "a disabled module contributes no prompt
2125 // sections" — only assembled at all under
2126 // `[experimental] module_registry = true` (§5.3 risk 2: flag-off is
2127 // byte-for-byte today's behavior, and today's behavior never emits
2128 // this section, since it doesn't exist pre-P3). Gated further by
2129 // D-7 itself: `core.skills` requires a read pathway (`read_file` or
2130 // `bash`) — absent either, no section is appended, matching the
2131 // hard-dependency shape `configfile::validate_modules` enforces at
2132 // resolve time.
2133 //
2134 // BP-6 (catalog D2 "Skills (progressive-disclosure packages)"): the
2135 // section is now the discovered SKILL.md INDEX — each package's
2136 // frontmatter `name` and `description`, nothing else. A body is
2137 // never assembled here; it is read on invocation only, which is
2138 // what "progressive disclosure" means. The `[core.prompts]`
2139 // template names keep their own sub-list below it.
2140 let skills = crate::skills::load_for_config(&config);
2141 if config.module_registry && config.skills_enabled {
2142 let has_read_pathway = config
2143 .core_tools_enabled
2144 .iter()
2145 .any(|t| t == "read_file" || t == "bash");
2146 if has_read_pathway {
2147 system.push_str(&skills_prompt_section(&config, &skills));
2148 }
2149 }
2150 // BP-5: the style layer lands LAST, where cc puts it ("output styles
2151 // append custom instructions to the END of the system prompt").
2152 // Empty for a neutral style (`default`/`none`) and for a style that
2153 // already replaced the base above.
2154 if let Some(style) = output_style.as_ref().filter(|s| !s.replaces_base) {
2155 system.push_str(&style.section());
2156 }
2157 // BP-5: the SCOPED rules travel with the tool context, which is
2158 // where a "a tool touched a matching file" event can see them.
2159 ctx.path_rules = std::sync::Arc::new(path_rules);
2160 let history = vec![ChatMessage::system(system)];
2161 // P4e (§1.6/§3.1 `core.session.git_metadata`, catalog:331): captured
2162 // once here, alongside `env_context`'s own git probe — `false` (the
2163 // default) is a no-op, byte-identical to today's behavior.
2164 let git_metadata = if config.session_git_metadata {
2165 crate::git_metadata::capture(&config.cwd, now_ms())
2166 } else {
2167 None
2168 };
2169 // BP-7 (catalog §4a "Named agent definitions as data"): discover
2170 // `<cwd>/.claude/agents/*.md` for EVERY harness that turns the
2171 // subagents module on, not only the Claude emulate/resume path —
2172 // that restriction was the second half of the ledger row's
2173 // residue.
2174 merge_project_agent_definitions(&mut config);
2175 // P5-3: captured before `config` moves into the literal below (a
2176 // `usize` field READ, not a move, but it must happen before the
2177 // `config` shorthand field consumes the binding).
2178 let subagent_depth = config.subagent_depth;
2179 // BP-5 (catalog D2 "Shell-output injection in templates/skills"):
2180 // the authorization every `` !`cmd` `` in a skill/command body is
2181 // evaluated under — this config's own permission rules, resolved
2182 // once. Disabled unless `[core.skills] shell_injection` is on.
2183 let shell_injection = crate::skills::ShellInjection::from_config(&config);
2184 // BP-8 (catalog:151): `[capabilities.session_tree] enabled` finally
2185 // has a reader. An armed tree starts empty and grows one node per
2186 // recorded message — the degenerate single-path case, byte-for-byte
2187 // the same conversation, until a rewind or branch actually forks it.
2188 let session_tree = if config.session_tree_enabled {
2189 Some(supercode_interchange::session_tree::SessionTree::new())
2190 } else {
2191 None
2192 };
2193 // BP-7: resolved once here so the request path never re-does the
2194 // lookup, and so `Self::model_price` is `None` exactly when this
2195 // build cannot price the model.
2196 let model_price = crate::pricing::resolve(
2197 &config.model,
2198 config.price_input_per_mtok,
2199 config.price_output_per_mtok,
2200 );
2201 // BP-10: same reason — built before `config` moves into the
2202 // literal. `Config::permissions_approvals_persist` off (the
2203 // default) makes this the pre-BP-10 in-memory cache and touches no
2204 // filesystem.
2205 let permissions_approval_cache = crate::permissions::cache_for_config(&config);
2206 Agent {
2207 config,
2208 provider: std::sync::Arc::from(provider),
2209 registry,
2210 history,
2211 ctx,
2212 total_output_tokens: 0,
2213 activated_tools: HashSet::new(),
2214 recorder: None,
2215 journal: None,
2216 session_tree,
2217 rewind_undo: Vec::new(),
2218 journaled_plan: Vec::new(),
2219 reduction_policy: None,
2220 reduction_log: ReductionLog::default(),
2221 imported_prefix_len: None,
2222 compacting_manually: false,
2223 env_context_live,
2224 base_prompt_live,
2225 shell_injection,
2226 spliced_context_blocks: Vec::new(),
2227 span_summarizer: None,
2228 last_tool_schema_tier_signature: None,
2229 context_limit: None,
2230 requests_issued: false,
2231 last_cache_activity_ms: None,
2232 cache_established: false,
2233 pending_cache_turn: (false, false, None),
2234 session_titler: None,
2235 usage_log: Vec::new(),
2236 turn_index: 0,
2237 turn_records: Vec::new(),
2238 retry_log: std::sync::Arc::new(crate::provider::RetryLog::default()),
2239 model_price,
2240 total_cost_usd: 0.0,
2241 total_steps: 0,
2242 reaped_subagents: std::collections::HashMap::new(),
2243 goal: None,
2244 steer_queue: std::sync::Arc::new(std::sync::Mutex::new(SteerInbox::default())),
2245 follow_up_queue: std::collections::VecDeque::new(),
2246 doom_loop_last_call: None,
2247 doom_loop_streak: 0,
2248 model_change_log: Vec::new(),
2249 git_metadata,
2250 permissions_approval_cache,
2251 permissions_approval_handler: None,
2252 mcp_prompts: std::collections::HashMap::new(),
2253 skills,
2254 subagent_depth,
2255 subagent_concurrency_gauge: std::sync::Arc::new(std::sync::atomic::AtomicUsize::new(0)),
2256 background_subagents: std::collections::HashMap::new(),
2257 pending_child_approvals: std::sync::Arc::new(std::sync::Mutex::new(Vec::new())),
2258 child_approval_handler_factory: None,
2259 subagent_store: None,
2260 claude_runtime_manifest: None,
2261 background_jobs: std::collections::HashMap::new(),
2262 background_concurrency_gauge: std::sync::Arc::new(std::sync::atomic::AtomicUsize::new(
2263 0,
2264 )),
2265 checkpoint_observer,
2266 lsp_manager,
2267 }
2268 }
2269
2270 /// A handle to this agent's model transport, for sharing with subagents.
2271 pub fn provider_arc(&self) -> std::sync::Arc<dyn Provider> {
2272 self.provider.clone()
2273 }
2274
2275 /// Read-only access to this agent's resolved [`Config`] — e.g. so a
2276 /// caller (`crates/cli`'s `attach_mcp`) can consult
2277 /// [`Config::module_registry`]/[`Config::module_activation`] AFTER
2278 /// construction without having to separately thread the config through
2279 /// every call site that builds an `Agent` and later needs it again.
2280 /// Same trust boundary as every other already-public `Agent` accessor
2281 /// (`history`, `provider_arc`) — the caller is the same process that
2282 /// built this `Config` in the first place, not a new exposure surface.
2283 pub fn config(&self) -> &Config {
2284 &self.config
2285 }
2286
2287 /// P5-9 (§2 module 20 `checkpoint`): this agent's checkpoint engine, if
2288 /// `Config::checkpoint_enabled` is `true` and the shadow store opened
2289 /// successfully — `None` otherwise (the default-off case, or a
2290 /// graceful-degrade after an I/O failure). A caller (CLI/TUI/embedder)
2291 /// uses this to `list`/`turn_diff`/`restore` WITHOUT re-deriving the
2292 /// shadow-store root itself. Deliberately named `checkpoint_observer`,
2293 /// not `checkpoint` — [`Self::checkpoint`] already names the unrelated
2294 /// in-memory conversation-position marker (see that method's doc
2295 /// comment).
2296 pub fn checkpoint_observer(&self) -> Option<&crate::checkpoint::CheckpointObserver> {
2297 self.checkpoint_observer.as_deref()
2298 }
2299
2300 /// P5-11 (§2 module 28 `lsp`): this agent's LSP server registry, if
2301 /// `Config::lsp_enabled` is `true` — `None` otherwise (the default-off
2302 /// case). `impl Drop for Agent` already covers production teardown via
2303 /// [`crate::lsp::LspManager::kill_all_sync`] (a real, group-killing OS
2304 /// process kill — see `crate::lsp`'s module doc). This accessor exists
2305 /// for an OPTIONAL caller (CLI/TUI/embedder) that manages its own
2306 /// `Agent` lifecycle and additionally wants to reach
2307 /// [`crate::lsp::LspManager::shutdown_all`] for a graceful LSP
2308 /// `shutdown`/`exit` handshake BEFORE dropping the agent — nothing
2309 /// calls `shutdown_all` automatically today.
2310 pub fn lsp_manager(&self) -> Option<&crate::lsp::LspManager> {
2311 self.lsp_manager.as_deref()
2312 }
2313
2314 /// Spawn a subagent that shares this agent's model transport, runs `task`
2315 /// to completion with its own fresh conversation (seeded with `system`), and
2316 /// returns its final answer. The analog of `Agent` / `spawn_agent`.
2317 pub async fn run_subagent(
2318 &self,
2319 system: impl Into<String>,
2320 task: impl Into<String>,
2321 ) -> Result<String> {
2322 let mut sub_config = Config::builder()
2323 .model(self.config.model.clone())
2324 .system_prompt(system)
2325 .cwd(self.config.cwd.clone())
2326 .sandbox(self.config.sandbox)
2327 .max_iterations(self.config.max_iterations)
2328 .build();
2329 sub_config.base_url = self.config.base_url.clone();
2330 let mut sub = Agent::with_provider_arc(sub_config, self.provider.clone());
2331 sub.send(task).await
2332 }
2333
2334 /// Like [`Self::with_provider`] but sharing an existing transport handle.
2335 pub fn with_provider_arc(mut config: Config, provider: std::sync::Arc<dyn Provider>) -> Self {
2336 let (ctx, checkpoint_observer, lsp_manager) = build_tool_context(&config);
2337 let history = vec![ChatMessage::system(config.system_prompt.clone())];
2338 // P3 (design §5.2): see the `Self::new` doc note — a no-op when
2339 // `config.module_registry` is off (the default).
2340 let mut registry = ToolRegistry::from_config(&config);
2341 // P5-12: see `Self::with_parts`'s identical call — a no-op when
2342 // `config.plugins_enabled` is `false` (the default).
2343 crate::plugins::register_into(&config, &mut registry);
2344 // P4e: see `Self::with_parts`'s identical capture.
2345 let git_metadata = if config.session_git_metadata {
2346 crate::git_metadata::capture(&config.cwd, now_ms())
2347 } else {
2348 None
2349 };
2350 // BP-7 (catalog §4a "Named agent definitions as data"): discover
2351 // `<cwd>/.claude/agents/*.md` for EVERY harness that turns the
2352 // subagents module on, not only the Claude emulate/resume path —
2353 // that restriction was the second half of the ledger row's
2354 // residue.
2355 merge_project_agent_definitions(&mut config);
2356 // P5-3: captured before `config` moves into the literal below (a
2357 // `usize` field READ, not a move, but it must happen before the
2358 // `config` shorthand field consumes the binding).
2359 let subagent_depth = config.subagent_depth;
2360 // BP-6: this constructor assembles no prompt sections at all (it
2361 // takes `config.system_prompt` verbatim), so there is no skills
2362 // INDEX here — but the discovered set still rides along, so an
2363 // explicit invocation (`/name`, `$slug`, the `skill` tool) resolves
2364 // the same packages the registry's own `skill` tool holds.
2365 let skills = crate::skills::load_for_config(&config);
2366 let base_prompt_live = config.system_prompt.clone();
2367 let shell_injection = crate::skills::ShellInjection::from_config(&config);
2368 // BP-8 (catalog:151): `[capabilities.session_tree] enabled` finally
2369 // has a reader. An armed tree starts empty and grows one node per
2370 // recorded message — the degenerate single-path case, byte-for-byte
2371 // the same conversation, until a rewind or branch actually forks it.
2372 let session_tree = if config.session_tree_enabled {
2373 Some(supercode_interchange::session_tree::SessionTree::new())
2374 } else {
2375 None
2376 };
2377 // BP-7: resolved once here so the request path never re-does the
2378 // lookup, and so `Self::model_price` is `None` exactly when this
2379 // build cannot price the model.
2380 let model_price = crate::pricing::resolve(
2381 &config.model,
2382 config.price_input_per_mtok,
2383 config.price_output_per_mtok,
2384 );
2385 // BP-10: see the sibling constructor — built before `config` moves.
2386 let permissions_approval_cache = crate::permissions::cache_for_config(&config);
2387 Agent {
2388 config,
2389 provider,
2390 registry,
2391 history,
2392 ctx,
2393 total_output_tokens: 0,
2394 activated_tools: HashSet::new(),
2395 recorder: None,
2396 journal: None,
2397 session_tree,
2398 rewind_undo: Vec::new(),
2399 journaled_plan: Vec::new(),
2400 reduction_policy: None,
2401 reduction_log: ReductionLog::default(),
2402 imported_prefix_len: None,
2403 compacting_manually: false,
2404 env_context_live: None,
2405 // BP-5: this constructor assembles no prompt sections (see the
2406 // skills note above) — `config.system_prompt` IS the whole
2407 // system message, so that is what a later `set_model` would
2408 // have to replace.
2409 base_prompt_live,
2410 shell_injection,
2411 spliced_context_blocks: Vec::new(),
2412 span_summarizer: None,
2413 last_tool_schema_tier_signature: None,
2414 context_limit: None,
2415 requests_issued: false,
2416 last_cache_activity_ms: None,
2417 cache_established: false,
2418 pending_cache_turn: (false, false, None),
2419 session_titler: None,
2420 usage_log: Vec::new(),
2421 turn_index: 0,
2422 turn_records: Vec::new(),
2423 retry_log: std::sync::Arc::new(crate::provider::RetryLog::default()),
2424 model_price,
2425 total_cost_usd: 0.0,
2426 total_steps: 0,
2427 reaped_subagents: std::collections::HashMap::new(),
2428 goal: None,
2429 steer_queue: std::sync::Arc::new(std::sync::Mutex::new(SteerInbox::default())),
2430 follow_up_queue: std::collections::VecDeque::new(),
2431 doom_loop_last_call: None,
2432 doom_loop_streak: 0,
2433 model_change_log: Vec::new(),
2434 git_metadata,
2435 permissions_approval_cache,
2436 permissions_approval_handler: None,
2437 mcp_prompts: std::collections::HashMap::new(),
2438 skills,
2439 subagent_depth,
2440 subagent_concurrency_gauge: std::sync::Arc::new(std::sync::atomic::AtomicUsize::new(0)),
2441 background_subagents: std::collections::HashMap::new(),
2442 pending_child_approvals: std::sync::Arc::new(std::sync::Mutex::new(Vec::new())),
2443 child_approval_handler_factory: None,
2444 subagent_store: None,
2445 claude_runtime_manifest: None,
2446 background_jobs: std::collections::HashMap::new(),
2447 background_concurrency_gauge: std::sync::Arc::new(std::sync::atomic::AtomicUsize::new(
2448 0,
2449 )),
2450 checkpoint_observer,
2451 lsp_manager,
2452 }
2453 }
2454
2455 /// Run a prompt on a background task, returning a handle that resolves to
2456 /// the final answer (and the agent, so the caller can continue it). The
2457 /// analog of background/async agent runs.
2458 pub fn run_in_background(
2459 mut self,
2460 prompt: impl Into<String>,
2461 ) -> tokio::task::JoinHandle<(Self, Result<String>)>
2462 where
2463 Self: Send + 'static,
2464 {
2465 let prompt = prompt.into();
2466 tokio::spawn(async move {
2467 let result = self.send(prompt).await;
2468 (self, result)
2469 })
2470 }
2471
2472 /// Build an agent and seed it with a previously-recorded [`Session`] so it
2473 /// can continue where Claude Code or Codex left off.
2474 pub fn resume(config: Config, session: Session) -> Result<Self> {
2475 let mut agent = Agent::new(config)?;
2476 agent.load_session(session);
2477 Ok(agent)
2478 }
2479
2480 /// Like [`Self::resume`], but also begins recording (A2/A3): a fresh
2481 /// native-v2 sidecar is created at `sidecar_path` from `session` (header +
2482 /// `session.raw` verbatim — the imported prefix's own fidelity), and every
2483 /// subsequent turn this agent produces is appended to it at full fidelity,
2484 /// independent of whatever `cap_tool_output`/`maybe_compact` (D6) do to
2485 /// `history`.
2486 ///
2487 /// Invariant this establishes ONLY once [`Self::set_reduction_policy`] is
2488 /// also called (the D6/A7 supersession gate, `Self::run_loop`): at any
2489 /// instant, `Session::from_native_str(sidecar).messages` equals
2490 /// `session.messages` (the imported prefix) followed by every message
2491 /// appended since — i.e. `self.history()[1..]` (`history[0]` is this
2492 /// agent's own system prompt, per [`Self::load_session`]; it is never
2493 /// part of `session` and is never written to the sidecar). Recording
2494 /// alone (no policy) leaves the gate off: `cap_tool_output` still runs on
2495 /// oversized tool results, and `history` can diverge from the sidecar for
2496 /// them — honestly, via the notice's "full output in session sidecar"
2497 /// label, never silently.
2498 pub fn resume_recorded(
2499 config: Config,
2500 session: Session,
2501 sidecar_path: &std::path::Path,
2502 ) -> Result<Self> {
2503 let mut agent = Agent::new(config)?;
2504 let recorder = SidecarWriter::create(sidecar_path, &session)?;
2505 agent.load_session(session);
2506 agent.recorder = Some(recorder);
2507 Ok(agent)
2508 }
2509
2510 /// Install (or replace) this agent's sidecar recorder (A3).
2511 pub fn set_recorder(&mut self, w: SidecarWriter) {
2512 self.recorder = Some(w);
2513 }
2514
2515 // ---- BP-8: the append-only journal (catalog:150/152/154/156) --------
2516
2517 /// Install (or replace) this agent's append-only session journal — the
2518 /// durable, flush-per-record log of every message it produces plus
2519 /// every queue/rewind/plan operation performed on it. Installing one
2520 /// alone changes nothing about the conversation; it only makes the
2521 /// session survive a crash mid-turn.
2522 pub fn set_journal(&mut self, journal: crate::session_journal::SessionJournal) {
2523 self.journal = Some(std::sync::Arc::new(std::sync::Mutex::new(journal)));
2524 }
2525
2526 /// Whether an append-only journal is installed.
2527 pub fn has_journal(&self) -> bool {
2528 self.journal.is_some()
2529 }
2530
2531 /// Append one operation to the journal, if installed. Best-effort by
2532 /// design: losing a durability record must never fail the turn it
2533 /// describes, so the failure is logged and the loop continues — the
2534 /// same contract the compaction-marker `record` call keeps.
2535 fn journal_op(&self, op: crate::session_journal::JournalOp) {
2536 let Some(journal) = &self.journal else { return };
2537 let mut guard = journal
2538 .lock()
2539 .unwrap_or_else(std::sync::PoisonError::into_inner);
2540 if let Err(error) = guard.append(op) {
2541 tracing::warn!("failed to append a session-journal record: {error}");
2542 }
2543 }
2544
2545 /// Declare the durable view caught up: `<name>.jsonl` now holds
2546 /// `messages` messages and every journal record before this point is
2547 /// already in it. Everything journaled AFTER the last such record is
2548 /// exactly what a crash would have lost — see
2549 /// [`crate::session_journal::JournalState::unpersisted`].
2550 pub fn journal_checkpoint(&self, messages: usize) {
2551 self.journal_op(crate::session_journal::JournalOp::Checkpoint { messages });
2552 }
2553
2554 /// BP-13: record one per-turn usage entry in the append-only journal.
2555 pub fn journal_usage(&self, record: &crate::usage_log::UsageRecord) {
2556 self.journal_op(crate::session_journal::JournalOp::Usage {
2557 record: record.clone(),
2558 });
2559 }
2560
2561 /// BP-13: record one mid-session model change in the append-only
2562 /// journal — the ONE persisted home for a routing record (BP-8's
2563 /// journal), never a second file.
2564 pub fn journal_model_change(&self, record: &crate::model_change::ModelChangeRecord) {
2565 self.journal_op(crate::session_journal::JournalOp::ModelChange {
2566 record: record.clone(),
2567 });
2568 }
2569
2570 /// BP-8 (catalog:151): this session's conversation tree, when the
2571 /// module is on.
2572 pub fn session_tree(&self) -> Option<&supercode_interchange::session_tree::SessionTree> {
2573 self.session_tree.as_ref()
2574 }
2575
2576 /// Install a tree loaded from the store (a resume), replacing whatever
2577 /// this agent built. A no-op when the module is off — a session whose
2578 /// preset does not enable `session_tree` must not acquire one through
2579 /// the back door of an old sidecar.
2580 pub fn set_session_tree(&mut self, tree: supercode_interchange::session_tree::SessionTree) {
2581 if self.config.session_tree_enabled {
2582 self.session_tree = Some(tree);
2583 }
2584 }
2585
2586 /// BP-8: rebuild the tree from the current linear history — used after
2587 /// a resume that loaded a transcript but had no `.tree.json` to restore
2588 /// (every session recorded before the module was on).
2589 pub fn rebuild_session_tree_from_history(&mut self) {
2590 if !self.config.session_tree_enabled {
2591 return;
2592 }
2593 let linear: Vec<ChatMessage> = self.history.iter().skip(1).cloned().collect();
2594 self.session_tree =
2595 Some(supercode_interchange::session_tree::SessionTree::from_linear(&linear, now_ms()));
2596 }
2597
2598 /// BP-8 (catalog:152 "Rewind/rollback conversation"): move THIS
2599 /// conversation back to an earlier point — the whole row, not the
2600 /// last-exchange special case [`Self::rewind_to`] serves and not
2601 /// `sessions fork --at`, which makes a different session.
2602 ///
2603 /// `keep` is a message count (index into `history`), so `keep = 1`
2604 /// leaves only the system message. Three things happen, in this order:
2605 ///
2606 /// 1. the removed tail is pushed onto an undo stack, so
2607 /// [`Self::undo_rewind`] can put it back;
2608 /// 2. a [`crate::session_journal::JournalOp::Rewind`] record is
2609 /// APPENDED — nothing is deleted from disk, so the rewound-away
2610 /// messages remain recoverable from the log;
2611 /// 3. when the tree module is on, the active branch's leaf moves to the
2612 /// node at `keep`, and the old leaf is preserved under a fresh
2613 /// sibling branch — the next message appended forks there rather
2614 /// than overwriting.
2615 ///
2616 /// Returns what it did. Rewinding to a point at or past the end is a
2617 /// no-op with `removed = 0`, never an error.
2618 pub fn rewind_conversation(&mut self, keep: usize) -> RewindOutcome {
2619 let keep = keep.max(1).min(self.history.len());
2620 let removed: Vec<ChatMessage> = self.history.split_off(keep);
2621 if removed.is_empty() {
2622 return RewindOutcome {
2623 kept: self.history.len(),
2624 removed: 0,
2625 preserved_branch: None,
2626 };
2627 }
2628 let removed_count = removed.len();
2629 self.rewind_undo.push(removed);
2630 // `keep` counts the system message; the journal records only
2631 // `history[1..]`, so its own view is one shorter.
2632 self.journal_op(crate::session_journal::JournalOp::Rewind { to: keep - 1 });
2633 let preserved_branch = self.session_tree.as_mut().and_then(|tree| {
2634 let path = tree.active_path().unwrap_or_default();
2635 // `keep - 1` messages remain after the system message, so the
2636 // new leaf is the node at index `keep - 2`.
2637 match keep.checked_sub(2).and_then(|i| path.get(i).cloned()) {
2638 Some(node) => tree.rewind(&node, now_ms()).ok().flatten(),
2639 None => None,
2640 }
2641 });
2642 RewindOutcome {
2643 kept: self.history.len(),
2644 removed: removed_count,
2645 preserved_branch,
2646 }
2647 }
2648
2649 /// BP-8: invert the most recent [`Self::rewind_conversation`] — the
2650 /// messages come back, and the inversion is itself an appended journal
2651 /// record. `false` when there is nothing to undo.
2652 pub fn undo_rewind(&mut self) -> bool {
2653 let Some(mut tail) = self.rewind_undo.pop() else {
2654 return false;
2655 };
2656 self.history.append(&mut tail);
2657 self.journal_op(crate::session_journal::JournalOp::Unrewind);
2658 if self.config.session_tree_enabled {
2659 self.rebuild_session_tree_from_history();
2660 }
2661 true
2662 }
2663
2664 /// BP-8 (catalog:150): append messages recovered from the journal
2665 /// after a crash — they were already recorded, so this deliberately
2666 /// does NOT re-journal them; it puts the live conversation back where
2667 /// the interrupted process left it.
2668 pub fn append_recovered_messages(&mut self, messages: &[ChatMessage]) {
2669 for msg in messages {
2670 if let Some(tree) = self.session_tree.as_mut() {
2671 tree.append_message(msg.clone(), now_ms());
2672 }
2673 self.history.push(msg.clone());
2674 }
2675 }
2676
2677 /// BP-8: how many rewinds are currently undoable.
2678 pub fn undoable_rewinds(&self) -> usize {
2679 self.rewind_undo.len()
2680 }
2681
2682 /// BP-8: restore the undo stack a previous process left in the journal,
2683 /// so `/rewind undo` works across a restart.
2684 pub fn restore_rewind_undo(&mut self, stack: Vec<Vec<ChatMessage>>) {
2685 self.rewind_undo = stack;
2686 }
2687
2688 /// BP-8 (catalog:154 "Queued-prompt persistence"): re-queue pending
2689 /// inputs recovered from the journal WITHOUT re-recording them — they
2690 /// are already in the log, and journaling them again would double them
2691 /// on the next restart.
2692 pub fn restore_queues(&mut self, steer: &[String], follow_up: &[String]) {
2693 for message in steer {
2694 self.steer_queue
2695 .lock()
2696 .unwrap_or_else(std::sync::PoisonError::into_inner)
2697 .queue_unchecked(message.clone());
2698 }
2699 for message in follow_up {
2700 self.follow_up_queue.push_back(message.clone());
2701 }
2702 }
2703
2704 /// BP-8 (catalog:156 "Todos/plan persisted per session"): the session's
2705 /// current `update_plan` checklist.
2706 pub fn plan(&self) -> Vec<crate::session_journal::PlanEntry> {
2707 self.ctx.plan_snapshot()
2708 }
2709
2710 /// BP-8: restore a plan read back from the store on resume. Marked as
2711 /// already-journaled, so a resume that changes nothing writes nothing.
2712 pub fn set_plan(&mut self, steps: Vec<crate::session_journal::PlanEntry>) {
2713 self.ctx.set_plan(steps.clone());
2714 self.journaled_plan = steps;
2715 }
2716
2717 /// BP-8 (catalog:154): record that `count` pending inputs left `queue`
2718 /// and became conversation. A no-op when nothing was taken, or when
2719 /// queue persistence is off.
2720 fn journal_queue_drain(&self, queue: crate::session_journal::QueueKind, count: usize) {
2721 if count == 0 || !self.config.session_queue_persist {
2722 return;
2723 }
2724 self.journal_op(crate::session_journal::JournalOp::Dequeue { queue, count });
2725 }
2726
2727 /// BP-8: journal the plan if `update_plan` changed it since the last
2728 /// time this ran. Called at every loop boundary — a plan that a crash
2729 /// would otherwise strand in the tool's memory is on disk within one
2730 /// iteration of being written.
2731 fn journal_plan_if_changed(&mut self) {
2732 if !self.config.todos_persist {
2733 return;
2734 }
2735 let current = self.ctx.plan_snapshot();
2736 if current == self.journaled_plan {
2737 return;
2738 }
2739 self.journaled_plan.clone_from(¤t);
2740 self.journal_op(crate::session_journal::JournalOp::Plan { steps: current });
2741 }
2742
2743 /// Install (or replace) this agent's reduction policy (A5/A7/A10). Once
2744 /// set, every provider request is built from a *projected* view of
2745 /// `history[1..]` (`reduce::project_messages`) rather than `history`
2746 /// verbatim — `history` itself is never shrunk or mutated by this; only
2747 /// the request view does.
2748 pub fn set_reduction_policy(&mut self, policy: ReductionPolicy) {
2749 self.reduction_policy = Some(policy);
2750 }
2751
2752 /// This agent's reduction policy, if one is installed.
2753 pub fn reduction_policy(&self) -> Option<&ReductionPolicy> {
2754 self.reduction_policy.as_ref()
2755 }
2756
2757 /// Change the global tool-schema tier (TR-8/T5) mid-session. Takes effect
2758 /// starting with the NEXT request this agent builds. Under
2759 /// [`CachePlan::ImportedPrefix`], the first request built after a change
2760 /// is flagged as a cache-bust event and its cache-control annotation is
2761 /// skipped for that one request (see [`provider::tier_change_is_cache_bust`],
2762 /// consulted in `Self::build_request_messages`) — normal annotation
2763 /// resumes on the next request if the tier doesn't change again.
2764 pub fn set_schema_tier(&mut self, tier: crate::tools::SchemaTier) {
2765 self.config.tool_schema_tier = tier;
2766 }
2767
2768 /// Override the schema tier for a single tool (TR-8/T5) mid-session, same
2769 /// cache-bust interaction as [`Self::set_schema_tier`].
2770 pub fn set_tool_schema_tier(
2771 &mut self,
2772 name: impl Into<String>,
2773 tier: crate::tools::SchemaTier,
2774 ) {
2775 self.config
2776 .tool_overrides
2777 .entry(name.into())
2778 .or_default()
2779 .schema_tier = Some(tier);
2780 }
2781
2782 /// A deterministic fingerprint of the current tool-schema tier
2783 /// configuration (global knob + every per-tool override), used to detect
2784 /// a mid-session tier change (TR-8/T5, dev/05). Order-independent over
2785 /// `tool_overrides` (sorted by name before hashing) so insertion order
2786 /// never spuriously changes the signature.
2787 fn schema_tier_signature(&self) -> u64 {
2788 use std::hash::{Hash, Hasher};
2789 let mut hasher = std::collections::hash_map::DefaultHasher::new();
2790 self.config.tool_schema_tier.hash(&mut hasher);
2791 let mut overrides: Vec<(&str, crate::tools::SchemaTier)> = self
2792 .config
2793 .tool_overrides
2794 .iter()
2795 .filter_map(|(name, o)| o.schema_tier.map(|t| (name.as_str(), t)))
2796 .collect();
2797 overrides.sort_by_key(|(name, _)| *name);
2798 for (name, tier) in overrides {
2799 name.hash(&mut hasher);
2800 tier.hash(&mut hasher);
2801 }
2802 hasher.finish()
2803 }
2804
2805 /// Install (or replace) this agent's TR-7 span summarizer — the
2806 /// injectable side-call `Self::build_request_messages` uses to turn an
2807 /// A10 `TurnsCleared` span into an LLM-written summary paragraph when
2808 /// `policy.summarize_cleared_turns` is on. Installing one alone changes
2809 /// nothing: [`ReductionPolicy::summarize_cleared_turns`] (off by
2810 /// default) is the actual gate, so tests/callers that want the
2811 /// deterministic stub can simply never call this.
2812 pub fn set_span_summarizer(
2813 &mut self,
2814 summarizer: impl reduce::summarize::SpanSummarizer + Send + Sync + 'static,
2815 ) {
2816 self.span_summarizer = Some(std::sync::Arc::new(summarizer));
2817 }
2818
2819 /// Install an already-shared summarizer — same seam as
2820 /// [`Self::set_span_summarizer`], for callers (and tests) that need to
2821 /// keep their own handle on it.
2822 pub fn set_span_summarizer_arc(
2823 &mut self,
2824 summarizer: std::sync::Arc<dyn reduce::summarize::SpanSummarizer + Send + Sync>,
2825 ) {
2826 self.span_summarizer = Some(summarizer);
2827 }
2828
2829 /// Prepare TR-7 metadata with this agent's installed summarizer for a
2830 /// projection performed by an outer driver before session history/log
2831 /// are loaded (the CLI foreign-resume preflight). `None` preserves the
2832 /// deterministic fallback when the gate is off, no summarizer exists,
2833 /// the span is below the cost floor, or the side-call fails.
2834 pub fn prepare_cleared_turns_summary(
2835 &self,
2836 msgs: &[ChatMessage],
2837 policy: &ReductionPolicy,
2838 prior: &ReductionLog,
2839 ) -> Option<reduce::PreparedClearSummary> {
2840 let summarizer = self.span_summarizer.as_deref()?;
2841 reduce::prepare_cleared_turns_summary(msgs, policy, prior, summarizer)
2842 }
2843
2844 /// P5-4: install (or replace) this agent's [`crate::EventSink`] AFTER
2845 /// construction — `Config::event_sink` is otherwise only set at
2846 /// `Config`-build time (before `Agent::new`), which is too early for a
2847 /// `tui` embedder that only knows it's activating (and needs to
2848 /// replace whatever print-mode/REPL sink was already installed with
2849 /// one that feeds its own render loop instead of writing straight to
2850 /// stdout) once it already holds a live `Agent`. Mirrors [`Self::
2851 /// set_permissions_approval_handler`]'s "installing one alone changes
2852 /// nothing beyond what already consults `Config::event_sink`" pattern
2853 /// — this is a plain replacement, not a new activation gate.
2854 pub fn set_event_sink(&mut self, sink: crate::EventSink) {
2855 self.config.event_sink = Some(sink);
2856 }
2857
2858 /// P5-1: install (or replace) this agent's permissions-engine approval
2859 /// handler — see [`crate::permissions::PermissionsApprovalHandler`].
2860 /// This is the non-interactive decision seam a CLI/TUI/SDK embedder
2861 /// implements for the `Ask`-tier prompt; the TUI's actual interactive
2862 /// UI is a separate module (P5 row 4), not built here. Installing one
2863 /// alone changes nothing: [`Config::permissions_enabled`] (off by
2864 /// default) is the actual gate — with no handler installed, every
2865 /// `Ask`-tier decision denies (fail-closed, see that trait's doc
2866 /// comment).
2867 ///
2868 /// P5-10 (§2 module 12, `escalation = "ask"`): the SAME handler also
2869 /// backs a sandbox-unenforceable `ask` decision
2870 /// (`crate::sandbox::decide_fs`'s `approval` parameter) — one installed
2871 /// seam serves both `permissions.rules`' `Ask` tier and
2872 /// `permissions.sandbox`'s `escalation = "ask"`, rather than requiring
2873 /// an embedder to install two near-identical handlers. Kept in sync on
2874 /// `self.ctx` (not just `self.permissions_approval_handler`) because
2875 /// `BashTool::execute`/`PersistentShellTool::execute` only ever see
2876 /// `&ToolContext`, never `&Agent` — see `ToolContext::
2877 /// sandbox_approval_handler`'s doc comment.
2878 /// BP-10 (catalog row "Session approval caching"): this agent's
2879 /// approval cache — the door an embedder/TUI uses to inspect or REVOKE
2880 /// remembered grants (`ApprovalCache::clear` forgets every one, in
2881 /// memory and on disk, and the next matching call asks again). Also
2882 /// how a test proves a grant really did survive the process:
2883 /// `store_path()` names the file a second agent reads back.
2884 pub fn permissions_approval_cache(&self) -> &crate::permissions::ApprovalCache {
2885 &self.permissions_approval_cache
2886 }
2887
2888 pub fn set_permissions_approval_handler(
2889 &mut self,
2890 handler: impl crate::permissions::PermissionsApprovalHandler + 'static,
2891 ) {
2892 let handler: std::sync::Arc<dyn crate::permissions::PermissionsApprovalHandler> =
2893 std::sync::Arc::new(handler);
2894 self.permissions_approval_handler = Some(handler.clone());
2895 self.ctx.sandbox_approval_handler =
2896 Some(crate::sandbox::SandboxApprovalHandler(handler.clone()));
2897 // BP-3 (§2 module 8): `exit_plan_mode` presents its plan on this
2898 // same door — one approval seam for the session, not a second one
2899 // the operator would have to answer separately.
2900 self.ctx.approval_handler = Some(crate::tools::ToolApprovalHandler(handler));
2901 }
2902
2903 /// BP-3 (§2 module 6 `tools.question`): install the door `ask_user`
2904 /// asks the human through — the `elicitation/create` handler the design
2905 /// names as the module's protocol side. SDK-owned frontends pass the
2906 /// broker-backed handler (`crate::server::FrontendRequestBridge::
2907 /// elicitation_handler`), which is what makes the question a real
2908 /// frontend request the turn waits on. `None` (the default, nothing
2909 /// installed) leaves the tool deny-default: it reports that nobody can
2910 /// be asked instead of blocking.
2911 ///
2912 /// Installing one alone changes nothing about whether the tool EXISTS —
2913 /// `[capabilities.tools_question]` is that gate, applied by
2914 /// `crate::tools::ToolRegistry::from_config`.
2915 pub fn set_user_question_handler(
2916 &mut self,
2917 handler: std::sync::Arc<dyn crate::mcp::McpElicitationHandler>,
2918 ) {
2919 self.ctx.question_handler = Some(crate::tools::UserQuestionHandler(handler));
2920 }
2921
2922 /// BP-3 (§2 module 8): the shared plan-mode state, so a frontend (the
2923 /// REPL's `/plan`, a TUI toggle) can enter or leave the read-only
2924 /// research phase the same tools and permission gate see.
2925 pub fn plan_mode(&self) -> &std::sync::Arc<crate::tools::PlanModeState> {
2926 &self.ctx.plan_mode
2927 }
2928
2929 /// Install the compatibility approval seam used when the composable
2930 /// permissions engine is disabled. SDK-owned interactive frontends call
2931 /// this alongside [`Self::set_permissions_approval_handler`] so the same
2932 /// authenticated request channel works under either policy engine; the
2933 /// selected engine remains entirely a configuration decision.
2934 pub fn set_legacy_approval_handler(&mut self, handler: crate::config::ApprovalHandler) {
2935 self.config.approval_handler = Some(handler);
2936 }
2937
2938 /// P5-3 (§2 module 9 D5 "subagent transcripts… persisted + linked"):
2939 /// install a [`crate::store::SessionStore`] (+ this agent's own session
2940 /// name in it) so `spawn_subagent` persists each child's transcript
2941 /// (via [`crate::store::SessionStore::save_subagent_transcript`]) and
2942 /// lineage record (via
2943 /// [`crate::store::SessionStore::save_subagent_lineage`]) once the
2944 /// child finishes. Installing one alone changes nothing about whether
2945 /// spawning WORKS — [`Config::subagents_enabled`] is the actual gate;
2946 /// this only controls whether a completed spawn's transcript additionally
2947 /// lands on disk.
2948 pub fn set_subagent_store(
2949 &mut self,
2950 store: std::sync::Arc<crate::store::SessionStore>,
2951 session_name: impl Into<String>,
2952 ) {
2953 self.subagent_store = Some((store, session_name.into()));
2954 }
2955
2956 /// Seed the Claude runtime manifest reconstructed during resume.
2957 ///
2958 /// Installing state enables the matching Claude runtime tool schemas so
2959 /// a disk-reloaded continuation does not lose that vocabulary, but never
2960 /// starts a timer by itself. The supplied execution posture is preserved:
2961 /// an embedding scheduler may deliberately activate before installing it.
2962 pub fn set_claude_runtime_manifest(
2963 &mut self,
2964 manifest: crate::claude_runtime_state::ClaudeRuntimeManifest,
2965 ) {
2966 // A persisted manifest is itself the compatibility capability marker.
2967 // Reopening a Supercode session must not retain its timers while
2968 // silently dropping Claude's Cron*/ScheduleWakeup vocabulary.
2969 self.config.claude_runtime_tools_enabled = true;
2970 self.claude_runtime_manifest = Some(manifest);
2971 }
2972
2973 /// Reinstall project-scoped Claude named-agent definitions when a
2974 /// Supercode continuation carrying a Claude runtime manifest is reopened
2975 /// from disk. The manifest is the durable capability marker; definitions
2976 /// themselves remain authoritative in `<cwd>/.claude/agents/*.md`.
2977 pub fn restore_claude_project_agents(&mut self) -> Result<usize> {
2978 let definitions = crate::claude_compat::load_project_agents(&self.config.cwd)?;
2979 crate::claude_compat::enable_claude_subagent_compatibility(&mut self.config);
2980 for imported in &definitions {
2981 self.config.subagents_definitions.insert(
2982 imported.definition.name.clone(),
2983 imported.definition.clone(),
2984 );
2985 }
2986 Ok(definitions.len())
2987 }
2988
2989 /// Current imported Claude runtime state, including paused mutations made
2990 /// by `Cron*`/`ScheduleWakeup`, for persistence by the embedding loop.
2991 pub fn claude_runtime_manifest(
2992 &self,
2993 ) -> Option<&crate::claude_runtime_state::ClaudeRuntimeManifest> {
2994 self.claude_runtime_manifest.as_ref()
2995 }
2996
2997 /// Mutable access for an embedding scheduler driver to atomically claim
2998 /// due events and persist the resulting manifest. Merely borrowing this
2999 /// state does not start a timer; execution remains the driver's explicit
3000 /// responsibility.
3001 pub fn claude_runtime_manifest_mut(
3002 &mut self,
3003 ) -> Option<&mut crate::claude_runtime_state::ClaudeRuntimeManifest> {
3004 self.claude_runtime_manifest.as_mut()
3005 }
3006
3007 /// P5-4 (tui, closes the P5-3 §2.2 C6 deferred chain): install a
3008 /// factory this agent's `Self::run_spawn_subagent` calls (with the
3009 /// fresh child's own id and this agent's shared
3010 /// [`Self::pending_child_approvals`] queue) to build the
3011 /// `PermissionsApprovalHandler` a `background_prompts = "parent"`
3012 /// child gets, INSTEAD of the default
3013 /// [`crate::subagents::ParentQueueApprovalHandler`]. Installing one
3014 /// alone changes nothing about whether background spawning works —
3015 /// [`Config::subagents_background_prompts`] being
3016 /// [`crate::subagents::BackgroundPromptsPolicy::Parent`] is the actual
3017 /// gate that reaches this factory at all; a `Parent`-policy child
3018 /// spawned before this is installed (or on an agent that never installs
3019 /// it) still gets the immediate-deny default, unchanged.
3020 ///
3021 /// **Security note.** The factory only controls WHICH handler answers
3022 /// an `Ask`-tier request — it can never widen what gets asked in the
3023 /// first place: [`crate::permissions::approval::resolve_ask`] only
3024 /// calls a handler's `ask` when the rule engine has already resolved
3025 /// the call to `Ask` (`Deny` short-circuits before any handler is
3026 /// consulted; `Allow` never needs one), so a parent's "allow" answer
3027 /// here can only grant what the policy already routed to a prompt —
3028 /// never override a `Deny` the engine already decided.
3029 pub fn set_child_approval_handler_factory(
3030 &mut self,
3031 factory: impl Fn(
3032 String,
3033 std::sync::Arc<std::sync::Mutex<Vec<crate::subagents::QueuedApproval>>>,
3034 ) -> std::sync::Arc<dyn crate::permissions::PermissionsApprovalHandler>
3035 + Send
3036 + Sync
3037 + 'static,
3038 ) {
3039 self.child_approval_handler_factory = Some(std::sync::Arc::new(factory));
3040 }
3041
3042 /// P5-3 (§2.2 C6 "parent-surfaced queue"): every approval request a
3043 /// `background_prompts = "parent"` child has raised so far, oldest
3044 /// first — a read-only audit view, not a mutable queue the caller
3045 /// answers. Under P5-3's own default handler (no P5-4 TUI factory
3046 /// installed) every entry here WAS already resolved `Deny` (a
3047 /// background call can't wait for an answer with no handler
3048 /// installed) — but once a `crate::tui::TuiChildApprovalHandler`
3049 /// factory is installed (P5-4,
3050 /// [`Self::set_child_approval_handler_factory`]), the underlying call
3051 /// genuinely blocks and may resolve `Allow`/`AllowForSession`; this
3052 /// method still records the SAME entry for the audit trail either
3053 /// way, so "queued here" no longer implies "was denied" in general —
3054 /// see [`crate::subagents::QueuedApproval`]'s doc comment.
3055 pub fn pending_child_approvals(&self) -> Vec<crate::subagents::QueuedApproval> {
3056 self.pending_child_approvals
3057 .lock()
3058 .map(|q| q.clone())
3059 .unwrap_or_default()
3060 }
3061
3062 /// P4b: install (or replace) this agent's auto-title side-call — see
3063 /// [`crate::session_title::SessionTitler`]. Installing one alone changes
3064 /// nothing: [`Config::auto_title`] (off by default) is the actual gate a
3065 /// caller should consult before calling [`Self::auto_title`].
3066 pub fn set_session_titler(
3067 &mut self,
3068 titler: impl crate::session_title::SessionTitler + Send + Sync + 'static,
3069 ) {
3070 self.session_titler = Some(std::sync::Arc::new(titler));
3071 }
3072
3073 /// P4b: produce a title for this agent's current conversation via the
3074 /// installed [`Self::set_session_titler`] side-call. Returns `None` (never
3075 /// panics, never blocks longer than the titler itself does) if no
3076 /// titler is installed, or the side-call itself declined (see
3077 /// [`crate::session_title::auto_title`]). Does NOT consult
3078 /// [`Config::auto_title`] itself — that gate is the caller's
3079 /// responsibility, matching `Self::span_summarizer`'s precedent of
3080 /// keeping the mechanism and the policy gate separate.
3081 pub fn auto_title(&self) -> Option<String> {
3082 let titler = self.session_titler.as_deref()?;
3083 crate::session_title::auto_title(&self.history, titler)
3084 }
3085
3086 /// P4b (§1.6, catalog §4a "persisted per-turn usage records"): every
3087 /// [`crate::usage_log::UsageRecord`] this agent has accumulated so far.
3088 pub fn usage_records(&self) -> &[crate::usage_log::UsageRecord] {
3089 &self.usage_log
3090 }
3091
3092 /// P4b: persist this agent's accumulated usage log to `store` under
3093 /// `name` — a thin wrapper over [`crate::store::SessionStore::save_usage_log`]
3094 /// so callers don't need to import both types.
3095 pub fn save_usage_log(&self, store: &crate::store::SessionStore, name: &str) -> Result<()> {
3096 store.save_usage_log(name, &self.usage_log)
3097 }
3098
3099 /// BP-7 (catalog §4a "Turn/step bracketing records"): every
3100 /// [`crate::turn_record::TurnRecord`] this agent has accumulated —
3101 /// the context/usage/finish brackets of each model round-trip plus the
3102 /// retry, abort, effort and goal markers between them.
3103 pub fn turn_records(&self) -> &[crate::turn_record::TurnRecord] {
3104 &self.turn_records
3105 }
3106
3107 /// BP-7: persist the marker log to `store` under `name`
3108 /// (`<name>.events.jsonl`), the same thin-wrapper shape
3109 /// [`Self::save_usage_log`] has.
3110 pub fn save_turn_records(&self, store: &crate::store::SessionStore, name: &str) -> Result<()> {
3111 store.save_turn_records(name, &self.turn_records)
3112 }
3113
3114 /// BP-7 (catalog §4a "Per-turn cost/usage accounting"): dollars this
3115 /// agent has spent so far. `0.0` when the model is unpriceable — read
3116 /// [`Self::model_priced`] to tell "free" from "unknown".
3117 pub fn total_cost_usd(&self) -> f64 {
3118 self.total_cost_usd
3119 }
3120
3121 /// BP-7: whether this build can price this agent's model, i.e. whether
3122 /// [`Self::total_cost_usd`] is a real figure rather than a floor.
3123 pub fn model_priced(&self) -> bool {
3124 self.model_price.is_some()
3125 }
3126
3127 /// BP-7 (catalog §4a "Turn/budget caps"): tool calls this agent has
3128 /// executed so far — the counter [`Config::max_steps`] bounds.
3129 pub fn total_steps(&self) -> usize {
3130 self.total_steps
3131 }
3132
3133 /// BP-7 (catalog §4a "Interrupt/abort with state preserved"): record
3134 /// that the in-flight turn was interrupted.
3135 ///
3136 /// Called by whoever owns the cancellation (the CLI's Ctrl-C race), NOT
3137 /// by the loop itself: a cancelled `send` future is dropped mid-await,
3138 /// so the loop never runs another line. The partial work already
3139 /// appended to the transcript stands; this marker is what makes the
3140 /// interruption a persisted FACT — the residue the ledger row named —
3141 /// rather than something a reader has to infer from a dangling tool
3142 /// call on reload. Emits [`AgentEvent::TurnAborted`] as the live
3143 /// counterpart.
3144 pub fn note_abort(&mut self, source: &str) {
3145 let messages = self.history.len();
3146 self.emit(AgentEvent::TurnAborted {
3147 source: source.to_string(),
3148 });
3149 self.push_turn_marker(crate::turn_record::TurnMarker::Aborted {
3150 source: source.to_string(),
3151 messages,
3152 });
3153 }
3154
3155 // ---- BP-7: goals (catalog §4a "Goals — persistent objective across
3156 // turns"; §2 module 7 `todos`, §3.1 `capabilities.todos.goals`) ----
3157
3158 /// Set (or revise) this session's standing objective.
3159 ///
3160 /// Returns `false`, changing nothing, when `capabilities.todos.goals`
3161 /// is off — the module gate, not a silent success. A goal restates
3162 /// itself at the tail of every request until [`Self::clear_goal`], and
3163 /// each change appends a `goal` marker to the turn-record log.
3164 pub fn set_goal(&mut self, objective: impl Into<String>) -> bool {
3165 if !self.config.goals_enabled {
3166 return false;
3167 }
3168 let objective = objective.into();
3169 let now = now_ms();
3170 match &mut self.goal {
3171 Some(goal) => goal.revise(objective.clone(), now),
3172 slot @ None => *slot = Some(crate::goals::GoalRecord::new(objective.clone(), now)),
3173 }
3174 self.push_turn_marker(crate::turn_record::TurnMarker::Goal { objective });
3175 true
3176 }
3177
3178 /// This session's standing objective, if one is set.
3179 pub fn goal(&self) -> Option<&crate::goals::GoalRecord> {
3180 self.goal.as_ref()
3181 }
3182
3183 /// Drop the standing objective. `true` when there was one to drop.
3184 pub fn clear_goal(&mut self) -> bool {
3185 if self.goal.take().is_none() {
3186 return false;
3187 }
3188 self.push_turn_marker(crate::turn_record::TurnMarker::Goal {
3189 objective: String::new(),
3190 });
3191 true
3192 }
3193
3194 /// BP-7: persist (or, when cleared, remove) the standing objective
3195 /// beside the session — same thin-wrapper shape as
3196 /// [`Self::save_usage_log`].
3197 pub fn save_goal(&self, store: &crate::store::SessionStore, name: &str) -> Result<()> {
3198 match &self.goal {
3199 Some(goal) => store.save_goal(name, goal),
3200 None => store.clear_goal(name),
3201 }
3202 }
3203
3204 /// BP-7: adopt a goal loaded from the store (a resumed session picks up
3205 /// exactly where it left off). Bypasses the module gate on purpose: a
3206 /// goal already persisted is data to restore, not a new capability
3207 /// being turned on, and dropping it silently would lose session state.
3208 pub fn restore_goal(&mut self, goal: Option<crate::goals::GoalRecord>) {
3209 self.goal = goal;
3210 }
3211
3212 // ---- BP-7: extended-thinking control (catalog §4a "Extended thinking
3213 // control": "Reasoning on/off/levels mid-session") ----
3214
3215 /// The reasoning-effort level in force for the NEXT request, or `None`
3216 /// when extended thinking is off.
3217 pub fn effort(&self) -> Option<&str> {
3218 self.config.effort.as_deref()
3219 }
3220
3221 /// Change the reasoning-effort level mid-session.
3222 ///
3223 /// `Some(level)` sets the level; `None` turns extended thinking OFF —
3224 /// the on/off toggle the ledger row named as distinct from the level.
3225 /// `run_loop` reads `self.config.effort` fresh when it builds each
3226 /// `ChatRequest`, so this takes effect on the very next request with no
3227 /// other copy to update (the same contract [`Self::set_model`] has).
3228 /// The change is appended to the turn-record log as an `effort` marker,
3229 /// the extended-thinking analog of the `model_change` log.
3230 ///
3231 /// Returns the PREVIOUS setting.
3232 pub fn set_effort(&mut self, effort: Option<String>) -> Option<String> {
3233 let previous = self.config.effort.clone();
3234 if previous == effort {
3235 return previous;
3236 }
3237 self.config.effort = effort.clone();
3238 self.push_turn_marker(crate::turn_record::TurnMarker::Effort {
3239 from: previous.clone(),
3240 to: effort,
3241 });
3242 previous
3243 }
3244
3245 // ---- BP-7: review mode (catalog §4a "Review mode — dedicated
3246 // code-review flow"; §3.1 `[core.prompts]`) ----
3247
3248 /// The purpose-built review turn's prompt: the `code-review` template
3249 /// from [`Config::prompts`] with `{args}` replaced by `args`.
3250 ///
3251 /// `None` when the resolved config carries no `code-review` template —
3252 /// the preset decides whether this harness has a review mode, and the
3253 /// template IS the report format (both parity presets pin one).
3254 pub fn review_prompt(&self, args: &str) -> Option<String> {
3255 self.config
3256 .prompts
3257 .get(REVIEW_PROMPT_NAME)
3258 .map(|template| template.replace("{args}", args.trim()))
3259 }
3260
3261 /// Run the review turn: an ordinary [`Self::send`] of
3262 /// [`Self::review_prompt`], so the review's request, tools, transcript
3263 /// and records are the session's own — a purpose-built TURN, not a
3264 /// second agent.
3265 pub async fn review(&mut self, args: &str) -> Result<String> {
3266 let prompt = self.review_prompt(args).ok_or_else(|| {
3267 Error::Other(format!(
3268 "no `{REVIEW_PROMPT_NAME}` prompt template is configured for this harness"
3269 ))
3270 })?;
3271 self.send(prompt).await
3272 }
3273
3274 // ---- BP-7: side/ephemeral Q&A (catalog §4a "Side/ephemeral Q&A":
3275 // "Tool-less question over full context, never enters history") ----
3276
3277 /// Answer `question` over this session's FULL current context without
3278 /// recording anything.
3279 ///
3280 /// Three properties, all load-bearing and all asserted by this build's
3281 /// tests: the request carries the whole conversation as the next turn
3282 /// would see it; it advertises NO tools, so the model can only answer;
3283 /// and neither `history`, the sidecar recorder, the usage log nor the
3284 /// turn-record log is touched — `&self`, not `&mut self`, is the type
3285 /// system saying so. cc's `/btw` and cx's `/side`.
3286 pub async fn side_question(&self, question: &str) -> Result<String> {
3287 let mut messages = self.history.clone();
3288 if let Some(goal) = &self.goal {
3289 messages.push(ChatMessage::system(goal.reminder()));
3290 }
3291 messages.push(ChatMessage::user(format!(
3292 "{SIDE_QUESTION_PREAMBLE}
3293
3294{question}"
3295 )));
3296 let mut req = ChatRequest {
3297 model: self.config.model.clone(),
3298 messages,
3299 tools: Vec::new(),
3300 temperature: self.config.temperature,
3301 max_tokens: self.config.max_tokens,
3302 effort: self.config.effort.clone(),
3303 response_format: None,
3304 service_tier: None,
3305 thinking_budget: None,
3306 extra_body: self.config.extra_body.clone(),
3307 };
3308 // BP-13: a side question is still a request to THIS model, so it
3309 // carries the same routing decisions the loop's own requests do.
3310 self.apply_routing(&mut req);
3311 let (assistant, _usage) = self.provider.complete(&req, &|_: &str| {}).await?;
3312 Ok(assistant.content.unwrap_or_default())
3313 }
3314
3315 /// BP-7: append one marker against the NEXT round-trip's index — the
3316 /// right frame for a marker written between turns (a goal change, an
3317 /// effort change, an abort).
3318 fn push_turn_marker(&mut self, marker: crate::turn_record::TurnMarker) {
3319 self.push_turn_marker_at(self.turn_index, marker);
3320 }
3321
3322 /// BP-7: append one marker against an explicit round-trip index — used
3323 /// inside `Self::run_loop`, where markers are written on both sides of
3324 /// the `turn_index` advance and must all carry the round-trip they
3325 /// describe.
3326 fn push_turn_marker_at(&mut self, turn: usize, marker: crate::turn_record::TurnMarker) {
3327 self.turn_records.push(crate::turn_record::TurnRecord::new(
3328 turn,
3329 &self.config.model,
3330 now_ms(),
3331 marker,
3332 ));
3333 }
3334
3335 /// P4b (§1.7, pi§3 semantics): queue a mid-turn steering message —
3336 /// delivered "after current tool calls" (pi's phrasing): at the top of
3337 /// `Self::run_loop`'s NEXT iteration, before the next model request is
3338 /// built, regardless of whether this turn is still mid-flight with
3339 /// pending tool calls. Drained per [`Config::steering_mode`].
3340 pub fn queue_steer(&self, message: impl Into<String>) {
3341 let message = message.into();
3342 // BP-8 (catalog:154 "Queued-prompt persistence"): the input is
3343 // recorded BEFORE it is queued, so the window in which a crash
3344 // could lose it is zero. A no-op when `core.session.queue_persist`
3345 // is off (cx-parity: stock Codex has no queue-operation records).
3346 if self.config.session_queue_persist {
3347 self.journal_op(crate::session_journal::JournalOp::Enqueue {
3348 queue: crate::session_journal::QueueKind::Steer,
3349 text: message.clone(),
3350 });
3351 }
3352 self.steer_queue
3353 .lock()
3354 .unwrap_or_else(std::sync::PoisonError::into_inner)
3355 .queue_unchecked(message);
3356 }
3357
3358 /// Crate-internal shared steering handle used by the canonical SDK
3359 /// runtime. It remains writable while an active turn holds `&mut Agent`,
3360 /// allowing local and remote frontends to steer without owning the loop.
3361 pub(crate) fn steer_queue_handle(&self) -> std::sync::Arc<std::sync::Mutex<SteerInbox>> {
3362 self.steer_queue.clone()
3363 }
3364
3365 /// P4b: queue a follow-up message — delivered "at idle" (pi's phrasing):
3366 /// only once `Self::run_loop` would otherwise return a final answer
3367 /// (no more tool calls pending). Drained per [`Config::follow_up_mode`].
3368 pub fn queue_follow_up(&mut self, message: impl Into<String>) {
3369 let message = message.into();
3370 // BP-8 (catalog:154): same record-then-queue order as
3371 // [`Self::queue_steer`].
3372 if self.config.session_queue_persist {
3373 self.journal_op(crate::session_journal::JournalOp::Enqueue {
3374 queue: crate::session_journal::QueueKind::FollowUp,
3375 text: message.clone(),
3376 });
3377 }
3378 self.follow_up_queue.push_back(message);
3379 }
3380
3381 /// P4b: how many steering messages are currently queued (mid-turn +
3382 /// follow-up combined) — mostly for tests/diagnostics.
3383 pub fn queued_steer_count(&self) -> usize {
3384 self.steer_queue
3385 .lock()
3386 .unwrap_or_else(std::sync::PoisonError::into_inner)
3387 .len()
3388 + self.follow_up_queue.len()
3389 }
3390
3391 /// The accumulating reduction log (A5) — every reduction applied to any
3392 /// projected request view so far. Combined with a full-fidelity sidecar
3393 /// Session, this is enough to `reduce::invert` any projected view back to
3394 /// the exact original.
3395 pub fn reduction_log(&self) -> &ReductionLog {
3396 &self.reduction_log
3397 }
3398
3399 /// PARITY-18 D4 — arm the per-send context guard: `Self::run_loop`
3400 /// will refuse (via [`Error::ContextLimitExceeded`]) to build and issue
3401 /// ANY request — the first or any later turn — whose
3402 /// [`supercode_runtime::context_guard`] verdict is "does not fit" against
3403 /// `limit`. Call this once the target model's context-window size is
3404 /// known (`resume --reduced`'s preflight already computes it). Leaving
3405 /// this unset (the default) is a no-op: no guard runs, exactly today's
3406 /// pre-PARITY-18 behavior.
3407 pub fn set_context_limit(&mut self, limit: u64) {
3408 self.context_limit = Some(limit);
3409 }
3410
3411 /// This agent's armed context limit, if [`Self::set_context_limit`] has
3412 /// been called.
3413 pub fn context_limit(&self) -> Option<u64> {
3414 self.context_limit
3415 }
3416
3417 /// The model identifier this agent sends on its next request
3418 /// ([`Config::model`], as of construction/resume or the last
3419 /// [`Self::set_model`] call).
3420 pub fn model(&self) -> &str {
3421 &self.config.model
3422 }
3423
3424 /// UX-30 dev/02 — switch the model this agent sends, starting with the
3425 /// NEXT request it builds (and every one after, until changed again).
3426 /// `Self::run_loop` reads `self.config.model` fresh on every request
3427 /// (see its `ChatRequest` construction), so this alone is enough —
3428 /// there is no cached/baked-in copy anywhere else to also update.
3429 /// Takes effect immediately; safe to call only between turns (the
3430 /// REPL's `/model` picker runs at the prompt, never mid-turn). Touches
3431 /// nothing else: history, the sidecar, and reduction state are exactly
3432 /// as untouched as [`Self::set_schema_tier`] leaves them for a
3433 /// mid-session tier change.
3434 ///
3435 /// P4c-review note: this is the LOW-LEVEL primitive — it swaps
3436 /// [`Config::model`] and nothing else. It does NOT run dep 8's
3437 /// reasoning-artifact filter
3438 /// ([`reduce::rehydrate::filter_reasoning_artifacts`]) and does NOT
3439 /// create a [`crate::model_change::ModelChangeRecord`], so calling it
3440 /// directly for a mid-session handoff between two DIFFERENT models
3441 /// leaves model-A's reasoning artifacts in `history` for model-B to
3442 /// inherit. [`Self::switch_model`] is the safe superset — gated by
3443 /// [`Config::model_switch_allow_switch`], it filters and records the
3444 /// switch before delegating to this method — and is what callers
3445 /// performing a governed mid-session model switch should use instead.
3446 pub fn set_model(&mut self, model: impl Into<String>) {
3447 self.config.model = model.into();
3448 // BP-5 (catalog D2 "Per-model-family base-prompt selection"): the
3449 // family's base prompt follows the model. Codex re-selects
3450 // `base_instructions` when the model changes; leaving model-A's
3451 // base prompt in front of model-B is exactly the mismatch the row
3452 // exists to prevent. Same locate-and-replace mechanism
3453 // `refresh_env_context` uses, and a no-op whenever the selection
3454 // did not actually change (always, for a config with no family
3455 // table).
3456 self.refresh_base_prompt();
3457 // BP-7: the price follows the model, or the per-turn cost figure
3458 // would keep billing the OLD model's rates after a switch.
3459 self.model_price = crate::pricing::resolve(
3460 &self.config.model,
3461 self.config.price_input_per_mtok,
3462 self.config.price_output_per_mtok,
3463 );
3464 }
3465
3466 /// P4c (§1.10/§3.1 `core.model_switch.allow_switch`, D9 row, dep 8,
3467 /// design's "core NEW-significant" item): the mid-session model
3468 /// switch — a superset of [`Self::set_model`] gated by
3469 /// [`Config::model_switch_allow_switch`].
3470 ///
3471 /// **`allow_switch = false` (the default): EXACTLY [`Self::set_model`]**
3472 /// — same single field write, nothing else touched, no
3473 /// [`crate::model_change::ModelChangeRecord`] created. Byte-identical to
3474 /// calling `set_model` directly.
3475 ///
3476 /// **`allow_switch = true`:** additionally, before the swap takes
3477 /// effect, runs [`reduce::rehydrate::filter_reasoning_artifacts`] over
3478 /// [`Self::history`] — model-A's reasoning/thinking artifacts (any
3479 /// [`supercode_interchange::ChatMessage::metadata`] key in
3480 /// [`reduce::rehydrate::REASONING_METADATA_KEYS`], any `content_parts`
3481 /// block whose `"type"` is in
3482 /// [`reduce::rehydrate::REASONING_CONTENT_PART_TYPES`]) are stripped
3483 /// BEFORE model-B ever builds a request from this history — then
3484 /// appends a typed, translatable [`crate::model_change::ModelChangeRecord`]
3485 /// to [`Self::model_change_records`] (persist it via
3486 /// [`Self::save_model_change_log`]). A switch TO the current model
3487 /// (`model == Self::model()`) is treated as a no-op — still exactly
3488 /// `set_model`'s mechanics, no record for a switch that didn't actually
3489 /// change anything (and nothing to filter FOR, since there was no
3490 /// handoff).
3491 pub fn switch_model(&mut self, model: impl Into<String>) {
3492 let to = model.into();
3493 if !self.config.model_switch_allow_switch || self.config.model == to {
3494 self.set_model(to);
3495 return;
3496 }
3497 let from = self.config.model.clone();
3498 self.record_model_change(&from, &to, None);
3499 }
3500
3501 /// BP-13 — the ONE place a mid-session model change is performed and
3502 /// recorded, shared by [`Self::switch_model`] (a user asked) and the
3503 /// run loop's fallback pass (a provider failed).
3504 ///
3505 /// It does four things, in this order, and nothing else: strips model-A
3506 /// reasoning artifacts out of the live history (dep 8 — model B must
3507 /// never inherit them), moves [`Config::model`], appends the typed
3508 /// [`crate::model_change::ModelChangeRecord`], and writes that same
3509 /// record into the append-only session journal (BP-8) — which is where
3510 /// every persisted routing record lives; there is no second file. The
3511 /// change is also EMITTED, so a surface that renders events shows the
3512 /// switch instead of silently answering as a different model.
3513 pub fn record_model_change(&mut self, from: &str, to: &str, reason: Option<&str>) {
3514 if from == to {
3515 return;
3516 }
3517 let touched = reduce::rehydrate::filter_reasoning_artifacts(&mut self.history);
3518 self.set_model(to.to_string());
3519 let record = crate::model_change::ModelChangeRecord::new(
3520 self.turn_index,
3521 from,
3522 to,
3523 true,
3524 touched,
3525 now_ms(),
3526 )
3527 .with_reason(reason.map(str::to_string));
3528 self.journal_model_change(&record);
3529 self.model_change_log.push(record);
3530 self.emit(AgentEvent::ModelChanged {
3531 from: from.to_string(),
3532 to: to.to_string(),
3533 reason: reason.map(str::to_string),
3534 });
3535 // A switch also re-injects the switch NOTICE when the config asks
3536 // for one (Codex's own mid-session behavior: the conversation is
3537 // told the model changed, so the new model reads the handoff rather
3538 // than inferring it from a style break).
3539 if self.config.model_switch_notice {
3540 let notice = ChatMessage::user(format!(
3541 "[model changed: {from} -> {to}{}]",
3542 match reason {
3543 Some(r) => format!(" ({r})"),
3544 None => String::new(),
3545 }
3546 ));
3547 let _ = self.record(¬ice);
3548 self.history.push(notice);
3549 }
3550 }
3551
3552 /// BP-13 (catalog D9 "Fast mode / service tiers"): set or clear the
3553 /// session-level service-tier override. `Some(tier)` WINS over the
3554 /// `[capabilities.model_catalog] service_tier` rule for every
3555 /// subsequent request (it is the live toggle the user just pulled);
3556 /// `None` puts the configured rule back in charge. Takes effect on the
3557 /// next request the loop builds, like [`Self::set_model`].
3558 pub fn set_service_tier(&mut self, tier: Option<String>) {
3559 self.config.service_tier = tier;
3560 }
3561
3562 /// BP-13 — apply the routing table to a request that already names its
3563 /// model: effort LEVEL (per-model override of `[core] effort`, clamped
3564 /// by whichever effort cap applies), thinking-token BUDGET, and service
3565 /// TIER (the live `/fast` override winning over the configured rule).
3566 /// Called for every request the loop builds AND again for every
3567 /// fallback hop, so a hop to a different model gets that model's
3568 /// routing rather than the previous model's.
3569 fn apply_routing(&self, req: &mut ChatRequest) {
3570 let routing = &self.config.model_routing;
3571 let rules = routing.rules_for(&req.model);
3572 // Plan mode's own effort tier, when the mode is live, is the
3573 // session level for this request — Codex's `/plan` is effort
3574 // steering (cx§6), so planning need not think at the executing
3575 // level. It is still clamped by whatever effort cap applies,
3576 // because `effective_effort` does the clamping, not this line.
3577 let session_effort = match (
3578 self.ctx.plan_mode.is_active(),
3579 self.config.plan_mode_effort.as_deref(),
3580 ) {
3581 (true, Some(effort)) => Some(effort),
3582 _ => self.config.effort.as_deref(),
3583 };
3584 req.effort = routing.effective_effort(&req.model, session_effort);
3585 req.thinking_budget = rules.thinking_budget;
3586 req.service_tier = self.config.service_tier.clone().or(rules.service_tier);
3587 }
3588
3589 /// BP-13 — send `req`, walking [`Config::model_fallback`] when the
3590 /// failure is one another model could plausibly answer.
3591 ///
3592 /// Returns the final outcome plus the hops actually taken, so the
3593 /// caller (which owns `&mut self`) can record each one. Each hop
3594 /// re-applies routing for the new model and strips model-A reasoning
3595 /// artifacts from the request's own message copy before model B sees
3596 /// them — the same dep-8 guarantee [`Self::record_model_change`] gives
3597 /// the live history.
3598 async fn complete_with_fallback(
3599 &self,
3600 req: &mut ChatRequest,
3601 on_delta: &(dyn for<'a> Fn(&'a str) + Send + Sync),
3602 ) -> (Result<(ChatMessage, provider::Usage)>, Vec<FallbackHop>) {
3603 let mut hops = Vec::new();
3604 let mut result = self.provider.complete(req, on_delta).await;
3605 for next in &self.config.model_fallback {
3606 let Err(error) = &result else {
3607 break;
3608 };
3609 if !is_failover_worthy(error) {
3610 break;
3611 }
3612 if next.is_empty() || next == &req.model {
3613 continue;
3614 }
3615 let reason = error.to_string();
3616 let from = std::mem::replace(&mut req.model, next.clone());
3617 reduce::rehydrate::filter_reasoning_artifacts(&mut req.messages);
3618 self.apply_routing(req);
3619 hops.push(FallbackHop {
3620 from,
3621 to: next.clone(),
3622 reason,
3623 });
3624 result = self.provider.complete(req, on_delta).await;
3625 }
3626 (result, hops)
3627 }
3628
3629 /// P4c: every [`crate::model_change::ModelChangeRecord`] this agent has
3630 /// accumulated so far (via [`Self::switch_model`] with `allow_switch`
3631 /// on). Empty when the knob is off or no switch has happened yet.
3632 pub fn model_change_records(&self) -> &[crate::model_change::ModelChangeRecord] {
3633 &self.model_change_log
3634 }
3635
3636 /// P4c: persist this agent's accumulated model-change log to `store`
3637 /// under `name` — the [`crate::model_change::ModelChangeRecord`] analog
3638 /// of [`Self::save_usage_log`].
3639 pub fn save_model_change_log(
3640 &self,
3641 store: &crate::store::SessionStore,
3642 name: &str,
3643 ) -> Result<()> {
3644 store.save_model_change_log(name, &self.model_change_log)
3645 }
3646
3647 /// P4e (§1.6/§3.1 `core.session.git_metadata`, catalog:331): this
3648 /// agent's captured git provenance, if [`Config::session_git_metadata`]
3649 /// was on at construction and the best-effort probe found a repo.
3650 pub fn git_metadata(&self) -> Option<&crate::git_metadata::GitMetadataRecord> {
3651 self.git_metadata.as_ref()
3652 }
3653
3654 /// P4e: persist this agent's captured git metadata to `store` under
3655 /// `name` — a thin wrapper over
3656 /// [`crate::store::SessionStore::save_git_metadata`], the
3657 /// [`crate::git_metadata::GitMetadataRecord`] analog of
3658 /// [`Self::save_usage_log`]. A no-op (`Ok(())`, nothing written) when
3659 /// [`Self::git_metadata`] is `None`.
3660 pub fn save_git_metadata(&self, store: &crate::store::SessionStore, name: &str) -> Result<()> {
3661 match &self.git_metadata {
3662 Some(record) => store.save_git_metadata(name, record),
3663 None => Ok(()),
3664 }
3665 }
3666
3667 /// P4e DEFECT-FIX (independent Fable-5 review of P4e: `core.session.persist`
3668 /// had a `Config` field and CLI plumbing at `ConfigProfile` → `Config` but
3669 /// no consumer at all): whether a CLI caller's session-store save sites
3670 /// (`persist_session`, `persist_full_view`) should actually write to
3671 /// disk. `true` (the default) is byte-identical to pre-fix behavior —
3672 /// every session persists. `false` makes a session ephemeral: it runs
3673 /// exactly as before, but no `<name>.jsonl`/sidecar family is ever
3674 /// written for it. A plain getter, same posture as [`Self::model`] —
3675 /// this crate itself never reads or enforces it; the CLI's save sites do.
3676 pub fn session_persist(&self) -> bool {
3677 self.config.session_persist
3678 }
3679
3680 /// P4e DEFECT-FIX (independent Fable-5 review of P4e: `core.session.name`
3681 /// had a `Config` field and CLI plumbing but no consumer): the
3682 /// caller-configured session name, if `[core.session] name` was set.
3683 /// `None` (the default) leaves session naming exactly as before —
3684 /// `mint_session_name`'s auto-generated `<tag>-<adjective>-<noun>` shape.
3685 /// A plain getter, same posture as [`Self::session_persist`].
3686 pub fn session_name(&self) -> Option<&str> {
3687 self.config.session_name.as_deref()
3688 }
3689
3690 /// PARITY-18 D3 — whether this agent has actually issued at least one
3691 /// live request to its [`Provider`] so far (set the instant
3692 /// `Self::run_loop` reaches its real send site, regardless of whether
3693 /// that call then succeeds or fails). Callers should report
3694 /// "request sent" from THIS, never from having merely passed the
3695 /// context guard or having called [`Self::send`] — either of those can
3696 /// happen with zero requests actually issued (a guard refusal, an
3697 /// interactive session quit before any turn completes).
3698 pub fn request_issued(&self) -> bool {
3699 self.requests_issued
3700 }
3701
3702 /// P5-2 (§2.2 C2): whether this agent currently considers its
3703 /// [`CachePlan::ImportedPrefix`] cache entry warm — mirrors
3704 /// [`Self::request_issued`]'s read-only-observability precedent, so a
3705 /// caller (or a test) can confirm [`Self::register_tool`]'s C2
3706 /// invalidation actually took effect without reaching into private
3707 /// state.
3708 pub fn cache_established(&self) -> bool {
3709 self.cache_established
3710 }
3711
3712 /// B7: length of the imported-prefix protected by [`CachePlan::ImportedPrefix`]
3713 /// (this agent's own system message plus every message of a
3714 /// previously-imported session), set by [`Self::load_session`]. `None`
3715 /// until a session has been loaded.
3716 pub fn imported_prefix_len(&self) -> Option<usize> {
3717 self.imported_prefix_len
3718 }
3719
3720 /// Replace this agent's accumulating reduction log (C4: `/expand`/`/reduce`
3721 /// mutate the log directly via `reduce::invert_one`/`reduce::project_messages`
3722 /// and must feed the result back here so the *next* request build or
3723 /// persist sees the updated state instead of silently recomputing from an
3724 /// empty log). Also lets a caller (`resume_cmd`, C1) seed the log with the
3725 /// initial projection it already computed for the entry banner, so
3726 /// `reduction_log()` reflects reality even before this agent's first
3727 /// `send()` (which is otherwise the only place `build_request_messages`
3728 /// populates it).
3729 pub fn set_reduction_log(&mut self, log: ReductionLog) {
3730 self.reduction_log = log;
3731 }
3732
3733 /// Replace the conversation with a loaded session, keeping this agent's own
3734 /// system prompt at the front. The session's own system/developer turns are
3735 /// preserved after it for context.
3736 pub fn load_session(&mut self, session: Session) {
3737 let system = self.history.first().cloned();
3738 self.history.clear();
3739 if let Some(sys) = system {
3740 self.history.push(sys);
3741 }
3742 self.history.extend(session.messages);
3743 // B7: the whole of `history` at this point — this agent's own system
3744 // message plus every imported message — is the stable prefix a
3745 // resumed session resends byte-identically every turn.
3746 self.imported_prefix_len = Some(self.history.len());
3747 // UX-26 (B7-warn): a freshly loaded prefix has no established cache
3748 // entry of THIS agent's own making yet (even if this agent was
3749 // resumed once before — that earlier prefix is gone). Seed the
3750 // activity clock from the loaded session's own last message
3751 // timestamp (walking backward past any trailing message that
3752 // carries none), so a session that's been sitting idle since
3753 // Claude Code/Codex/a prior supercode run last touched it is
3754 // correctly treated as already-cold on its very first turn here —
3755 // `None` (no timestamp anywhere in the loaded messages) leaves the
3756 // TTL check disarmed rather than guessing.
3757 self.cache_established = false;
3758 self.last_cache_activity_ms = self
3759 .history
3760 .iter()
3761 .rev()
3762 .find_map(|m| m.metadata.get("timestamp"))
3763 .and_then(|ts| supercode_interchange::sidecar::rfc3339_to_ms(ts));
3764 }
3765
3766 /// Append `msg` to the sidecar recorder (A3), if one is installed — a
3767 /// no-op, at zero cost, when `recorder` is `None` (today's behavior).
3768 fn record(&mut self, msg: &ChatMessage) -> Result<()> {
3769 if let Some(recorder) = self.recorder.as_mut() {
3770 recorder.append(msg)?;
3771 }
3772 // BP-8 (catalog:150): the append-only half — written and FLUSHED
3773 // here, at the moment the message exists, not at the end of the
3774 // turn. A journal failure is logged, never fatal: durability
3775 // bookkeeping must not be able to fail a turn.
3776 if let Some(journal) = &self.journal {
3777 let mut guard = journal
3778 .lock()
3779 .unwrap_or_else(std::sync::PoisonError::into_inner);
3780 if let Err(error) = guard.append_message(msg) {
3781 tracing::warn!("failed to journal a message: {error}");
3782 }
3783 }
3784 // BP-8 (catalog:151): the same message becomes a tree node, so the
3785 // tree and the linear history never disagree about what was said.
3786 if let Some(tree) = self.session_tree.as_mut() {
3787 tree.append_message(msg.clone(), now_ms());
3788 }
3789 Ok(())
3790 }
3791
3792 /// Persist the live conversation to `path` as JSONL (one [`ChatMessage`]
3793 /// per line) so the session can be resumed later — supercode's own sessions
3794 /// become first-class, resumable artifacts.
3795 pub fn save_transcript(&self, path: impl AsRef<std::path::Path>) -> Result<()> {
3796 let mut out = String::new();
3797 for m in &self.history {
3798 out.push_str(&serde_json::to_string(m).map_err(Error::Decode)?);
3799 out.push('\n');
3800 }
3801 std::fs::write(path, out)?;
3802 Ok(())
3803 }
3804
3805 /// Restore a conversation previously written with [`Self::save_transcript`],
3806 /// replacing the current history.
3807 pub fn load_transcript(&mut self, path: impl AsRef<std::path::Path>) -> Result<()> {
3808 let text = std::fs::read_to_string(path)?;
3809 let mut history = Vec::new();
3810 for line in text.lines().map(str::trim).filter(|l| !l.is_empty()) {
3811 history.push(serde_json::from_str::<ChatMessage>(line).map_err(Error::Decode)?);
3812 }
3813 self.history = history;
3814 Ok(())
3815 }
3816
3817 /// Take a checkpoint of the current conversation position. Pass it to
3818 /// [`Self::rewind_to`] to discard everything sent since (the rewind/undo
3819 /// analog of `fork`/checkpoint).
3820 pub fn checkpoint(&self) -> usize {
3821 self.history.len()
3822 }
3823
3824 /// Rewind the conversation to a [`Self::checkpoint`], discarding later turns.
3825 pub fn rewind_to(&mut self, checkpoint: usize) {
3826 self.history.truncate(checkpoint.min(self.history.len()));
3827 }
3828
3829 /// Send a message with file inputs attached — the `--file` / `-i` analog.
3830 /// Each file's contents are injected into the prompt: UTF-8 text inline,
3831 /// binary (e.g. images) noted with a size marker. (Native image *vision*
3832 /// would additionally require multimodal content parts.)
3833 pub async fn send_with_files(
3834 &mut self,
3835 text: impl Into<String>,
3836 files: &[std::path::PathBuf],
3837 ) -> Result<String> {
3838 let mut prompt = text.into();
3839 for path in files {
3840 let block = match std::fs::read(path) {
3841 Ok(bytes) => match String::from_utf8(bytes.clone()) {
3842 Ok(s) => format!("\n\n[file: {}]\n{}", path.display(), s),
3843 Err(_) => format!(
3844 "\n\n[file: {} — {} bytes, binary content omitted]",
3845 path.display(),
3846 bytes.len()
3847 ),
3848 },
3849 Err(e) => format!("\n\n[file: {} — could not read: {e}]", path.display()),
3850 };
3851 prompt.push_str(&block);
3852 }
3853 let expanded = self.expand_prompt_async(&prompt).await;
3854 let msg = ChatMessage::user(expanded);
3855 self.guard_candidate_message(&msg)?;
3856 self.record(&msg)?;
3857 self.history.push(msg);
3858 self.run_loop().await
3859 }
3860
3861 /// Send a message with image inputs to a vision model — the `-i/--image`
3862 /// analog. `image_urls` may be `https://…` links or `data:image/…;base64,…`
3863 /// URLs; they're attached as multimodal `image_url` content parts.
3864 pub async fn send_with_images(
3865 &mut self,
3866 text: impl Into<String>,
3867 image_urls: &[String],
3868 ) -> Result<String> {
3869 let expanded = self.expand_prompt_async(&text.into()).await;
3870 let msg = ChatMessage::user_with_images(expanded, image_urls);
3871 self.guard_candidate_message(&msg)?;
3872 self.record(&msg)?;
3873 self.history.push(msg);
3874 self.run_loop().await
3875 }
3876
3877 /// Expand a `/<name> <args>` slash command against the registered prompt
3878 /// templates (`{args}` is replaced with the trailing text). Non-matching
3879 /// input is returned unchanged.
3880 /// BP-6 additionally resolves SKILL.md invocations here, after the
3881 /// template table misses: `/skill:name args` (pi§2 "Skill commands"),
3882 /// `/name args` when the config follows Claude Code (cc§7: "a `SKILL.md`
3883 /// in a directory = a `/name` command"), and `$slug` mentions (cx§7).
3884 /// `$ARGUMENTS` in the body is replaced with the trailing text.
3885 pub fn expand_prompt(&self, input: &str) -> String {
3886 // BP-5 (catalog D2 "@-file mentions / attachments"): `@path`
3887 // expansion happens FIRST, so a mention works in a bare message, in
3888 // a slash-command's arguments, and in the text a `$slug` mention
3889 // appends to — one rule, every prompt shape.
3890 let input = &self.expand_file_mentions(input);
3891 let trimmed = input.trim_start();
3892 let Some(rest) = trimmed.strip_prefix('/') else {
3893 return self.expand_skill_mentions(input);
3894 };
3895 let (name, args) = match rest.split_once(char::is_whitespace) {
3896 Some((n, a)) => (n, a.trim()),
3897 None => (rest, ""),
3898 };
3899 match self.config.prompts.get(name) {
3900 Some(template) => template.replace("{args}", args),
3901 None => match self.expand_skill_command(name, args) {
3902 Some(expanded) => expanded,
3903 None => self.expand_skill_mentions(input),
3904 },
3905 }
3906 }
3907
3908 /// BP-5 (catalog D2 "@-file mentions / attachments"; cc§2 "`@` in the
3909 /// prompt triggers file-path autocomplete and injects file context …
3910 /// Read deny rules best-effort apply to `@file` mentions"; cx§2
3911 /// "`@`-mentions (files)"): replace each `@path` token in `input` with
3912 /// that file's contents.
3913 ///
3914 /// **Deny-rule aware, through the one permissions engine.** Each
3915 /// mention is resolved with
3916 /// [`crate::permissions::evaluate_path_safe`] — the same
3917 /// traversal/symlink-resolving check a `read_file` tool call goes
3918 /// through — against this config's own rules and protected-path floor.
3919 /// Anything short of `Allow` inlines the refusal instead of the file, so
3920 /// `@.env` under a preset whose protected paths cover it says so rather
3921 /// than quietly leaking it.
3922 ///
3923 /// A token that names nothing readable is left exactly as the user typed
3924 /// it: an email address, a decorator, or a `@`-prefixed word in prose is
3925 /// not a file mention, and must survive untouched.
3926 /// Off by default (`[core.file_mentions]`).
3927 fn expand_file_mentions(&self, input: &str) -> String {
3928 if !self.config.file_mentions || !input.contains('@') {
3929 return input.to_string();
3930 }
3931 let mut attachments = String::new();
3932 let mut seen: Vec<String> = Vec::new();
3933 for token in input.split_whitespace() {
3934 let Some(rel) = token.strip_prefix('@') else {
3935 continue;
3936 };
3937 let rel = rel.trim_end_matches([',', ';', ':', '.', ')', ']', '"', '\'']);
3938 if rel.is_empty() || seen.iter().any(|s| s == rel) {
3939 continue;
3940 }
3941 let path = if std::path::Path::new(rel).is_absolute() {
3942 std::path::PathBuf::from(rel)
3943 } else {
3944 self.config.cwd.join(rel)
3945 };
3946 if !path.is_file() {
3947 continue;
3948 }
3949 seen.push(rel.to_string());
3950 attachments.push_str(&self.render_mention(rel, &path));
3951 if seen.len() >= MAX_FILE_MENTIONS_PER_MESSAGE {
3952 break;
3953 }
3954 }
3955 if attachments.is_empty() {
3956 return input.to_string();
3957 }
3958 format!("{input}{attachments}")
3959 }
3960
3961 /// One mention's block: the permission verdict first, then the bytes.
3962 /// Text is inlined; a binary file is named with its size, the same
3963 /// shape [`Self::send_with_files`] already uses for an explicit
3964 /// attachment, so a mention and a `--file` read the same way.
3965 fn render_mention(&self, shown: &str, path: &std::path::Path) -> String {
3966 use crate::permissions::{Decision, PathKind};
3967 let rules = crate::permissions::rules_for_config(&self.config);
3968 // BP-10's multi-root form: a mention is checked against every
3969 // granted root (cwd + `additional_dirs`), folded to the strictest —
3970 // the same call the tool-dispatch gate makes for a `read_file`
3971 // path, so a mention can never reach a file a read could not.
3972 let mut roots = vec![self.config.cwd.clone()];
3973 roots.extend(self.config.additional_dirs.iter().cloned());
3974 let decision = crate::permissions::evaluate_path_safe_roots(
3975 &rules,
3976 PathKind::Read,
3977 &roots,
3978 &path.to_string_lossy(),
3979 Decision::Allow,
3980 );
3981 if decision != Decision::Allow {
3982 return format!(
3983 "\n\n[file: {shown} — not attached; the permission rules for this session \
3984 resolve reading it to {decision:?}]"
3985 );
3986 }
3987 match std::fs::read(path) {
3988 Ok(bytes) => match String::from_utf8(bytes) {
3989 Ok(text) => {
3990 let mut text = text;
3991 if text.len() > MAX_FILE_MENTION_BYTES {
3992 let mut cut = MAX_FILE_MENTION_BYTES;
3993 while cut > 0 && !text.is_char_boundary(cut) {
3994 cut -= 1;
3995 }
3996 text.truncate(cut);
3997 text.push_str("\n[file truncated]");
3998 }
3999 format!("\n\n[file: {shown}]\n{text}")
4000 }
4001 Err(e) => format!(
4002 "\n\n[file: {shown} — {} bytes, binary content omitted]",
4003 e.into_bytes().len()
4004 ),
4005 },
4006 Err(e) => format!("\n\n[file: {shown} — could not read: {e}]"),
4007 }
4008 }
4009
4010 /// The SKILL.md packages this agent discovered (frontmatter only) — the
4011 /// exact set its prompt index lists and its `skill` tool can load.
4012 pub fn skills(&self) -> &[crate::skills::LoopSkill] {
4013 &self.skills
4014 }
4015
4016 /// BP-6: resolve a slash command against the discovered skills.
4017 ///
4018 /// `/skill:<name>` is pi's own form and is accepted under every config
4019 /// (it can never collide with a template name, which cannot contain a
4020 /// colon-prefixed `skill` segment by construction). The BARE `/<name>`
4021 /// form is Claude Code's — there, a skill IS a slash command — so it is
4022 /// honored only when the config reads Claude Code's roots; under
4023 /// `cx-parity`, where Codex has no skill slash commands, `/deploy` stays
4024 /// the literal text the user typed.
4025 fn expand_skill_command(&self, name: &str, args: &str) -> Option<String> {
4026 if self.skills.is_empty() {
4027 return None;
4028 }
4029 let bare = match name.strip_prefix("skill:") {
4030 Some(rest) => rest,
4031 None if self.config.skills_harness.as_deref()
4032 == Some(crate::HarnessId::CLAUDE_CODE) =>
4033 {
4034 name
4035 }
4036 None => return None,
4037 };
4038 let skill = self.find_skill(bare)?;
4039 skill
4040 .body_with_shell(args, &self.shell_injection)
4041 .ok()
4042 .map(|body| crate::skills::render_skill(skill, &body))
4043 }
4044
4045 /// BP-6: `$slug` mentions (cx§7 `TOOL_MENTION_SIGIL = '$'`) and — only
4046 /// under `[core.skills] implicit_match` — a description match.
4047 ///
4048 /// The user's own text is never replaced: a loaded body is APPENDED, the
4049 /// way Codex splices a skill into the turn. Mentions are only honored
4050 /// for a config that reads Codex's roots; `$WORD` is ordinary shell text
4051 /// everywhere else.
4052 fn expand_skill_mentions(&self, input: &str) -> String {
4053 if self.skills.is_empty() {
4054 return input.to_string();
4055 }
4056 let mut loaded: Vec<String> = Vec::new();
4057 let mut names: Vec<String> = Vec::new();
4058 if self.config.skills_harness.as_deref() == Some(crate::HarnessId::CODEX) {
4059 for token in input.split_whitespace() {
4060 let Some(slug) = token.strip_prefix('$') else {
4061 continue;
4062 };
4063 let slug =
4064 slug.trim_matches(|c: char| !c.is_alphanumeric() && c != '-' && c != ':');
4065 if slug.is_empty() {
4066 continue;
4067 }
4068 let Some(skill) = self.find_skill(slug) else {
4069 continue;
4070 };
4071 if names.contains(&skill.name) || loaded.len() >= MAX_SKILL_LOADS_PER_MESSAGE {
4072 continue;
4073 }
4074 if let Ok(body) = skill.body_with_shell("", &self.shell_injection) {
4075 names.push(skill.name.clone());
4076 loaded.push(crate::skills::render_skill(skill, &body));
4077 }
4078 }
4079 }
4080 if loaded.is_empty() && self.config.skills_implicit_match {
4081 if let Some(skill) = crate::skills::implicit_skill_match(&self.skills, input) {
4082 if let Ok(body) = skill.body_with_shell("", &self.shell_injection) {
4083 loaded.push(crate::skills::render_skill(skill, &body));
4084 }
4085 }
4086 }
4087 if loaded.is_empty() {
4088 return input.to_string();
4089 }
4090 format!("{input}\n\n{}", loaded.join("\n\n"))
4091 }
4092
4093 /// Resolve one invocation name against the discovered set — the same
4094 /// resolver the `skill` tool uses, so every door agrees on what a name
4095 /// means.
4096 fn find_skill(&self, name: &str) -> Option<&crate::skills::LoopSkill> {
4097 crate::skills::find_skill(&self.skills, name)
4098 }
4099
4100 /// P5-2 (§2 module 15 D7 row 4 "prompts-as-commands"): like
4101 /// [`Self::expand_prompt`], but also consults MCP-server-sourced
4102 /// prompts registered via [`Self::register_mcp_prompt`] when the local
4103 /// `Config::prompts` table has no match — a live `prompts/get`
4104 /// round-trip, which is why this is async and [`Self::expand_prompt`]
4105 /// itself stays synchronous (its public sync signature is unchanged,
4106 /// for every existing caller that doesn't need MCP prompts).
4107 ///
4108 /// **Argument mapping (a scope decision, not a protocol requirement —
4109 /// the MCP spec leaves "how does free CLI text become named prompt
4110 /// arguments" to the client):** a prompt with zero or one declared
4111 /// arguments gets the whole trailing text (empty string if the prompt
4112 /// takes no arguments and none was given); a prompt with two or more
4113 /// declared arguments expects `key=value` pairs, whitespace-separated
4114 /// (`/mcp__server__prompt lang=rust topic=async`) — an unparseable pair
4115 /// (no `=`) is simply skipped, never a hard error (matches this
4116 /// method's "non-matching input passes through" fail-open posture for
4117 /// the LOCAL-prompt case above).
4118 pub async fn expand_prompt_async(&self, input: &str) -> String {
4119 let local = self.expand_prompt(input);
4120 if local != input {
4121 return local; // a local `Config::prompts` template matched
4122 }
4123 let trimmed = input.trim_start();
4124 let Some(rest) = trimmed.strip_prefix('/') else {
4125 return input.to_string();
4126 };
4127 let (name, args) = match rest.split_once(char::is_whitespace) {
4128 Some((n, a)) => (n, a.trim()),
4129 None => (rest, ""),
4130 };
4131 let Some(source) = self.mcp_prompts.get(name) else {
4132 return input.to_string();
4133 };
4134 let arg_map = match source.arg_names() {
4135 [] => std::collections::BTreeMap::new(),
4136 [single] => {
4137 let mut m = std::collections::BTreeMap::new();
4138 if !args.is_empty() {
4139 m.insert(single.clone(), args.to_string());
4140 }
4141 m
4142 }
4143 _ => args
4144 .split_whitespace()
4145 .filter_map(|pair| pair.split_once('='))
4146 .map(|(k, v)| (k.to_string(), v.to_string()))
4147 .collect(),
4148 };
4149 match source.render(arg_map).await {
4150 Ok(rendered) => rendered,
4151 Err(e) => format!("Error: mcp prompt `{name}` failed: {e}"),
4152 }
4153 }
4154
4155 /// P5-2 (§2 module 15 D7 row 4): register an MCP server's prompt as a
4156 /// slash-command source — `command_name` MUST already be the
4157 /// namespaced `mcp__<server>__<prompt>` form
4158 /// ([`crate::mcp::McpServerHandle::prompts`] produces exactly that
4159 /// shape); this method does not re-namespace or validate it, so a
4160 /// caller that hands it a bare name defeats the collision protection
4161 /// [`crate::mcp::McpPromptSource`]'s doc comment describes. Overwrites
4162 /// any prior registration under the same command name (re-attaching
4163 /// the same server replaces its own earlier prompt list; this can
4164 /// never touch a NON-`mcp__`-prefixed key, i.e. never a local
4165 /// `Config::prompts` entry).
4166 pub fn register_mcp_prompt(
4167 &mut self,
4168 command_name: impl Into<String>,
4169 source: impl crate::sdk::SdkPromptSource + 'static,
4170 ) {
4171 self.mcp_prompts
4172 .insert(command_name.into(), Box::new(source));
4173 }
4174
4175 /// P5-2 (§2 module 15 D7 row 5 "instructions"): fold an MCP server's
4176 /// `initialize`-time instructions (or any other free-text note) into
4177 /// this agent's system message — the context-assembly site every other
4178 /// `core.*`/`capabilities.*` prompt-section append already uses
4179 /// (`Self::with_parts`), except this one fires AFTER construction
4180 /// (attaching MCP servers happens once the agent already exists — see
4181 /// `crates/cli/src/main.rs`'s `attach_mcp`). A no-op if `history` is
4182 /// somehow empty or its first message isn't a system message (never
4183 /// true for an `Agent` built via `Self::new`/`Self::with_parts`, but
4184 /// checked rather than assumed).
4185 pub fn append_system_note(&mut self, text: &str) {
4186 if let Some(system) = self.history.first_mut() {
4187 if system.role == Role::System {
4188 system
4189 .content
4190 .get_or_insert_with(String::new)
4191 .push_str(text);
4192 }
4193 }
4194 }
4195
4196 /// BP-4 (catalog:90, cx§2 `<environment_context>` "re-emitted on
4197 /// change"): re-derive the `# Environment` block and, if anything in it
4198 /// moved — cwd, the approval/sandbox policy, the git branch or its
4199 /// dirty state, the date — replace the stale copy in the system message
4200 /// with the fresh one. Returns whether the block changed.
4201 ///
4202 /// A no-op (and free — no git subprocess) when `core.env_context` is
4203 /// off, which is the default and every non-parity config. Replacing in
4204 /// place rather than appending a second block is deliberate: two
4205 /// `# Environment` sections disagreeing about cwd is worse context than
4206 /// one stale one, and the system message is re-sent on every request,
4207 /// so the rewrite IS the re-emission the model sees.
4208 pub fn refresh_env_context(&mut self) -> bool {
4209 if !self.config.env_context {
4210 return false;
4211 }
4212 let fresh = env_context_block(&self.config);
4213 let Some(stale) = self.env_context_live.clone() else {
4214 // Nothing was spliced at construction (e.g. `with_provider_arc`);
4215 // splice it now rather than silently never emitting one.
4216 self.append_system_note(&fresh);
4217 self.env_context_live = Some(fresh);
4218 return true;
4219 };
4220 if stale == fresh {
4221 return false;
4222 }
4223 if let Some(system) = self.history.first_mut() {
4224 if system.role == Role::System {
4225 if let Some(content) = system.content.as_mut() {
4226 if let Some(at) = content.find(&stale) {
4227 content.replace_range(at..at + stale.len(), &fresh);
4228 self.env_context_live = Some(fresh);
4229 return true;
4230 }
4231 }
4232 }
4233 }
4234 false
4235 }
4236
4237 /// BP-5 (catalog D2 "Per-model-family base-prompt selection"): re-select
4238 /// the base prompt for the model now in force and replace the stale one
4239 /// in place. Returns whether the system message changed.
4240 ///
4241 /// A no-op — not even a string search — when the selection is unchanged,
4242 /// which is every config that sets no `base_prompts` table.
4243 fn refresh_base_prompt(&mut self) -> bool {
4244 let fresh = base_prompt_for_config(&self.config);
4245 if fresh == self.base_prompt_live {
4246 return false;
4247 }
4248 let stale = std::mem::replace(&mut self.base_prompt_live, fresh.clone());
4249 if stale.is_empty() {
4250 return false;
4251 }
4252 if let Some(system) = self.history.first_mut() {
4253 if system.role == Role::System {
4254 if let Some(content) = system.content.as_mut() {
4255 if let Some(at) = content.find(&stale) {
4256 content.replace_range(at..at + stale.len(), &fresh);
4257 return true;
4258 }
4259 }
4260 }
4261 }
4262 false
4263 }
4264
4265 /// BP-5: assemble the request from an already-built message list and
4266 /// tool-schema list. The ONE place a [`ChatRequest`] is constructed from
4267 /// this agent's config, so the request `Self::run_loop` issues and the
4268 /// request [`Self::model_input`] renders cannot drift apart.
4269 fn chat_request(&self, messages: Vec<ChatMessage>, tools: Vec<ToolSchema>) -> ChatRequest {
4270 let mut req = ChatRequest {
4271 model: self.config.model.clone(),
4272 messages,
4273 tools,
4274 temperature: self.config.temperature,
4275 max_tokens: self.config.max_tokens,
4276 effort: self.config.effort.clone(),
4277 response_format: self.config.response_format.clone(),
4278 service_tier: None,
4279 thinking_budget: None,
4280 extra_body: self.config.extra_body.clone(),
4281 };
4282 // BP-13 (catalog Domain 9): the per-request routing decisions —
4283 // effort LEVEL, thinking-token BUDGET and service TIER — all come
4284 // out of `Config::model_routing` keyed by the model this request is
4285 // actually going to. Applied HERE so `model_input`'s rendering and
4286 // the loop's own send can never disagree about what would be sent,
4287 // and so a mid-session switch re-decides all three for the new
4288 // model on the next pass.
4289 self.apply_routing(&mut req);
4290 req
4291 }
4292
4293 /// BP-5 (catalog D2 "Prompt-input debugging": *render the exact
4294 /// model-visible input for inspection*; cx§2 `codex debug prompt-input`,
4295 /// which "renders the exact model-visible input list as JSON"): the
4296 /// request this agent would send next.
4297 ///
4298 /// Built by the SAME two calls the loop makes
4299 /// ([`Self::build_request_messages`], [`Self::tool_schemas`]) and
4300 /// assembled by the SAME [`Self::chat_request`] — it is the real
4301 /// request, not a reconstruction of one. `&mut self` because
4302 /// `build_request_messages` is: rendering the input is exactly as
4303 /// stateful as building it for a send.
4304 pub fn model_input(&mut self) -> ChatRequest {
4305 let tools = self.tool_schemas();
4306 let messages = self.build_request_messages();
4307 self.chat_request(messages, tools)
4308 }
4309
4310 /// BP-5: [`Self::model_input`] for a turn that has not been sent —
4311 /// `prompt` is expanded exactly as [`Self::send`] would expand it
4312 /// (slash templates, skills, `@path` mentions, MCP prompts) and appended
4313 /// to the conversation IN MEMORY, then the request is rendered.
4314 ///
4315 /// Deliberately not recorded: this door inspects an input, it does not
4316 /// take a turn. Nothing is written to the session store, no journal
4317 /// entry is made, and no request is issued.
4318 pub async fn model_input_for(&mut self, prompt: &str) -> ChatRequest {
4319 let expanded = self.expand_prompt_async(prompt).await;
4320 self.history.push(ChatMessage::user(expanded));
4321 self.model_input()
4322 }
4323
4324 /// BP-5: a [`ChatRequest`] as the JSON a human (or `jq`) inspects — the
4325 /// system prompt, every message in order, and every advertised tool
4326 /// schema, plus the sampling controls that travel with them.
4327 pub fn render_model_input(req: &ChatRequest) -> serde_json::Value {
4328 serde_json::json!({
4329 "model": req.model,
4330 "temperature": req.temperature,
4331 "max_tokens": req.max_tokens,
4332 "effort": req.effort,
4333 "response_format": req.response_format,
4334 // Serialized through `ChatMessage`'s OWN wire serializer and
4335 // `ToolSchema`'s own — i.e. the exact bytes the provider is
4336 // handed, not a second rendering of them.
4337 "messages": serde_json::to_value(&req.messages).unwrap_or(serde_json::Value::Null),
4338 "tools": serde_json::to_value(&req.tools).unwrap_or(serde_json::Value::Null),
4339 })
4340 }
4341
4342 /// BP-5 (catalog D2 "Per-model-family base-prompt selection"): the base
4343 /// system prompt currently in force for this agent's model.
4344 pub fn base_prompt(&self) -> &str {
4345 &self.base_prompt_live
4346 }
4347
4348 /// BP-4 (catalog:91 "Synthetic context-injection blocks"): splice one
4349 /// named ambient block into the live context — the seam a hook's
4350 /// `additionalContext`, a frontend nudge or an orchestrator's brief
4351 /// enters through, mid-session, after construction.
4352 ///
4353 /// Requires `core.context_injections` (returns `false` otherwise): the
4354 /// gate governs the whole registry, not just its startup half. The
4355 /// block is appended to the system message and remembered, so it is
4356 /// carried by every later request and re-rendered by
4357 /// [`crate::context_injection::assemble`] wherever the prompt is
4358 /// rebuilt.
4359 pub fn inject_context_block(
4360 &mut self,
4361 name: impl Into<String>,
4362 content: impl Into<String>,
4363 ) -> bool {
4364 if !self.config.context_injections {
4365 return false;
4366 }
4367 let block = crate::config::ContextInjectionBlock::new(name, content);
4368 let rendered = crate::context_injection::render(std::slice::from_ref(&block));
4369 self.spliced_context_blocks.push(block);
4370 self.append_system_note(&rendered);
4371 true
4372 }
4373
4374 /// The blocks spliced in since construction — see
4375 /// [`Self::inject_context_block`].
4376 pub fn spliced_context_blocks(&self) -> &[crate::config::ContextInjectionBlock] {
4377 &self.spliced_context_blocks
4378 }
4379
4380 /// Compact the conversation if it has grown past the configured
4381 /// threshold.
4382 ///
4383 /// **Re-founded (A10):** with a [`ReductionPolicy`] installed
4384 /// ([`Self::set_reduction_policy`]), this no longer touches `self.history`
4385 /// at all. It derives `policy.clear_turns_older_than` from
4386 /// `compact_after_messages` so the *next* projected request view
4387 /// (`reduce::project_messages`, built in `Self::run_loop`) collapses the
4388 /// old turns into one reversible `TurnsCleared` stub instead —
4389 /// `history()` and the sidecar keep every message forever; only the view
4390 /// shrinks. Returns whether the live (unreduced) history currently
4391 /// exceeds the threshold, i.e. whether a clearing will actually be
4392 /// visible in the next projected view.
4393 ///
4394 /// **Legacy path (no policy) — LOSSY, kept only for byte-identical
4395 /// backward compatibility (D6):** destructively rewrites `self.history`,
4396 /// permanently discarding the dropped middle turns (replaced by a single
4397 /// non-reversible summary marker that becomes their SOLE remaining copy —
4398 /// exactly the lossy compaction this reduction layer differentiates
4399 /// against). Once a sidecar/recorder or a [`ReductionPolicy`] is in play,
4400 /// prefer installing a policy so this method takes the re-founded path
4401 /// above instead.
4402 pub fn maybe_compact(&mut self) -> bool {
4403 // P4e (§1.5/§3.1 `core.compaction.enabled`, "no master gate exists
4404 // yet"): checked FIRST, before either trigger — `false` disables
4405 // every auto-compaction trigger unconditionally (message-count AND
4406 // pressure), composing with them rather than replacing their own
4407 // logic. `true` (the default, matching today's pre-P4e behavior,
4408 // where nothing ever gated compaction) falls straight through to
4409 // the existing trigger checks below, unchanged.
4410 if !self.config.compaction_enabled {
4411 return false;
4412 }
4413 let threshold = self.config.compact_after_messages;
4414 // P4b (§1.5/§3.1 `core.compaction.reserve_tokens`, pi§2 shape): a
4415 // SECOND, independent trigger — context-window pressure — alongside
4416 // (not instead of) the message-count one above. `None` (the
4417 // default) is byte-identical to today's message-count-only
4418 // behavior; this whole block is a no-op then.
4419 let message_trigger = threshold.is_some_and(|t| self.history.len() > t);
4420 let pressure_trigger = self.compaction_pressure_triggered();
4421 if threshold.is_none() && self.config.compaction_reserve_tokens.is_none() {
4422 return false;
4423 }
4424 if !message_trigger && !pressure_trigger {
4425 return false;
4426 }
4427 if let Some(policy) = self.reduction_policy.as_mut() {
4428 if let Some(t) = threshold {
4429 policy.clear_turns_older_than = Some(t);
4430 }
4431 // P4b scope note: the token-PRESSURE trigger's "how much to
4432 // clear" derivation (below, for the legacy in-place path) has no
4433 // `ReductionPolicy`/A10 analog yet — that mechanism decides its
4434 // own clearing window once `clear_turns_older_than` is set, so
4435 // pressure firing alone (no message threshold configured) has
4436 // nothing new to hand it in this pass. Report the message-count
4437 // verdict only, matching today's pre-P4b behavior exactly when
4438 // only `threshold` is set.
4439 return message_trigger;
4440 }
4441 // Legacy in-place compaction (no `ReductionPolicy` installed) below.
4442 // `keep_recent`: the message-count trigger's own `threshold / 2`
4443 // shape when it's what fired (or both fired); otherwise (pressure
4444 // fired alone) a token-budget-derived count.
4445 let keep_recent = if message_trigger {
4446 (threshold.unwrap() / 2).max(2)
4447 } else {
4448 self.keep_recent_count_by_tokens()
4449 };
4450 self.compact_in_place(keep_recent, None)
4451 }
4452
4453 /// BP-4 (catalog:98 "Manual compact with focus instructions", cc§2 /
4454 /// cx§2 `/compact [instructions]`): compact NOW, regardless of whether
4455 /// either automatic trigger has fired — the mechanism behind the REPL's
4456 /// `/compact [focus]`.
4457 ///
4458 /// `focus` is this invocation's steering text: it overrides the standing
4459 /// `core.compaction.focus_instructions` for this compaction only, is
4460 /// carried into the SUMMARIZER's input (so the model-written summary
4461 /// preserves what the user asked for), and is stated on the marker. An
4462 /// empty/whitespace `focus` falls back to the configured standing value,
4463 /// which is what a bare `/compact` means.
4464 ///
4465 /// Returns whether anything was compacted (`false` when the history is
4466 /// already at or below the keep-window, or when a [`ReductionPolicy`] is
4467 /// installed — under a policy the reversible A10 path owns clearing, and
4468 /// a manual compact would be the lossy one).
4469 pub fn compact_now(&mut self, focus: Option<&str>) -> bool {
4470 self.compacting_manually = true;
4471 let compacted = self.compact_now_inner(focus);
4472 self.compacting_manually = false;
4473 compacted
4474 }
4475
4476 fn compact_now_inner(&mut self, focus: Option<&str>) -> bool {
4477 if self.reduction_policy.is_some() {
4478 return false;
4479 }
4480 // A manual compact must actually compact. The token budget alone
4481 // (`core.compaction.keep_recent_tokens`, 20k) keeps EVERYTHING on
4482 // any ordinary conversation, which is right for the pressure
4483 // trigger (it fires only when the window is nearly full) and wrong
4484 // for `/compact`, whose whole point is compacting before the
4485 // pressure arrives. So the keep-window is the tighter of the two:
4486 // the token budget, and the message-count trigger's own established
4487 // "keep the most recent half" shape (`maybe_compact`'s
4488 // `threshold / 2`, floor 2).
4489 let keep_recent = self
4490 .keep_recent_count_by_tokens()
4491 .min((self.history.len() / 2).max(2));
4492 let focus = focus.map(str::trim).filter(|f| !f.is_empty());
4493 self.compact_in_place(keep_recent, focus)
4494 }
4495
4496 /// The legacy (no-[`ReductionPolicy`]) in-place compaction both
4497 /// [`Self::maybe_compact`] and [`Self::compact_now`] run: collapse
4498 /// `history[first..cut)` into one marker, keeping the newest
4499 /// `keep_recent` messages.
4500 fn compact_in_place(&mut self, keep_recent: usize, focus_override: Option<&str>) -> bool {
4501 if self.history.len() <= keep_recent {
4502 return false;
4503 }
4504 // Indices: 0 is the system prompt; collapse [first .. len-keep_recent).
4505 // `first` is 1 (only the system prompt is ever auto-preserved) unless
4506 // B7's coordination clamp widens it.
4507 let mut first = 1usize;
4508 // B7 coordination clamp: this legacy (no-`ReductionPolicy`) path
4509 // mutates `self.history` directly, so — unlike the re-founded A10
4510 // path (clamped inside `reduce::project_messages`, threaded from
4511 // `build_request_messages`) — it must clamp itself. Widening `first`
4512 // (not `cut`) is what actually protects the imported prefix: the
4513 // drop range is `[first, cut)`, so raising `cut` alone would only
4514 // drop MORE messages, not fewer. `imported_prefix_len` is already an
4515 // absolute `history` index count (it protects `history[0..len]`), so
4516 // no offset conversion is needed here.
4517 if matches!(self.config.cache_plan, CachePlan::ImportedPrefix) {
4518 if let Some(protected) = self.imported_prefix_len {
4519 first = first.max(protected);
4520 }
4521 }
4522 let mut cut = self.history.len() - keep_recent;
4523 if cut <= first {
4524 return false;
4525 }
4526 // Never begin the kept window on a tool result: its originating
4527 // assistant turn (with the matching `tool_calls`) is about to be
4528 // dropped, which would orphan the tool message and make the replayed
4529 // conversation invalid. Advance past any leading tool results.
4530 while cut < self.history.len() && self.history[cut].role == Role::Tool {
4531 cut += 1;
4532 }
4533 if cut >= self.history.len() {
4534 return false;
4535 }
4536 let dropped = cut - first;
4537 // BP-11: the compaction is decided from here on — the one point both
4538 // the automatic triggers and `/compact` pass through — so this is
4539 // where `pre_compact` observers hear about it.
4540 self.fire_lifecycle(&crate::config::LifecycleEvent::PreCompact {
4541 messages: self.history.len(),
4542 dropped,
4543 manual: focus_override.is_some() || self.compacting_manually,
4544 });
4545 // P4b (§1.5/§3.1 `core.compaction.focus_instructions`, catalog D2
4546 // "no instruction steering" gap): appended to the marker whenever
4547 // set, regardless of which trigger fired. `None` (the default)
4548 // leaves this byte-identical to the pre-P4b marker text.
4549 //
4550 // BP-4: `focus_override` — the per-invocation `/compact <focus>`
4551 // text — wins over the standing config value for THIS compaction,
4552 // which is what "`/compact [instructions]` steers what's preserved"
4553 // means. Neither is required.
4554 let focus: Option<String> = focus_override.map(str::to_string).or_else(|| {
4555 self.config
4556 .compaction_focus_instructions
4557 .clone()
4558 .filter(|f| !f.is_empty())
4559 });
4560 // BP-1 (§1.5/§3.1 `core.compaction.summarize`): the verb the marker
4561 // uses is now the config's to state. `true` (the default, and what
4562 // every preset sets) keeps the historical "summarized" text
4563 // byte-identical; `false` says only what actually happened to the
4564 // span, so a config that turns summarization off does not leave a
4565 // marker claiming a summary exists.
4566 let verb = if self.config.compaction_summarize {
4567 "summarized"
4568 } else {
4569 "cleared"
4570 };
4571 // BP-4 (catalog:107 "LLM summaries of cleared spans", design §1.5:
4572 // obligation 5 is "auto-compaction … + a persisted marker + AN LLM
4573 // SUMMARY OF THE COMPACTED SPAN", knob `[core.compaction] summarize`
4574 // — "the summary side-call depends on a utility model … core falls
4575 // back to the main model"). The side-call is therefore CORE, not a
4576 // reduction-module privilege: when `core.compaction.summarize` is on
4577 // and a summarizer is installed, the span is summarized by the model
4578 // and the marker carries that summary instead of only a count.
4579 //
4580 // Every failure mode degrades to the count-only marker: no
4581 // summarizer installed, an `Err` from the side-call, or an empty
4582 // reply. It never blocks or fails compaction — the same contract
4583 // TR-7's own side-call site keeps.
4584 let summary_body = if self.config.compaction_summarize {
4585 self.summarize_span(first..cut, focus.as_deref())
4586 } else {
4587 None
4588 };
4589 // BP-4 (catalog:99 "Compaction markers persisted in transcript"):
4590 // the marker states where the originals went, which is the whole
4591 // point of a boundary record — a reader must be able to tell a
4592 // reversible compaction from a lossy one without knowing which
4593 // modules were on.
4594 let retention = if self.recorder.is_some() {
4595 "The compacted messages remain in this session's transcript sidecar."
4596 } else {
4597 "No transcript sidecar is attached, so this marker is the only remaining record of them."
4598 };
4599 let mut summary_text = format!(
4600 "[earlier conversation compacted: {dropped} message(s) {verb} to save context]\n{retention}"
4601 );
4602 if let Some(focus) = &focus {
4603 summary_text.push_str(&format!("\n\nFocus: {focus}"));
4604 }
4605 if let Some(body) = &summary_body {
4606 summary_text.push_str(&format!("\n\nSummary of the compacted span:\n{body}"));
4607 }
4608 let summary = ChatMessage::system(summary_text);
4609 // BP-4 (catalog:99): PERSIST the boundary. Before this the legacy
4610 // path rewrote `self.history` and never called `record`, so the
4611 // marker existed only in the live window and a resumed session had
4612 // no on-disk trace that a compaction ever happened. A recorder
4613 // failure is logged, never fatal — losing the boundary record must
4614 // not lose the compaction.
4615 if let Err(error) = self.record(&summary) {
4616 tracing::warn!("failed to persist the compaction marker: {error}");
4617 }
4618 let mut new_history = Vec::with_capacity(first + keep_recent + 2);
4619 new_history.extend(self.history[..first].iter().cloned());
4620 new_history.push(summary);
4621 new_history.extend(self.history.split_off(cut));
4622 self.history = new_history;
4623 self.fire_lifecycle(&crate::config::LifecycleEvent::PostCompact {
4624 messages: self.history.len(),
4625 dropped,
4626 });
4627 // BP-8 (catalog:150): compaction RESHAPES the live view rather than
4628 // appending to it, so the journal's "everything since the last
4629 // checkpoint is unpersisted" accounting has to be re-based here —
4630 // otherwise a crash-recovery replay would re-append messages this
4631 // compaction deliberately set aside. The set-aside messages' own
4632 // bytes stay in the log above, untouched.
4633 self.journal_checkpoint(self.history.len());
4634 true
4635 }
4636
4637 /// BP-4 (catalog:107): run the installed [`reduce::summarize::SpanSummarizer`]
4638 /// over `history[span]`, with `focus` (the `/compact <focus>` text)
4639 /// carried into the summarizer's INPUT so the model-written summary
4640 /// preserves what the user asked to keep.
4641 ///
4642 /// `None` — never an error — whenever no summarizer is installed, the
4643 /// span renders empty, the side-call fails, or it returns nothing. The
4644 /// caller falls back to the count-only marker.
4645 fn summarize_span(&self, span: std::ops::Range<usize>, focus: Option<&str>) -> Option<String> {
4646 let summarizer = self.span_summarizer.as_deref()?;
4647 let mut span_text = String::new();
4648 // The focus rides at the head of the span text (the trait's one
4649 // input) as an explicit, labeled line rather than a silent prompt
4650 // mutation: the fixed prompt's "do not state anything not present
4651 // in the span" still holds, because the focus IS present in it.
4652 if let Some(focus) = focus {
4653 span_text.push_str(&format!("[compaction focus requested: {focus}]\n\n"));
4654 }
4655 for msg in self.history.get(span)? {
4656 let role = match msg.role {
4657 Role::System => "system",
4658 Role::User => "user",
4659 Role::Assistant => "assistant",
4660 Role::Tool => "tool",
4661 };
4662 span_text.push_str(role);
4663 span_text.push_str(": ");
4664 span_text.push_str(msg.content.as_deref().unwrap_or(""));
4665 span_text.push('\n');
4666 }
4667 match summarizer.summarize(&span_text) {
4668 Ok(text) if !text.trim().is_empty() => Some(text.trim().to_string()),
4669 Ok(_) => None,
4670 Err(error) => {
4671 tracing::warn!("compaction span summarizer failed: {error}");
4672 None
4673 }
4674 }
4675 }
4676
4677 /// BP-4 (catalog:106 "Handoff (fresh objective + curated keep-set)",
4678 /// cx§1 `new_context`): reset the live working view to a fresh
4679 /// objective plus a curated keep-set, in-session.
4680 ///
4681 /// The new view is: the system prompt (plus any imported prefix a
4682 /// `CachePlan::ImportedPrefix` config protects — same clamp compaction
4683 /// uses), then a handoff marker stating the objective and what was set
4684 /// aside, then the most recent `keep_recent` messages (`None` = the
4685 /// token-budget-derived count `core.compaction.keep_recent_tokens`
4686 /// already governs, so the keep-set is curated by the same budget the
4687 /// rest of the compaction machinery uses, not by a magic number). The
4688 /// keep-set never begins on a tool result, so no tool message is left
4689 /// orphaned from its originating assistant turn.
4690 ///
4691 /// Returns how many messages were set aside. Like compaction, the
4692 /// marker is PERSISTED through the recorder, so a resumed session can
4693 /// see where the handoff happened; and like compaction, the set-aside
4694 /// messages remain in the transcript sidecar whenever one is attached.
4695 ///
4696 /// Scope note: this is the in-session `new_context` mechanism, NOT
4697 /// `Config::handoff_enabled`'s reversible ReductionLog snapshot (the
4698 /// offline `supercode handoff` projection) — that one is the reduction
4699 /// module's, and stays there.
4700 pub fn new_context(&mut self, objective: &str, keep_recent: Option<usize>) -> usize {
4701 let keep_recent = keep_recent.unwrap_or_else(|| self.keep_recent_count_by_tokens());
4702 let mut first = 1usize;
4703 if matches!(self.config.cache_plan, CachePlan::ImportedPrefix) {
4704 if let Some(protected) = self.imported_prefix_len {
4705 first = first.max(protected);
4706 }
4707 }
4708 let mut cut = self.history.len().saturating_sub(keep_recent).max(first);
4709 while cut < self.history.len() && self.history[cut].role == Role::Tool {
4710 cut += 1;
4711 }
4712 let dropped = cut.saturating_sub(first);
4713 let objective = objective.trim();
4714 let retention = if self.recorder.is_some() {
4715 "They remain in this session's transcript sidecar."
4716 } else {
4717 "No transcript sidecar is attached, so they are not retained."
4718 };
4719 let marker = ChatMessage::system(format!(
4720 "[handoff: a fresh working context starts here]\nObjective: {objective}\n\
4721 {dropped} earlier message(s) were set aside; the most recent {kept} were kept. \
4722 {retention}",
4723 kept = self.history.len() - cut,
4724 ));
4725 if let Err(error) = self.record(&marker) {
4726 tracing::warn!("failed to persist the handoff marker: {error}");
4727 }
4728 let mut new_history = Vec::with_capacity(first + keep_recent + 2);
4729 new_history.extend(self.history[..first].iter().cloned());
4730 new_history.push(marker);
4731 new_history.extend(self.history.split_off(cut));
4732 self.history = new_history;
4733 // BP-8: same re-basing as `maybe_compact` — see its comment.
4734 self.journal_checkpoint(self.history.len());
4735 dropped
4736 }
4737
4738 /// P4b (§1.5/§3.1 `core.compaction.reserve_tokens`, pi§2 shape:
4739 /// `contextTokens > contextWindow - reserveTokens`): whether the
4740 /// estimated token size of the live history is within `reserve_tokens`
4741 /// of the model's context window. `false` when
4742 /// [`Config::compaction_reserve_tokens`] is unset (the default).
4743 fn compaction_pressure_triggered(&self) -> bool {
4744 let Some(reserve) = self.config.compaction_reserve_tokens else {
4745 return false;
4746 };
4747 let limit = provider::model_context_limit(&self.config.model)
4748 .unwrap_or(provider::UNKNOWN_MODEL_CONTEXT_FLOOR);
4749 let used = supercode_runtime::estimate_view_tokens(&self.history);
4750 used.saturating_add(reserve) > limit
4751 }
4752
4753 /// P4b (§1.5/§3.1 `core.compaction.keep_recent_tokens`): how many of the
4754 /// most recent messages (walking backward from the end of `self.history`,
4755 /// skipping the system prompt) fit within the configured token budget
4756 /// (default 20,000, pi§6 precedent). Always keeps at least 2 messages,
4757 /// matching the message-count trigger's own floor.
4758 fn keep_recent_count_by_tokens(&self) -> usize {
4759 let budget = self.config.compaction_keep_recent_tokens.unwrap_or(20_000);
4760 let mut used = 0u64;
4761 let mut count = 0usize;
4762 for msg in self.history.iter().skip(1).rev() {
4763 let t = supercode_runtime::estimate_view_tokens(std::slice::from_ref(msg));
4764 if used.saturating_add(t) > budget && count > 0 {
4765 break;
4766 }
4767 used = used.saturating_add(t);
4768 count += 1;
4769 }
4770 count.max(2)
4771 }
4772
4773 /// Register an additional tool (e.g. your own capability).
4774 ///
4775 /// P5-2 (§2.2 C2 "connect invalidates cache prefix"): registering a
4776 /// tool AFTER this agent has already issued a request
4777 /// ([`Self::request_issued`]) changes the tools schema every
4778 /// subsequent request carries — the exact prefix-churn shape C2
4779 /// describes, MCP-sourced or not. Resets [`Self::cache_established`] so
4780 /// the next cache-warmth check (`provider::cache_cold_reason`) doesn't
4781 /// wrongly assume the entry is still warm. A no-op call before the
4782 /// first request (the common case: `attach_mcp` registers tools once at
4783 /// startup, before any turn runs) changes nothing — byte-identical to
4784 /// today.
4785 pub fn register_tool(&mut self, tool: impl crate::tools::Tool + 'static) {
4786 self.registry.register(tool);
4787 if self.requests_issued {
4788 self.cache_established = false;
4789 }
4790 }
4791
4792 /// The current conversation, including the system prompt.
4793 pub fn history(&self) -> &[ChatMessage] {
4794 &self.history
4795 }
4796
4797 /// Send a user message and run the loop until the model produces a final
4798 /// answer (text with no tool calls) or the iteration budget is exhausted.
4799 pub async fn send(&mut self, user_input: impl Into<String>) -> Result<String> {
4800 let expanded = self.expand_prompt_async(&user_input.into()).await;
4801 let msg = ChatMessage::user(expanded);
4802 self.guard_candidate_message(&msg)?;
4803 self.record(&msg)?;
4804 self.history.push(msg);
4805 self.run_loop().await
4806 }
4807
4808 /// BP-4 (catalog:109 "Context-usage introspection", cc§2 `/context`
4809 /// grid, cx§8 `/status` + `get_context_remaining`): the LIVE
4810 /// context-window accounting for this session — the same numbers
4811 /// `resume --dry-run`'s preflight already computes
4812 /// (`tokens::estimate_request_tokens` / `tokens::context_guard`), read
4813 /// out mid-session instead of only before one.
4814 ///
4815 /// Pure: it projects the request view exactly as
4816 /// [`Self::guard_candidate_message`] does (reduction stubs included,
4817 /// cache annotation included) without mutating the reduction log, so
4818 /// asking "how full am I?" can never change what the next request
4819 /// carries.
4820 pub fn context_usage(&self) -> ContextUsage {
4821 let messages = self.projected_view(None);
4822 let tools = self.tool_schemas();
4823 let message_tokens = supercode_runtime::estimate_view_tokens(&messages);
4824 let request_tokens = supercode_runtime::estimate_request_tokens(&messages, &tools);
4825 let limit = self.context_limit.or_else(|| {
4826 crate::provider::model_context_limit(&self.config.model)
4827 .or(Some(crate::provider::UNKNOWN_MODEL_CONTEXT_FLOOR))
4828 });
4829 let projected_tokens = supercode_runtime::with_guard_margin(request_tokens);
4830 let reserve = supercode_runtime::CONTEXT_RESPONSE_RESERVE_TOKENS;
4831 let (fits, remaining_tokens, used_pct) = match limit {
4832 Some(limit) => (
4833 projected_tokens.saturating_add(reserve) <= limit,
4834 limit
4835 .saturating_sub(reserve)
4836 .saturating_sub(projected_tokens),
4837 if limit == 0 {
4838 0
4839 } else {
4840 (projected_tokens as f64 / limit as f64 * 100.0).round() as u32
4841 },
4842 ),
4843 None => (true, 0, 0),
4844 };
4845 ContextUsage {
4846 model: self.config.model.clone(),
4847 messages: messages.len(),
4848 message_tokens,
4849 tool_count: tools.len(),
4850 tool_schema_tokens: request_tokens.saturating_sub(message_tokens),
4851 request_tokens,
4852 projected_tokens,
4853 response_reserve_tokens: reserve,
4854 context_limit: limit,
4855 remaining_tokens,
4856 used_pct,
4857 fits,
4858 }
4859 }
4860
4861 /// The messages a request would carry right now — the read-only half of
4862 /// [`Self::guard_candidate_message`]/[`Self::build_request_messages`],
4863 /// with `candidate` optionally appended as a not-yet-committed turn.
4864 /// Never mutates `self`.
4865 fn projected_view(&self, candidate: Option<&ChatMessage>) -> Vec<ChatMessage> {
4866 let messages = match &self.reduction_policy {
4867 None => {
4868 let mut messages = self.history.clone();
4869 if let Some(candidate) = candidate {
4870 messages.push(candidate.clone());
4871 }
4872 messages
4873 }
4874 Some(policy) => {
4875 let has_system = self.history.first().is_some_and(|m| m.role == Role::System);
4876 let mut reducible = self.history[usize::from(has_system)..].to_vec();
4877 if let Some(candidate) = candidate {
4878 reducible.push(candidate.clone());
4879 }
4880 let mut prepared = policy.clone();
4881 reduce::prepare_read_freshness(&mut prepared, &reducible);
4882 let (view, _) =
4883 reduce::project_messages(&reducible, &prepared, &self.reduction_log);
4884 let mut messages = Vec::with_capacity(view.len() + usize::from(has_system));
4885 if has_system {
4886 messages.push(self.history[0].clone());
4887 }
4888 messages.extend(view);
4889 messages
4890 }
4891 };
4892 provider::apply_cache_plan(&messages, self.config.cache_plan, self.imported_prefix_len)
4893 }
4894
4895 /// Refuse an oversized new user turn before it mutates canonical history
4896 /// or an attached sidecar. The in-loop guard remains authoritative for
4897 /// every actual request; this preflight closes the first-request seam
4898 /// where `send*` used to record/push the message before that guard ran.
4899 fn guard_candidate_message(&self, msg: &ChatMessage) -> Result<()> {
4900 let Some(limit) = self.context_limit else {
4901 return Ok(());
4902 };
4903
4904 let messages = self.projected_view(Some(msg));
4905 let tools = self.tool_schemas();
4906 let (fits, projected_tokens) = supercode_runtime::context_guard(&messages, &tools, limit);
4907 if !fits {
4908 return Err(Error::ContextLimitExceeded {
4909 projected_tokens,
4910 reserve_tokens: supercode_runtime::CONTEXT_RESPONSE_RESERVE_TOKENS,
4911 context_limit: limit,
4912 model: self.config.model.clone(),
4913 });
4914 }
4915 Ok(())
4916 }
4917
4918 /// The messages a provider request should carry for the CURRENT turn
4919 /// (A5/A7/A8/A10): with no [`ReductionPolicy`] installed, exactly
4920 /// `self.history.clone()` — byte-identical to every version of this
4921 /// method before reduction landed. With a policy installed, `history[0]`
4922 /// (this agent's own system prompt, never a reduction target) followed by
4923 /// [`reduce::project_messages`]'s projected view of `history[1..]`, fed
4924 /// with `self.reduction_log` so already-applied reductions reproduce
4925 /// verbatim across turns (prefix stability, A5) — the updated log is
4926 /// stored back onto `self` so the NEXT call (this turn, next turn, or a
4927 /// later `send`) sees the same accumulating state. `self.history` itself
4928 /// is never read back into or mutated by this: it stays the full
4929 /// canonical view, in lockstep with the sidecar (A3).
4930 ///
4931 /// When `policy.elide_stale_reads` is set, this re-runs
4932 /// [`reduce::probe_read_freshness`] (the one place A8's disk I/O happens)
4933 /// against `history[1..]` before projecting, so every request sees
4934 /// up-to-date freshness verdicts — `project_messages` itself stays pure.
4935 ///
4936 /// Finally, B7's [`provider::apply_cache_plan`] runs over the assembled
4937 /// view (regardless of whether a [`ReductionPolicy`] is installed) — a
4938 /// pure, cloning annotation step, so this method's `&mut self` mutations
4939 /// above (`self.reduction_log`) are already committed before it runs and
4940 /// its own output is never written back onto `self.history` or the log:
4941 /// purity for B7's cache breakpoints holds independently of A5's.
4942 fn build_request_messages(&mut self) -> Vec<ChatMessage> {
4943 let messages = match self.reduction_policy.clone() {
4944 None => self.history.clone(),
4945 Some(mut policy) => {
4946 reduce::prepare_read_freshness(&mut policy, &self.history[1..]);
4947 // B7 coordination clamp: while `CachePlan::ImportedPrefix` is
4948 // active, A10 turn-clearing must never establish a range
4949 // that dips into the imported prefix (protects the cache
4950 // breakpoint the request build will place there below).
4951 // `imported_prefix_len` counts `history[0]` (this agent's own
4952 // system message) plus the imported messages, but
4953 // `project_messages` only ever sees `history[1..]` — hence
4954 // the `- 1`.
4955 if matches!(self.config.cache_plan, CachePlan::ImportedPrefix) {
4956 policy.protect_imported_prefix =
4957 self.imported_prefix_len.map(|n| n.saturating_sub(1));
4958 }
4959 // TR-7 (T20): the one side-call site, run BEFORE
4960 // `project_messages` (which stays pure/I-O-free) — mirrors
4961 // `elide_stale_reads`/`probe_read_freshness` immediately
4962 // above. Only ever does anything when both the policy gate
4963 // AND a summarizer are present; either being absent means
4964 // `cleared_turns_summary` stays `None` and `project_messages`
4965 // renders the deterministic stub, same as before TR-7
4966 // existed.
4967 if policy.summarize_cleared_turns {
4968 if let Some(summarizer) = self.span_summarizer.as_deref() {
4969 policy.cleared_turns_summary = reduce::prepare_cleared_turns_summary(
4970 &self.history[1..],
4971 &policy,
4972 &self.reduction_log,
4973 summarizer,
4974 );
4975 }
4976 }
4977 let (view, log) =
4978 reduce::project_messages(&self.history[1..], &policy, &self.reduction_log);
4979 self.reduction_log = log;
4980 let mut messages = Vec::with_capacity(view.len() + 1);
4981 messages.push(self.history[0].clone());
4982 messages.extend(view);
4983 messages
4984 }
4985 };
4986 // TR-8 (T5): a tool-schema tier change since the last request is a
4987 // cache-bust event under `CachePlan::ImportedPrefix` — the `tools`
4988 // array is part of the cache key alongside `messages`, so flag it by
4989 // skipping this one request's cache annotation rather than claiming
4990 // a prefix hit that won't actually land. Recorded unconditionally
4991 // (even under `CachePlan::Off`) so the signature stays current
4992 // regardless of which plan is active.
4993 let tier_sig = self.schema_tier_signature();
4994 let busted =
4995 provider::tier_change_is_cache_bust(self.last_tool_schema_tier_signature, tier_sig);
4996 self.last_tool_schema_tier_signature = Some(tier_sig);
4997 let effective_cache_plan = if busted {
4998 CachePlan::Off
4999 } else {
5000 self.config.cache_plan
5001 };
5002 // UX-26 (B7-warn): mirror `apply_cache_plan`'s own placement gate
5003 // (`ImportedPrefix` AND a non-zero prefix) to know whether THIS
5004 // request will actually carry a `cache_control` annotation. `busted`
5005 // requests (schema-tier change) and `CachePlan::Off` never annotate,
5006 // so `provider::cache_cold_reason` can never flag them — there was
5007 // nothing to reuse, by construction. `idle_secs` is computed
5008 // whenever a signal exists at all (even before this agent's first
5009 // annotated send — see `Self::last_cache_activity_ms`'s doc comment
5010 // on why the pre-establishment case matters); `cache_established`
5011 // additionally gates the usage-ratio check specifically (see
5012 // `provider::cache_cold_reason`'s doc comment for why those two
5013 // checks need independent gates).
5014 let will_annotate = matches!(effective_cache_plan, CachePlan::ImportedPrefix)
5015 && self.imported_prefix_len.is_some_and(|n| n > 0);
5016 let idle_secs = self
5017 .last_cache_activity_ms
5018 .map(|last| (now_ms() - last).max(0) / 1000);
5019 self.pending_cache_turn = (will_annotate, self.cache_established, idle_secs);
5020 let mut messages =
5021 provider::apply_cache_plan(&messages, effective_cache_plan, self.imported_prefix_len);
5022 // BP-7 (catalog §4a "Goals"): the standing objective, restated at
5023 // the TAIL of the request — after the cache annotation, which sits
5024 // on the PREFIX, so a goal that changes mid-session never busts the
5025 // cached prefix. Request-view only: `history` is untouched, so the
5026 // persisted transcript is exactly the conversation and a translator
5027 // never has to invent a message for a harness-tracked goal.
5028 if let Some(goal) = &self.goal {
5029 messages.push(ChatMessage::system(goal.reminder()));
5030 }
5031 messages
5032 }
5033
5034 /// Run the model/tool loop over the current history until a final answer or
5035 /// the iteration budget is exhausted. (Shared by `send`, `send_with_files`,
5036 /// and `send_with_images`.)
5037 /// P4b (§1.7, pi§3 semantics): pop the next message(s) to deliver from
5038 /// `queue` per `mode` — `All` drains everything and joins it with a
5039 /// blank line, `OneAtATime` pops exactly one. `None` when `queue` is
5040 /// empty (the default state, at zero cost).
5041 fn drain_steer_queue(
5042 queue: &mut std::collections::VecDeque<String>,
5043 mode: SteeringMode,
5044 ) -> Option<String> {
5045 if queue.is_empty() {
5046 return None;
5047 }
5048 match mode {
5049 SteeringMode::All => Some(queue.drain(..).collect::<Vec<_>>().join("\n\n")),
5050 SteeringMode::OneAtATime => queue.pop_front(),
5051 }
5052 }
5053
5054 async fn run_loop(&mut self) -> Result<String> {
5055 let _steer_turn = SteerTurnGuard::new(self.steer_queue.clone());
5056 let mut output_tokens_used: u64 = 0;
5057
5058 // BP-7 (catalog §4a "Turn/budget caps"): the SPEND cap, checked
5059 // before this `send` can issue anything. Unlike
5060 // `max_total_output_tokens` (a per-`send` allowance, unchanged),
5061 // spend accumulates over the agent's whole lifetime — a dollar
5062 // budget that resets on every prompt is not a budget. A cap reached
5063 // MID-loop ends that loop cleanly with a `spend_budget` finish
5064 // marker (below); a cap already exhausted at entry is an error,
5065 // because there is nothing to return.
5066 if let Some(budget) = self.config.max_budget_usd.filter(|b| *b > 0.0) {
5067 if self.total_cost_usd >= budget {
5068 return Err(Error::BudgetExhausted {
5069 spent_usd: self.total_cost_usd,
5070 budget_usd: budget,
5071 });
5072 }
5073 }
5074
5075 // P5-9 (§2 module 20, cc's "per-prompt file-history-snapshot"):
5076 // open a fresh checkpoint for THIS turn — `run_loop` is called
5077 // exactly once per `send`/`send_with_files`/`send_with_images`
5078 // call (never recursively for the same turn), so this fires once
5079 // per user prompt, matching the design's per-prompt granularity.
5080 // `self.history.last()` is the user message that call just pushed.
5081 // `None` (`checkpoint_observer` unset, the default) is a no-op —
5082 // zero cost, no disk touched.
5083 if let Some(cp) = &self.checkpoint_observer {
5084 let label = self
5085 .history
5086 .last()
5087 .and_then(|m| m.content.as_deref())
5088 .unwrap_or("")
5089 .to_string();
5090 cp.begin_turn(&label);
5091 }
5092
5093 // BP-4 (catalog:90, cx§2 "re-emitted on change"): once per user
5094 // turn — not per loop iteration — re-derive the environment block
5095 // so a cwd change, an approval/sandbox policy change or a branch
5096 // switch since the last turn reaches the model instead of leaving
5097 // it reading the startup snapshot. A no-op, with no subprocess, for
5098 // every config that doesn't set `core.env_context`.
5099 self.refresh_env_context();
5100
5101 for _ in 0..self.config.max_iterations {
5102 // BP-8 (catalog:156): flush a plan `update_plan` wrote during
5103 // the previous iteration's tool calls. A no-op when
5104 // `todos.persist` is off or the plan did not change.
5105 self.journal_plan_if_changed();
5106 // BP-7: the index of the round-trip this iteration is about to
5107 // make. Captured here because `self.turn_index` advances the
5108 // moment the usage record is written, and every marker in this
5109 // iteration — including the ones written after that point —
5110 // must carry the SAME index, or the marker log would not join
5111 // to the usage log on `turn`.
5112 let round_trip = self.turn_index;
5113 self.maybe_compact();
5114 // P4b (§1.7, pi§3 "steer = after current tool calls"): drain any
5115 // queued mid-turn steering message(s) BEFORE building the next
5116 // request — the top of every loop iteration is exactly "after
5117 // whatever tool calls the previous iteration just ran" (or, on
5118 // the very first iteration, before anything has happened yet,
5119 // which is an equally valid "deliver immediately" reading).
5120 // Empty queue (today's default state) is a no-op.
5121 let (steer_msg, steer_taken) = {
5122 let mut inbox = self
5123 .steer_queue
5124 .lock()
5125 .unwrap_or_else(std::sync::PoisonError::into_inner);
5126 let before = inbox.len();
5127 let drained = inbox.drain(self.config.steering_mode);
5128 let taken = before - inbox.len();
5129 (drained, taken)
5130 };
5131 if let Some(steer_msg) = steer_msg {
5132 // BP-8 (catalog:154): the queue record's other half —
5133 // without it a replayed journal would keep re-delivering an
5134 // input the conversation already consumed.
5135 self.journal_queue_drain(crate::session_journal::QueueKind::Steer, steer_taken);
5136 let msg = ChatMessage::user(steer_msg);
5137 self.record(&msg)?;
5138 self.history.push(msg);
5139 }
5140 // Recomputed every iteration (not hoisted): under `Deferred`
5141 // advertising, a `tool_search` call earlier in this same loop
5142 // activates tools that must be advertised starting with the very
5143 // next request (B6).
5144 let tools = self.tool_schemas();
5145 let messages = self.build_request_messages();
5146
5147 // PARITY-18 D4 — re-check the context guard before EVERY
5148 // request this loop builds, not just the caller's one-shot
5149 // preflight: interactive turns 2+, `/expand all`, and any
5150 // mid-loop tool round-trip that grows `messages` can push a
5151 // barely-passing session over the limit between sends. Only
5152 // armed when a caller has opted in via `set_context_limit`.
5153 // Uses the exact same `tokens::context_guard`
5154 // formula the CLI preflight uses, so the two can never disagree.
5155 if let Some(limit) = self.context_limit {
5156 let (fits, projected_tokens) =
5157 supercode_runtime::context_guard(&messages, &tools, limit);
5158 if !fits {
5159 return Err(Error::ContextLimitExceeded {
5160 projected_tokens,
5161 reserve_tokens: supercode_runtime::CONTEXT_RESPONSE_RESERVE_TOKENS,
5162 context_limit: limit,
5163 model: self.config.model.clone(),
5164 });
5165 }
5166 }
5167
5168 // BP-7 (catalog §4a "Turn/step bracketing records"): the
5169 // OPENING bracket, written before the request is issued so it
5170 // survives a request that never returns (a cancelled turn keeps
5171 // its `context` marker with no `usage`/`finish` after it).
5172 // Uses `tokens::estimate_request_tokens` — the same estimator
5173 // the context guard above uses, so the two can never disagree.
5174 self.push_turn_marker_at(
5175 round_trip,
5176 crate::turn_record::TurnMarker::Context {
5177 messages: messages.len(),
5178 tools: tools.len(),
5179 estimated_tokens: supercode_runtime::estimate_request_tokens(&messages, &tools),
5180 },
5181 );
5182
5183 let mut req = self.chat_request(messages, tools);
5184
5185 let fallback_hops: Vec<FallbackHop>;
5186 let completion = {
5187 let sink = self.config.event_sink.as_ref();
5188 let on_delta = move |s: &str| {
5189 if let Some(sink) = sink {
5190 sink(AgentEvent::TextDelta(s.to_string()));
5191 }
5192 };
5193 // PARITY-18 D3 — the real send site: flip the flag
5194 // immediately before issuing the request, regardless of
5195 // whether `complete` then succeeds or fails, so
5196 // `request_issued()` truthfully reflects "a live request
5197 // was attempted" rather than "the run reached this line and
5198 // later succeeded."
5199 self.requests_issued = true;
5200 // P4b (§1.1/§3.1 `core.retry`, pi§3 shape): retry-with-
5201 // backoff already lives at the TRANSPORT layer
5202 // (`provider::OpenAiProvider::send_with_retry`, pre-existing
5203 // — connection failures and 5xx responses are retried
5204 // there); `Config.retry_*` (see `Agent::new`) makes that
5205 // EXISTING mechanism config-file-settable instead of
5206 // duplicating a second retry loop here, which would nest
5207 // retries confusingly on top of the transport's own.
5208 // BP-13 (D9 "Failure fallback model chains"): the chain is
5209 // EXECUTED here, not merely resolved. On a failure another
5210 // model could plausibly answer (overload / rate limit /
5211 // unavailability — `is_failover_worthy`), the request is
5212 // re-sent against the next entry of
5213 // `Config::model_fallback`, with routing re-applied for
5214 // that model. A 4xx that is not a rate limit is the caller's
5215 // problem, not the model's, and is never retried elsewhere.
5216 // The transport-level retry above has already run and given
5217 // up by the time a hop is considered.
5218 let (result, hops) = self.complete_with_fallback(&mut req, &on_delta).await;
5219 fallback_hops = hops;
5220 result
5221 };
5222 // BP-7: drained whether the request succeeded or failed, and
5223 // BEFORE the `?` — a request that exhausted its retries and
5224 // then errored is exactly the case a retry record exists for.
5225 for notice in self.retry_log.drain() {
5226 self.emit(AgentEvent::ProviderRetry {
5227 attempt: notice.attempt,
5228 delay_ms: notice.delay_ms,
5229 reason: notice.reason.clone(),
5230 });
5231 self.push_turn_marker_at(
5232 round_trip,
5233 crate::turn_record::TurnMarker::Retry {
5234 attempt: notice.attempt,
5235 delay_ms: notice.delay_ms,
5236 reason: notice.reason,
5237 },
5238 );
5239 }
5240 // The switch the fallback pass performed is a real mid-session
5241 // model change: it moves `Config::model` for every subsequent
5242 // request and is recorded exactly like a user-driven `/model`
5243 // switch (typed record + journal line), never as a silent retry.
5244 for hop in fallback_hops {
5245 self.record_model_change(&hop.from, &hop.to, Some(hop.reason.as_str()));
5246 }
5247 let (mut assistant, usage) = completion?;
5248 // Persist the actual generating model on the message itself.
5249 // A resumed foreign session keeps its original model in
5250 // `SessionMeta`; using only that session-level value on export
5251 // misattributes every Supercode continuation turn to the source
5252 // harness model. Per-message provenance lets native exporters
5253 // preserve the boundary accurately (for example, Claude history
5254 // followed by a GLM continuation).
5255 assistant
5256 .metadata
5257 .insert("model".to_string(), self.config.model.clone());
5258 output_tokens_used += usage.completion_tokens;
5259 self.total_output_tokens += usage.completion_tokens;
5260
5261 // UX-26 (B7-warn): consult the verdict computed at build time
5262 // (before this request was sent) now that `usage` — the only
5263 // piece that couldn't be known pre-send — is in hand. Gated on
5264 // `Config::cache_warnings` (default on; `--no-cache-warnings` /
5265 // `SUPERCODE_CACHE_WARNINGS=0` at the CLI layer, dev/03) so this
5266 // stays a zero-behavior-change no-op for every caller that
5267 // hasn't opted into `CachePlan::ImportedPrefix` in the first
5268 // place (`pending_cache_turn.0` is `false` whenever
5269 // `CachePlan::Off`, so the predicate always returns `None` then
5270 // regardless of this flag).
5271 let (will_annotate, cache_established, idle_secs) = self.pending_cache_turn;
5272 // UX-26 T2 (accuracy fold-in): `CacheColdReason::message` asserts
5273 // Anthropic-specific facts (a fixed 5-minute ephemeral TTL, and
5274 // cache-read-ratio semantics that assume Anthropic's exact-count
5275 // billing) that are only true for Anthropic-family models. This
5276 // is a WARNING-only gate, deliberately not folded into
5277 // `will_annotate`/the breakpoint-placement gate above: whether a
5278 // `cache_control` breakpoint is safe/inert to send to a
5279 // non-Anthropic model through OpenRouter is a separate cache-
5280 // behavior question this ticket doesn't touch (see
5281 // `.volter/tracker/markdown/UX-26.md`'s T2 note) — narrowing only
5282 // the warning keeps this fix scoped to warning ACCURACY, with
5283 // zero change to what gets sent on the wire.
5284 let warning_applies_to_this_model =
5285 provider::is_anthropic_family_model(&self.config.model);
5286 if self.config.cache_warnings && warning_applies_to_this_model {
5287 if let Some(reason) =
5288 provider::cache_cold_reason(will_annotate, cache_established, idle_secs, &usage)
5289 {
5290 self.emit(AgentEvent::CacheWarning {
5291 message: reason.message(),
5292 });
5293 }
5294 }
5295 // Refresh the activity clock / establish-once flag for the NEXT
5296 // turn's comparison, but only when THIS request actually carried
5297 // the annotation — an unannotated (busted/Off) request neither
5298 // warms nor cools a cache entry it never touched.
5299 if will_annotate {
5300 self.last_cache_activity_ms = Some(now_ms());
5301 self.cache_established = true;
5302 }
5303
5304 // UX-23: emitted before `TurnCompleted` so a `--trace`/
5305 // `stream-json` consumer sees "this round-trip cost N tokens"
5306 // land right alongside the round-trip it describes, rather than
5307 // needing to correlate it with a later event.
5308 self.emit(AgentEvent::Usage(usage.clone()));
5309 self.emit(AgentEvent::TurnCompleted);
5310 // P4b (§1.6, catalog §4a "persisted per-turn usage records"):
5311 // EventSink already streamed `Usage` above — this durably
5312 // accumulates the same data as a typed record (see
5313 // `Self::usage_records`/`Self::save_usage_log`), never a lossy
5314 // display-only channel.
5315 // BP-7 (catalog §4a "Per-turn cost/usage accounting"): the
5316 // record now carries the round-trip's DOLLAR cost too — the
5317 // half the row's semantics name alongside tokens — whenever
5318 // this build can price the model.
5319 // BP-13 (D9 "Model-served-vs-requested provenance"): the record
5320 // now carries BOTH sides — the model this agent asked for and,
5321 // when the provider reported one, the model that actually
5322 // answered. They can genuinely differ (a gateway aliasing a
5323 // name to a dated snapshot, a fallback hop, a routed tier), and
5324 // a record that can only ever state the request cannot show it.
5325 let served = assistant
5326 .metadata
5327 .get(crate::provider::SERVED_MODEL_KEY)
5328 .cloned();
5329 let record = crate::usage_log::UsageRecord::from_usage(
5330 self.turn_index,
5331 &self.config.model,
5332 &usage,
5333 now_ms(),
5334 )
5335 .priced(self.model_price)
5336 .with_served_model(served);
5337 self.total_cost_usd += record.cost_usd.unwrap_or(0.0);
5338 // BP-7: the round-trip's usage bracket, written from the same
5339 // point as the usage record so the two logs never disagree.
5340 self.push_turn_marker_at(
5341 round_trip,
5342 crate::turn_record::TurnMarker::Usage {
5343 prompt_tokens: record.prompt_tokens,
5344 completion_tokens: record.completion_tokens,
5345 total_tokens: record.total_tokens,
5346 cached_tokens: record.cached_tokens,
5347 cost_usd: record.cost_usd,
5348 },
5349 );
5350 self.journal_usage(&record);
5351 self.usage_log.push(record);
5352 self.turn_index += 1;
5353 self.record(&assistant)?;
5354 self.history.push(assistant.clone());
5355
5356 let calls = assistant.tool_calls().to_vec();
5357 if calls.is_empty() {
5358 // Close steering acceptance under the same lock as the last
5359 // drain. A message accepted before this boundary extends the
5360 // current turn; anything later is rejected by the SDK and
5361 // can never leak into a future turn.
5362 let (steer_msg, steer_taken) = {
5363 let mut inbox = self
5364 .steer_queue
5365 .lock()
5366 .unwrap_or_else(std::sync::PoisonError::into_inner);
5367 let before = inbox.len();
5368 let drained = inbox.drain_or_close(self.config.steering_mode);
5369 let taken = before - inbox.len();
5370 (drained, taken)
5371 };
5372 if let Some(steer_msg) = steer_msg {
5373 self.journal_queue_drain(crate::session_journal::QueueKind::Steer, steer_taken);
5374 let msg = ChatMessage::user(steer_msg);
5375 self.record(&msg)?;
5376 self.history.push(msg);
5377 continue;
5378 }
5379 // P4b (§1.7, pi§3 "follow-up = at idle"): a queued follow-up
5380 // message takes priority over the stop-gate — it's more
5381 // input to answer, not a veto of an answer already given.
5382 let follow_up_before = self.follow_up_queue.len();
5383 if let Some(follow_up_msg) =
5384 Self::drain_steer_queue(&mut self.follow_up_queue, self.config.follow_up_mode)
5385 {
5386 self.journal_queue_drain(
5387 crate::session_journal::QueueKind::FollowUp,
5388 follow_up_before - self.follow_up_queue.len(),
5389 );
5390 let msg = ChatMessage::user(follow_up_msg);
5391 self.record(&msg)?;
5392 self.history.push(msg);
5393 continue;
5394 }
5395 // P4b (§1.9/§3.1 `[core] stop_gate`, D3 "stop/completion
5396 // gating"): consulted exactly once per iteration that would
5397 // otherwise return — computed into an owned `Option<String>`
5398 // so the immutable borrow of `self.config.stop_gate` ends
5399 // before the `self.record`/`self.history.push` calls below
5400 // need `&mut self`.
5401 let final_content = assistant.content.clone().unwrap_or_default();
5402 let veto_reason: Option<String> = self
5403 .config
5404 .stop_gate
5405 .as_ref()
5406 .and_then(|gate| gate(&final_content));
5407 if let Some(reason) = veto_reason {
5408 let msg = ChatMessage::user(reason);
5409 self.record(&msg)?;
5410 self.history.push(msg);
5411 continue;
5412 }
5413 // BP-8 (catalog:156): the last iteration's tool calls are
5414 // the ones the top-of-loop flush above never sees.
5415 self.journal_plan_if_changed();
5416 self.push_turn_marker_at(
5417 round_trip,
5418 crate::turn_record::TurnMarker::Finish {
5419 reason: crate::turn_record::FinishReason::EndTurn,
5420 },
5421 );
5422 return Ok(assistant.content.unwrap_or_default());
5423 }
5424
5425 // BP-7: this round-trip ended by asking for tool calls; the
5426 // loop continues. The budget arms below mark the LOOP's end
5427 // separately when one of them stops it here.
5428 self.push_turn_marker_at(
5429 round_trip,
5430 crate::turn_record::TurnMarker::Finish {
5431 reason: crate::turn_record::FinishReason::ToolCalls,
5432 },
5433 );
5434
5435 // Output-token budget (output only — input tokens are not counted,
5436 // so this does not bound cost): stop spawning further model turns
5437 // once the cumulative output-token budget for this `send` is
5438 // exhausted.
5439 if let Some(budget) = self.config.max_total_output_tokens {
5440 if output_tokens_used >= budget {
5441 // The assistant turn we just pushed carries unanswered
5442 // tool_calls. Leaving them dangling yields an invalid
5443 // history (assistant tool_calls with no tool results) that
5444 // the provider rejects on the next `send`/resume. Emit
5445 // synthetic results so the transcript stays well-formed.
5446 for call in &calls {
5447 let msg = ChatMessage::tool_result(
5448 call.id.clone(),
5449 call.function.name.clone(),
5450 "[skipped: output token budget reached]".to_string(),
5451 );
5452 self.record(&msg)?;
5453 self.history.push(msg);
5454 }
5455 self.push_turn_marker_at(
5456 round_trip,
5457 crate::turn_record::TurnMarker::Finish {
5458 reason: crate::turn_record::FinishReason::OutputTokenBudget,
5459 },
5460 );
5461 return Ok(assistant.content.clone().unwrap_or_default());
5462 }
5463 }
5464
5465 // BP-7 (catalog §4a "Turn/budget caps"): the SPEND and STEP
5466 // caps, at the same point and with the same shape as the
5467 // output-token cap above — checked before this turn's tool
5468 // calls run, with synthetic results so the transcript stays
5469 // well-formed for a resume.
5470 let spend_exhausted = self
5471 .config
5472 .max_budget_usd
5473 .is_some_and(|b| b > 0.0 && self.total_cost_usd >= b);
5474 let steps_exhausted = self
5475 .config
5476 .max_steps
5477 .is_some_and(|n| n > 0 && self.total_steps + calls.len() > n);
5478 if spend_exhausted || steps_exhausted {
5479 let (label, reason) = if spend_exhausted {
5480 (
5481 "[skipped: spend budget reached]",
5482 crate::turn_record::FinishReason::SpendBudget,
5483 )
5484 } else {
5485 (
5486 "[skipped: step budget reached]",
5487 crate::turn_record::FinishReason::StepBudget,
5488 )
5489 };
5490 for call in &calls {
5491 let msg = ChatMessage::tool_result(
5492 call.id.clone(),
5493 call.function.name.clone(),
5494 label.to_string(),
5495 );
5496 self.record(&msg)?;
5497 self.history.push(msg);
5498 }
5499 self.push_turn_marker_at(
5500 round_trip,
5501 crate::turn_record::TurnMarker::Finish { reason },
5502 );
5503 return Ok(assistant.content.clone().unwrap_or_default());
5504 }
5505 self.total_steps += calls.len();
5506
5507 // P4e (§3.1 `core.parallel_tool_calls`, catalog:59): off (the
5508 // default) or a single call takes the EXACT pre-P4e sequential
5509 // path below, byte-identical. Only `true` with 2+ calls in this
5510 // turn takes `Self::run_tools_concurrently` — see its doc
5511 // comment for exactly what does and doesn't run concurrently.
5512 if self.config.parallel_tool_calls && calls.len() > 1 {
5513 for call in &calls {
5514 self.emit(AgentEvent::tool_started(call));
5515 }
5516 let results = self.run_tools_concurrently(&calls).await;
5517 for (call, (output, is_error)) in calls.iter().zip(results) {
5518 self.emit(AgentEvent::ToolCallCompleted {
5519 id: call.id.clone(),
5520 name: call.function.name.clone(),
5521 output: output.clone(),
5522 is_error,
5523 });
5524 self.apply_tool_result(call, output, is_error)?;
5525 }
5526 } else {
5527 for call in &calls {
5528 self.emit(AgentEvent::tool_started(call));
5529 let (output, is_error) = self.run_tool(call).await;
5530 self.emit(AgentEvent::ToolCallCompleted {
5531 id: call.id.clone(),
5532 name: call.function.name.clone(),
5533 output: output.clone(),
5534 is_error,
5535 });
5536 self.apply_tool_result(call, output, is_error)?;
5537 }
5538 }
5539 // BP-3 (catalog row "Context-budget tools"): a `new_context`
5540 // call parks its request on the shared budget; this is where
5541 // the agent — the one owner of `history` — applies it, so the
5542 // NEXT request built by this loop is already the fresh window.
5543 // No parked request (every session that never calls the tool)
5544 // is a single `Option` check.
5545 self.apply_pending_new_context();
5546 }
5547
5548 self.push_turn_marker_at(
5549 self.turn_index.saturating_sub(1),
5550 crate::turn_record::TurnMarker::Finish {
5551 reason: crate::turn_record::FinishReason::MaxIterations,
5552 },
5553 );
5554 Err(Error::MaxIterations(self.config.max_iterations))
5555 }
5556
5557 /// BP-3: apply a parked [`crate::tools::NewContextRequest`], if any.
5558 ///
5559 /// The rewrite itself is BP-4's [`Self::new_context`] — the SAME
5560 /// mechanism the operator's `/handoff` runs, so the model's door and the
5561 /// human's door can never drift into two different notions of "a fresh
5562 /// window". This function is only the hand-off point between the tool
5563 /// that asked and the agent that owns `history`.
5564 fn apply_pending_new_context(&mut self) {
5565 let Some(request) = self.ctx.context_budget.take_new_context() else {
5566 return;
5567 };
5568 self.new_context(&request.objective, request.keep_recent);
5569 }
5570
5571 /// The exact post-execution handling every tool result gets, regardless
5572 /// of whether it was produced by the sequential loop or
5573 /// [`Self::run_tools_concurrently`] — factored out of `Self::run_loop`'s
5574 /// tool-dispatch section (P4e) so both paths share one copy: multimodal
5575 /// image-marker detection, A7 output capping (gated exactly as before),
5576 /// TR-10 error stamping, and the `record`/`history` append. Always
5577 /// called in ORIGINAL call order, one call at a time, so the lossless
5578 /// sidecar's append-order invariant (S1.13) holds regardless of which
5579 /// dispatch path produced the result.
5580 fn apply_tool_result(
5581 &mut self,
5582 call: &supercode_interchange::ToolCall,
5583 output: String,
5584 is_error: bool,
5585 ) -> Result<()> {
5586 // P4c (§1.2 `core.tools.read_file.multimodal` / `view_image`):
5587 // a successful tool result carrying the image-data-URL
5588 // marker becomes a `content_parts` image block instead of
5589 // plain text — checked BEFORE `cap_tool_output` (a data URL
5590 // is not meaningfully "capped" by a byte-length text notice)
5591 // and recorded identically on both the full and history
5592 // copies, mirroring `ImageRedacted`'s "images are their own
5593 // axis, orthogonal to A7 text truncation" treatment
5594 // (reduce.rs). An ERRORED call never carries the marker (a
5595 // tool only emits it on success), so `is_error` is not
5596 // re-checked here.
5597 if let Some(data_url) = output.strip_prefix(crate::tools::MULTIMODAL_IMAGE_MARKER) {
5598 let notice = format!("[{}: image content attached below]", call.function.name);
5599 let full_result = ChatMessage::tool_result_with_image(
5600 call.id.clone(),
5601 call.function.name.clone(),
5602 notice.clone(),
5603 data_url.to_string(),
5604 );
5605 let hist_result = ChatMessage::tool_result_with_image(
5606 call.id.clone(),
5607 call.function.name.clone(),
5608 notice,
5609 data_url.to_string(),
5610 );
5611 self.record(&full_result)?;
5612 self.history.push(hist_result);
5613 return Ok(());
5614 }
5615 // Record the FULL output before capping (A3): what the
5616 // sidecar keeps must never be the already-lossy, truncated
5617 // copy (#8/#40) — `history` alone governs what shrinks.
5618 let mut full_result =
5619 ChatMessage::tool_result(call.id.clone(), call.function.name.clone(), output.clone());
5620 // D6/A7 supersession gate (TR-12 land-blocker fix): `history`
5621 // is the exact slice `reduce::project_messages` mints A7/A10
5622 // reduction hashes from (`Self::build_request_messages`
5623 // below). Capping it here — as this unconditionally used to
5624 // do — would silently shrink the bytes those hashes cover, so
5625 // a hash minted now could never recompute the same way once
5626 // the sidecar is reloaded from disk later (`verify_log`/
5627 // `invert`, offline). Gate `cap_tool_output` off in exactly
5628 // the combination where reductions can be minted over
5629 // `history` AND the full bytes are durably retained: a
5630 // recorder AND a `ReductionPolicy` both installed. A7 then
5631 // owns tool-output bounding, reversibly, at projection time
5632 // (SPEC.md D6/A7) — `history`/the sidecar keep everything,
5633 // only the request view shrinks. With a policy but no
5634 // recorder (constructible via `set_reduction_policy` alone),
5635 // nothing durable backs the full bytes, so capping stays on —
5636 // the same honest-labeling spirit as `cap_tool_output`'s own
5637 // retention branch below, just applied at the gate instead of
5638 // the notice text. With no policy at all, this is untouched:
5639 // today's byte-identical legacy cap.
5640 let for_history = if self.recorder.is_some() && self.reduction_policy.is_some() {
5641 output
5642 } else {
5643 self.cap_tool_output(output)
5644 };
5645 let mut hist_result =
5646 ChatMessage::tool_result(call.id.clone(), call.function.name.clone(), for_history);
5647 if is_error {
5648 // TR-10: the reduction layer's success/failure boundary
5649 // (`ReductionKind::ToolInputElided` must never target an
5650 // errored call — TR-6's territory) has no other
5651 // structural signal on `ChatMessage`; stamp both the
5652 // recorded copy (so it survives a sidecar round-trip via
5653 // `NativeTurn`) and the live-history copy (so an
5654 // in-process `project_messages` sees it immediately).
5655 reduce::mark_tool_error(&mut full_result);
5656 reduce::mark_tool_error(&mut hist_result);
5657 }
5658 self.record(&full_result)?;
5659 self.history.push(hist_result);
5660 Ok(())
5661 }
5662
5663 /// Truncate an oversized tool result so a single runaway command can't blow
5664 /// up the context window. Cuts on a char boundary and appends a notice.
5665 fn cap_tool_output(&self, output: String) -> String {
5666 let Some(max) = self.config.max_tool_output_bytes else {
5667 return output;
5668 };
5669 if max == 0 || output.len() <= max {
5670 return output;
5671 }
5672 // Find the largest char boundary <= max.
5673 let mut end = max;
5674 while end > 0 && !output.is_char_boundary(end) {
5675 end -= 1;
5676 }
5677 let total = output.len();
5678 let mut s = output[..end].to_string();
5679 // BP-2 (catalog:58, `core.tool_output_spill`): write the full bytes
5680 // to a per-session file the model can read back. Off (the default)
5681 // leaves the notice byte-identical to before.
5682 let spill = if self.config.tool_output_spill {
5683 self.spill_tool_output(&output)
5684 } else {
5685 None
5686 };
5687 // Honest retention labeling (D6, B10-AC4): only claim the sidecar has
5688 // the full output when a recorder is actually installed — or, BP-2,
5689 // that the spill file has it when one was actually written.
5690 let retention = if self.recorder.is_some() {
5691 "full output in session sidecar"
5692 } else if spill.is_some() {
5693 "full output on disk"
5694 } else {
5695 "full output not retained"
5696 };
5697 let recovery = match &spill {
5698 // The door is named in the notice, so it works WITHOUT
5699 // `capabilities.reduction`: under a preset with a read tool
5700 // that is `read_file`; under a shell-only preset (cx-parity,
5701 // whose whole read pathway is the shell) it is `cat`.
5702 Some(path) => {
5703 let door = if self.registry.get("read_file").is_some() {
5704 "read it with `read_file`"
5705 } else {
5706 "read it with `cat`"
5707 };
5708 format!("; full output spilled to {} — {door}", path.display())
5709 }
5710 None => String::new(),
5711 };
5712 s.push_str(&format!(
5713 "{CAP_NOTICE_MARKER}{total} bytes total, showing first {end}; {retention}{recovery}]"
5714 ));
5715 s
5716 }
5717
5718 /// BP-2 (catalog:58 "Oversized output truncated; full content kept
5719 /// reachable"): write `full` to this session's spill directory and
5720 /// return the path, or `None` if it could not be written (a spill is a
5721 /// recovery convenience — it must never fail the tool call).
5722 ///
5723 /// The file is named by content hash, so the same output spilled twice
5724 /// costs one file and a re-run of an identical command reuses it.
5725 fn spill_tool_output(&self, full: &str) -> Option<std::path::PathBuf> {
5726 let dir = self.spill_dir();
5727 std::fs::create_dir_all(&dir).ok()?;
5728 let digest = blake3::hash(full.as_bytes()).to_hex();
5729 let path = dir.join(format!("tool-output-{}.txt", &digest[..16]));
5730 if !path.exists() {
5731 std::fs::write(&path, full).ok()?;
5732 }
5733 Some(path)
5734 }
5735
5736 /// BP-2: where this agent's spilled outputs live — beside the session
5737 /// sidecar when one is recording (per-SESSION, the same identity the
5738 /// sidecar has), else a per-PROCESS temp directory, which is as
5739 /// specific as an agent with no sidecar can honestly be.
5740 fn spill_dir(&self) -> std::path::PathBuf {
5741 if let Some(recorder) = &self.recorder {
5742 let path = recorder.path();
5743 if let (Some(parent), Some(stem)) = (path.parent(), path.file_stem()) {
5744 return parent.join(format!("{}.spill", stem.to_string_lossy()));
5745 }
5746 }
5747 std::env::temp_dir().join(format!("supercode-spill-{}", std::process::id()))
5748 }
5749
5750 /// P5-3 note on the signature: written as a plain fn returning an
5751 /// explicitly boxed future (`Pin<Box<dyn Future + Send>>`) rather than
5752 /// as `async fn`. `spawn_subagent` makes this function genuinely
5753 /// recursive at the TYPE level: `run_tool` -> `run_spawn_subagent` ->
5754 /// (a child) `Agent::send` -> `run_loop` -> `run_tool` again — an
5755 /// `async fn`'s return type is an anonymous, compiler-inferred
5756 /// self-referential state machine, and inferring one that embeds
5757 /// itself (even indirectly, through several other functions) is a
5758 /// compile error (an infinitely-sized/cyclic opaque type). Declaring
5759 /// `run_tool`'s return type EXPLICITLY as a boxed trait object breaks
5760 /// the cycle: every other function on the call graph now embeds a
5761 /// concrete, already-known type here instead of one the compiler would
5762 /// otherwise need to (cyclically) infer. Callers are unaffected —
5763 /// `self.run_tool(call).await` reads identically either way.
5764 fn run_tool<'a>(
5765 &'a mut self,
5766 call: &'a supercode_interchange::ToolCall,
5767 ) -> std::pin::Pin<Box<dyn std::future::Future<Output = (String, bool)> + Send + 'a>> {
5768 Box::pin(async move {
5769 let translated_builtin = if self.config.claude_runtime_tools_enabled {
5770 match self.translate_claude_builtin_call(call) {
5771 Ok(translated) => translated,
5772 Err(error) => return (format!("Error: {error}"), true),
5773 }
5774 } else {
5775 None
5776 };
5777 let call = translated_builtin.as_ref().unwrap_or(call);
5778 if self.config.claude_runtime_tools_enabled
5779 && matches!(
5780 call.function.name.as_str(),
5781 CLAUDE_CRON_CREATE
5782 | CLAUDE_CRON_DELETE
5783 | CLAUDE_CRON_LIST
5784 | CLAUDE_SCHEDULE_WAKEUP
5785 )
5786 {
5787 return self.run_claude_runtime_tool(call);
5788 }
5789 // P5-3: `spawn_subagent`/`subagent_status` need full async
5790 // `&mut self` access (running a child agent's loop, or
5791 // awaiting an already-finished background `JoinHandle`) —
5792 // `prepare_tool_call` is purely synchronous, so these are
5793 // intercepted HERE, one level above it, rather than inside it
5794 // like `TOOL_SEARCH`/`EXPAND_REDUCTION`/`SIDECAR_SEARCH`.
5795 if call.function.name == CLAUDE_AGENT && self.config.subagents_claude_agent_alias {
5796 return match self.translate_claude_agent_call(call) {
5797 Ok(translated) => self.run_spawn_subagent(&translated).await,
5798 Err(error) => (format!("Error: {error}"), true),
5799 };
5800 }
5801 if call.function.name == SPAWN_SUBAGENT {
5802 return self.run_spawn_subagent(call).await;
5803 }
5804 // P5-3 safety-hardening fix (Fable-5 review, LOW "wrong error
5805 // when disabled"): gated on `subagents_enabled`, matching
5806 // `run_spawn_subagent`'s own already-correct disabled behavior
5807 // (that one gates INTERNALLY, at its own top; this one gates
5808 // HERE, at the interception point, because unlike
5809 // `spawn_subagent` it has no other reason to run any logic at
5810 // all when subagents are off). When disabled, a hallucinated
5811 // `subagent_status` call must NOT be intercepted — it falls
5812 // through to `prepare_tool_call`'s normal unknown-tool path
5813 // below, which returns `Error::UnknownTool("subagent_status")`,
5814 // byte-identical to the pre-P5-3 (and disabled-spawn_subagent)
5815 // error text — never `Error::SubagentNotFound`'s "unknown
5816 // subagent id" text, which would wrongly imply subagents are on
5817 // but this particular id is bogus.
5818 if call.function.name == SUBAGENT_STATUS && self.config.subagents_enabled {
5819 return self.run_subagent_status(call).await;
5820 }
5821 // BP-7: same interception shape and same `subagents_enabled`
5822 // gate as `SUBAGENT_STATUS` above — when the module is off a
5823 // hallucinated call falls through to the ordinary unknown-tool
5824 // error rather than a misleading "unknown subagent id".
5825 if call.function.name == SEND_MESSAGE && self.config.subagents_enabled {
5826 return self.run_send_message(call).await;
5827 }
5828 if call.function.name == SUBAGENT_RESUME && self.config.subagents_enabled {
5829 return self.run_subagent_resume(call).await;
5830 }
5831 match self.prepare_tool_call(call) {
5832 PreparedCall::Done(result) => result,
5833 PreparedCall::Ready { name, args } => {
5834 // `prepare_tool_call` already confirmed the registry has
5835 // this tool.
5836 let tool = self.registry.get(&name).expect("prepared as Ready");
5837 let (output, is_error) = match tool.execute(args, &self.ctx).await {
5838 Ok(out) => (out, false),
5839 Err(e) => (format!("Error: {e}"), true),
5840 };
5841 if let Some(hook) = &self.config.post_tool_hook {
5842 hook(&name, &output, is_error);
5843 }
5844 (output, is_error)
5845 }
5846 }
5847 })
5848 }
5849
5850 /// P4e (§3.1 `core.parallel_tool_calls`, catalog:59): the SYNCHRONOUS
5851 /// half of dispatching one tool call — everything `Self::run_tool` did
5852 /// BEFORE its single `tool.execute(...).await`, factored out so
5853 /// [`Self::run_tools_concurrently`] can run these cheap, stateful,
5854 /// `&mut self` checks (agent intrinsics, unknown-tool, approval,
5855 /// doom-loop, pre-tool-hook) SEQUENTIALLY and in ORIGINAL call order —
5856 /// exactly as `run_tool` always has — before handing the remaining
5857 /// calls' `execute()` futures to `join_all`. `Self::run_tool` itself is
5858 /// now a thin wrapper over this (a pure refactor: byte-identical
5859 /// observable behavior, verified by the existing test suite).
5860 fn prepare_tool_call(&mut self, call: &supercode_interchange::ToolCall) -> PreparedCall {
5861 let name = &call.function.name;
5862 // BP-3 (catalog row "Context-budget tools"): hand the model's
5863 // `get_context_remaining` the agent's OWN accounting — BP-4's
5864 // [`Self::context_usage`], the same struct `/context` prints and
5865 // the same estimates the context guard enforces, so what the model
5866 // reads and what refuses an oversized turn can never disagree.
5867 // Computed at the moment the question is asked (the freshest
5868 // possible view) and ONLY then: `context_usage` projects the whole
5869 // request view, which is not a cost to pay on unrelated calls.
5870 if name == crate::tools::GET_CONTEXT_REMAINING {
5871 if let Ok(usage) = serde_json::to_value(self.context_usage()) {
5872 self.ctx.context_budget.publish(usage);
5873 }
5874 }
5875 if name == TOOL_SEARCH {
5876 // Agent intrinsic (B6): intercepted before registry lookup, since
5877 // `Tool::execute` has no access to the registry or `activated_tools`.
5878 return PreparedCall::Done(self.run_tool_search(call));
5879 }
5880 if name == EXPAND_REDUCTION {
5881 // Agent intrinsic (T12/TR-1): intercepted before registry lookup,
5882 // same reason — resolves against `self.reduction_log`/`self.history`,
5883 // which `Tool::execute` has no access to.
5884 return PreparedCall::Done(self.run_expand_reduction(call));
5885 }
5886 if name == SIDECAR_SEARCH {
5887 return PreparedCall::Done(self.run_sidecar_search(call));
5888 }
5889 // P5-6 (§2 module 4 `tools.background`): gated at the interception
5890 // point itself (not internally, at each method's own top) —
5891 // mirroring `SUBAGENT_STATUS`'s own fix (Fable-5 review, LOW "wrong
5892 // error when disabled"): a hallucinated call when the module is off
5893 // must fall through to the plain `Error::UnknownTool` path below,
5894 // never a background-specific error that would wrongly imply the
5895 // module is on. Unlike `SPAWN_SUBAGENT`/`SUBAGENT_STATUS`, none of
5896 // these four need async `&mut self` access (spawning a process,
5897 // `Child::try_wait`, and `Child::start_kill` are all synchronous),
5898 // so they're intercepted here in `prepare_tool_call` rather than in
5899 // `Self::run_tool`.
5900 if self.config.tools_background_enabled {
5901 if name == BACKGROUND_EXEC {
5902 return PreparedCall::Done(self.run_background_exec(call));
5903 }
5904 if name == BACKGROUND_STATUS {
5905 return PreparedCall::Done(self.run_background_status(call));
5906 }
5907 if name == BACKGROUND_LIST {
5908 return PreparedCall::Done(self.run_background_list(call));
5909 }
5910 if name == BACKGROUND_KILL {
5911 return PreparedCall::Done(self.run_background_kill(call));
5912 }
5913 }
5914 if self.registry.get(name).is_none() {
5915 let err = Error::UnknownTool(name.clone());
5916 return PreparedCall::Done((format!("Error: {err}"), true));
5917 }
5918
5919 // P5-1 (§2 modules 10-11, integration point named in
5920 // COMPOSABLE-HARNESS-DESIGN.md's activation set): the permissions
5921 // ENGINE governs the gate when `capabilities.permissions.enabled`
5922 // is on; every other config resolves this to `false`
5923 // (`Config::default`), which takes the `else` branch below —
5924 // the EXACT pre-P5-1 code, untouched, so the default posture
5925 // (approval=never/sandbox=none) and every existing test's observed
5926 // behavior is byte-for-byte unchanged.
5927 if self.config.permissions_enabled {
5928 // The engine needs the command/path TEXT the legacy tool-name-
5929 // only gate below never looked at, so args must be parsed
5930 // BEFORE the gate here (not after, like the legacy branch).
5931 let args = match call.function.parsed_arguments() {
5932 Ok(v) => v,
5933 Err(e) => {
5934 let err = Error::InvalidArguments {
5935 tool: name.clone(),
5936 message: e.to_string(),
5937 };
5938 return PreparedCall::Done((format!("Error: {err}"), true));
5939 }
5940 };
5941 // P4c doom-loop breaker, unchanged, still before any gate.
5942 if let Some(reason) = self.check_doom_loop(name, &args) {
5943 return PreparedCall::Done((format!("Error: {reason}"), true));
5944 }
5945 // BP-10: the hook runs BEFORE the engine on this path, so its
5946 // rewrite is what the rules see and its allow/ask/deny is a
5947 // tier inside them — see `run_pre_tool_hook`.
5948 let (args, hook_decision) = match self.run_pre_tool_hook(name, args) {
5949 Ok(pair) => pair,
5950 Err(done) => return done,
5951 };
5952 if let Some(reason) = self.permissions_gate_denial(name, &args, hook_decision) {
5953 return PreparedCall::Done((format!("Error: {reason}"), true));
5954 }
5955 PreparedCall::Ready {
5956 name: name.clone(),
5957 args,
5958 }
5959 } else {
5960 // ---- pre-P5-1 gate, byte-for-byte unchanged ----
5961 // Approval gate: if the policy requires it, consult the handler
5962 // (absent handler denies, so an OnRequest/Untrusted policy is
5963 // fail-closed).
5964 if self.config.needs_approval(name) {
5965 let approved = self
5966 .config
5967 .approval_handler
5968 .as_ref()
5969 .map(|h| h(call))
5970 .unwrap_or(false);
5971 if !approved {
5972 return PreparedCall::Done((
5973 format!("Error: tool `{name}` was not approved for execution"),
5974 true,
5975 ));
5976 }
5977 }
5978 let args = match call.function.parsed_arguments() {
5979 Ok(v) => v,
5980 Err(e) => {
5981 let err = Error::InvalidArguments {
5982 tool: name.clone(),
5983 message: e.to_string(),
5984 };
5985 return PreparedCall::Done((format!("Error: {err}"), true));
5986 }
5987 };
5988 self.finish_prepare(name.clone(), args)
5989 }
5990 }
5991
5992 /// P5-1: the shared tail of [`Self::prepare_tool_call`] — doom-loop
5993 /// check, pre-tool hook, `Ready` construction — factored out so both
5994 /// the legacy gate and the new permissions-engine gate run the exact
5995 /// same downstream checks in the exact same order (§5.3 risk 1: the
5996 /// permissions engine changes WHO gets to run, never what happens once
5997 /// they're approved).
5998 fn finish_prepare(&mut self, name: String, args: serde_json::Value) -> PreparedCall {
5999 // P4c (§5.2 P4 "doom-loop breaker", oc UNIQUE `doom_loop` row,
6000 // catalog D3): a default, always-available veto point distinct from
6001 // `Config.pre_tool_hook` (a single user-installable slot — the
6002 // breaker must coexist with a caller's own hook, not compete for the
6003 // one slot). `None`/`Some(0|1)` is a no-op — byte-identical to
6004 // today (no repetition tracking, no call is ever refused on this
6005 // basis).
6006 if let Some(reason) = self.check_doom_loop(&name, &args) {
6007 return PreparedCall::Done((format!("Error: {reason}"), true));
6008 }
6009 // Pre-tool hook may block the call. BP-10: the pre-P5-1 gate has
6010 // no permissions engine for a hook's `Allow`/`Ask` to be a tier
6011 // OF, so only the deny half can mean anything here — an
6012 // `updated_args` rewrite still applies (it is a property of the
6013 // call, not of any gate), and `Allow`/`Ask` are no-ops, exactly
6014 // as `None` was before BP-10.
6015 let mut args = args;
6016 if let Some(hook) = &self.config.pre_tool_hook {
6017 let outcome = hook(&name, &args);
6018 if let Some(rewritten) = outcome.updated_args {
6019 args = rewritten;
6020 }
6021 if outcome.decision == crate::config::HookDecision::Deny {
6022 let reason = outcome.reason.unwrap_or_else(|| "denied".to_string());
6023 return PreparedCall::Done((
6024 format!("Error: blocked by pre-tool hook: {reason}"),
6025 true,
6026 ));
6027 }
6028 }
6029 PreparedCall::Ready { name, args }
6030 }
6031
6032 /// BP-10 (catalog row "Hook/plugin permission veto"): fire the pre-tool
6033 /// hook for the permissions-engine path, where it runs BEFORE the gate
6034 /// (CC's own order: a `PreToolUse` hook answers the permission question
6035 /// rather than being asked after it). Returns the possibly-REWRITTEN
6036 /// arguments plus the [`crate::config::HookDecision`] the engine folds
6037 /// in, or the finished denial when the hook refused outright.
6038 ///
6039 /// The rewrite lands BEFORE the gate deliberately: the engine must
6040 /// evaluate what will actually run, so a hook cannot launder a denied
6041 /// command by rewriting it past the rules.
6042 #[allow(clippy::type_complexity)]
6043 fn run_pre_tool_hook(
6044 &self,
6045 name: &str,
6046 args: serde_json::Value,
6047 ) -> std::result::Result<(serde_json::Value, crate::config::HookDecision), PreparedCall> {
6048 let Some(hook) = &self.config.pre_tool_hook else {
6049 return Ok((args, crate::config::HookDecision::Pass));
6050 };
6051 let outcome = hook(name, &args);
6052 let args = outcome.updated_args.unwrap_or(args);
6053 if outcome.decision == crate::config::HookDecision::Deny {
6054 let reason = outcome.reason.unwrap_or_else(|| "denied".to_string());
6055 return Err(PreparedCall::Done((
6056 format!("Error: blocked by pre-tool hook: {reason}"),
6057 true,
6058 )));
6059 }
6060 Ok((args, outcome.decision))
6061 }
6062
6063 /// D-2 (Fable-5 delta review — LOW-MEDIUM, "over-grant residual"): the
6064 /// built-in tools whose `command`/`path`/`patch` arg IS semantically
6065 /// the whole call — the ONLY tools [`Self::permissions_gate_denial`]
6066 /// is allowed to turn into an [`permissions::ApprovalRequest::subject`]
6067 /// (see that method's own doc comment on the `subject` line for the
6068 /// full story). Bash-family (`bash`, and `shell` —
6069 /// [`crate::tools::builtins::PersistentShellTool`]'s registered name,
6070 /// what the review's "persistent-shell" refers to), the file tools
6071 /// (`read_file`/`write_file`/`edit_file`/`view_image`, whose `path` IS
6072 /// the subject), and `apply_patch` (whose `patch` envelope is handled
6073 /// separately but is unconditionally this tool only, see the `patch`
6074 /// local a few lines below). Deliberately NOT `list_dir`/`glob`/
6075 /// `search` — this crate's F2 fix (`ApprovalCache::key_for_request`)
6076 /// already falls back to a full-args digest for anything not on this
6077 /// list, which is a strictly SAFER (if slightly less cache-granular)
6078 /// default than guessing at more built-ins that weren't part of this
6079 /// finding.
6080 const SUBJECT_BEARING_BUILTIN_TOOLS: &'static [&'static str] = &[
6081 "bash",
6082 "shell",
6083 "read_file",
6084 "write_file",
6085 "edit_file",
6086 "view_image",
6087 "apply_patch",
6088 ];
6089
6090 /// P5-1: evaluate `name`'s call (with parsed `args`) against the
6091 /// permissions engine (`crate::permissions`) — builds the
6092 /// [`crate::permissions::RuleSet`] from `Config`'s deny/ask/allow
6093 /// pattern lists (folding [`Config::permissions_protected_paths`] into
6094 /// the `deny` tier, module 13), picks a command-, path-, or name-only
6095 /// evaluation depending on what `args` carries, resolves an `Ask`
6096 /// decision via the session cache + THIS agent's own installed
6097 /// [`crate::permissions::PermissionsApprovalHandler`], and returns
6098 /// `Some(reason)` when the call is refused (`None` = proceed). Thin
6099 /// wrapper over [`Self::permissions_gate_denial_impl`] — see that
6100 /// method's doc comment for why the handler is a parameter there.
6101 fn permissions_gate_denial(
6102 &self,
6103 name: &str,
6104 args: &serde_json::Value,
6105 hook: crate::config::HookDecision,
6106 ) -> Option<String> {
6107 self.permissions_gate_denial_impl(
6108 name,
6109 args,
6110 self.permissions_approval_handler.as_deref(),
6111 hook,
6112 )
6113 }
6114
6115 /// P5-6 (§2.2 C6, build brief "wire to the P5-1 engine's non-
6116 /// interactive fail-closed path"): the SAME rule-evaluation body as
6117 /// [`Self::permissions_gate_denial`], but the approval `handler` is a
6118 /// PARAMETER instead of always reading `self.permissions_approval_handler`
6119 /// — `Agent::background_permission_denial` calls this with `handler:
6120 /// None` (or a [`crate::subagents::ParentQueueApprovalHandler`]) so a
6121 /// `background_exec` call's `Ask`-tier decisions resolve exactly like a
6122 /// P5-3 background child's do (`crate::permissions::resolve_ask`'s
6123 /// pre-existing "no handler ⇒ deny" contract), REGARDLESS of whether
6124 /// this agent itself has an interactive handler installed for its own
6125 /// foreground calls — a background job must never block on a prompt it
6126 /// has no way to answer, even if the agent hosting it could otherwise
6127 /// answer one. The rule SET and default-policy baseline are otherwise
6128 /// identical to a foreground call's — only how an `Ask` decision
6129 /// resolves ever differs, and only in the strictly-narrower direction
6130 /// (never escalates past what a foreground call of the same command is
6131 /// allowed).
6132 fn permissions_gate_denial_impl(
6133 &self,
6134 name: &str,
6135 args: &serde_json::Value,
6136 handler: Option<&dyn crate::permissions::PermissionsApprovalHandler>,
6137 hook: crate::config::HookDecision,
6138 ) -> Option<String> {
6139 use crate::permissions::{self, Decision, PathKind};
6140
6141 // `Config.tool_deny_patterns`/`tool_allow_patterns` (P4a) ARE the
6142 // engine's deny/allow tiers — the same `capabilities.permissions.
6143 // rules.deny`/`.allow` keys, one source of truth, no duplication.
6144 // Protected paths (module 13) are an unconditional deny floor,
6145 // folded in here rather than checked separately, so they benefit
6146 // from the SAME first-match deny-wins priority every other deny
6147 // rule gets.
6148 // BP-5: the rule set itself is `permissions::rules_for_config` —
6149 // ONE construction shared with every other surface that has to ask
6150 // this engine a question (see that function's doc comment). Plan
6151 // mode's narrowing is layered on top of it here, because it is
6152 // this agent's live state, not the config's.
6153 let mut rules = permissions::rules_for_config(&self.config);
6154 // BP-3 (§2 module 8 `plan_mode`, dependency edge `plan_mode →
6155 // permissions.rules|sandbox`): the read-only research phase IS a
6156 // narrowing of this rule set — while the mode is active the write
6157 // and execution tools join the deny tier, and get the same
6158 // first-match, never-overridable treatment every other deny rule
6159 // gets. Inactive (the default, and the only state a config without
6160 // the module can reach) contributes NOTHING, so the rule set is
6161 // byte-identical to before.
6162 rules
6163 .deny
6164 .extend(crate::tools::plan_mode::deny_rules(&self.ctx.plan_mode));
6165
6166 // The baseline decision when NO rule matches at all — derived from
6167 // `ApprovalPolicy`, the same per-policy shape
6168 // `Config::needs_approval` uses for the legacy gate (see
6169 // `ApprovalPolicy::ModelRequested`'s doc comment for why this
6170 // richer gate approximates Codex's real "mostly silent" posture
6171 // instead of that method's conservative OnRequest-alike treatment
6172 // — explicit deny/ask rules still apply on top regardless).
6173 let default = permissions::default_decision(&self.config, name);
6174
6175 // BP-10: path rules are evaluated relative to EVERY granted root
6176 // (cwd + `core.additional_dirs`/`--add-dir`), folded to the
6177 // strictest — see `permissions::evaluate_path_safe_roots`'s doc
6178 // comment for why a grant must not also remove the root-relative
6179 // protected-path floor inside the granted directory. With no extra
6180 // dirs (the default) this is the single `cwd` list, byte-identical
6181 // to before.
6182 let mut roots = vec![self.config.cwd.clone()];
6183 roots.extend(self.config.additional_dirs.iter().cloned());
6184
6185 let command = args.get("command").and_then(|v| v.as_str());
6186 let path = args.get("path").and_then(|v| v.as_str());
6187 // F4 (Fable-5 adversarial review): `apply_patch`'s args carry a
6188 // patch ENVELOPE body (`args["patch"]`), not a `command` or a
6189 // `path` — the two branches above never fire for it, which is
6190 // exactly how a patch touching a protected path bypassed
6191 // `protected_paths` entirely. Only consulted for the `apply_patch`
6192 // tool specifically (a `patch`-shaped arg on some other tool is not
6193 // this envelope format and isn't given this treatment).
6194 let patch = (name == "apply_patch")
6195 .then(|| args.get("patch").and_then(|v| v.as_str()))
6196 .flatten();
6197 let decision = if let Some(command) = command {
6198 permissions::evaluate_command(&rules, name, command, default)
6199 } else if let Some(patch) = patch {
6200 // Same dual-check shape as the path branch below (pseudo-tool
6201 // `write(...)` rules from `protected_paths`, AND a rule
6202 // authored against the real `apply_patch` tool name), applied
6203 // to EVERY path the envelope's ops touch (`Add`/`Delete`/
6204 // `Update`'s `path`, plus `*** Move to:`). A patch that fails
6205 // to parse can't be proven to avoid a protected path — fail
6206 // closed to at least `Ask`, the same floor an unparseable bash
6207 // command gets in `permissions::evaluate_command`, rather than
6208 // silently let it through on `default`.
6209 let mut d = rules.evaluate(name, None).unwrap_or(default);
6210 match crate::tools::patch_target_paths(patch) {
6211 Ok(paths) => {
6212 for p in &paths {
6213 // SECURITY (CRITICAL fix): route both checks through
6214 // the safe-path-resolving variants — a patch target
6215 // like `x/../.git/config` must be caught exactly
6216 // like a `write_file`/`edit_file` `path` argument
6217 // would be (see `evaluate_path_safe`'s doc comment).
6218 let pseudo = permissions::evaluate_path_safe_roots(
6219 &rules,
6220 PathKind::Write,
6221 &roots,
6222 p,
6223 Decision::Allow,
6224 );
6225 let real_tool = permissions::evaluate_path_subject_safe_roots(
6226 &rules,
6227 name,
6228 &roots,
6229 p,
6230 Decision::Allow,
6231 );
6232 d = d.stricter(pseudo).stricter(real_tool);
6233 }
6234 }
6235 Err(_) => {
6236 d = d.stricter(Decision::Ask);
6237 }
6238 }
6239 d
6240 } else if let Some(path) = path {
6241 let kind = if matches!(name, "write_file" | "edit_file") {
6242 PathKind::Write
6243 } else {
6244 PathKind::Read
6245 };
6246 // TWO independent sources of path-shaped rules can apply to the
6247 // same call, and BOTH must be checked:
6248 // (a) the `read(...)`/`write(...)` pseudo-tool (tool-agnostic —
6249 // applies no matter WHICH tool touches the path; this is
6250 // what `Config::permissions_protected_paths`/module 13
6251 // expands into, via `protected_path_deny_rules`);
6252 // (b) a rule authored against the REAL tool name with the path
6253 // as its subject — design §4.4's own oc-parity worked
6254 // example writes exactly this shape (`"read_file(*.env)"`,
6255 // not a pseudo-tool), matching how `bash(cmdglob)` rules
6256 // are authored. `RuleSet::evaluate`'s bare-tool-name-glob
6257 // branch (no parens) ALSO fires here regardless of
6258 // `subject`, so this one call additionally covers a
6259 // blanket "deny this tool entirely" rule — no separate
6260 // `rules.evaluate(name, None)` call is needed.
6261 // SECURITY (CRITICAL fix, guarantor audit): both checks now
6262 // route through the safe-path-resolving variants (see
6263 // `evaluate_path_safe`'s doc comment) instead of glob-matching
6264 // the raw model-supplied `path` string directly — this is what
6265 // closes the traversal bypass (`write_file
6266 // path="x/../.git/config"`) and the analogous symlink escape.
6267 let pseudo_decision =
6268 permissions::evaluate_path_safe_roots(&rules, kind, &roots, path, default);
6269 let real_tool_decision =
6270 permissions::evaluate_path_subject_safe_roots(&rules, name, &roots, path, default);
6271 pseudo_decision.stricter(real_tool_decision)
6272 } else {
6273 rules.evaluate(name, None).unwrap_or(default)
6274 };
6275
6276 // BP-10 (catalog row "Hook/plugin permission veto"): the hook's
6277 // verdict is a TIER of this engine, folded in the one direction
6278 // that is always safe — `Ask` tightens (`stricter` never loosens),
6279 // `Deny` is a floor. `Allow` deliberately does NOT change the
6280 // decision here: it answers the `Ask` tier below (the
6281 // `PermissionRequest`-class reply CC's hooks give on the user's
6282 // behalf), so a `deny` rule still refuses the call outright — a
6283 // hook may skip a prompt, never a floor.
6284 let decision = match hook {
6285 crate::config::HookDecision::Deny => Decision::Deny,
6286 crate::config::HookDecision::Ask => decision.stricter(Decision::Ask),
6287 crate::config::HookDecision::Allow | crate::config::HookDecision::Pass => decision,
6288 };
6289
6290 // BP-10 (catalog row "Sandbox-escalation path", cx§4
6291 // `sandbox_permissions: "require_escalated"` + justification): the
6292 // model's channel to ASK for an unsandboxed run is a rule inside
6293 // this one engine, not a switch beside it. A call carrying
6294 // `with_escalated_permissions: true` is forced to at least the
6295 // `Ask` tier — never below whatever the rules already decided, so
6296 // a denied command cannot escalate its way out (`stricter` only
6297 // tightens), and never silently allowed under
6298 // `ApprovalPolicy::ModelRequested`/`Never`, whose `Allow` baseline
6299 // is exactly what made "the model requests escalation" a no-op
6300 // before. The `justification` rides in `raw_args` below, so the
6301 // approval door shows the user the model's own reason.
6302 let decision = if args
6303 .get("with_escalated_permissions")
6304 .and_then(|v| v.as_bool())
6305 .unwrap_or(false)
6306 {
6307 decision.stricter(Decision::Ask)
6308 } else {
6309 decision
6310 };
6311
6312 // D-2 (Fable-5 delta review — LOW-MEDIUM): `command`/`path`/`patch`
6313 // above are extracted (and used to DRIVE the decision above) for
6314 // ANY tool that happens to carry one of those arg names — that
6315 // part is unchanged and correct (a rule authored against, say, an
6316 // MCP tool's own name legitimately wants to glob-match its
6317 // `command`-shaped arg too). But the narrower single-field
6318 // `subject` handed to the cache/handler below must NOT do the
6319 // same for a non-built-in tool: an MCP (or other) tool's
6320 // `command`/`path` is just one field among potentially several
6321 // that together define what the call actually does — collapsing
6322 // an `AllowForSession` grant down to that one field would silently
6323 // auto-allow a later call with the SAME `command` but different
6324 // OTHER args (e.g. `{"command":"sync","target":"staging"}`
6325 // auto-allowing `{"command":"sync","target":"production"}`).
6326 // Restricting this to the known built-ins whose `subject` really
6327 // IS the whole call leaves every other tool with `subject: None`,
6328 // which routes it through `ApprovalCache::key_for_request`'s
6329 // full-args-digest fallback (F2) instead.
6330 let subject = Self::SUBJECT_BEARING_BUILTIN_TOOLS
6331 .contains(&name)
6332 .then(|| command.or(path).or(patch))
6333 .flatten();
6334 let req = permissions::ApprovalRequest {
6335 tool: name,
6336 subject,
6337 raw_args: args,
6338 };
6339 let approved = permissions::decision_to_approved(decision, || {
6340 // BP-10: a hook `Allow` answers this ask without a prompt (and
6341 // without a cache entry — the hook is consulted on every call,
6342 // so caching its answer would be a second, staler copy of the
6343 // same decision).
6344 if hook == crate::config::HookDecision::Allow {
6345 return true;
6346 }
6347 permissions::resolve_ask(&self.permissions_approval_cache, handler, &req)
6348 });
6349 if approved {
6350 None
6351 } else {
6352 Some(format!(
6353 "tool `{name}` was not approved for execution (permissions engine: {decision:?})"
6354 ))
6355 }
6356 }
6357
6358 /// P5-6 (§2.2 C6, build brief "a bg `rm -rf` subject to the same deny
6359 /// rules... must never escalate past what a foreground exec of the
6360 /// same command is allowed"): the permission gate `background_exec`
6361 /// runs BEFORE spawning anything. Evaluated against the tool name
6362 /// `"bash"` (not `"background_exec"`) deliberately — so any
6363 /// `bash(...)`-authored deny/ask/allow rule (or protected-path floor)
6364 /// applies to a background command byte-for-byte identically to a
6365 /// foreground `bash` call, the SAME rule set + default baseline
6366 /// [`Self::permissions_gate_denial`] would use for one.
6367 ///
6368 /// The one deliberate difference (C6 itself): an `Ask`-tier decision
6369 /// NEVER reaches an interactive handler here — a background job has no
6370 /// way to block on a prompt it can't answer. When
6371 /// [`Config::subagents_background_prompts`] is
6372 /// [`crate::subagents::BackgroundPromptsPolicy::Parent`], the denied
6373 /// request is additionally queued onto [`Self::pending_child_approvals`]
6374 /// (via [`crate::subagents::ParentQueueApprovalHandler`], reused
6375 /// verbatim — the SAME "parent-surfaced queue" §2.2 C6 names for
6376 /// `subagents.background`, with the job id standing in for a child
6377 /// agent id) for later inspection; any other configuration (including
6378 /// no `background_prompts` set at all) resolves via `handler: None` —
6379 /// [`crate::permissions::resolve_ask`]'s pre-existing "no handler ⇒
6380 /// deny" fail-closed default, identical to `subagents`'s own
6381 /// `AutoPolicy` reading. Either way, `Ask` always denies; only `Allow`
6382 /// (from the rule engine itself, or a PRIOR interactively-granted
6383 /// `AllowForSession` cache entry) ever lets a background command run —
6384 /// so this can only ever be as-or-more restrictive than a foreground
6385 /// call, never looser, regardless of configuration.
6386 ///
6387 /// Covers BOTH gate generations: when [`Config::permissions_enabled`]
6388 /// is on, the P5-1 engine (above) is used; otherwise the legacy
6389 /// [`Config::needs_approval`] gate is consulted but its
6390 /// `approval_handler` closure is NEVER invoked (that closure could
6391 /// itself block, e.g. a real interactive prompt) — an approval-required
6392 /// legacy policy simply denies a background command outright, the same
6393 /// never-hang guarantee under the older gate.
6394 fn background_permission_denial(
6395 &self,
6396 command: &str,
6397 job_id: &str,
6398 hook: crate::config::HookDecision,
6399 ) -> Option<String> {
6400 let args = serde_json::json!({ "command": command });
6401 if self.config.permissions_enabled {
6402 if let Some(crate::subagents::BackgroundPromptsPolicy::Parent) =
6403 self.config.subagents_background_prompts
6404 {
6405 let handler = crate::subagents::ParentQueueApprovalHandler {
6406 child_agent_id: format!("bg:{job_id}"),
6407 queue: self.pending_child_approvals.clone(),
6408 };
6409 self.permissions_gate_denial_impl("bash", &args, Some(&handler), hook)
6410 } else {
6411 self.permissions_gate_denial_impl("bash", &args, None, hook)
6412 }
6413 } else if self.config.needs_approval("bash") {
6414 Some(
6415 "tool `bash` requires approval, which a background job cannot request \
6416 interactively (§2.2 C6: auto-policy denies)"
6417 .to_string(),
6418 )
6419 } else {
6420 None
6421 }
6422 }
6423
6424 /// P4e (§3.1 `core.parallel_tool_calls`, catalog:59 "Independent
6425 /// sibling calls run concurrently"): runs `calls`' `Tool::execute()`
6426 /// futures CONCURRENTLY via `futures::future::join_all`, for whichever
6427 /// calls [`Self::prepare_tool_call`] resolves to [`PreparedCall::Ready`]
6428 /// — i.e. every plain (non-intrinsic) registry-tool call that passes
6429 /// its synchronous approval/doom-loop/pre-tool-hook checks. A call that
6430 /// resolves to [`PreparedCall::Done`] (an intrinsic, an unknown tool, a
6431 /// denied/blocked call) is NOT parallelized — its result is already in
6432 /// hand from the synchronous prepare pass. Every prepare check still
6433 /// runs sequentially, in original call order, before ANY `execute()`
6434 /// future starts (only the actual tool I/O overlaps) — so doom-loop
6435 /// bookkeeping and pre-tool-hook vetoes see the exact same call order
6436 /// they would under the sequential path. Returns results in the SAME
6437 /// order as `calls`, so callers can always `zip` the two. Post-tool
6438 /// hooks fire per call, in original order, once every result is in
6439 /// hand — a caller-visible timing difference from the sequential path
6440 /// ONLY when this method runs at all (i.e. only when
6441 /// `Config::parallel_tool_calls` is on): hooks see "this batch
6442 /// finished" ordering rather than "this one call finished" ordering.
6443 /// Documented, not a bug.
6444 async fn run_tools_concurrently(
6445 &mut self,
6446 calls: &[supercode_interchange::ToolCall],
6447 ) -> Vec<(String, bool)> {
6448 // P5-3: `spawn_subagent`/`subagent_status` need sequential `&mut
6449 // self` access `prepare_tool_call`'s synchronous-only signature
6450 // can't give them (see `Self::run_tool`'s identical interception).
6451 // A batch that includes one falls back to dispatching the WHOLE
6452 // batch sequentially via `Self::run_tool` — a documented, narrow
6453 // simplification (not a partial-parallelization attempt) rather
6454 // than restructuring `PreparedCall` to carry a future; a batch with
6455 // no subagent intrinsic is completely unaffected and still
6456 // parallelizes exactly as before.
6457 if calls.iter().any(|c| {
6458 c.function.name == SPAWN_SUBAGENT
6459 || c.function.name == SUBAGENT_STATUS
6460 || c.function.name == SEND_MESSAGE
6461 || c.function.name == SUBAGENT_RESUME
6462 || (self.config.claude_runtime_tools_enabled
6463 && matches!(
6464 c.function.name.as_str(),
6465 CLAUDE_CRON_CREATE
6466 | CLAUDE_CRON_DELETE
6467 | CLAUDE_CRON_LIST
6468 | CLAUDE_SCHEDULE_WAKEUP
6469 ))
6470 }) {
6471 let mut out = Vec::with_capacity(calls.len());
6472 for call in calls {
6473 out.push(self.run_tool(call).await);
6474 }
6475 return out;
6476 }
6477 let prepared: Vec<PreparedCall> = calls.iter().map(|c| self.prepare_tool_call(c)).collect();
6478 let mut slots: Vec<Option<(String, bool)>> = prepared
6479 .iter()
6480 .map(|p| match p {
6481 PreparedCall::Done(r) => Some(r.clone()),
6482 PreparedCall::Ready { .. } => None,
6483 })
6484 .collect();
6485
6486 let ready_idxs: Vec<usize> = prepared
6487 .iter()
6488 .enumerate()
6489 .filter(|(_, p)| matches!(p, PreparedCall::Ready { .. }))
6490 .map(|(i, _)| i)
6491 .collect();
6492
6493 if !ready_idxs.is_empty() {
6494 let futs = ready_idxs.iter().map(|&i| {
6495 let PreparedCall::Ready { name, args } = &prepared[i] else {
6496 unreachable!("filtered to Ready above")
6497 };
6498 // `self.registry.get` borrows `self.registry` immutably;
6499 // `self.ctx` is `Clone` (P4c precedent) so each future owns
6500 // its own copy rather than borrowing `self` across the
6501 // `.await` inside `join_all`.
6502 let tool = self.registry.get(name).expect("prepared as Ready");
6503 let args = args.clone();
6504 let ctx = self.ctx.clone();
6505 async move {
6506 match tool.execute(args, &ctx).await {
6507 Ok(out) => (out, false),
6508 Err(e) => (format!("Error: {e}"), true),
6509 }
6510 }
6511 });
6512 let results = futures::future::join_all(futs).await;
6513 for (idx, result) in ready_idxs.iter().zip(results) {
6514 slots[*idx] = Some(result);
6515 }
6516 }
6517
6518 let out: Vec<(String, bool)> = slots
6519 .into_iter()
6520 .map(|s| s.expect("every call resolved to Some above"))
6521 .collect();
6522 // Post-tool hook, in original order — only for calls that actually
6523 // reached `execute()` (matches `run_tool`'s existing behavior: an
6524 // intrinsic/denied/blocked call never fires the post-tool hook).
6525 let ready_set: std::collections::HashSet<usize> = ready_idxs.into_iter().collect();
6526 for (i, call) in calls.iter().enumerate() {
6527 if !ready_set.contains(&i) {
6528 continue;
6529 }
6530 let (output, is_error) = &out[i];
6531 if let Some(hook) = &self.config.post_tool_hook {
6532 hook(&call.function.name, output, *is_error);
6533 }
6534 }
6535 out
6536 }
6537
6538 /// P4c (§5.2 P4 "doom-loop breaker", §3.1 `core.doom_loop_threshold`):
6539 /// update the consecutive-identical-call streak for `(name, args)` and
6540 /// return `Some(reason)` the moment the streak reaches
6541 /// `Config.doom_loop_threshold` (a call whose name AND JSON-canonical
6542 /// arguments are byte-identical to the immediately preceding call
6543 /// extends the streak; anything else resets it to 1). `None`
6544 /// (`Config.doom_loop_threshold` unset, or `Some(n)` with `n < 2` — a
6545 /// threshold below 2 can never fire since the FIRST call already
6546 /// "repeats zero times") never touches the streak fields at all.
6547 fn check_doom_loop(&mut self, name: &str, args: &serde_json::Value) -> Option<String> {
6548 let threshold = self.config.doom_loop_threshold?;
6549 if threshold < 2 {
6550 return None;
6551 }
6552 // `serde_json::Value::Object` is a `BTreeMap` in this workspace (no
6553 // `preserve_order` feature), so `to_string()` is already
6554 // key-order-canonical — two calls that differ only in argument key
6555 // order are still treated as identical.
6556 let key = (name.to_string(), args.to_string());
6557 if self.doom_loop_last_call.as_ref() == Some(&key) {
6558 self.doom_loop_streak += 1;
6559 } else {
6560 self.doom_loop_last_call = Some(key);
6561 self.doom_loop_streak = 1;
6562 }
6563 if self.doom_loop_streak >= threshold {
6564 Some(format!(
6565 "doom-loop breaker: `{name}` called with identical arguments {} times in a row \
6566 — try a different approach instead of repeating the same call",
6567 self.doom_loop_streak
6568 ))
6569 } else {
6570 None
6571 }
6572 }
6573
6574 /// Whether `name` is in the eagerly-advertised "core" set for the current
6575 /// [`ToolAdvertising`] mode: every enabled tool under `Full`, or the
6576 /// explicit `core` allowlist under `Deferred`.
6577 fn is_core_tool(&self, name: &str) -> bool {
6578 match &self.config.tool_advertising {
6579 ToolAdvertising::Full => true,
6580 ToolAdvertising::Deferred { core } => core.iter().any(|c| c == name),
6581 }
6582 }
6583
6584 /// The schema advertised on the wire for `t`: the raw (as-shipped)
6585 /// schema with TR-8/T5's per-tool schema tier applied. This is what
6586 /// [`Self::tool_schemas`] sends every request.
6587 fn schema_for(&self, t: &dyn crate::tools::Tool) -> ToolSchema {
6588 let raw = self.raw_schema_for(t);
6589 let tier = self.config.schema_tier_for(t.name());
6590 let (description, parameters) =
6591 crate::tools::tiers::minify(&raw.description, &raw.parameters, tier);
6592 ToolSchema {
6593 name: raw.name,
6594 description,
6595 parameters,
6596 }
6597 }
6598
6599 /// The ORIGINAL, as-shipped schema for `t` — never tier-minified. This is
6600 /// the full contract [`Self::run_tool_search`] hands back on activation
6601 /// (TR-8/T5 dev/03: the B6 fetch path is the invert of tiering, so a
6602 /// model that fetched a tool via `tool_search` always sees the complete
6603 /// schema, byte-equal to `t.description()`/`t.parameters()` — modulo the
6604 /// pre-existing [`crate::Config::tool_description`] override, which is
6605 /// orthogonal to tiering).
6606 fn raw_schema_for(&self, t: &dyn crate::tools::Tool) -> ToolSchema {
6607 ToolSchema {
6608 name: t.name().to_string(),
6609 description: self
6610 .config
6611 .tool_description(t.name(), t.description())
6612 .to_string(),
6613 parameters: t.parameters(),
6614 }
6615 }
6616
6617 /// The synthetic `tool_search` schema advertised under `Deferred` (B6).
6618 fn tool_search_schema() -> ToolSchema {
6619 ToolSchema {
6620 name: TOOL_SEARCH.to_string(),
6621 description: "Search for additional tools not currently advertised (the deferred \
6622 MCP surface and any other non-core tools). Matches keywords case-insensitively \
6623 against each tool's name and description. Matched tools become callable starting \
6624 with your NEXT message, not this one."
6625 .to_string(),
6626 parameters: serde_json::json!({
6627 "type": "object",
6628 "properties": {
6629 "query": {
6630 "type": "string",
6631 "description": "Keyword(s) to search for in tool names and descriptions."
6632 },
6633 "max_results": {
6634 "type": "integer",
6635 "description": "Maximum number of matching tools to return."
6636 }
6637 },
6638 "required": ["query"],
6639 "additionalProperties": false
6640 }),
6641 }
6642 }
6643
6644 /// The tool-schema array this agent would advertise on its NEXT
6645 /// request, exactly as `Self::run_loop` computes it. Public
6646 /// (PARITY-18 D1) so a caller can measure the real request-token cost
6647 /// of an agent's tool surface — including the current
6648 /// [`crate::config::ToolAdvertising`] mode's core/deferred split and
6649 /// the synthetic `tool_search`/`expand_reduction`/`sidecar_search`
6650 /// schemas — BEFORE ever calling [`Self::send`], e.g. for a preflight
6651 /// context-guard check.
6652 pub fn tool_schemas(&self) -> Vec<ToolSchema> {
6653 let mut out = match &self.config.tool_advertising {
6654 ToolAdvertising::Full => self
6655 .registry
6656 .iter()
6657 .filter(|t| self.config.tool_enabled(t.name()))
6658 .map(|t| self.schema_for(t))
6659 .collect(),
6660 ToolAdvertising::Deferred { .. } => {
6661 let mut out: Vec<ToolSchema> = self
6662 .registry
6663 .iter()
6664 .filter(|t| self.config.tool_enabled(t.name()))
6665 .filter(|t| {
6666 self.is_core_tool(t.name()) || self.activated_tools.contains(t.name())
6667 })
6668 .map(|t| self.schema_for(t))
6669 .collect();
6670 out.push(Self::tool_search_schema());
6671 out
6672 }
6673 };
6674 // T12/TR-1: `expand_reduction`/`sidecar_search` are orthogonal to
6675 // `tool_advertising` (which governs the ordinary tool surface) —
6676 // advertised whenever a `ReductionPolicy` is installed, regardless of
6677 // Full/Deferred, since only a reduced session ever has anything to
6678 // expand or search (SPEC.md TR-1 dev/01).
6679 if self.reduction_policy.is_some() {
6680 out.push(Self::expand_reduction_schema());
6681 out.push(Self::sidecar_search_schema());
6682 }
6683 // P5-3 (§2 module 9): `spawn_subagent`/`subagent_status` are
6684 // orthogonal to `tool_advertising` too, same reasoning as
6685 // `expand_reduction`/`sidecar_search` above — advertised whenever
6686 // `Config::subagents_enabled` is on, Full or Deferred alike.
6687 // `false` (the default) never appends either, so a config that
6688 // never turns the module on gets byte-identical tool schemas to
6689 // today.
6690 if self.config.subagents_enabled {
6691 out.push(self.spawn_subagent_schema());
6692 if self.config.subagents_claude_agent_alias {
6693 out.push(self.claude_agent_schema());
6694 }
6695 if self.config.subagents_background {
6696 out.push(Self::subagent_status_schema());
6697 // BP-7 (catalog §4a "Background subagents + resume"): the
6698 // two halves the row named as missing — a mailbox into a
6699 // still-running child, and a resume of a finished one with
6700 // its context intact.
6701 out.push(Self::send_message_schema());
6702 out.push(Self::subagent_resume_schema());
6703 }
6704 }
6705 if self.config.claude_runtime_tools_enabled {
6706 out.extend(self.claude_builtin_tool_schemas());
6707 out.push(Self::claude_cron_create_schema());
6708 out.push(Self::claude_cron_delete_schema());
6709 out.push(Self::claude_cron_list_schema());
6710 out.push(Self::claude_schedule_wakeup_schema());
6711 }
6712 // P5-6 (§2 module 4 `tools.background`): same orthogonal-to-
6713 // `tool_advertising` treatment, advertised whenever
6714 // `Config::tools_background_enabled` is on. `false` (the default)
6715 // never appends any of the four, so a config that never turns the
6716 // module on gets byte-identical tool schemas to today.
6717 if self.config.tools_background_enabled {
6718 out.push(Self::background_exec_schema());
6719 out.push(Self::background_status_schema());
6720 out.push(Self::background_list_schema());
6721 out.push(Self::background_kill_schema());
6722 }
6723 // BP-10 (catalog row "Tool hiding via policy", cc§4 "bare-name
6724 // deny"): a policy deny does not merely REFUSE the call at
6725 // dispatch — it removes the tool from the model's view. Applied
6726 // once, here, over the finished array, so every family appended
6727 // above (`spawn_subagent`, `background_*`, the Claude aliases,
6728 // `expand_reduction`, …) is hidden by the same one rule, not by a
6729 // per-family repeat of it. See [`Self::policy_hides_tool`] for
6730 // which deny tier is consulted and why.
6731 out.retain(|schema| !self.policy_hides_tool(&schema.name));
6732 out
6733 }
6734
6735 /// BP-10: whether the CONFIG-DECLARED deny tier hides `name` from the
6736 /// model's tool surface entirely (cc§4: CC's bare-name deny "removes
6737 /// the tool from the model's view", where an ordinary rule only
6738 /// refuses the call).
6739 ///
6740 /// The ONE engine decides: this is
6741 /// [`crate::permissions::RuleSet::evaluate`] with `subject: None`, so
6742 /// exactly the patterns that can be satisfied by a tool NAME ALONE
6743 /// (`"bash"`, `"mcp_*"`, `"*"`) hide; a rule that names a
6744 /// command/path constraint (`"bash(rm -rf*)"`, `"write(.git/**)"`) is
6745 /// not satisfiable without a subject and therefore never hides a tool
6746 /// — the same `rule_matches` contract the dispatch gate uses.
6747 ///
6748 /// **Which deny tier.** `Config::tool_deny_patterns` — the
6749 /// `capabilities.permissions.rules.deny` array — and NOT the two
6750 /// runtime narrowings the dispatch gate folds in beside it:
6751 /// `protected_paths` expands to `read(...)`/`write(...)` patterns that
6752 /// carry a subject by construction (so they could never match here
6753 /// anyway), and `plan_mode::deny_rules` is a MODE, not a policy — CC's
6754 /// plan mode refuses a write, it does not make Write disappear and
6755 /// reappear as the mode toggles mid-session. Hiding is a property of
6756 /// the configured policy, which is fixed for the run.
6757 ///
6758 /// Gated on [`Config::permissions_enabled`]: a config that never turns
6759 /// the module on gets byte-identical schemas to before this existed.
6760 fn policy_hides_tool(&self, name: &str) -> bool {
6761 if !self.config.permissions_enabled || self.config.tool_deny_patterns.is_empty() {
6762 return false;
6763 }
6764 let rules = crate::permissions::RuleSet {
6765 deny: self.config.tool_deny_patterns.clone(),
6766 ..Default::default()
6767 };
6768 rules.evaluate(name, None) == Some(crate::permissions::Decision::Deny)
6769 }
6770
6771 fn claude_builtin_tool_schemas(&self) -> Vec<ToolSchema> {
6772 let mut schemas = Vec::new();
6773 let mut push = |alias: &str, native: &str, description: &str, parameters| {
6774 if self.registry.get(native).is_some() && self.config.tool_enabled(native) {
6775 schemas.push(ToolSchema {
6776 name: alias.to_string(),
6777 description: description.to_string(),
6778 parameters,
6779 });
6780 }
6781 };
6782 push(
6783 CLAUDE_BASH,
6784 "bash",
6785 "Claude Code-compatible shell command execution.",
6786 serde_json::json!({
6787 "type": "object",
6788 "properties": {
6789 "command": {"type": "string"},
6790 "timeout": {"type": "integer", "description": "Timeout in milliseconds."},
6791 "description": {"type": "string"}
6792 },
6793 "required": ["command"],
6794 "additionalProperties": true
6795 }),
6796 );
6797 push(
6798 CLAUDE_READ,
6799 "read_file",
6800 "Claude Code-compatible file reader.",
6801 serde_json::json!({
6802 "type": "object",
6803 "properties": {
6804 "file_path": {"type": "string"},
6805 "offset": {"type": "integer"},
6806 "limit": {"type": "integer"}
6807 },
6808 "required": ["file_path"],
6809 "additionalProperties": false
6810 }),
6811 );
6812 push(
6813 CLAUDE_WRITE,
6814 "write_file",
6815 "Claude Code-compatible file writer.",
6816 serde_json::json!({
6817 "type": "object",
6818 "properties": {"file_path": {"type": "string"}, "content": {"type": "string"}},
6819 "required": ["file_path", "content"],
6820 "additionalProperties": false
6821 }),
6822 );
6823 push(
6824 CLAUDE_EDIT,
6825 "edit_file",
6826 "Claude Code-compatible exact file edit.",
6827 serde_json::json!({
6828 "type": "object",
6829 "properties": {
6830 "file_path": {"type": "string"},
6831 "old_string": {"type": "string"},
6832 "new_string": {"type": "string"},
6833 "replace_all": {"type": "boolean"}
6834 },
6835 "required": ["file_path", "old_string", "new_string"],
6836 "additionalProperties": false
6837 }),
6838 );
6839 push(
6840 CLAUDE_GLOB,
6841 "glob",
6842 "Claude Code-compatible file glob.",
6843 serde_json::json!({
6844 "type": "object",
6845 "properties": {"pattern": {"type": "string"}, "path": {"type": "string"}},
6846 "required": ["pattern"],
6847 "additionalProperties": false
6848 }),
6849 );
6850 push(
6851 CLAUDE_GREP,
6852 "search",
6853 "Claude Code-compatible content search.",
6854 serde_json::json!({
6855 "type": "object",
6856 "properties": {"pattern": {"type": "string"}, "path": {"type": "string"}},
6857 "required": ["pattern"],
6858 "additionalProperties": true
6859 }),
6860 );
6861 schemas
6862 }
6863
6864 fn translate_claude_builtin_call(
6865 &self,
6866 call: &supercode_interchange::ToolCall,
6867 ) -> Result<Option<supercode_interchange::ToolCall>> {
6868 let native = match call.function.name.as_str() {
6869 CLAUDE_BASH => "bash",
6870 CLAUDE_READ => "read_file",
6871 CLAUDE_WRITE => "write_file",
6872 CLAUDE_EDIT => "edit_file",
6873 CLAUDE_GLOB => "glob",
6874 CLAUDE_GREP => "search",
6875 _ => return Ok(None),
6876 };
6877 let mut args = call.function.parsed_arguments()?;
6878 let object = args
6879 .as_object_mut()
6880 .ok_or_else(|| Error::InvalidArguments {
6881 tool: call.function.name.clone(),
6882 message: "expected a JSON object".to_string(),
6883 })?;
6884 if let Some(path) = object.remove("file_path") {
6885 object.entry("path".to_string()).or_insert(path);
6886 }
6887 if call.function.name == CLAUDE_BASH {
6888 if let Some(timeout) = object.remove("timeout") {
6889 object.entry("timeout_ms".to_string()).or_insert(timeout);
6890 }
6891 }
6892 if call.function.name == CLAUDE_GLOB {
6893 if let Some(path) = object
6894 .remove("path")
6895 .and_then(|value| value.as_str().map(str::to_owned))
6896 {
6897 if let Some(pattern) = object.get_mut("pattern") {
6898 if let Some(value) = pattern.as_str() {
6899 if !std::path::Path::new(value).is_absolute() {
6900 *pattern = serde_json::Value::String(
6901 std::path::Path::new(&path)
6902 .join(value)
6903 .to_string_lossy()
6904 .into_owned(),
6905 );
6906 }
6907 }
6908 }
6909 }
6910 }
6911 let mut translated = call.clone();
6912 translated.function.name = native.to_string();
6913 translated.function.arguments = serde_json::to_string(&args)?;
6914 Ok(Some(translated))
6915 }
6916
6917 fn claude_cron_create_schema() -> ToolSchema {
6918 ToolSchema {
6919 name: CLAUDE_CRON_CREATE.to_string(),
6920 description: "Record a Claude-compatible cron job in the imported runtime manifest. \
6921 The job inherits the manifest's ACTIVE or PAUSED posture; an embedding scheduler, \
6922 not this agent loop, owns execution."
6923 .to_string(),
6924 parameters: serde_json::json!({
6925 "type": "object",
6926 "properties": {
6927 "cron": {"type": "string", "description": "Cron expression to preserve."},
6928 "prompt": {"type": "string", "description": "Prompt associated with the job."},
6929 "recurring": {"type": "boolean", "default": false},
6930 "durable": {"type": "boolean", "default": false}
6931 },
6932 "required": ["cron", "prompt"],
6933 "additionalProperties": false
6934 }),
6935 }
6936 }
6937
6938 fn claude_cron_delete_schema() -> ToolSchema {
6939 ToolSchema {
6940 name: CLAUDE_CRON_DELETE.to_string(),
6941 description: "Delete a Claude-compatible cron job from the imported manifest. \
6942 This updates state only; an embedding scheduler owns execution."
6943 .to_string(),
6944 parameters: serde_json::json!({
6945 "type": "object",
6946 "properties": {"id": {"type": "string"}},
6947 "required": ["id"],
6948 "additionalProperties": false
6949 }),
6950 }
6951 }
6952
6953 fn claude_cron_list_schema() -> ToolSchema {
6954 ToolSchema {
6955 name: CLAUDE_CRON_LIST.to_string(),
6956 description: "List imported Claude cron jobs and their explicit ACTIVE or PAUSED \
6957 manifest posture. This agent loop itself does not run a scheduler."
6958 .to_string(),
6959 parameters: serde_json::json!({
6960 "type": "object",
6961 "properties": {},
6962 "additionalProperties": false
6963 }),
6964 }
6965 }
6966
6967 fn claude_schedule_wakeup_schema() -> ToolSchema {
6968 ToolSchema {
6969 name: CLAUDE_SCHEDULE_WAKEUP.to_string(),
6970 description: "Replace the one-shot wakeup stored in the imported Claude manifest. \
6971 The wakeup inherits the manifest's ACTIVE or PAUSED posture; an embedding scheduler \
6972 owns timer execution."
6973 .to_string(),
6974 parameters: serde_json::json!({
6975 "type": "object",
6976 "properties": {
6977 "delaySeconds": {"type": "integer", "minimum": 0},
6978 "reason": {"type": "string"},
6979 "prompt": {"type": "string"}
6980 },
6981 "required": ["delaySeconds"],
6982 "additionalProperties": false
6983 }),
6984 }
6985 }
6986
6987 /// The `background_exec` schema (P5-6, §2 module 4, D1 "background
6988 /// exec").
6989 fn background_exec_schema() -> ToolSchema {
6990 ToolSchema {
6991 name: BACKGROUND_EXEC.to_string(),
6992 description: "Run a shell command in the BACKGROUND: spawns it as a detached \
6993 process and returns a `job_id` IMMEDIATELY, before the command finishes — this \
6994 call never returns the command's output. Poll `background_status` with the \
6995 `job_id` to check progress and retrieve captured output; use `background_kill` \
6996 to cancel it early. The command goes through the exact same sandbox/permission \
6997 checks as a foreground `bash` call, and any check that would need an \
6998 interactive approval is denied automatically (a background job cannot wait for \
6999 one)."
7000 .to_string(),
7001 parameters: serde_json::json!({
7002 "type": "object",
7003 "properties": {
7004 "command": {
7005 "type": "string",
7006 "description": "Shell command to run in the background via `sh -c`."
7007 }
7008 },
7009 "required": ["command"],
7010 "additionalProperties": false
7011 }),
7012 }
7013 }
7014
7015 /// The `background_status` schema (P5-6, D1 "monitor/event feed").
7016 fn background_status_schema() -> ToolSchema {
7017 ToolSchema {
7018 name: BACKGROUND_STATUS.to_string(),
7019 description: "Check on a background job spawned via background_exec: its \
7020 running/exited/killed status, exit code (once known), and the command's \
7021 captured stdout/stderr so far (bounded — very large output is truncated with a \
7022 marker). Once the job has exited or been killed, this call also reaps it (it \
7023 will no longer appear in background_list or accept further status polls)."
7024 .to_string(),
7025 parameters: serde_json::json!({
7026 "type": "object",
7027 "properties": {
7028 "job_id": {
7029 "type": "string",
7030 "description": "The id `background_exec` returned when this job was \
7031 started."
7032 }
7033 },
7034 "required": ["job_id"],
7035 "additionalProperties": false
7036 }),
7037 }
7038 }
7039
7040 /// The `background_list` schema (P5-6, D10 "bg-manager").
7041 fn background_list_schema() -> ToolSchema {
7042 ToolSchema {
7043 name: BACKGROUND_LIST.to_string(),
7044 description: "List every background job currently tracked (running, or finished \
7045 but not yet polled via background_status) — job id, command, status, pid, and \
7046 start time for each. Does not retrieve output or reap anything."
7047 .to_string(),
7048 parameters: serde_json::json!({
7049 "type": "object",
7050 "properties": {},
7051 "additionalProperties": false
7052 }),
7053 }
7054 }
7055
7056 /// The `background_kill` schema (P5-6, D10 "bg-manager").
7057 fn background_kill_schema() -> ToolSchema {
7058 ToolSchema {
7059 name: BACKGROUND_KILL.to_string(),
7060 description: "Kill a background job's real process immediately (a no-op, not an \
7061 error, if it already exited on its own) and reap it."
7062 .to_string(),
7063 parameters: serde_json::json!({
7064 "type": "object",
7065 "properties": {
7066 "job_id": {
7067 "type": "string",
7068 "description": "The id `background_exec` returned when this job was \
7069 started."
7070 }
7071 },
7072 "required": ["job_id"],
7073 "additionalProperties": false
7074 }),
7075 }
7076 }
7077
7078 /// The `spawn_subagent` schema (P5-3, §2 module 9 D1 "spawn tool").
7079 /// Lists every configured `agent_type` name so the model knows what's
7080 /// available, but `agent_type` stays optional — an ad-hoc spawn with an
7081 /// inline `system_prompt` is always allowed too.
7082 fn spawn_subagent_schema(&self) -> ToolSchema {
7083 let mut names: Vec<&str> = self
7084 .config
7085 .subagents_definitions
7086 .keys()
7087 .map(String::as_str)
7088 .collect();
7089 names.sort_unstable();
7090 let agent_type_desc = if names.is_empty() {
7091 "Optional named subagent type to run (none configured — omit this and pass \
7092 `system_prompt` instead)."
7093 .to_string()
7094 } else {
7095 format!(
7096 "Optional named subagent type to run: {}. Omit to run an ad-hoc subagent with \
7097 your own `system_prompt` instead.",
7098 names.join(", ")
7099 )
7100 };
7101 let background_desc = if self.config.subagents_background {
7102 "Run this subagent in the background instead of waiting for it — this call \
7103 returns immediately with a `subagent_id`; poll `subagent_status` with that id for \
7104 the result."
7105 } else {
7106 "Background subagents are disabled for this agent — this must be omitted or false."
7107 };
7108 ToolSchema {
7109 name: SPAWN_SUBAGENT.to_string(),
7110 description: "Spawn a subagent to work on a self-contained task and (by default) \
7111 wait for its final answer, which is returned as this call's result. The \
7112 subagent runs its own independent reasoning/tool loop; it does not see your \
7113 conversation except for the `task` text you give it here."
7114 .to_string(),
7115 parameters: serde_json::json!({
7116 "type": "object",
7117 "properties": {
7118 "task": {
7119 "type": "string",
7120 "description": "The self-contained task/prompt for the subagent."
7121 },
7122 "agent_type": {
7123 "type": "string",
7124 "description": agent_type_desc
7125 },
7126 "system_prompt": {
7127 "type": "string",
7128 "description": "Inline system prompt for an ad-hoc subagent (ignored \
7129 if `agent_type` is given — the named type's own prompt is used \
7130 instead)."
7131 },
7132 "background": {
7133 "type": "boolean",
7134 "description": background_desc
7135 }
7136 },
7137 "required": ["task"],
7138 "additionalProperties": false
7139 }),
7140 }
7141 }
7142
7143 /// Claude Code-compatible alias for [`Self::spawn_subagent_schema`].
7144 fn claude_agent_schema(&self) -> ToolSchema {
7145 let mut names: Vec<String> = self.config.subagents_definitions.keys().cloned().collect();
7146 names.push("general-purpose".into());
7147 names.sort_unstable();
7148 names.dedup();
7149 ToolSchema {
7150 name: CLAUDE_AGENT.to_string(),
7151 description: "Claude Code-compatible subagent dispatcher. Runs a named or ad-hoc \
7152 child agent; children default to background execution in this compatibility mode."
7153 .to_string(),
7154 parameters: serde_json::json!({
7155 "type": "object",
7156 "properties": {
7157 "prompt": {"type": "string", "description": "Self-contained child task."},
7158 "subagent_type": {
7159 "type": "string",
7160 "description": format!("Named agent type. Available: {}", names.join(", "))
7161 },
7162 "description": {
7163 "type": "string",
7164 "description": "Short human-facing task label; preserved as descriptive input."
7165 },
7166 "model": {
7167 "type": "string",
7168 "description": "Optional model alias or full provider slug for this child."
7169 },
7170 "run_in_background": {
7171 "type": "boolean",
7172 "description": "Whether to return immediately with a child id (default true)."
7173 }
7174 },
7175 "required": ["prompt"],
7176 "additionalProperties": false
7177 }),
7178 }
7179 }
7180
7181 /// Translate Claude's `Agent` arguments to the native subagent intrinsic.
7182 fn translate_claude_agent_call(
7183 &self,
7184 call: &supercode_interchange::ToolCall,
7185 ) -> Result<supercode_interchange::ToolCall> {
7186 let args = call
7187 .function
7188 .parsed_arguments()
7189 .map_err(|error| Error::InvalidArguments {
7190 tool: CLAUDE_AGENT.to_string(),
7191 message: error.to_string(),
7192 })?;
7193 let object = args.as_object().ok_or_else(|| Error::InvalidArguments {
7194 tool: CLAUDE_AGENT.to_string(),
7195 message: "arguments must be an object".to_string(),
7196 })?;
7197 let mut translated = serde_json::Map::new();
7198 if let Some(value) = object.get("prompt") {
7199 translated.insert("task".to_string(), value.clone());
7200 }
7201 if let Some(value) = object.get("subagent_type") {
7202 // `general-purpose` is a built-in Claude agent, not a project
7203 // definition file. Supercode's equivalent is an ad-hoc child
7204 // using the inherited default system prompt, represented by an
7205 // omitted `agent_type`.
7206 if value.as_str() != Some("general-purpose") {
7207 translated.insert("agent_type".to_string(), value.clone());
7208 }
7209 }
7210 if let Some(value) = object.get("model") {
7211 translated.insert("model".to_string(), value.clone());
7212 }
7213 translated.insert(
7214 "background".to_string(),
7215 object
7216 .get("run_in_background")
7217 .cloned()
7218 .unwrap_or(serde_json::Value::Bool(true)),
7219 );
7220 Ok(supercode_interchange::ToolCall {
7221 id: call.id.clone(),
7222 kind: call.kind.clone(),
7223 function: supercode_interchange::FunctionCall {
7224 name: SPAWN_SUBAGENT.to_string(),
7225 arguments: serde_json::Value::Object(translated).to_string(),
7226 },
7227 })
7228 }
7229
7230 /// Execute Claude's scheduling vocabulary against the imported manifest.
7231 ///
7232 /// This is intentionally a state editor, not a scheduler: it owns no
7233 /// timer/task handle, nothing downstream of it fires, and every
7234 /// successful response says so, so the model is never told a job it just
7235 /// created will run here.
7236 fn run_claude_runtime_tool(
7237 &mut self,
7238 call: &supercode_interchange::ToolCall,
7239 ) -> (String, bool) {
7240 let args = match call.function.parsed_arguments() {
7241 Ok(value) if value.is_object() => value,
7242 Ok(_) => {
7243 return (
7244 format!("Error: {} arguments must be an object", call.function.name),
7245 true,
7246 )
7247 }
7248 Err(error) => return (format!("Error: {error}"), true),
7249 };
7250 let object = args.as_object().expect("checked object above");
7251
7252 let Some(manifest) = self.claude_runtime_manifest.as_mut() else {
7253 return (
7254 "Error: Claude runtime compatibility was enabled without an imported runtime \
7255 manifest; refusing to invent scheduler state"
7256 .to_string(),
7257 true,
7258 );
7259 };
7260 // Every imported schedule is carried and inert. `state` is the single
7261 // fact the model is told about it, in the manifest's own vocabulary.
7262 let state = "paused";
7263
7264 match call.function.name.as_str() {
7265 CLAUDE_CRON_LIST => {
7266 let jobs: Vec<serde_json::Value> = manifest
7267 .active_crons
7268 .iter()
7269 .map(|job| {
7270 serde_json::json!({
7271 "id": job.id,
7272 "cron": job.schedule,
7273 "prompt": job.prompt,
7274 "recurring": job.recurring,
7275 "durable": job.durable_requested,
7276 "state": state
7277 })
7278 })
7279 .collect();
7280 let notice = "Imported jobs are preserved but no scheduler is running.";
7281 (
7282 serde_json::json!({
7283 "execution_state": state,
7284 "execution_notice": notice,
7285 "jobs": jobs
7286 })
7287 .to_string(),
7288 false,
7289 )
7290 }
7291 CLAUDE_CRON_CREATE => {
7292 let Some(schedule) = object.get("cron").and_then(serde_json::Value::as_str) else {
7293 return ("Error: CronCreate requires string `cron`".to_string(), true);
7294 };
7295 let Some(prompt) = object.get("prompt").and_then(serde_json::Value::as_str) else {
7296 return (
7297 "Error: CronCreate requires string `prompt`".to_string(),
7298 true,
7299 );
7300 };
7301 let recurring = object
7302 .get("recurring")
7303 .and_then(serde_json::Value::as_bool)
7304 .unwrap_or(false);
7305 let durable_requested = object
7306 .get("durable")
7307 .and_then(serde_json::Value::as_bool)
7308 .unwrap_or(false);
7309 let mut sequence = 1_u64;
7310 let id = loop {
7311 let candidate = format!("sc{sequence:06}");
7312 if !manifest.active_crons.iter().any(|job| job.id == candidate) {
7313 break candidate;
7314 }
7315 sequence += 1;
7316 };
7317 let kind = if recurring { "recurring " } else { "" };
7318 let result = format!(
7319 "Scheduled {kind}job {id} ({schedule}) in PAUSED state. The job is preserved \
7320 in the continuation manifest but no scheduler is running and it will not execute."
7321 );
7322 manifest
7323 .active_crons
7324 .push(crate::claude_runtime_state::ClaudeCronJob {
7325 id: id.clone(),
7326 tool_use_id: call.id.clone(),
7327 schedule: schedule.to_string(),
7328 recurring,
7329 durable_requested,
7330 prompt: prompt.to_string(),
7331 // The creation instant is a fact about this
7332 // continuation, recorded like every other manifest
7333 // field. Nothing consults it as a due time.
7334 created_at: Some(supercode_interchange::sidecar::ms_to_rfc3339(now_ms())),
7335 expires_after_seconds: None,
7336 creation_result: result.clone(),
7337 });
7338 manifest
7339 .active_crons
7340 .sort_by(|left, right| left.id.cmp(&right.id));
7341 (result, false)
7342 }
7343 CLAUDE_CRON_DELETE => {
7344 let Some(id) = object.get("id").and_then(serde_json::Value::as_str) else {
7345 return ("Error: CronDelete requires string `id`".to_string(), true);
7346 };
7347 let Some(index) = manifest.active_crons.iter().position(|job| job.id == id) else {
7348 return (
7349 format!("Error: unknown {state} Claude cron job `{id}`"),
7350 true,
7351 );
7352 };
7353 manifest.active_crons.remove(index);
7354 (
7355 format!("Cancelled job {id}. The job was PAUSED; no execution occurred."),
7356 false,
7357 )
7358 }
7359 CLAUDE_SCHEDULE_WAKEUP => {
7360 let Some(delay_seconds) = object
7361 .get("delaySeconds")
7362 .and_then(serde_json::Value::as_u64)
7363 else {
7364 return (
7365 "Error: ScheduleWakeup requires integer `delaySeconds`".to_string(),
7366 true,
7367 );
7368 };
7369 let reason = object
7370 .get("reason")
7371 .and_then(serde_json::Value::as_str)
7372 .map(str::to_string);
7373 let prompt = object
7374 .get("prompt")
7375 .and_then(serde_json::Value::as_str)
7376 .map(str::to_string);
7377 let now = now_ms();
7378 let created_at = Some(supercode_interchange::sidecar::ms_to_rfc3339(now));
7379 // The instant the wakeup asks for, recorded as the request
7380 // made it. No timer consults it here.
7381 let delay_ms = i64::try_from(delay_seconds)
7382 .unwrap_or(i64::MAX)
7383 .saturating_mul(1_000);
7384 let scheduled_for =
7385 supercode_interchange::sidecar::ms_to_rfc3339(now.saturating_add(delay_ms));
7386 let result = format!(
7387 "Next wakeup recorded for {scheduled_for} (in {delay_seconds}s) in PAUSED \
7388 state. The request replaced the prior wakeup in the manifest, but no timer \
7389 is running and it will not execute."
7390 );
7391 manifest.pending_wakeups.clear();
7392 manifest
7393 .pending_wakeups
7394 .push(crate::claude_runtime_state::ClaudeWakeup {
7395 tool_use_id: call.id.clone(),
7396 delay_seconds,
7397 reason,
7398 prompt,
7399 created_at,
7400 scheduled_for: Some(scheduled_for),
7401 creation_result: result.clone(),
7402 });
7403 (result, false)
7404 }
7405 _ => unreachable!("runtime tool dispatch is name-gated"),
7406 }
7407 }
7408
7409 /// The `subagent_status` schema (P5-3, D3 "background+resume").
7410 fn subagent_status_schema() -> ToolSchema {
7411 ToolSchema {
7412 name: SUBAGENT_STATUS.to_string(),
7413 description: "Check on (and, once finished, retrieve the result of) a background \
7414 subagent spawned via spawn_subagent with background=true. Pass the \
7415 `subagent_id` that spawn returned."
7416 .to_string(),
7417 parameters: serde_json::json!({
7418 "type": "object",
7419 "properties": {
7420 "subagent_id": {
7421 "type": "string",
7422 "description": "The id `spawn_subagent` returned when this subagent \
7423 was spawned."
7424 }
7425 },
7426 "required": ["subagent_id"],
7427 "additionalProperties": false
7428 }),
7429 }
7430 }
7431
7432 /// The `send_message` schema.
7433 fn send_message_schema() -> ToolSchema {
7434 ToolSchema {
7435 name: SEND_MESSAGE.to_string(),
7436 description: "Send a message to another agent: one of your background subagents \
7437 that is STILL RUNNING (by the id `spawn_subagent` returned; it receives it at the \
7438 start of its next step), or another session of any harness, on this machine or an \
7439 enrolled one (by its name, name@machine, or sc: address; `supercode message list` \
7440 shows them). A session's reply comes back to you as a message. Send only what \
7441 asks something or carries a result; no acknowledgements."
7442 .to_string(),
7443 parameters: serde_json::json!({
7444 "type": "object",
7445 "properties": {
7446 "to": {
7447 "type": "string",
7448 "description": "A running subagent's id, or a session's name, name@machine or sc: address."
7449 },
7450 "message": {
7451 "type": "string",
7452 "description": "What to tell it."
7453 },
7454 "notify_when_idle": {
7455 "type": "boolean",
7456 "description": "For a session: also get one notice when its next turn ends."
7457 }
7458 },
7459 "required": ["to", "message"],
7460 "additionalProperties": false
7461 }),
7462 }
7463 }
7464
7465 /// BP-7: the `subagent_resume` schema.
7466 fn subagent_resume_schema() -> ToolSchema {
7467 ToolSchema {
7468 name: SUBAGENT_RESUME.to_string(),
7469 description: "Continue a subagent that has already FINISHED, with its own previous conversation restored, so it keeps everything it learned instead of being briefed again from scratch. Pass the id it was spawned with and the next task."
7470 .to_string(),
7471 parameters: serde_json::json!({
7472 "type": "object",
7473 "properties": {
7474 "subagent_id": {
7475 "type": "string",
7476 "description": "The id of a subagent that has already finished."
7477 },
7478 "task": {
7479 "type": "string",
7480 "description": "What the resumed subagent should do next."
7481 }
7482 },
7483 "required": ["subagent_id", "task"],
7484 "additionalProperties": false
7485 }),
7486 }
7487 }
7488
7489 /// BP-7 (catalog §4a "Background subagents + resume"): deliver a
7490 /// message into a still-running background child's mailbox.
7491 ///
7492 /// The mailbox is the child's own `SteerInbox` — the seam P4b built for
7493 /// mid-turn steering, which is writable while the child's turn holds
7494 /// `&mut Agent`. So delivery ordering is already defined: the message
7495 /// arrives at the top of the child's next loop iteration, i.e. after
7496 /// whatever tool calls it is currently running, per its
7497 /// `steering_mode`. A child that has already FINISHED is refused with
7498 /// a pointer at `subagent_resume`, which is the operation for that
7499 /// case — never silently dropped.
7500 async fn run_send_message(&mut self, call: &supercode_interchange::ToolCall) -> (String, bool) {
7501 let args = match call.function.parsed_arguments() {
7502 Ok(v) => v,
7503 Err(e) => {
7504 let err = Error::InvalidArguments {
7505 tool: SEND_MESSAGE.to_string(),
7506 message: e.to_string(),
7507 };
7508 return (format!("Error: {err}"), true);
7509 }
7510 };
7511 let Some(id) = args.get("to").and_then(serde_json::Value::as_str) else {
7512 let err = Error::InvalidArguments {
7513 tool: SEND_MESSAGE.to_string(),
7514 message: "`to` is required".to_string(),
7515 };
7516 return (format!("Error: {err}"), true);
7517 };
7518 let message = args
7519 .get("message")
7520 .and_then(serde_json::Value::as_str)
7521 .unwrap_or("");
7522 if message.is_empty() {
7523 let err = Error::InvalidArguments {
7524 tool: SEND_MESSAGE.to_string(),
7525 message: "`message` is required and must be non-empty".to_string(),
7526 };
7527 return (format!("Error: {err}"), true);
7528 }
7529 let Some(entry) = self.background_subagents.get(id) else {
7530 // Not one of this agent's children: another session.
7531 let notify_when_idle = args
7532 .get("notify_when_idle")
7533 .and_then(serde_json::Value::as_bool)
7534 .unwrap_or(false);
7535 return send_to_session(id, message, notify_when_idle).await;
7536 };
7537 if entry.handle.is_finished() {
7538 let out = serde_json::json!({
7539 "subagent_id": id,
7540 "status": "finished",
7541 "delivered": false,
7542 "hint": "this subagent already finished — collect it with subagent_status, then continue it with subagent_resume",
7543 });
7544 return (out.to_string(), false);
7545 }
7546 entry
7547 .mailbox
7548 .lock()
7549 .unwrap_or_else(std::sync::PoisonError::into_inner)
7550 .queue_unchecked(message.to_string());
7551 let out = serde_json::json!({
7552 "subagent_id": id,
7553 "status": "running",
7554 "delivered": true,
7555 });
7556 (out.to_string(), false)
7557 }
7558
7559 /// BP-7 (catalog §4a "Background subagents + resume": "resumable with
7560 /// context intact"): continue a finished child over its OWN transcript.
7561 ///
7562 /// The context comes from the reap (kept in-process) or, for a session
7563 /// that attached a subagent store, from that child's persisted
7564 /// `<parent>.subagents/<id>.sidecar.jsonl`. Either way the resumed
7565 /// child is rebuilt through `build_child_config` from the SAME named
7566 /// definition it was spawned with, so its permission posture on resume
7567 /// is the one it had originally — never a fresh, looser default.
7568 async fn run_subagent_resume(
7569 &mut self,
7570 call: &supercode_interchange::ToolCall,
7571 ) -> (String, bool) {
7572 let args = match call.function.parsed_arguments() {
7573 Ok(v) => v,
7574 Err(e) => {
7575 let err = Error::InvalidArguments {
7576 tool: SUBAGENT_RESUME.to_string(),
7577 message: e.to_string(),
7578 };
7579 return (format!("Error: {err}"), true);
7580 }
7581 };
7582 let Some(id) = args
7583 .get("subagent_id")
7584 .and_then(serde_json::Value::as_str)
7585 .map(String::from)
7586 else {
7587 let err = Error::InvalidArguments {
7588 tool: SUBAGENT_RESUME.to_string(),
7589 message: "`subagent_id` is required".to_string(),
7590 };
7591 return (format!("Error: {err}"), true);
7592 };
7593 let task = args
7594 .get("task")
7595 .and_then(serde_json::Value::as_str)
7596 .unwrap_or("")
7597 .to_string();
7598 if task.is_empty() {
7599 let err = Error::InvalidArguments {
7600 tool: SUBAGENT_RESUME.to_string(),
7601 message: "`task` is required and must be non-empty".to_string(),
7602 };
7603 return (format!("Error: {err}"), true);
7604 }
7605 if self
7606 .background_subagents
7607 .get(&id)
7608 .is_some_and(|e| !e.handle.is_finished())
7609 {
7610 let err = Error::tool(
7611 SUBAGENT_RESUME,
7612 format!(
7613 "subagent `{id}` is still running — send it a message with send_message, or collect it with subagent_status first"
7614 ),
7615 );
7616 return (format!("Error: {err}"), true);
7617 }
7618 let lineage = self
7619 .subagent_store
7620 .as_ref()
7621 .and_then(|(store, parent)| store.load_subagent_lineage(parent, &id).ok().flatten());
7622 let Some(prior) = self.prior_subagent_transcript(&id) else {
7623 let err = Error::SubagentNotFound(id.clone());
7624 return (format!("Error: {err}"), true);
7625 };
7626
7627 let agent_type = lineage.as_ref().and_then(|l| l.agent_type.clone());
7628 let definition = agent_type
7629 .as_ref()
7630 .and_then(|name| self.config.subagents_definitions.get(name).cloned());
7631 let Some(guard) = crate::subagents::try_acquire(
7632 &self.subagent_concurrency_gauge,
7633 self.config.subagents_max_concurrent,
7634 ) else {
7635 let err = Error::SubagentConcurrencyExceeded {
7636 max_concurrent: self.config.subagents_max_concurrent,
7637 };
7638 return (format!("Error: {err}"), true);
7639 };
7640 let child_config = self.build_child_config(
7641 definition.as_ref(),
7642 None,
7643 lineage
7644 .as_ref()
7645 .map(|l| l.model.clone())
7646 .or_else(|| definition.as_ref().and_then(|d| d.model.clone())),
7647 );
7648 let mut child = Agent::with_provider_arc(child_config, self.provider.clone());
7649 child.subagent_depth = self.subagent_depth + 1;
7650 child.subagent_concurrency_gauge = self.subagent_concurrency_gauge.clone();
7651 // Context intact: the child's own prior messages, appended after
7652 // its (re-derived, identical) system prompt.
7653 child.history.extend(prior);
7654
7655 let result = child.send(task).await;
7656 let transcript = child.history()[1..].to_vec();
7657 if let Some(lineage) = &lineage {
7658 self.persist_subagent_transcript(&id, lineage, &transcript);
7659 }
7660 self.reaped_subagents.insert(id.clone(), transcript);
7661 drop(guard);
7662 match result {
7663 Ok(text) => {
7664 let out = serde_json::json!({
7665 "subagent_id": id,
7666 "status": "done",
7667 "resumed": true,
7668 "result": text,
7669 });
7670 (out.to_string(), false)
7671 }
7672 Err(e) => {
7673 let out = serde_json::json!({
7674 "subagent_id": id,
7675 "status": "error",
7676 "resumed": true,
7677 "message": e.to_string(),
7678 });
7679 (out.to_string(), true)
7680 }
7681 }
7682 }
7683
7684 /// BP-7 (catalog §4a "Named agent definitions as data"): the child
7685 /// `Config` a `spawn_subagent` of `agent_type` would build — the
7686 /// resolved posture a named definition actually produces, including
7687 /// its [`crate::subagents::AgentPermissions`] bundle applied through
7688 /// the tightening-only rules. `None` when no definition of that name
7689 /// is registered or discovered.
7690 ///
7691 /// Exposed so a caller (and this build's tests) can ask what a named
7692 /// agent WOULD run as without spawning it and paying for a turn.
7693 pub fn child_config_for_agent_type(&self, agent_type: &str) -> Option<Config> {
7694 let definition = self.config.subagents_definitions.get(agent_type)?.clone();
7695 Some(self.build_child_config(Some(&definition), None, definition.model.clone()))
7696 }
7697
7698 /// BP-7 (catalog §4a "Background subagents + resume"): the ids of
7699 /// children that have finished and been reaped, and can therefore be
7700 /// continued with [`SUBAGENT_RESUME`].
7701 pub fn reaped_subagent_ids(&self) -> Vec<String> {
7702 let mut ids: Vec<String> = self.reaped_subagents.keys().cloned().collect();
7703 ids.sort();
7704 ids
7705 }
7706
7707 /// BP-7: a finished child's own messages — from the in-process reap
7708 /// cache first, then this session's subagent store.
7709 fn prior_subagent_transcript(&self, id: &str) -> Option<Vec<ChatMessage>> {
7710 if let Some(messages) = self.reaped_subagents.get(id) {
7711 return Some(messages.clone());
7712 }
7713 let (store, parent) = self.subagent_store.as_ref()?;
7714 let jsonl = store.load_subagent_transcript(parent, id).ok()??;
7715 let session = supercode_interchange::session::Session::from_sidecar_str(&jsonl).ok()?;
7716 Some(
7717 session
7718 .messages
7719 .into_iter()
7720 .filter(|m| m.role != supercode_interchange::Role::System)
7721 .collect(),
7722 )
7723 }
7724
7725 /// Build the CHILD `Config` a `spawn_subagent` call constructs its
7726 /// [`Agent`] from. The whole point of this method (§5.3-style
7727 /// "monotonic posture", build-brief "a subagent inherits or narrows —
7728 /// never widens — the parent's permission posture"): every field that
7729 /// governs what the child is ALLOWED to do (sandbox, approval,
7730 /// tool_overrides, deny/allow patterns, protected paths, the subagents
7731 /// caps themselves) is copied VERBATIM from `self.config` — never
7732 /// loosened — and the only NARROWING lever is `definition.tools`
7733 /// (intersected with whatever the parent already had enabled, never
7734 /// unioned in anything new).
7735 ///
7736 /// P5-3 safety hardening (Fable-5 review, LOW-MEDIUM "child safety-limit
7737 /// inheritance"): the monotonic-posture guarantee above was, before this
7738 /// fix, scoped to PERMISSION fields only — a child could still silently
7739 /// get a LOOSER safety BUDGET/BREAKER than its parent, because
7740 /// `max_total_output_tokens`/`max_tool_output_bytes`/`max_tokens`/
7741 /// `doom_loop_threshold`/`edit_file_require_read_before_edit` were never
7742 /// copied and so fell back to `Config::default()`'s (looser/uncapped)
7743 /// values on every spawn regardless of what the parent had configured.
7744 /// These are now copied verbatim alongside the permission-posture
7745 /// fields — a parent that capped its own output/tool-output/doom-loop
7746 /// exposure, or required read-before-edit, gets a child that is bound
7747 /// by the exact same ceiling, never a wider one.
7748 ///
7749 /// **Full field-by-field accounting** (every [`Config`] field, so this
7750 /// doc comment stays the single place that answers "did we forget
7751 /// one?"): fields already copied above/below this note (permission
7752 /// posture: `sandbox`/`approval`/`tool_overrides`/`auto_approved_tools`/
7753 /// `tool_deny_patterns`/`tool_allow_patterns`/`permissions_enabled`/
7754 /// `permissions_ask_patterns`/`permissions_protected_paths`/
7755 /// `network_policy`/`core_tools_enabled`/`module_registry`/
7756 /// `module_activation`/every `subagents_*` field; safety limits:
7757 /// `max_iterations`/`max_total_output_tokens`/`max_tool_output_bytes`/
7758 /// `max_tokens`/`doom_loop_threshold`/`edit_file_require_read_before_edit`;
7759 /// identity/transport: `model`/`system_prompt`/`cwd`/`base_url`/
7760 /// `api_key`/`api_key_env`/`api_key_cmd`) are the ones that gate
7761 /// harm/spend/hazard exposure. Every OTHER field is deliberately left at
7762 /// `Config::default()` because none of them is a safety ceiling the
7763 /// child could "loosen" by missing it:
7764 /// - `temperature`/`effort`/`response_format`/`extra_body`/`extra_headers`/
7765 /// `tool_advertising`/`tool_schema_tier`/`cache_plan`/`cache_warnings`/
7766 /// `reduction_policy`/
7767 /// `session_*`/`small_model`/`model_fallback`/`env_context`/
7768 /// `project_root_markers`/`project_doc_max_bytes`/`instruction_imports`/
7769 /// `retry_*`/`compaction_*`/`auto_title`/`steering_mode`/
7770 /// `follow_up_mode`/`read_file_multimodal`/`edit_file_notebook_aware`/
7771 /// `shell_env_snapshot`/`nested_instructions`/`model_switch_allow_switch`/
7772 /// `context_injections`/`context_injection_blocks`/`parallel_tool_calls`
7773 /// are behavior/cost-shaping or presentation knobs, not hard guards —
7774 /// a child defaulting on any of these can do LESS (e.g. no multimodal
7775 /// read, no notebook-aware edits, no proactive compaction) or the same,
7776 /// never something the parent hadn't already exposed it to. Several
7777 /// default to their OFF/conservative state (`false`/`None`), which is
7778 /// the tight direction, not the loose one.
7779 /// - `additional_dirs`: governs which extra roots are reachable at all
7780 /// (`presets.rs`'s `[core] additional_dirs` note) — a child that
7781 /// doesn't inherit it has FEWER reachable roots than its parent, i.e.
7782 /// strictly tighter, never looser.
7783 /// - `load_project_context`: whether instruction files are auto-loaded
7784 /// into the system prompt — a read-time convenience, not an access
7785 /// grant (`sandbox`/`permissions_protected_paths` already gate actual
7786 /// file access).
7787 /// - `prompts`: named `/slash` command templates for THIS agent's own
7788 /// user-facing input surface, not something the model can invoke
7789 /// against the child's tool surface.
7790 /// - `stop_gate`/`post_tool_hook`/`approval_handler`/`event_sink`:
7791 /// code-only `Box<dyn Fn>` callbacks (see the `pre_tool_hook` note
7792 /// immediately below — same non-`Clone` shape) that are observational
7793 /// or terminate-only, not a call-time veto over what a tool is allowed
7794 /// to do; `approval_handler` specifically is ALREADY documented at
7795 /// this method's call site (`Self::run_spawn_subagent`) as
7796 /// intentionally never set here — a foreground child gets no handler
7797 /// by design, an embedder installs its own after spawn if it wants
7798 /// one.
7799 ///
7800 /// **`pre_tool_hook` cannot propagate, and this is deliberate + named,
7801 /// not a silent gap**: `Config::pre_tool_hook` is a `Box<dyn Fn(&str,
7802 /// &serde_json::Value) -> Option<String> + Send + Sync>` — an
7803 /// embedder's own call-time veto over every tool call. `Box<dyn Fn>` is
7804 /// not `Clone` (there is no generic way to duplicate an opaque closure),
7805 /// so it genuinely CANNOT be copied into a child `Config` the way every
7806 /// `Clone`-able field above is — there is no fix that makes this one
7807 /// "verbatim copy" like the others. An embedder relying on a
7808 /// `pre_tool_hook` veto reaching spawned children as well as the parent
7809 /// MUST re-install one on the child explicitly (e.g. via a
7810 /// `spawn_subagent`-adjacent hook of their own, or by not relying on
7811 /// `pre_tool_hook` alone for anything safety-critical across a spawn
7812 /// boundary) — named here so this is a documented contract, not a gap
7813 /// an embedder discovers by a child silently misbehaving.
7814 fn build_child_config(
7815 &self,
7816 definition: Option<&crate::subagents::NamedAgentDefinition>,
7817 inline_system_prompt: Option<String>,
7818 model_override: Option<String>,
7819 ) -> Config {
7820 let system_prompt = definition
7821 .map(|d| d.system_prompt.clone())
7822 .filter(|s| !s.is_empty())
7823 .or(inline_system_prompt)
7824 .unwrap_or_else(|| self.config.system_prompt.clone());
7825 let model = model_override.unwrap_or_else(|| self.config.model.clone());
7826
7827 let mut child = Config::builder()
7828 .model(model)
7829 .system_prompt(system_prompt)
7830 .cwd(self.config.cwd.clone())
7831 // Monotonic: verbatim, never loosened.
7832 .sandbox(self.config.sandbox)
7833 .approval(self.config.approval)
7834 .max_iterations(self.config.max_iterations)
7835 .build();
7836 child.base_url = self.config.base_url.clone();
7837 child.api_key = self.config.api_key.clone();
7838 child.api_key_env = self.config.api_key_env.clone();
7839 child.api_key_cmd = self.config.api_key_cmd.clone();
7840 // P5-3 safety hardening (Fable-5 review, LOW-MEDIUM "child
7841 // safety-limit inheritance"): the monotonic-posture spirit extends
7842 // to safety BUDGETS/BREAKERS, not just permissions — a child must
7843 // not get a looser cap/breaker than its parent by simply falling
7844 // back to `Config::default()`'s (looser) values. See this method's
7845 // doc comment for the full field-by-field accounting.
7846 child.max_total_output_tokens = self.config.max_total_output_tokens;
7847 child.max_tool_output_bytes = self.config.max_tool_output_bytes;
7848 child.max_tokens = self.config.max_tokens;
7849 child.doom_loop_threshold = self.config.doom_loop_threshold;
7850 child.edit_file_require_read_before_edit = self.config.edit_file_require_read_before_edit;
7851 // Monotonic tool posture: start from the PARENT's own overrides
7852 // (so anything the parent already disabled stays disabled), then
7853 // narrow further if a named definition restricts the tool set.
7854 child.tool_overrides = self.config.tool_overrides.clone();
7855 child.auto_approved_tools = self.config.auto_approved_tools.clone();
7856 child.tool_deny_patterns = self.config.tool_deny_patterns.clone();
7857 child.tool_allow_patterns = self.config.tool_allow_patterns.clone();
7858 child.permissions_enabled = self.config.permissions_enabled;
7859 child.permissions_ask_patterns = self.config.permissions_ask_patterns.clone();
7860 child.permissions_protected_paths = self.config.permissions_protected_paths.clone();
7861 child.network_policy = self.config.network_policy.clone();
7862 // P5-10 (§2 module 12): same monotonic-posture treatment as
7863 // `sandbox`/`approval` above — a subagent must inherit its
7864 // parent's OS-sandbox posture verbatim, never a looser
7865 // `Config::default()` fallback (`sandbox_os_enabled: None`,
7866 // `escalation: Deny`, `env_policy: Inherit` would otherwise be
7867 // right back to "confine only when the tier itself says so" for a
7868 // child whose parent explicitly forced the backstop on/off).
7869 child.sandbox_os_enabled = self.config.sandbox_os_enabled;
7870 child.sandbox_escalation = self.config.sandbox_escalation;
7871 child.sandbox_env_policy = self.config.sandbox_env_policy;
7872 if let Some(def) = definition {
7873 if let Some(allowed) = &def.tools {
7874 for name in &self.config.core_tools_enabled {
7875 if !allowed.iter().any(|t| t == name) {
7876 child
7877 .tool_overrides
7878 .entry(name.clone())
7879 .or_default()
7880 .enabled = Some(false);
7881 }
7882 }
7883 }
7884 // BP-7 (catalog §4a "Named agent definitions as data": the
7885 // `permissions` component of `prompt+model+tools+permissions`).
7886 // Every arm below can only TIGHTEN — the two policy values go
7887 // through the SAME strictness ranks `configfile::
7888 // clamp_project_permissions` uses for the untrusted project
7889 // layer (a looser value is ignored, never honored), the
7890 // auto-approve list is INTERSECTED with the parent's, and the
7891 // deny list is a union. A definition may come from a
7892 // `.claude/agents/*.md` file in the repo, so it sits at the
7893 // project trust tier and must never be an escalation door.
7894 if let Some(perms) = &def.permissions {
7895 if let Some(approval) = perms.approval {
7896 if crate::configfile::approval_rank(approval)
7897 < crate::configfile::approval_rank(child.approval)
7898 {
7899 child.approval = approval;
7900 }
7901 }
7902 if let Some(sandbox) = perms.sandbox {
7903 if crate::configfile::sandbox_rank(sandbox)
7904 < crate::configfile::sandbox_rank(child.sandbox)
7905 {
7906 child.sandbox = sandbox;
7907 }
7908 }
7909 if let Some(allowed) = &perms.auto_approved_tools {
7910 child
7911 .auto_approved_tools
7912 .retain(|tool| allowed.iter().any(|a| a == tool));
7913 }
7914 for pattern in &perms.deny {
7915 if !child.tool_deny_patterns.iter().any(|p| p == pattern) {
7916 child.tool_deny_patterns.push(pattern.clone());
7917 }
7918 }
7919 }
7920 }
7921 child.core_tools_enabled = self.config.core_tools_enabled.clone();
7922 child.module_registry = self.config.module_registry;
7923 child.module_activation = self.config.module_activation.clone();
7924 // The subagents module itself never widens either: a child spawned
7925 // at depth d+1 inherits the SAME caps (never a looser depth/
7926 // concurrency/background posture than its own parent).
7927 child.subagents_enabled = self.config.subagents_enabled;
7928 child.subagents_max_depth = self.config.subagents_max_depth;
7929 child.subagents_max_concurrent = self.config.subagents_max_concurrent;
7930 child.subagents_background = self.config.subagents_background;
7931 child.subagents_background_prompts = self.config.subagents_background_prompts;
7932 child.subagents_claude_agent_alias = self.config.subagents_claude_agent_alias;
7933 child.subagents_definitions = self.config.subagents_definitions.clone();
7934 child.subagent_depth = self.subagent_depth + 1;
7935 child
7936 }
7937
7938 /// Execute the `spawn_subagent` intrinsic (P5-3, §2 module 9). See
7939 /// `Self::build_child_config` for the monotonic-posture guarantee and
7940 /// `crate::subagents` for the depth/concurrency resource bounds and the
7941 /// §2.2 C6 background-policy enforcement.
7942 async fn run_spawn_subagent(
7943 &mut self,
7944 call: &supercode_interchange::ToolCall,
7945 ) -> (String, bool) {
7946 // BP-11: `subagent_start`/`subagent_stop` bracket a call that passed
7947 // the same validation the runner applies (subagents on, non-empty
7948 // task); a refused call fires neither.
7949 let task = if self.config.subagents_enabled {
7950 call.function
7951 .parsed_arguments()
7952 .ok()
7953 .and_then(|v| {
7954 v.get("task")
7955 .and_then(serde_json::Value::as_str)
7956 .map(str::to_string)
7957 })
7958 .filter(|t| !t.is_empty())
7959 } else {
7960 None
7961 };
7962 if let Some(task) = &task {
7963 self.fire_lifecycle(&crate::config::LifecycleEvent::SubagentStart {
7964 task: task.clone(),
7965 });
7966 }
7967 let (output, is_error) = self.run_spawn_subagent_inner(call).await;
7968 if let Some(task) = task {
7969 self.fire_lifecycle(&crate::config::LifecycleEvent::SubagentStop {
7970 task,
7971 is_error,
7972 output_len: output.len(),
7973 });
7974 }
7975 (output, is_error)
7976 }
7977
7978 /// Hands a lifecycle moment to the installed observer, if any (BP-11).
7979 fn fire_lifecycle(&self, event: &crate::config::LifecycleEvent) {
7980 if let Some(hook) = self.config.lifecycle_hook.as_ref() {
7981 hook(event);
7982 }
7983 }
7984
7985 /// Installs the lifecycle observer (compaction and subagent boundaries).
7986 pub fn set_lifecycle_hook(&mut self, hook: crate::config::LifecycleHook) {
7987 self.config.lifecycle_hook = Some(hook);
7988 }
7989
7990 async fn run_spawn_subagent_inner(
7991 &mut self,
7992 call: &supercode_interchange::ToolCall,
7993 ) -> (String, bool) {
7994 if !self.config.subagents_enabled {
7995 let err = Error::UnknownTool(SPAWN_SUBAGENT.to_string());
7996 return (format!("Error: {err}"), true);
7997 }
7998 let args = match call.function.parsed_arguments() {
7999 Ok(v) => v,
8000 Err(e) => {
8001 let err = Error::InvalidArguments {
8002 tool: SPAWN_SUBAGENT.to_string(),
8003 message: e.to_string(),
8004 };
8005 return (format!("Error: {err}"), true);
8006 }
8007 };
8008 let task = args
8009 .get("task")
8010 .and_then(serde_json::Value::as_str)
8011 .unwrap_or("")
8012 .to_string();
8013 if task.is_empty() {
8014 let err = Error::InvalidArguments {
8015 tool: SPAWN_SUBAGENT.to_string(),
8016 message: "`task` is required and must be non-empty".to_string(),
8017 };
8018 return (format!("Error: {err}"), true);
8019 }
8020 let agent_type = args
8021 .get("agent_type")
8022 .and_then(serde_json::Value::as_str)
8023 .map(String::from);
8024 let inline_system_prompt = args
8025 .get("system_prompt")
8026 .and_then(serde_json::Value::as_str)
8027 .map(String::from);
8028 let background = args
8029 .get("background")
8030 .and_then(serde_json::Value::as_bool)
8031 .unwrap_or(false);
8032 let requested_model = args
8033 .get("model")
8034 .and_then(serde_json::Value::as_str)
8035 .map(|model| crate::model_catalog::resolve_alias(model));
8036
8037 let definition = match &agent_type {
8038 Some(name) => match self.config.subagents_definitions.get(name) {
8039 Some(d) => Some(d.clone()),
8040 None => {
8041 let err = Error::SubagentDefinitionNotFound(name.clone());
8042 return (format!("Error: {err}"), true);
8043 }
8044 },
8045 None => None,
8046 };
8047
8048 if background {
8049 if !self.config.subagents_background {
8050 let err = Error::tool(
8051 SPAWN_SUBAGENT,
8052 "background=true requires capabilities.subagents.background = true",
8053 );
8054 return (format!("Error: {err}"), true);
8055 }
8056 // §2.2 C6, defensive re-check (belt-and-suspenders — see
8057 // `Error::SubagentBackgroundPolicyMissing`'s doc comment for why
8058 // this can't just trust the resolver already checked it).
8059 if self.config.subagents_background_prompts.is_none() {
8060 let err = Error::SubagentBackgroundPolicyMissing;
8061 return (format!("Error: {err}"), true);
8062 }
8063 }
8064
8065 // Resource bounds (fail-closed): depth first (cheap, no side
8066 // effect on failure), THEN concurrency (holds a slot — must be the
8067 // LAST check before actually spawning, so a refused spawn never
8068 // leaves a stray slot held).
8069 if let Err(e) =
8070 crate::subagents::check_depth(self.subagent_depth, self.config.subagents_max_depth)
8071 {
8072 return (format!("Error: {e}"), true);
8073 }
8074 let Some(guard) = crate::subagents::try_acquire(
8075 &self.subagent_concurrency_gauge,
8076 self.config.subagents_max_concurrent,
8077 ) else {
8078 let err = Error::SubagentConcurrencyExceeded {
8079 max_concurrent: self.config.subagents_max_concurrent,
8080 };
8081 return (format!("Error: {err}"), true);
8082 };
8083
8084 let child_id = next_subagent_id();
8085 let child_config = self.build_child_config(
8086 definition.as_ref(),
8087 inline_system_prompt,
8088 requested_model.or_else(|| definition.as_ref().and_then(|d| d.model.clone())),
8089 );
8090 let child_model = child_config.model.clone();
8091 let mut child = Agent::with_provider_arc(child_config, self.provider.clone());
8092 child.subagent_depth = self.subagent_depth + 1;
8093 child.subagent_concurrency_gauge = self.subagent_concurrency_gauge.clone();
8094
8095 // §2.2 C6: a background child NEVER gets a BLOCKING-BY-DEFAULT
8096 // interactive approval handler — either no handler at all
8097 // (`AutoPolicy`: the engine's pre-existing "no handler ⇒ deny"
8098 // fail-closed default), or (`Parent`) the never-blocking
8099 // `ParentQueueApprovalHandler`, UNLESS a `tui` embedder has
8100 // installed [`Self::child_approval_handler_factory`] (P5-4), in
8101 // which case THAT builds the handler instead — see
8102 // [`Self::set_child_approval_handler_factory`]'s doc comment for
8103 // why this can't escalate past what the rule engine already routed
8104 // to `Ask`. A foreground child also gets no handler here (today's
8105 // existing default posture; an embedder that wants an interactive
8106 // child installs its own via `set_permissions_approval_handler`
8107 // after this call returns, out of this method's scope).
8108 if background {
8109 if let Some(crate::subagents::BackgroundPromptsPolicy::Parent) =
8110 self.config.subagents_background_prompts
8111 {
8112 let handler: std::sync::Arc<dyn crate::permissions::PermissionsApprovalHandler> =
8113 match &self.child_approval_handler_factory {
8114 Some(factory) => {
8115 factory(child_id.clone(), self.pending_child_approvals.clone())
8116 }
8117 None => std::sync::Arc::new(crate::subagents::ParentQueueApprovalHandler {
8118 child_agent_id: child_id.clone(),
8119 queue: self.pending_child_approvals.clone(),
8120 }),
8121 };
8122 child.ctx.sandbox_approval_handler =
8123 Some(crate::sandbox::SandboxApprovalHandler(handler.clone()));
8124 child.permissions_approval_handler = Some(handler);
8125 }
8126 }
8127
8128 let lineage = crate::subagents::SubagentLineage {
8129 child_agent_id: child_id.clone(),
8130 parent_session_id: self.subagent_store.as_ref().map(|(_, name)| name.clone()),
8131 parent_tool_use_id: call.id.clone(),
8132 depth: self.subagent_depth + 1,
8133 agent_type: agent_type.clone(),
8134 task: task.clone(),
8135 background,
8136 spawned_at_ms: now_ms(),
8137 model: child_model,
8138 };
8139 if let Some((store, parent_name)) = &self.subagent_store {
8140 let _ = store.save_subagent_lineage(parent_name, &child_id, &lineage);
8141 }
8142
8143 if background {
8144 let spawned_task_text = task.clone();
8145 // BP-7: captured BEFORE `child` moves into the task — this is
8146 // the handle `send_message` writes into.
8147 let mailbox = child.steer_queue_handle();
8148 self.background_subagents.insert(
8149 child_id.clone(),
8150 BackgroundSubagent {
8151 handle: tokio::spawn(async move {
8152 // The concurrency slot lives for exactly as long as
8153 // this future runs — moved in here, dropped when the
8154 // child's `send` (and this future) finishes.
8155 let _guard = guard;
8156 let result = child.send(spawned_task_text).await;
8157 let transcript = child.history()[1..].to_vec();
8158 (child_id, result, transcript)
8159 }),
8160 task,
8161 agent_type,
8162 started_at_ms: lineage.spawned_at_ms,
8163 mailbox,
8164 },
8165 );
8166 let out = serde_json::json!({
8167 "subagent_id": lineage.child_agent_id,
8168 "status": "spawned",
8169 "background": true,
8170 });
8171 return (out.to_string(), false);
8172 }
8173
8174 // Foreground: run to completion now, guard held until this
8175 // function returns (then drops, freeing the slot).
8176 let result = child.send(task).await;
8177 let transcript = child.history()[1..].to_vec();
8178 self.persist_subagent_transcript(&child_id, &lineage, &transcript);
8179 // BP-7: kept in-process so `subagent_resume` can restore this
8180 // child's context even with no session store attached.
8181 self.reaped_subagents
8182 .insert(child_id.clone(), transcript.clone());
8183 drop(guard);
8184 match result {
8185 Ok(text) => (text, false),
8186 Err(e) => (format!("Error: subagent `{child_id}` failed: {e}"), true),
8187 }
8188 }
8189
8190 /// Execute the `subagent_status` intrinsic (P5-3, D3
8191 /// "background+resume"): poll a background child; once its `JoinHandle`
8192 /// is finished, reap it (removing it from `Self::background_subagents`
8193 /// and persisting its transcript, same as the foreground path).
8194 async fn run_subagent_status(
8195 &mut self,
8196 call: &supercode_interchange::ToolCall,
8197 ) -> (String, bool) {
8198 let args = match call.function.parsed_arguments() {
8199 Ok(v) => v,
8200 Err(e) => {
8201 let err = Error::InvalidArguments {
8202 tool: SUBAGENT_STATUS.to_string(),
8203 message: e.to_string(),
8204 };
8205 return (format!("Error: {err}"), true);
8206 }
8207 };
8208 let Some(id) = args.get("subagent_id").and_then(serde_json::Value::as_str) else {
8209 let err = Error::InvalidArguments {
8210 tool: SUBAGENT_STATUS.to_string(),
8211 message: "`subagent_id` is required".to_string(),
8212 };
8213 return (format!("Error: {err}"), true);
8214 };
8215 let Some(entry) = self.background_subagents.get(id) else {
8216 let err = Error::SubagentNotFound(id.to_string());
8217 return (format!("Error: {err}"), true);
8218 };
8219 if !entry.handle.is_finished() {
8220 let out = serde_json::json!({
8221 "subagent_id": id,
8222 "status": "pending",
8223 "task": entry.task,
8224 "agent_type": entry.agent_type,
8225 "started_at_ms": entry.started_at_ms,
8226 });
8227 return (out.to_string(), false);
8228 }
8229 // Finished — reap it. `.await` on an already-finished handle
8230 // resolves immediately (never actually blocks).
8231 let entry = self
8232 .background_subagents
8233 .remove(id)
8234 .expect("checked Some above");
8235 let (child_id, result, transcript) = match entry.handle.await {
8236 Ok(v) => v,
8237 Err(join_err) => {
8238 let err = Error::tool(
8239 SUBAGENT_STATUS,
8240 format!("subagent `{id}` task panicked: {join_err}"),
8241 );
8242 return (format!("Error: {err}"), true);
8243 }
8244 };
8245 // Re-derive the lineage record for persistence (cheap; the fields
8246 // are all still in hand) — mirrors the foreground path's single
8247 // `persist_subagent_transcript` call site.
8248 if let Some((store, parent_name)) = self.subagent_store.clone() {
8249 if let Ok(Some(lineage)) = store.load_subagent_lineage(&parent_name, &child_id) {
8250 self.persist_subagent_transcript(&child_id, &lineage, &transcript);
8251 }
8252 }
8253 // BP-7: see the foreground path's identical line.
8254 self.reaped_subagents
8255 .insert(child_id.clone(), transcript.clone());
8256 match result {
8257 Ok(text) => {
8258 let out = serde_json::json!({
8259 "subagent_id": child_id,
8260 "status": "done",
8261 "result": text,
8262 });
8263 (out.to_string(), false)
8264 }
8265 Err(e) => {
8266 let out = serde_json::json!({
8267 "subagent_id": child_id,
8268 "status": "error",
8269 "message": e.to_string(),
8270 });
8271 (out.to_string(), true)
8272 }
8273 }
8274 }
8275
8276 /// Execute the `background_exec` intrinsic (P5-6, §2 module 4, D1
8277 /// "background exec"): spawn `args.command` as a detached OS process
8278 /// via `crate::tools::build_sandboxed_sh` — the SAME sandboxed-spawn
8279 /// path [`crate::tools::BashTool::execute`] uses — and return its job
8280 /// id IMMEDIATELY, never the command's output. Gated by the same
8281 /// permission check a foreground `bash` call gets
8282 /// ([`Self::background_permission_denial`]), then a fail-closed
8283 /// concurrency cap ([`Config::tools_background_max_concurrent`]), THEN
8284 /// the actual spawn — in that order, so a refused call never holds a
8285 /// concurrency slot and never touches the process table.
8286 fn run_background_exec(&mut self, call: &supercode_interchange::ToolCall) -> (String, bool) {
8287 let args = match call.function.parsed_arguments() {
8288 Ok(v) => v,
8289 Err(e) => {
8290 let err = Error::InvalidArguments {
8291 tool: BACKGROUND_EXEC.to_string(),
8292 message: e.to_string(),
8293 };
8294 return (format!("Error: {err}"), true);
8295 }
8296 };
8297 let command = args
8298 .get("command")
8299 .and_then(serde_json::Value::as_str)
8300 .unwrap_or("")
8301 .to_string();
8302 if command.is_empty() {
8303 let err = Error::InvalidArguments {
8304 tool: BACKGROUND_EXEC.to_string(),
8305 message: "`command` is required and must be non-empty".to_string(),
8306 };
8307 return (format!("Error: {err}"), true);
8308 }
8309
8310 // A job id up front (before spawning) — used both as the audit
8311 // handle for a §2.2 C6 `Parent`-policy queued denial (this call may
8312 // never actually reach the spawn below) and, if the call proceeds,
8313 // as `Self::background_jobs`'s real key.
8314 let job_id = supercode_runtime::background::next_job_id(now_ms());
8315
8316 // Fable-5 review (LOW, "pre_tool_hook + doom-loop don't cover
8317 // background_exec"): this intrinsic is intercepted in
8318 // `Self::prepare_tool_call` and returns before `Self::finish_prepare`
8319 // ever runs, so — unlike a foreground `bash` call — it was reaching
8320 // this real spawn below WITHOUT ever offering `Config.pre_tool_hook`
8321 // a chance to veto it. `background_exec` runs a REAL command (unlike
8322 // the purely in-process meta-intrinsics `tool_search`/
8323 // `expand_reduction`/`sidecar_search`, which have no such gap to
8324 // close), so it belongs behind the same security-relevant veto a
8325 // foreground call gets. Scoped to this one call site — the other
8326 // meta-intrinsics are unchanged. The doom-loop counter
8327 // (`Self::check_doom_loop`) is deliberately NOT wired here: it is a
8328 // foreground repetition breaker keyed on `(self.doom_loop_last_call,
8329 // self.doom_loop_streak)`, a single piece of state shared with the
8330 // ordinary tool-call loop — folding background jobs into that same
8331 // streak would make an interleaved foreground/background pattern
8332 // trip (or fail to trip) the breaker in ways that have nothing to
8333 // do with the foreground loop actually repeating itself; the
8334 // pre_tool_hook veto below is the security-relevant half of this
8335 // fix, the doom-loop breaker is not.
8336 // BP-10: the hook now runs BEFORE this path's permissions gate, the
8337 // same order the foreground path uses — so a rewrite is what the
8338 // rules evaluate and what actually runs, and the hook's
8339 // `Allow`/`Ask` are tiers inside the engine rather than a second
8340 // verdict beside it.
8341 let mut command = command;
8342 let mut hook_decision = crate::config::HookDecision::Pass;
8343 if let Some(hook) = &self.config.pre_tool_hook {
8344 let outcome = hook(BACKGROUND_EXEC, &args);
8345 if outcome.decision == crate::config::HookDecision::Deny {
8346 let reason = outcome.reason.unwrap_or_else(|| "denied".to_string());
8347 return (format!("Error: blocked by pre-tool hook: {reason}"), true);
8348 }
8349 if let Some(rewritten) = outcome.updated_args {
8350 command = rewritten
8351 .get("command")
8352 .and_then(|v| v.as_str())
8353 .unwrap_or(&command)
8354 .to_string();
8355 }
8356 hook_decision = outcome.decision;
8357 }
8358
8359 if let Some(reason) = self.background_permission_denial(&command, &job_id, hook_decision) {
8360 return (format!("Error: {reason}"), true);
8361 }
8362
8363 let Some(guard) = crate::subagents::try_acquire(
8364 &self.background_concurrency_gauge,
8365 self.config.tools_background_max_concurrent,
8366 ) else {
8367 let err = Error::BackgroundJobConcurrencyExceeded {
8368 max_concurrent: self.config.tools_background_max_concurrent,
8369 };
8370 return (format!("Error: {err}"), true);
8371 };
8372
8373 let mut cmd = match crate::tools::build_sandboxed_sh(&command, &self.ctx) {
8374 Ok(cmd) => cmd,
8375 Err(e) => return (format!("Error: {e}"), true),
8376 };
8377 cmd.current_dir(&self.ctx.cwd)
8378 .stdin(std::process::Stdio::null())
8379 .stdout(std::process::Stdio::piped())
8380 .stderr(std::process::Stdio::piped())
8381 // Defense-in-depth for the "must be killed on drop" guarantee —
8382 // see `impl Drop for Agent`'s doc comment; the EXPLICIT
8383 // `start_kill()` loop there is what makes the guarantee
8384 // provable, this is a second, independent line of defense for
8385 // the same outcome.
8386 .kill_on_drop(true);
8387 // Fable-5 review (HIGH, "grandchildren orphaned on kill AND
8388 // agent-drop"): `Child::start_kill` only signals the DIRECT child.
8389 // A background command that spawns a surviving subprocess (a `&`
8390 // job, a pipeline, a double-forking daemon — or, on macOS, the
8391 // `sandbox-exec` wrapper itself in `build_sandboxed_sh`, whose real
8392 // `sh` and ITS children are all grandchildren of the tracked pid)
8393 // leaves those processes running, reparented to init, after the
8394 // tracked job is "killed". Putting this job in its OWN new process
8395 // group (`pgid == its own pid`, since every descendant inherits the
8396 // group unless it explicitly opts out) lets `kill_job_process_group`
8397 // below signal the WHOLE tree at kill/drop time, not just the one
8398 // pid we happen to be tracking. No portable equivalent on Windows —
8399 // see `kill_job_process_group`'s `#[cfg(not(unix))]` fallback.
8400 #[cfg(unix)]
8401 cmd.process_group(0);
8402 // P4c (`core.shell_env_snapshot`)/P5-10 (`env_policy`):
8403 // `build_sandboxed_sh` (above) already applied both via its own
8404 // `apply_sandbox_env_policy` last step — no separate `ctx.shell_env`
8405 // application here (that would re-add a secret `Filtered`/`None`
8406 // just stripped, on top of the already-`env_clear`'d command).
8407
8408 let mut child = match cmd.spawn() {
8409 Ok(c) => c,
8410 Err(e) => {
8411 drop(guard);
8412 let err = Error::tool(
8413 BACKGROUND_EXEC,
8414 format!("failed to spawn background command: {e}"),
8415 );
8416 return (format!("Error: {err}"), true);
8417 }
8418 };
8419 let pid = child.id();
8420 let output = std::sync::Arc::new(supercode_runtime::background::CapturedOutput::new());
8421 let cap = self.config.tools_background_max_output_bytes;
8422 // Fire-and-forget: the reader tasks outlive this method call and
8423 // exit on their own at pipe EOF — see `spawn_output_reader`'s doc
8424 // comment. Bound to named (not `_`) locals only to keep clippy's
8425 // `let_underscore_future` lint quiet; neither handle is awaited or
8426 // aborted anywhere.
8427 if let Some(stdout) = child.stdout.take() {
8428 let _stdout_reader = spawn_output_reader(stdout, output.clone(), cap);
8429 }
8430 if let Some(stderr) = child.stderr.take() {
8431 let _stderr_reader = spawn_output_reader(stderr, output.clone(), cap);
8432 }
8433
8434 let started_at_ms = now_ms();
8435 self.background_jobs.insert(
8436 job_id.clone(),
8437 BackgroundJob {
8438 child,
8439 command: command.clone(),
8440 pid,
8441 output,
8442 started_at_ms,
8443 killed: false,
8444 _guard: guard,
8445 },
8446 );
8447
8448 let out = serde_json::json!({
8449 "job_id": job_id,
8450 "status": "running",
8451 "pid": pid,
8452 "command": command,
8453 });
8454 (out.to_string(), false)
8455 }
8456
8457 /// Execute the `background_status` intrinsic (P5-6, D1 "monitor/event
8458 /// feed"): non-blocking poll of one job's run status (via
8459 /// `Child::try_wait`), drain its output captured since the LAST poll
8460 /// and emit it as an [`AgentEvent::BackgroundOutput`] event (the
8461 /// "event feed" — a real `EventSink` consumer sees each poll's new
8462 /// output live), and return the full captured output (bounded, per
8463 /// [`Config::tools_background_max_output_bytes`]) so far either way.
8464 /// Once the job is terminal (exited or killed), this reaps it — removes
8465 /// it from [`Self::background_jobs`], freeing its concurrency slot —
8466 /// same "poll once more to reap" contract [`Self::run_subagent_status`]
8467 /// already established for background subagents.
8468 fn run_background_status(&mut self, call: &supercode_interchange::ToolCall) -> (String, bool) {
8469 let args = match call.function.parsed_arguments() {
8470 Ok(v) => v,
8471 Err(e) => {
8472 let err = Error::InvalidArguments {
8473 tool: BACKGROUND_STATUS.to_string(),
8474 message: e.to_string(),
8475 };
8476 return (format!("Error: {err}"), true);
8477 }
8478 };
8479 let Some(job_id) = args.get("job_id").and_then(serde_json::Value::as_str) else {
8480 let err = Error::InvalidArguments {
8481 tool: BACKGROUND_STATUS.to_string(),
8482 message: "`job_id` is required".to_string(),
8483 };
8484 return (format!("Error: {err}"), true);
8485 };
8486 let job_id = job_id.to_string();
8487
8488 // Scoped so the mutable borrow of `self.background_jobs` ends
8489 // before `self.emit(...)`/`self.background_jobs.remove(...)` below
8490 // need their own (mutable) access to `self`.
8491 let (command, pid, started_at_ms, status, output_so_far, truncated, delta) = {
8492 let Some(job) = self.background_jobs.get_mut(&job_id) else {
8493 let err = Error::BackgroundJobNotFound(job_id);
8494 return (format!("Error: {err}"), true);
8495 };
8496 let status = background_job_status(job);
8497 let (output_so_far, truncated) = job.output.snapshot();
8498 let delta = job.output.drain_new();
8499 (
8500 job.command.clone(),
8501 job.pid,
8502 job.started_at_ms,
8503 status,
8504 output_so_far,
8505 truncated,
8506 delta,
8507 )
8508 };
8509
8510 if !delta.is_empty() {
8511 self.emit(AgentEvent::BackgroundOutput {
8512 job_id: job_id.clone(),
8513 chunk: delta,
8514 truncated,
8515 });
8516 }
8517
8518 let exit_code = match status {
8519 supercode_runtime::background::JobStatus::Exited(code) => code,
8520 _ => None,
8521 };
8522 let out = serde_json::json!({
8523 "job_id": job_id,
8524 "command": command,
8525 "status": status.as_str(),
8526 "exit_code": exit_code,
8527 "pid": pid,
8528 "started_at_ms": started_at_ms,
8529 "output": output_so_far,
8530 "output_truncated": truncated,
8531 });
8532 if !matches!(status, supercode_runtime::background::JobStatus::Running) {
8533 self.background_jobs.remove(&job_id);
8534 }
8535 (out.to_string(), false)
8536 }
8537
8538 /// Execute the `background_list` intrinsic (P5-6, D10 "bg-manager"):
8539 /// list every background job this agent is currently tracking, without
8540 /// draining output or reaping anything (a read-only listing —
8541 /// `background_status` is the reaping poll).
8542 fn run_background_list(&mut self, _call: &supercode_interchange::ToolCall) -> (String, bool) {
8543 let mut jobs = Vec::new();
8544 for (job_id, job) in self.background_jobs.iter_mut() {
8545 let status = background_job_status(job);
8546 jobs.push(serde_json::json!({
8547 "job_id": job_id,
8548 "command": job.command,
8549 "status": status.as_str(),
8550 "pid": job.pid,
8551 "started_at_ms": job.started_at_ms,
8552 }));
8553 }
8554 let out = serde_json::json!({ "jobs": jobs });
8555 (out.to_string(), false)
8556 }
8557
8558 /// Execute the `background_kill` intrinsic (P5-6, D10 "bg-manager",
8559 /// build brief "kill/cancel a job"): request REAL termination of a
8560 /// background job's OS process AND its whole process group (see
8561 /// [`kill_job_process_group`] — Fable-5 review, HIGH, "grandchildren
8562 /// orphaned on kill"; a documented no-op if the process already
8563 /// exited) and reap it immediately, freeing its concurrency slot.
8564 fn run_background_kill(&mut self, call: &supercode_interchange::ToolCall) -> (String, bool) {
8565 let args = match call.function.parsed_arguments() {
8566 Ok(v) => v,
8567 Err(e) => {
8568 let err = Error::InvalidArguments {
8569 tool: BACKGROUND_KILL.to_string(),
8570 message: e.to_string(),
8571 };
8572 return (format!("Error: {err}"), true);
8573 }
8574 };
8575 let Some(job_id) = args.get("job_id").and_then(serde_json::Value::as_str) else {
8576 let err = Error::InvalidArguments {
8577 tool: BACKGROUND_KILL.to_string(),
8578 message: "`job_id` is required".to_string(),
8579 };
8580 return (format!("Error: {err}"), true);
8581 };
8582 let job_id = job_id.to_string();
8583 let Some(mut job) = self.background_jobs.remove(&job_id) else {
8584 let err = Error::BackgroundJobNotFound(job_id);
8585 return (format!("Error: {err}"), true);
8586 };
8587 kill_job_process_group(&mut job);
8588 job.killed = true;
8589 let out = serde_json::json!({
8590 "job_id": job_id,
8591 "status": "killed",
8592 "pid": job.pid,
8593 });
8594 // `job` (and its `ConcurrencyGuard`) drops here, freeing the slot.
8595 (out.to_string(), false)
8596 }
8597
8598 /// P5-3 (D5 "subagent transcripts… persisted + linked"): write a
8599 /// finished child's transcript to `Self::subagent_store`, if one is
8600 /// installed — a no-op otherwise (see that field's doc comment). Builds
8601 /// the child's `Session` the same way `to_native_jsonl_v2`'s doc
8602 /// comment describes (an empty imported prefix + `transcript` as
8603 /// `appended` `NativeTurn`s), with `meta.agent_id`/`parent_tool_use_id`/
8604 /// `lineage` populated from `lineage` so the native-v2 header carries
8605 /// the full lineage record on disk (see `Session::to_native_jsonl_v2`'s
8606 /// P5-3 doc note).
8607 ///
8608 /// P5-3 safety-hardening fix (Fable-5 review, LOW "translation-fidelity
8609 /// cosmetic"): `Session::from_claude_code_str("")` is used ONLY to get
8610 /// a blank `raw`/`messages` skeleton cheaply (an empty string parses
8611 /// identically under any loader) — it is NOT claiming this child's
8612 /// session actually came from Claude Code. Before this fix, that
8613 /// borrowed constructor's `meta.source` (`SessionSource::ClaudeCode`)
8614 /// leaked straight through to the persisted sidecar's `source` header,
8615 /// mislabeling a native `spawn_subagent` child as an imported CC
8616 /// session. Corrected to `SessionSource::Native` immediately after —
8617 /// see that variant's doc comment.
8618 fn persist_subagent_transcript(
8619 &self,
8620 child_id: &str,
8621 lineage: &crate::subagents::SubagentLineage,
8622 transcript: &[ChatMessage],
8623 ) {
8624 let Some((store, parent_name)) = &self.subagent_store else {
8625 return;
8626 };
8627 let mut session = match Session::from_claude_code_str("") {
8628 Ok(s) => s,
8629 Err(_) => return,
8630 };
8631 session.meta.source = supercode_interchange::session::SessionSource::Native;
8632 session.meta.agent_id = Some(lineage.child_agent_id.clone());
8633 session.meta.parent_tool_use_id = Some(lineage.parent_tool_use_id.clone());
8634 session.meta.lineage = lineage.to_lineage_map();
8635 let sidecar_jsonl = session.to_native_jsonl_v2(transcript);
8636 let _ = store.save_subagent_transcript(parent_name, child_id, &sidecar_jsonl);
8637 let _ = store.save_subagent_lineage(parent_name, child_id, lineage);
8638 }
8639
8640 /// Execute the `tool_search` intrinsic (B6): case-insensitive keyword
8641 /// match over `name` + `description` of every registered, enabled,
8642 /// non-core, not-yet-activated tool (builtin and `mcp__*` alike). Matches
8643 /// are activated (advertised starting with the next request) and
8644 /// returned as a JSON array of their full [`ToolSchema`]s.
8645 fn run_tool_search(&mut self, call: &supercode_interchange::ToolCall) -> (String, bool) {
8646 let args = match call.function.parsed_arguments() {
8647 Ok(v) => v,
8648 Err(e) => {
8649 let err = Error::InvalidArguments {
8650 tool: TOOL_SEARCH.to_string(),
8651 message: e.to_string(),
8652 };
8653 return (format!("Error: {err}"), true);
8654 }
8655 };
8656 let query = args
8657 .get("query")
8658 .and_then(serde_json::Value::as_str)
8659 .unwrap_or("")
8660 .to_lowercase();
8661 let max_results = args
8662 .get("max_results")
8663 .and_then(serde_json::Value::as_u64)
8664 .map(|n| n as usize);
8665
8666 let mut matches: Vec<ToolSchema> = self
8667 .registry
8668 .iter()
8669 .filter(|t| self.config.tool_enabled(t.name()))
8670 .filter(|t| !self.is_core_tool(t.name()))
8671 .filter(|t| !self.activated_tools.contains(t.name()))
8672 .filter(|t| {
8673 query.is_empty()
8674 || t.name().to_lowercase().contains(&query)
8675 || self
8676 .config
8677 .tool_description(t.name(), t.description())
8678 .to_lowercase()
8679 .contains(&query)
8680 })
8681 // TR-8/T5 dev/03: the on-demand fetch always returns the ORIGINAL
8682 // full schema, never the tier-minified one — that's the invert.
8683 .map(|t| self.raw_schema_for(t))
8684 .collect();
8685
8686 if let Some(max) = max_results {
8687 matches.truncate(max);
8688 }
8689
8690 for m in &matches {
8691 self.activated_tools.insert(m.name.clone());
8692 }
8693
8694 let result = serde_json::to_string(&matches).unwrap_or_else(|_| "[]".to_string());
8695 (result, false)
8696 }
8697
8698 /// The `expand_reduction` schema (T12/TR-1), advertised whenever a
8699 /// [`ReductionPolicy`] is installed.
8700 ///
8701 /// The description deliberately never spells the literal stub sentinel
8702 /// prefix: A11's export leak guard is unconditional, so an assistant
8703 /// turn that quoted a stub line verbatim (which teaching the syntax
8704 /// invites) would permanently fail export for that session. Stubs are
8705 /// described abstractly and the model is told to pass ids only.
8706 fn expand_reduction_schema() -> ToolSchema {
8707 ToolSchema {
8708 name: EXPAND_REDUCTION.to_string(),
8709 description: "Fetch back the original content hidden behind a reduction stub in \
8710 your current view — a truncated tool output, cleared old turns, or an elided \
8711 file read that was hidden to save context. Each stub line names a reduction id \
8712 like r0042-9f3c: pass ONLY that id here, and never quote or repeat a stub line \
8713 itself in your replies. The original is durably kept in the session sidecar. \
8714 Pass `byte_range` to fetch a slice of a large one at a time instead of all of \
8715 it at once; ranged results are prefixed with a `bytes start..end of total` \
8716 header so you can plan the next slice."
8717 .to_string(),
8718 parameters: serde_json::json!({
8719 "type": "object",
8720 "properties": {
8721 "reduction_id": {
8722 "type": "string",
8723 "description": "The reduction id named in the stub line, e.g. \
8724 \"r0042-9f3c\". Pass the id alone."
8725 },
8726 "byte_range": {
8727 "type": "array",
8728 "items": {"type": "integer"},
8729 "minItems": 2,
8730 "maxItems": 2,
8731 "description": "Optional [start, end) byte offsets within the original \
8732 content to fetch instead of all of it. Exactly two non-negative \
8733 integers with start <= end."
8734 }
8735 },
8736 "required": ["reduction_id"],
8737 "additionalProperties": false
8738 }),
8739 }
8740 }
8741
8742 /// The `sidecar_search` schema (T12/TR-1), advertised whenever a
8743 /// [`ReductionPolicy`] is installed. Same no-literal-sentinel rule as
8744 /// [`Self::expand_reduction_schema`].
8745 fn sidecar_search_schema() -> ToolSchema {
8746 ToolSchema {
8747 name: SIDECAR_SEARCH.to_string(),
8748 description: "Search content currently hidden from your view by reduction stubs \
8749 (large tool outputs, cleared old turns, elided file reads) for a substring or \
8750 regex. Only hidden content is searched, never what you can already see. \
8751 Returns match snippets with each match's reduction_id for use with \
8752 expand_reduction; refer to results by their reduction id rather than quoting \
8753 stub lines. Results are capped — if `truncated` is true, narrow the query."
8754 .to_string(),
8755 parameters: serde_json::json!({
8756 "type": "object",
8757 "properties": {
8758 "query": {
8759 "type": "string",
8760 "description": "Non-empty substring or regex to search for \
8761 (case-insensitive)."
8762 }
8763 },
8764 "required": ["query"],
8765 "additionalProperties": false
8766 }),
8767 }
8768 }
8769
8770 /// Reload the recorder's full recorded messages from disk (TR-1's
8771 /// `recorded` resolution source). Since TR-12's D6/A7 supersession gate
8772 /// (`Self::run_loop`), a `expand_reduction`/`sidecar_search` call only
8773 /// ever exists alongside an active [`ReductionPolicy`] (see
8774 /// [`EXPAND_REDUCTION`]'s doc), and pairing one with a recorder — as the
8775 /// CLI's reduced mode always does — means the gate is already on and
8776 /// `history[1..]` holds the same full bytes as this reload: this upgrade
8777 /// is then a dormant no-op (`reduce::rehydrate::prefer_recorded` sees
8778 /// `recorded == minted` and keeps `minted`). It stops being a no-op —
8779 /// defense in depth, not the common path — for a **legacy** sidecar
8780 /// recorded before this gate existed, or for a policy-without-recorder
8781 /// agent (gate off, so `history[1..]` still carries
8782 /// [`Self::cap_tool_output`]-capped copies): only there can `history[1..]`
8783 /// diverge from the sidecar, and only there does consulting this reload
8784 /// actually recover bytes `history[1..]` alone couldn't. `Ok(None)` when
8785 /// no recorder is attached (rehydration then resolves from history alone,
8786 /// whose capped copies — if any — carry their own honest cap notice). A
8787 /// disk-level reload is fine here regardless: these intrinsic calls are
8788 /// rare, model-initiated events, not per-request work.
8789 fn recorded_messages(&self) -> std::result::Result<Option<Vec<ChatMessage>>, String> {
8790 let Some(recorder) = &self.recorder else {
8791 return Ok(None);
8792 };
8793 let raw = std::fs::read_to_string(recorder.path())
8794 .map_err(|e| format!("failed to read the session sidecar: {e}"))?;
8795 let session = Session::from_sidecar_str(&raw)
8796 .map_err(|e| format!("failed to parse the session sidecar: {e}"))?;
8797 Ok(Some(session.messages))
8798 }
8799
8800 /// Execute the `expand_reduction` intrinsic (T12/TR-1): resolves against
8801 /// `self.reduction_log` + `self.history[1..]` (the hash-minting source),
8802 /// upgraded to the recorder's full recorded bytes for cap-diverged
8803 /// content ([`Self::recorded_messages`]; the two-source contract is
8804 /// documented on `reduce::rehydrate`). `byte_range` is validated
8805 /// strictly — any malformed shape is a model-recoverable error naming
8806 /// the expected form and the original's true size, never a silent
8807 /// whole-content (or empty) return.
8808 fn run_expand_reduction(&mut self, call: &supercode_interchange::ToolCall) -> (String, bool) {
8809 let args = match call.function.parsed_arguments() {
8810 Ok(v) => v,
8811 Err(e) => {
8812 let err = Error::InvalidArguments {
8813 tool: EXPAND_REDUCTION.to_string(),
8814 message: e.to_string(),
8815 };
8816 return (format!("Error: {err}"), true);
8817 }
8818 };
8819 let Some(id) = args.get("reduction_id").and_then(serde_json::Value::as_str) else {
8820 return (
8821 "Error: expand_reduction requires a `reduction_id` string argument".to_string(),
8822 true,
8823 );
8824 };
8825 let recorded = match self.recorded_messages() {
8826 Ok(r) => r,
8827 Err(e) => return (format!("Error: expand_reduction: {e}"), true),
8828 };
8829 let recorded = recorded.as_deref();
8830
8831 // B3: strict shape validation — exactly two non-negative integers.
8832 // Anything else errors (with the true total when resolvable) rather
8833 // than silently degrading to a whole-content expand.
8834 let byte_range = match args.get("byte_range") {
8835 None | Some(serde_json::Value::Null) => None,
8836 Some(v) => {
8837 let parsed = v
8838 .as_array()
8839 .filter(|a| a.len() == 2)
8840 .and_then(|a| Some((a[0].as_u64()? as usize, a[1].as_u64()? as usize)));
8841 match parsed {
8842 Some(range) => Some(range),
8843 None => {
8844 let total = reduce::rehydrate::reduction_total_bytes(
8845 &self.reduction_log,
8846 &self.history[1..],
8847 recorded,
8848 id,
8849 )
8850 .map(|n| format!("; the original is {n} bytes"))
8851 .unwrap_or_default();
8852 return (
8853 format!(
8854 "Error: expand_reduction: malformed byte_range {v} — expected \
8855 [start, end): exactly two non-negative integers with \
8856 start <= end{total}"
8857 ),
8858 true,
8859 );
8860 }
8861 }
8862 }
8863 };
8864 match reduce::rehydrate::expand_reduction(
8865 &self.reduction_log,
8866 &self.history[1..],
8867 recorded,
8868 id,
8869 byte_range,
8870 ) {
8871 // A ranged result carries a provenance header naming the slice
8872 // and the true total, so the model can plan its next slice; a
8873 // whole-content expand stays byte-exact (TR-1 dev/01).
8874 Ok(outcome) => match outcome.range {
8875 Some((start, end)) => (
8876 format!(
8877 "[{id}: bytes {start}..{end} of {total}]\n{content}",
8878 total = outcome.total_bytes,
8879 content = outcome.content
8880 ),
8881 false,
8882 ),
8883 None => (outcome.content, false),
8884 },
8885 Err(e) => (format!("Error: {e}"), true),
8886 }
8887 }
8888
8889 /// Execute the `sidecar_search` intrinsic (T12/TR-1); same two-source
8890 /// resolution as [`Self::run_expand_reduction`]. The result is bounded
8891 /// by construction (`reduce::rehydrate::SidecarSearchResult`'s caps), so
8892 /// a broad query can never re-inflate the context or bloat the sidecar
8893 /// the recorder appends this result to.
8894 fn run_sidecar_search(&mut self, call: &supercode_interchange::ToolCall) -> (String, bool) {
8895 let args = match call.function.parsed_arguments() {
8896 Ok(v) => v,
8897 Err(e) => {
8898 let err = Error::InvalidArguments {
8899 tool: SIDECAR_SEARCH.to_string(),
8900 message: e.to_string(),
8901 };
8902 return (format!("Error: {err}"), true);
8903 }
8904 };
8905 let query = args
8906 .get("query")
8907 .and_then(serde_json::Value::as_str)
8908 .unwrap_or("");
8909 if query.trim().is_empty() {
8910 return (
8911 "Error: sidecar_search requires a non-empty `query` string argument".to_string(),
8912 true,
8913 );
8914 }
8915 let recorded = match self.recorded_messages() {
8916 Ok(r) => r,
8917 Err(e) => return (format!("Error: sidecar_search: {e}"), true),
8918 };
8919 match reduce::rehydrate::sidecar_search(
8920 &self.reduction_log,
8921 &self.history[1..],
8922 recorded.as_deref(),
8923 query,
8924 ) {
8925 Ok(result) => (
8926 serde_json::to_string(&result).unwrap_or_else(|_| "{}".to_string()),
8927 false,
8928 ),
8929 Err(e) => (format!("Error: {e}"), true),
8930 }
8931 }
8932
8933 fn emit(&self, event: AgentEvent) {
8934 if let Some(sink) = &self.config.event_sink {
8935 sink(event);
8936 }
8937 }
8938
8939 /// Number of non-system messages exchanged so far.
8940 pub fn turn_count(&self) -> usize {
8941 self.history
8942 .iter()
8943 .filter(|m| m.role != Role::System)
8944 .count()
8945 }
8946
8947 /// Cumulative output (completion) tokens reported by the provider across
8948 /// every `send` on this agent. Zero if the provider reports no usage.
8949 pub fn total_output_tokens(&self) -> u64 {
8950 self.total_output_tokens
8951 }
8952}
8953
8954/// Deliver an agent's `send_message` to another session through the one
8955/// send every sender uses. The sender is this process's own session (a
8956/// runtime supercode hosts, found from its ancestry), so replies can come
8957/// back.
8958#[cfg(feature = "adapter-api")]
8959async fn send_to_session(to: &str, message: &str, notify_when_idle: bool) -> (String, bool) {
8960 let homes = crate::HarnessHomes::default();
8961 let caller =
8962 match crate::mail_route::resolve_caller(&homes, &crate::mail_route::process_ancestry()) {
8963 Ok(caller) => caller,
8964 Err(_) => {
8965 return (
8966 format!(
8967 "Not sent: `{to}` is not one of your running subagents, and this session has \
8968 no address other sessions can reply to (supercode does not host it), so it \
8969 can message only its own subagents."
8970 ),
8971 true,
8972 )
8973 }
8974 };
8975 let options = crate::mail_send::SendOptions {
8976 notify_when_idle,
8977 ..Default::default()
8978 };
8979 match crate::mail_send::send(&homes, &caller, to, message, options).await {
8980 Ok(outcome) => (outcome.text, outcome.code != 0),
8981 Err(error) => (format!("Not sent to {to}: {error}"), true),
8982 }
8983}
8984
8985#[cfg(not(feature = "adapter-api"))]
8986async fn send_to_session(to: &str, _message: &str, _notify_when_idle: bool) -> (String, bool) {
8987 (
8988 format!("Not sent: `{to}` is not one of your running subagents, and this build has no cross-session messaging."),
8989 true,
8990 )
8991}
8992
8993#[cfg(test)]
8994mod bp2_spill_tests {
8995 //! BP-2 (`.volter/tracker/markdown/BP-2.md`, catalog:58): under the
8996 //! parity presets a capped tool output stays RECOVERABLE by the model —
8997 //! without `capabilities.reduction`, which both presets leave off.
8998
8999 use super::*;
9000 use crate::configfile::{resolve, ResolveOptions};
9001
9002 /// Never called — these tests drive `cap_tool_output` directly.
9003 #[derive(Debug)]
9004 struct NeverCalledProvider;
9005
9006 #[async_trait::async_trait]
9007 impl Provider for NeverCalledProvider {
9008 async fn complete(
9009 &self,
9010 _req: &ChatRequest,
9011 _on_delta: &(dyn for<'a> Fn(&'a str) + Send + Sync),
9012 ) -> crate::Result<(ChatMessage, crate::provider::Usage)> {
9013 unreachable!("BP-2 spill tests never issue a request")
9014 }
9015 }
9016
9017 fn resolved(preset: &str) -> crate::configfile::Resolved {
9018 let toml = crate::presets::lookup(preset).unwrap();
9019 resolve(toml, None, &ResolveOptions { strict: true })
9020 .unwrap_or_else(|e| panic!("{preset} resolves: {e}"))
9021 }
9022
9023 /// The residue this closes: the cap notice said the full output was
9024 /// "not retained" / "in session sidecar" and the only door to it —
9025 /// `expand_reduction` — is advertised solely when a `ReductionPolicy`
9026 /// is installed, which `capabilities.reduction = false` never does. So
9027 /// under both parity presets a truncated output was simply lost.
9028 ///
9029 /// Now the notice NAMES a spill file, and the door is the preset's own
9030 /// read pathway: `read_file` under cc-parity, the shell under
9031 /// cx-parity (which registers no file tools at all).
9032 #[tokio::test]
9033 async fn parity_presets_spill_capped_output_and_name_a_door_the_preset_has() {
9034 for (preset, expected_door) in [
9035 ("cc-parity", "read it with `read_file`"),
9036 ("cx-parity", "read it with `cat`"),
9037 ] {
9038 let r = resolved(preset);
9039 assert!(
9040 r.config.tool_output_spill,
9041 "{preset} must set `core.tool_output_spill`"
9042 );
9043 assert_eq!(
9044 r.modules.get("reduction"),
9045 Some(&false),
9046 "{preset} leaves `capabilities.reduction` off — the spill must not depend on it"
9047 );
9048 let mut config = resolved(preset).config;
9049 config.max_tool_output_bytes = Some(1024);
9050 let registry = crate::tools::ToolRegistry::from_config(&config);
9051 let agent = Agent::with_parts(config, Box::new(NeverCalledProvider), registry);
9052 assert!(
9053 agent.reduction_policy.is_none(),
9054 "no reduction policy is installed under {preset}"
9055 );
9056
9057 let full = "R".repeat(50_000);
9058 let capped = agent.cap_tool_output(full.clone());
9059 assert!(capped.len() < full.len(), "{preset}: output must be capped");
9060 assert!(capped.contains(expected_door), "{preset}: {capped:?}");
9061
9062 // The path in the notice must actually hold the full bytes.
9063 let marker = capped.split("spilled to ").nth(1).unwrap_or_default();
9064 let path = marker.split(" — ").next().unwrap_or_default();
9065 assert!(!path.is_empty(), "{preset}: no spill path in {capped:?}");
9066 assert_eq!(
9067 std::fs::read_to_string(path).unwrap(),
9068 full,
9069 "{preset}: the spill file must hold the FULL output"
9070 );
9071
9072 // And the model can actually walk through that door: the
9073 // preset's own read pathway returns the spilled content.
9074 let ctx = build_tool_context(agent.config()).0;
9075 let recovered = match registry_read_tool(&agent) {
9076 Some(("read_file", tool)) => tool
9077 .execute(serde_json::json!({"path": path}), &ctx)
9078 .await
9079 .unwrap(),
9080 Some(("bash", tool)) => tool
9081 .execute(serde_json::json!({"command": format!("cat {path}")}), &ctx)
9082 .await
9083 .unwrap(),
9084 _ => panic!("{preset}: no read door registered"),
9085 };
9086 assert!(
9087 recovered.contains(&"R".repeat(2000)),
9088 "{preset}: the door must return the spilled output"
9089 );
9090 let _ = std::fs::remove_file(path);
9091 }
9092 }
9093
9094 /// The preset's read pathway: `read_file` where it exists, else the
9095 /// shell — the same choice the cap notice's wording makes.
9096 fn registry_read_tool<'a>(
9097 agent: &'a Agent,
9098 ) -> Option<(&'static str, &'a dyn crate::tools::Tool)> {
9099 if let Some(tool) = agent.registry.get("read_file") {
9100 return Some(("read_file", tool));
9101 }
9102 agent.registry.get("bash").map(|tool| ("bash", tool))
9103 }
9104
9105 /// Off (the default, every non-parity preset and every SDK embedder):
9106 /// no spill file, and the notice is byte-identical to before BP-2.
9107 #[test]
9108 fn spill_off_leaves_the_notice_unchanged_and_writes_nothing() {
9109 let config = Config::builder().max_tool_output_bytes(1024).build();
9110 assert!(!config.tool_output_spill);
9111 let agent = Agent::with_provider(config, Box::new(NeverCalledProvider));
9112 let capped = agent.cap_tool_output("S".repeat(50_000));
9113 assert!(capped.contains("full output not retained"), "{capped:?}");
9114 assert!(!capped.contains("spilled to"), "{capped:?}");
9115 }
9116}
9117
9118#[cfg(test)]
9119mod bp3_new_core_tool_tests {
9120 //! BP-3 (`.volter/tracker/markdown/BP-3.md`): the two behaviours the
9121 //! new tools can only have INSIDE the agent — plan mode narrowing the
9122 //! permissions engine, and `new_context` re-founding the request view
9123 //! through `reduce`'s handoff projection — driven over the RESOLVED
9124 //! parity presets.
9125
9126 use super::*;
9127 use crate::configfile::{resolve, ResolveOptions};
9128
9129 #[derive(Debug)]
9130 struct NeverCalledProvider;
9131
9132 #[async_trait::async_trait]
9133 impl Provider for NeverCalledProvider {
9134 async fn complete(
9135 &self,
9136 _req: &ChatRequest,
9137 _on_delta: &(dyn for<'a> Fn(&'a str) + Send + Sync),
9138 ) -> crate::Result<(ChatMessage, crate::provider::Usage)> {
9139 unreachable!("BP-3 tests never issue a request")
9140 }
9141 }
9142
9143 fn resolved(preset: &str) -> Config {
9144 let toml = crate::presets::lookup(preset).unwrap();
9145 resolve(toml, None, &ResolveOptions { strict: true })
9146 .unwrap_or_else(|e| panic!("{preset} resolves: {e}"))
9147 .config
9148 }
9149
9150 fn agent_for(preset: &str) -> Agent {
9151 Agent::with_provider(resolved(preset), Box::new(NeverCalledProvider))
9152 }
9153
9154 /// An approval door that says yes to everything — so a denial in these
9155 /// tests can only come from the DENY tier, never from cc-parity's
9156 /// `approval = "untrusted"` ask default.
9157 struct AlwaysAllow;
9158 impl crate::permissions::PermissionsApprovalHandler for AlwaysAllow {
9159 fn ask(
9160 &self,
9161 _req: &crate::permissions::ApprovalRequest,
9162 ) -> crate::permissions::ApprovalOutcome {
9163 crate::permissions::ApprovalOutcome::Allow
9164 }
9165 }
9166
9167 fn call(name: &str, args: serde_json::Value) -> supercode_interchange::ToolCall {
9168 supercode_interchange::ToolCall {
9169 id: format!("call-{name}"),
9170 kind: "function".to_string(),
9171 function: supercode_interchange::FunctionCall {
9172 name: name.to_string(),
9173 arguments: args.to_string(),
9174 },
9175 }
9176 }
9177
9178 /// The row `plan-mode-read-only-research-phase` claims: under the
9179 /// resolved `cc-parity` preset, entering plan mode makes the
9180 /// permissions engine REFUSE write and execution tools — and the
9181 /// refusal survives an approval door that allows everything, because
9182 /// the mode contributes DENY rules, the tier no approval can override.
9183 #[test]
9184 fn cc_parity_plan_mode_denies_writes_through_the_permissions_engine() {
9185 let mut agent = agent_for("cc-parity");
9186 assert!(
9187 agent.config.permissions_enabled,
9188 "cc-parity runs the permissions engine; plan mode narrows it"
9189 );
9190 agent.set_permissions_approval_handler(AlwaysAllow);
9191
9192 let write = serde_json::json!({"path": "notes.txt", "content": "x"});
9193 let bash = serde_json::json!({"command": "echo hi"});
9194 assert!(
9195 agent
9196 .permissions_gate_denial("write_file", &write, crate::config::HookDecision::Pass)
9197 .is_none(),
9198 "outside plan mode an allowed write must pass"
9199 );
9200 assert!(agent
9201 .permissions_gate_denial("bash", &bash, crate::config::HookDecision::Pass)
9202 .is_none());
9203
9204 agent.plan_mode().enter(Some("research first"));
9205
9206 let denial = agent
9207 .permissions_gate_denial("write_file", &write, crate::config::HookDecision::Pass)
9208 .expect("plan mode must refuse a write");
9209 assert!(denial.contains("Deny"), "{denial}");
9210 assert!(agent
9211 .permissions_gate_denial("bash", &bash, crate::config::HookDecision::Pass)
9212 .is_some());
9213 assert!(agent
9214 .permissions_gate_denial(
9215 "apply_patch",
9216 &serde_json::json!({"patch": "*** Begin Patch\n*** End Patch"}),
9217 crate::config::HookDecision::Pass
9218 )
9219 .is_some());
9220
9221 // The research surface, and the way out, stay open.
9222 for (tool, args) in [
9223 ("read_file", serde_json::json!({"path": "notes.txt"})),
9224 ("glob", serde_json::json!({"pattern": "*.rs"})),
9225 ("exit_plan_mode", serde_json::json!({"plan": "the plan"})),
9226 ("ask_user", serde_json::json!({"questions": []})),
9227 ] {
9228 assert!(
9229 agent
9230 .permissions_gate_denial(tool, &args, crate::config::HookDecision::Pass)
9231 .is_none(),
9232 "plan mode must leave `{tool}` reachable"
9233 );
9234 }
9235
9236 agent.plan_mode().exit();
9237 assert!(
9238 agent
9239 .permissions_gate_denial("write_file", &write, crate::config::HookDecision::Pass)
9240 .is_none(),
9241 "leaving plan mode restores the write surface"
9242 );
9243 }
9244
9245 /// The `context-budget-tools` row's read half: the figure
9246 /// `get_context_remaining` reports is the agent's OWN accounting,
9247 /// computed at the moment the tool asks for it.
9248 #[test]
9249 fn cx_parity_publishes_its_context_accounting_when_the_budget_tool_runs() {
9250 let mut agent = agent_for("cx-parity");
9251 assert!(agent.ctx.context_budget.snapshot().is_none(), "nothing yet");
9252 agent
9253 .history
9254 .push(ChatMessage::user("x".repeat(4000).to_string()));
9255 let _ = agent.prepare_tool_call(&call("current_time", serde_json::json!({})));
9256 assert!(
9257 agent.ctx.context_budget.snapshot().is_none(),
9258 "an unrelated tool call must not pay for the accounting"
9259 );
9260
9261 let _ = agent.prepare_tool_call(&call("get_context_remaining", serde_json::json!({})));
9262 let published = agent
9263 .ctx
9264 .context_budget
9265 .snapshot()
9266 .expect("the budget tool's own call publishes it");
9267 // What the model reads IS `Agent::context_usage()` — the same
9268 // struct `/context` prints and the guard enforces, not a second
9269 // estimate that could disagree with it.
9270 assert_eq!(
9271 published,
9272 serde_json::to_value(agent.context_usage()).unwrap()
9273 );
9274 assert!(published["context_limit"].as_u64().unwrap() > 0);
9275 assert!(published["remaining_tokens"].as_u64().unwrap() > 0);
9276 }
9277
9278 /// The `context-budget-tools` row's write half, over the resolved
9279 /// `cx-parity` preset: a parked `new_context` request re-founds the
9280 /// request view through `reduce`'s handoff projection — objective in
9281 /// the leading system message, the tail kept, the rest covered by
9282 /// `TurnsCleared` spans that land in this agent's own reduction log.
9283 #[test]
9284 fn cx_parity_new_context_rebuilds_the_window_through_the_same_handoff_the_operator_gets() {
9285 let mut agent = agent_for("cx-parity");
9286 agent.history.push(ChatMessage::system("system"));
9287 for i in 0..12 {
9288 agent.history.push(ChatMessage::user(format!("turn {i}")));
9289 }
9290 let before = agent.history.clone();
9291
9292 agent
9293 .ctx
9294 .context_budget
9295 .request_new_context(crate::tools::NewContextRequest {
9296 objective: "finish the parser".to_string(),
9297 keep_recent: Some(2),
9298 });
9299 agent.apply_pending_new_context();
9300
9301 // Exactly what `/handoff` produces — the model's door and the
9302 // operator's door run one mechanism, so this compares against it.
9303 let mut expected =
9304 Agent::with_provider(resolved("cx-parity"), Box::new(NeverCalledProvider));
9305 expected.history = before.clone();
9306 expected.new_context("finish the parser", Some(2));
9307 assert_eq!(agent.history, expected.history);
9308
9309 assert!(
9310 agent.history.len() < before.len(),
9311 "the window must actually shrink: {} -> {}",
9312 before.len(),
9313 agent.history.len()
9314 );
9315 assert_eq!(agent.history[0], before[0], "the system prompt survives");
9316 let marker = agent.history[1].content.clone().unwrap_or_default();
9317 assert!(marker.contains("fresh working context"), "{marker}");
9318 assert!(marker.contains("finish the parser"), "{marker}");
9319 assert_eq!(
9320 agent.history[agent.history.len() - 2..],
9321 before[before.len() - 2..],
9322 "the requested tail is kept verbatim"
9323 );
9324 assert!(
9325 agent.ctx.context_budget.take_new_context().is_none(),
9326 "the request is consumed exactly once"
9327 );
9328 }
9329
9330 /// `new_context` never trades recoverability for a smaller window: with
9331 /// no sidecar recorder the request is refused, the reason is handed
9332 /// back to the model, and the transcript keeps every turn.
9333 #[test]
9334 fn cx_parity_new_context_states_the_retention_it_actually_has() {
9335 let mut agent = agent_for("cx-parity");
9336 agent.history.push(ChatMessage::system("system"));
9337 for i in 0..12 {
9338 agent.history.push(ChatMessage::user(format!("turn {i}")));
9339 }
9340 agent
9341 .ctx
9342 .context_budget
9343 .request_new_context(crate::tools::NewContextRequest {
9344 objective: "finish the parser".to_string(),
9345 keep_recent: Some(2),
9346 });
9347 agent.apply_pending_new_context();
9348
9349 // No recorder is installed here, and the marker says so rather than
9350 // implying the set-aside turns are still somewhere.
9351 let marker = agent.history[1].content.clone().unwrap_or_default();
9352 assert!(
9353 marker.contains("No transcript sidecar is attached"),
9354 "the marker must not overstate retention: {marker}"
9355 );
9356 }
9357}
9358
9359/// BP-8 (catalog:152): what [`Agent::rewind_conversation`] did.
9360#[derive(Debug, Clone, PartialEq, Eq)]
9361pub struct RewindOutcome {
9362 /// Messages remaining, including the system message at index 0.
9363 pub kept: usize,
9364 /// Messages removed from the live conversation (still on disk, in the
9365 /// journal, and still in the tree under `preserved_branch`).
9366 pub removed: usize,
9367 /// When the tree module is on and the rewind actually moved the leaf:
9368 /// the sibling branch the old leaf was preserved under, so the rewound
9369 /// path stays independently addressable.
9370 pub preserved_branch: Option<String>,
9371}
9372
9373#[cfg(test)]
9374mod bp1_compaction_tests {
9375 //! BP-1 (`.volter/tracker/markdown/BP-1.md` AC2): `cx-parity` fires
9376 //! [`Agent::maybe_compact`] at its own trigger, and
9377 //! `core.compaction.summarize` reaches [`Config`] and is what the
9378 //! compaction marker says.
9379
9380 use super::*;
9381 use crate::configfile::{resolve, ResolveOptions};
9382
9383 /// Never called — these tests drive `maybe_compact` directly, which
9384 /// makes no request.
9385 #[derive(Debug)]
9386 struct NeverCalledProvider;
9387
9388 #[async_trait::async_trait]
9389 impl Provider for NeverCalledProvider {
9390 async fn complete(
9391 &self,
9392 _req: &ChatRequest,
9393 _on_delta: &(dyn for<'a> Fn(&'a str) + Send + Sync),
9394 ) -> crate::Result<(ChatMessage, crate::provider::Usage)> {
9395 unreachable!("BP-1 compaction tests never issue a request")
9396 }
9397 }
9398
9399 fn resolved_cx_parity() -> Config {
9400 let toml = crate::presets::lookup("cx-parity").unwrap();
9401 resolve(toml, None, &ResolveOptions { strict: true })
9402 .expect("cx-parity resolves")
9403 .config
9404 }
9405
9406 /// Enough history to sit inside `reserve_tokens` of ANY model context
9407 /// window (`estimate_view_tokens` is size-proportional, and the largest
9408 /// window in the catalog is far below this).
9409 fn stuff_history(agent: &mut Agent) {
9410 agent.history.push(ChatMessage::system("system"));
9411 for i in 0..400 {
9412 agent
9413 .history
9414 .push(ChatMessage::user(format!("turn {i}: {}", "x".repeat(8000))));
9415 }
9416 }
9417
9418 /// The defect: `cx-parity` armed NEITHER compaction trigger, so
9419 /// `maybe_compact` returned `false` on its
9420 /// `threshold.is_none() && compaction_reserve_tokens.is_none()` guard
9421 /// no matter how large the conversation grew.
9422 #[test]
9423 fn cx_parity_fires_maybe_compact_at_its_pressure_trigger() {
9424 let config = resolved_cx_parity();
9425 assert!(config.compaction_enabled);
9426 assert_eq!(config.compaction_reserve_tokens, Some(16384));
9427 assert!(config.compaction_summarize);
9428
9429 let mut agent = Agent::with_provider(config, Box::new(NeverCalledProvider));
9430 stuff_history(&mut agent);
9431 let before = agent.history.len();
9432 assert!(
9433 agent.maybe_compact(),
9434 "cx-parity must compact under context pressure"
9435 );
9436 assert!(agent.history.len() < before, "history must actually shrink");
9437 let marker = agent
9438 .history
9439 .iter()
9440 .find(|m| {
9441 m.content
9442 .as_deref()
9443 .is_some_and(|c| c.contains("earlier conversation compacted"))
9444 })
9445 .expect("a compaction marker must be present");
9446 assert!(
9447 marker
9448 .content
9449 .as_deref()
9450 .unwrap()
9451 .contains("summarized to save context"),
9452 "cx-parity sets `core.compaction.summarize = true`"
9453 );
9454 }
9455
9456 /// `core.compaction.summarize = false` reaches `Config` too, and the
9457 /// marker then states only what actually happened to the span.
9458 #[test]
9459 fn compaction_summarize_false_reaches_config_and_changes_the_marker() {
9460 let toml = "extends = \"cx-parity\"\n[core.compaction]\nsummarize = false\n";
9461 let config = resolve(toml, None, &ResolveOptions::default())
9462 .expect("resolves")
9463 .config;
9464 assert!(!config.compaction_summarize);
9465
9466 let mut agent = Agent::with_provider(config, Box::new(NeverCalledProvider));
9467 stuff_history(&mut agent);
9468 assert!(agent.maybe_compact());
9469 let marker = agent
9470 .history
9471 .iter()
9472 .find(|m| {
9473 m.content
9474 .as_deref()
9475 .is_some_and(|c| c.contains("earlier conversation compacted"))
9476 })
9477 .expect("a compaction marker must be present");
9478 let text = marker.content.as_deref().unwrap();
9479 assert!(text.contains("cleared to save context"), "got: {text}");
9480 assert!(!text.contains("summarized"));
9481 }
9482}
9483
9484#[cfg(test)]
9485mod api_key_cmd_tests {
9486 //! P4 (design §5.2, §1.8 D6 row): `api_key_cmd` credential-helper
9487 //! resolution. `Agent::new` never makes a network call, so these tests
9488 //! exercise the real resolution chain end-to-end without mocking.
9489
9490 use super::*;
9491
9492 /// Default-off: with no `api_key`/`api_key_cmd` set and an env var that
9493 /// isn't set either, resolution fails exactly as it always has —
9494 /// `api_key_cmd` being a brand-new field changes nothing when unset.
9495 #[test]
9496 fn default_none_falls_through_to_missing_api_key_error() {
9497 let config = Config::builder()
9498 .api_key_env("SUPERCODE_TEST_UNSET_VAR_API_KEY_CMD")
9499 .build();
9500 assert!(config.api_key.is_none());
9501 assert!(config.api_key_cmd.is_none());
9502 let err = Agent::new(config).err().expect("no key source configured");
9503 assert!(matches!(err, Error::MissingApiKey(_)));
9504 }
9505
9506 /// Happy path: `api_key_cmd` alone (no `api_key`, no matching env var)
9507 /// is enough for `Agent::new` to succeed — the helper's stdout is
9508 /// resolved and used.
9509 #[test]
9510 fn api_key_cmd_alone_resolves_successfully() {
9511 let config = Config::builder()
9512 .api_key_cmd("echo sk-test-from-helper")
9513 .api_key_env("SUPERCODE_TEST_UNSET_VAR_API_KEY_CMD_2")
9514 .build();
9515 assert!(Agent::new(config).is_ok());
9516 }
9517
9518 /// A failing helper command (non-zero exit, or empty stdout) falls
9519 /// through to `api_key_env` rather than propagating the helper's own
9520 /// failure — same "try the next source" posture as every other layer.
9521 #[test]
9522 fn api_key_cmd_failure_falls_through_to_env() {
9523 std::env::set_var(
9524 "SUPERCODE_TEST_API_KEY_CMD_FALLBACK",
9525 "sk-from-env-fallback",
9526 );
9527 let config = Config::builder()
9528 .api_key_cmd("exit 1")
9529 .api_key_env("SUPERCODE_TEST_API_KEY_CMD_FALLBACK")
9530 .build();
9531 assert!(Agent::new(config).is_ok());
9532 std::env::remove_var("SUPERCODE_TEST_API_KEY_CMD_FALLBACK");
9533 }
9534
9535 /// A failing helper AND no fallback env var still produces the same
9536 /// `MissingApiKey` error today's no-key path always produced — the new
9537 /// source never turns a hard failure into a silent empty key.
9538 #[test]
9539 fn api_key_cmd_failure_with_no_fallback_still_errors() {
9540 let config = Config::builder()
9541 .api_key_cmd("exit 1")
9542 .api_key_env("SUPERCODE_TEST_UNSET_VAR_API_KEY_CMD_3")
9543 .build();
9544 let err = Agent::new(config)
9545 .err()
9546 .expect("helper failed, no env fallback");
9547 assert!(matches!(err, Error::MissingApiKey(_)));
9548 }
9549
9550 /// `run_api_key_cmd` directly: happy path trims trailing whitespace/
9551 /// newline from the command's stdout.
9552 #[test]
9553 fn run_api_key_cmd_trims_output() {
9554 assert_eq!(run_api_key_cmd("echo ' sk-abc123 '"), "sk-abc123");
9555 }
9556
9557 /// `run_api_key_cmd` directly: a nonexistent binary fails to spawn and
9558 /// returns an empty string rather than panicking.
9559 #[test]
9560 fn run_api_key_cmd_spawn_failure_returns_empty() {
9561 // `sh -c` itself always spawns; feed it a command that can't run.
9562 assert_eq!(
9563 run_api_key_cmd("/no/such/binary/at/all --flag"),
9564 String::new()
9565 );
9566 }
9567}
9568
9569#[cfg(test)]
9570mod bp4_prompt_context_tests {
9571 //! BP-4 (`.volter/tracker/markdown/BP-4.md`): the prompt/context knobs
9572 //! the parity presets never set, proved over the RESOLVED `cc-parity` /
9573 //! `cx-parity` configs (not over hand-built `Config`s — a preset that
9574 //! doesn't arm the knob would pass that weaker test).
9575
9576 use super::*;
9577 use crate::configfile::{resolve, ResolveOptions};
9578
9579 fn resolved(preset: &str) -> Config {
9580 let toml = crate::presets::lookup(preset).unwrap();
9581 resolve(toml, None, &ResolveOptions { strict: true })
9582 .unwrap_or_else(|e| panic!("{preset} resolves: {e}"))
9583 .config
9584 }
9585
9586 /// Never called — these tests assemble prompts and drive
9587 /// `refresh_env_context`/`inject_context_block`, none of which issue a
9588 /// request.
9589 #[derive(Debug)]
9590 struct NoProvider;
9591
9592 #[async_trait::async_trait]
9593 impl Provider for NoProvider {
9594 async fn complete(
9595 &self,
9596 _req: &ChatRequest,
9597 _on_delta: &(dyn for<'a> Fn(&'a str) + Send + Sync),
9598 ) -> crate::Result<(ChatMessage, crate::provider::Usage)> {
9599 unreachable!("BP-4 prompt/context tests never issue a request")
9600 }
9601 }
9602
9603 fn scratch(tag: &str) -> std::path::PathBuf {
9604 let dir = std::env::temp_dir().join(format!(
9605 "supercode-bp4-{tag}-{}-{}",
9606 std::process::id(),
9607 std::time::SystemTime::now()
9608 .duration_since(std::time::UNIX_EPOCH)
9609 .unwrap()
9610 .as_nanos()
9611 ));
9612 std::fs::create_dir_all(&dir).unwrap();
9613 dir
9614 }
9615
9616 /// Row `project-instruction-files-w-directory-walk`: a CLAUDE.md above
9617 /// the working directory is discovered, ordered root→cwd (nearest wins
9618 /// by appearing last), and the climb STOPS at the `.git` root — the
9619 /// directory above it is never read.
9620 #[test]
9621 fn presets_walk_ancestors_up_to_the_git_root_nearest_last() {
9622 for preset in ["cc-parity", "cx-parity"] {
9623 let base = scratch("walk");
9624 let above = base.join("above");
9625 let root = above.join("repo");
9626 let deep = root.join("crates").join("thing");
9627 std::fs::create_dir_all(&deep).unwrap();
9628 std::fs::create_dir_all(root.join(".git")).unwrap();
9629 std::fs::write(above.join("CLAUDE.md"), "ABOVE-THE-ROOT-MARKER").unwrap();
9630 std::fs::write(above.join("AGENTS.md"), "ABOVE-THE-ROOT-MARKER").unwrap();
9631 std::fs::write(root.join("CLAUDE.md"), "REPO-ROOT-MARKER").unwrap();
9632 std::fs::write(root.join("AGENTS.md"), "REPO-ROOT-MARKER").unwrap();
9633 std::fs::write(deep.join("CLAUDE.md"), "NEAREST-DIR-MARKER").unwrap();
9634 std::fs::write(deep.join("AGENTS.md"), "NEAREST-DIR-MARKER").unwrap();
9635
9636 let mut config = resolved(preset);
9637 config.cwd = deep.clone();
9638 let blob = assemble_project_instructions(&config);
9639
9640 let root_at = blob
9641 .find("REPO-ROOT-MARKER")
9642 .unwrap_or_else(|| panic!("{preset}: the ancestor repo root was not walked"));
9643 let near_at = blob
9644 .find("NEAREST-DIR-MARKER")
9645 .unwrap_or_else(|| panic!("{preset}: cwd's own file was not loaded"));
9646 assert!(
9647 root_at < near_at,
9648 "{preset}: nearest-to-cwd must win by appearing LAST (root→cwd)"
9649 );
9650 assert!(
9651 !blob.contains("ABOVE-THE-ROOT-MARKER"),
9652 "{preset}: the walk must stop at the `.git` root"
9653 );
9654 let _ = std::fs::remove_dir_all(&base);
9655 }
9656 }
9657
9658 /// The walk is bounded even with no root marker anywhere: it terminates
9659 /// at the filesystem root instead of looping.
9660 #[test]
9661 fn walk_terminates_without_a_root_marker() {
9662 let base = scratch("nomarker");
9663 let deep = base.join("a").join("b").join("c");
9664 std::fs::create_dir_all(&deep).unwrap();
9665 let mut config = resolved("cx-parity");
9666 config.cwd = deep.clone();
9667 let roots = instruction_walk_roots(&config);
9668 assert!(roots.len() <= MAX_INSTRUCTION_WALK_DEPTH);
9669 assert_eq!(roots.last().unwrap(), &deep, "cwd is the LAST root");
9670 let _ = std::fs::remove_dir_all(&base);
9671 }
9672
9673 /// Row `instruction-file-hygiene-controls`, cx half: `cx-parity` sets
9674 /// the documented 32 KiB `project_doc_max_bytes`, and it binds per file
9675 /// AND over the aggregate, each with its own notice.
9676 #[test]
9677 fn cx_parity_enforces_the_documented_instruction_byte_cap() {
9678 let config = resolved("cx-parity");
9679 assert_eq!(
9680 config.project_doc_max_bytes,
9681 Some(32_768),
9682 "cx-parity must arm cx§2's documented 32 KiB cap"
9683 );
9684
9685 let base = scratch("cap");
9686 std::fs::create_dir_all(base.join(".git")).unwrap();
9687 std::fs::write(base.join("CLAUDE.md"), "x".repeat(40_000)).unwrap();
9688 std::fs::write(base.join("AGENTS.md"), "y".repeat(40_000)).unwrap();
9689 let mut config = config;
9690 config.cwd = base.clone();
9691 let blob = assemble_project_instructions(&config);
9692 assert!(
9693 blob.contains("[supercode: file truncated at core.project_doc_max_bytes]"),
9694 "the per-file cap must fire with a notice"
9695 );
9696 assert!(
9697 blob.contains(
9698 "[supercode: instruction content truncated at core.project_doc_max_bytes]"
9699 ),
9700 "the aggregate cap must fire with a notice"
9701 );
9702 assert!(blob.len() < 33_200, "aggregate blob stayed over the cap");
9703 let _ = std::fs::remove_dir_all(&base);
9704 }
9705
9706 /// Row `instruction-file-hygiene-controls`, cc half: `cc-parity` arms
9707 /// CC's own two levers — HTML-comment stripping and `claudeMdExcludes`
9708 /// — and arms NO byte cap, because CC documents none.
9709 #[test]
9710 fn cc_parity_strips_html_comments_and_honours_excludes() {
9711 let mut config = resolved("cc-parity");
9712 assert!(config.project_doc_strip_comments, "cc strips `<!-- … -->`");
9713 assert_eq!(
9714 config.project_doc_max_bytes, None,
9715 "`project_doc_max_bytes = 0` is §3.1's spelling for uncapped"
9716 );
9717
9718 let base = scratch("hygiene");
9719 std::fs::create_dir_all(base.join(".git")).unwrap();
9720 std::fs::write(
9721 base.join("CLAUDE.md"),
9722 "KEEP-THIS<!-- MAINTAINER-NOTE -->AND-THIS",
9723 )
9724 .unwrap();
9725 std::fs::write(base.join("AGENTS.md"), "EXCLUDED-FILE-MARKER").unwrap();
9726 config.cwd = base.clone();
9727 config.project_doc_excludes = vec!["AGENTS.md".to_string()];
9728 let blob = assemble_project_instructions(&config);
9729 assert!(blob.contains("KEEP-THIS") && blob.contains("AND-THIS"));
9730 assert!(
9731 !blob.contains("MAINTAINER-NOTE"),
9732 "block HTML comments must be stripped before injection"
9733 );
9734 assert!(
9735 !blob.contains("EXCLUDED-FILE-MARKER"),
9736 "an excluded instruction file must never be read into the prompt"
9737 );
9738 let _ = std::fs::remove_dir_all(&base);
9739 }
9740
9741 /// A project layer must not be able to suppress the user's own global
9742 /// instruction files by adding an exclude pattern (§3.3 trust boundary).
9743 #[test]
9744 fn project_layer_cannot_set_instruction_excludes() {
9745 let hc = crate::configfile::HarnessConfig::from_toml_str(
9746 "schema_version = 1\n[core]\nproject_doc_excludes = [\"CLAUDE.md\"]\n",
9747 )
9748 .unwrap();
9749 let (sanitized, dropped) = crate::configfile::sanitize_for_project(&hc);
9750 assert!(sanitized.core.project_doc_excludes.is_none());
9751 assert!(dropped.iter().any(|d| d == "core.project_doc_excludes"));
9752 }
9753
9754 /// Row `environment-context-block`: both presets emit the policy line
9755 /// the row's semantics name, and the block is RE-EMITTED when the thing
9756 /// it describes moves (cx§2 "re-emitted on change").
9757 #[test]
9758 fn env_context_block_carries_policy_and_re_emits_on_change() {
9759 for preset in ["cc-parity", "cx-parity"] {
9760 let base = scratch("env");
9761 // A SIBLING, not a child: `contains` assertions below must not
9762 // be satisfiable by a path prefix.
9763 let here = base.join("here");
9764 let other = base.join("elsewhere");
9765 std::fs::create_dir_all(&here).unwrap();
9766 std::fs::create_dir_all(&other).unwrap();
9767
9768 let mut config = resolved(preset);
9769 config.cwd = here.clone();
9770 assert!(config.env_context, "{preset} must set core.env_context");
9771 let expected_policy = format!(
9772 "approval policy: {} · sandbox: {}",
9773 approval_policy_label(config.approval),
9774 sandbox_policy_label(config.sandbox),
9775 );
9776
9777 let mut agent = Agent::with_provider(config, Box::new(NoProvider));
9778 let system = agent.history[0].content.clone().unwrap_or_default();
9779 assert!(
9780 system.contains(&expected_policy),
9781 "{preset}: the environment block must state the approval/sandbox policy — {system}"
9782 );
9783 assert!(system.contains(&format!("cwd: {}", here.display())));
9784
9785 // Nothing moved ⇒ no churn (the prompt cache is not busted for
9786 // free).
9787 assert!(!agent.refresh_env_context(), "{preset}: spurious re-emit");
9788
9789 // cwd + policy move mid-session.
9790 agent.config.cwd = other.clone();
9791 agent.config.approval = crate::config::ApprovalPolicy::Untrusted;
9792 assert!(
9793 agent.refresh_env_context(),
9794 "{preset}: change not re-emitted"
9795 );
9796 let system = agent.history[0].content.clone().unwrap_or_default();
9797 assert!(
9798 system.contains(&format!("cwd: {}", other.display())),
9799 "{preset}: the fresh cwd must reach the model"
9800 );
9801 assert!(system.contains("approval policy: untrusted"));
9802 assert!(
9803 !system.contains(&format!("cwd: {}", here.display())),
9804 "{preset}: the stale block must be REPLACED, not duplicated"
9805 );
9806 assert_eq!(
9807 system.matches("# Environment").count(),
9808 1,
9809 "{preset}: exactly one environment block"
9810 );
9811 let _ = std::fs::remove_dir_all(&base);
9812 }
9813 }
9814
9815 /// Row `synthetic-context-injection-blocks`: both presets arm the
9816 /// registry, the built-in blocks reach the assembled system prompt, and
9817 /// a block spliced mid-session reaches it too.
9818 #[test]
9819 fn presets_splice_builtin_and_runtime_context_blocks() {
9820 for preset in ["cc-parity", "cx-parity"] {
9821 let config = resolved(preset);
9822 assert!(
9823 config.context_injections,
9824 "{preset} must set core.context_injections"
9825 );
9826 let mut agent = Agent::with_provider(config, Box::new(NoProvider));
9827 let system = agent.history[0].content.clone().unwrap_or_default();
9828 assert!(
9829 system.contains("# Task list"),
9830 "{preset}: a built-in ambient block must reach the prompt"
9831 );
9832
9833 assert!(agent.inject_context_block("Mid session", "SPLICED-BODY-MARKER"));
9834 let system = agent.history[0].content.clone().unwrap_or_default();
9835 assert!(
9836 system.contains("# Mid session") && system.contains("SPLICED-BODY-MARKER"),
9837 "{preset}: a runtime splice must reach the prompt"
9838 );
9839 assert_eq!(agent.spliced_context_blocks().len(), 1);
9840 }
9841 }
9842
9843 /// A deterministic stand-in for the CLI's real provider-backed
9844 /// summarizer: it records exactly what it was asked to summarize, so the
9845 /// test can prove the `/compact <focus>` text reached the summarizer's
9846 /// INPUT and not only the marker.
9847 #[derive(Debug, Default)]
9848 struct RecordingSummarizer {
9849 seen: std::sync::Mutex<Vec<String>>,
9850 }
9851
9852 impl reduce::summarize::SpanSummarizer for RecordingSummarizer {
9853 fn summarize(&self, span_text: &str) -> reduce::Result<String> {
9854 self.seen
9855 .lock()
9856 .unwrap_or_else(std::sync::PoisonError::into_inner)
9857 .push(span_text.to_string());
9858 Ok("MODEL-WRITTEN-SUMMARY".to_string())
9859 }
9860
9861 fn model_id(&self) -> &str {
9862 "test-summarizer"
9863 }
9864 }
9865
9866 fn stuffed_agent(preset: &str) -> Agent {
9867 let config = resolved(preset);
9868 let mut agent = Agent::with_provider(config, Box::new(NoProvider));
9869 for i in 0..40 {
9870 agent.history.push(ChatMessage::user(format!("turn {i}")));
9871 agent
9872 .history
9873 .push(ChatMessage::assistant(format!("reply {i}")));
9874 }
9875 agent
9876 }
9877
9878 /// Row `manual-compact-with-focus-instructions`: `/compact <focus>`
9879 /// compacts on demand (no trigger needed) and the focus text lands in
9880 /// BOTH the summarizer's input and the marker.
9881 #[test]
9882 fn presets_manual_compact_carries_focus_into_the_summarizer_and_the_marker() {
9883 for preset in ["cc-parity", "cx-parity"] {
9884 let config = resolved(preset);
9885 assert!(
9886 config.compaction_focus_instructions.is_some(),
9887 "{preset} must state core.compaction.focus_instructions"
9888 );
9889 let mut agent = stuffed_agent(preset);
9890 let summarizer = std::sync::Arc::new(RecordingSummarizer::default());
9891 agent.set_span_summarizer_arc(summarizer.clone());
9892
9893 let before = agent.history().len();
9894 assert!(
9895 agent.compact_now(Some("keep the migration steps")),
9896 "{preset}: /compact must compact on demand"
9897 );
9898 assert!(agent.history().len() < before, "{preset}: nothing dropped");
9899
9900 let marker = agent
9901 .history()
9902 .iter()
9903 .find_map(|m| m.content.as_deref())
9904 .filter(|c| c.contains("earlier conversation compacted"))
9905 .or_else(|| {
9906 agent
9907 .history()
9908 .iter()
9909 .filter_map(|m| m.content.as_deref())
9910 .find(|c| c.contains("earlier conversation compacted"))
9911 })
9912 .unwrap_or_else(|| panic!("{preset}: no compaction marker"))
9913 .to_string();
9914 assert!(
9915 marker.contains("Focus: keep the migration steps"),
9916 "{preset}: {marker}"
9917 );
9918
9919 let seen = summarizer
9920 .seen
9921 .lock()
9922 .unwrap_or_else(std::sync::PoisonError::into_inner);
9923 assert_eq!(seen.len(), 1, "{preset}: exactly one side-call");
9924 assert!(
9925 seen[0].contains("keep the migration steps"),
9926 "{preset}: the focus must reach the summarizer INPUT — {}",
9927 &seen[0][..seen[0].len().min(200)]
9928 );
9929 }
9930 }
9931
9932 /// Row `llm-summaries-of-cleared-spans`: the model-written summary is
9933 /// produced under the presets WITHOUT `capabilities.reduction` — design
9934 /// §1.5 puts "an LLM summary of the compacted span" in core obligation
9935 /// 5, knob `[core.compaction] summarize`.
9936 #[test]
9937 fn presets_summarize_the_cleared_span_without_the_reduction_module() {
9938 for preset in ["cc-parity", "cx-parity"] {
9939 let toml = crate::presets::lookup(preset).unwrap();
9940 let r = resolve(toml, None, &ResolveOptions { strict: true }).unwrap();
9941 assert_eq!(
9942 r.modules.get("reduction"),
9943 Some(&false),
9944 "{preset}: this row must hold with the reduction module OFF"
9945 );
9946 assert!(r.config.compaction_summarize);
9947
9948 let mut agent = stuffed_agent(preset);
9949 agent.set_span_summarizer_arc(std::sync::Arc::new(RecordingSummarizer::default()));
9950 assert!(agent.compact_now(None));
9951 let marker = agent
9952 .history()
9953 .iter()
9954 .filter_map(|m| m.content.as_deref())
9955 .find(|c| c.contains("earlier conversation compacted"))
9956 .unwrap_or_else(|| panic!("{preset}: no compaction marker"));
9957 assert!(
9958 marker.contains("MODEL-WRITTEN-SUMMARY"),
9959 "{preset}: the marker must carry the model-written summary — {marker}"
9960 );
9961 }
9962 }
9963
9964 /// BP-11 (catalog "Lifecycle hooks, config-registered"): the compaction
9965 /// boundary is observable — `pre_compact` fires once the compaction is
9966 /// decided (with the manual/auto trigger named) and `post_compact` once
9967 /// the window has been rewritten, under both parity presets.
9968 #[test]
9969 fn compaction_fires_pre_and_post_lifecycle_events_under_both_presets() {
9970 use crate::config::LifecycleEvent;
9971 for preset in ["cc-parity", "cx-parity"] {
9972 let seen = std::sync::Arc::new(std::sync::Mutex::new(Vec::new()));
9973 let mut agent = stuffed_agent(preset);
9974 let sink = seen.clone();
9975 agent.set_lifecycle_hook(Box::new(move |event| {
9976 sink.lock().unwrap().push(event.clone());
9977 }));
9978 let before = agent.history().len();
9979 assert!(agent.compact_now(Some("keep the plan")), "{preset}");
9980 let after = agent.history().len();
9981 let seen = seen.lock().unwrap();
9982 assert_eq!(seen.len(), 2, "{preset}: exactly pre + post — {seen:?}");
9983 match &seen[0] {
9984 LifecycleEvent::PreCompact {
9985 messages,
9986 dropped,
9987 manual,
9988 } => {
9989 assert_eq!(*messages, before, "{preset}");
9990 assert!(*dropped > 0, "{preset}");
9991 assert!(*manual, "{preset}: /compact is the manual trigger");
9992 }
9993 other => panic!("{preset}: first event must be PreCompact, got {other:?}"),
9994 }
9995 match &seen[1] {
9996 LifecycleEvent::PostCompact { messages, dropped } => {
9997 assert_eq!(*messages, after, "{preset}");
9998 assert_eq!(
9999 *dropped,
10000 before - after + 1,
10001 "{preset}: dropped span + 1 marker"
10002 );
10003 }
10004 other => panic!("{preset}: second event must be PostCompact, got {other:?}"),
10005 }
10006 }
10007 }
10008
10009 /// The automatic trigger reports itself as such, and a window too small
10010 /// to compact fires nothing at all (no pre without a post).
10011 #[test]
10012 fn automatic_compaction_reports_the_auto_trigger_and_a_no_op_fires_nothing() {
10013 use crate::config::LifecycleEvent;
10014 let seen = std::sync::Arc::new(std::sync::Mutex::new(Vec::new()));
10015 let mut agent = stuffed_agent("cc-parity");
10016 let sink = seen.clone();
10017 agent.set_lifecycle_hook(Box::new(move |event| {
10018 sink.lock().unwrap().push(event.clone());
10019 }));
10020 agent.config.compact_after_messages = Some(10);
10021 assert!(agent.maybe_compact());
10022 assert!(matches!(
10023 seen.lock().unwrap()[0],
10024 LifecycleEvent::PreCompact { manual: false, .. }
10025 ));
10026 seen.lock().unwrap().clear();
10027 let mut small = Agent::with_provider(resolved("cc-parity"), Box::new(NoProvider));
10028 let sink = seen.clone();
10029 small.set_lifecycle_hook(Box::new(move |event| {
10030 sink.lock().unwrap().push(event.clone());
10031 }));
10032 assert!(!small.compact_now(None));
10033 assert!(seen.lock().unwrap().is_empty());
10034 }
10035
10036 /// With no summarizer installed the marker degrades to the count-only
10037 /// form — the side-call never blocks or fails compaction.
10038 #[test]
10039 fn compaction_without_a_summarizer_keeps_the_count_only_marker() {
10040 let mut agent = stuffed_agent("cc-parity");
10041 assert!(agent.compact_now(None));
10042 let marker = agent
10043 .history()
10044 .iter()
10045 .filter_map(|m| m.content.as_deref())
10046 .find(|c| c.contains("earlier conversation compacted"))
10047 .unwrap();
10048 assert!(!marker.contains("Summary of the compacted span"));
10049 }
10050
10051 /// Row `compaction-markers-persisted-in-transcript`: under the presets
10052 /// (reduction OFF) the boundary marker is written to the session
10053 /// sidecar, and says where the originals went.
10054 #[test]
10055 fn presets_persist_the_compaction_marker_to_the_transcript() {
10056 for preset in ["cc-parity", "cx-parity"] {
10057 let dir = scratch("marker");
10058 let path = dir.join("session.jsonl");
10059 let empty = supercode_interchange::session::Session::from_claude_code_str("").unwrap();
10060 let writer =
10061 supercode_interchange::sidecar::SidecarWriter::create(&path, &empty).unwrap();
10062 let mut agent = stuffed_agent(preset);
10063 agent.set_recorder(writer);
10064 assert!(agent.reduction_policy().is_none(), "{preset}");
10065
10066 assert!(agent.compact_now(None));
10067 let on_disk = std::fs::read_to_string(&path).unwrap();
10068 assert!(
10069 on_disk.contains("earlier conversation compacted"),
10070 "{preset}: the marker must reach the transcript on disk"
10071 );
10072 assert!(
10073 on_disk.contains("remain in this session's transcript sidecar"),
10074 "{preset}: the marker must say where the originals went"
10075 );
10076 let _ = std::fs::remove_dir_all(&dir);
10077 }
10078 }
10079
10080 /// Row `handoff-fresh-objective-curated-keep-set`: an in-session
10081 /// `new_context` — fresh objective, curated recent keep-set, persisted
10082 /// marker — under cx-parity, where `capabilities.reduction` is off.
10083 #[test]
10084 fn cx_parity_handoff_seeds_a_fresh_objective_with_a_curated_keep_set() {
10085 let mut agent = stuffed_agent("cx-parity");
10086 assert!(!agent.config().handoff_enabled, "reduction handoff is off");
10087 agent.history.push(ChatMessage::user("LAST-USER-TURN"));
10088 let before = agent.history().len();
10089
10090 let dropped = agent.new_context("ship the migration", Some(3));
10091 assert!(dropped > 0, "messages must be set aside");
10092 assert!(agent.history().len() < before);
10093 let system_prompt = agent.history()[0].content.clone().unwrap_or_default();
10094 assert!(
10095 system_prompt.contains("supercode") || !system_prompt.is_empty(),
10096 "the system prompt survives a handoff"
10097 );
10098 let marker = agent
10099 .history()
10100 .iter()
10101 .filter_map(|m| m.content.as_deref())
10102 .find(|c| c.contains("[handoff:"))
10103 .expect("handoff marker");
10104 assert!(marker.contains("Objective: ship the migration"));
10105 assert!(
10106 agent
10107 .history()
10108 .iter()
10109 .any(|m| m.content.as_deref() == Some("LAST-USER-TURN")),
10110 "the curated keep-set must carry the most recent turns"
10111 );
10112 }
10113
10114 /// Row `context-usage-introspection`: a live breakdown, from the same
10115 /// estimator the context guard enforces, without sending anything.
10116 #[test]
10117 fn presets_report_live_context_usage() {
10118 for preset in ["cc-parity", "cx-parity"] {
10119 let agent = stuffed_agent(preset);
10120 let usage = agent.context_usage();
10121 assert_eq!(usage.messages, agent.history().len(), "{preset}");
10122 assert!(usage.message_tokens > 0, "{preset}");
10123 assert_eq!(
10124 usage.request_tokens,
10125 usage.message_tokens + usage.tool_schema_tokens,
10126 "{preset}: the breakdown must add up"
10127 );
10128 assert!(usage.projected_tokens >= usage.request_tokens, "{preset}");
10129 assert!(usage.context_limit.is_some(), "{preset}: window known");
10130 assert!(usage.fits, "{preset}");
10131 let line = usage.summary_line();
10132 assert!(line.contains('%') && line.contains(&usage.model), "{line}");
10133
10134 // Pure: asking must not change what the next request carries.
10135 let again = agent.context_usage();
10136 assert_eq!(usage, again, "{preset}");
10137 }
10138 }
10139}
10140
10141#[cfg(test)]
10142#[path = "agent_bp10_tests.rs"]
10143mod bp10_permissions_tests;