Skip to main content

vtcode_config/core/agent/
harness.rs

1//! Harness execution, verification, and tool-result-clearing configuration.
2
3use serde::{Deserialize, Serialize};
4
5use crate::constants::defaults;
6use crate::constants::tool_limits;
7
8use super::approval::{AsyncApprovalConfig, ConfidenceEscalationConfig, SkepticPanelConfig};
9use super::{ContextResetMode, ContinuationPolicy, HarnessOrchestrationMode};
10
11#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
12#[derive(Debug, Clone, Deserialize, Serialize)]
13pub struct AgentHarnessConfig {
14    /// Maximum number of tool calls allowed per turn. Defaults to `120`.
15    /// Set to `0` to disable the cap.
16    #[serde(default = "default_harness_max_tool_calls_per_turn")]
17    pub max_tool_calls_per_turn: usize,
18    /// Maximum wall clock time (seconds) for tool execution in a turn
19    #[serde(default = "default_harness_max_tool_wall_clock_secs")]
20    pub max_tool_wall_clock_secs: u64,
21    /// Maximum retries for retryable tool errors
22    #[serde(default = "default_harness_max_tool_retries")]
23    pub max_tool_retries: u32,
24    /// Maximum number of tool calls that may execute concurrently within a single parallel batch.
25    /// Set to `0` to disable the cap (unlimited concurrency).  Default: 4.
26    #[serde(default = "default_harness_max_parallel_tool_calls")]
27    pub max_parallel_tool_calls: usize,
28    /// Enable automatic context compaction when token pressure crosses threshold.
29    ///
30    /// Enabled by default. When disabled, normal threshold-triggered automatic
31    /// compaction is skipped; the bounded post-tool recovery path may still
32    /// compact the older prefix as a safety fallback after a provider failure.
33    #[serde(default = "default_harness_auto_compaction_enabled")]
34    pub auto_compaction_enabled: bool,
35    /// Optional absolute compaction threshold (tokens) for native and local compaction.
36    ///
37    /// When set, this may lower the trigger but cannot bypass the resolved
38    /// model capacity or `context.max_context_tokens` safety ceiling.
39    /// The next response budget is reserved before deriving the trigger.
40    #[serde(default)]
41    pub auto_compaction_threshold_tokens: Option<u64>,
42    /// Optional custom instructions for the compaction summarization prompt.
43    /// When set, replaces the default Anthropic compaction prompt entirely.
44    /// Useful for tool-use scenarios to prevent the model from calling tools
45    /// during summarization. Only applies to Anthropic provider.
46    #[serde(default)]
47    pub auto_compaction_instructions: Option<String>,
48    /// Whether to pause after compaction (Anthropic only).
49    /// When true and compaction triggers, the API returns early with
50    /// `stop_reason: "compaction"` and only the compaction block.
51    /// The caller can then insert additional messages before the model
52    /// generates its text response.
53    #[serde(default)]
54    pub auto_compaction_pause_after: bool,
55    /// Automatically compact conversation context when the main session model
56    /// or provider is switched mid-conversation, so the newly selected model
57    /// starts from a summary instead of the outgoing model's raw trace.
58    /// Default: true. Disable to keep the full history across a model switch.
59    #[serde(default = "default_harness_compact_on_model_switch")]
60    pub compact_on_model_switch: bool,
61    /// Provider-native tool-result clearing policy. When enabled, old tool
62    /// results are stripped from the context once it grows past
63    /// `trigger_tokens`, keeping only the most recent `keep_tool_uses` results
64    /// and always retaining at least `clear_at_least_tokens`. This bounds
65    /// per-turn context growth and is enabled by default.
66    #[serde(default)]
67    pub tool_result_clearing: ToolResultClearingConfig,
68    /// Optional maximum estimated API cost in USD before VT Code stops the session.
69    #[serde(default)]
70    pub max_budget_usd: Option<f64>,
71    /// Fraction of `max_budget_usd` at which VT Code emits a one-time
72    /// near-budget warning. Ignored when `max_budget_usd` is unset.
73    #[serde(default = "default_harness_budget_warning_threshold")]
74    pub budget_warning_threshold: f64,
75    /// Controls whether harness-managed continuation loops are enabled.
76    #[serde(default)]
77    pub continuation_policy: ContinuationPolicy,
78    /// When to trigger a context reset — starting a clean session from
79    /// external artifacts only, discarding conversation history. Distinct
80    /// from compaction (which preserves conversational continuity).
81    /// Default: `off` (carry forward history as before).
82    #[serde(default)]
83    pub context_reset_mode: ContextResetMode,
84    /// Number of consecutive stall turns before `on_stall` context reset
85    /// triggers. Ignored unless `context_reset_mode = "on_stall"`.
86    /// Default: 2.
87    #[serde(default = "default_harness_context_reset_stall_threshold")]
88    pub context_reset_stall_threshold: u32,
89    /// Optional compatibility/export JSONL path for harness events.
90    /// Canonical events are always stored under the workspace session store;
91    /// unset configuration creates no global harness file.
92    #[serde(default)]
93    pub event_log_path: Option<String>,
94    /// Select the exec/full-auto harness orchestration path.
95    #[serde(default)]
96    pub orchestration_mode: HarnessOrchestrationMode,
97    /// Maximum generator revision rounds after evaluator rejection.
98    #[serde(default = "default_harness_max_revision_rounds")]
99    pub max_revision_rounds: usize,
100    /// Confidence-based escalation for autonomous decisions.
101    ///
102    /// Implements the escalation decision rule:
103    ///   Escalate iff p_success < tau_conf OR action in A_irreversible OR cost > B_auto
104    ///
105    /// When a tool call is classified as irreversible or below the confidence
106    /// threshold, the harness escalates to blocked-handoff instead of proceeding
107    /// autonomously.  Opt-in (default: disabled).
108    #[serde(default)]
109    pub confidence_escalation: ConfidenceEscalationConfig,
110    /// Async (out-of-band) approval for deferred tool execution requests.
111    ///
112    /// When enabled, approval requests exceeding the auto-approve cost threshold
113    /// write a blocker file and notify the user out-of-band rather than blocking
114    /// on terminal input.  Opt-in (default: disabled).
115    #[serde(default)]
116    async_approval: AsyncApprovalConfig,
117    /// Adversarial multi-model evaluator panel.  When enabled, the harness
118    /// runs the evaluator prompt against every listed model in parallel and
119    /// aggregates the strictest verdict/scorecard across the panel.
120    /// Opt-in (default: disabled).
121    #[serde(default)]
122    pub skeptic_panel: SkepticPanelConfig,
123    /// Autonomous recovery for the anti-blind-editing verification gate.
124    /// When the model emits text instead of running a verifier, the harness
125    /// grants bounded directive retries and (optionally) runs the detected
126    /// project verifier itself instead of forcing manual `continue`.
127    #[serde(default)]
128    pub verification: VerificationAutoRecoveryConfig,
129    /// Tracker-aware auto-continuation: when `task_tracker` still has
130    /// incomplete steps, keep looping / auto-queue the next turn instead of
131    /// ending and nudging the user to resume.
132    #[serde(default)]
133    pub continuation: TrackerContinuationConfig,
134}
135
136impl Default for AgentHarnessConfig {
137    fn default() -> Self {
138        Self {
139            max_tool_calls_per_turn: default_harness_max_tool_calls_per_turn(),
140            max_tool_wall_clock_secs: default_harness_max_tool_wall_clock_secs(),
141            max_tool_retries: default_harness_max_tool_retries(),
142            max_parallel_tool_calls: default_harness_max_parallel_tool_calls(),
143            auto_compaction_enabled: default_harness_auto_compaction_enabled(),
144            auto_compaction_threshold_tokens: None,
145            auto_compaction_instructions: None,
146            auto_compaction_pause_after: false,
147            compact_on_model_switch: default_harness_compact_on_model_switch(),
148            tool_result_clearing: ToolResultClearingConfig::default(),
149            max_budget_usd: None,
150            budget_warning_threshold: default_harness_budget_warning_threshold(),
151            continuation_policy: ContinuationPolicy::default(),
152            context_reset_mode: ContextResetMode::default(),
153            context_reset_stall_threshold: default_harness_context_reset_stall_threshold(),
154            event_log_path: None,
155            orchestration_mode: HarnessOrchestrationMode::default(),
156            max_revision_rounds: default_harness_max_revision_rounds(),
157            confidence_escalation: ConfidenceEscalationConfig::default(),
158            async_approval: AsyncApprovalConfig::default(),
159            skeptic_panel: SkepticPanelConfig::default(),
160            verification: VerificationAutoRecoveryConfig::default(),
161            continuation: TrackerContinuationConfig::default(),
162        }
163    }
164}
165/// Tracker-aware auto-continuation policy.
166///
167/// When `task_tracker` still has incomplete items, the binary runloop
168/// continues in-turn for status-only responses and, after recoverable
169/// budget/recovery turn ends (or on session resume), auto-queues a bounded
170/// follow-up turn instead of requiring the user to type `continue`.
171#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
172#[derive(Debug, Clone, Deserialize, Serialize)]
173pub struct TrackerContinuationConfig {
174    /// Auto-continue while the task tracker has incomplete steps.
175    /// Default: true.
176    #[serde(default = "default_tracker_auto_continue")]
177    pub auto_continue_tracker: bool,
178    /// Bounded cross-turn auto-continue turns after a recoverable end
179    /// (budget/preview/tool-free recovery) or on resume while tracker work
180    /// remains. The episode budget **progress-resets** when any tracker step
181    /// completes. `0` disables cross-turn tracker auto-queue (in-turn
182    /// continuation still applies when `auto_continue_tracker` is true).
183    /// Default: 32.
184    #[serde(default = "default_tracker_cross_turn_turns")]
185    pub cross_turn_turns: u8,
186}
187
188impl Default for TrackerContinuationConfig {
189    fn default() -> Self {
190        Self {
191            auto_continue_tracker: default_tracker_auto_continue(),
192            cross_turn_turns: default_tracker_cross_turn_turns(),
193        }
194    }
195}
196
197#[inline]
198const fn default_tracker_auto_continue() -> bool {
199    true
200}
201
202#[inline]
203const fn default_tracker_cross_turn_turns() -> u8 {
204    32
205}
206/// Autonomous recovery policy for the anti-blind-editing verification gate.
207///
208/// After 6 consecutive successful mutations without verification, text-only
209/// responses no longer block the turn immediately: the harness grants bounded
210/// directive retries naming the exact project verifier, then (when
211/// `auto_execute` is set) runs that verifier itself through the normal tool
212/// pipeline instead of forcing the user to type `continue`. A verifier that
213/// keeps failing escalates to a manual blocked handoff carrying the failure
214/// log after `max_consecutive_failures` consecutive failures.
215#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
216#[derive(Debug, Clone, Deserialize, Serialize)]
217pub struct VerificationAutoRecoveryConfig {
218    /// Run the detected project verifier through the normal tool pipeline
219    /// when the model exhausts its directive retries without verifying.
220    /// Default: true. Set to false to restore directive-only recovery.
221    #[serde(default = "default_verification_auto_execute")]
222    pub auto_execute: bool,
223    /// Bounded in-turn directive retries granted when the model emits text
224    /// instead of a verifier while the gate is pending. Each grant resets the
225    /// text-response streak once and injects a project-aware directive.
226    /// Default: 2.
227    #[serde(default = "default_verification_in_turn_attempts")]
228    pub in_turn_attempts: u8,
229    /// Autonomous cross-turn recovery turns scheduled after a
230    /// verification-blocked turn before a manual blocked handoff is written.
231    /// Default: 2.
232    #[serde(default = "default_verification_cross_turn_turns")]
233    pub cross_turn_turns: u8,
234    /// Explicit verifier command overriding project-marker detection
235    /// (e.g. `"cargo nextest run -p mycrate"`). Must be a standalone verifier
236    /// or pure `&&` chain; pipes and `;`/`||` joins are rejected at use.
237    #[serde(default)]
238    pub default_verifier_override: Option<String>,
239    /// Consecutive failed harness auto-verifications before escalation to a
240    /// manual blocked handoff carrying the failure log. Reset by any success,
241    /// completed turn, or fresh user input. Default: 3.
242    #[serde(default = "default_verification_max_consecutive_failures")]
243    pub max_consecutive_failures: u8,
244}
245
246impl Default for VerificationAutoRecoveryConfig {
247    fn default() -> Self {
248        Self {
249            auto_execute: default_verification_auto_execute(),
250            in_turn_attempts: default_verification_in_turn_attempts(),
251            cross_turn_turns: default_verification_cross_turn_turns(),
252            default_verifier_override: None,
253            max_consecutive_failures: default_verification_max_consecutive_failures(),
254        }
255    }
256}
257#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
258#[derive(Debug, Clone, Deserialize, Serialize)]
259pub struct ToolResultClearingConfig {
260    #[serde(default = "default_tool_result_clearing_enabled")]
261    pub enabled: bool,
262    #[serde(default = "default_tool_result_clearing_trigger_tokens")]
263    pub trigger_tokens: u64,
264    #[serde(default = "default_tool_result_clearing_keep_tool_uses")]
265    pub keep_tool_uses: u32,
266    #[serde(default = "default_tool_result_clearing_clear_at_least_tokens")]
267    pub clear_at_least_tokens: u64,
268    /// Replace paired `tool_calls[].function.arguments` for stubbed results.
269    /// Leaving inputs in place keeps full `apply_patch`/`write_file` bodies on
270    /// every request after the result was already reclaimed. Defaults to
271    /// `true`; set `false` to opt out. Bare `#[serde(default)]` on a `bool`
272    /// is `false`, so this field names its default fn explicitly.
273    #[serde(default = "default_tool_result_clearing_clear_tool_inputs")]
274    pub clear_tool_inputs: bool,
275}
276
277impl Default for ToolResultClearingConfig {
278    fn default() -> Self {
279        Self {
280            enabled: default_tool_result_clearing_enabled(),
281            trigger_tokens: default_tool_result_clearing_trigger_tokens(),
282            keep_tool_uses: default_tool_result_clearing_keep_tool_uses(),
283            clear_at_least_tokens: default_tool_result_clearing_clear_at_least_tokens(),
284            clear_tool_inputs: default_tool_result_clearing_clear_tool_inputs(),
285        }
286    }
287}
288
289impl ToolResultClearingConfig {
290    pub(crate) fn validate(&self) -> Result<(), String> {
291        if self.trigger_tokens == 0 {
292            return Err("tool_result_clearing.trigger_tokens must be greater than 0".to_string());
293        }
294        if self.keep_tool_uses == 0 {
295            return Err("tool_result_clearing.keep_tool_uses must be greater than 0".to_string());
296        }
297        if self.clear_at_least_tokens == 0 {
298            return Err("tool_result_clearing.clear_at_least_tokens must be greater than 0".to_string());
299        }
300        Ok(())
301    }
302}
303#[inline]
304const fn default_harness_max_tool_calls_per_turn() -> usize {
305    tool_limits::DEFAULT_MAX_TOOL_CALLS_PER_TURN
306}
307
308#[inline]
309const fn default_harness_max_tool_wall_clock_secs() -> u64 {
310    defaults::DEFAULT_MAX_TOOL_WALL_CLOCK_SECS
311}
312
313#[inline]
314const fn default_harness_max_tool_retries() -> u32 {
315    defaults::DEFAULT_MAX_TOOL_RETRIES
316}
317
318#[inline]
319const fn default_harness_max_parallel_tool_calls() -> usize {
320    4 // Cap parallel fan-out at 4; set to 0 in vtcode.toml to remove the limit.
321}
322
323#[inline]
324const fn default_harness_auto_compaction_enabled() -> bool {
325    true
326}
327
328const fn default_harness_compact_on_model_switch() -> bool {
329    true
330}
331
332#[inline]
333const fn default_harness_context_reset_stall_threshold() -> u32 {
334    2
335}
336
337#[inline]
338const fn default_verification_auto_execute() -> bool {
339    true
340}
341
342#[inline]
343const fn default_verification_in_turn_attempts() -> u8 {
344    2
345}
346
347#[inline]
348const fn default_verification_cross_turn_turns() -> u8 {
349    2
350}
351
352#[inline]
353const fn default_verification_max_consecutive_failures() -> u8 {
354    3
355}
356
357#[inline]
358const fn default_tool_result_clearing_enabled() -> bool {
359    true
360}
361
362#[inline]
363const fn default_tool_result_clearing_trigger_tokens() -> u64 {
364    // 40k: research/audit turns were observed at ~1M input tokens/turn with
365    // the old 100k trigger — tool results piled up long before any clearing.
366    40_000
367}
368
369#[inline]
370const fn default_tool_result_clearing_keep_tool_uses() -> u32 {
371    2
372}
373
374#[inline]
375const fn default_tool_result_clearing_clear_at_least_tokens() -> u64 {
376    30_000
377}
378
379#[inline]
380const fn default_tool_result_clearing_clear_tool_inputs() -> bool {
381    true
382}
383
384#[inline]
385const fn default_harness_max_revision_rounds() -> usize {
386    2
387}
388
389#[inline]
390const fn default_harness_budget_warning_threshold() -> f64 {
391    0.75
392}