vtcode_config/core/agent/harness.rs
1//! Harness execution, verification, and tool-result-clearing configuration.
2
3use serde::{Deserialize, Serialize};
4
5use crate::constants::defaults;
6use crate::constants::tool_limits;
7
8use super::approval::{AsyncApprovalConfig, ConfidenceEscalationConfig, SkepticPanelConfig};
9use super::{ContextResetMode, ContinuationPolicy, HarnessOrchestrationMode};
10
11#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
12#[derive(Debug, Clone, Deserialize, Serialize)]
13pub struct AgentHarnessConfig {
14 /// Maximum number of tool calls allowed per turn. Defaults to `120`.
15 /// Set to `0` to disable the cap.
16 #[serde(default = "default_harness_max_tool_calls_per_turn")]
17 pub max_tool_calls_per_turn: usize,
18 /// Maximum wall clock time (seconds) for tool execution in a turn
19 #[serde(default = "default_harness_max_tool_wall_clock_secs")]
20 pub max_tool_wall_clock_secs: u64,
21 /// Maximum retries for retryable tool errors
22 #[serde(default = "default_harness_max_tool_retries")]
23 pub max_tool_retries: u32,
24 /// Maximum number of tool calls that may execute concurrently within a single parallel batch.
25 /// Set to `0` to disable the cap (unlimited concurrency). Default: 4.
26 #[serde(default = "default_harness_max_parallel_tool_calls")]
27 pub max_parallel_tool_calls: usize,
28 /// Enable automatic context compaction when token pressure crosses threshold.
29 ///
30 /// Enabled by default. When disabled, normal threshold-triggered automatic
31 /// compaction is skipped; the bounded post-tool recovery path may still
32 /// compact the older prefix as a safety fallback after a provider failure.
33 #[serde(default = "default_harness_auto_compaction_enabled")]
34 pub auto_compaction_enabled: bool,
35 /// Optional absolute compaction threshold (tokens) for native and local compaction.
36 ///
37 /// When set, this may lower the trigger but cannot bypass the resolved
38 /// model capacity or `context.max_context_tokens` safety ceiling.
39 /// The next response budget is reserved before deriving the trigger.
40 #[serde(default)]
41 pub auto_compaction_threshold_tokens: Option<u64>,
42 /// Optional custom instructions for the compaction summarization prompt.
43 /// When set, replaces the default Anthropic compaction prompt entirely.
44 /// Useful for tool-use scenarios to prevent the model from calling tools
45 /// during summarization. Only applies to Anthropic provider.
46 #[serde(default)]
47 pub auto_compaction_instructions: Option<String>,
48 /// Whether to pause after compaction (Anthropic only).
49 /// When true and compaction triggers, the API returns early with
50 /// `stop_reason: "compaction"` and only the compaction block.
51 /// The caller can then insert additional messages before the model
52 /// generates its text response.
53 #[serde(default)]
54 pub auto_compaction_pause_after: bool,
55 /// Automatically compact conversation context when the main session model
56 /// or provider is switched mid-conversation, so the newly selected model
57 /// starts from a summary instead of the outgoing model's raw trace.
58 /// Default: true. Disable to keep the full history across a model switch.
59 #[serde(default = "default_harness_compact_on_model_switch")]
60 pub compact_on_model_switch: bool,
61 /// Provider-native tool-result clearing policy. When enabled, old tool
62 /// results are stripped from the context once it grows past
63 /// `trigger_tokens`, keeping only the most recent `keep_tool_uses` results
64 /// and always retaining at least `clear_at_least_tokens`. This bounds
65 /// per-turn context growth and is enabled by default.
66 #[serde(default)]
67 pub tool_result_clearing: ToolResultClearingConfig,
68 /// Optional maximum estimated API cost in USD before VT Code stops the session.
69 #[serde(default)]
70 pub max_budget_usd: Option<f64>,
71 /// Fraction of `max_budget_usd` at which VT Code emits a one-time
72 /// near-budget warning. Ignored when `max_budget_usd` is unset.
73 #[serde(default = "default_harness_budget_warning_threshold")]
74 pub budget_warning_threshold: f64,
75 /// Controls whether harness-managed continuation loops are enabled.
76 #[serde(default)]
77 pub continuation_policy: ContinuationPolicy,
78 /// When to trigger a context reset — starting a clean session from
79 /// external artifacts only, discarding conversation history. Distinct
80 /// from compaction (which preserves conversational continuity).
81 /// Default: `off` (carry forward history as before).
82 #[serde(default)]
83 pub context_reset_mode: ContextResetMode,
84 /// Number of consecutive stall turns before `on_stall` context reset
85 /// triggers. Ignored unless `context_reset_mode = "on_stall"`.
86 /// Default: 2.
87 #[serde(default = "default_harness_context_reset_stall_threshold")]
88 pub context_reset_stall_threshold: u32,
89 /// Optional compatibility/export JSONL path for harness events.
90 /// Canonical events are always stored under the workspace session store;
91 /// unset configuration creates no global harness file.
92 #[serde(default)]
93 pub event_log_path: Option<String>,
94 /// Select the exec/full-auto harness orchestration path.
95 #[serde(default)]
96 pub orchestration_mode: HarnessOrchestrationMode,
97 /// Maximum generator revision rounds after evaluator rejection.
98 #[serde(default = "default_harness_max_revision_rounds")]
99 pub max_revision_rounds: usize,
100 /// Confidence-based escalation for autonomous decisions.
101 ///
102 /// Implements the escalation decision rule:
103 /// Escalate iff p_success < tau_conf OR action in A_irreversible OR cost > B_auto
104 ///
105 /// When a tool call is classified as irreversible or below the confidence
106 /// threshold, the harness escalates to blocked-handoff instead of proceeding
107 /// autonomously. Opt-in (default: disabled).
108 #[serde(default)]
109 pub confidence_escalation: ConfidenceEscalationConfig,
110 /// Async (out-of-band) approval for deferred tool execution requests.
111 ///
112 /// When enabled, approval requests exceeding the auto-approve cost threshold
113 /// write a blocker file and notify the user out-of-band rather than blocking
114 /// on terminal input. Opt-in (default: disabled).
115 #[serde(default)]
116 async_approval: AsyncApprovalConfig,
117 /// Adversarial multi-model evaluator panel. When enabled, the harness
118 /// runs the evaluator prompt against every listed model in parallel and
119 /// aggregates the strictest verdict/scorecard across the panel.
120 /// Opt-in (default: disabled).
121 #[serde(default)]
122 pub skeptic_panel: SkepticPanelConfig,
123 /// Autonomous recovery for the anti-blind-editing verification gate.
124 /// When the model emits text instead of running a verifier, the harness
125 /// grants bounded directive retries and (optionally) runs the detected
126 /// project verifier itself instead of forcing manual `continue`.
127 #[serde(default)]
128 pub verification: VerificationAutoRecoveryConfig,
129 /// Tracker-aware auto-continuation: when `task_tracker` still has
130 /// incomplete steps, keep looping / auto-queue the next turn instead of
131 /// ending and nudging the user to resume.
132 #[serde(default)]
133 pub continuation: TrackerContinuationConfig,
134}
135
136impl Default for AgentHarnessConfig {
137 fn default() -> Self {
138 Self {
139 max_tool_calls_per_turn: default_harness_max_tool_calls_per_turn(),
140 max_tool_wall_clock_secs: default_harness_max_tool_wall_clock_secs(),
141 max_tool_retries: default_harness_max_tool_retries(),
142 max_parallel_tool_calls: default_harness_max_parallel_tool_calls(),
143 auto_compaction_enabled: default_harness_auto_compaction_enabled(),
144 auto_compaction_threshold_tokens: None,
145 auto_compaction_instructions: None,
146 auto_compaction_pause_after: false,
147 compact_on_model_switch: default_harness_compact_on_model_switch(),
148 tool_result_clearing: ToolResultClearingConfig::default(),
149 max_budget_usd: None,
150 budget_warning_threshold: default_harness_budget_warning_threshold(),
151 continuation_policy: ContinuationPolicy::default(),
152 context_reset_mode: ContextResetMode::default(),
153 context_reset_stall_threshold: default_harness_context_reset_stall_threshold(),
154 event_log_path: None,
155 orchestration_mode: HarnessOrchestrationMode::default(),
156 max_revision_rounds: default_harness_max_revision_rounds(),
157 confidence_escalation: ConfidenceEscalationConfig::default(),
158 async_approval: AsyncApprovalConfig::default(),
159 skeptic_panel: SkepticPanelConfig::default(),
160 verification: VerificationAutoRecoveryConfig::default(),
161 continuation: TrackerContinuationConfig::default(),
162 }
163 }
164}
165/// Tracker-aware auto-continuation policy.
166///
167/// When `task_tracker` still has incomplete items, the binary runloop
168/// continues in-turn for status-only responses and, after recoverable
169/// budget/recovery turn ends (or on session resume), auto-queues a bounded
170/// follow-up turn instead of requiring the user to type `continue`.
171#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
172#[derive(Debug, Clone, Deserialize, Serialize)]
173pub struct TrackerContinuationConfig {
174 /// Auto-continue while the task tracker has incomplete steps.
175 /// Default: true.
176 #[serde(default = "default_tracker_auto_continue")]
177 pub auto_continue_tracker: bool,
178 /// Bounded cross-turn auto-continue turns after a recoverable end
179 /// (budget/preview/tool-free recovery) or on resume while tracker work
180 /// remains. The episode budget **progress-resets** when any tracker step
181 /// completes. `0` disables cross-turn tracker auto-queue (in-turn
182 /// continuation still applies when `auto_continue_tracker` is true).
183 /// Default: 32.
184 #[serde(default = "default_tracker_cross_turn_turns")]
185 pub cross_turn_turns: u8,
186}
187
188impl Default for TrackerContinuationConfig {
189 fn default() -> Self {
190 Self {
191 auto_continue_tracker: default_tracker_auto_continue(),
192 cross_turn_turns: default_tracker_cross_turn_turns(),
193 }
194 }
195}
196
197#[inline]
198const fn default_tracker_auto_continue() -> bool {
199 true
200}
201
202#[inline]
203const fn default_tracker_cross_turn_turns() -> u8 {
204 32
205}
206/// Autonomous recovery policy for the anti-blind-editing verification gate.
207///
208/// After 6 consecutive successful mutations without verification, text-only
209/// responses no longer block the turn immediately: the harness grants bounded
210/// directive retries naming the exact project verifier, then (when
211/// `auto_execute` is set) runs that verifier itself through the normal tool
212/// pipeline instead of forcing the user to type `continue`. A verifier that
213/// keeps failing escalates to a manual blocked handoff carrying the failure
214/// log after `max_consecutive_failures` consecutive failures.
215#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
216#[derive(Debug, Clone, Deserialize, Serialize)]
217pub struct VerificationAutoRecoveryConfig {
218 /// Run the detected project verifier through the normal tool pipeline
219 /// when the model exhausts its directive retries without verifying.
220 /// Default: true. Set to false to restore directive-only recovery.
221 #[serde(default = "default_verification_auto_execute")]
222 pub auto_execute: bool,
223 /// Bounded in-turn directive retries granted when the model emits text
224 /// instead of a verifier while the gate is pending. Each grant resets the
225 /// text-response streak once and injects a project-aware directive.
226 /// Default: 2.
227 #[serde(default = "default_verification_in_turn_attempts")]
228 pub in_turn_attempts: u8,
229 /// Autonomous cross-turn recovery turns scheduled after a
230 /// verification-blocked turn before a manual blocked handoff is written.
231 /// Default: 2.
232 #[serde(default = "default_verification_cross_turn_turns")]
233 pub cross_turn_turns: u8,
234 /// Explicit verifier command overriding project-marker detection
235 /// (e.g. `"cargo nextest run -p mycrate"`). Must be a standalone verifier
236 /// or pure `&&` chain; pipes and `;`/`||` joins are rejected at use.
237 #[serde(default)]
238 pub default_verifier_override: Option<String>,
239 /// Consecutive failed harness auto-verifications before escalation to a
240 /// manual blocked handoff carrying the failure log. Reset by any success,
241 /// completed turn, or fresh user input. Default: 3.
242 #[serde(default = "default_verification_max_consecutive_failures")]
243 pub max_consecutive_failures: u8,
244}
245
246impl Default for VerificationAutoRecoveryConfig {
247 fn default() -> Self {
248 Self {
249 auto_execute: default_verification_auto_execute(),
250 in_turn_attempts: default_verification_in_turn_attempts(),
251 cross_turn_turns: default_verification_cross_turn_turns(),
252 default_verifier_override: None,
253 max_consecutive_failures: default_verification_max_consecutive_failures(),
254 }
255 }
256}
257#[cfg_attr(feature = "schema", derive(schemars::JsonSchema))]
258#[derive(Debug, Clone, Deserialize, Serialize)]
259pub struct ToolResultClearingConfig {
260 #[serde(default = "default_tool_result_clearing_enabled")]
261 pub enabled: bool,
262 #[serde(default = "default_tool_result_clearing_trigger_tokens")]
263 pub trigger_tokens: u64,
264 #[serde(default = "default_tool_result_clearing_keep_tool_uses")]
265 pub keep_tool_uses: u32,
266 #[serde(default = "default_tool_result_clearing_clear_at_least_tokens")]
267 pub clear_at_least_tokens: u64,
268 /// Replace paired `tool_calls[].function.arguments` for stubbed results.
269 /// Leaving inputs in place keeps full `apply_patch`/`write_file` bodies on
270 /// every request after the result was already reclaimed. Defaults to
271 /// `true`; set `false` to opt out. Bare `#[serde(default)]` on a `bool`
272 /// is `false`, so this field names its default fn explicitly.
273 #[serde(default = "default_tool_result_clearing_clear_tool_inputs")]
274 pub clear_tool_inputs: bool,
275}
276
277impl Default for ToolResultClearingConfig {
278 fn default() -> Self {
279 Self {
280 enabled: default_tool_result_clearing_enabled(),
281 trigger_tokens: default_tool_result_clearing_trigger_tokens(),
282 keep_tool_uses: default_tool_result_clearing_keep_tool_uses(),
283 clear_at_least_tokens: default_tool_result_clearing_clear_at_least_tokens(),
284 clear_tool_inputs: default_tool_result_clearing_clear_tool_inputs(),
285 }
286 }
287}
288
289impl ToolResultClearingConfig {
290 pub(crate) fn validate(&self) -> Result<(), String> {
291 if self.trigger_tokens == 0 {
292 return Err("tool_result_clearing.trigger_tokens must be greater than 0".to_string());
293 }
294 if self.keep_tool_uses == 0 {
295 return Err("tool_result_clearing.keep_tool_uses must be greater than 0".to_string());
296 }
297 if self.clear_at_least_tokens == 0 {
298 return Err("tool_result_clearing.clear_at_least_tokens must be greater than 0".to_string());
299 }
300 Ok(())
301 }
302}
303#[inline]
304const fn default_harness_max_tool_calls_per_turn() -> usize {
305 tool_limits::DEFAULT_MAX_TOOL_CALLS_PER_TURN
306}
307
308#[inline]
309const fn default_harness_max_tool_wall_clock_secs() -> u64 {
310 defaults::DEFAULT_MAX_TOOL_WALL_CLOCK_SECS
311}
312
313#[inline]
314const fn default_harness_max_tool_retries() -> u32 {
315 defaults::DEFAULT_MAX_TOOL_RETRIES
316}
317
318#[inline]
319const fn default_harness_max_parallel_tool_calls() -> usize {
320 4 // Cap parallel fan-out at 4; set to 0 in vtcode.toml to remove the limit.
321}
322
323#[inline]
324const fn default_harness_auto_compaction_enabled() -> bool {
325 true
326}
327
328const fn default_harness_compact_on_model_switch() -> bool {
329 true
330}
331
332#[inline]
333const fn default_harness_context_reset_stall_threshold() -> u32 {
334 2
335}
336
337#[inline]
338const fn default_verification_auto_execute() -> bool {
339 true
340}
341
342#[inline]
343const fn default_verification_in_turn_attempts() -> u8 {
344 2
345}
346
347#[inline]
348const fn default_verification_cross_turn_turns() -> u8 {
349 2
350}
351
352#[inline]
353const fn default_verification_max_consecutive_failures() -> u8 {
354 3
355}
356
357#[inline]
358const fn default_tool_result_clearing_enabled() -> bool {
359 true
360}
361
362#[inline]
363const fn default_tool_result_clearing_trigger_tokens() -> u64 {
364 // 40k: research/audit turns were observed at ~1M input tokens/turn with
365 // the old 100k trigger — tool results piled up long before any clearing.
366 40_000
367}
368
369#[inline]
370const fn default_tool_result_clearing_keep_tool_uses() -> u32 {
371 2
372}
373
374#[inline]
375const fn default_tool_result_clearing_clear_at_least_tokens() -> u64 {
376 30_000
377}
378
379#[inline]
380const fn default_tool_result_clearing_clear_tool_inputs() -> bool {
381 true
382}
383
384#[inline]
385const fn default_harness_max_revision_rounds() -> usize {
386 2
387}
388
389#[inline]
390const fn default_harness_budget_warning_threshold() -> f64 {
391 0.75
392}