vtcode_core/compaction/strategy.rs
1//! Compaction strategy selection and manual-compaction options.
2
3use super::native_inline::compact_history_native_inline;
4use super::*;
5
6/// How the manual `/compact` command compacts for a given provider/model.
7#[derive(Debug, Clone, Copy, PartialEq, Eq)]
8pub enum CompactionStrategy {
9 /// Provider exposes a standalone on-demand compaction endpoint
10 /// (OpenAI `/responses/compact`). Delegates to `LLMProvider::compact_history_with_options`.
11 NativeStandalone,
12 /// Provider compacts inline via request fields, threshold-triggered
13 /// (Anthropic `compact_20260112`). Invoked through the capability-aware
14 /// one-shot response collector with `context_management` set and
15 /// `pause_after_compaction`.
16 NativeInline,
17 /// Universal fallback: summarize history via the capability-aware one-shot
18 /// response collector and rebuild as a summary message plus retained recent
19 /// user messages. Works for every provider.
20 Local,
21}
22
23/// Select the manual-compaction strategy for a provider/model.
24///
25/// `NativeStandalone` when the provider opts in via `supports_manual_openai_compaction`
26/// (e.g. OpenAI `/responses/compact`), `NativeInline` when the provider reports
27/// inline compaction support via `supports_native_inline_compaction` (e.g.
28/// Anthropic `compact_20260112`), otherwise `Local`.
29///
30/// Note: `supports_responses_compaction` is intentionally *not* the discriminator
31/// for `NativeInline`. It is overloaded — true for both OpenAI-compatible
32/// standalone compaction and Anthropic inline compaction — so OpenAI-compatible
33/// custom endpoints (which report it but cannot serve an Anthropic
34/// `compact_20260112` edit) would otherwise be misrouted to `NativeInline` and
35/// waste a rejected `generate` call before falling back to `Local`.
36pub fn manual_compaction_strategy(provider: &dyn LLMProvider, model: &str) -> CompactionStrategy {
37 if provider.supports_manual_openai_compaction(model) {
38 CompactionStrategy::NativeStandalone
39 } else if provider.supports_native_inline_compaction(model) {
40 CompactionStrategy::NativeInline
41 } else {
42 CompactionStrategy::Local
43 }
44}
45
46/// Whether a compaction result is worth keeping.
47///
48/// Local compaction always appends framing around the summarized history (a
49/// summary message plus the persisted session-memory envelope), so a result
50/// whose message count is not strictly smaller than the input grows the
51/// conversation instead of compacting it. This happens when the continuity
52/// tail already covers the whole history (small, tool-heavy sessions where a
53/// handful of user turns anchor dozens of assistant/tool messages): the
54/// rebuild keeps every message and adds framing on top (observed as
55/// `86 -> 88`). Such results must be discarded in favor of the already-compact
56/// path so callers never report growth as a successful compaction.
57///
58/// The check is intentionally message-count based: that is the unit surfaced
59/// in user-facing progress (`86 -> 12`) and harness `compact_boundary`
60/// events, so the persisted history must never exceed it.
61#[must_use]
62pub fn compacted_history_shrinks(original_len: usize, compacted_len: usize, mode: CompactionMode) -> bool {
63 // Local mode injects one envelope message into the returned history during
64 // persistence; account for it here so the *final* history is what shrinks.
65 // Provider-native windows (and the `Unknown` catch-all, which never arises
66 // from a live compaction pass) are canonical replay state with no envelope
67 // injected, so the raw lengths compare directly.
68 let envelope_messages = match mode {
69 CompactionMode::Local => 1,
70 CompactionMode::Provider | CompactionMode::Unknown => 0,
71 };
72 compacted_len.saturating_add(envelope_messages) < original_len
73}
74
75/// Universally meaningful manual-compaction options.
76///
77/// Provider-specific extras (OpenAI `service_tier` / `prompt_cache_key` / `store` /
78/// `include`) are intentionally absent: the manual `/compact` command exposes only
79/// the options that apply across every provider.
80#[derive(Debug, Clone, Default, PartialEq, Eq)]
81pub struct ManualCompactionOptions {
82 /// Overrides the default summary/compaction prompt when set.
83 pub instructions: Option<String>,
84 /// Caps the summary/compaction output length on every provider.
85 pub max_output_tokens: Option<u32>,
86 /// Optional reasoning effort override for the compaction pass.
87 pub reasoning_effort: Option<ReasoningEffortLevel>,
88 /// Permit an explicit lower supported effort when the requested level is
89 /// unavailable on the selected provider/model route. The default is
90 /// strict blocking so compaction never silently changes reasoning
91 /// fidelity.
92 pub allow_reasoning_effort_downgrade: bool,
93 /// Optional verbosity override for the compaction output.
94 pub verbosity: Option<VerbosityLevel>,
95}
96
97impl From<ManualCompactionOptions> for ResponsesCompactionOptions {
98 fn from(options: ManualCompactionOptions) -> Self {
99 Self {
100 instructions: options.instructions,
101 max_output_tokens: options.max_output_tokens,
102 reasoning_effort: options.reasoning_effort,
103 verbosity: options.verbosity,
104 responses_include: None,
105 response_store: None,
106 service_tier: None,
107 prompt_cache_key: None,
108 }
109 }
110}
111
112impl CompactionConfig {
113 /// Return a config with the manual options' instructions applied as the
114 /// summary prompt override. The remaining option fields
115 /// (`max_output_tokens`, `reasoning_effort`, `verbosity`) are applied to the
116 /// summary `LLMRequest` directly by `summarize_locally`, not stored here.
117 pub(crate) fn with_manual_overrides(self, options: &ManualCompactionOptions) -> Self {
118 let summary_prompt = options
119 .instructions
120 .clone()
121 .map(|instructions| instructions.trim().to_string())
122 .filter(|instructions| !instructions.is_empty())
123 .unwrap_or(self.summary_prompt);
124 Self { summary_prompt, ..self }
125 }
126}
127
128/// Compact history for the manual `/compact` command using provider-native
129/// compaction when available, falling back to local summarization otherwise.
130///
131/// Returns the compacted messages and the `CompactionMode` that produced them
132/// (`Provider` for native compaction, `Local` for client-side summarization).
133#[cfg_attr(feature = "profiling", hotpath::measure)]
134pub async fn compact_history_manual(
135 provider: &dyn LLMProvider,
136 model: &str,
137 history: &[Message],
138 config: &CompactionConfig,
139 options: &ManualCompactionOptions,
140) -> Result<(Vec<Message>, CompactionMode)> {
141 compact_history_manual_with_budget(provider, model, history, config, options, None).await
142}
143
144/// Manual compaction with the resolved session context budget applied to native
145/// request inputs and locally rebuilt summaries. Native provider responses are
146/// returned unchanged so opaque continuation state remains valid.
147pub async fn compact_history_manual_with_budget(
148 provider: &dyn LLMProvider,
149 model: &str,
150 history: &[Message],
151 config: &CompactionConfig,
152 options: &ManualCompactionOptions,
153 context_budget: Option<usize>,
154) -> Result<(Vec<Message>, CompactionMode)> {
155 compact_history_manual_with_parent_context(provider, model, history, config, options, context_budget, None).await
156}
157
158/// Manual compaction that reuses the parent segment's cached prefix.
159///
160/// Pass the live request envelope's system prompt + ordered tools as `parent`
161/// so the local-summary fork (`summarize_locally`) hits the provider prompt
162/// cache instead of re-paying full input cost for the entire history.
163pub async fn compact_history_manual_with_parent_context(
164 provider: &dyn LLMProvider,
165 model: &str,
166 history: &[Message],
167 config: &CompactionConfig,
168 options: &ManualCompactionOptions,
169 context_budget: Option<usize>,
170 parent: Option<&CompactionParentContext>,
171) -> Result<(Vec<Message>, CompactionMode)> {
172 if history.is_empty() {
173 return Ok((Vec::new(), CompactionMode::Local));
174 }
175 let options = resolve_manual_compaction_options(provider, model, options)?;
176 match manual_compaction_strategy(provider, model) {
177 CompactionStrategy::NativeStandalone => {
178 let responses_options: ResponsesCompactionOptions = options.clone().into();
179 // Bound the native input the same way as the local fork: a
180 // near-full history would otherwise exceed the summarizer window
181 // and fail the whole `/compact` command.
182 let native_source = bound_history_for_native_compaction(
183 history,
184 options.instructions.as_deref().unwrap_or(""),
185 compaction_history_budget(provider, model, context_budget),
186 );
187 let compacted = provider
188 .compact_history_with_options(model, &native_source, &responses_options)
189 .await
190 .context("Failed to compact history via provider-native compaction")?;
191 // A standalone compaction response is the provider's canonical
192 // next context window. Keep its retained items and opaque
193 // compaction items intact; the next provider request must receive
194 // this window as returned rather than a locally pruned variant.
195 Ok((compacted, CompactionMode::Provider))
196 }
197 CompactionStrategy::NativeInline => {
198 compact_history_native_inline(provider, model, history, config, &options, context_budget, parent).await
199 }
200 CompactionStrategy::Local => {
201 let compacted =
202 summarize_locally(provider, model, history, config, &options, context_budget, parent).await?;
203 Ok((compacted, CompactionMode::Local))
204 }
205 }
206}
207
208/// Resolve a manual compaction effort before selecting a provider strategy.
209/// Every strategy eventually emits an `LLMRequest` or provider compaction
210/// options, so validating once at this boundary prevents native, inline, and
211/// hierarchical paths from silently coercing unsupported levels.
212fn resolve_manual_compaction_options(
213 provider: &dyn LLMProvider,
214 model: &str,
215 options: &ManualCompactionOptions,
216) -> Result<ManualCompactionOptions> {
217 let Some(requested) = options.reasoning_effort else {
218 return Ok(options.clone());
219 };
220
221 let mapping = ReasoningEffortMapper::resolve(provider, model, requested, options.allow_reasoning_effort_downgrade)
222 .with_context(|| {
223 format!("Failed to resolve compaction reasoning effort for {} / {}", provider.name(), model)
224 })?;
225
226 if mapping.degraded() {
227 tracing::warn!(
228 provider = provider.name(),
229 model,
230 requested = %mapping.requested,
231 effective = %mapping.effective,
232 "Compaction reasoning effort explicitly downgraded"
233 );
234 }
235
236 let mut resolved = options.clone();
237 resolved.reasoning_effort = Some(mapping.effective);
238 Ok(resolved)
239}