use super::*;
pub(crate) fn compaction_summary_request(
model: &str,
history: &[Message],
instructions: &str,
max_output_tokens: Option<u32>,
reasoning_effort: Option<ReasoningEffortLevel>,
verbosity: Option<VerbosityLevel>,
supports_turn_scoped: bool,
parent: Option<&CompactionParentContext>,
) -> LLMRequest {
let mut forked = build_cache_safe_compaction_history(history, instructions);
if !supports_turn_scoped {
for message in forked.iter_mut() {
if message.clear_at.is_some() {
message.clear_at = None;
}
}
}
LLMRequest {
messages: Arc::new(forked),
model: model.to_string(),
system_prompt: parent.and_then(|parent| parent.system_prompt.clone()),
tools: parent.and_then(|parent| parent.tools.clone()).filter(|tools| !tools.is_empty()),
tool_choice: Some(ToolChoice::none()),
max_tokens: max_output_tokens,
reasoning_effort,
verbosity,
..Default::default()
}
}
#[cfg_attr(feature = "profiling", hotpath::measure)]
pub async fn compact_history(
provider: &dyn LLMProvider,
model: &str,
history: &[Message],
config: &CompactionConfig,
) -> Result<Vec<Message>> {
compact_history_with_budget(provider, model, history, config, None).await
}
pub async fn compact_history_with_budget(
provider: &dyn LLMProvider,
model: &str,
history: &[Message],
config: &CompactionConfig,
context_budget: Option<usize>,
) -> Result<Vec<Message>> {
if history.is_empty() {
return Ok(Vec::new());
}
if !config.always_summarize && history.len() <= config.keep_last_messages {
return Ok(bound_compacted_history_to_context(history.to_vec(), provider, model, context_budget));
}
if !config.always_summarize && provider.supports_manual_openai_compaction(model) {
let native_source = bound_history_for_native_compaction(
history,
"",
compaction_history_budget(provider, model, context_budget),
);
let compacted = provider
.compact_history(model, &native_source)
.await
.context("Failed to compact history via Responses compact endpoint")?;
return Ok(compacted);
}
let effective_config = context_bounded_compaction_config(provider, model, history, config, context_budget);
let history_budget = compaction_history_budget(provider, model, context_budget);
let tail_target = route_tail_target_tokens(provider, model, context_budget);
let pruned_history = prune_oversized_tool_outputs(history);
let summary_source =
bound_history_for_summarization(&pruned_history, &effective_config.summary_prompt, history_budget);
let summary = generate_local_summary_with_retry(
provider,
model,
&pruned_history,
&summary_source,
&effective_config.summary_prompt,
history_budget,
&ManualCompactionOptions::default(),
None,
)
.await?;
Ok(bound_compacted_history_to_context(
build_local_compacted_history(
history,
&summary,
effective_config.retained_user_message_tokens,
effective_config.retained_user_messages,
true,
tail_target,
),
provider,
model,
context_budget,
))
}
pub(crate) async fn summarize_locally(
provider: &dyn LLMProvider,
model: &str,
history: &[Message],
config: &CompactionConfig,
options: &ManualCompactionOptions,
context_budget: Option<usize>,
parent: Option<&CompactionParentContext>,
) -> Result<Vec<Message>> {
if config.hierarchical {
return summarize_locally_hierarchical(provider, model, history, config, options, context_budget, parent).await;
}
let effective_config = context_bounded_compaction_config(
provider,
model,
history,
&config.clone().with_manual_overrides(options),
context_budget,
);
let history_budget = compaction_history_budget(provider, model, context_budget);
let tail_target = route_tail_target_tokens(provider, model, context_budget);
let pruned_history = prune_oversized_tool_outputs(history);
let summary_source =
bound_history_for_summarization(&pruned_history, &effective_config.summary_prompt, history_budget);
let summary = generate_local_summary_with_retry(
provider,
model,
&pruned_history,
&summary_source,
&effective_config.summary_prompt,
history_budget,
options,
parent,
)
.await?;
Ok(bound_compacted_history_to_context(
build_summary_compacted_history(history, summary, &effective_config, true, tail_target),
provider,
model,
context_budget,
))
}
pub(crate) async fn generate_summary_with_capacity_retry(
provider: &dyn LLMProvider,
model: &str,
history: &[Message],
instructions: &str,
first_source: &[Message],
history_budget: Option<usize>,
max_overflow_retries: u32,
error_context: &'static str,
make_request: impl Fn(&[Message]) -> LLMRequest,
) -> Result<String> {
let first_tokens = first_source.iter().map(Message::estimate_tokens).sum::<usize>();
let mut current_source: &[Message] = first_source;
let mut current_tokens = first_tokens;
let mut current_budget = history_budget;
let mut remaining_retries = max_overflow_retries;
let mut retry_storage: Vec<Message>;
loop {
match collect_single_response(provider, make_request(current_source)).await {
Ok(response) => {
let summary = response.content.unwrap_or_default().trim().to_string();
if summary.is_empty() {
return Err(anyhow::anyhow!(
"{error_context} for {} / {}: provider returned an empty summary (input_tokens={current_tokens}, budget={current_budget:?})",
provider.name(),
model,
));
}
return Ok(summary);
}
Err(error) if is_context_capacity_message(&error.to_string()) && remaining_retries > 0 => {
current_budget = match current_budget {
Some(budget) => Some((budget / 2).max(4)),
None => Some((current_tokens / 2).max(4)),
};
retry_storage = bound_history_for_summarization(history, instructions, current_budget);
let retry_tokens = retry_storage.iter().map(Message::estimate_tokens).sum::<usize>();
if retry_tokens >= current_tokens {
return Err(anyhow::Error::from(error).context(format!(
"{error_context} for {} / {} (input_tokens={current_tokens}, budget={current_budget:?})",
provider.name(),
model,
)));
}
tracing::warn!(
provider = provider.name(),
model,
input_tokens = current_tokens,
retry_tokens,
budget = ?current_budget,
error = ?error,
"{error_context}: input exceeded context capacity; retrying with a halved input budget"
);
current_source = &retry_storage;
current_tokens = retry_tokens;
remaining_retries -= 1;
}
Err(error) => {
return Err(anyhow::Error::from(error).context(format!(
"{error_context} for {} / {} (input_tokens={current_tokens}, budget={current_budget:?})",
provider.name(),
model,
)));
}
}
}
}
async fn generate_local_summary_with_retry(
provider: &dyn LLMProvider,
model: &str,
history: &[Message],
summary_source: &[Message],
instructions: &str,
history_budget: Option<usize>,
options: &ManualCompactionOptions,
parent: Option<&CompactionParentContext>,
) -> Result<String> {
let supports_turn_scoped = provider.supports_turn_scoped_system_messages(model);
generate_summary_with_capacity_retry(
provider,
model,
history,
instructions,
summary_source,
history_budget,
CompactionRoutePolicy::resolve(provider.name(), model).max_overflow_retries,
"Failed to generate compaction summary",
|source| {
compaction_summary_request(
model,
source,
instructions,
options.max_output_tokens,
options.reasoning_effort,
options.verbosity,
supports_turn_scoped,
parent,
)
},
)
.await
}
const ABSTRACT_BAND_INTRO: &str =
"In 1-2 sentences, what was the overall goal and major progress in this portion of the conversation?\n\n";
fn band_summary_request(
model: &str,
prompt: String,
max_tokens: Option<u32>,
options: &ManualCompactionOptions,
parent: Option<&CompactionParentContext>,
) -> LLMRequest {
LLMRequest {
messages: Arc::new(vec![Message::user(prompt)]),
model: model.to_string(),
system_prompt: parent.and_then(|parent| parent.system_prompt.clone()),
tools: parent.and_then(|parent| parent.tools.clone()).filter(|tools| !tools.is_empty()),
tool_choice: Some(ToolChoice::none()),
max_tokens,
reasoning_effort: options.reasoning_effort,
verbosity: options.verbosity,
..Default::default()
}
}
async fn summarize_locally_hierarchical(
provider: &dyn LLMProvider,
model: &str,
history: &[Message],
config: &CompactionConfig,
options: &ManualCompactionOptions,
context_budget: Option<usize>,
parent: Option<&CompactionParentContext>,
) -> Result<Vec<Message>> {
let effective_config = context_bounded_compaction_config(
provider,
model,
history,
&config.clone().with_manual_overrides(options),
context_budget,
);
let (summary_history, _) =
split_continuity_history_with_target(history, route_tail_target_tokens(provider, model, context_budget));
let total = summary_history.len();
let band_size = total / 3;
let abstract_end = band_size;
let detail_end = band_size * 2;
let history_budget = compaction_history_budget(provider, model, context_budget);
let max_overflow_retries = CompactionRoutePolicy::resolve(provider.name(), model).max_overflow_retries;
let pruned_abstract = prune_oversized_tool_outputs(&summary_history[..abstract_end]);
let abstract_band = bound_history_for_summarization(&pruned_abstract, "", history_budget);
let abstract_summary = generate_summary_with_capacity_retry(
provider,
model,
&pruned_abstract,
"",
&abstract_band,
history_budget,
max_overflow_retries,
"Failed to generate abstract summary",
|band| {
band_summary_request(
model,
format!("{ABSTRACT_BAND_INTRO}{}", build_summary_prompt(band, "")),
Some(150),
options,
parent,
)
},
)
.await?;
let pruned_detail = prune_oversized_tool_outputs(&summary_history[abstract_end..detail_end]);
let detail_band = bound_history_for_summarization(&pruned_detail, &effective_config.summary_prompt, history_budget);
let detail_summary = generate_summary_with_capacity_retry(
provider,
model,
&pruned_detail,
&effective_config.summary_prompt,
&detail_band,
history_budget,
max_overflow_retries,
"Failed to generate detail summary",
|band| {
band_summary_request(
model,
build_summary_prompt(band, &effective_config.summary_prompt),
options.max_output_tokens,
options,
parent,
)
},
)
.await?;
let recent_band = &summary_history[detail_end..];
let retained = collect_retained_user_messages(
recent_band,
effective_config.retained_user_message_tokens,
effective_config.retained_user_messages,
);
let mut new_history = Vec::with_capacity(2 + retained.len());
new_history.push(Message::system(format!("{ABSTRACT_PREFIX}{abstract_summary}")));
new_history.push(Message::system(format!("{DETAIL_PREFIX}{detail_summary}")));
new_history.extend(retained);
for message in continuity_tail_with_target(history, route_tail_target_tokens(provider, model, context_budget)) {
new_history.push(message.clone());
}
Ok(bound_compacted_history_to_context(new_history, provider, model, context_budget))
}