Skip to main content

navi_core/
compact.rs

1use crate::config::HarnessConfig;
2use crate::model::{ModelMessage, ModelProvider, ModelRequest, ModelRole, ThinkingConfig};
3use anyhow::Result;
4use std::collections::HashSet;
5use std::time::{SystemTime, UNIX_EPOCH};
6
7/// Result of a successful conversation compaction.
8#[derive(Debug, Clone, PartialEq, Eq)]
9pub struct CompactOutcome {
10    /// Estimated tokens removed from the live context.
11    pub tokens_saved: u64,
12    /// Model-produced summary that replaces older turns.
13    pub summary: String,
14    /// How many non-system conversation messages were kept after the summary.
15    pub kept_recent_messages: usize,
16}
17
18/// Build the standard post-compact user message that carries the summary.
19pub fn compact_summary_user_message(summary: &str) -> ModelMessage {
20    ModelMessage::user(format!(
21        "Here is a summary of the conversation so far:\n\n{}",
22        summary
23    ))
24}
25
26const READ_ONLY_TOOLS: &[&str] = &[
27    "read_file",
28    "read",
29    "search",
30    "fs_browser",
31    "grep",
32    "list_dir",
33    "glob",
34    "tool_search",
35    "code",
36    "ast_search",
37    "symbol_goto",
38    "symbol_references",
39    "repo_explore",
40    "current_time",
41    "get_context_remaining",
42    "view_image",
43    "question",
44    "plan",
45];
46
47/// Removes read-only tool results from older messages when idle time exceeds
48/// the gap threshold. Returns the number of messages cleared.
49pub fn micro_compact(messages: &mut [ModelMessage], gap_threshold_minutes: u64) -> usize {
50    let now = current_unix_millis();
51    let gap_threshold_ms = gap_threshold_minutes * 60 * 1000;
52
53    let last_assistant_ts = messages
54        .iter()
55        .rev()
56        .find(|m| m.role == ModelRole::Assistant)
57        .and_then(|m| m.created_at);
58
59    let Some(last_ts) = last_assistant_ts else {
60        return 0;
61    };
62
63    if now.saturating_sub(last_ts) < gap_threshold_ms {
64        return 0;
65    }
66
67    let mut cleared = 0;
68    for msg in messages.iter_mut() {
69        if msg.role == ModelRole::Tool
70            && let Some(ref tool_name) = msg.tool_name
71            && READ_ONLY_TOOLS.contains(&tool_name.as_str())
72            && !msg.content.contains("[Old tool result content cleared]")
73        {
74            msg.content = "[Old tool result content cleared]".to_string();
75            // Free multimodal payload (e.g. view_image base64) along with text.
76            msg.content_parts.clear();
77            cleared += 1;
78        }
79    }
80    cleared
81}
82
83pub const AUTOCOMPACT_BUFFER_TOKENS: u64 = 13_000;
84pub const WARNING_THRESHOLD_BUFFER_TOKENS: u64 = 20_000;
85pub const ERROR_THRESHOLD_BUFFER_TOKENS: u64 = 20_000;
86pub const MAX_OUTPUT_TOKENS_FOR_SUMMARY: u64 = 20_000;
87pub const MAX_CONSECUTIVE_FAILURES: u32 = 3;
88/// auto-compact when context usage reaches this percent of the window.
89pub const AUTO_COMPACT_THRESHOLD_PERCENT: u8 = 80;
90
91/// Context usage severity level used to trigger compact warnings and errors.
92#[derive(Debug, Clone, Copy, PartialEq, Eq)]
93pub enum CompactThreshold {
94    /// Context usage is within normal bounds.
95    Normal,
96    /// Context usage is approaching the limit; a warning should be shown.
97    Warning,
98    /// Context usage is critically close to the limit.
99    Error,
100    /// Compact has failed too many times; further attempts are blocked.
101    CircuitOpen,
102}
103
104impl std::fmt::Display for CompactThreshold {
105    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
106        match self {
107            CompactThreshold::Normal => write!(f, "ok"),
108            CompactThreshold::Warning => write!(f, "warning"),
109            CompactThreshold::Error => write!(f, "error"),
110            CompactThreshold::CircuitOpen => write!(f, "circuit-open"),
111        }
112    }
113}
114
115/// Tracks token usage and compact failure state for autocompact decisions.
116///
117/// Measurement model (aligned with modern coding CLI TUI):
118/// - **context tokens used** = last API `input_tokens` (ground truth for the
119/// current context) + client preflight for unsent bytes (`bytes/4`)
120/// - **window** = model `context_window` from registry
121/// - **usage %** = used / window
122/// - **total before compaction** = cumulative input tokens across turns
123/// (historical; does not reset when the bar drops after compact)
124#[derive(Debug, Clone)]
125pub struct CompactState {
126    /// Token count from the last model response usage (current context size).
127    pub last_input_tokens: Option<u64>,
128    /// Output tokens from the last model response (UI turn label).
129    pub last_output_tokens: Option<u64>,
130    /// Estimated bytes of new messages not yet sent to the model.
131    pub estimated_unsent_bytes: usize,
132    /// Context window size in tokens for the current model.
133    pub context_window: u64,
134    /// Cumulative input tokens processed before/during compaction cycles.
135    pub total_tokens_before_compaction: u64,
136    /// Number of compaction runs in this session.
137    pub compaction_count: u32,
138    /// Auto-compact when `context_percentage >= this` (default: 85).
139    pub auto_compact_threshold_percent: u8,
140    /// Number of consecutive compact failures.
141    pub consecutive_failures: u32,
142    pub summary: Option<String>,
143    pub summary_message_count: usize,
144    /// Latest long-horizon rebuild context that must stay attached to the
145    /// system prompt across subsequent turns.
146    pub rebuild_context: Option<String>,
147    /// List of checkpoint thresholds crossed in the current context cycle.
148    pub crossed_thresholds: Vec<f64>,
149    /// Fingerprints of messages already copied into long-horizon history.
150    pub history_synced_message_keys: HashSet<u64>,
151}
152
153impl Default for CompactState {
154    fn default() -> Self {
155        Self {
156            last_input_tokens: None,
157            last_output_tokens: None,
158            estimated_unsent_bytes: 0,
159            context_window: 0,
160            total_tokens_before_compaction: 0,
161            compaction_count: 0,
162            auto_compact_threshold_percent: AUTO_COMPACT_THRESHOLD_PERCENT,
163            consecutive_failures: 0,
164            summary: None,
165            summary_message_count: 0,
166            rebuild_context: None,
167            crossed_thresholds: Vec::new(),
168            history_synced_message_keys: HashSet::new(),
169        }
170    }
171}
172
173fn format_token_short(t: u64) -> String {
174    if t >= 1_000_000 {
175        let m = t as f64 / 1_000_000.0;
176        // Prefer `1M` over `1.0M` when close to a whole million.
177        if (m - m.round()).abs() < 0.05 {
178            format!("{}M", m.round() as u64)
179        } else {
180            format!("{:.1}M", m)
181        }
182    } else if t >= 10_000 {
183        format!("{}k", t / 1_000)
184    } else if t >= 1_000 {
185        // Keep one decimal under 10k so 1.5k doesn't collapse to `1k`.
186        let k = t as f64 / 1_000.0;
187        if (k - k.floor()).abs() < 0.05 {
188            format!("{}k", k.floor() as u64)
189        } else {
190            format!("{:.1}k", k)
191        }
192    } else {
193        t.to_string()
194    }
195}
196
197/// Effective prompt tokens for the context-window meter.
198///
199/// Handles providers that split non-cached vs cached prompt tokens (Anthropic
200/// and some OpenAI-compat aggregators). Without this, a session can show
201/// `430 / 1M` while the real fill is ~64k.
202pub fn context_tokens_for_meter(
203    input_tokens: Option<u64>,
204    cache_creation_tokens: u64,
205    cache_read_tokens: u64,
206) -> Option<u64> {
207    let input = input_tokens.unwrap_or(0);
208    if input == 0 && cache_creation_tokens == 0 && cache_read_tokens == 0 {
209        return None;
210    }
211    // Inclusive (OpenAI): prompt_tokens already includes cached_tokens.
212    if cache_read_tokens > 0 && input >= cache_read_tokens {
213        return Some(input.saturating_add(cache_creation_tokens));
214    }
215    // Exclusive: non-cached input + cache create + cache read.
216    Some(
217        input
218            .saturating_add(cache_creation_tokens)
219            .saturating_add(cache_read_tokens),
220    )
221}
222
223impl CompactState {
224    pub fn new(context_window: u64) -> Self {
225        Self {
226            context_window,
227            ..Default::default()
228        }
229    }
230
231    pub fn add_unsent_bytes(&mut self, bytes: usize) {
232        self.estimated_unsent_bytes += bytes;
233    }
234
235    pub fn clear_unsent_bytes(&mut self) {
236        self.estimated_unsent_bytes = 0;
237    }
238
239    /// Current context size (live context usage).
240    pub fn context_tokens_used(&self, pending_input_bytes: usize) -> u64 {
241        self.total_estimated_tokens(pending_input_bytes)
242    }
243
244    /// Context window size (context window size).
245    pub fn context_window_tokens(&self) -> u64 {
246        self.context_window
247    }
248
249    /// Integer percent of the window in use (context window usage percent).
250    pub fn context_window_usage(&self, pending_input_bytes: usize) -> u8 {
251        self.context_percentage(pending_input_bytes)
252    }
253
254    /// Estimate tokens for the next request:
255    /// API last-input (server) + unsent/pending client bytes as `ceil(bytes/4)`.
256    pub fn total_estimated_tokens(&self, pending_input_bytes: usize) -> u64 {
257        let server_tokens = self.last_input_tokens.unwrap_or(0);
258        let client_bytes = self.estimated_unsent_bytes + pending_input_bytes;
259        let client_tokens = (client_bytes.saturating_add(3) / 4) as u64;
260        server_tokens + client_tokens
261    }
262
263    pub fn threshold_level(&self, pending_input_bytes: usize) -> CompactThreshold {
264        if self.consecutive_failures >= MAX_CONSECUTIVE_FAILURES {
265            return CompactThreshold::CircuitOpen;
266        }
267        let total_tokens = self.total_estimated_tokens(pending_input_bytes);
268        if total_tokens == 0 {
269            return CompactThreshold::Normal;
270        }
271        let remaining = self.context_window.saturating_sub(total_tokens);
272        if remaining <= ERROR_THRESHOLD_BUFFER_TOKENS {
273            CompactThreshold::Error
274        } else if remaining <= WARNING_THRESHOLD_BUFFER_TOKENS + AUTOCOMPACT_BUFFER_TOKENS {
275            CompactThreshold::Warning
276        } else {
277            CompactThreshold::Normal
278        }
279    }
280
281    pub fn should_autocompact(&self, buffer_tokens: u64) -> bool {
282        if self.consecutive_failures >= MAX_CONSECUTIVE_FAILURES {
283            return false;
284        }
285        if self.context_window == 0 {
286            return false;
287        }
288        // compact near 80% of the window (includes unsent preflight).
289        let used = self.total_estimated_tokens(0);
290        let pct = (used as f64 / self.context_window as f64) * 100.0;
291        if pct >= f64::from(self.auto_compact_threshold_percent) {
292            return true;
293        }
294        // Hard ceiling: last API input + reserved buffer would fill the window.
295        let Some(input_tokens) = self.last_input_tokens else {
296            return false;
297        };
298        input_tokens.saturating_add(buffer_tokens) >= self.context_window
299    }
300
301    pub fn context_percentage(&self, pending_input_bytes: usize) -> u8 {
302        if self.context_window == 0 {
303            return 0;
304        }
305        let total_tokens = self.total_estimated_tokens(pending_input_bytes);
306        let percentage = (total_tokens as f64 / self.context_window as f64) * 100.0;
307        percentage.clamp(0.0, 100.0) as u8
308    }
309
310    /// Compact context meter for the composer footer: `3.2k / 200k`.
311    /// Percentage is revealed on hover (see `usage_label_with_percent`).
312    pub fn usage_label(&self, pending_input_bytes: usize) -> String {
313        self.usage_label_compact(pending_input_bytes)
314    }
315
316    /// Token counts only — default (non-hover) display.
317    pub fn usage_label_compact(&self, pending_input_bytes: usize) -> String {
318        let total_tokens = self.total_estimated_tokens(pending_input_bytes);
319        format!(
320            "{} / {}",
321            format_token_short(total_tokens),
322            format_token_short(self.context_window),
323        )
324    }
325
326    /// Token counts + percent — shown while the context chip is hovered.
327    pub fn usage_label_with_percent(&self, pending_input_bytes: usize) -> String {
328        let pct = self.context_percentage(pending_input_bytes);
329        format!(
330            "{} ({}%)",
331            self.usage_label_compact(pending_input_bytes),
332            pct
333        )
334    }
335
336    /// Record provider usage for this turn (called every stream that reports usage).
337    pub fn update_usage(&mut self, input_tokens: u64) {
338        self.update_usage_full(input_tokens, 0);
339    }
340
341    /// Record full turn usage and refresh live context metrics.
342    pub fn update_usage_full(&mut self, input_tokens: u64, output_tokens: u64) {
343        self.last_input_tokens = Some(input_tokens);
344        self.last_output_tokens = Some(output_tokens);
345        self.total_tokens_before_compaction = self
346            .total_tokens_before_compaction
347            .saturating_add(input_tokens);
348        self.clear_unsent_bytes();
349    }
350
351    /// Label shown in the TUI footer — updates every turn after `update_usage*`.
352    pub fn turn_usage_label(&self) -> Option<String> {
353        let input = self.last_input_tokens?;
354        let output = self.last_output_tokens.unwrap_or(0);
355        Some(format!(
356            "{}→{}",
357            format_token_short(input),
358            format_token_short(output)
359        ))
360    }
361
362    pub async fn auto_compact(
363        &mut self,
364        messages: &mut Vec<ModelMessage>,
365        model_provider: &dyn ModelProvider,
366        model_name: &str,
367        harness_config: &HarnessConfig,
368    ) -> Result<Option<CompactOutcome>> {
369        self.auto_compact_inner(messages, model_provider, model_name, harness_config, false)
370            .await
371    }
372
373    /// Force compaction with the session model even when below the threshold.
374    /// Manual Compact always fully replaces conversation history (keep_ratio=0).
375    pub async fn force_compact(
376        &mut self,
377        messages: &mut Vec<ModelMessage>,
378        model_provider: &dyn ModelProvider,
379        model_name: &str,
380        harness_config: &HarnessConfig,
381    ) -> Result<Option<CompactOutcome>> {
382        self.auto_compact_inner(messages, model_provider, model_name, harness_config, true)
383            .await
384    }
385
386    async fn auto_compact_inner(
387        &mut self,
388        messages: &mut Vec<ModelMessage>,
389        model_provider: &dyn ModelProvider,
390        model_name: &str,
391        harness_config: &HarnessConfig,
392        force: bool,
393    ) -> Result<Option<CompactOutcome>> {
394        if !force && !self.should_autocompact(harness_config.autocompact_buffer_tokens) {
395            return Ok(None);
396        }
397
398        // Split: system + developer messages first (the prompt prefix),
399        // then conversation messages.
400        let system_msgs: Vec<ModelMessage> = messages
401            .iter()
402            .filter(|m| m.role == ModelRole::System || m.role == ModelRole::Developer)
403            .cloned()
404            .collect();
405        let conversation_msgs: Vec<ModelMessage> = messages
406            .iter()
407            .filter(|m| m.role != ModelRole::System && m.role != ModelRole::Developer)
408            .cloned()
409            .collect();
410
411        if conversation_msgs.is_empty() {
412            return Ok(None);
413        }
414
415        // KeepRatio: keep the last N% of conversation turns intact.
416        // Forced/manual compact always fully replaces so context actually shrinks.
417        let keep_ratio = if force {
418            0.0
419        } else {
420            harness_config.autocompact_keep_ratio.clamp(0.0, 0.9)
421        };
422        let total = conversation_msgs.len();
423        let keep_count = if force || total < 2 {
424            0
425        } else {
426            let keep_count = (total as f64 * keep_ratio).round() as usize;
427            // Always keep at least 2 messages (1 user + 1 assistant) and at most
428            // total - 2 (so there's something to summarize).
429            keep_count.clamp(2.min(total), total.saturating_sub(2).max(2.min(total)))
430        };
431        let split_at = total.saturating_sub(keep_count);
432
433        // Old messages → summarize. Recent messages → keep intact.
434        let (old_msgs, recent_msgs) = conversation_msgs.split_at(split_at);
435        let old_text = build_conversation_text(old_msgs);
436
437        if old_text.trim().is_empty() {
438            // Nothing old to summarize; just keep everything.
439            return Ok(None);
440        }
441
442        let prompt = if let Some(ref prev_summary) = self.summary {
443            PARTIAL_COMPACT_PROMPT
444                .replace("{previous_summary}", prev_summary)
445                .replace("{new_conversation}", &old_text)
446        } else {
447            format!(
448                "{}\n\nConversation to summarize:\n{}",
449                COMPACT_PROMPT, old_text
450            )
451        };
452
453        let request = ModelRequest {
454            model: model_name.to_string(),
455            instructions: None,
456            messages: vec![
457                ModelMessage::system("You are a precise conversation summarizer."),
458                ModelMessage::user(prompt),
459            ],
460            thinking: ThinkingConfig::Off,
461            tools: vec![],
462            session_id: None,
463        };
464
465        match model_provider.complete(request).await {
466            Ok(response) => {
467                let summary = response.text.trim().to_string();
468                if summary.is_empty() {
469                    self.consecutive_failures += 1;
470                    anyhow::bail!("compaction model returned an empty summary");
471                }
472                let previous_tokens = self.last_input_tokens.unwrap_or_else(|| {
473                    messages
474                        .iter()
475                        .map(|message| message.content.len() as u64)
476                        .sum::<u64>()
477                        .saturating_add(3)
478                        / 4
479                });
480                let kept_recent_messages = recent_msgs.len();
481
482                // Reassemble: system + summary + recent turns kept intact.
483                messages.clear();
484                messages.extend(system_msgs);
485                messages.push(compact_summary_user_message(&summary));
486                messages.extend(recent_msgs.iter().cloned());
487
488                self.summary = Some(summary.clone());
489                self.summary_message_count = messages.len();
490                self.consecutive_failures = 0;
491                self.last_input_tokens = None;
492                self.last_output_tokens = None;
493                self.clear_unsent_bytes();
494                self.compaction_count = self.compaction_count.saturating_add(1);
495
496                let tokens_saved = previous_tokens
497                    .saturating_sub(estimate_messages_tokens(messages))
498                    .max(1);
499                tracing::info!(
500                    tokens_saved,
501                    old_turns = old_msgs.len(),
502                    kept_turns = kept_recent_messages,
503                    force,
504                    "auto-compact completed"
505                );
506
507                Ok(Some(CompactOutcome {
508                    tokens_saved,
509                    summary,
510                    kept_recent_messages,
511                }))
512            }
513            Err(e) => {
514                self.consecutive_failures += 1;
515                tracing::warn!(
516                    failures = self.consecutive_failures,
517                    error = %e,
518                    "auto-compact failed"
519                );
520                Err(e)
521            }
522        }
523    }
524
525    pub fn apply_manual_summary(
526        &mut self,
527        messages: &mut Vec<ModelMessage>,
528        summary: String,
529    ) -> CompactOutcome {
530        let system_msgs: Vec<ModelMessage> = messages
531            .iter()
532            .filter(|m| m.role == ModelRole::System || m.role == ModelRole::Developer)
533            .cloned()
534            .collect();
535        let estimated_previous_tokens = self.last_input_tokens.unwrap_or_else(|| {
536            messages
537                .iter()
538                .map(|message| message.content.len() as u64)
539                .sum::<u64>()
540                .saturating_add(3)
541                / 4
542        });
543
544        messages.clear();
545        messages.extend(system_msgs);
546        messages.push(compact_summary_user_message(&summary));
547
548        self.summary = Some(summary.clone());
549        self.summary_message_count = messages.len();
550        self.consecutive_failures = 0;
551        self.last_input_tokens = None;
552        self.last_output_tokens = None;
553        self.compaction_count = self.compaction_count.saturating_add(1);
554        self.clear_unsent_bytes();
555
556        let tokens_saved = estimated_previous_tokens
557            .saturating_sub(estimate_messages_tokens(messages))
558            .max(1);
559
560        CompactOutcome {
561            tokens_saved,
562            summary,
563            kept_recent_messages: 0,
564        }
565    }
566}
567
568fn estimate_messages_tokens(messages: &[ModelMessage]) -> u64 {
569    messages
570        .iter()
571        .map(|message| message.content.len() as u64)
572        .sum::<u64>()
573        .saturating_add(3)
574        / 4
575}
576
577fn build_conversation_text(messages: &[ModelMessage]) -> String {
578    let mut text = String::new();
579    for msg in messages {
580        let role_label = match msg.role {
581            ModelRole::User => "User",
582            ModelRole::Assistant => "Assistant",
583            ModelRole::Tool => "Tool",
584            ModelRole::System | ModelRole::Developer => continue,
585        };
586        if msg.role == ModelRole::Tool {
587            if let Some(ref tool_name) = msg.tool_name {
588                text.push_str(&format!("[Tool({})]: {}\n", tool_name, msg.content));
589            } else {
590                text.push_str(&format!("[Tool]: {}\n", msg.content));
591            }
592        } else {
593            let image_note = if msg.content_parts.iter().any(|p| p.is_image()) {
594                let count = msg.content_parts.iter().filter(|p| p.is_image()).count();
595                format!(" [{} image(s) attached]", count)
596            } else {
597                String::new()
598            };
599            text.push_str(&format!(
600                "[{}]: {}{}\n",
601                role_label, msg.content, image_note
602            ));
603        }
604    }
605    text
606}
607
608pub const COMPACT_PROMPT: &str = r#"You are summarizing a conversation between a user and an AI coding assistant (NAVI). Create a detailed summary with these exact sections:
609
610## 1. Primary Request and Intent
611## 2. Key Technical Concepts
612## 3. Files and Code Snippets
613## 4. Errors and Fixes
614## 5. Problem Resolution
615## 6. All User Messages
616## 7. Pending Tasks
617## 8. Current Work
618## 9. Active Work Plan
619If the conversation has an active plan (via the plan tool) or an in-progress Plan-mode proposal, include:
620- Plan ID and title (if any)
621- All steps with completion status
622- Which step to work on next
623If there is no active plan, skip this section.
624Also note any active thread goal (create_goal/update_goal) separately if present — do not conflate plan and goal.
625## 10. Next Step (Optional)
626List the next step you would take on the current task.
627
628Be thorough and specific. The summary must contain enough detail to continue the conversation seamlessly.
629
630IMPORTANT: If there is an active plan that is not completed or abandoned, continue it after reading this summary UNLESS the user clearly redirected to a different task — in that case, note the redirect in section 8/10 and do not restart the old plan. Do not create a new plan unless the prior plan is completed, abandoned, or the user asked for a new one."#;
631
632pub const PARTIAL_COMPACT_PROMPT: &str = r#"You are extending an existing conversation summary with new content. Preserve the existing summary sections and update them with new information. Add any new user messages to section 6. Update sections 8 and 9 based on the most recent work.
633
634Existing summary:
635{previous_summary}
636
637New conversation to summarize:
638{new_conversation}
639
640Return the complete updated summary with all 10 sections (including Active Work Plan if applicable).
641
642IMPORTANT: If there is an active plan, preserve plan details and step completion status. Continue that plan only if the user has not redirected; note any redirect instead of forcing the old plan."#;
643
644fn current_unix_millis() -> u64 {
645    SystemTime::now()
646        .duration_since(UNIX_EPOCH)
647        .unwrap_or_default()
648        .as_millis() as u64
649}
650
651#[cfg(test)]
652mod tests {
653    use super::*;
654    use crate::model::ModelMessage;
655
656    #[test]
657    fn micro_compact_clears_read_only_tools_after_gap() {
658        let now = current_unix_millis();
659        let gap_ms: u64 = 61 * 60 * 1000;
660
661        let mut messages = vec![
662            ModelMessage::system("system"),
663            ModelMessage::user("task"),
664            {
665                let mut m = ModelMessage::assistant("response");
666                m.created_at = Some(now.saturating_sub(gap_ms));
667                m
668            },
669            ModelMessage::tool_result("call-1", "read_file", "file content here".to_string()),
670            ModelMessage::tool_result("call-2", "write_file", "written content".to_string()),
671            ModelMessage::tool_result("call-3", "grep", "match results".to_string()),
672            ModelMessage::tool_result("call-5", "bash", "command output".to_string()),
673        ];
674
675        let cleared = micro_compact(&mut messages, 60);
676        assert_eq!(cleared, 2);
677        assert!(
678            messages[3]
679                .content
680                .contains("[Old tool result content cleared]")
681        );
682        assert_eq!(messages[4].content, "written content");
683        assert!(
684            messages[5]
685                .content
686                .contains("[Old tool result content cleared]")
687        );
688        assert_eq!(messages[6].content, "command output");
689    }
690
691    #[test]
692    fn micro_compact_no_gap_returns_zero() {
693        let mut messages = vec![
694            ModelMessage::system("system"),
695            ModelMessage::user("task"),
696            ModelMessage::assistant("response"),
697            ModelMessage::tool_result("call-1", "read_file", "content".to_string()),
698        ];
699
700        let cleared = micro_compact(&mut messages, 60);
701        assert_eq!(cleared, 0);
702    }
703
704    #[test]
705    fn micro_compact_no_double_clear() {
706        let now = current_unix_millis();
707        let gap_ms: u64 = 61 * 60 * 1000;
708
709        let mut messages = vec![
710            ModelMessage::system("system"),
711            {
712                let mut m = ModelMessage::assistant("response");
713                m.created_at = Some(now.saturating_sub(gap_ms));
714                m
715            },
716            ModelMessage::tool_result(
717                "call-1",
718                "read_file",
719                "[Old tool result content cleared]".to_string(),
720            ),
721        ];
722
723        let cleared = micro_compact(&mut messages, 60);
724        assert_eq!(cleared, 0);
725    }
726
727    #[test]
728    fn compact_state_threshold_normal() {
729        let state = CompactState {
730            last_input_tokens: Some(50_000),
731            context_window: 200_000,
732            ..Default::default()
733        };
734        assert_eq!(state.threshold_level(0), CompactThreshold::Normal);
735    }
736
737    #[test]
738    fn compact_state_threshold_warning() {
739        let state = CompactState {
740            last_input_tokens: Some(170_000),
741            context_window: 200_000,
742            ..Default::default()
743        };
744        assert_eq!(state.threshold_level(0), CompactThreshold::Warning);
745    }
746
747    #[test]
748    fn compact_state_threshold_error() {
749        let state = CompactState {
750            last_input_tokens: Some(181_000),
751            context_window: 200_000,
752            ..Default::default()
753        };
754        assert_eq!(state.threshold_level(0), CompactThreshold::Error);
755    }
756
757    #[test]
758    fn compact_state_circuit_breaker() {
759        let state = CompactState {
760            last_input_tokens: Some(50_000),
761            context_window: 200_000,
762            consecutive_failures: 3,
763            ..Default::default()
764        };
765        assert_eq!(state.threshold_level(0), CompactThreshold::CircuitOpen);
766        assert!(!state.should_autocompact(AUTOCOMPACT_BUFFER_TOKENS));
767    }
768
769    #[test]
770    fn compact_state_should_autocompact() {
771        let state = CompactState {
772            last_input_tokens: Some(190_000),
773            context_window: 200_000,
774            ..Default::default()
775        };
776        assert!(state.should_autocompact(AUTOCOMPACT_BUFFER_TOKENS));
777    }
778
779    #[test]
780    fn compact_state_autocompact_at_eighty_percent() {
781        // 160k / 200k = 80% triggers even when buffer would not.
782        let state = CompactState {
783            last_input_tokens: Some(160_000),
784            context_window: 200_000,
785            auto_compact_threshold_percent: 80,
786            ..Default::default()
787        };
788        assert!(state.should_autocompact(0));
789        assert_eq!(state.context_window_usage(0), 80);
790
791        // Just under threshold does not fire without the hard buffer ceiling.
792        let below = CompactState {
793            last_input_tokens: Some(159_000),
794            context_window: 200_000,
795            auto_compact_threshold_percent: 80,
796            ..Default::default()
797        };
798        assert!(!below.should_autocompact(0));
799    }
800
801    #[test]
802    fn compact_state_update_usage_full_tracks_cumulative() {
803        let mut state = CompactState::new(200_000);
804        state.update_usage_full(10_000, 500);
805        state.update_usage_full(12_000, 800);
806        assert_eq!(state.last_input_tokens, Some(12_000));
807        assert_eq!(state.last_output_tokens, Some(800));
808        assert_eq!(state.total_tokens_before_compaction, 22_000);
809        assert_eq!(state.turn_usage_label().as_deref(), Some("12k→800"));
810    }
811
812    #[test]
813    fn compact_state_manual_summary_compacts_below_threshold() {
814        let mut state = CompactState {
815            last_input_tokens: Some(10_000),
816            context_window: 200_000,
817            consecutive_failures: 2,
818            estimated_unsent_bytes: 4096,
819            ..Default::default()
820        };
821        assert!(!state.should_autocompact(AUTOCOMPACT_BUFFER_TOKENS));
822        let mut messages = vec![
823            ModelMessage::system("system"),
824            ModelMessage::user("task"),
825            ModelMessage::assistant("response"),
826        ];
827
828        let outcome = state.apply_manual_summary(&mut messages, "Manual summary".to_string());
829
830        assert!(outcome.tokens_saved >= 1);
831        assert_eq!(outcome.kept_recent_messages, 0);
832        assert_eq!(outcome.summary, "Manual summary");
833        assert_eq!(messages.len(), 2);
834        assert_eq!(messages[0].role, ModelRole::System);
835        assert!(messages[1].content.contains("Manual summary"));
836        assert_eq!(state.summary.as_deref(), Some("Manual summary"));
837        assert_eq!(state.summary_message_count, 2);
838        assert_eq!(state.consecutive_failures, 0);
839        assert_eq!(state.estimated_unsent_bytes, 0);
840        assert!(state.last_input_tokens.is_none());
841    }
842
843    #[test]
844    fn compact_state_context_percentage() {
845        let state = CompactState {
846            last_input_tokens: Some(100_000),
847            context_window: 200_000,
848            ..Default::default()
849        };
850        assert_eq!(state.context_percentage(0), 50);
851    }
852
853    #[test]
854    fn compact_state_no_usage_returns_zero_percent() {
855        let state = CompactState {
856            last_input_tokens: None,
857            context_window: 200_000,
858            ..Default::default()
859        };
860        assert_eq!(state.context_percentage(0), 0);
861    }
862
863    #[test]
864    fn compact_state_usage_label_shows_real_context_usage() {
865        let state = CompactState {
866            last_input_tokens: Some(2_000),
867            context_window: 128_000,
868            ..Default::default()
869        };
870        // Default (composer) is counts only; percent is a hover-only affordance.
871        assert_eq!(state.usage_label(0), "2k / 128k");
872        assert_eq!(state.usage_label_with_percent(0), "2k / 128k (1%)");
873    }
874
875    #[test]
876    fn context_tokens_for_meter_sums_exclusive_cache() {
877        // Charm-style undercount: tiny non-cached prompt + large cache hit.
878        assert_eq!(context_tokens_for_meter(Some(430), 0, 63_570), Some(64_000));
879    }
880
881    #[test]
882    fn context_tokens_for_meter_keeps_openai_inclusive() {
883        // OpenAI: prompt_tokens already includes cached_tokens.
884        assert_eq!(
885            context_tokens_for_meter(Some(64_000), 0, 63_570),
886            Some(64_000)
887        );
888    }
889
890    #[test]
891    fn format_token_short_million_window() {
892        assert_eq!(format_token_short(1_048_576), "1M");
893        assert_eq!(format_token_short(64_000), "64k");
894    }
895
896    #[test]
897    fn write_tool_preserved_in_micro_compact() {
898        let now = current_unix_millis();
899        let gap_ms: u64 = 61 * 60 * 1000;
900
901        let mut messages = vec![
902            ModelMessage::system("system"),
903            {
904                let mut m = ModelMessage::assistant("response");
905                m.created_at = Some(now.saturating_sub(gap_ms));
906                m
907            },
908            ModelMessage::tool_result("call-1", "write_file", "content written".to_string()),
909            ModelMessage::tool_result("call-2", "apply_patch", "patch applied".to_string()),
910        ];
911
912        let cleared = micro_compact(&mut messages, 60);
913        assert_eq!(cleared, 0);
914        assert_eq!(messages[2].content, "content written");
915        assert_eq!(messages[3].content, "patch applied");
916    }
917
918    // ── Regression tests ──────────────────────────────────────────────────────
919
920    #[test]
921    fn regression_micro_compact_no_assistant_messages_returns_zero() {
922        let mut messages = vec![
923            ModelMessage::system("system"),
924            ModelMessage::user("hello"),
925            ModelMessage::tool_result("c1", "read_file", "content"),
926        ];
927        let cleared = micro_compact(&mut messages, 60);
928        assert_eq!(cleared, 0);
929    }
930
931    #[test]
932    fn regression_micro_compact_preserves_non_readonly_tools() {
933        let now = current_unix_millis();
934        let gap_ms: u64 = 61 * 60 * 1000;
935
936        let mut messages = vec![
937            ModelMessage::system("system"),
938            {
939                let mut m = ModelMessage::assistant("response");
940                m.created_at = Some(now.saturating_sub(gap_ms));
941                m
942            },
943            ModelMessage::tool_result("c1", "write_file", "file written"),
944            ModelMessage::tool_result("c2", "package_manager", "deps ok"),
945            ModelMessage::tool_result("c3", "apply_patch", "patch applied"),
946            ModelMessage::tool_result("c4", "bash", "command ok"),
947        ];
948
949        let cleared = micro_compact(&mut messages, 60);
950        assert_eq!(cleared, 0, "non-read-only tools must not be cleared");
951        assert_eq!(messages[2].content, "file written");
952        assert_eq!(messages[3].content, "deps ok");
953        assert_eq!(messages[4].content, "patch applied");
954        assert_eq!(messages[5].content, "command ok");
955    }
956
957    #[test]
958    fn regression_compact_state_percentage_clamps_to_100() {
959        let mut state = CompactState::new(1000);
960        // Simulate more tokens than window
961        state.last_input_tokens = Some(2000);
962        let pct = state.context_percentage(0);
963        assert!(pct <= 100, "percentage must clamp to 100, got {pct}");
964    }
965
966    #[test]
967    fn regression_compact_state_usage_label_formats_millions() {
968        let state = CompactState {
969            last_input_tokens: Some(1_500_000),
970            context_window: 2_000_000,
971            ..Default::default()
972        };
973        let label = state.usage_label(0);
974        assert!(label.contains("M"), "should use M format for millions");
975    }
976
977    #[test]
978    fn regression_build_conversation_text_excludes_system() {
979        let messages = vec![
980            ModelMessage::system("you are a helpful assistant"),
981            ModelMessage::user("hello"),
982            ModelMessage::assistant("hi there"),
983        ];
984        let text = build_conversation_text(&messages);
985        assert!(
986            !text.contains("you are a helpful assistant"),
987            "system message must be excluded"
988        );
989        assert!(text.contains("hello"));
990        assert!(text.contains("hi there"));
991    }
992
993    #[test]
994    fn regression_build_conversation_text_includes_tool_name() {
995        let messages = vec![
996            ModelMessage::user("read file"),
997            ModelMessage::tool_result("c1", "read_file", "file content"),
998        ];
999        let text = build_conversation_text(&messages);
1000        assert!(
1001            text.contains("read_file"),
1002            "tool name must be included in conversation text"
1003        );
1004    }
1005
1006    #[test]
1007    fn keep_ratio_clamps_valid_range() {
1008        let config = HarnessConfig {
1009            autocompact_keep_ratio: 0.25,
1010            ..Default::default()
1011        };
1012        assert_eq!(config.autocompact_keep_ratio, 0.25);
1013    }
1014
1015    #[test]
1016    fn keep_ratio_default_is_25_percent() {
1017        let config = HarnessConfig::default();
1018        assert_eq!(config.autocompact_keep_ratio, 0.25);
1019    }
1020
1021    #[test]
1022    fn build_conversation_text_preserves_order() {
1023        let messages = vec![
1024            ModelMessage::user("first"),
1025            ModelMessage::assistant("second"),
1026            ModelMessage::user("third"),
1027            ModelMessage::assistant("fourth"),
1028        ];
1029        let text = build_conversation_text(&messages);
1030        let first_pos = text.find("first").unwrap();
1031        let second_pos = text.find("second").unwrap();
1032        let third_pos = text.find("third").unwrap();
1033        let fourth_pos = text.find("fourth").unwrap();
1034        assert!(first_pos < second_pos);
1035        assert!(second_pos < third_pos);
1036        assert!(third_pos < fourth_pos);
1037    }
1038}