Skip to main content

navi_core/
compact.rs

1use crate::config::HarnessConfig;
2use crate::model::{ModelMessage, ModelProvider, ModelRequest, ModelRole, ThinkingConfig};
3use anyhow::Result;
4use std::collections::HashSet;
5use std::time::{SystemTime, UNIX_EPOCH};
6
7const READ_ONLY_TOOLS: &[&str] = &[
8    "read_file",
9    "read",
10    "search",
11    "fs_browser",
12    "grep",
13    "list_dir",
14    "glob",
15    "tool_search",
16    "code",
17    "ast_search",
18    "symbol_goto",
19    "symbol_references",
20    "repo_explore",
21    "current_time",
22    "get_context_remaining",
23    "view_image",
24    "question",
25    "plan",
26];
27
28/// Removes read-only tool results from older messages when idle time exceeds
29/// the gap threshold. Returns the number of messages cleared.
30pub fn micro_compact(messages: &mut [ModelMessage], gap_threshold_minutes: u64) -> usize {
31    let now = current_unix_millis();
32    let gap_threshold_ms = gap_threshold_minutes * 60 * 1000;
33
34    let last_assistant_ts = messages
35        .iter()
36        .rev()
37        .find(|m| m.role == ModelRole::Assistant)
38        .and_then(|m| m.created_at);
39
40    let Some(last_ts) = last_assistant_ts else {
41        return 0;
42    };
43
44    if now.saturating_sub(last_ts) < gap_threshold_ms {
45        return 0;
46    }
47
48    let mut cleared = 0;
49    for msg in messages.iter_mut() {
50        if msg.role == ModelRole::Tool
51            && let Some(ref tool_name) = msg.tool_name
52            && READ_ONLY_TOOLS.contains(&tool_name.as_str())
53            && !msg.content.contains("[Old tool result content cleared]")
54        {
55            msg.content = "[Old tool result content cleared]".to_string();
56            // Free multimodal payload (e.g. view_image base64) along with text.
57            msg.content_parts.clear();
58            cleared += 1;
59        }
60    }
61    cleared
62}
63
64pub const AUTOCOMPACT_BUFFER_TOKENS: u64 = 13_000;
65pub const WARNING_THRESHOLD_BUFFER_TOKENS: u64 = 20_000;
66pub const ERROR_THRESHOLD_BUFFER_TOKENS: u64 = 20_000;
67pub const MAX_OUTPUT_TOKENS_FOR_SUMMARY: u64 = 20_000;
68pub const MAX_CONSECUTIVE_FAILURES: u32 = 3;
69/// auto-compact when context usage reaches this percent of the window.
70pub const AUTO_COMPACT_THRESHOLD_PERCENT: u8 = 85;
71
72/// Context usage severity level used to trigger compact warnings and errors.
73#[derive(Debug, Clone, Copy, PartialEq, Eq)]
74pub enum CompactThreshold {
75    /// Context usage is within normal bounds.
76    Normal,
77    /// Context usage is approaching the limit; a warning should be shown.
78    Warning,
79    /// Context usage is critically close to the limit.
80    Error,
81    /// Compact has failed too many times; further attempts are blocked.
82    CircuitOpen,
83}
84
85impl std::fmt::Display for CompactThreshold {
86    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
87        match self {
88            CompactThreshold::Normal => write!(f, "ok"),
89            CompactThreshold::Warning => write!(f, "warning"),
90            CompactThreshold::Error => write!(f, "error"),
91            CompactThreshold::CircuitOpen => write!(f, "circuit-open"),
92        }
93    }
94}
95
96/// Tracks token usage and compact failure state for autocompact decisions.
97///
98/// Measurement model (aligned with modern coding CLI TUI):
99/// - **context tokens used** = last API `input_tokens` (ground truth for the
100/// current context) + client preflight for unsent bytes (`bytes/4`)
101/// - **window** = model `context_window` from registry
102/// - **usage %** = used / window
103/// - **total before compaction** = cumulative input tokens across turns
104/// (historical; does not reset when the bar drops after compact)
105#[derive(Debug, Clone)]
106pub struct CompactState {
107    /// Token count from the last model response usage (current context size).
108    pub last_input_tokens: Option<u64>,
109    /// Output tokens from the last model response (UI turn label).
110    pub last_output_tokens: Option<u64>,
111    /// Estimated bytes of new messages not yet sent to the model.
112    pub estimated_unsent_bytes: usize,
113    /// Context window size in tokens for the current model.
114    pub context_window: u64,
115    /// Cumulative input tokens processed before/during compaction cycles.
116    pub total_tokens_before_compaction: u64,
117    /// Number of compaction runs in this session.
118    pub compaction_count: u32,
119    /// Auto-compact when `context_percentage >= this` (default: 85).
120    pub auto_compact_threshold_percent: u8,
121    /// Number of consecutive compact failures.
122    pub consecutive_failures: u32,
123    pub summary: Option<String>,
124    pub summary_message_count: usize,
125    /// Latest long-horizon rebuild context that must stay attached to the
126    /// system prompt across subsequent turns.
127    pub rebuild_context: Option<String>,
128    /// List of checkpoint thresholds crossed in the current context cycle.
129    pub crossed_thresholds: Vec<f64>,
130    /// Fingerprints of messages already copied into long-horizon history.
131    pub history_synced_message_keys: HashSet<u64>,
132}
133
134impl Default for CompactState {
135    fn default() -> Self {
136        Self {
137            last_input_tokens: None,
138            last_output_tokens: None,
139            estimated_unsent_bytes: 0,
140            context_window: 0,
141            total_tokens_before_compaction: 0,
142            compaction_count: 0,
143            auto_compact_threshold_percent: AUTO_COMPACT_THRESHOLD_PERCENT,
144            consecutive_failures: 0,
145            summary: None,
146            summary_message_count: 0,
147            rebuild_context: None,
148            crossed_thresholds: Vec::new(),
149            history_synced_message_keys: HashSet::new(),
150        }
151    }
152}
153
154fn format_token_short(t: u64) -> String {
155    if t >= 1_000_000 {
156        let m = t as f64 / 1_000_000.0;
157        // Prefer `1M` over `1.0M` when close to a whole million.
158        if (m - m.round()).abs() < 0.05 {
159            format!("{}M", m.round() as u64)
160        } else {
161            format!("{:.1}M", m)
162        }
163    } else if t >= 10_000 {
164        format!("{}k", t / 1_000)
165    } else if t >= 1_000 {
166        // Keep one decimal under 10k so 1.5k doesn't collapse to `1k`.
167        let k = t as f64 / 1_000.0;
168        if (k - k.floor()).abs() < 0.05 {
169            format!("{}k", k.floor() as u64)
170        } else {
171            format!("{:.1}k", k)
172        }
173    } else {
174        t.to_string()
175    }
176}
177
178/// Effective prompt tokens for the context-window meter.
179///
180/// Handles providers that split non-cached vs cached prompt tokens (Anthropic
181/// and some OpenAI-compat aggregators). Without this, a session can show
182/// `430 / 1M` while the real fill is ~64k.
183pub fn context_tokens_for_meter(
184    input_tokens: Option<u64>,
185    cache_creation_tokens: u64,
186    cache_read_tokens: u64,
187) -> Option<u64> {
188    let input = input_tokens.unwrap_or(0);
189    if input == 0 && cache_creation_tokens == 0 && cache_read_tokens == 0 {
190        return None;
191    }
192    // Inclusive (OpenAI): prompt_tokens already includes cached_tokens.
193    if cache_read_tokens > 0 && input >= cache_read_tokens {
194        return Some(input.saturating_add(cache_creation_tokens));
195    }
196    // Exclusive: non-cached input + cache create + cache read.
197    Some(
198        input
199            .saturating_add(cache_creation_tokens)
200            .saturating_add(cache_read_tokens),
201    )
202}
203
204impl CompactState {
205    pub fn new(context_window: u64) -> Self {
206        Self {
207            context_window,
208            ..Default::default()
209        }
210    }
211
212    pub fn add_unsent_bytes(&mut self, bytes: usize) {
213        self.estimated_unsent_bytes += bytes;
214    }
215
216    pub fn clear_unsent_bytes(&mut self) {
217        self.estimated_unsent_bytes = 0;
218    }
219
220    /// Current context size (live context usage).
221    pub fn context_tokens_used(&self, pending_input_bytes: usize) -> u64 {
222        self.total_estimated_tokens(pending_input_bytes)
223    }
224
225    /// Context window size (context window size).
226    pub fn context_window_tokens(&self) -> u64 {
227        self.context_window
228    }
229
230    /// Integer percent of the window in use (context window usage percent).
231    pub fn context_window_usage(&self, pending_input_bytes: usize) -> u8 {
232        self.context_percentage(pending_input_bytes)
233    }
234
235    /// Estimate tokens for the next request:
236    /// API last-input (server) + unsent/pending client bytes as `ceil(bytes/4)`.
237    pub fn total_estimated_tokens(&self, pending_input_bytes: usize) -> u64 {
238        let server_tokens = self.last_input_tokens.unwrap_or(0);
239        let client_bytes = self.estimated_unsent_bytes + pending_input_bytes;
240        let client_tokens = (client_bytes.saturating_add(3) / 4) as u64;
241        server_tokens + client_tokens
242    }
243
244    pub fn threshold_level(&self, pending_input_bytes: usize) -> CompactThreshold {
245        if self.consecutive_failures >= MAX_CONSECUTIVE_FAILURES {
246            return CompactThreshold::CircuitOpen;
247        }
248        let total_tokens = self.total_estimated_tokens(pending_input_bytes);
249        if total_tokens == 0 {
250            return CompactThreshold::Normal;
251        }
252        let remaining = self.context_window.saturating_sub(total_tokens);
253        if remaining <= ERROR_THRESHOLD_BUFFER_TOKENS {
254            CompactThreshold::Error
255        } else if remaining <= WARNING_THRESHOLD_BUFFER_TOKENS + AUTOCOMPACT_BUFFER_TOKENS {
256            CompactThreshold::Warning
257        } else {
258            CompactThreshold::Normal
259        }
260    }
261
262    pub fn should_autocompact(&self, buffer_tokens: u64) -> bool {
263        if self.consecutive_failures >= MAX_CONSECUTIVE_FAILURES {
264            return false;
265        }
266        if self.context_window == 0 {
267            return false;
268        }
269        // compact near 85% of the window (includes unsent preflight).
270        let used = self.total_estimated_tokens(0);
271        let pct = (used as f64 / self.context_window as f64) * 100.0;
272        if pct >= f64::from(self.auto_compact_threshold_percent) {
273            return true;
274        }
275        // Hard ceiling: last API input + reserved buffer would fill the window.
276        let Some(input_tokens) = self.last_input_tokens else {
277            return false;
278        };
279        input_tokens.saturating_add(buffer_tokens) >= self.context_window
280    }
281
282    pub fn context_percentage(&self, pending_input_bytes: usize) -> u8 {
283        if self.context_window == 0 {
284            return 0;
285        }
286        let total_tokens = self.total_estimated_tokens(pending_input_bytes);
287        let percentage = (total_tokens as f64 / self.context_window as f64) * 100.0;
288        percentage.clamp(0.0, 100.0) as u8
289    }
290
291    /// Compact context meter for the composer footer: `3.2k / 200k`.
292    /// Percentage is revealed on hover (see `usage_label_with_percent`).
293    pub fn usage_label(&self, pending_input_bytes: usize) -> String {
294        self.usage_label_compact(pending_input_bytes)
295    }
296
297    /// Token counts only — default (non-hover) display.
298    pub fn usage_label_compact(&self, pending_input_bytes: usize) -> String {
299        let total_tokens = self.total_estimated_tokens(pending_input_bytes);
300        format!(
301            "{} / {}",
302            format_token_short(total_tokens),
303            format_token_short(self.context_window),
304        )
305    }
306
307    /// Token counts + percent — shown while the context chip is hovered.
308    pub fn usage_label_with_percent(&self, pending_input_bytes: usize) -> String {
309        let pct = self.context_percentage(pending_input_bytes);
310        format!(
311            "{} ({}%)",
312            self.usage_label_compact(pending_input_bytes),
313            pct
314        )
315    }
316
317    /// Record provider usage for this turn (called every stream that reports usage).
318    pub fn update_usage(&mut self, input_tokens: u64) {
319        self.update_usage_full(input_tokens, 0);
320    }
321
322    /// Record full turn usage and refresh live context metrics.
323    pub fn update_usage_full(&mut self, input_tokens: u64, output_tokens: u64) {
324        self.last_input_tokens = Some(input_tokens);
325        self.last_output_tokens = Some(output_tokens);
326        self.total_tokens_before_compaction = self
327            .total_tokens_before_compaction
328            .saturating_add(input_tokens);
329        self.clear_unsent_bytes();
330    }
331
332    /// Label shown in the TUI footer — updates every turn after `update_usage*`.
333    pub fn turn_usage_label(&self) -> Option<String> {
334        let input = self.last_input_tokens?;
335        let output = self.last_output_tokens.unwrap_or(0);
336        Some(format!(
337            "{}→{}",
338            format_token_short(input),
339            format_token_short(output)
340        ))
341    }
342
343    pub async fn auto_compact(
344        &mut self,
345        messages: &mut Vec<ModelMessage>,
346        model_provider: &dyn ModelProvider,
347        model_name: &str,
348        harness_config: &HarnessConfig,
349    ) -> Result<Option<u64>> {
350        if !self.should_autocompact(harness_config.autocompact_buffer_tokens) {
351            return Ok(None);
352        }
353
354        // Split: system + developer messages first (the prompt prefix),
355        // then conversation messages.
356        let system_msgs: Vec<ModelMessage> = messages
357            .iter()
358            .filter(|m| m.role == ModelRole::System || m.role == ModelRole::Developer)
359            .cloned()
360            .collect();
361        let conversation_msgs: Vec<ModelMessage> = messages
362            .iter()
363            .filter(|m| m.role != ModelRole::System && m.role != ModelRole::Developer)
364            .cloned()
365            .collect();
366
367        if conversation_msgs.is_empty() {
368            return Ok(None);
369        }
370
371        // KeepRatio: keep the last N% of conversation turns intact.
372        let keep_ratio = harness_config.autocompact_keep_ratio.clamp(0.0, 0.9);
373        let total = conversation_msgs.len();
374        let keep_count = (total as f64 * keep_ratio).round() as usize;
375        // Always keep at least 2 messages (1 user + 1 assistant) and at most
376        // total - 2 (so there's something to summarize).
377        let keep_count = keep_count.clamp(2.min(total), total.saturating_sub(2).max(2.min(total)));
378        let split_at = total.saturating_sub(keep_count);
379
380        // Old messages → summarize. Recent messages → keep intact.
381        let (old_msgs, recent_msgs) = conversation_msgs.split_at(split_at);
382        let old_text = build_conversation_text(old_msgs);
383
384        if old_text.trim().is_empty() {
385            // Nothing old to summarize; just keep everything.
386            return Ok(None);
387        }
388
389        let prompt = if let Some(ref prev_summary) = self.summary {
390            PARTIAL_COMPACT_PROMPT
391                .replace("{previous_summary}", prev_summary)
392                .replace("{new_conversation}", &old_text)
393        } else {
394            format!(
395                "{}\n\nConversation to summarize:\n{}",
396                COMPACT_PROMPT, old_text
397            )
398        };
399
400        let request = ModelRequest {
401            model: model_name.to_string(),
402            instructions: None,
403            messages: vec![
404                ModelMessage::system("You are a precise conversation summarizer."),
405                ModelMessage::user(prompt),
406            ],
407            thinking: ThinkingConfig::Off,
408            tools: vec![],
409            session_id: None,
410        };
411
412        match model_provider.complete(request).await {
413            Ok(response) => {
414                let summary = response.text;
415                let previous_tokens = self.last_input_tokens.unwrap_or(0);
416
417                // Reassemble: system + summary + recent turns kept intact.
418                messages.clear();
419                messages.extend(system_msgs);
420                messages.push(ModelMessage::user(format!(
421                    "Here is a summary of the conversation so far:\n\n{}",
422                    summary
423                )));
424                messages.extend(recent_msgs.iter().cloned());
425
426                self.summary = Some(summary);
427                self.summary_message_count = messages.len();
428                self.consecutive_failures = 0;
429                self.last_input_tokens = None;
430                self.last_output_tokens = None;
431                self.compaction_count = self.compaction_count.saturating_add(1);
432
433                let tokens_saved =
434                    previous_tokens.saturating_sub(harness_config.autocompact_max_output_tokens);
435                tracing::info!(
436                    tokens_saved,
437                    old_turns = old_msgs.len(),
438                    kept_turns = recent_msgs.len(),
439                    "auto-compact completed"
440                );
441
442                Ok(Some(tokens_saved))
443            }
444            Err(e) => {
445                self.consecutive_failures += 1;
446                tracing::warn!(
447                    failures = self.consecutive_failures,
448                    error = %e,
449                    "auto-compact failed"
450                );
451                Err(e)
452            }
453        }
454    }
455
456    pub fn apply_manual_summary(
457        &mut self,
458        messages: &mut Vec<ModelMessage>,
459        summary: String,
460    ) -> u64 {
461        let system_msgs: Vec<ModelMessage> = messages
462            .iter()
463            .filter(|m| m.role == ModelRole::System)
464            .cloned()
465            .collect();
466        let estimated_previous_tokens = self.last_input_tokens.unwrap_or_else(|| {
467            messages
468                .iter()
469                .map(|message| message.content.len() as u64)
470                .sum::<u64>()
471                .saturating_add(3)
472                / 4
473        });
474
475        messages.clear();
476        messages.extend(system_msgs);
477        messages.push(ModelMessage::user(format!(
478            "Here is a summary of the conversation so far:\n\n{}",
479            summary
480        )));
481
482        self.summary = Some(summary);
483        self.summary_message_count = messages.len();
484        self.consecutive_failures = 0;
485        self.last_input_tokens = None;
486        self.last_output_tokens = None;
487        self.compaction_count = self.compaction_count.saturating_add(1);
488        self.clear_unsent_bytes();
489
490        estimated_previous_tokens
491    }
492}
493
494fn build_conversation_text(messages: &[ModelMessage]) -> String {
495    let mut text = String::new();
496    for msg in messages {
497        let role_label = match msg.role {
498            ModelRole::User => "User",
499            ModelRole::Assistant => "Assistant",
500            ModelRole::Tool => "Tool",
501            ModelRole::System | ModelRole::Developer => continue,
502        };
503        if msg.role == ModelRole::Tool {
504            if let Some(ref tool_name) = msg.tool_name {
505                text.push_str(&format!("[Tool({})]: {}\n", tool_name, msg.content));
506            } else {
507                text.push_str(&format!("[Tool]: {}\n", msg.content));
508            }
509        } else {
510            let image_note = if msg.content_parts.iter().any(|p| p.is_image()) {
511                let count = msg.content_parts.iter().filter(|p| p.is_image()).count();
512                format!(" [{} image(s) attached]", count)
513            } else {
514                String::new()
515            };
516            text.push_str(&format!(
517                "[{}]: {}{}\n",
518                role_label, msg.content, image_note
519            ));
520        }
521    }
522    text
523}
524
525pub const COMPACT_PROMPT: &str = r#"You are summarizing a conversation between a user and an AI coding assistant (NAVI). Create a detailed summary with these exact sections:
526
527## 1. Primary Request and Intent
528## 2. Key Technical Concepts
529## 3. Files and Code Snippets
530## 4. Errors and Fixes
531## 5. Problem Resolution
532## 6. All User Messages
533## 7. Pending Tasks
534## 8. Current Work
535## 9. Active Work Plan
536If the conversation has an active plan (via the plan tool) or an in-progress Plan-mode proposal, include:
537- Plan ID and title (if any)
538- All steps with completion status
539- Which step to work on next
540If there is no active plan, skip this section.
541Also note any active thread goal (set_goal) separately if present — do not conflate plan and goal.
542## 10. Next Step (Optional)
543List the next step you would take on the current task.
544
545Be thorough and specific. The summary must contain enough detail to continue the conversation seamlessly.
546
547IMPORTANT: If there is an active plan that is not completed or abandoned, continue it after reading this summary UNLESS the user clearly redirected to a different task — in that case, note the redirect in section 8/10 and do not restart the old plan. Do not create a new plan unless the prior plan is completed, abandoned, or the user asked for a new one."#;
548
549pub const PARTIAL_COMPACT_PROMPT: &str = r#"You are extending an existing conversation summary with new content. Preserve the existing summary sections and update them with new information. Add any new user messages to section 6. Update sections 8 and 9 based on the most recent work.
550
551Existing summary:
552{previous_summary}
553
554New conversation to summarize:
555{new_conversation}
556
557Return the complete updated summary with all 10 sections (including Active Work Plan if applicable).
558
559IMPORTANT: If there is an active plan, preserve plan details and step completion status. Continue that plan only if the user has not redirected; note any redirect instead of forcing the old plan."#;
560
561fn current_unix_millis() -> u64 {
562    SystemTime::now()
563        .duration_since(UNIX_EPOCH)
564        .unwrap_or_default()
565        .as_millis() as u64
566}
567
568#[cfg(test)]
569mod tests {
570    use super::*;
571    use crate::model::ModelMessage;
572
573    #[test]
574    fn micro_compact_clears_read_only_tools_after_gap() {
575        let now = current_unix_millis();
576        let gap_ms: u64 = 61 * 60 * 1000;
577
578        let mut messages = vec![
579            ModelMessage::system("system"),
580            ModelMessage::user("task"),
581            {
582                let mut m = ModelMessage::assistant("response");
583                m.created_at = Some(now.saturating_sub(gap_ms));
584                m
585            },
586            ModelMessage::tool_result("call-1", "read_file", "file content here".to_string()),
587            ModelMessage::tool_result("call-2", "write_file", "written content".to_string()),
588            ModelMessage::tool_result("call-3", "grep", "match results".to_string()),
589            ModelMessage::tool_result("call-5", "bash", "command output".to_string()),
590        ];
591
592        let cleared = micro_compact(&mut messages, 60);
593        assert_eq!(cleared, 2);
594        assert!(
595            messages[3]
596                .content
597                .contains("[Old tool result content cleared]")
598        );
599        assert_eq!(messages[4].content, "written content");
600        assert!(
601            messages[5]
602                .content
603                .contains("[Old tool result content cleared]")
604        );
605        assert_eq!(messages[6].content, "command output");
606    }
607
608    #[test]
609    fn micro_compact_no_gap_returns_zero() {
610        let mut messages = vec![
611            ModelMessage::system("system"),
612            ModelMessage::user("task"),
613            ModelMessage::assistant("response"),
614            ModelMessage::tool_result("call-1", "read_file", "content".to_string()),
615        ];
616
617        let cleared = micro_compact(&mut messages, 60);
618        assert_eq!(cleared, 0);
619    }
620
621    #[test]
622    fn micro_compact_no_double_clear() {
623        let now = current_unix_millis();
624        let gap_ms: u64 = 61 * 60 * 1000;
625
626        let mut messages = vec![
627            ModelMessage::system("system"),
628            {
629                let mut m = ModelMessage::assistant("response");
630                m.created_at = Some(now.saturating_sub(gap_ms));
631                m
632            },
633            ModelMessage::tool_result(
634                "call-1",
635                "read_file",
636                "[Old tool result content cleared]".to_string(),
637            ),
638        ];
639
640        let cleared = micro_compact(&mut messages, 60);
641        assert_eq!(cleared, 0);
642    }
643
644    #[test]
645    fn compact_state_threshold_normal() {
646        let state = CompactState {
647            last_input_tokens: Some(50_000),
648            context_window: 200_000,
649            ..Default::default()
650        };
651        assert_eq!(state.threshold_level(0), CompactThreshold::Normal);
652    }
653
654    #[test]
655    fn compact_state_threshold_warning() {
656        let state = CompactState {
657            last_input_tokens: Some(170_000),
658            context_window: 200_000,
659            ..Default::default()
660        };
661        assert_eq!(state.threshold_level(0), CompactThreshold::Warning);
662    }
663
664    #[test]
665    fn compact_state_threshold_error() {
666        let state = CompactState {
667            last_input_tokens: Some(181_000),
668            context_window: 200_000,
669            ..Default::default()
670        };
671        assert_eq!(state.threshold_level(0), CompactThreshold::Error);
672    }
673
674    #[test]
675    fn compact_state_circuit_breaker() {
676        let state = CompactState {
677            last_input_tokens: Some(50_000),
678            context_window: 200_000,
679            consecutive_failures: 3,
680            ..Default::default()
681        };
682        assert_eq!(state.threshold_level(0), CompactThreshold::CircuitOpen);
683        assert!(!state.should_autocompact(AUTOCOMPACT_BUFFER_TOKENS));
684    }
685
686    #[test]
687    fn compact_state_should_autocompact() {
688        let state = CompactState {
689            last_input_tokens: Some(190_000),
690            context_window: 200_000,
691            ..Default::default()
692        };
693        assert!(state.should_autocompact(AUTOCOMPACT_BUFFER_TOKENS));
694    }
695
696    #[test]
697    fn compact_state_autocompact_at_eighty_five_percent() {
698        // 170k / 200k = 85% triggers even when buffer would not.
699        let state = CompactState {
700            last_input_tokens: Some(170_000),
701            context_window: 200_000,
702            auto_compact_threshold_percent: 85,
703            ..Default::default()
704        };
705        assert!(state.should_autocompact(0));
706        assert_eq!(state.context_window_usage(0), 85);
707    }
708
709    #[test]
710    fn compact_state_update_usage_full_tracks_cumulative() {
711        let mut state = CompactState::new(200_000);
712        state.update_usage_full(10_000, 500);
713        state.update_usage_full(12_000, 800);
714        assert_eq!(state.last_input_tokens, Some(12_000));
715        assert_eq!(state.last_output_tokens, Some(800));
716        assert_eq!(state.total_tokens_before_compaction, 22_000);
717        assert_eq!(state.turn_usage_label().as_deref(), Some("12k→800"));
718    }
719
720    #[test]
721    fn compact_state_manual_summary_compacts_below_threshold() {
722        let mut state = CompactState {
723            last_input_tokens: Some(10_000),
724            context_window: 200_000,
725            consecutive_failures: 2,
726            estimated_unsent_bytes: 4096,
727            ..Default::default()
728        };
729        assert!(!state.should_autocompact(AUTOCOMPACT_BUFFER_TOKENS));
730        let mut messages = vec![
731            ModelMessage::system("system"),
732            ModelMessage::user("task"),
733            ModelMessage::assistant("response"),
734        ];
735
736        let tokens_saved = state.apply_manual_summary(&mut messages, "Manual summary".to_string());
737
738        assert_eq!(tokens_saved, 10_000);
739        assert_eq!(messages.len(), 2);
740        assert_eq!(messages[0].role, ModelRole::System);
741        assert!(messages[1].content.contains("Manual summary"));
742        assert_eq!(state.summary.as_deref(), Some("Manual summary"));
743        assert_eq!(state.summary_message_count, 2);
744        assert_eq!(state.consecutive_failures, 0);
745        assert_eq!(state.estimated_unsent_bytes, 0);
746        assert!(state.last_input_tokens.is_none());
747    }
748
749    #[test]
750    fn compact_state_context_percentage() {
751        let state = CompactState {
752            last_input_tokens: Some(100_000),
753            context_window: 200_000,
754            ..Default::default()
755        };
756        assert_eq!(state.context_percentage(0), 50);
757    }
758
759    #[test]
760    fn compact_state_no_usage_returns_zero_percent() {
761        let state = CompactState {
762            last_input_tokens: None,
763            context_window: 200_000,
764            ..Default::default()
765        };
766        assert_eq!(state.context_percentage(0), 0);
767    }
768
769    #[test]
770    fn compact_state_usage_label_shows_real_context_usage() {
771        let state = CompactState {
772            last_input_tokens: Some(2_000),
773            context_window: 128_000,
774            ..Default::default()
775        };
776        // Default (composer) is counts only; percent is a hover-only affordance.
777        assert_eq!(state.usage_label(0), "2k / 128k");
778        assert_eq!(state.usage_label_with_percent(0), "2k / 128k (1%)");
779    }
780
781    #[test]
782    fn context_tokens_for_meter_sums_exclusive_cache() {
783        // Charm-style undercount: tiny non-cached prompt + large cache hit.
784        assert_eq!(context_tokens_for_meter(Some(430), 0, 63_570), Some(64_000));
785    }
786
787    #[test]
788    fn context_tokens_for_meter_keeps_openai_inclusive() {
789        // OpenAI: prompt_tokens already includes cached_tokens.
790        assert_eq!(
791            context_tokens_for_meter(Some(64_000), 0, 63_570),
792            Some(64_000)
793        );
794    }
795
796    #[test]
797    fn format_token_short_million_window() {
798        assert_eq!(format_token_short(1_048_576), "1M");
799        assert_eq!(format_token_short(64_000), "64k");
800    }
801
802    #[test]
803    fn write_tool_preserved_in_micro_compact() {
804        let now = current_unix_millis();
805        let gap_ms: u64 = 61 * 60 * 1000;
806
807        let mut messages = vec![
808            ModelMessage::system("system"),
809            {
810                let mut m = ModelMessage::assistant("response");
811                m.created_at = Some(now.saturating_sub(gap_ms));
812                m
813            },
814            ModelMessage::tool_result("call-1", "write_file", "content written".to_string()),
815            ModelMessage::tool_result("call-2", "apply_patch", "patch applied".to_string()),
816        ];
817
818        let cleared = micro_compact(&mut messages, 60);
819        assert_eq!(cleared, 0);
820        assert_eq!(messages[2].content, "content written");
821        assert_eq!(messages[3].content, "patch applied");
822    }
823
824    // ── Regression tests ──────────────────────────────────────────────────────
825
826    #[test]
827    fn regression_micro_compact_no_assistant_messages_returns_zero() {
828        let mut messages = vec![
829            ModelMessage::system("system"),
830            ModelMessage::user("hello"),
831            ModelMessage::tool_result("c1", "read_file", "content"),
832        ];
833        let cleared = micro_compact(&mut messages, 60);
834        assert_eq!(cleared, 0);
835    }
836
837    #[test]
838    fn regression_micro_compact_preserves_non_readonly_tools() {
839        let now = current_unix_millis();
840        let gap_ms: u64 = 61 * 60 * 1000;
841
842        let mut messages = vec![
843            ModelMessage::system("system"),
844            {
845                let mut m = ModelMessage::assistant("response");
846                m.created_at = Some(now.saturating_sub(gap_ms));
847                m
848            },
849            ModelMessage::tool_result("c1", "write_file", "file written"),
850            ModelMessage::tool_result("c2", "package_manager", "deps ok"),
851            ModelMessage::tool_result("c3", "apply_patch", "patch applied"),
852            ModelMessage::tool_result("c4", "bash", "command ok"),
853        ];
854
855        let cleared = micro_compact(&mut messages, 60);
856        assert_eq!(cleared, 0, "non-read-only tools must not be cleared");
857        assert_eq!(messages[2].content, "file written");
858        assert_eq!(messages[3].content, "deps ok");
859        assert_eq!(messages[4].content, "patch applied");
860        assert_eq!(messages[5].content, "command ok");
861    }
862
863    #[test]
864    fn regression_compact_state_percentage_clamps_to_100() {
865        let mut state = CompactState::new(1000);
866        // Simulate more tokens than window
867        state.last_input_tokens = Some(2000);
868        let pct = state.context_percentage(0);
869        assert!(pct <= 100, "percentage must clamp to 100, got {pct}");
870    }
871
872    #[test]
873    fn regression_compact_state_usage_label_formats_millions() {
874        let state = CompactState {
875            last_input_tokens: Some(1_500_000),
876            context_window: 2_000_000,
877            ..Default::default()
878        };
879        let label = state.usage_label(0);
880        assert!(label.contains("M"), "should use M format for millions");
881    }
882
883    #[test]
884    fn regression_build_conversation_text_excludes_system() {
885        let messages = vec![
886            ModelMessage::system("you are a helpful assistant"),
887            ModelMessage::user("hello"),
888            ModelMessage::assistant("hi there"),
889        ];
890        let text = build_conversation_text(&messages);
891        assert!(
892            !text.contains("you are a helpful assistant"),
893            "system message must be excluded"
894        );
895        assert!(text.contains("hello"));
896        assert!(text.contains("hi there"));
897    }
898
899    #[test]
900    fn regression_build_conversation_text_includes_tool_name() {
901        let messages = vec![
902            ModelMessage::user("read file"),
903            ModelMessage::tool_result("c1", "read_file", "file content"),
904        ];
905        let text = build_conversation_text(&messages);
906        assert!(
907            text.contains("read_file"),
908            "tool name must be included in conversation text"
909        );
910    }
911
912    #[test]
913    fn keep_ratio_clamps_valid_range() {
914        let config = HarnessConfig {
915            autocompact_keep_ratio: 0.25,
916            ..Default::default()
917        };
918        assert_eq!(config.autocompact_keep_ratio, 0.25);
919    }
920
921    #[test]
922    fn keep_ratio_default_is_25_percent() {
923        let config = HarnessConfig::default();
924        assert_eq!(config.autocompact_keep_ratio, 0.25);
925    }
926
927    #[test]
928    fn build_conversation_text_preserves_order() {
929        let messages = vec![
930            ModelMessage::user("first"),
931            ModelMessage::assistant("second"),
932            ModelMessage::user("third"),
933            ModelMessage::assistant("fourth"),
934        ];
935        let text = build_conversation_text(&messages);
936        let first_pos = text.find("first").unwrap();
937        let second_pos = text.find("second").unwrap();
938        let third_pos = text.find("third").unwrap();
939        let fourth_pos = text.find("fourth").unwrap();
940        assert!(first_pos < second_pos);
941        assert!(second_pos < third_pos);
942        assert!(third_pos < fourth_pos);
943    }
944}