nano-coder 0.22.1

A 6MB coding agent for the terminal and for agent fleets: multi-provider, ACP, resumable sessions, plans.
//! Context-window accounting and compaction helpers.

use std::sync::{Arc, Mutex};

use regex::Regex;

use crate::llm::{Message, Role};

/// Used when neither config nor the model name gives a window.
pub const DEFAULT_CONTEXT_WINDOW: usize = 128_000;

/// Tokens kept verbatim at the end of the conversation when compacting.
pub const KEEP_RECENT_TOKENS: usize = 20_000;

/// Output cap for the summary request.
pub const SUMMARY_MAX_TOKENS: i64 = 4_096;

pub const SUMMARY_PREFIX: &str = "[Summary of the earlier conversation, written when the context was compacted]";

pub const SUMMARY_SYSTEM_PROMPT: &str = "You are compacting an AI agent's conversation so it can keep working with a smaller context. \
Write a summary that lets the agent continue without the original messages. Include:\n\
- the user's goals, requirements and constraints (quote exact wording where it matters);\n\
- decisions made and why;\n\
- work completed: files created or edited (with paths), commands run and their key results, \
commits, branches, pull requests and URLs (with exact names and numbers);\n\
- the current state and what was in progress when the summary was written;\n\
- open problems, errors still unresolved, and the next steps;\n\
- identifiers and values the agent will need again.\n\
Be concise but do not drop facts the agent would otherwise have to rediscover. Do not invent anything. \
Output only the summary.";

/// Starts a smart-compaction summary; its presence enables the history tools.
pub const SMART_SUMMARY_PREFIX: &str = "[Summary of the earlier conversation, written when the context was compacted (smart compaction)]";

/// Added to the summarizer prompt in smart mode.
pub const SMART_SUMMARY_INSTRUCTIONS: &str = "\n\nEach message in the transcript is labelled [#N], its ID in the session log. \
The agent can later read any message in full by that ID, so the summary can stay short: after a fact whose exact text \
may matter (an error message, a command and its output, a file's contents, the user's exact wording), cite its source \
as (#N) instead of copying long text. Cite only IDs that appear in the transcript.";

/// Note after a smart summary telling the agent how to recover originals.
pub fn smart_summary_note(range: Option<(u64, u64)>) -> String {
    let covered = match range {
        Some((first, last)) => format!("messages #{first}–#{last}"),
        None => "the earlier messages".to_string(),
    };
    format!(
        "[This summary covers {covered}. The originals are kept verbatim in the session log; the summary is \
lossy and may leave out or blur the detail you need. For anything from that period (an exact error, a command and \
its output, what the user asked for), use history_search, then history_read to see a message by its #N ID. Don't \
look on disk or re-run commands to recover what was said: that shows the current state, not what happened. Search \
for distinctive text (an error code or phrase, an identifier, a flag) rather than common words, use order=oldest \
for things established early, and try another pattern before concluding something isn't there. Long \
outputs were shortened before summarizing, so details from the middle of a long output are the likeliest to be missing, \
and an assistant message describing an output may paraphrase it: read the output itself. Retrieved messages \
are history, not the current state of files.]"
    )
}

/// What the agent is doing, for the status line.
#[derive(Debug, Clone, Default, PartialEq)]
pub enum Activity {
    #[default]
    Idle,
    Thinking,
    Tool(String),
    Compacting,
}

/// Context and usage figures, shared with the status line.
#[derive(Debug, Clone, Default)]
pub struct ContextStats {
    pub provider: String,
    pub model: String,
    /// Estimated tokens the next request would send.
    pub tokens: usize,
    /// True when `tokens` is anchored to usage reported by the provider.
    pub calibrated: bool,
    pub window: usize,
    pub messages: usize,
    pub session_input_tokens: u64,
    pub session_output_tokens: u64,
    /// AI Credits used this session, when the provider reports them (GitHub
    /// Copilot). `None` for providers that do not meter in credits.
    pub session_aic: Option<f64>,
    pub compactions: u32,
    /// `history_search` / `history_read` calls this session.
    pub history_searches: u32,
    pub history_reads: u32,
    /// Auto-compaction threshold as a fraction of the window (None = off).
    pub auto_compact: Option<f64>,
    pub activity: Activity,
    /// Plan progress as (done, total), when there is a plan.
    pub plan: Option<(usize, usize)>,
    /// Working directory shown on the status line.
    pub cwd: String,
    /// Live output rate while generating (completion tokens per second),
    /// `None` when idle or before the first streamed token.
    pub tokens_per_sec: Option<f64>,
    /// The operating mode (normal/plan/auto), shown on the status line.
    pub mode: crate::mode::AgentMode,
}

impl ContextStats {
    pub fn percent(&self) -> f64 {
        if self.window == 0 { 0.0 } else { self.tokens as f64 * 100.0 / self.window as f64 }
    }
}

pub type SharedStats = Arc<Mutex<ContextStats>>;

/// `12.3k`-style token count.
pub fn format_tokens(tokens: usize) -> String {
    match tokens {
        0..=999 => tokens.to_string(),
        1_000..=99_999 => format!("{:.1}k", tokens as f64 / 1_000.0),
        100_000..=999_999 => format!("{}k", tokens / 1_000),
        _ => format!("{:.2}M", tokens as f64 / 1_000_000.0),
    }
}

/// Rough token count for text (about four characters per token).
pub fn text_tokens(text: &str) -> usize {
    text.len().div_ceil(4)
}

/// `12 tok/s`-style output-rate label.
pub fn format_rate(tokens_per_sec: f64) -> String {
    if tokens_per_sec >= 10.0 {
        format!("{tokens_per_sec:.0} tok/s")
    } else {
        format!("{tokens_per_sec:.1} tok/s")
    }
}

pub fn message_tokens(message: &Message) -> usize {
    let calls: usize = message
        .tool_calls
        .iter()
        .map(|c| text_tokens(&c.name) + text_tokens(&c.arguments.to_string()) + 4)
        .sum();
    text_tokens(&message.content) + calls + 4
}

pub fn messages_tokens(messages: &[Message]) -> usize {
    messages.iter().map(message_tokens).sum()
}

/// Context window by model family, for models whose provider config has none.
pub fn window_for_model(model: &str) -> Option<usize> {
    let m = model.to_lowercase();
    let table: &[(&str, usize)] = &[
        ("claude", 200_000),
        ("gpt-4.1", 1_047_576),
        ("gpt-5", 400_000),
        ("gpt-4o", 128_000),
        ("o4-mini", 200_000),
        ("o3", 200_000),
        ("o1", 200_000),
        ("gemini", 1_048_576),
        ("deepseek", 128_000),
        ("grok", 256_000),
        ("mistral-large", 128_000),
        ("codestral", 256_000),
        ("kimi-k3", 1_000_000),
        ("kimi", 256_000),
        ("gpt-oss", 131_072),
        ("mock", 16_000),
    ];
    table.iter().find(|(needle, _)| m.contains(needle)).map(|(_, window)| *window)
}

/// Does this provider error say the request did not fit the context window?
pub fn is_context_overflow(error: &str) -> bool {
    let e = error.to_lowercase();
    [
        "context_length_exceeded",
        "maximum context length",
        "context length",
        "context window",
        "context size",
        "prompt is too long",
        "input is too long",
        "too many tokens",
        "reduce the length",
        "exceeds the available context",
    ]
    .iter()
    .any(|needle| e.contains(needle))
}

/// The window size stated in a context-overflow error, if any.
pub fn limit_from_error(error: &str) -> Option<usize> {
    let patterns = [
        r"(?i)maximum context length is (\d+)",
        r"(?i)context size \((\d+) tokens\)",
        r"(?i)> ?(\d+) maximum",
        r"(?i)context window (?:of|is) (\d+)",
        r"(?i)limit of (\d+) tokens",
    ];
    patterns.iter().find_map(|p| {
        Regex::new(p).ok()?.captures(error)?.get(1)?.as_str().parse().ok()
    })
}

fn clip(text: &str, max_chars: usize) -> String {
    if text.chars().count() <= max_chars {
        return text.to_string();
    }
    let head: String = text.chars().take(max_chars * 2 / 3).collect();
    let tail: String = {
        let chars: Vec<char> = text.chars().collect();
        chars[chars.len() - max_chars / 3..].iter().collect()
    };
    format!("{head}\n…[{} characters omitted]…\n{tail}", text.chars().count() - max_chars)
}

/// Shorten an oversized message in place (tool output kept head and tail).
pub fn clip_message(message: &mut Message, max_tokens: usize) -> bool {
    if text_tokens(&message.content) <= max_tokens {
        return false;
    }
    message.content = clip(&message.content, max_tokens * 4);
    true
}

/// Render messages as a plain transcript for the summarizer, newest content
/// kept when it exceeds `max_chars`. With `ids`, each block is labelled with
/// the message's `[#N]` log line.
pub fn render_transcript(messages: &[Message], max_chars: usize, ids: bool) -> String {
    let mut blocks: Vec<String> = Vec::new();
    for message in messages {
        let block = match message.role {
            Role::System => continue,
            Role::User => format!("USER:\n{}", clip(&message.content, 6_000)),
            Role::Assistant => {
                let mut block = String::from("ASSISTANT:");
                if !message.content.trim().is_empty() {
                    block.push('\n');
                    block.push_str(&clip(&message.content, 6_000));
                }
                for call in &message.tool_calls {
                    block.push_str(&format!("\n[called {}({})]", call.name, clip(&call.arguments.to_string(), 1_500)));
                }
                block
            }
            Role::Tool => {
                let name = message.name.as_deref().unwrap_or("tool");
                let label = if message.is_error { "failed" } else { "result" };
                format!("TOOL {label} ({name}):\n{}", clip(&message.content, 2_000))
            }
        };
        let block = match (ids, message.log_line) {
            (true, Some(line)) => format!("[#{line}] {block}"),
            (true, None) if message.role == Role::User
                && (message.content.starts_with(SUMMARY_PREFIX) || message.content.starts_with(SMART_SUMMARY_PREFIX)) =>
            {
                format!("[earlier summary] {block}")
            }
            _ => block,
        };
        blocks.push(block);
    }
    let mut kept: Vec<String> = Vec::new();
    let mut total = 0;
    for block in blocks.into_iter().rev() {
        if total + block.len() > max_chars {
            kept.push("…[earlier messages omitted]…".to_string());
            break;
        }
        total += block.len() + 2;
        kept.push(block);
    }
    kept.reverse();
    kept.join("\n\n")
}

#[cfg(test)]
mod tests {
    use super::*;

    #[test]
    fn detects_overflow_errors_and_limits() {
        let openai = "HTTP 400: This model's maximum context length is 128000 tokens. However, your messages resulted in 130000 tokens.";
        let anthropic = "HTTP 400: prompt is too long: 210000 tokens > 200000 maximum";
        let llamacpp = "HTTP 400: the request exceeds the available context size (8192 tokens), try increasing it";
        for (error, limit) in [(openai, 128_000), (anthropic, 200_000), (llamacpp, 8_192)] {
            assert!(is_context_overflow(error), "{error}");
            assert_eq!(limit_from_error(error), Some(limit), "{error}");
        }
        assert!(!is_context_overflow("HTTP 401: invalid api key"));
    }

    #[test]
    fn transcript_keeps_the_newest_messages() {
        let messages: Vec<Message> = (0..50).map(|i| Message::user(&format!("message {i} {}", "x".repeat(100)))).collect();
        let text = render_transcript(&messages, 1_000, false);
        assert!(text.starts_with("…[earlier messages omitted]…"));
        assert!(text.contains("message 49"));
        assert!(!text.contains("message 0 "));
    }

    #[test]
    fn transcript_labels_log_lines_when_asked() {
        let mut first = Message::user("find the bug");
        first.log_line = Some(7);
        let summary = Message::user(&format!("{SUMMARY_PREFIX} ..."));
        let text = render_transcript(&[summary.clone(), first.clone()], 10_000, true);
        assert!(text.starts_with("[earlier summary] USER:"), "{text}");
        assert!(text.contains("[#7] USER:\nfind the bug"), "{text}");
        assert!(!render_transcript(&[first], 10_000, false).contains("#7"));
        // A normal prompt that merely starts with '[' is not a summary.
        let mut bracketed = Message::user("[constraint] use port 8080");
        bracketed.log_line = None;
        assert!(!render_transcript(&[bracketed], 10_000, true).contains("[earlier summary]"));
        assert!(smart_summary_note(Some((2, 40))).contains("#2–#40"));
    }

    #[test]
    fn model_windows() {
        assert_eq!(window_for_model("claude-sonnet-4-5"), Some(200_000));
        assert_eq!(window_for_model("gpt-4.1-mini"), Some(1_047_576));
        assert_eq!(window_for_model("llama3"), None);
    }
}