supercode-reduce 0.5.45

Optional lossless, reversible session reduction for Volter Harness
Documentation
//! T30/TR-4 — terminal-noise normalization (`ReductionKind::OutputNormalized`):
//! a small, deterministic line-buffer terminal simulator that collapses ANSI
//! color/style codes and carriage-return/erase-line/cursor-up redraws down to
//! the FINAL rendered content of each line — the same content a human
//! watching the build would actually see, without the hundreds of
//! intermediate redraws a captured progress bar otherwise leaves in the
//! transcript.
//!
//! **Not a full `vte` emulation** (per SPEC.md TR-4's approach sketch): this
//! supports exactly the sequences that dominate real `cargo`/`npm`/`pip`/
//! `docker` output —
//!
//! - SGR (`ESC[...m`, colors/styles) — stripped; it never prints or moves.
//! - CR (`\r`) — cursor to column 0 of the current row.
//! - LF (`\n`) — cursor to column 0 of the NEXT row (a deliberate
//!   simplification: real LF preserves column, but every real capture in
//!   this codebase's fixtures pairs LF with either a preceding CR or content
//!   that starts a fresh line anyway, so this never diverges from the
//!   fixtures' actual rendering and keeps the model trivial to reason about).
//! - EL (`ESC[K`, `ESC[0K`, `ESC[1K`, `ESC[2K`) — erase to end / to start /
//!   whole line.
//! - CUU / CUD (`ESC[<n>A` / `ESC[<n>B`) — cursor up/down `n` rows (the
//!   multi-line redraw idiom `docker pull` uses for concurrent layers).
//! - CHA (`ESC[<n>G`) — cursor to absolute column `n` (1-based); the idiom
//!   modern `npm`'s spinner uses instead of `\r`.
//! - DEC private mode set/reset (`ESC[?...h` / `ESC[?...l`) — e.g. `?25l`/
//!   `?25h` (cursor hide/show around a spinner): dropped silently. This is a
//!   deliberately narrow carve-out (real DEC private modes can do much more,
//!   e.g. the alternate screen buffer) but build-tool output never uses those
//!   — see the module-level safety note below.
//!
//! Everything else — including a CSI sequence with a final byte this module
//! doesn't recognize (device status report, cursor-position, scroll, etc.),
//! an OSC sequence, or any escape truncated by an upstream byte cap before
//! its terminator — is passed through **verbatim, as literal printable
//! text**, landing in the rendered output unchanged. "Never guess": an
//! unrecognized sequence is never assumed to be a no-op, so its bytes are
//! never silently dropped, and the reduction is content-preserving even for
//! escape vocabulary this module has never seen.
//!
//! # Safety / no-panic guarantee
//!
//! [`normalize`] never panics on any input, including a `&str` truncated
//! mid-escape-sequence (the TR-1 gotcha this module was warned about:
//! `Agent::cap_tool_output`'s 100 KB history cap can slice a raw tool output
//! anywhere, including through the middle of a CSI/OSC sequence, before this
//! module ever sees it). A truncated sequence at the end of the input is
//! detected (no terminator found before the string ends) and copied through
//! as literal text, same as any other unrecognized sequence — never a panic,
//! never an out-of-bounds slice. See `tests::never_panics_on_malformed_input`
//! for a sweep over adversarial byte patterns (including sequences chopped at
//! every possible byte boundary).
//!
//! All scanning here is on `&str` byte offsets, but every control byte this
//! module inspects (`ESC` 0x1B, `CR` 0x0D, `LF` 0x0A, CSI param/final bytes
//! 0x20-0x7E) is ASCII — and ASCII bytes are never a continuation byte
//! (0x80-0xBF) or a lead byte (0xC0-0xFF) of a multi-byte UTF-8 sequence, so
//! every position this module treats as a slice boundary is guaranteed to
//! already be a valid `char` boundary in a well-formed `&str`. Regular
//! (non-control) runs between control bytes are therefore always safe to
//! slice directly.

use std::fmt::Write as _;

/// Minimum byte savings (`original.len() - normalized.len()`) for
/// `project_messages` to accept a normalization candidate —
/// SPEC.md TR-4's "savings floor" knob, mirrored as
/// `ReductionPolicy::terminal_output_min_savings`. Exposed
/// here as the documented default; the policy field is what callers actually
/// tune.
pub const DEFAULT_MIN_SAVINGS: usize = 128;

/// Tool names T30/TR-4's candidate rule treats as "terminal/exec" — a result
/// from one of these is eligible for [`super::ReductionKind::OutputNormalized`].
/// Compared against the INVOKING tool call's function name (see
/// `detect_normalize_candidates`), the same pattern the A8 read-tool list
/// uses for read-type tools:
///
/// - `"bash"` — this SDK's own built-in (`tools/builtins.rs`'s
///   `BashTool::name`); `B6` must keep this in sync with any built-in tool
///   rename.
/// - `"shell"` — the other shell-tool name this SDK already anticipates for
///   embedder-registered tools (see `tools/mod.rs`'s
///   `shell_sandbox_unenforceable`, which checks the identical pair).
/// - `"exec_command"` — Codex's own native exec tool name, so a Codex log
///   loaded via `Session::from_codex` (whose `function_call`/
///   `function_call_output` records never carry a `ChatMessage::name` at
///   all — see `detect_normalize_candidates`'s doc comment) is covered too.
///
pub const NORMALIZE_TOOLS: &[&str] = &["bash", "shell", "exec_command"];

/// One simulated terminal row: a flat char buffer supporting index-based
/// overwrite (what CR/EL/cursor-up redraws need) without tracking style —
/// SGR is stripped at parse time, never simulated as row state.
type Row = Vec<char>;

/// The minimal line-buffer terminal state [`normalize`] drives.
struct Screen {
    rows: Vec<Row>,
    row: usize,
    col: usize,
}

impl Screen {
    fn new() -> Self {
        Screen {
            rows: vec![Vec::new()],
            row: 0,
            col: 0,
        }
    }

    /// Ensure row `r` exists, extending with empty rows as needed. Used by
    /// LF and CUD, both of which can move onto a row not yet materialized.
    fn ensure_row(&mut self, r: usize) {
        while self.rows.len() <= r {
            self.rows.push(Vec::new());
        }
    }

    /// Write one printable char at the cursor, overwriting in place (the
    /// redraw semantics this whole module exists for), padding with spaces
    /// if the cursor sits past the row's current end (e.g. after a
    /// cursor-up onto a shorter row). Then advances the cursor one column.
    fn write_char(&mut self, c: char) {
        let row = &mut self.rows[self.row];
        match self.col.cmp(&row.len()) {
            std::cmp::Ordering::Less => row[self.col] = c,
            std::cmp::Ordering::Equal => row.push(c),
            std::cmp::Ordering::Greater => {
                row.resize(self.col, ' ');
                row.push(c);
            }
        }
        self.col += 1;
    }

    /// Write a run of printable text (no control bytes) starting at the
    /// cursor — char-by-char, so multi-byte UTF-8 content (a spinner's
    /// Braille glyphs, non-ASCII build output) is never split.
    fn write_str(&mut self, s: &str) {
        for c in s.chars() {
            self.write_char(c);
        }
    }

    fn carriage_return(&mut self) {
        self.col = 0;
    }

    fn line_feed(&mut self) {
        self.row += 1;
        self.ensure_row(self.row);
        self.col = 0;
    }

    fn cursor_up(&mut self, n: usize) {
        self.row = self.row.saturating_sub(n);
    }

    fn cursor_down(&mut self, n: usize) {
        self.row = (self.row + n).min(self.rows.len().saturating_sub(1));
        self.ensure_row(self.row);
    }

    fn cursor_col_absolute(&mut self, n: usize) {
        // CHA is 1-based; column 0 is `n == 1`. `n == 0` is out of spec but
        // never guessed at — clamp to column 0 rather than underflowing.
        self.col = n.saturating_sub(1);
    }

    /// EL — erase in line. `param` is the parsed numeric argument (default
    /// `0` when absent, ECMA-48's own default for `K`).
    fn erase_line(&mut self, param: u32) {
        let row = &mut self.rows[self.row];
        match param {
            // 0: cursor to end of line.
            0 => row.truncate(self.col.min(row.len())),
            // 1: start of line to cursor, inclusive.
            1 => {
                let end = (self.col + 1).min(row.len());
                for cell in row.iter_mut().take(end) {
                    *cell = ' ';
                }
            }
            // 2 (or anything else we don't special-case): whole line.
            _ => row.clear(),
        }
    }

    /// Render the final settled content: one line per row, joined by `\n` —
    /// exactly what a plain-text capture of the same input (no CR/ESC at
    /// all) would already look like, which is what makes [`normalize`] a
    /// byte-exact no-op on plain output (SPEC.md TR-4 dev/04).
    fn render(&self) -> String {
        self.rows
            .iter()
            .map(|r| r.iter().collect::<String>())
            .collect::<Vec<_>>()
            .join("\n")
    }
}

/// Is `b` a CSI parameter byte (ECMA-48: `0x30..=0x3F`, i.e. digits, `;`,
/// `:`, `<`, `=`, `>`, `?`)?
fn is_csi_param_byte(b: u8) -> bool {
    (0x30..=0x3F).contains(&b)
}

/// Is `b` a CSI final byte (ECMA-48: `0x40..=0x7E`)?
fn is_csi_final_byte(b: u8) -> bool {
    (0x40..=0x7E).contains(&b)
}

/// Parse the (at most one) leading numeric parameter of a CSI param string,
/// ignoring everything after the first `;` (none of the sequences this
/// module simulates take more than one meaningful parameter) and any leading
/// `?` (DEC private-mode prefix, stripped by the caller's own dispatch, but
/// tolerated here too so a stray `?` never breaks the digit parse).
fn first_param(params: &str) -> Option<u32> {
    let digits: String = params
        .split(&[';', ':'][..])
        .next()
        .unwrap_or("")
        .chars()
        .filter(|c| c.is_ascii_digit())
        .collect();
    if digits.is_empty() {
        None
    } else {
        digits.parse().ok()
    }
}

/// Normalize `input`: strip ANSI SGR, simulate CR/EL/CUU/CUD/CHA redraws, and
/// return the final rendered text. Deterministic and pure — same bytes in,
/// byte-identical text out, every call (SPEC.md TR-4's determinism
/// requirement; see `tests::deterministic_across_repeated_runs`).
///
/// Never panics (see the module doc comment's safety note): a malformed or
/// truncated escape sequence is copied through as literal text rather than
/// ever indexing out of bounds or asserting on unexpected structure.
pub fn normalize(input: &str) -> String {
    let bytes = input.as_bytes();
    let mut screen = Screen::new();
    let mut i = 0usize;
    let n = bytes.len();

    while i < n {
        match bytes[i] {
            b'\r' => {
                screen.carriage_return();
                i += 1;
            }
            b'\n' => {
                screen.line_feed();
                i += 1;
            }
            0x1B => {
                // ESC. Every branch below either fully consumes a
                // recognized sequence, or falls back to copying whatever
                // bytes it looked at as literal text — there is no path
                // that advances `i` without having accounted for the bytes
                // in between.
                if i + 1 < n && bytes[i + 1] == b'[' {
                    i = consume_csi(input, &mut screen, i);
                } else if i + 1 < n && bytes[i + 1] == b']' {
                    i = consume_osc(input, &mut screen, i);
                } else {
                    // A bare ESC (not CSI/OSC), or ESC as the very last
                    // byte (truncated). Never guessed at: pass the ESC
                    // itself through as literal text; whatever follows (if
                    // anything) is reprocessed independently on the next
                    // loop iteration.
                    screen.write_char('\u{1B}');
                    i += 1;
                }
            }
            _ => {
                // A run of regular (non-control) text up to the next
                // control byte or end of input. Safe to slice directly —
                // see the module doc comment on ASCII control-byte
                // boundaries.
                let start = i;
                while i < n && !matches!(bytes[i], b'\r' | b'\n' | 0x1B) {
                    i += 1;
                }
                screen.write_str(&input[start..i]);
            }
        }
    }

    screen.render()
}

/// Consume one CSI sequence (`ESC [ params final`) starting at `esc_pos`
/// (the index of the `ESC` byte, with `bytes[esc_pos + 1] == b'['` already
/// verified by the caller). Dispatches recognized final bytes to `screen`;
/// anything else — including a sequence with no final byte before the input
/// ends (truncated) — is written through as literal text. Returns the index
/// to resume scanning from.
fn consume_csi(input: &str, screen: &mut Screen, esc_pos: usize) -> usize {
    let bytes = input.as_bytes();
    let n = bytes.len();
    let params_start = esc_pos + 2; // past `ESC [`
    let mut j = params_start;
    while j < n && is_csi_param_byte(bytes[j]) {
        j += 1;
    }
    if j >= n || !is_csi_final_byte(bytes[j]) {
        // No final byte found before the string ends: a truncated CSI
        // sequence (the TR-1-flagged 100KB-cap boundary case). Pass
        // everything from ESC to the end of input through verbatim and
        // stop — there is nothing left to parse.
        screen.write_str(&input[esc_pos..]);
        return n;
    }

    let params = &input[params_start..j];
    let final_byte = bytes[j];
    let private = params.starts_with('?');

    match final_byte {
        b'm' => {} // SGR: stripped, no rendered effect.
        b'K' => screen.erase_line(first_param(params).unwrap_or(0)),
        b'A' => screen.cursor_up(first_param(params).unwrap_or(1).max(1) as usize),
        b'B' => screen.cursor_down(first_param(params).unwrap_or(1).max(1) as usize),
        b'G' => screen.cursor_col_absolute(first_param(params).unwrap_or(1) as usize),
        b'h' | b'l' if private => {
            // DEC private mode set/reset (`?25l`/`?25h` cursor hide/show,
            // `?2004h/l` bracketed paste, etc.) — no rendered-content
            // effect for the modes real build tools use. See the module
            // doc comment's scoped carve-out.
        }
        _ => {
            // Recognized CSI *shape*, unrecognized final byte (cursor
            // position, device status report, erase-display, scroll,
            // ...). Never guessed at: the whole sequence, verbatim.
            screen.write_str(&input[esc_pos..=j]);
        }
    }
    j + 1
}

/// Consume one OSC sequence (`ESC ] ... (BEL | ESC \\)`) starting at
/// `esc_pos` (with `bytes[esc_pos + 1] == b']'` already verified). OSC
/// payloads (window title, etc.) never move the cursor or print visible
/// content themselves, but this module still does not special-case them —
/// "never guess" applies to properties like OSC-8 hyperlinks wrapping
/// visible text, which real build tools do not use but this module has no
/// way to rule out categorically. So: pass the whole sequence through
/// verbatim, same as any other unrecognized escape. An OSC with no
/// terminator before the input ends is likewise passed through verbatim to
/// the end (the truncated-sequence case). Returns the index to resume
/// scanning from.
fn consume_osc(input: &str, screen: &mut Screen, esc_pos: usize) -> usize {
    let bytes = input.as_bytes();
    let n = bytes.len();
    let mut j = esc_pos + 2; // past `ESC ]`
    while j < n {
        if bytes[j] == 0x07 {
            // BEL terminator, inclusive.
            screen.write_str(&input[esc_pos..=j]);
            return j + 1;
        }
        if bytes[j] == 0x1B && j + 1 < n && bytes[j + 1] == b'\\' {
            // ST (`ESC \`) terminator, inclusive.
            screen.write_str(&input[esc_pos..=(j + 1)]);
            return j + 2;
        }
        j += 1;
    }
    // Truncated: no terminator before the input ends.
    screen.write_str(&input[esc_pos..]);
    n
}

/// Format the honesty trailer's summary text (SPEC.md TR-4: "normalized text
/// must remain honest"): plain ASCII, one line, no `]` — same constraints
/// the reduction-stub formatter already enforces on every summary, so
/// this is folded into the shared `[sc-reduced output-normalized <id>: ...]`
/// grammar (see `reduce.rs`'s `OutputNormalized` candidate pass) rather than
/// a bespoke sentinel — that keeps the existing leak-guard (A11),
/// `stub::parse` (`sessions show-reductions`), and `Kind::from` dispatch all
/// working for this kind with no special-casing.
pub fn summary(original_bytes: usize, normalized_bytes: usize) -> String {
    let mut s = String::new();
    let _ = write!(
        s,
        "ANSI/redraw collapsed, {}B -> {}B - raw output in session sidecar",
        supercode_interchange::format_commas(original_bytes),
        supercode_interchange::format_commas(normalized_bytes),
    );
    s
}