Skip to main content

supercode_reduce/engine/
normalize.rs

1//! T30/TR-4 — terminal-noise normalization (`ReductionKind::OutputNormalized`):
2//! a small, deterministic line-buffer terminal simulator that collapses ANSI
3//! color/style codes and carriage-return/erase-line/cursor-up redraws down to
4//! the FINAL rendered content of each line — the same content a human
5//! watching the build would actually see, without the hundreds of
6//! intermediate redraws a captured progress bar otherwise leaves in the
7//! transcript.
8//!
9//! **Not a full `vte` emulation** (per SPEC.md TR-4's approach sketch): this
10//! supports exactly the sequences that dominate real `cargo`/`npm`/`pip`/
11//! `docker` output —
12//!
13//! - SGR (`ESC[...m`, colors/styles) — stripped; it never prints or moves.
14//! - CR (`\r`) — cursor to column 0 of the current row.
15//! - LF (`\n`) — cursor to column 0 of the NEXT row (a deliberate
16//!   simplification: real LF preserves column, but every real capture in
17//!   this codebase's fixtures pairs LF with either a preceding CR or content
18//!   that starts a fresh line anyway, so this never diverges from the
19//!   fixtures' actual rendering and keeps the model trivial to reason about).
20//! - EL (`ESC[K`, `ESC[0K`, `ESC[1K`, `ESC[2K`) — erase to end / to start /
21//!   whole line.
22//! - CUU / CUD (`ESC[<n>A` / `ESC[<n>B`) — cursor up/down `n` rows (the
23//!   multi-line redraw idiom `docker pull` uses for concurrent layers).
24//! - CHA (`ESC[<n>G`) — cursor to absolute column `n` (1-based); the idiom
25//!   modern `npm`'s spinner uses instead of `\r`.
26//! - DEC private mode set/reset (`ESC[?...h` / `ESC[?...l`) — e.g. `?25l`/
27//!   `?25h` (cursor hide/show around a spinner): dropped silently. This is a
28//!   deliberately narrow carve-out (real DEC private modes can do much more,
29//!   e.g. the alternate screen buffer) but build-tool output never uses those
30//!   — see the module-level safety note below.
31//!
32//! Everything else — including a CSI sequence with a final byte this module
33//! doesn't recognize (device status report, cursor-position, scroll, etc.),
34//! an OSC sequence, or any escape truncated by an upstream byte cap before
35//! its terminator — is passed through **verbatim, as literal printable
36//! text**, landing in the rendered output unchanged. "Never guess": an
37//! unrecognized sequence is never assumed to be a no-op, so its bytes are
38//! never silently dropped, and the reduction is content-preserving even for
39//! escape vocabulary this module has never seen.
40//!
41//! # Safety / no-panic guarantee
42//!
43//! [`normalize`] never panics on any input, including a `&str` truncated
44//! mid-escape-sequence (the TR-1 gotcha this module was warned about:
45//! `Agent::cap_tool_output`'s 100 KB history cap can slice a raw tool output
46//! anywhere, including through the middle of a CSI/OSC sequence, before this
47//! module ever sees it). A truncated sequence at the end of the input is
48//! detected (no terminator found before the string ends) and copied through
49//! as literal text, same as any other unrecognized sequence — never a panic,
50//! never an out-of-bounds slice. See `tests::never_panics_on_malformed_input`
51//! for a sweep over adversarial byte patterns (including sequences chopped at
52//! every possible byte boundary).
53//!
54//! All scanning here is on `&str` byte offsets, but every control byte this
55//! module inspects (`ESC` 0x1B, `CR` 0x0D, `LF` 0x0A, CSI param/final bytes
56//! 0x20-0x7E) is ASCII — and ASCII bytes are never a continuation byte
57//! (0x80-0xBF) or a lead byte (0xC0-0xFF) of a multi-byte UTF-8 sequence, so
58//! every position this module treats as a slice boundary is guaranteed to
59//! already be a valid `char` boundary in a well-formed `&str`. Regular
60//! (non-control) runs between control bytes are therefore always safe to
61//! slice directly.
62
63use std::fmt::Write as _;
64
65/// Minimum byte savings (`original.len() - normalized.len()`) for
66/// `project_messages` to accept a normalization candidate —
67/// SPEC.md TR-4's "savings floor" knob, mirrored as
68/// `ReductionPolicy::terminal_output_min_savings`. Exposed
69/// here as the documented default; the policy field is what callers actually
70/// tune.
71pub const DEFAULT_MIN_SAVINGS: usize = 128;
72
73/// Tool names T30/TR-4's candidate rule treats as "terminal/exec" — a result
74/// from one of these is eligible for [`super::ReductionKind::OutputNormalized`].
75/// Compared against the INVOKING tool call's function name (see
76/// `detect_normalize_candidates`), the same pattern the A8 read-tool list
77/// uses for read-type tools:
78///
79/// - `"bash"` — this SDK's own built-in (`tools/builtins.rs`'s
80///   `BashTool::name`); `B6` must keep this in sync with any built-in tool
81///   rename.
82/// - `"shell"` — the other shell-tool name this SDK already anticipates for
83///   embedder-registered tools (see `tools/mod.rs`'s
84///   `shell_sandbox_unenforceable`, which checks the identical pair).
85/// - `"exec_command"` — Codex's own native exec tool name, so a Codex log
86///   loaded via `Session::from_codex` (whose `function_call`/
87///   `function_call_output` records never carry a `ChatMessage::name` at
88///   all — see `detect_normalize_candidates`'s doc comment) is covered too.
89///
90pub const NORMALIZE_TOOLS: &[&str] = &["bash", "shell", "exec_command"];
91
92/// One simulated terminal row: a flat char buffer supporting index-based
93/// overwrite (what CR/EL/cursor-up redraws need) without tracking style —
94/// SGR is stripped at parse time, never simulated as row state.
95type Row = Vec<char>;
96
97/// The minimal line-buffer terminal state [`normalize`] drives.
98struct Screen {
99    rows: Vec<Row>,
100    row: usize,
101    col: usize,
102}
103
104impl Screen {
105    fn new() -> Self {
106        Screen {
107            rows: vec![Vec::new()],
108            row: 0,
109            col: 0,
110        }
111    }
112
113    /// Ensure row `r` exists, extending with empty rows as needed. Used by
114    /// LF and CUD, both of which can move onto a row not yet materialized.
115    fn ensure_row(&mut self, r: usize) {
116        while self.rows.len() <= r {
117            self.rows.push(Vec::new());
118        }
119    }
120
121    /// Write one printable char at the cursor, overwriting in place (the
122    /// redraw semantics this whole module exists for), padding with spaces
123    /// if the cursor sits past the row's current end (e.g. after a
124    /// cursor-up onto a shorter row). Then advances the cursor one column.
125    fn write_char(&mut self, c: char) {
126        let row = &mut self.rows[self.row];
127        match self.col.cmp(&row.len()) {
128            std::cmp::Ordering::Less => row[self.col] = c,
129            std::cmp::Ordering::Equal => row.push(c),
130            std::cmp::Ordering::Greater => {
131                row.resize(self.col, ' ');
132                row.push(c);
133            }
134        }
135        self.col += 1;
136    }
137
138    /// Write a run of printable text (no control bytes) starting at the
139    /// cursor — char-by-char, so multi-byte UTF-8 content (a spinner's
140    /// Braille glyphs, non-ASCII build output) is never split.
141    fn write_str(&mut self, s: &str) {
142        for c in s.chars() {
143            self.write_char(c);
144        }
145    }
146
147    fn carriage_return(&mut self) {
148        self.col = 0;
149    }
150
151    fn line_feed(&mut self) {
152        self.row += 1;
153        self.ensure_row(self.row);
154        self.col = 0;
155    }
156
157    fn cursor_up(&mut self, n: usize) {
158        self.row = self.row.saturating_sub(n);
159    }
160
161    fn cursor_down(&mut self, n: usize) {
162        self.row = (self.row + n).min(self.rows.len().saturating_sub(1));
163        self.ensure_row(self.row);
164    }
165
166    fn cursor_col_absolute(&mut self, n: usize) {
167        // CHA is 1-based; column 0 is `n == 1`. `n == 0` is out of spec but
168        // never guessed at — clamp to column 0 rather than underflowing.
169        self.col = n.saturating_sub(1);
170    }
171
172    /// EL — erase in line. `param` is the parsed numeric argument (default
173    /// `0` when absent, ECMA-48's own default for `K`).
174    fn erase_line(&mut self, param: u32) {
175        let row = &mut self.rows[self.row];
176        match param {
177            // 0: cursor to end of line.
178            0 => row.truncate(self.col.min(row.len())),
179            // 1: start of line to cursor, inclusive.
180            1 => {
181                let end = (self.col + 1).min(row.len());
182                for cell in row.iter_mut().take(end) {
183                    *cell = ' ';
184                }
185            }
186            // 2 (or anything else we don't special-case): whole line.
187            _ => row.clear(),
188        }
189    }
190
191    /// Render the final settled content: one line per row, joined by `\n` —
192    /// exactly what a plain-text capture of the same input (no CR/ESC at
193    /// all) would already look like, which is what makes [`normalize`] a
194    /// byte-exact no-op on plain output (SPEC.md TR-4 dev/04).
195    fn render(&self) -> String {
196        self.rows
197            .iter()
198            .map(|r| r.iter().collect::<String>())
199            .collect::<Vec<_>>()
200            .join("\n")
201    }
202}
203
204/// Is `b` a CSI parameter byte (ECMA-48: `0x30..=0x3F`, i.e. digits, `;`,
205/// `:`, `<`, `=`, `>`, `?`)?
206fn is_csi_param_byte(b: u8) -> bool {
207    (0x30..=0x3F).contains(&b)
208}
209
210/// Is `b` a CSI final byte (ECMA-48: `0x40..=0x7E`)?
211fn is_csi_final_byte(b: u8) -> bool {
212    (0x40..=0x7E).contains(&b)
213}
214
215/// Parse the (at most one) leading numeric parameter of a CSI param string,
216/// ignoring everything after the first `;` (none of the sequences this
217/// module simulates take more than one meaningful parameter) and any leading
218/// `?` (DEC private-mode prefix, stripped by the caller's own dispatch, but
219/// tolerated here too so a stray `?` never breaks the digit parse).
220fn first_param(params: &str) -> Option<u32> {
221    let digits: String = params
222        .split(&[';', ':'][..])
223        .next()
224        .unwrap_or("")
225        .chars()
226        .filter(|c| c.is_ascii_digit())
227        .collect();
228    if digits.is_empty() {
229        None
230    } else {
231        digits.parse().ok()
232    }
233}
234
235/// Normalize `input`: strip ANSI SGR, simulate CR/EL/CUU/CUD/CHA redraws, and
236/// return the final rendered text. Deterministic and pure — same bytes in,
237/// byte-identical text out, every call (SPEC.md TR-4's determinism
238/// requirement; see `tests::deterministic_across_repeated_runs`).
239///
240/// Never panics (see the module doc comment's safety note): a malformed or
241/// truncated escape sequence is copied through as literal text rather than
242/// ever indexing out of bounds or asserting on unexpected structure.
243pub fn normalize(input: &str) -> String {
244    let bytes = input.as_bytes();
245    let mut screen = Screen::new();
246    let mut i = 0usize;
247    let n = bytes.len();
248
249    while i < n {
250        match bytes[i] {
251            b'\r' => {
252                screen.carriage_return();
253                i += 1;
254            }
255            b'\n' => {
256                screen.line_feed();
257                i += 1;
258            }
259            0x1B => {
260                // ESC. Every branch below either fully consumes a
261                // recognized sequence, or falls back to copying whatever
262                // bytes it looked at as literal text — there is no path
263                // that advances `i` without having accounted for the bytes
264                // in between.
265                if i + 1 < n && bytes[i + 1] == b'[' {
266                    i = consume_csi(input, &mut screen, i);
267                } else if i + 1 < n && bytes[i + 1] == b']' {
268                    i = consume_osc(input, &mut screen, i);
269                } else {
270                    // A bare ESC (not CSI/OSC), or ESC as the very last
271                    // byte (truncated). Never guessed at: pass the ESC
272                    // itself through as literal text; whatever follows (if
273                    // anything) is reprocessed independently on the next
274                    // loop iteration.
275                    screen.write_char('\u{1B}');
276                    i += 1;
277                }
278            }
279            _ => {
280                // A run of regular (non-control) text up to the next
281                // control byte or end of input. Safe to slice directly —
282                // see the module doc comment on ASCII control-byte
283                // boundaries.
284                let start = i;
285                while i < n && !matches!(bytes[i], b'\r' | b'\n' | 0x1B) {
286                    i += 1;
287                }
288                screen.write_str(&input[start..i]);
289            }
290        }
291    }
292
293    screen.render()
294}
295
296/// Consume one CSI sequence (`ESC [ params final`) starting at `esc_pos`
297/// (the index of the `ESC` byte, with `bytes[esc_pos + 1] == b'['` already
298/// verified by the caller). Dispatches recognized final bytes to `screen`;
299/// anything else — including a sequence with no final byte before the input
300/// ends (truncated) — is written through as literal text. Returns the index
301/// to resume scanning from.
302fn consume_csi(input: &str, screen: &mut Screen, esc_pos: usize) -> usize {
303    let bytes = input.as_bytes();
304    let n = bytes.len();
305    let params_start = esc_pos + 2; // past `ESC [`
306    let mut j = params_start;
307    while j < n && is_csi_param_byte(bytes[j]) {
308        j += 1;
309    }
310    if j >= n || !is_csi_final_byte(bytes[j]) {
311        // No final byte found before the string ends: a truncated CSI
312        // sequence (the TR-1-flagged 100KB-cap boundary case). Pass
313        // everything from ESC to the end of input through verbatim and
314        // stop — there is nothing left to parse.
315        screen.write_str(&input[esc_pos..]);
316        return n;
317    }
318
319    let params = &input[params_start..j];
320    let final_byte = bytes[j];
321    let private = params.starts_with('?');
322
323    match final_byte {
324        b'm' => {} // SGR: stripped, no rendered effect.
325        b'K' => screen.erase_line(first_param(params).unwrap_or(0)),
326        b'A' => screen.cursor_up(first_param(params).unwrap_or(1).max(1) as usize),
327        b'B' => screen.cursor_down(first_param(params).unwrap_or(1).max(1) as usize),
328        b'G' => screen.cursor_col_absolute(first_param(params).unwrap_or(1) as usize),
329        b'h' | b'l' if private => {
330            // DEC private mode set/reset (`?25l`/`?25h` cursor hide/show,
331            // `?2004h/l` bracketed paste, etc.) — no rendered-content
332            // effect for the modes real build tools use. See the module
333            // doc comment's scoped carve-out.
334        }
335        _ => {
336            // Recognized CSI *shape*, unrecognized final byte (cursor
337            // position, device status report, erase-display, scroll,
338            // ...). Never guessed at: the whole sequence, verbatim.
339            screen.write_str(&input[esc_pos..=j]);
340        }
341    }
342    j + 1
343}
344
345/// Consume one OSC sequence (`ESC ] ... (BEL | ESC \\)`) starting at
346/// `esc_pos` (with `bytes[esc_pos + 1] == b']'` already verified). OSC
347/// payloads (window title, etc.) never move the cursor or print visible
348/// content themselves, but this module still does not special-case them —
349/// "never guess" applies to properties like OSC-8 hyperlinks wrapping
350/// visible text, which real build tools do not use but this module has no
351/// way to rule out categorically. So: pass the whole sequence through
352/// verbatim, same as any other unrecognized escape. An OSC with no
353/// terminator before the input ends is likewise passed through verbatim to
354/// the end (the truncated-sequence case). Returns the index to resume
355/// scanning from.
356fn consume_osc(input: &str, screen: &mut Screen, esc_pos: usize) -> usize {
357    let bytes = input.as_bytes();
358    let n = bytes.len();
359    let mut j = esc_pos + 2; // past `ESC ]`
360    while j < n {
361        if bytes[j] == 0x07 {
362            // BEL terminator, inclusive.
363            screen.write_str(&input[esc_pos..=j]);
364            return j + 1;
365        }
366        if bytes[j] == 0x1B && j + 1 < n && bytes[j + 1] == b'\\' {
367            // ST (`ESC \`) terminator, inclusive.
368            screen.write_str(&input[esc_pos..=(j + 1)]);
369            return j + 2;
370        }
371        j += 1;
372    }
373    // Truncated: no terminator before the input ends.
374    screen.write_str(&input[esc_pos..]);
375    n
376}
377
378/// Format the honesty trailer's summary text (SPEC.md TR-4: "normalized text
379/// must remain honest"): plain ASCII, one line, no `]` — same constraints
380/// the reduction-stub formatter already enforces on every summary, so
381/// this is folded into the shared `[sc-reduced output-normalized <id>: ...]`
382/// grammar (see `reduce.rs`'s `OutputNormalized` candidate pass) rather than
383/// a bespoke sentinel — that keeps the existing leak-guard (A11),
384/// `stub::parse` (`sessions show-reductions`), and `Kind::from` dispatch all
385/// working for this kind with no special-casing.
386pub fn summary(original_bytes: usize, normalized_bytes: usize) -> String {
387    let mut s = String::new();
388    let _ = write!(
389        s,
390        "ANSI/redraw collapsed, {}B -> {}B - raw output in session sidecar",
391        supercode_interchange::format_commas(original_bytes),
392        supercode_interchange::format_commas(normalized_bytes),
393    );
394    s
395}