Skip to main content

supercode_frontend_tui/foundation/
table_detect.rs

1// Derived from OpenAI Codex: codex-rs/tui/src/table_detect.rs
2// Pinned source: 8604689ec5e3437eb79802d8d72249b7722fbf5b
3// Copyright 2025 OpenAI
4// Licensed under the Apache License, Version 2.0.
5// Modified by the Supercode contributors; see docs/legal/codex-frontend-extraction.toml.
6
7//! Canonical pipe-table structure detection and fenced-code-block tracking for
8//! raw markdown source.
9//!
10//! Both the streaming controller (`streaming/controller.rs`) and the
11//! markdown-fence unwrapper (`markdown.rs`) need to identify pipe-table
12//! structure and fenced code blocks in raw markdown source.  This module
13//! provides the canonical implementations so fixes only need to happen in one
14//! place.
15//!
16//! ## Concepts
17//!
18//! A GFM pipe table is a sequence of lines where:
19//! - A **header line** contains pipe-separated segments with at least one
20//!   non-empty cell.
21//! - A **delimiter line** immediately follows the header and contains only
22//!   alignment markers (`---`, `:---`, `---:`, `:---:`), each with at least
23//!   three dashes.
24//! - **Body rows** follow the delimiter.
25//!
26//! A **fenced code block** starts with 3+ backticks or tildes and ends with a
27//! matching close marker.  [`FenceTracker`] classifies each line as
28//! [`FenceKind::Outside`], [`FenceKind::Markdown`], or [`FenceKind::Other`]
29//! so callers can skip pipe characters that appear inside non-markdown fences.
30//!
31//! The table functions operate on single lines and do not maintain cross-line
32//! state.  Callers (the streaming controller and fence unwrapper) are
33//! responsible for pairing consecutive lines to confirm a table.
34
35/// Split a pipe-delimited line into trimmed segments.
36///
37/// Returns `None` if the line is empty or has no unescaped separator marker.
38/// Leading/trailing pipes are stripped before splitting.
39///
40/// This is intentionally a structural parser, not a renderer. It preserves
41/// escaped pipes inside the returned segments because callers only care about
42/// whether the line can participate in a table, not how the cell text should
43/// finally be displayed.
44pub fn parse_table_segments(line: &str) -> Option<Vec<&str>> {
45    let trimmed = line.trim();
46    if trimmed.is_empty() {
47        return None;
48    }
49
50    let has_outer_pipe = trimmed.starts_with('|') || trimmed.ends_with('|');
51    let content = trimmed.strip_prefix('|').unwrap_or(trimmed);
52    let content = content.strip_suffix('|').unwrap_or(content);
53    let raw_segments = split_unescaped_pipe(content);
54    if !has_outer_pipe && raw_segments.len() <= 1 {
55        return None;
56    }
57
58    let segments: Vec<&str> = raw_segments.into_iter().map(str::trim).collect();
59    (!segments.is_empty()).then_some(segments)
60}
61
62/// Split `content` on unescaped `|` characters.
63///
64/// A pipe preceded by `\` is treated as literal text, not a column separator.
65/// The backslash remains in the segment (this is structure detection, not
66/// rendering).
67fn split_unescaped_pipe(content: &str) -> Vec<&str> {
68    let mut segments = Vec::with_capacity(8);
69    let mut start = 0;
70    let bytes = content.as_bytes();
71    let mut i = 0;
72    while i < bytes.len() {
73        if bytes[i] == b'\\' {
74            // Skip the escaped character.
75            i += 2;
76        } else if bytes[i] == b'|' {
77            segments.push(&content[start..i]);
78            start = i + 1;
79            i += 1;
80        } else {
81            i += 1;
82        }
83    }
84    segments.push(&content[start..]);
85    segments
86}
87
88// Small table-detection helpers inlined for the streaming hot path — they are
89// called on every source line during incremental holdback scanning.
90
91/// Whether `line` looks like a table header row (has pipe-separated
92/// segments with at least one non-empty cell).
93#[inline]
94pub fn is_table_header_line(line: &str) -> bool {
95    parse_table_segments(line).is_some_and(|segments| segments.iter().any(|s| !s.is_empty()))
96}
97
98/// Whether a single segment matches the `---`, `:---`, `---:`, or `:---:`
99/// alignment-colon syntax used in markdown table delimiter rows.
100#[inline]
101fn is_table_delimiter_segment(segment: &str) -> bool {
102    let trimmed = segment.trim();
103    if trimmed.is_empty() {
104        return false;
105    }
106    let without_leading = trimmed.strip_prefix(':').unwrap_or(trimmed);
107    let without_ends = without_leading.strip_suffix(':').unwrap_or(without_leading);
108    without_ends.len() >= 3 && without_ends.chars().all(|c| c == '-')
109}
110
111/// Whether `line` is a valid table delimiter row (every segment passes
112/// [`is_table_delimiter_segment`]).
113#[inline]
114pub fn is_table_delimiter_line(line: &str) -> bool {
115    parse_table_segments(line)
116        .is_some_and(|segments| segments.into_iter().all(is_table_delimiter_segment))
117}
118
119// ---------------------------------------------------------------------------
120// Fenced code block tracking
121// ---------------------------------------------------------------------------
122
123/// Where a source line sits relative to fenced code blocks.
124///
125/// Table holdback only applies to lines that are `Outside` or inside a
126/// `Markdown` fence. Lines inside `Other` fences (e.g. `sh`, `rust`) are
127/// ignored by the table scanner because their pipe characters are code, not
128/// table syntax.
129#[derive(Clone, Copy, Debug, PartialEq, Eq)]
130pub enum FenceKind {
131    /// Not inside any fenced code block.
132    Outside,
133    /// Inside a `` ```md `` or `` ```markdown `` fence.
134    Markdown,
135    /// Inside a fence with a non-markdown info string.
136    Other,
137}
138
139/// Incremental tracker for fenced-code-block open/close transitions.
140///
141/// Feed lines one at a time via [`advance`](Self::advance); query the current
142/// context with [`kind`](Self::kind).  The tracker handles leading-whitespace
143/// limits (>3 spaces → not a fence), blockquote prefix stripping, and
144/// backtick/tilde marker matching.
145///
146/// The tracker reports the fence context that applies to the current line
147/// before that line mutates the state. Callers rely on that when deciding
148/// whether the current raw line can open or continue a table.
149pub struct FenceTracker {
150    state: Option<(char, usize, FenceKind)>,
151}
152
153impl Default for FenceTracker {
154    fn default() -> Self {
155        Self::new()
156    }
157}
158
159impl FenceTracker {
160    #[inline]
161    pub fn new() -> Self {
162        Self { state: None }
163    }
164
165    /// Process one raw source line and update fence state.
166    ///
167    /// Lines with >3 leading spaces are ignored (indented code blocks, not
168    /// fences).  Blockquote prefixes (`>`) are stripped before scanning.
169    pub fn advance(&mut self, raw_line: &str) {
170        let leading_spaces = raw_line
171            .as_bytes()
172            .iter()
173            .take_while(|byte| **byte == b' ')
174            .count();
175        if leading_spaces > 3 {
176            return;
177        }
178
179        let trimmed = &raw_line[leading_spaces..];
180        let fence_scan_text = strip_blockquote_prefix(trimmed);
181        if let Some((marker, len)) = parse_fence_marker(fence_scan_text) {
182            if let Some((open_char, open_len, _)) = self.state {
183                // Close the current fence if the marker matches.
184                if marker == open_char
185                    && len >= open_len
186                    && fence_scan_text[len..].trim().is_empty()
187                {
188                    self.state = None;
189                }
190            } else {
191                // Opening a new fence.
192                let kind = if is_markdown_fence_info(fence_scan_text, len) {
193                    FenceKind::Markdown
194                } else {
195                    FenceKind::Other
196                };
197                self.state = Some((marker, len, kind));
198            }
199        }
200    }
201
202    /// Current fence context for the most-recently-advanced line.
203    #[inline]
204    pub fn kind(&self) -> FenceKind {
205        self.state.map_or(FenceKind::Outside, |(_, _, k)| k)
206    }
207}
208
209/// Return fence marker character and run length for a potential fence line.
210///
211/// Recognises backtick and tilde fences with a minimum run of 3.
212/// The input should already have leading whitespace and blockquote prefixes
213/// stripped.
214#[inline]
215pub fn parse_fence_marker(line: &str) -> Option<(char, usize)> {
216    let first = line.as_bytes().first().copied()?;
217    if first != b'`' && first != b'~' {
218        return None;
219    }
220    let len = line.bytes().take_while(|&b| b == first).count();
221    if len < 3 {
222        return None;
223    }
224    Some((first as char, len))
225}
226
227/// Whether the info string after a fence marker indicates markdown content.
228///
229/// Matches `md` and `markdown` (case-insensitive).
230#[inline]
231pub fn is_markdown_fence_info(trimmed_line: &str, marker_len: usize) -> bool {
232    let info = trimmed_line[marker_len..]
233        .split_whitespace()
234        .next()
235        .unwrap_or_default();
236    info.eq_ignore_ascii_case("md") || info.eq_ignore_ascii_case("markdown")
237}
238
239/// Peel all leading `>` blockquote markers from a line.
240///
241/// Tables can appear inside blockquotes (`> | A | B |`), so the holdback
242/// scanner must strip these markers before checking for table syntax.
243#[inline]
244pub fn strip_blockquote_prefix(line: &str) -> &str {
245    let mut rest = line.trim_start();
246    loop {
247        let Some(stripped) = rest.strip_prefix('>') else {
248            return rest;
249        };
250        rest = stripped.strip_prefix(' ').unwrap_or(stripped).trim_start();
251    }
252}