Skip to main content

mobench_report/
lib.rs

1//! Canonical Mobench report models and context-specific renderers.
2//!
3//! This crate is the report Module for Mobench. Provider and command Modules
4//! produce typed report data; JSON, Markdown, CSV, JUnit, comparison, and CI
5//! adapters render that data without reimplementing report semantics.
6
7mod github;
8mod model;
9mod render;
10
11pub use github::{
12    CheckRunAnnotation, CheckRunOutput, CheckRunRequest, GITHUB_CHECK_ANNOTATION_LIMIT,
13    render_sticky_comment,
14};
15
16pub use model::{
17    BenchmarkFailureStats, BenchmarkResourceUsage, BenchmarkStats, CanonicalSummaryV2,
18    CompareReport, CompareRow, DeviceSummary, RegressionFinding, RunOutcome, SummaryReport,
19};
20pub use render::{
21    MEMORY_BASELINE_GAP_NOTE, compare_summaries, comparison_json, detect_regressions,
22    format_cpu_total_duration_ms, format_duration_smart, format_failure_elapsed_ms, format_ms,
23    render_compare_markdown, render_csv_summary, render_junit_report, render_markdown_summary,
24};
25
26/// Encode untrusted text for an inline Markdown context, including headings.
27///
28/// The returned text renders as the original plain text while Markdown and
29/// HTML parsers see only inert character references. Line breaks are folded
30/// to spaces so the value cannot open a new block-level construct.
31#[must_use]
32pub fn markdown_inline_text(input: &str) -> String {
33    let mut encoded = String::with_capacity(input.len());
34    let mut chars = input.chars().peekable();
35
36    while let Some(ch) = chars.next() {
37        if ch == '\r' {
38            if chars.peek() == Some(&'\n') {
39                chars.next();
40            }
41            encoded.push(' ');
42            continue;
43        }
44        if ch == '\n' {
45            encoded.push(' ');
46            continue;
47        }
48        if ch.is_control() {
49            encoded.push(' ');
50            continue;
51        }
52
53        encoded.push_str(match ch {
54            '&' => "&",
55            '<' => "&lt;",
56            '>' => "&gt;",
57            '!' => "&#33;",
58            '"' => "&#34;",
59            '#' => "&#35;",
60            '(' => "&#40;",
61            ')' => "&#41;",
62            '*' => "&#42;",
63            '+' => "&#43;",
64            '-' => "&#45;",
65            '.' => "&#46;",
66            '/' => "&#47;",
67            ':' => "&#58;",
68            '@' => "&#64;",
69            '[' => "&#91;",
70            '\\' => "&#92;",
71            ']' => "&#93;",
72            '_' => "&#95;",
73            '`' => "&#96;",
74            '{' => "&#123;",
75            '|' => "&#124;",
76            '}' => "&#125;",
77            '~' => "&#126;",
78            _ => {
79                encoded.push(ch);
80                continue;
81            }
82        });
83    }
84
85    encoded
86}
87
88fn markdown_text_requires_encoding(input: &str) -> bool {
89    let lower = input.to_ascii_lowercase();
90    lower.contains("http://")
91        || lower.contains("https://")
92        || lower.contains("mailto:")
93        || contains_markdown_underscore_delimiter(input)
94        || input.chars().any(|ch| {
95            ch.is_control()
96                || matches!(
97                    ch,
98                    '&' | '<'
99                        | '>'
100                        | '!'
101                        | '#'
102                        | '('
103                        | ')'
104                        | '*'
105                        | '['
106                        | '\\'
107                        | ']'
108                        | '`'
109                        | '{'
110                        | '|'
111                        | '}'
112                        | '~'
113                )
114        })
115}
116
117fn contains_markdown_underscore_delimiter(input: &str) -> bool {
118    let chars = input.chars().collect::<Vec<_>>();
119    chars.iter().enumerate().any(|(index, ch)| {
120        *ch == '_'
121            && (index == 0
122                || index + 1 == chars.len()
123                || !chars[index - 1].is_alphanumeric()
124                || !chars[index + 1].is_alphanumeric())
125    })
126}
127
128/// Encode untrusted text for a GitHub-flavored Markdown table cell.
129///
130/// This deliberately has a separate interface from [`markdown_inline_text`]
131/// so table-specific behavior can evolve without callers choosing a generic
132/// escape function. The current encoding is equally strict in both contexts.
133#[must_use]
134pub fn markdown_table_cell_text(input: &str) -> String {
135    markdown_inline_text(input)
136}
137
138/// Preserve benign report fields while encoding Markdown-active input.
139///
140/// This compatibility-oriented wrapper keeps ordinary legacy report values
141/// byte-for-byte stable. Values containing Markdown, HTML, control, or
142/// autolink syntax delegate to the strict inline encoder.
143#[must_use]
144pub fn markdown_inline_field_text(input: &str) -> String {
145    if markdown_text_requires_encoding(input) {
146        markdown_inline_text(input)
147    } else {
148        input.to_string()
149    }
150}
151
152/// Wrap untrusted text in a safe inline-code span.
153///
154/// Ordinary values retain the legacy single-backtick representation. Control
155/// characters and line endings are folded to spaces, and values containing
156/// backticks use a delimiter longer than every run in the value. Padding keeps
157/// the delimiter separate from a leading or trailing backtick; CommonMark
158/// removes that padding when the span is rendered.
159#[must_use]
160pub fn markdown_inline_code(input: &str) -> String {
161    let mut normalized = String::with_capacity(input.len());
162    let mut chars = input.chars().peekable();
163    while let Some(ch) = chars.next() {
164        if ch == '\r' {
165            if chars.peek() == Some(&'\n') {
166                chars.next();
167            }
168            normalized.push(' ');
169        } else if ch == '\n' || ch.is_control() {
170            normalized.push(' ');
171        } else {
172            normalized.push(ch);
173        }
174    }
175
176    let max_backtick_run = normalized
177        .split(|ch| ch != '`')
178        .map(str::len)
179        .max()
180        .unwrap_or(0);
181    if max_backtick_run == 0 {
182        if normalized.is_empty() {
183            return "` `".to_string();
184        }
185        return format!("`{normalized}`");
186    }
187
188    let delimiter = "`".repeat(max_backtick_run + 1);
189    format!("{delimiter} {normalized} {delimiter}")
190}
191
192/// Preserve benign table fields while encoding Markdown-active input.
193///
194/// This is the table-context counterpart to [`markdown_inline_field_text`].
195#[must_use]
196pub fn markdown_table_field_text(input: &str) -> String {
197    if markdown_text_requires_encoding(input) {
198        markdown_table_cell_text(input)
199    } else {
200        input.to_string()
201    }
202}
203
204/// Encode an untrusted relative path for a Markdown link destination.
205///
206/// RFC 3986 unreserved bytes and non-leading path separators are preserved so
207/// normal artifact links stay readable. All other bytes are percent-encoded;
208/// leading separators are encoded to prevent protocol-relative destinations.
209#[must_use]
210pub fn markdown_link_destination(input: &str) -> String {
211    const HEX: &[u8; 16] = b"0123456789ABCDEF";
212
213    let bytes = input.as_bytes();
214    let leading_slashes = bytes.iter().take_while(|byte| **byte == b'/').count();
215    let mut encoded = String::with_capacity(bytes.len());
216
217    for (index, byte) in bytes.iter().copied().enumerate() {
218        let is_unreserved =
219            byte.is_ascii_alphanumeric() || matches!(byte, b'-' | b'.' | b'_' | b'~');
220        let is_safe_separator = byte == b'/' && index >= leading_slashes;
221        if is_unreserved || is_safe_separator {
222            encoded.push(char::from(byte));
223        } else {
224            encoded.push('%');
225            encoded.push(char::from(HEX[usize::from(byte >> 4)]));
226            encoded.push(char::from(HEX[usize::from(byte & 0x0f)]));
227        }
228    }
229
230    encoded
231}
232
233/// Encode one untrusted CSV field according to RFC 4180.
234///
235/// Fields containing commas, quotes, CR, or LF are double-quoted and embedded
236/// quotes are doubled. To prevent spreadsheet formula execution, an apostrophe
237/// is prefixed when the first non-whitespace, non-control character is one of
238/// `=`, `+`, `-`, or `@`. Prefixing at byte zero also closes leading
239/// whitespace/control-character bypasses while preserving the original text.
240#[must_use]
241pub fn csv_field(input: &str) -> String {
242    let formula_like = input
243        .chars()
244        .find(|ch| !ch.is_whitespace() && !ch.is_control())
245        .is_some_and(|ch| matches!(ch, '=' | '+' | '-' | '@'));
246
247    let neutralized = if formula_like {
248        let mut value = String::with_capacity(input.len() + 1);
249        value.push('\'');
250        value.push_str(input);
251        value
252    } else {
253        input.to_string()
254    };
255
256    if !neutralized
257        .chars()
258        .any(|ch| matches!(ch, ',' | '"' | '\r' | '\n'))
259    {
260        return neutralized;
261    }
262
263    let mut encoded = String::with_capacity(neutralized.len() + 2);
264    encoded.push('"');
265    for ch in neutralized.chars() {
266        if ch == '"' {
267            encoded.push('"');
268        }
269        encoded.push(ch);
270    }
271    encoded.push('"');
272    encoded
273}
274
275#[cfg(test)]
276mod tests {
277    use super::*;
278
279    const ADVERSARIAL_MARKDOWN_CORPUS: &[&str] = &[
280        "# heading\n> quote\n- list item",
281        "**bold** _emphasis_ `code` ~~strike~~",
282        "[link](https://evil.invalid) ![image](x) <script>alert(1)</script>",
283        "left|right\r\nmailto:owner@example.com\\*escaped*",
284    ];
285
286    #[test]
287    fn inline_text_neutralizes_markdown_html_and_line_breaks() {
288        let adversarial =
289            "# heading\r\n![image](https://evil.invalid/x) | <script>*owned*</script>";
290
291        let encoded = markdown_inline_text(adversarial);
292
293        assert_eq!(
294            encoded,
295            "&#35; heading &#33;&#91;image&#93;&#40;https&#58;&#47;&#47;evil&#46;invalid&#47;x&#41; &#124; &lt;script&gt;&#42;owned&#42;&lt;&#47;script&gt;"
296        );
297    }
298
299    #[test]
300    fn table_cell_text_cannot_break_the_row_or_create_links() {
301        let adversarial = "left|right\n[link](mailto:owner@example.com) <img src=x>";
302
303        let encoded = markdown_table_cell_text(adversarial);
304
305        assert_eq!(
306            encoded,
307            "left&#124;right &#91;link&#93;&#40;mailto&#58;owner&#64;example&#46;com&#41; &lt;img src=x&gt;"
308        );
309    }
310
311    #[test]
312    fn both_context_encoders_neutralize_the_adversarial_corpus() {
313        for adversarial in ADVERSARIAL_MARKDOWN_CORPUS {
314            for encoded in [
315                markdown_inline_text(adversarial),
316                markdown_table_cell_text(adversarial),
317            ] {
318                for raw_syntax in [
319                    "\r", "\n", "<", ">", "|", "[", "]", "(", ")", "!", "*", "_", "`", "~", "\\",
320                    "https://", "mailto:",
321                ] {
322                    assert!(
323                        !encoded.contains(raw_syntax),
324                        "encoded output retained `{raw_syntax}` from `{adversarial}`: {encoded}"
325                    );
326                }
327            }
328        }
329    }
330
331    #[test]
332    fn field_encoders_preserve_benign_report_text_exactly() {
333        let benign = [
334            "2026-03-26T00:00:00Z",
335            "Google Pixel 8-14.0",
336            "basic_benchmark::bench_fibonacci",
337            "plots/alpha-ios.svg",
338        ];
339
340        for input in benign {
341            assert_eq!(markdown_inline_field_text(input), input);
342            assert_eq!(markdown_table_field_text(input), input);
343        }
344    }
345
346    #[test]
347    fn field_encoders_neutralize_underscore_emphasis_without_rewriting_identifiers() {
348        assert_eq!(
349            markdown_inline_field_text("prefix _emphasis_ suffix"),
350            "prefix &#95;emphasis&#95; suffix"
351        );
352        assert_eq!(
353            markdown_table_field_text("__strong__"),
354            "&#95;&#95;strong&#95;&#95;"
355        );
356        assert_eq!(
357            markdown_inline_field_text("basic_benchmark::bench_fibonacci"),
358            "basic_benchmark::bench_fibonacci"
359        );
360    }
361
362    #[test]
363    fn strict_encoders_remain_canonical_for_benign_report_text() {
364        let input = "provekit::passport";
365
366        assert_eq!(markdown_inline_text(input), "provekit&#58;&#58;passport");
367        assert_eq!(
368            markdown_table_cell_text(input),
369            "provekit&#58;&#58;passport"
370        );
371    }
372
373    #[test]
374    fn inline_code_preserves_benign_fields_and_contains_backtick_injection() {
375        assert_eq!(
376            markdown_inline_code("provekit::passport"),
377            "`provekit::passport`"
378        );
379        assert_eq!(markdown_inline_code("line\r\nbreak"), "`line break`");
380        assert_eq!(
381            markdown_inline_code("value`\n- forged"),
382            "`` value` - forged ``"
383        );
384        assert_eq!(markdown_inline_code("two``ticks"), "``` two``ticks ```");
385        assert_eq!(markdown_inline_code(""), "` `");
386    }
387
388    #[test]
389    fn csv_field_applies_rfc_4180_quoting_and_formula_neutralization() {
390        let cases = [
391            ("plain", "plain"),
392            ("comma,value", "\"comma,value\""),
393            ("quote\"value", "\"quote\"\"value\""),
394            ("line\r\nbreak", "\"line\r\nbreak\""),
395            ("=SUM(A1:A2)", "'=SUM(A1:A2)"),
396            ("+cmd", "'+cmd"),
397            ("-1+2", "'-1+2"),
398            ("@IMPORT", "'@IMPORT"),
399            (" \t=SUM(1,2)", "\"' \t=SUM(1,2)\""),
400            ("\u{0000}@cmd", "'\u{0000}@cmd"),
401        ];
402
403        for (input, expected) in cases {
404            assert_eq!(csv_field(input), expected, "input: {input:?}");
405        }
406    }
407
408    #[test]
409    fn csv_field_handles_the_combined_adversarial_corpus() {
410        let input = " \t=HYPERLINK(\"https://evil.invalid/a,b\")\r\nnext";
411
412        assert_eq!(
413            csv_field(input),
414            "\"' \t=HYPERLINK(\"\"https://evil.invalid/a,b\"\")\r\nnext\""
415        );
416    }
417
418    #[test]
419    fn markdown_link_destination_preserves_safe_relative_paths_and_encodes_syntax() {
420        let cases = [
421            ("plots/alpha.svg", "plots/alpha.svg"),
422            ("javascript:alert(1)", "javascript%3Aalert%281%29"),
423            ("//evil.invalid/x", "%2F%2Fevil.invalid/x"),
424            ("plots/a b[1].svg", "plots/a%20b%5B1%5D.svg"),
425            ("plots/é.svg", "plots/%C3%A9.svg"),
426            ("plots/x\r\n# heading.svg", "plots/x%0D%0A%23%20heading.svg"),
427        ];
428
429        for (input, expected) in cases {
430            assert_eq!(
431                markdown_link_destination(input),
432                expected,
433                "input: {input:?}"
434            );
435        }
436    }
437}