Skip to main content

rotulus_layout/
markdown.rs

1//! The markdown inline scanner.
2//!
3//! Implements the restricted subset in docs/design.md "Markdown" —
4//! `**bold**`, `*italic*` / `_italic_`, `` `code` ``, `~~strike~~`,
5//! `[label](url)`, and backslash escapes. Block constructs (fenced code,
6//! `>` quotes) are recognised by [`split_blocks`] before this scanner
7//! runs; everything else CommonMark defines is deliberately absent.
8//!
9//! **Why hand-written rather than `pulldown-cmark`.** Three reasons: it
10//! has no inline-only mode, so we would be filtering
11//! a block-level event stream and fighting CommonMark's block rules to
12//! suppress headings, thematic breaks and setext underlines — all of
13//! which occur constantly in real chat prose (`# 1`, `---`, `====`); its
14//! event stream would still need converting into byte-ranged
15//! [`Span`](crate::span::Span)s, which is most of the work; and a scanner
16//! for six constructs is small enough to be exhaustively tested and
17//! predictable on pathological input.
18//!
19//! **What is deliberately not supported**, and why: headings (`#` opens
20//! far too many ordinary chat lines), images (`![]()` — inline media has
21//! a server-validated upload/download pipeline and must not be
22//! bypassable by an arbitrary URL), tables, raw HTML, reference links,
23//! footnotes, thematic breaks, setext headings, and autolinking (that is
24//! [`crate::linkify`]'s, which also supplies the scheme list a
25//! `[label](url)` is checked against).
26
27use crate::linkify::Linkifier;
28use crate::span::{Attrs, ParsedText, SpanBuilder, Style};
29use std::ops::Range;
30
31/// Cap on nesting depth. `**a *b* a**` is depth 2. Real messages never
32/// approach this; the cap exists so a line of 5,000 asterisks can't
33/// recurse the parser into the stack guard.
34const MAX_DEPTH: u8 = 8;
35
36/// Whether a `[label](url)` link may point at `url` under the default
37/// scheme list.
38///
39/// Anything else — `javascript:`, `data:`, `file:`, or an unrecognized
40/// scheme — makes the whole construct render as literal text, delimiters
41/// included, so the user sees exactly what was typed rather than a link
42/// they can't inspect. A view with its own scheme list uses
43/// [`parse_inline_with`].
44pub fn scheme_allowed(url: &str) -> bool {
45    Linkifier::default().allows(url)
46}
47
48/// A block-level piece of a message body.
49#[derive(Clone, PartialEq, Eq, Debug)]
50pub enum RawBlock {
51    /// Ordinary text, to be run through [`parse_inline`].
52    Paragraph(String),
53    /// A fenced code block. Contents are inert.
54    Code {
55        text: String,
56        language: Option<String>,
57    },
58    /// One or more `>`-prefixed lines, already stripped of their markers.
59    Quote { text: String, depth: u8 },
60}
61
62/// Split a message body into block-level pieces.
63///
64/// Only two block constructs exist in the subset, and both are
65/// unambiguous at line starts, so this is a line scanner rather than a
66/// parser. An unterminated fence runs to the end of the body — the
67/// alternative (treating it as literal) means a message someone is
68/// mid-way through typing flickers between two renderings.
69pub fn split_blocks(body: &str) -> Vec<RawBlock> {
70    let mut out: Vec<RawBlock> = Vec::new();
71    let mut para = String::new();
72    let mut lines = body.split('\n').peekable();
73
74    let flush = |para: &mut String, out: &mut Vec<RawBlock>| {
75        if !para.is_empty() {
76            out.push(RawBlock::Paragraph(std::mem::take(para)));
77        }
78    };
79
80    while let Some(line) = lines.next() {
81        let trimmed = line.trim_start();
82
83        if let Some(rest) = trimmed.strip_prefix("```") {
84            flush(&mut para, &mut out);
85
86            // A fence that opens and closes on one line — ```like this```
87            // — is by far the most common way someone types a code block
88            // in a chat box, because chat boxes send on Enter. Treating
89            // it as an *opening* fence made the rest of the line the
90            // "language", scanned for a close that never came, and
91            // produced an empty code block: a blank row where the user's
92            // text should be.
93            if let Some(inner) = rest.strip_suffix("```") {
94                out.push(RawBlock::Code {
95                    text: inner.to_string(),
96                    language: None,
97                });
98                continue;
99            }
100
101            let language = {
102                let l = rest.trim();
103                if l.is_empty() {
104                    None
105                } else {
106                    Some(l.to_string())
107                }
108            };
109            let mut code = String::new();
110            for l in lines.by_ref() {
111                if l.trim_start().starts_with("```") {
112                    break;
113                }
114                if !code.is_empty() {
115                    code.push('\n');
116                }
117                code.push_str(l);
118            }
119            out.push(RawBlock::Code {
120                text: code,
121                language,
122            });
123            continue;
124        }
125
126        if trimmed.starts_with('>') {
127            flush(&mut para, &mut out);
128            let mut depth = 0u8;
129            let mut rest = trimmed;
130            while let Some(r) = rest.strip_prefix('>') {
131                depth = depth.saturating_add(1);
132                rest = r.trim_start();
133            }
134            let mut quoted = rest.to_string();
135            // Consume following lines at the same depth.
136            while let Some(next) = lines.peek() {
137                let nt = next.trim_start();
138                if !nt.starts_with('>') {
139                    break;
140                }
141                let mut d = 0u8;
142                let mut r = nt;
143                while let Some(s) = r.strip_prefix('>') {
144                    d = d.saturating_add(1);
145                    r = s.trim_start();
146                }
147                if d != depth {
148                    break;
149                }
150                quoted.push('\n');
151                quoted.push_str(r);
152                lines.next();
153            }
154            out.push(RawBlock::Quote {
155                text: quoted,
156                depth,
157            });
158            continue;
159        }
160
161        if !para.is_empty() {
162            para.push('\n');
163        }
164        para.push_str(line);
165    }
166    flush(&mut para, &mut out);
167    out
168}
169
170/// Parse inline markdown into styled text.
171///
172/// Never fails and never panics: any construct that doesn't close
173/// renders as the literal characters that were typed.
174pub fn parse_inline(src: &str) -> ParsedText {
175    parse_inline_with(src, &Linkifier::default())
176}
177
178/// [`parse_inline`], with `[label](url)` links checked against `links`
179/// rather than the default scheme list.
180pub fn parse_inline_with(src: &str, links: &Linkifier) -> ParsedText {
181    let mut b = SpanBuilder::new();
182    scan(src, Style::default(), 0, links, &mut b);
183    let out = b.finish();
184    out.debug_assert_well_formed();
185    out
186}
187
188fn scan(src: &str, base: Style, depth: u8, links: &Linkifier, out: &mut SpanBuilder) {
189    let bytes = src.as_bytes();
190    let mut i = 0usize;
191    // Start of the current literal run, flushed lazily so plain text
192    // costs one push rather than one per character.
193    let mut lit = 0usize;
194
195    macro_rules! flush_lit {
196        ($upto:expr) => {
197            if $upto > lit {
198                out.push(&src[lit..$upto], base);
199            }
200        };
201    }
202
203    while i < bytes.len() {
204        let c = bytes[i];
205
206        // Backslash escape: the next character is literal, whatever it is.
207        if c == b'\\' && i + 1 < bytes.len() {
208            let next = i + 1;
209            let ch_end = next + utf8_len(bytes[next]);
210            let ch_end = ch_end.min(bytes.len());
211            if is_escapable(&src[next..ch_end]) {
212                flush_lit!(i);
213                out.push(&src[next..ch_end], base);
214                i = ch_end;
215                lit = i;
216                continue;
217            }
218            i += 1;
219            continue;
220        }
221
222        // `code` — inert contents, so it is tried before everything else.
223        if c == b'`' {
224            if let Some((inner, end)) = code_span(bytes, i) {
225                flush_lit!(i);
226                out.push(strip_code_pad(&src[inner]), base.with_attrs(Attrs::CODE));
227                i = end;
228                lit = i;
229                continue;
230            }
231            // An unmatched run is literal text. Skip the *whole* run, not
232            // one byte: retrying at the second backtick of ``` would let
233            // a shorter run inside it close against something later.
234            i += backtick_run(bytes, i);
235            continue;
236        }
237
238        if depth < MAX_DEPTH {
239            // **bold** and ~~strike~~ — two-character delimiters first, so
240            // `**` is never mistaken for an empty `*` pair.
241            let two = match c {
242                b'*' if bytes.get(i + 1) == Some(&b'*') => Some((Attrs::BOLD, "**")),
243                b'~' if bytes.get(i + 1) == Some(&b'~') => Some((Attrs::STRIKETHROUGH, "~~")),
244                _ => None,
245            };
246            if let Some((attr, delim)) = two {
247                if can_open(bytes, i + 2) {
248                    if let Some(close) = find_delim(src, i + 2, delim) {
249                        flush_lit!(i);
250                        scan(
251                            &src[i + 2..close],
252                            base.with_attrs(attr),
253                            depth + 1,
254                            links,
255                            out,
256                        );
257                        i = close + 2;
258                        lit = i;
259                        continue;
260                    }
261                }
262                i += 2;
263                continue;
264            }
265
266            // *italic* / _italic_.
267            //
268            // `_` additionally requires non-alphanumeric neighbours so
269            // that snake_case identifiers, which turn up constantly in
270            // this project's chat, survive intact.
271            if (c == b'*' && can_open(bytes, i + 1))
272                || (c == b'_' && can_open(bytes, i + 1) && intraword_ok(bytes, i))
273            {
274                if let Some(close) = find_italic_close(src, i + 1, c) {
275                    flush_lit!(i);
276                    scan(
277                        &src[i + 1..close],
278                        base.with_attrs(Attrs::ITALIC),
279                        depth + 1,
280                        links,
281                        out,
282                    );
283                    i = close + 1;
284                    lit = i;
285                    continue;
286                }
287                i += 1;
288                continue;
289            }
290
291            // [label](url)
292            if c == b'[' {
293                if let Some((label, href, end)) = parse_link(src, i) {
294                    if links.allows(href) {
295                        flush_lit!(i);
296                        // The id must exist before the label is emitted
297                        // so it can ride in the label's Style; the
298                        // visible range is patched in afterwards.
299                        let id = out.reserve_link(href.to_string());
300                        let start = out.len();
301                        let mut label_style = base.with_attrs(Attrs::UNDERLINE);
302                        label_style.link = Some(id);
303                        // MAX_DEPTH, not depth + 1: a link label renders
304                        // as plain text. Nested links are invalid
305                        // markdown, and `parse_link` balances brackets,
306                        // so re-scanning the label would happily parse
307                        // the inner one.
308                        scan(label, label_style, MAX_DEPTH, links, out);
309                        out.set_link_range(id, start..out.len());
310                        i = end;
311                        lit = i;
312                        continue;
313                    }
314                    // Disallowed scheme: fall through so the whole
315                    // `[label](url)` renders as literal text. The user
316                    // sees what was typed rather than a link they have
317                    // no way to inspect.
318                }
319                i += 1;
320                continue;
321            }
322        }
323
324        i += 1;
325    }
326    flush_lit!(bytes.len());
327}
328
329/// Bytes markdown lets you escape. A backslash before anything else is
330/// itself literal — `C:\path` must not lose its separators.
331fn is_escapable(s: &str) -> bool {
332    matches!(
333        s,
334        "\\" | "*" | "_" | "`" | "~" | "[" | "]" | "(" | ")" | ">" | "#"
335    )
336}
337
338fn utf8_len(b: u8) -> usize {
339    if b < 0x80 {
340        1
341    } else if b >> 5 == 0b110 {
342        2
343    } else if b >> 4 == 0b1110 {
344        3
345    } else if b >> 3 == 0b11110 {
346        4
347    } else {
348        1
349    }
350}
351
352/// Length of the run of backticks starting at `at`.
353fn backtick_run(bytes: &[u8], at: usize) -> usize {
354    let mut n = 0usize;
355    while bytes.get(at + n) == Some(&b'`') {
356        n += 1;
357    }
358    n
359}
360
361/// A code span starting at `at`: the byte range of its contents, and the
362/// offset just past its closing run.
363///
364/// CommonMark's rule, and the reason `` `` `hello` `` `` works: a span
365/// opens with a run of N backticks and closes on the next run of
366/// *exactly* N. Matching a single backtick against the next single
367/// backtick — what this used to do — turns ```` ``hello`` ```` into two
368/// empty spans on either side of unstyled text, which is exactly how it
369/// rendered.
370///
371/// There is no escaping inside a code span and none in front of one: a
372/// `\``  is consumed by the backslash branch in `scan` before the
373/// backtick is ever seen here.
374fn code_span(bytes: &[u8], at: usize) -> Option<(Range<usize>, usize)> {
375    let n = backtick_run(bytes, at);
376    let mut j = at + n;
377    while j < bytes.len() {
378        if bytes[j] == b'`' {
379            let m = backtick_run(bytes, j);
380            if m == n {
381                return Some((at + n..j, j + m));
382            }
383            j += m;
384            continue;
385        }
386        j += 1;
387    }
388    None
389}
390
391/// Strip one leading and one trailing space, when both are there and the
392/// content is not all spaces.
393///
394/// This is what lets a span hold a backtick of its own: `` ` `` is a
395/// two-backtick span containing " ` ", and without the strip it would
396/// render with the padding the author only added to separate the
397/// delimiters.
398fn strip_code_pad(s: &str) -> &str {
399    let b = s.as_bytes();
400    if b.len() >= 2 && b[0] == b' ' && b[b.len() - 1] == b' ' && b.iter().any(|&c| c != b' ') {
401        &s[1..s.len() - 1]
402    } else {
403        s
404    }
405}
406
407/// Next unescaped occurrence of a multi-byte delimiter, skipping code
408/// spans so that `` **a `b**` c** `` closes at the last `**`.
409fn find_delim(src: &str, from: usize, delim: &str) -> Option<usize> {
410    let bytes = src.as_bytes();
411    let d = delim.as_bytes();
412    let mut i = from;
413    while i + d.len() <= bytes.len() {
414        if bytes[i] == b'\\' {
415            i += 2;
416            continue;
417        }
418        if bytes[i] == b'`' {
419            match code_span(bytes, i) {
420                Some((_, end)) => {
421                    i = end;
422                    continue;
423                }
424                None => return None,
425            }
426        }
427        if bytes[i..].starts_with(d) {
428            // An empty span (`****`) is not emphasis, and a closer must
429            // be right-flanking.
430            if i == from || !can_close(bytes, i) {
431                i += d.len();
432                continue;
433            }
434            return Some(i);
435        }
436        i += 1;
437    }
438    None
439}
440
441/// Close delimiter for single-character emphasis. Rejects a `**` run so
442/// that `*a**b*` doesn't close on the doubled pair.
443fn find_italic_close(src: &str, from: usize, open: u8) -> Option<usize> {
444    let bytes = src.as_bytes();
445    let mut i = from;
446    while i < bytes.len() {
447        if bytes[i] == b'\\' {
448            i += 2;
449            continue;
450        }
451        if bytes[i] == b'`' {
452            match code_span(bytes, i) {
453                Some((_, end)) => {
454                    i = end;
455                    continue;
456                }
457                None => return None,
458            }
459        }
460        if bytes[i] == open {
461            if bytes.get(i + 1) == Some(&open) {
462                i += 2;
463                continue;
464            }
465            if i == from || !can_close(bytes, i) {
466                i += 1;
467                continue;
468            }
469            if open == b'_' && !intraword_close_ok(bytes, i) {
470                i += 1;
471                continue;
472            }
473            return Some(i);
474        }
475        i += 1;
476    }
477    None
478}
479
480/// CommonMark's left-flanking rule, which is what stops `2 * 3 * 4`
481/// from becoming `2  3  4`.
482///
483/// An opening delimiter must be followed by non-whitespace. Arithmetic,
484/// bullet-ish prose ("see * the docs") and trailing asterisks all rely
485/// on this; without it the parser eats punctuation out of ordinary
486/// sentences, which is the single most annoying way a chat markdown
487/// implementation can be wrong.
488fn can_open(bytes: &[u8], after: usize) -> bool {
489    bytes.get(after).is_some_and(|b| !b.is_ascii_whitespace())
490}
491
492/// The right-flanking counterpart: a closing delimiter must be preceded
493/// by non-whitespace, so `a * b *` doesn't close on the trailing one.
494fn can_close(bytes: &[u8], at: usize) -> bool {
495    at > 0 && !bytes[at - 1].is_ascii_whitespace()
496}
497
498/// `_` opens emphasis only at a word boundary, so snake_case survives.
499fn intraword_ok(bytes: &[u8], i: usize) -> bool {
500    let before_ok = i == 0 || !bytes[i - 1].is_ascii_alphanumeric();
501    let after_ok = bytes.get(i + 1).is_some_and(|b| *b != b'_');
502    before_ok && after_ok
503}
504
505/// `_` closes emphasis only at a word boundary.
506fn intraword_close_ok(bytes: &[u8], i: usize) -> bool {
507    bytes.get(i + 1).is_none_or(|b| !b.is_ascii_alphanumeric())
508}
509
510/// Parse `[label](url)` starting at `open`. Returns label, href and the
511/// offset one past the closing paren.
512fn parse_link(src: &str, open: usize) -> Option<(&str, &str, usize)> {
513    let bytes = src.as_bytes();
514    // Label: balanced brackets, no nesting beyond one level needed.
515    let mut i = open + 1;
516    let mut depth = 1usize;
517    while i < bytes.len() {
518        match bytes[i] {
519            b'\\' => {
520                i += 2;
521                continue;
522            }
523            b'[' => depth += 1,
524            b']' => {
525                depth -= 1;
526                if depth == 0 {
527                    break;
528                }
529            }
530            _ => {}
531        }
532        i += 1;
533    }
534    if depth != 0 || i >= bytes.len() {
535        return None;
536    }
537    let label_end = i;
538    if bytes.get(label_end + 1) != Some(&b'(') {
539        return None;
540    }
541    let url_start = label_end + 2;
542    let mut j = url_start;
543    while j < bytes.len() && bytes[j] != b')' {
544        if bytes[j] == b'\\' {
545            j += 2;
546            continue;
547        }
548        // A URL never contains whitespace; bail rather than swallowing
549        // the rest of the line looking for a paren.
550        if bytes[j].is_ascii_whitespace() {
551            return None;
552        }
553        j += 1;
554    }
555    if j >= bytes.len() {
556        return None;
557    }
558    let label = &src[open + 1..label_end];
559    let href = &src[url_start..j];
560    if label.is_empty() || href.is_empty() {
561        return None;
562    }
563    Some((label, href, j + 1))
564}
565
566// ---- input tinting --------------------------------------------------
567
568/// One tintable region of *source* text, for the compose box.
569#[derive(Debug, Clone, PartialEq, Eq)]
570pub struct SourceSpan {
571    pub start: usize,
572    pub end: usize,
573    pub attrs: Attrs,
574    /// True for the delimiter characters themselves, which the input
575    /// box dims rather than styles.
576    pub delim: bool,
577}
578
579/// Locate the markdown delimiters in `src` that will actually be
580/// consumed, reporting ranges **in the source**.
581///
582/// This exists because [`parse_inline`] reports ranges in the *rendered*
583/// text, with the delimiters removed — exactly the offsets the compose
584/// box does not have. Rather than teach the renderer to carry source
585/// offsets through its recursion (a change to a heavily-tested function
586/// for a cosmetic feature), this is a separate, deliberately shallower
587/// pass that reuses the *rules* that are actually subtle: `can_open` /
588/// `can_close` flanking and the `_` intraword guards.
589///
590/// **It is an approximation, and that is fine here.** It doesn't nest,
591/// doesn't handle links, and takes the first valid closer. Being wrong
592/// in the compose box means a character is tinted that won't be, on text
593/// the user is still editing and can see; being wrong in the renderer
594/// would change what a message *says*. The two failure modes are not
595/// comparable, which is why they don't share a code path.
596pub fn scan_delims(src: &str) -> Vec<SourceSpan> {
597    let bytes = src.as_bytes();
598    let mut out = Vec::new();
599    let mut i = 0usize;
600
601    while i < bytes.len() {
602        // A backslash escape hides the next character from tinting, the
603        // same way it hides it from the parser.
604        if bytes[i] == b'\\' && i + 1 < bytes.len() {
605            i += 2;
606            continue;
607        }
608
609        let (run, attrs) = match bytes[i] {
610            b'*' if bytes.get(i + 1) == Some(&b'*') => (2usize, Attrs::BOLD),
611            b'*' => (1, Attrs::ITALIC),
612            b'_' if intraword_ok(bytes, i) => (1, Attrs::ITALIC),
613            b'`' => (backtick_run(bytes, i), Attrs::CODE),
614            _ => {
615                i += 1;
616                continue;
617            }
618        };
619
620        // Code spans are literal: no escapes, no nesting, and the closer
621        // is a run of exactly the same length. They also skip the
622        // flanking rules — `` ` `` is a legitimate span whose content is
623        // a space, and can_open would reject it.
624        let literal = attrs == Attrs::CODE;
625
626        if !literal && !can_open(bytes, i + run) {
627            i += run;
628            continue;
629        }
630        let mut j = i + run;
631        let close = loop {
632            if j >= bytes.len() {
633                break None;
634            }
635            if !literal && bytes[j] == b'\\' {
636                j += 2;
637                continue;
638            }
639            if literal {
640                if bytes[j] == b'`' {
641                    let m = backtick_run(bytes, j);
642                    if m == run {
643                        break Some(j);
644                    }
645                    j += m;
646                    continue;
647                }
648                j += 1;
649                continue;
650            }
651            let hit = match run {
652                2 => bytes[j] == b'*' && bytes.get(j + 1) == Some(&b'*'),
653                _ => bytes[j] == bytes[i] && (bytes[i] != b'_' || intraword_close_ok(bytes, j)),
654            };
655            if hit && can_close(bytes, j) {
656                break Some(j);
657            }
658            j += 1;
659        };
660
661        let Some(close) = close else {
662            i += run;
663            continue;
664        };
665
666        out.push(SourceSpan {
667            start: i,
668            end: i + run,
669            attrs,
670            delim: true,
671        });
672        out.push(SourceSpan {
673            start: i + run,
674            end: close,
675            attrs,
676            delim: false,
677        });
678        out.push(SourceSpan {
679            start: close,
680            end: close + run,
681            attrs,
682            delim: true,
683        });
684        i = close + run;
685    }
686    out
687}