rotulus_layout/markdown.rs
1//! The markdown inline scanner.
2//!
3//! Implements the restricted subset in docs/design.md "Markdown" —
4//! `**bold**`, `*italic*` / `_italic_`, `` `code` ``, `~~strike~~`,
5//! `[label](url)`, and backslash escapes. Block constructs (fenced code,
6//! `>` quotes) are recognised by [`split_blocks`] before this scanner
7//! runs; everything else CommonMark defines is deliberately absent.
8//!
9//! **Why hand-written rather than `pulldown-cmark`.** Three reasons: it
10//! has no inline-only mode, so we would be filtering
11//! a block-level event stream and fighting CommonMark's block rules to
12//! suppress headings, thematic breaks and setext underlines — all of
13//! which occur constantly in real chat prose (`# 1`, `---`, `====`); its
14//! event stream would still need converting into byte-ranged
15//! [`Span`](crate::span::Span)s, which is most of the work; and a scanner
16//! for six constructs is small enough to be exhaustively tested and
17//! predictable on pathological input.
18//!
19//! **What is deliberately not supported**, and why: headings (`#` opens
20//! far too many ordinary chat lines), images (`![]()` — inline media has
21//! a server-validated upload/download pipeline and must not be
22//! bypassable by an arbitrary URL), tables, raw HTML, reference links,
23//! footnotes, thematic breaks, setext headings, and autolinking (that is
24//! [`crate::linkify`]'s, which also supplies the scheme list a
25//! `[label](url)` is checked against).
26
27use crate::linkify::Linkifier;
28use crate::span::{Attrs, ParsedText, SpanBuilder, Style};
29use std::ops::Range;
30
31/// Cap on nesting depth. `**a *b* a**` is depth 2. Real messages never
32/// approach this; the cap exists so a line of 5,000 asterisks can't
33/// recurse the parser into the stack guard.
34const MAX_DEPTH: u8 = 8;
35
36/// Whether a `[label](url)` link may point at `url` under the default
37/// scheme list.
38///
39/// Anything else — `javascript:`, `data:`, `file:`, or an unrecognized
40/// scheme — makes the whole construct render as literal text, delimiters
41/// included, so the user sees exactly what was typed rather than a link
42/// they can't inspect. A view with its own scheme list uses
43/// [`parse_inline_with`].
44pub fn scheme_allowed(url: &str) -> bool {
45 Linkifier::default().allows(url)
46}
47
48/// A block-level piece of a message body.
49#[derive(Clone, PartialEq, Eq, Debug)]
50pub enum RawBlock {
51 /// Ordinary text, to be run through [`parse_inline`].
52 Paragraph(String),
53 /// A fenced code block. Contents are inert.
54 Code {
55 text: String,
56 language: Option<String>,
57 },
58 /// One or more `>`-prefixed lines, already stripped of their markers.
59 Quote { text: String, depth: u8 },
60}
61
62/// Split a message body into block-level pieces.
63///
64/// Only two block constructs exist in the subset, and both are
65/// unambiguous at line starts, so this is a line scanner rather than a
66/// parser. An unterminated fence runs to the end of the body — the
67/// alternative (treating it as literal) means a message someone is
68/// mid-way through typing flickers between two renderings.
69pub fn split_blocks(body: &str) -> Vec<RawBlock> {
70 let mut out: Vec<RawBlock> = Vec::new();
71 let mut para = String::new();
72 let mut lines = body.split('\n').peekable();
73
74 let flush = |para: &mut String, out: &mut Vec<RawBlock>| {
75 if !para.is_empty() {
76 out.push(RawBlock::Paragraph(std::mem::take(para)));
77 }
78 };
79
80 while let Some(line) = lines.next() {
81 let trimmed = line.trim_start();
82
83 if let Some(rest) = trimmed.strip_prefix("```") {
84 flush(&mut para, &mut out);
85
86 // A fence that opens and closes on one line — ```like this```
87 // — is by far the most common way someone types a code block
88 // in a chat box, because chat boxes send on Enter. Treating
89 // it as an *opening* fence made the rest of the line the
90 // "language", scanned for a close that never came, and
91 // produced an empty code block: a blank row where the user's
92 // text should be.
93 if let Some(inner) = rest.strip_suffix("```") {
94 out.push(RawBlock::Code {
95 text: inner.to_string(),
96 language: None,
97 });
98 continue;
99 }
100
101 let language = {
102 let l = rest.trim();
103 if l.is_empty() {
104 None
105 } else {
106 Some(l.to_string())
107 }
108 };
109 let mut code = String::new();
110 for l in lines.by_ref() {
111 if l.trim_start().starts_with("```") {
112 break;
113 }
114 if !code.is_empty() {
115 code.push('\n');
116 }
117 code.push_str(l);
118 }
119 out.push(RawBlock::Code {
120 text: code,
121 language,
122 });
123 continue;
124 }
125
126 if trimmed.starts_with('>') {
127 flush(&mut para, &mut out);
128 let mut depth = 0u8;
129 let mut rest = trimmed;
130 while let Some(r) = rest.strip_prefix('>') {
131 depth = depth.saturating_add(1);
132 rest = r.trim_start();
133 }
134 let mut quoted = rest.to_string();
135 // Consume following lines at the same depth.
136 while let Some(next) = lines.peek() {
137 let nt = next.trim_start();
138 if !nt.starts_with('>') {
139 break;
140 }
141 let mut d = 0u8;
142 let mut r = nt;
143 while let Some(s) = r.strip_prefix('>') {
144 d = d.saturating_add(1);
145 r = s.trim_start();
146 }
147 if d != depth {
148 break;
149 }
150 quoted.push('\n');
151 quoted.push_str(r);
152 lines.next();
153 }
154 out.push(RawBlock::Quote {
155 text: quoted,
156 depth,
157 });
158 continue;
159 }
160
161 if !para.is_empty() {
162 para.push('\n');
163 }
164 para.push_str(line);
165 }
166 flush(&mut para, &mut out);
167 out
168}
169
170/// Parse inline markdown into styled text.
171///
172/// Never fails and never panics: any construct that doesn't close
173/// renders as the literal characters that were typed.
174pub fn parse_inline(src: &str) -> ParsedText {
175 parse_inline_with(src, &Linkifier::default())
176}
177
178/// [`parse_inline`], with `[label](url)` links checked against `links`
179/// rather than the default scheme list.
180pub fn parse_inline_with(src: &str, links: &Linkifier) -> ParsedText {
181 let mut b = SpanBuilder::new();
182 scan(src, Style::default(), 0, links, &mut b);
183 let out = b.finish();
184 out.debug_assert_well_formed();
185 out
186}
187
188fn scan(src: &str, base: Style, depth: u8, links: &Linkifier, out: &mut SpanBuilder) {
189 let bytes = src.as_bytes();
190 let mut i = 0usize;
191 // Start of the current literal run, flushed lazily so plain text
192 // costs one push rather than one per character.
193 let mut lit = 0usize;
194
195 macro_rules! flush_lit {
196 ($upto:expr) => {
197 if $upto > lit {
198 out.push(&src[lit..$upto], base);
199 }
200 };
201 }
202
203 while i < bytes.len() {
204 let c = bytes[i];
205
206 // Backslash escape: the next character is literal, whatever it is.
207 if c == b'\\' && i + 1 < bytes.len() {
208 let next = i + 1;
209 let ch_end = next + utf8_len(bytes[next]);
210 let ch_end = ch_end.min(bytes.len());
211 if is_escapable(&src[next..ch_end]) {
212 flush_lit!(i);
213 out.push(&src[next..ch_end], base);
214 i = ch_end;
215 lit = i;
216 continue;
217 }
218 i += 1;
219 continue;
220 }
221
222 // `code` — inert contents, so it is tried before everything else.
223 if c == b'`' {
224 if let Some((inner, end)) = code_span(bytes, i) {
225 flush_lit!(i);
226 out.push(strip_code_pad(&src[inner]), base.with_attrs(Attrs::CODE));
227 i = end;
228 lit = i;
229 continue;
230 }
231 // An unmatched run is literal text. Skip the *whole* run, not
232 // one byte: retrying at the second backtick of ``` would let
233 // a shorter run inside it close against something later.
234 i += backtick_run(bytes, i);
235 continue;
236 }
237
238 if depth < MAX_DEPTH {
239 // **bold** and ~~strike~~ — two-character delimiters first, so
240 // `**` is never mistaken for an empty `*` pair.
241 let two = match c {
242 b'*' if bytes.get(i + 1) == Some(&b'*') => Some((Attrs::BOLD, "**")),
243 b'~' if bytes.get(i + 1) == Some(&b'~') => Some((Attrs::STRIKETHROUGH, "~~")),
244 _ => None,
245 };
246 if let Some((attr, delim)) = two {
247 if can_open(bytes, i + 2) {
248 if let Some(close) = find_delim(src, i + 2, delim) {
249 flush_lit!(i);
250 scan(
251 &src[i + 2..close],
252 base.with_attrs(attr),
253 depth + 1,
254 links,
255 out,
256 );
257 i = close + 2;
258 lit = i;
259 continue;
260 }
261 }
262 i += 2;
263 continue;
264 }
265
266 // *italic* / _italic_.
267 //
268 // `_` additionally requires non-alphanumeric neighbours so
269 // that snake_case identifiers, which turn up constantly in
270 // this project's chat, survive intact.
271 if (c == b'*' && can_open(bytes, i + 1))
272 || (c == b'_' && can_open(bytes, i + 1) && intraword_ok(bytes, i))
273 {
274 if let Some(close) = find_italic_close(src, i + 1, c) {
275 flush_lit!(i);
276 scan(
277 &src[i + 1..close],
278 base.with_attrs(Attrs::ITALIC),
279 depth + 1,
280 links,
281 out,
282 );
283 i = close + 1;
284 lit = i;
285 continue;
286 }
287 i += 1;
288 continue;
289 }
290
291 // [label](url)
292 if c == b'[' {
293 if let Some((label, href, end)) = parse_link(src, i) {
294 if links.allows(href) {
295 flush_lit!(i);
296 // The id must exist before the label is emitted
297 // so it can ride in the label's Style; the
298 // visible range is patched in afterwards.
299 let id = out.reserve_link(href.to_string());
300 let start = out.len();
301 let mut label_style = base.with_attrs(Attrs::UNDERLINE);
302 label_style.link = Some(id);
303 // MAX_DEPTH, not depth + 1: a link label renders
304 // as plain text. Nested links are invalid
305 // markdown, and `parse_link` balances brackets,
306 // so re-scanning the label would happily parse
307 // the inner one.
308 scan(label, label_style, MAX_DEPTH, links, out);
309 out.set_link_range(id, start..out.len());
310 i = end;
311 lit = i;
312 continue;
313 }
314 // Disallowed scheme: fall through so the whole
315 // `[label](url)` renders as literal text. The user
316 // sees what was typed rather than a link they have
317 // no way to inspect.
318 }
319 i += 1;
320 continue;
321 }
322 }
323
324 i += 1;
325 }
326 flush_lit!(bytes.len());
327}
328
329/// Bytes markdown lets you escape. A backslash before anything else is
330/// itself literal — `C:\path` must not lose its separators.
331fn is_escapable(s: &str) -> bool {
332 matches!(
333 s,
334 "\\" | "*" | "_" | "`" | "~" | "[" | "]" | "(" | ")" | ">" | "#"
335 )
336}
337
338fn utf8_len(b: u8) -> usize {
339 if b < 0x80 {
340 1
341 } else if b >> 5 == 0b110 {
342 2
343 } else if b >> 4 == 0b1110 {
344 3
345 } else if b >> 3 == 0b11110 {
346 4
347 } else {
348 1
349 }
350}
351
352/// Length of the run of backticks starting at `at`.
353fn backtick_run(bytes: &[u8], at: usize) -> usize {
354 let mut n = 0usize;
355 while bytes.get(at + n) == Some(&b'`') {
356 n += 1;
357 }
358 n
359}
360
361/// A code span starting at `at`: the byte range of its contents, and the
362/// offset just past its closing run.
363///
364/// CommonMark's rule, and the reason `` `` `hello` `` `` works: a span
365/// opens with a run of N backticks and closes on the next run of
366/// *exactly* N. Matching a single backtick against the next single
367/// backtick — what this used to do — turns ```` ``hello`` ```` into two
368/// empty spans on either side of unstyled text, which is exactly how it
369/// rendered.
370///
371/// There is no escaping inside a code span and none in front of one: a
372/// `\`` is consumed by the backslash branch in `scan` before the
373/// backtick is ever seen here.
374fn code_span(bytes: &[u8], at: usize) -> Option<(Range<usize>, usize)> {
375 let n = backtick_run(bytes, at);
376 let mut j = at + n;
377 while j < bytes.len() {
378 if bytes[j] == b'`' {
379 let m = backtick_run(bytes, j);
380 if m == n {
381 return Some((at + n..j, j + m));
382 }
383 j += m;
384 continue;
385 }
386 j += 1;
387 }
388 None
389}
390
391/// Strip one leading and one trailing space, when both are there and the
392/// content is not all spaces.
393///
394/// This is what lets a span hold a backtick of its own: `` ` `` is a
395/// two-backtick span containing " ` ", and without the strip it would
396/// render with the padding the author only added to separate the
397/// delimiters.
398fn strip_code_pad(s: &str) -> &str {
399 let b = s.as_bytes();
400 if b.len() >= 2 && b[0] == b' ' && b[b.len() - 1] == b' ' && b.iter().any(|&c| c != b' ') {
401 &s[1..s.len() - 1]
402 } else {
403 s
404 }
405}
406
407/// Next unescaped occurrence of a multi-byte delimiter, skipping code
408/// spans so that `` **a `b**` c** `` closes at the last `**`.
409fn find_delim(src: &str, from: usize, delim: &str) -> Option<usize> {
410 let bytes = src.as_bytes();
411 let d = delim.as_bytes();
412 let mut i = from;
413 while i + d.len() <= bytes.len() {
414 if bytes[i] == b'\\' {
415 i += 2;
416 continue;
417 }
418 if bytes[i] == b'`' {
419 match code_span(bytes, i) {
420 Some((_, end)) => {
421 i = end;
422 continue;
423 }
424 None => return None,
425 }
426 }
427 if bytes[i..].starts_with(d) {
428 // An empty span (`****`) is not emphasis, and a closer must
429 // be right-flanking.
430 if i == from || !can_close(bytes, i) {
431 i += d.len();
432 continue;
433 }
434 return Some(i);
435 }
436 i += 1;
437 }
438 None
439}
440
441/// Close delimiter for single-character emphasis. Rejects a `**` run so
442/// that `*a**b*` doesn't close on the doubled pair.
443fn find_italic_close(src: &str, from: usize, open: u8) -> Option<usize> {
444 let bytes = src.as_bytes();
445 let mut i = from;
446 while i < bytes.len() {
447 if bytes[i] == b'\\' {
448 i += 2;
449 continue;
450 }
451 if bytes[i] == b'`' {
452 match code_span(bytes, i) {
453 Some((_, end)) => {
454 i = end;
455 continue;
456 }
457 None => return None,
458 }
459 }
460 if bytes[i] == open {
461 if bytes.get(i + 1) == Some(&open) {
462 i += 2;
463 continue;
464 }
465 if i == from || !can_close(bytes, i) {
466 i += 1;
467 continue;
468 }
469 if open == b'_' && !intraword_close_ok(bytes, i) {
470 i += 1;
471 continue;
472 }
473 return Some(i);
474 }
475 i += 1;
476 }
477 None
478}
479
480/// CommonMark's left-flanking rule, which is what stops `2 * 3 * 4`
481/// from becoming `2 3 4`.
482///
483/// An opening delimiter must be followed by non-whitespace. Arithmetic,
484/// bullet-ish prose ("see * the docs") and trailing asterisks all rely
485/// on this; without it the parser eats punctuation out of ordinary
486/// sentences, which is the single most annoying way a chat markdown
487/// implementation can be wrong.
488fn can_open(bytes: &[u8], after: usize) -> bool {
489 bytes.get(after).is_some_and(|b| !b.is_ascii_whitespace())
490}
491
492/// The right-flanking counterpart: a closing delimiter must be preceded
493/// by non-whitespace, so `a * b *` doesn't close on the trailing one.
494fn can_close(bytes: &[u8], at: usize) -> bool {
495 at > 0 && !bytes[at - 1].is_ascii_whitespace()
496}
497
498/// `_` opens emphasis only at a word boundary, so snake_case survives.
499fn intraword_ok(bytes: &[u8], i: usize) -> bool {
500 let before_ok = i == 0 || !bytes[i - 1].is_ascii_alphanumeric();
501 let after_ok = bytes.get(i + 1).is_some_and(|b| *b != b'_');
502 before_ok && after_ok
503}
504
505/// `_` closes emphasis only at a word boundary.
506fn intraword_close_ok(bytes: &[u8], i: usize) -> bool {
507 bytes.get(i + 1).is_none_or(|b| !b.is_ascii_alphanumeric())
508}
509
510/// Parse `[label](url)` starting at `open`. Returns label, href and the
511/// offset one past the closing paren.
512fn parse_link(src: &str, open: usize) -> Option<(&str, &str, usize)> {
513 let bytes = src.as_bytes();
514 // Label: balanced brackets, no nesting beyond one level needed.
515 let mut i = open + 1;
516 let mut depth = 1usize;
517 while i < bytes.len() {
518 match bytes[i] {
519 b'\\' => {
520 i += 2;
521 continue;
522 }
523 b'[' => depth += 1,
524 b']' => {
525 depth -= 1;
526 if depth == 0 {
527 break;
528 }
529 }
530 _ => {}
531 }
532 i += 1;
533 }
534 if depth != 0 || i >= bytes.len() {
535 return None;
536 }
537 let label_end = i;
538 if bytes.get(label_end + 1) != Some(&b'(') {
539 return None;
540 }
541 let url_start = label_end + 2;
542 let mut j = url_start;
543 while j < bytes.len() && bytes[j] != b')' {
544 if bytes[j] == b'\\' {
545 j += 2;
546 continue;
547 }
548 // A URL never contains whitespace; bail rather than swallowing
549 // the rest of the line looking for a paren.
550 if bytes[j].is_ascii_whitespace() {
551 return None;
552 }
553 j += 1;
554 }
555 if j >= bytes.len() {
556 return None;
557 }
558 let label = &src[open + 1..label_end];
559 let href = &src[url_start..j];
560 if label.is_empty() || href.is_empty() {
561 return None;
562 }
563 Some((label, href, j + 1))
564}
565
566// ---- input tinting --------------------------------------------------
567
568/// One tintable region of *source* text, for the compose box.
569#[derive(Debug, Clone, PartialEq, Eq)]
570pub struct SourceSpan {
571 pub start: usize,
572 pub end: usize,
573 pub attrs: Attrs,
574 /// True for the delimiter characters themselves, which the input
575 /// box dims rather than styles.
576 pub delim: bool,
577}
578
579/// Locate the markdown delimiters in `src` that will actually be
580/// consumed, reporting ranges **in the source**.
581///
582/// This exists because [`parse_inline`] reports ranges in the *rendered*
583/// text, with the delimiters removed — exactly the offsets the compose
584/// box does not have. Rather than teach the renderer to carry source
585/// offsets through its recursion (a change to a heavily-tested function
586/// for a cosmetic feature), this is a separate, deliberately shallower
587/// pass that reuses the *rules* that are actually subtle: `can_open` /
588/// `can_close` flanking and the `_` intraword guards.
589///
590/// **It is an approximation, and that is fine here.** It doesn't nest,
591/// doesn't handle links, and takes the first valid closer. Being wrong
592/// in the compose box means a character is tinted that won't be, on text
593/// the user is still editing and can see; being wrong in the renderer
594/// would change what a message *says*. The two failure modes are not
595/// comparable, which is why they don't share a code path.
596pub fn scan_delims(src: &str) -> Vec<SourceSpan> {
597 let bytes = src.as_bytes();
598 let mut out = Vec::new();
599 let mut i = 0usize;
600
601 while i < bytes.len() {
602 // A backslash escape hides the next character from tinting, the
603 // same way it hides it from the parser.
604 if bytes[i] == b'\\' && i + 1 < bytes.len() {
605 i += 2;
606 continue;
607 }
608
609 let (run, attrs) = match bytes[i] {
610 b'*' if bytes.get(i + 1) == Some(&b'*') => (2usize, Attrs::BOLD),
611 b'*' => (1, Attrs::ITALIC),
612 b'_' if intraword_ok(bytes, i) => (1, Attrs::ITALIC),
613 b'`' => (backtick_run(bytes, i), Attrs::CODE),
614 _ => {
615 i += 1;
616 continue;
617 }
618 };
619
620 // Code spans are literal: no escapes, no nesting, and the closer
621 // is a run of exactly the same length. They also skip the
622 // flanking rules — `` ` `` is a legitimate span whose content is
623 // a space, and can_open would reject it.
624 let literal = attrs == Attrs::CODE;
625
626 if !literal && !can_open(bytes, i + run) {
627 i += run;
628 continue;
629 }
630 let mut j = i + run;
631 let close = loop {
632 if j >= bytes.len() {
633 break None;
634 }
635 if !literal && bytes[j] == b'\\' {
636 j += 2;
637 continue;
638 }
639 if literal {
640 if bytes[j] == b'`' {
641 let m = backtick_run(bytes, j);
642 if m == run {
643 break Some(j);
644 }
645 j += m;
646 continue;
647 }
648 j += 1;
649 continue;
650 }
651 let hit = match run {
652 2 => bytes[j] == b'*' && bytes.get(j + 1) == Some(&b'*'),
653 _ => bytes[j] == bytes[i] && (bytes[i] != b'_' || intraword_close_ok(bytes, j)),
654 };
655 if hit && can_close(bytes, j) {
656 break Some(j);
657 }
658 j += 1;
659 };
660
661 let Some(close) = close else {
662 i += run;
663 continue;
664 };
665
666 out.push(SourceSpan {
667 start: i,
668 end: i + run,
669 attrs,
670 delim: true,
671 });
672 out.push(SourceSpan {
673 start: i + run,
674 end: close,
675 attrs,
676 delim: false,
677 });
678 out.push(SourceSpan {
679 start: close,
680 end: close + run,
681 attrs,
682 delim: true,
683 });
684 i = close + run;
685 }
686 out
687}