rotulus-layout 0.1.0

The layout engine behind the Rotulus chat view: message model, wrapping, height index, scroll anchoring
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
//! The markdown inline scanner.
//!
//! Implements the restricted subset in docs/design.md "Markdown" —
//! `**bold**`, `*italic*` / `_italic_`, `` `code` ``, `~~strike~~`,
//! `[label](url)`, and backslash escapes. Block constructs (fenced code,
//! `>` quotes) are recognised by [`split_blocks`] before this scanner
//! runs; everything else CommonMark defines is deliberately absent.
//!
//! **Why hand-written rather than `pulldown-cmark`.** Three reasons: it
//! has no inline-only mode, so we would be filtering
//! a block-level event stream and fighting CommonMark's block rules to
//! suppress headings, thematic breaks and setext underlines — all of
//! which occur constantly in real chat prose (`# 1`, `---`, `====`); its
//! event stream would still need converting into byte-ranged
//! [`Span`](crate::span::Span)s, which is most of the work; and a scanner
//! for six constructs is small enough to be exhaustively tested and
//! predictable on pathological input.
//!
//! **What is deliberately not supported**, and why: headings (`#` opens
//! far too many ordinary chat lines), images (`![]()` — inline media has
//! a server-validated upload/download pipeline and must not be
//! bypassable by an arbitrary URL), tables, raw HTML, reference links,
//! footnotes, thematic breaks, setext headings, and autolinking (that is
//! [`crate::linkify`]'s, which also supplies the scheme list a
//! `[label](url)` is checked against).

use crate::linkify::Linkifier;
use crate::span::{Attrs, ParsedText, SpanBuilder, Style};
use std::ops::Range;

/// Cap on nesting depth. `**a *b* a**` is depth 2. Real messages never
/// approach this; the cap exists so a line of 5,000 asterisks can't
/// recurse the parser into the stack guard.
const MAX_DEPTH: u8 = 8;

/// Whether a `[label](url)` link may point at `url` under the default
/// scheme list.
///
/// Anything else — `javascript:`, `data:`, `file:`, or an unrecognized
/// scheme — makes the whole construct render as literal text, delimiters
/// included, so the user sees exactly what was typed rather than a link
/// they can't inspect. A view with its own scheme list uses
/// [`parse_inline_with`].
pub fn scheme_allowed(url: &str) -> bool {
    Linkifier::default().allows(url)
}

/// A block-level piece of a message body.
#[derive(Clone, PartialEq, Eq, Debug)]
pub enum RawBlock {
    /// Ordinary text, to be run through [`parse_inline`].
    Paragraph(String),
    /// A fenced code block. Contents are inert.
    Code {
        text: String,
        language: Option<String>,
    },
    /// One or more `>`-prefixed lines, already stripped of their markers.
    Quote { text: String, depth: u8 },
}

/// Split a message body into block-level pieces.
///
/// Only two block constructs exist in the subset, and both are
/// unambiguous at line starts, so this is a line scanner rather than a
/// parser. An unterminated fence runs to the end of the body — the
/// alternative (treating it as literal) means a message someone is
/// mid-way through typing flickers between two renderings.
pub fn split_blocks(body: &str) -> Vec<RawBlock> {
    let mut out: Vec<RawBlock> = Vec::new();
    let mut para = String::new();
    let mut lines = body.split('\n').peekable();

    let flush = |para: &mut String, out: &mut Vec<RawBlock>| {
        if !para.is_empty() {
            out.push(RawBlock::Paragraph(std::mem::take(para)));
        }
    };

    while let Some(line) = lines.next() {
        let trimmed = line.trim_start();

        if let Some(rest) = trimmed.strip_prefix("```") {
            flush(&mut para, &mut out);

            // A fence that opens and closes on one line — ```like this```
            // — is by far the most common way someone types a code block
            // in a chat box, because chat boxes send on Enter. Treating
            // it as an *opening* fence made the rest of the line the
            // "language", scanned for a close that never came, and
            // produced an empty code block: a blank row where the user's
            // text should be.
            if let Some(inner) = rest.strip_suffix("```") {
                out.push(RawBlock::Code {
                    text: inner.to_string(),
                    language: None,
                });
                continue;
            }

            let language = {
                let l = rest.trim();
                if l.is_empty() {
                    None
                } else {
                    Some(l.to_string())
                }
            };
            let mut code = String::new();
            for l in lines.by_ref() {
                if l.trim_start().starts_with("```") {
                    break;
                }
                if !code.is_empty() {
                    code.push('\n');
                }
                code.push_str(l);
            }
            out.push(RawBlock::Code {
                text: code,
                language,
            });
            continue;
        }

        if trimmed.starts_with('>') {
            flush(&mut para, &mut out);
            let mut depth = 0u8;
            let mut rest = trimmed;
            while let Some(r) = rest.strip_prefix('>') {
                depth = depth.saturating_add(1);
                rest = r.trim_start();
            }
            let mut quoted = rest.to_string();
            // Consume following lines at the same depth.
            while let Some(next) = lines.peek() {
                let nt = next.trim_start();
                if !nt.starts_with('>') {
                    break;
                }
                let mut d = 0u8;
                let mut r = nt;
                while let Some(s) = r.strip_prefix('>') {
                    d = d.saturating_add(1);
                    r = s.trim_start();
                }
                if d != depth {
                    break;
                }
                quoted.push('\n');
                quoted.push_str(r);
                lines.next();
            }
            out.push(RawBlock::Quote {
                text: quoted,
                depth,
            });
            continue;
        }

        if !para.is_empty() {
            para.push('\n');
        }
        para.push_str(line);
    }
    flush(&mut para, &mut out);
    out
}

/// Parse inline markdown into styled text.
///
/// Never fails and never panics: any construct that doesn't close
/// renders as the literal characters that were typed.
pub fn parse_inline(src: &str) -> ParsedText {
    parse_inline_with(src, &Linkifier::default())
}

/// [`parse_inline`], with `[label](url)` links checked against `links`
/// rather than the default scheme list.
pub fn parse_inline_with(src: &str, links: &Linkifier) -> ParsedText {
    let mut b = SpanBuilder::new();
    scan(src, Style::default(), 0, links, &mut b);
    let out = b.finish();
    out.debug_assert_well_formed();
    out
}

fn scan(src: &str, base: Style, depth: u8, links: &Linkifier, out: &mut SpanBuilder) {
    let bytes = src.as_bytes();
    let mut i = 0usize;
    // Start of the current literal run, flushed lazily so plain text
    // costs one push rather than one per character.
    let mut lit = 0usize;

    macro_rules! flush_lit {
        ($upto:expr) => {
            if $upto > lit {
                out.push(&src[lit..$upto], base);
            }
        };
    }

    while i < bytes.len() {
        let c = bytes[i];

        // Backslash escape: the next character is literal, whatever it is.
        if c == b'\\' && i + 1 < bytes.len() {
            let next = i + 1;
            let ch_end = next + utf8_len(bytes[next]);
            let ch_end = ch_end.min(bytes.len());
            if is_escapable(&src[next..ch_end]) {
                flush_lit!(i);
                out.push(&src[next..ch_end], base);
                i = ch_end;
                lit = i;
                continue;
            }
            i += 1;
            continue;
        }

        // `code` — inert contents, so it is tried before everything else.
        if c == b'`' {
            if let Some((inner, end)) = code_span(bytes, i) {
                flush_lit!(i);
                out.push(strip_code_pad(&src[inner]), base.with_attrs(Attrs::CODE));
                i = end;
                lit = i;
                continue;
            }
            // An unmatched run is literal text. Skip the *whole* run, not
            // one byte: retrying at the second backtick of ``` would let
            // a shorter run inside it close against something later.
            i += backtick_run(bytes, i);
            continue;
        }

        if depth < MAX_DEPTH {
            // **bold** and ~~strike~~ — two-character delimiters first, so
            // `**` is never mistaken for an empty `*` pair.
            let two = match c {
                b'*' if bytes.get(i + 1) == Some(&b'*') => Some((Attrs::BOLD, "**")),
                b'~' if bytes.get(i + 1) == Some(&b'~') => Some((Attrs::STRIKETHROUGH, "~~")),
                _ => None,
            };
            if let Some((attr, delim)) = two {
                if can_open(bytes, i + 2) {
                    if let Some(close) = find_delim(src, i + 2, delim) {
                        flush_lit!(i);
                        scan(
                            &src[i + 2..close],
                            base.with_attrs(attr),
                            depth + 1,
                            links,
                            out,
                        );
                        i = close + 2;
                        lit = i;
                        continue;
                    }
                }
                i += 2;
                continue;
            }

            // *italic* / _italic_.
            //
            // `_` additionally requires non-alphanumeric neighbours so
            // that snake_case identifiers, which turn up constantly in
            // this project's chat, survive intact.
            if (c == b'*' && can_open(bytes, i + 1))
                || (c == b'_' && can_open(bytes, i + 1) && intraword_ok(bytes, i))
            {
                if let Some(close) = find_italic_close(src, i + 1, c) {
                    flush_lit!(i);
                    scan(
                        &src[i + 1..close],
                        base.with_attrs(Attrs::ITALIC),
                        depth + 1,
                        links,
                        out,
                    );
                    i = close + 1;
                    lit = i;
                    continue;
                }
                i += 1;
                continue;
            }

            // [label](url)
            if c == b'[' {
                if let Some((label, href, end)) = parse_link(src, i) {
                    if links.allows(href) {
                        flush_lit!(i);
                        // The id must exist before the label is emitted
                        // so it can ride in the label's Style; the
                        // visible range is patched in afterwards.
                        let id = out.reserve_link(href.to_string());
                        let start = out.len();
                        let mut label_style = base.with_attrs(Attrs::UNDERLINE);
                        label_style.link = Some(id);
                        // MAX_DEPTH, not depth + 1: a link label renders
                        // as plain text. Nested links are invalid
                        // markdown, and `parse_link` balances brackets,
                        // so re-scanning the label would happily parse
                        // the inner one.
                        scan(label, label_style, MAX_DEPTH, links, out);
                        out.set_link_range(id, start..out.len());
                        i = end;
                        lit = i;
                        continue;
                    }
                    // Disallowed scheme: fall through so the whole
                    // `[label](url)` renders as literal text. The user
                    // sees what was typed rather than a link they have
                    // no way to inspect.
                }
                i += 1;
                continue;
            }
        }

        i += 1;
    }
    flush_lit!(bytes.len());
}

/// Bytes markdown lets you escape. A backslash before anything else is
/// itself literal — `C:\path` must not lose its separators.
fn is_escapable(s: &str) -> bool {
    matches!(
        s,
        "\\" | "*" | "_" | "`" | "~" | "[" | "]" | "(" | ")" | ">" | "#"
    )
}

fn utf8_len(b: u8) -> usize {
    if b < 0x80 {
        1
    } else if b >> 5 == 0b110 {
        2
    } else if b >> 4 == 0b1110 {
        3
    } else if b >> 3 == 0b11110 {
        4
    } else {
        1
    }
}

/// Length of the run of backticks starting at `at`.
fn backtick_run(bytes: &[u8], at: usize) -> usize {
    let mut n = 0usize;
    while bytes.get(at + n) == Some(&b'`') {
        n += 1;
    }
    n
}

/// A code span starting at `at`: the byte range of its contents, and the
/// offset just past its closing run.
///
/// CommonMark's rule, and the reason `` `` `hello` `` `` works: a span
/// opens with a run of N backticks and closes on the next run of
/// *exactly* N. Matching a single backtick against the next single
/// backtick — what this used to do — turns ```` ``hello`` ```` into two
/// empty spans on either side of unstyled text, which is exactly how it
/// rendered.
///
/// There is no escaping inside a code span and none in front of one: a
/// `\``  is consumed by the backslash branch in `scan` before the
/// backtick is ever seen here.
fn code_span(bytes: &[u8], at: usize) -> Option<(Range<usize>, usize)> {
    let n = backtick_run(bytes, at);
    let mut j = at + n;
    while j < bytes.len() {
        if bytes[j] == b'`' {
            let m = backtick_run(bytes, j);
            if m == n {
                return Some((at + n..j, j + m));
            }
            j += m;
            continue;
        }
        j += 1;
    }
    None
}

/// Strip one leading and one trailing space, when both are there and the
/// content is not all spaces.
///
/// This is what lets a span hold a backtick of its own: `` ` `` is a
/// two-backtick span containing " ` ", and without the strip it would
/// render with the padding the author only added to separate the
/// delimiters.
fn strip_code_pad(s: &str) -> &str {
    let b = s.as_bytes();
    if b.len() >= 2 && b[0] == b' ' && b[b.len() - 1] == b' ' && b.iter().any(|&c| c != b' ') {
        &s[1..s.len() - 1]
    } else {
        s
    }
}

/// Next unescaped occurrence of a multi-byte delimiter, skipping code
/// spans so that `` **a `b**` c** `` closes at the last `**`.
fn find_delim(src: &str, from: usize, delim: &str) -> Option<usize> {
    let bytes = src.as_bytes();
    let d = delim.as_bytes();
    let mut i = from;
    while i + d.len() <= bytes.len() {
        if bytes[i] == b'\\' {
            i += 2;
            continue;
        }
        if bytes[i] == b'`' {
            match code_span(bytes, i) {
                Some((_, end)) => {
                    i = end;
                    continue;
                }
                None => return None,
            }
        }
        if bytes[i..].starts_with(d) {
            // An empty span (`****`) is not emphasis, and a closer must
            // be right-flanking.
            if i == from || !can_close(bytes, i) {
                i += d.len();
                continue;
            }
            return Some(i);
        }
        i += 1;
    }
    None
}

/// Close delimiter for single-character emphasis. Rejects a `**` run so
/// that `*a**b*` doesn't close on the doubled pair.
fn find_italic_close(src: &str, from: usize, open: u8) -> Option<usize> {
    let bytes = src.as_bytes();
    let mut i = from;
    while i < bytes.len() {
        if bytes[i] == b'\\' {
            i += 2;
            continue;
        }
        if bytes[i] == b'`' {
            match code_span(bytes, i) {
                Some((_, end)) => {
                    i = end;
                    continue;
                }
                None => return None,
            }
        }
        if bytes[i] == open {
            if bytes.get(i + 1) == Some(&open) {
                i += 2;
                continue;
            }
            if i == from || !can_close(bytes, i) {
                i += 1;
                continue;
            }
            if open == b'_' && !intraword_close_ok(bytes, i) {
                i += 1;
                continue;
            }
            return Some(i);
        }
        i += 1;
    }
    None
}

/// CommonMark's left-flanking rule, which is what stops `2 * 3 * 4`
/// from becoming `2  3  4`.
///
/// An opening delimiter must be followed by non-whitespace. Arithmetic,
/// bullet-ish prose ("see * the docs") and trailing asterisks all rely
/// on this; without it the parser eats punctuation out of ordinary
/// sentences, which is the single most annoying way a chat markdown
/// implementation can be wrong.
fn can_open(bytes: &[u8], after: usize) -> bool {
    bytes.get(after).is_some_and(|b| !b.is_ascii_whitespace())
}

/// The right-flanking counterpart: a closing delimiter must be preceded
/// by non-whitespace, so `a * b *` doesn't close on the trailing one.
fn can_close(bytes: &[u8], at: usize) -> bool {
    at > 0 && !bytes[at - 1].is_ascii_whitespace()
}

/// `_` opens emphasis only at a word boundary, so snake_case survives.
fn intraword_ok(bytes: &[u8], i: usize) -> bool {
    let before_ok = i == 0 || !bytes[i - 1].is_ascii_alphanumeric();
    let after_ok = bytes.get(i + 1).is_some_and(|b| *b != b'_');
    before_ok && after_ok
}

/// `_` closes emphasis only at a word boundary.
fn intraword_close_ok(bytes: &[u8], i: usize) -> bool {
    bytes.get(i + 1).is_none_or(|b| !b.is_ascii_alphanumeric())
}

/// Parse `[label](url)` starting at `open`. Returns label, href and the
/// offset one past the closing paren.
fn parse_link(src: &str, open: usize) -> Option<(&str, &str, usize)> {
    let bytes = src.as_bytes();
    // Label: balanced brackets, no nesting beyond one level needed.
    let mut i = open + 1;
    let mut depth = 1usize;
    while i < bytes.len() {
        match bytes[i] {
            b'\\' => {
                i += 2;
                continue;
            }
            b'[' => depth += 1,
            b']' => {
                depth -= 1;
                if depth == 0 {
                    break;
                }
            }
            _ => {}
        }
        i += 1;
    }
    if depth != 0 || i >= bytes.len() {
        return None;
    }
    let label_end = i;
    if bytes.get(label_end + 1) != Some(&b'(') {
        return None;
    }
    let url_start = label_end + 2;
    let mut j = url_start;
    while j < bytes.len() && bytes[j] != b')' {
        if bytes[j] == b'\\' {
            j += 2;
            continue;
        }
        // A URL never contains whitespace; bail rather than swallowing
        // the rest of the line looking for a paren.
        if bytes[j].is_ascii_whitespace() {
            return None;
        }
        j += 1;
    }
    if j >= bytes.len() {
        return None;
    }
    let label = &src[open + 1..label_end];
    let href = &src[url_start..j];
    if label.is_empty() || href.is_empty() {
        return None;
    }
    Some((label, href, j + 1))
}

// ---- input tinting --------------------------------------------------

/// One tintable region of *source* text, for the compose box.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct SourceSpan {
    pub start: usize,
    pub end: usize,
    pub attrs: Attrs,
    /// True for the delimiter characters themselves, which the input
    /// box dims rather than styles.
    pub delim: bool,
}

/// Locate the markdown delimiters in `src` that will actually be
/// consumed, reporting ranges **in the source**.
///
/// This exists because [`parse_inline`] reports ranges in the *rendered*
/// text, with the delimiters removed — exactly the offsets the compose
/// box does not have. Rather than teach the renderer to carry source
/// offsets through its recursion (a change to a heavily-tested function
/// for a cosmetic feature), this is a separate, deliberately shallower
/// pass that reuses the *rules* that are actually subtle: `can_open` /
/// `can_close` flanking and the `_` intraword guards.
///
/// **It is an approximation, and that is fine here.** It doesn't nest,
/// doesn't handle links, and takes the first valid closer. Being wrong
/// in the compose box means a character is tinted that won't be, on text
/// the user is still editing and can see; being wrong in the renderer
/// would change what a message *says*. The two failure modes are not
/// comparable, which is why they don't share a code path.
pub fn scan_delims(src: &str) -> Vec<SourceSpan> {
    let bytes = src.as_bytes();
    let mut out = Vec::new();
    let mut i = 0usize;

    while i < bytes.len() {
        // A backslash escape hides the next character from tinting, the
        // same way it hides it from the parser.
        if bytes[i] == b'\\' && i + 1 < bytes.len() {
            i += 2;
            continue;
        }

        let (run, attrs) = match bytes[i] {
            b'*' if bytes.get(i + 1) == Some(&b'*') => (2usize, Attrs::BOLD),
            b'*' => (1, Attrs::ITALIC),
            b'_' if intraword_ok(bytes, i) => (1, Attrs::ITALIC),
            b'`' => (backtick_run(bytes, i), Attrs::CODE),
            _ => {
                i += 1;
                continue;
            }
        };

        // Code spans are literal: no escapes, no nesting, and the closer
        // is a run of exactly the same length. They also skip the
        // flanking rules — `` ` `` is a legitimate span whose content is
        // a space, and can_open would reject it.
        let literal = attrs == Attrs::CODE;

        if !literal && !can_open(bytes, i + run) {
            i += run;
            continue;
        }
        let mut j = i + run;
        let close = loop {
            if j >= bytes.len() {
                break None;
            }
            if !literal && bytes[j] == b'\\' {
                j += 2;
                continue;
            }
            if literal {
                if bytes[j] == b'`' {
                    let m = backtick_run(bytes, j);
                    if m == run {
                        break Some(j);
                    }
                    j += m;
                    continue;
                }
                j += 1;
                continue;
            }
            let hit = match run {
                2 => bytes[j] == b'*' && bytes.get(j + 1) == Some(&b'*'),
                _ => bytes[j] == bytes[i] && (bytes[i] != b'_' || intraword_close_ok(bytes, j)),
            };
            if hit && can_close(bytes, j) {
                break Some(j);
            }
            j += 1;
        };

        let Some(close) = close else {
            i += run;
            continue;
        };

        out.push(SourceSpan {
            start: i,
            end: i + run,
            attrs,
            delim: true,
        });
        out.push(SourceSpan {
            start: i + run,
            end: close,
            attrs,
            delim: false,
        });
        out.push(SourceSpan {
            start: close,
            end: close + run,
            attrs,
            delim: true,
        });
        i = close + run;
    }
    out
}