Skip to main content

leaf_core/
syntax.rs

1//! Syntax highlighting for a fenced code block: the lines of a block and the
2//! language its fence names in, a [`Token`] per byte range out.
3//!
4//! The grammars are Sublime Text's, run by syntect, and the set is `two-face`'s
5//! — bat's 213, which is the set `plates` publishes with, so a block that
6//! highlights in leaf highlights on the published page and the other way
7//! round. What comes out is not a colour and not a scope: it is the eight-way
8//! [`Token`] classing, which is all a palette can usefully tell apart, decided
9//! here once so that four frontends do not each read TextMate scope names.
10//!
11//! # How a scope becomes a token
12//!
13//! syntect's parser leaves a stack of scopes over every byte —
14//! `source.rust meta.function.rust string.quoted.double.rust
15//! punctuation.definition.string.begin.rust` over the `"` that opens a string
16//! literal in a function body. The token for a byte is decided the way a
17//! stylesheet over classed HTML decides it, because that is what `plates`'
18//! stylesheet is and the two should agree:
19//!
20//! - **Innermost scope first.** The nearest scope that names a token wins, as
21//!   an inner `<span>`'s own colour beats what it inherits from an outer one.
22//! - **Any atom, not the first.** `punctuation.definition.string.begin` has
23//!   both `punctuation` and `string` among its atoms, and the *later* of the
24//!   two in [`Token::ALL`] wins — so the quote reads as string, and a `//`
25//!   reads as comment.
26//! - **`meta` and `source` name nothing.** `meta.function` wraps a whole
27//!   function, signature and body together; colouring it would flood
28//!   everything nested inside. They are simply absent from the vocabulary.
29//!
30//! # What it costs
31//!
32//! The syntax set is unpacked from its embedded dump on first use — tens of
33//! milliseconds and about a megabyte of memory — and then shared for the life
34//! of the process. Parsing itself is per block, once per rebuild of that block:
35//! the WYSIWYG map's block cache keeps a block's rows across edits elsewhere in
36//! the document, so typing in a paragraph never re-highlights the code above
37//! it, and typing in a code block re-highlights that block alone.
38
39use std::ops::Range;
40use std::sync::OnceLock;
41
42use syntect::parsing::{ParseState, Scope, ScopeStack, SyntaxSet};
43
44use crate::style::Token;
45
46/// One line's highlighting: byte ranges *into that line*, ascending and
47/// non-overlapping, with the token over each. A byte no range covers carries
48/// no token.
49pub type LineTokens = Vec<(Range<usize>, Token)>;
50
51/// The grammars, unpacked once. `extra_newlines` rather than `extra_no_newlines`
52/// because these grammars match a line *with* its terminator — a `//` comment's
53/// pattern runs to `\n` — so [`highlight`] feeds each line one, and a set built
54/// for the other convention would match differently at every end of line.
55fn syntax_set() -> &'static SyntaxSet {
56    static SET: OnceLock<SyntaxSet> = OnceLock::new();
57    SET.get_or_init(two_face::syntax::extra_newlines)
58}
59
60/// The atom numbers that name each token, resolved once. A [`Scope`] is a
61/// packed list of interned atoms, and interning `keyword` here yields the same
62/// number the loaded grammars' `keyword.control.rust` carries in its first
63/// slot — scopes are dumped as strings and re-interned on load — so classing
64/// a scope is eight `u16` comparisons per atom, no string built.
65///
66/// Two tokens have two atoms each: `keyword`/`storage` are both
67/// [`Token::Keyword`] and `entity`/`variable` both [`Token::Entity`], as in
68/// `plates`' stylesheet.
69fn atoms() -> &'static [(u16, Token)] {
70    static ATOMS: OnceLock<Vec<(u16, Token)>> = OnceLock::new();
71    ATOMS.get_or_init(|| {
72        [
73            ("punctuation", Token::Punctuation),
74            ("keyword", Token::Keyword),
75            ("storage", Token::Keyword),
76            ("entity", Token::Entity),
77            ("variable", Token::Entity),
78            ("support", Token::Support),
79            ("constant", Token::Constant),
80            ("string", Token::String),
81            ("comment", Token::Comment),
82            ("invalid", Token::Invalid),
83        ]
84        .into_iter()
85        // A single-atom scope name always parses; `expect` documents that.
86        .map(|(name, t)| {
87            (
88                Scope::new(name).expect("a bare atom is a scope").atom_at(0),
89                t,
90            )
91        })
92        .collect()
93    })
94}
95
96/// The token one scope names, by the highest-precedence token any of its atoms
97/// names — or `None` for a scope like `meta.function` or `source.rust` that
98/// names no token at all.
99fn token_of_scope(scope: Scope) -> Option<Token> {
100    let atoms = atoms();
101    (0..scope.len() as usize)
102        .map(|i| scope.atom_at(i))
103        .filter_map(|a| atoms.iter().find(|(n, _)| *n == a).map(|(_, t)| *t))
104        .max_by_key(|t| t.index())
105}
106
107/// The token a scope stack puts on the bytes under it: the innermost scope
108/// that names one.
109fn token_of_stack(stack: &ScopeStack) -> Option<Token> {
110    stack
111        .as_slice()
112        .iter()
113        .rev()
114        .find_map(|s| token_of_scope(*s))
115}
116
117/// Whether `lang` — a fence's info string, trimmed — names a grammar: `rust`,
118/// `rs`, `zig` and `swift` do; `text`, `""` and `no-such-language` do not.
119pub fn knows(lang: &str) -> bool {
120    syntax_set().find_syntax_by_token(lang).is_some()
121}
122
123/// Highlight the lines of one fenced code block written in `lang`.
124///
125/// One `Vec` per line, in order, each a list of byte ranges *into that line*
126/// with the token over them, ascending and non-overlapping. A byte no range
127/// covers carries no token — an identifier the grammar leaves as plain
128/// `source.rust`, the space between two words — and draws in the code colour.
129/// The lines are the block's text split on `\n`, without terminators; the
130/// terminators are supplied here, since the grammars expect them.
131///
132/// `None` when `lang` names no grammar the set carries, which is the whole of
133/// the difference between "a language we can't highlight" and "a block with
134/// nothing worth colouring": a caller draws either in plain code colour, but
135/// only the second is a highlighted block.
136pub fn highlight(lang: &str, lines: &[&str]) -> Option<Vec<LineTokens>> {
137    let set = syntax_set();
138    let syntax = set.find_syntax_by_token(lang)?;
139    let mut state = ParseState::new(syntax);
140    let mut stack = ScopeStack::new();
141    let mut out = Vec::with_capacity(lines.len());
142    let mut buf = String::new();
143    for line in lines {
144        buf.clear();
145        buf.push_str(line);
146        buf.push('\n');
147        let mut spans: LineTokens = Vec::new();
148        // A grammar that fails mid-block — a stack that overflows, a pattern the
149        // fancy-regex engine will not run — loses colour from that line on
150        // rather than losing the block: the lines so far keep their tokens and
151        // the rest carry none, which is what an unknown language would draw.
152        let Ok(ops) = state.parse_line(&buf, set) else {
153            out.push(spans);
154            break;
155        };
156        let mut last = 0usize;
157        // Each op sits at the byte it takes effect from; the bytes since the
158        // previous op were under the stack as it stood. The terminator this
159        // function added is clipped off — the caller's line has no byte there.
160        for (pos, op) in &ops {
161            let pos = (*pos).min(line.len());
162            if pos > last {
163                push_span(&mut spans, last..pos, token_of_stack(&stack));
164                last = pos;
165            }
166            // `apply` fails only on a `Pop` past the stack's bottom, which a
167            // grammar does not produce; and the tokens for a line that somehow
168            // did would merely be off for the rest of the block.
169            let _ = stack.apply(op);
170        }
171        if line.len() > last {
172            push_span(&mut spans, last..line.len(), token_of_stack(&stack));
173        }
174        out.push(spans);
175    }
176    // A parse that gave up mid-block leaves the later lines unlisted; pad so a
177    // caller can index by line number.
178    out.resize_with(lines.len(), Vec::new);
179    Some(out)
180}
181
182/// Append a range with a token, merging into the previous range when it carries
183/// the same token — a run of ops inside one string literal would otherwise
184/// leave the literal as a dozen adjacent `String` spans, and a frontend that
185/// coalesces glyphs by style merges them anyway. A range with no token is not
186/// recorded at all.
187fn push_span(spans: &mut LineTokens, range: Range<usize>, token: Option<Token>) {
188    let Some(token) = token else { return };
189    if let Some((last, t)) = spans.last_mut()
190        && *t == token
191        && last.end == range.start
192    {
193        last.end = range.end;
194        return;
195    }
196    spans.push((range, token));
197}
198
199#[cfg(test)]
200mod tests {
201    use super::*;
202
203    /// The token over byte `at` of line `line`, or `None`.
204    fn at(spans: &[LineTokens], line: usize, at: usize) -> Option<Token> {
205        spans[line]
206            .iter()
207            .find(|(r, _)| r.contains(&at))
208            .map(|(_, t)| *t)
209    }
210
211    #[test]
212    fn a_rust_block_is_classed_the_way_a_reader_expects() {
213        let lines = [
214            "fn main() {",
215            "    let s = \"hi\"; // greet",
216            "    println!(\"{}\", 42);",
217            "}",
218        ];
219        let spans = highlight("rust", &lines).expect("rust is a known language");
220        assert_eq!(spans.len(), lines.len());
221        // `fn` and `let` are keywords; `main` at its definition is an entity.
222        assert_eq!(at(&spans, 0, 0), Some(Token::Keyword));
223        assert_eq!(at(&spans, 0, 3), Some(Token::Entity));
224        assert_eq!(at(&spans, 1, 4), Some(Token::Keyword));
225        // The string, quotes included — the opening quote's scope is
226        // `punctuation.definition.string.begin`, and string outranks punctuation.
227        let quote = lines[1].find('"').unwrap();
228        assert_eq!(at(&spans, 1, quote), Some(Token::String));
229        assert_eq!(at(&spans, 1, quote + 1), Some(Token::String));
230        // The comment, its `//` included.
231        let slash = lines[1].find("//").unwrap();
232        assert_eq!(at(&spans, 1, slash), Some(Token::Comment));
233        assert_eq!(at(&spans, 1, slash + 4), Some(Token::Comment));
234        // A number is a constant.
235        let num = lines[2].find("42").unwrap();
236        assert_eq!(at(&spans, 2, num), Some(Token::Constant));
237        // Braces are punctuation.
238        assert_eq!(at(&spans, 3, 0), Some(Token::Punctuation));
239    }
240
241    #[test]
242    fn spans_are_ascending_and_within_their_line() {
243        let lines = [
244            "const x: [u8; 3] = [1, 2, 3]; // n",
245            "",
246            "fn f() -> u8 { x[0] }",
247        ];
248        let spans = highlight("rust", &lines).unwrap();
249        for (i, line) in spans.iter().enumerate() {
250            let mut end = 0;
251            for (r, _) in line {
252                assert!(r.start >= end, "line {i}: {r:?} overlaps or goes backwards");
253                assert!(
254                    r.end <= lines[i].len(),
255                    "line {i}: {r:?} runs past the line"
256                );
257                assert!(r.start < r.end, "line {i}: {r:?} is empty");
258                end = r.end;
259            }
260        }
261        assert!(spans[1].is_empty(), "an empty line has nothing to class");
262    }
263
264    /// A block's grammar state runs across its lines: a block comment opened on
265    /// one line is still a comment on the next.
266    #[test]
267    fn state_carries_from_line_to_line() {
268        let lines = ["/* a", "   b */ let c = 1;"];
269        let spans = highlight("rust", &lines).unwrap();
270        assert_eq!(at(&spans, 1, 3), Some(Token::Comment));
271        assert_eq!(at(&spans, 1, 8), Some(Token::Keyword));
272    }
273
274    /// The languages this organisation is written in are all in the set —
275    /// the reason it is two-face's and not syntect's own.
276    #[test]
277    fn the_org_languages_are_known() {
278        for lang in [
279            "rust",
280            "rs",
281            "zig",
282            "swift",
283            "toml",
284            "typescript",
285            "ts",
286            "js",
287            "sh",
288            "md",
289        ] {
290            assert!(knows(lang), "{lang} should resolve to a grammar");
291        }
292        assert!(!knows(""));
293        assert!(!knows("no-such-language"));
294        assert!(highlight("no-such-language", &["x"]).is_none());
295    }
296
297    /// Alike tokens over adjacent bytes merge, so a string is one span even
298    /// though the grammar pushes and pops several scopes across it.
299    #[test]
300    fn adjacent_alike_spans_merge() {
301        let spans = highlight("rust", &["let s = \"a b c\";"]).unwrap();
302        let strings: Vec<_> = spans[0]
303            .iter()
304            .filter(|(_, t)| *t == Token::String)
305            .collect();
306        assert_eq!(strings.len(), 1, "{:?}", spans[0]);
307        assert_eq!(strings[0].0, 8..15);
308    }
309}