leaf_core/syntax.rs
1//! Syntax highlighting for a fenced code block: the lines of a block and the
2//! language its fence names in, a [`Token`] per byte range out.
3//!
4//! The grammars are Sublime Text's, run by syntect, and the set is `two-face`'s
5//! — bat's 213, which is the set `plates` publishes with, so a block that
6//! highlights in leaf highlights on the published page and the other way
7//! round. What comes out is not a colour and not a scope: it is the eight-way
8//! [`Token`] classing, which is all a palette can usefully tell apart, decided
9//! here once so that four frontends do not each read TextMate scope names.
10//!
11//! # How a scope becomes a token
12//!
13//! syntect's parser leaves a stack of scopes over every byte —
14//! `source.rust meta.function.rust string.quoted.double.rust
15//! punctuation.definition.string.begin.rust` over the `"` that opens a string
16//! literal in a function body. The token for a byte is decided the way a
17//! stylesheet over classed HTML decides it, because that is what `plates`'
18//! stylesheet is and the two should agree:
19//!
20//! - **Innermost scope first.** The nearest scope that names a token wins, as
21//! an inner `<span>`'s own colour beats what it inherits from an outer one.
22//! - **Any atom, not the first.** `punctuation.definition.string.begin` has
23//! both `punctuation` and `string` among its atoms, and the *later* of the
24//! two in [`Token::ALL`] wins — so the quote reads as string, and a `//`
25//! reads as comment.
26//! - **`meta` and `source` name nothing.** `meta.function` wraps a whole
27//! function, signature and body together; colouring it would flood
28//! everything nested inside. They are simply absent from the vocabulary.
29//!
30//! # What it costs
31//!
32//! The syntax set is unpacked from its embedded dump on first use — tens of
33//! milliseconds and about a megabyte of memory — and then shared for the life
34//! of the process. Parsing itself is per block, once per rebuild of that block:
35//! the WYSIWYG map's block cache keeps a block's rows across edits elsewhere in
36//! the document, so typing in a paragraph never re-highlights the code above
37//! it, and typing in a code block re-highlights that block alone.
38
39use std::ops::Range;
40use std::sync::OnceLock;
41
42use syntect::parsing::{ParseState, Scope, ScopeStack, SyntaxSet};
43
44use crate::style::Token;
45
46/// One line's highlighting: byte ranges *into that line*, ascending and
47/// non-overlapping, with the token over each. A byte no range covers carries
48/// no token.
49pub type LineTokens = Vec<(Range<usize>, Token)>;
50
51/// The grammars, unpacked once. `extra_newlines` rather than `extra_no_newlines`
52/// because these grammars match a line *with* its terminator — a `//` comment's
53/// pattern runs to `\n` — so [`highlight`] feeds each line one, and a set built
54/// for the other convention would match differently at every end of line.
55fn syntax_set() -> &'static SyntaxSet {
56 static SET: OnceLock<SyntaxSet> = OnceLock::new();
57 SET.get_or_init(two_face::syntax::extra_newlines)
58}
59
60/// The atom numbers that name each token, resolved once. A [`Scope`] is a
61/// packed list of interned atoms, and interning `keyword` here yields the same
62/// number the loaded grammars' `keyword.control.rust` carries in its first
63/// slot — scopes are dumped as strings and re-interned on load — so classing
64/// a scope is eight `u16` comparisons per atom, no string built.
65///
66/// Two tokens have two atoms each: `keyword`/`storage` are both
67/// [`Token::Keyword`] and `entity`/`variable` both [`Token::Entity`], as in
68/// `plates`' stylesheet.
69fn atoms() -> &'static [(u16, Token)] {
70 static ATOMS: OnceLock<Vec<(u16, Token)>> = OnceLock::new();
71 ATOMS.get_or_init(|| {
72 [
73 ("punctuation", Token::Punctuation),
74 ("keyword", Token::Keyword),
75 ("storage", Token::Keyword),
76 ("entity", Token::Entity),
77 ("variable", Token::Entity),
78 ("support", Token::Support),
79 ("constant", Token::Constant),
80 ("string", Token::String),
81 ("comment", Token::Comment),
82 ("invalid", Token::Invalid),
83 ]
84 .into_iter()
85 // A single-atom scope name always parses; `expect` documents that.
86 .map(|(name, t)| {
87 (
88 Scope::new(name).expect("a bare atom is a scope").atom_at(0),
89 t,
90 )
91 })
92 .collect()
93 })
94}
95
96/// The token one scope names, by the highest-precedence token any of its atoms
97/// names — or `None` for a scope like `meta.function` or `source.rust` that
98/// names no token at all.
99fn token_of_scope(scope: Scope) -> Option<Token> {
100 let atoms = atoms();
101 (0..scope.len() as usize)
102 .map(|i| scope.atom_at(i))
103 .filter_map(|a| atoms.iter().find(|(n, _)| *n == a).map(|(_, t)| *t))
104 .max_by_key(|t| t.index())
105}
106
107/// The token a scope stack puts on the bytes under it: the innermost scope
108/// that names one.
109fn token_of_stack(stack: &ScopeStack) -> Option<Token> {
110 stack
111 .as_slice()
112 .iter()
113 .rev()
114 .find_map(|s| token_of_scope(*s))
115}
116
117/// Whether `lang` — a fence's info string, trimmed — names a grammar: `rust`,
118/// `rs`, `zig` and `swift` do; `text`, `""` and `no-such-language` do not.
119pub fn knows(lang: &str) -> bool {
120 syntax_set().find_syntax_by_token(lang).is_some()
121}
122
123/// Highlight the lines of one fenced code block written in `lang`.
124///
125/// One `Vec` per line, in order, each a list of byte ranges *into that line*
126/// with the token over them, ascending and non-overlapping. A byte no range
127/// covers carries no token — an identifier the grammar leaves as plain
128/// `source.rust`, the space between two words — and draws in the code colour.
129/// The lines are the block's text split on `\n`, without terminators; the
130/// terminators are supplied here, since the grammars expect them.
131///
132/// `None` when `lang` names no grammar the set carries, which is the whole of
133/// the difference between "a language we can't highlight" and "a block with
134/// nothing worth colouring": a caller draws either in plain code colour, but
135/// only the second is a highlighted block.
136pub fn highlight(lang: &str, lines: &[&str]) -> Option<Vec<LineTokens>> {
137 let set = syntax_set();
138 let syntax = set.find_syntax_by_token(lang)?;
139 let mut state = ParseState::new(syntax);
140 let mut stack = ScopeStack::new();
141 let mut out = Vec::with_capacity(lines.len());
142 let mut buf = String::new();
143 for line in lines {
144 buf.clear();
145 buf.push_str(line);
146 buf.push('\n');
147 let mut spans: LineTokens = Vec::new();
148 // A grammar that fails mid-block — a stack that overflows, a pattern the
149 // fancy-regex engine will not run — loses colour from that line on
150 // rather than losing the block: the lines so far keep their tokens and
151 // the rest carry none, which is what an unknown language would draw.
152 let Ok(ops) = state.parse_line(&buf, set) else {
153 out.push(spans);
154 break;
155 };
156 let mut last = 0usize;
157 // Each op sits at the byte it takes effect from; the bytes since the
158 // previous op were under the stack as it stood. The terminator this
159 // function added is clipped off — the caller's line has no byte there.
160 for (pos, op) in &ops {
161 let pos = (*pos).min(line.len());
162 if pos > last {
163 push_span(&mut spans, last..pos, token_of_stack(&stack));
164 last = pos;
165 }
166 // `apply` fails only on a `Pop` past the stack's bottom, which a
167 // grammar does not produce; and the tokens for a line that somehow
168 // did would merely be off for the rest of the block.
169 let _ = stack.apply(op);
170 }
171 if line.len() > last {
172 push_span(&mut spans, last..line.len(), token_of_stack(&stack));
173 }
174 out.push(spans);
175 }
176 // A parse that gave up mid-block leaves the later lines unlisted; pad so a
177 // caller can index by line number.
178 out.resize_with(lines.len(), Vec::new);
179 Some(out)
180}
181
182/// Append a range with a token, merging into the previous range when it carries
183/// the same token — a run of ops inside one string literal would otherwise
184/// leave the literal as a dozen adjacent `String` spans, and a frontend that
185/// coalesces glyphs by style merges them anyway. A range with no token is not
186/// recorded at all.
187fn push_span(spans: &mut LineTokens, range: Range<usize>, token: Option<Token>) {
188 let Some(token) = token else { return };
189 if let Some((last, t)) = spans.last_mut()
190 && *t == token
191 && last.end == range.start
192 {
193 last.end = range.end;
194 return;
195 }
196 spans.push((range, token));
197}
198
199#[cfg(test)]
200mod tests {
201 use super::*;
202
203 /// The token over byte `at` of line `line`, or `None`.
204 fn at(spans: &[LineTokens], line: usize, at: usize) -> Option<Token> {
205 spans[line]
206 .iter()
207 .find(|(r, _)| r.contains(&at))
208 .map(|(_, t)| *t)
209 }
210
211 #[test]
212 fn a_rust_block_is_classed_the_way_a_reader_expects() {
213 let lines = [
214 "fn main() {",
215 " let s = \"hi\"; // greet",
216 " println!(\"{}\", 42);",
217 "}",
218 ];
219 let spans = highlight("rust", &lines).expect("rust is a known language");
220 assert_eq!(spans.len(), lines.len());
221 // `fn` and `let` are keywords; `main` at its definition is an entity.
222 assert_eq!(at(&spans, 0, 0), Some(Token::Keyword));
223 assert_eq!(at(&spans, 0, 3), Some(Token::Entity));
224 assert_eq!(at(&spans, 1, 4), Some(Token::Keyword));
225 // The string, quotes included — the opening quote's scope is
226 // `punctuation.definition.string.begin`, and string outranks punctuation.
227 let quote = lines[1].find('"').unwrap();
228 assert_eq!(at(&spans, 1, quote), Some(Token::String));
229 assert_eq!(at(&spans, 1, quote + 1), Some(Token::String));
230 // The comment, its `//` included.
231 let slash = lines[1].find("//").unwrap();
232 assert_eq!(at(&spans, 1, slash), Some(Token::Comment));
233 assert_eq!(at(&spans, 1, slash + 4), Some(Token::Comment));
234 // A number is a constant.
235 let num = lines[2].find("42").unwrap();
236 assert_eq!(at(&spans, 2, num), Some(Token::Constant));
237 // Braces are punctuation.
238 assert_eq!(at(&spans, 3, 0), Some(Token::Punctuation));
239 }
240
241 #[test]
242 fn spans_are_ascending_and_within_their_line() {
243 let lines = [
244 "const x: [u8; 3] = [1, 2, 3]; // n",
245 "",
246 "fn f() -> u8 { x[0] }",
247 ];
248 let spans = highlight("rust", &lines).unwrap();
249 for (i, line) in spans.iter().enumerate() {
250 let mut end = 0;
251 for (r, _) in line {
252 assert!(r.start >= end, "line {i}: {r:?} overlaps or goes backwards");
253 assert!(
254 r.end <= lines[i].len(),
255 "line {i}: {r:?} runs past the line"
256 );
257 assert!(r.start < r.end, "line {i}: {r:?} is empty");
258 end = r.end;
259 }
260 }
261 assert!(spans[1].is_empty(), "an empty line has nothing to class");
262 }
263
264 /// A block's grammar state runs across its lines: a block comment opened on
265 /// one line is still a comment on the next.
266 #[test]
267 fn state_carries_from_line_to_line() {
268 let lines = ["/* a", " b */ let c = 1;"];
269 let spans = highlight("rust", &lines).unwrap();
270 assert_eq!(at(&spans, 1, 3), Some(Token::Comment));
271 assert_eq!(at(&spans, 1, 8), Some(Token::Keyword));
272 }
273
274 /// The languages this organisation is written in are all in the set —
275 /// the reason it is two-face's and not syntect's own.
276 #[test]
277 fn the_org_languages_are_known() {
278 for lang in [
279 "rust",
280 "rs",
281 "zig",
282 "swift",
283 "toml",
284 "typescript",
285 "ts",
286 "js",
287 "sh",
288 "md",
289 ] {
290 assert!(knows(lang), "{lang} should resolve to a grammar");
291 }
292 assert!(!knows(""));
293 assert!(!knows("no-such-language"));
294 assert!(highlight("no-such-language", &["x"]).is_none());
295 }
296
297 /// Alike tokens over adjacent bytes merge, so a string is one span even
298 /// though the grammar pushes and pops several scopes across it.
299 #[test]
300 fn adjacent_alike_spans_merge() {
301 let spans = highlight("rust", &["let s = \"a b c\";"]).unwrap();
302 let strings: Vec<_> = spans[0]
303 .iter()
304 .filter(|(_, t)| *t == Token::String)
305 .collect();
306 assert_eq!(strings.len(), 1, "{:?}", spans[0]);
307 assert_eq!(strings[0].0, 8..15);
308 }
309}