1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
//! [`Kind`]: the one kind type every forged language uses for its tokens and
//! nodes.
use fmt;
use TokenKind;
/// The kind of a token or node in a forged language's syntax tree.
///
/// A schematic declares its vocabulary as text — rule names, keywords,
/// punctuation — so a forged language cannot have a Rust `enum` of its own.
/// `Kind` stands in for one: a small `Copy` value, assigned when the language is
/// forged, that compares as cheaply as an enum discriminant. Tokens and nodes
/// share the type, the model [`syntax_lang`] is built on.
///
/// Kinds are looked up by name with [`Language::kind`](crate::Language::kind)
/// and named with [`Language::kind_name`](crate::Language::kind_name). Look a
/// kind up once, keep it, and compare against it while walking trees: the
/// comparison is a single integer compare.
///
/// | Name | Kind of |
/// |---|---|
/// | a rule name, such as `let_stmt` | the node the rule builds |
/// | a literal's text, such as `let` or `+=` | the keyword or symbol token |
/// | a Pratt level's `node`, or `binary`, `prefix`, `postfix` | an operator node |
/// | `IDENT`, `NUMBER`, `STRING`, `NEWLINE` | the built-in token classes |
/// | `WHITESPACE`, `COMMENT` | trivia tokens |
/// | `UNKNOWN` | a character the lexer did not recognize (trivia) |
/// | `ERROR` | a node wrapping tokens the parser skipped |
///
/// A kind belongs to the language that produced it. Comparing kinds from two
/// different languages is meaningless, and naming one language's kind with
/// another language returns the wrong name or `"<unknown>"`.
///
/// # Trivia
///
/// [`TokenKind::is_trivia`] answers without the language: whitespace, comments,
/// unrecognized characters, and — unless the schematic sets `newlines = true` —
/// line breaks are trivia. Trivia is kept in the tree, so it stays lossless,
/// but the parser never sees it.
///
/// # Examples
///
/// ```
/// use lang_forge::Language;
/// use lang_forge::syntax_lang::TokenKind;
///
/// let lang = Language::from_lsf(
/// r#"
/// [language]
/// name = "sum"
///
/// [rules]
/// sum = "NUMBER ('+' NUMBER)*"
/// "#,
/// )?;
///
/// let plus = lang.kind("+").expect("the grammar uses '+'");
/// let space = lang.kind("WHITESPACE").expect("built in");
/// assert_eq!(lang.kind_name(plus), "+");
/// assert!(space.is_trivia());
/// assert!(!plus.is_trivia());
///
/// let parse = lang.parse("1 + 2");
/// let pluses = parse.tree().tokens().filter(|t| *t.kind() == plus).count();
/// assert_eq!(pluses, 1);
/// # Ok::<(), lang_forge::Error>(())
/// ```
;
/// The bit that marks a kind as trivia. Kept inside the value so that
/// [`TokenKind::is_trivia`] needs no access to the language.
const TRIVIA: u16 = 0x8000;
/// The largest number of kinds a language can have: indexes use the 15 bits
/// below the trivia flag.
pub const MAX_KINDS: usize = TRIVIA as usize;