Skip to main content

ktrs_parser/builder/
psi_builder.rs

1//! Port of the parse-time half of `PsiBuilderImpl` (token cursor, markers, errors).
2//! Tree construction (`getTreeBuilt`) lives in `tree.rs`.
3
4use ktrs_syntax::SyntaxKind;
5
6use super::marker::Marker;
7use super::production::Production;
8
9/// Constructed (from recycled buffers) in `pool.rs`.
10pub struct PsiBuilder {
11    pub(crate) text: String,
12    /// `myLexStarts`: `lexeme_count + 1` entries, the last one is `text.len()`.
13    pub(crate) lex_starts: Vec<u32>,
14    /// `myLexTypes`; mutable because of `remapCurrentToken`.
15    pub(crate) lex_types: Vec<SyntaxKind>,
16    /// The lexer's kinds before any remap, for chameleons to reuse (see `LazyLeaf`).
17    pub(crate) orig_types: Vec<SyntaxKind>,
18    pub(crate) current_lexeme: usize,
19    pub(super) token_type_checked: bool,
20    pub(crate) production: Production,
21    /// Scratch for `prepareLightTree`'s duplicate-error pass (see `tree.rs`).
22    pub(crate) skipped_errors: Vec<bool>,
23}
24
25impl PsiBuilder {
26    pub(crate) fn lexeme_count(&self) -> usize {
27        self.lex_types.len()
28    }
29
30    pub fn get_original_text(&self) -> &str {
31        &self.text
32    }
33
34    /// No remapper is ever installed for Kotlin, so the cached-type machinery reduces to this.
35    #[inline]
36    pub fn get_token_type(&mut self) -> Option<SyntaxKind> {
37        if self.eof() { None } else { Some(self.lex_types[self.current_lexeme]) }
38    }
39
40    /// Whitespace = `KtTokens.WHITESPACES`, comments = `KtTokens.COMMENTS` (KotlinParserDefinition);
41    /// the KDoc sub-builder uses the same Kotlin language definition.
42    pub fn is_whitespace_or_comment(&self, kind: SyntaxKind) -> bool {
43        kind.is_trivia()
44    }
45
46    pub fn remap_current_token(&mut self, kind: SyntaxKind) {
47        self.lex_types[self.current_lexeme] = kind;
48    }
49
50    pub fn look_ahead(&mut self, steps: i32) -> Option<SyntaxKind> {
51        let mut cur = self.shift_over_whitespace_forward(self.current_lexeme);
52        let mut steps = steps;
53        while steps > 0 {
54            cur = self.shift_over_whitespace_forward(cur + 1);
55            steps -= 1;
56        }
57        self.lex_types.get(cur).copied()
58    }
59
60    pub(crate) fn shift_over_whitespace_forward(&self, lex_index: usize) -> usize {
61        let mut lex_index = lex_index;
62        while lex_index < self.lexeme_count() && self.is_whitespace_or_comment(self.lex_types[lex_index]) {
63            lex_index += 1;
64        }
65        lex_index
66    }
67
68    pub fn raw_lookup(&self, steps: i32) -> Option<SyntaxKind> {
69        let cur = self.current_lexeme as i64 + steps as i64;
70        if cur >= 0 && (cur as usize) < self.lexeme_count() { Some(self.lex_types[cur as usize]) } else { None }
71    }
72
73    pub fn raw_token_type_start(&self, steps: i32) -> i32 {
74        let cur = self.current_lexeme as i64 + steps as i64;
75        if cur < 0 {
76            return -1;
77        }
78        if cur as usize >= self.lexeme_count() {
79            return self.text.len() as i32;
80        }
81        self.lex_starts[cur as usize] as i32
82    }
83
84    pub fn raw_token_index(&self) -> i32 {
85        self.current_lexeme as i32
86    }
87
88    pub fn raw_advance_lexer(&mut self, steps: i32) {
89        assert!(steps >= 0, "Steps must be a positive integer - lexer can only be advanced.");
90        if steps == 0 {
91            return;
92        }
93        self.current_lexeme = (self.current_lexeme + steps as usize).min(self.lexeme_count());
94        self.token_type_checked = false;
95    }
96
97    pub fn advance_lexer(&mut self) {
98        if self.eof() {
99            return;
100        }
101        self.token_type_checked = false;
102        self.current_lexeme += 1;
103    }
104
105    fn skip_whitespace(&mut self) {
106        while self.current_lexeme < self.lexeme_count()
107            && self.is_whitespace_or_comment(self.lex_types[self.current_lexeme])
108        {
109            self.current_lexeme += 1;
110        }
111    }
112
113    pub fn get_current_offset(&mut self) -> i32 {
114        if self.eof() {
115            return self.text.len() as i32;
116        }
117        self.lex_starts[self.current_lexeme] as i32
118    }
119
120    pub fn get_token_text(&mut self) -> Option<&str> {
121        if self.eof() {
122            return None;
123        }
124        let (start, end) = (self.lex_starts[self.current_lexeme], self.lex_starts[self.current_lexeme + 1]);
125        Some(&self.text[start as usize..end as usize])
126    }
127
128    /// Skips whitespace first unless this is the very first (root) marker.
129    pub fn mark(&mut self) -> Marker {
130        if !self.production.is_empty() {
131            self.skip_whitespace();
132        }
133        let id = self.production.allocate(false, self.current_lexeme as i32);
134        self.production.add_marker(id);
135        Marker(id)
136    }
137
138    #[inline]
139    pub fn eof(&mut self) -> bool {
140        if !self.token_type_checked {
141            self.token_type_checked = true;
142            self.skip_whitespace();
143        }
144        self.current_lexeme >= self.lexeme_count()
145    }
146
147    /// Adds an empty error element at the current (possibly pre-whitespace) lexeme; a second
148    /// error at the same lexeme right after the first is ignored.
149    pub fn error(&mut self, message: &str) {
150        let size = self.production.size();
151        if let Some(last) = size.checked_sub(1).and_then(|i| self.production.get_start_marker_at(i)) {
152            let last = self.production.marker(last);
153            if last.is_error_item && last.lexeme == self.current_lexeme as i32 {
154                return;
155            }
156        }
157        let id = self.production.allocate(true, self.current_lexeme as i32);
158        self.production.set_message(id, message);
159        self.production.add_marker(id);
160    }
161
162    /// `PsiBuilderImpl.hasErrorsAfter`.
163    pub fn has_errors_after(&self, marker: Marker) -> bool {
164        self.production.has_errors_after(marker.0)
165    }
166
167    pub(crate) fn precede(&mut self, marker: Marker) -> Marker {
168        let lexeme = self.production.marker(marker.0).lexeme;
169        assert!(lexeme >= 0, "Preceding disposed marker");
170        let id = self.production.allocate(false, lexeme);
171        self.production.add_before(id, marker.0);
172        Marker(id)
173    }
174
175    pub(crate) fn rollback_to(&mut self, marker: Marker) {
176        let lexeme = self.production.marker(marker.0).lexeme;
177        assert!(lexeme >= 0, "The marker is already disposed");
178        self.current_lexeme = lexeme as usize;
179        self.token_type_checked = true;
180        self.production.rollback_to(marker.0);
181    }
182
183    pub(crate) fn drop_marker(&mut self, marker: Marker) {
184        self.production.drop_marker(marker.0);
185    }
186
187    /// `processDone`: `kind` is `myType`, already chosen by the caller (`done`/`error`/...).
188    pub(crate) fn process_done(
189        &mut self,
190        marker: Marker,
191        kind: SyntaxKind,
192        error_message: Option<&str>,
193        before: Option<Marker>,
194    ) {
195        let id = marker.0;
196        assert!(!self.production.marker(id).is_done(), "Marker already done.");
197        let done_lexeme = match before {
198            None => self.current_lexeme as i32,
199            Some(before) => self.production.marker(before.0).lexeme,
200        };
201        let start = self.production.marker(id).lexeme;
202        let left_bound_empty = is_left_bound(kind) && self.is_empty(start, done_lexeme);
203        if let Some(message) = error_message {
204            self.production.set_message(id, message);
205        }
206        let data = self.production.marker_mut(id);
207        data.kind = Some(kind);
208        if left_bound_empty {
209            data.left_binder = Some(super::EdgeBinder::DefaultRight);
210        }
211        data.done_lexeme = done_lexeme;
212        self.production.add_done(id, before.map(|b| b.0));
213    }
214
215    fn is_empty(&self, start_idx: i32, end_idx: i32) -> bool {
216        (start_idx.max(0)..end_idx.max(0)).all(|i| self.is_whitespace_or_comment(self.lex_types[i as usize]))
217    }
218
219    /// `doneBefore(type, before, errorMessage)`: an error item at `before`, then the done.
220    pub(crate) fn done_before_with_error_item(
221        &mut self,
222        marker: Marker,
223        kind: SyntaxKind,
224        before: Marker,
225        error_message: &str,
226    ) {
227        let lexeme = self.production.marker(before.0).lexeme;
228        let error_id = self.production.allocate(true, lexeme);
229        self.production.set_message(error_id, error_message);
230        self.production.add_before(error_id, before.0);
231        self.process_done(marker, kind, None, Some(before));
232    }
233
234    pub(crate) fn mark_collapsed(&mut self, marker: Marker) {
235        self.production.marker_mut(marker.0).collapsed = true;
236    }
237
238    pub(crate) fn set_binders(
239        &mut self,
240        marker: Marker,
241        left: Option<super::EdgeBinder>,
242        right: Option<super::EdgeBinder>,
243    ) {
244        let data = self.production.marker_mut(marker.0);
245        if left.is_some() {
246            data.left_binder = left;
247        }
248        if right.is_some() {
249            data.right_binder = right;
250        }
251    }
252}
253
254/// `IElementType.isLeftBound()`: true for `TokenType.ERROR_ELEMENT` and `KtLeftBoundNodeType`s.
255fn is_left_bound(kind: SyntaxKind) -> bool {
256    matches!(
257        kind,
258        SyntaxKind::ERROR_ELEMENT | SyntaxKind::CONSTRUCTOR_DELEGATION_CALL | SyntaxKind::CONSTRUCTOR_DELEGATION_REFERENCE
259    )
260}