Skip to main content

ktrs_parser/builder/
binders.rs

1//! `WhitespacesAndCommentsBinder` implementations: IntelliJ's `WhitespacesBinders` plus Kotlin's
2//! `KotlinWhitespaceAndCommentsBinders.kt`, as one closed enum. None of them is recursive
3//! (`isRecursive()`), so `MarkerProduction.confineMarkersToMaxLexeme` is never needed.
4
5use ktrs_syntax::SyntaxKind::{self, *};
6
7use crate::kt_tokens::COMMENTS;
8
9/// Decides where an element edge lands inside a run of whitespace/comment tokens:
10/// the result is an index into `tokens` (0 = before all of them, `tokens.len()` = after all).
11#[derive(Clone, Copy, Debug, PartialEq, Eq)]
12pub enum EdgeBinder {
13    /// `WhitespacesBinders.DEFAULT_LEFT_BINDER` (= `GREEDY_RIGHT_BINDER`).
14    DefaultLeft,
15    /// `WhitespacesBinders.DEFAULT_RIGHT_BINDER` (= `GREEDY_LEFT_BINDER`).
16    DefaultRight,
17    PrecedingComments,
18    PrecedingDocComments,
19    TrailingComments,
20    /// `PRECEDING_ALL_COMMENTS_BINDER` (`AllCommentsBinder(isTrailing = false)`).
21    PrecedingAllComments,
22    /// `TRAILING_ALL_COMMENTS_BINDER` (`AllCommentsBinder(isTrailing = true)`).
23    TrailingAllComments,
24    DoNotBindAnything,
25    BindFirstShebangWithWhitespaceOnly,
26    /// `PRECEDING_ALL_BINDER` (`BindAll(isTrailing = false)`).
27    PrecedingAll,
28    /// `TRAILING_ALL_BINDER` (`BindAll(isTrailing = true)`).
29    TrailingAll,
30}
31
32pub const GREEDY_LEFT_BINDER: EdgeBinder = EdgeBinder::DefaultRight;
33pub const GREEDY_RIGHT_BINDER: EdgeBinder = EdgeBinder::DefaultLeft;
34
35impl EdgeBinder {
36    pub fn get_edge_position<'t>(
37        self,
38        tokens: &[SyntaxKind],
39        _at_stream_edge: bool,
40        getter: &dyn Fn(usize) -> &'t str,
41    ) -> usize {
42        match self {
43            EdgeBinder::DefaultLeft => tokens.len(),
44            EdgeBinder::DefaultRight => 0,
45            EdgeBinder::PrecedingComments => preceding_comments(tokens, getter),
46            EdgeBinder::PrecedingDocComments => {
47                if tokens.is_empty() {
48                    return 0;
49                }
50                tokens.iter().rposition(|&t| t == DOC_COMMENT).unwrap_or(tokens.len())
51            }
52            EdgeBinder::TrailingComments => trailing_comments(tokens, getter),
53            EdgeBinder::PrecedingAllComments => all_comments(tokens, false),
54            EdgeBinder::TrailingAllComments => all_comments(tokens, true),
55            EdgeBinder::DoNotBindAnything => 0,
56            EdgeBinder::BindFirstShebangWithWhitespaceOnly => {
57                if tokens.first() == Some(&SHEBANG_COMMENT) {
58                    return if tokens.get(1) == Some(&WHITE_SPACE) { 2 } else { 1 };
59                }
60                0
61            }
62            EdgeBinder::PrecedingAll => 0,
63            EdgeBinder::TrailingAll => tokens.len(),
64        }
65    }
66}
67
68fn preceding_comments<'t>(tokens: &[SyntaxKind], getter: &dyn Fn(usize) -> &'t str) -> usize {
69    if tokens.is_empty() {
70        return 0;
71    }
72
73    // 1. bind doc comment
74    if let Some(idx) = tokens.iter().rposition(|&t| t == DOC_COMMENT) {
75        return idx;
76    }
77
78    // 2. bind plain comments
79    let mut result = tokens.len();
80    for idx in (0..tokens.len()).rev() {
81        let token_type = tokens[idx];
82        if token_type == WHITE_SPACE {
83            if get_line_break_count(getter(idx)) > 1 {
84                break;
85            }
86        } else if COMMENTS.contains(token_type) {
87            if idx == 0 || tokens[idx - 1] == WHITE_SPACE && contains_line_break(getter(idx - 1)) {
88                result = idx;
89            }
90        } else {
91            break;
92        }
93    }
94    result
95}
96
97fn trailing_comments<'t>(tokens: &[SyntaxKind], getter: &dyn Fn(usize) -> &'t str) -> usize {
98    if tokens.is_empty() {
99        return 0;
100    }
101
102    let mut result = 0;
103    for (idx, &token_type) in tokens.iter().enumerate() {
104        match token_type {
105            WHITE_SPACE => {
106                if contains_line_break(getter(idx)) {
107                    break;
108                }
109            }
110            EOL_COMMENT | BLOCK_COMMENT => result = idx + 1,
111            _ => break,
112        }
113    }
114    result
115}
116
117fn all_comments(tokens: &[SyntaxKind], is_trailing: bool) -> usize {
118    if tokens.is_empty() {
119        return 0;
120    }
121    let size = tokens.len();
122    // Skip one whitespace if needed. Expect that there can't be several consecutive whitespaces
123    let end_token = tokens[if is_trailing { size - 1 } else { 0 }];
124    let shift = usize::from(end_token == WHITE_SPACE);
125    if is_trailing { size - shift } else { shift }
126}
127
128/// `StringUtil.getLineBreakCount`: `\r\n`, `\r` and `\n` each count once.
129fn get_line_break_count(text: &str) -> usize {
130    let bytes = text.as_bytes();
131    let mut count = 0;
132    let mut i = 0;
133    while i < bytes.len() {
134        match bytes[i] {
135            b'\n' => count += 1,
136            b'\r' => {
137                if bytes.get(i + 1) == Some(&b'\n') {
138                    i += 1;
139                }
140                count += 1;
141            }
142            _ => {}
143        }
144        i += 1;
145    }
146    count
147}
148
149/// `StringUtil.containsLineBreak`.
150fn contains_line_break(text: &str) -> bool {
151    text.bytes().any(|b| b == b'\n' || b == b'\r')
152}