Skip to main content

moxy_token/lex/
cursor.rs

1use crate::Span;
2use crate::span::fallback;
3
4use super::LexError;
5
6/// Zero-copy immutable cursor over source text.
7/// Each parse step returns a new advanced cursor.
8#[derive(Copy, Clone)]
9pub struct Cursor<'a> {
10    rest: &'a str,
11    offset: u32,
12}
13
14impl<'a> Cursor<'a> {
15    pub fn new(src: &'a str, offset: u32) -> Self {
16        Self { rest: src, offset }
17    }
18
19    pub fn rest(&self) -> &'a str {
20        self.rest
21    }
22
23    pub fn offset(&self) -> u32 {
24        self.offset
25    }
26
27    pub fn is_empty(&self) -> bool {
28        self.rest.is_empty()
29    }
30
31    pub fn first(&self) -> Option<char> {
32        self.rest.chars().next()
33    }
34
35    pub fn starts_with(&self, s: &str) -> bool {
36        self.rest.starts_with(s)
37    }
38
39    pub fn span(&self) -> Span {
40        fallback::Span::new(self.offset, self.offset + 1).into()
41    }
42
43    /// Create a fallback::Span from this cursor to another.
44    pub fn span_to(&self, end: &Cursor<'_>) -> Span {
45        fallback::Span::new(self.offset, end.offset).into()
46    }
47
48    /// Get a slice of text from the current cursor to the provided end.
49    pub fn slice_to(&self, end: Cursor<'_>) -> &'a str {
50        let len = (end.offset - self.offset) as usize;
51        &self.rest[..len]
52    }
53
54    /// Create an error at the current span
55    pub fn error(&self) -> LexError {
56        LexError::new(self.span())
57    }
58
59    /// Advance by 1 byte, counting characters for the offset.
60    pub fn advance(&self) -> Self {
61        self.advance_by(1)
62    }
63
64    /// Advance by `n` bytes, counting characters for the offset.
65    pub fn advance_by(&self, n: usize) -> Self {
66        Self {
67            rest: &self.rest[n..],
68            offset: self.offset + n as u32,
69        }
70    }
71
72    /// Advance while predicate holds on chars.
73    pub fn skip_while(&self, mut pred: impl FnMut(char) -> bool) -> Self {
74        let mut bytes = 0;
75
76        for ch in self.rest.chars() {
77            if !pred(ch) {
78                break;
79            }
80
81            bytes += ch.len_utf8();
82        }
83
84        self.advance_by(bytes)
85    }
86
87    pub fn skip_whitespace(mut self) -> Self {
88        loop {
89            // Whitespace
90            let next = self.skip_while(|ch| ch.is_whitespace());
91
92            if next.offset() != self.offset() {
93                self = next;
94                continue;
95            }
96
97            // Line comment — skip plain `//` and `////+`, but NOT doc `///`/`//!`.
98            if self.starts_with("//") && !self.is_line_doc() {
99                self = self.skip_while(|ch| ch != '\n');
100
101                if self.starts_with("\n") {
102                    self = self.advance();
103                }
104
105                continue;
106            }
107
108            // Block comment (nested) — skip plain `/*`, but NOT doc `/**`/`/*!`.
109            if self.starts_with("/*") && !self.is_block_doc() {
110                match self.skip_comment() {
111                    None => break, // unterminated — let the main parser deal with it
112                    Some(next) => {
113                        self = next;
114                        continue;
115                    }
116                }
117            }
118
119            break;
120        }
121
122        self
123    }
124
125    /// True at a line doc comment: `///...` (but not `////...`) or `//!...`.
126    pub fn is_line_doc(&self) -> bool {
127        (self.starts_with("///") && !self.starts_with("////")) || self.starts_with("//!")
128    }
129
130    /// True at a block doc comment: `/**...` (but not `/***`/`/**/`) or `/*!...`.
131    pub fn is_block_doc(&self) -> bool {
132        self.starts_with("/*!") || (self.starts_with("/**") && !self.starts_with("/***") && !self.starts_with("/**/"))
133    }
134
135    /// If positioned at a doc comment, return `(cursor after it, is_inner, text)`.
136    pub fn doc_comment(&self) -> Option<(Self, bool, String)> {
137        if self.is_line_doc() {
138            let inner = self.starts_with("//!");
139            let body = self.advance_by(3); // skip /// or //!
140            let end = body.skip_while(|ch| ch != '\n');
141            let text: String = body.rest()[..(end.offset() - body.offset()) as usize].to_string();
142            let next = if end.starts_with("\n") { end.advance() } else { end };
143            return Some((next, inner, text.trim().to_string()));
144        }
145
146        if self.is_block_doc() {
147            let inner = self.starts_with("/*!");
148            let body = self.advance_by(3); // skip /** or /*!
149            let close = body.skip_comment_to_close()?;
150            // close is positioned just after `*/`; text is between body and `*/`.
151            let len = (close.offset() - body.offset()) as usize - 2;
152            let text: String = body.rest()[..len].to_string();
153            return Some((close, inner, text.trim().to_string()));
154        }
155
156        None
157    }
158
159    pub fn skip_comment(&self) -> Option<Self> {
160        let mut cur = self.advance_by(2); // skip /*
161        let mut depth = 1u32;
162
163        while !cur.is_empty() {
164            if cur.starts_with("/*") {
165                depth += 1;
166                cur = cur.advance_by(2);
167            } else if cur.starts_with("*/") {
168                depth -= 1;
169                cur = cur.advance_by(2);
170
171                if depth == 0 {
172                    return Some(cur);
173                }
174            } else {
175                let ch = cur.first().unwrap();
176                cur = cur.advance_by(ch.len_utf8());
177            }
178        }
179
180        None
181    }
182
183    /// Skip to just past the matching `*/` of a (non-nested) block comment body.
184    fn skip_comment_to_close(&self) -> Option<Self> {
185        let mut cur = *self;
186
187        while !cur.is_empty() {
188            if cur.starts_with("*/") {
189                return Some(cur.advance_by(2));
190            }
191
192            let ch = cur.first().unwrap();
193            cur = cur.advance_by(ch.len_utf8());
194        }
195
196        None
197    }
198}