Skip to main content

moxy_token/lex/
cursor.rs

1use crate::Span;
2use crate::span::fallback;
3
4use super::LexError;
5
6/// Zero-copy immutable cursor over source text.
7/// Each parse step returns a new advanced cursor.
8/// A copyable cursor over source text during lexing.
9#[derive(Copy, Clone)]
10pub struct Cursor<'a> {
11    rest: &'a str,
12    offset: u32,
13}
14
15impl<'a> Cursor<'a> {
16    pub fn new(src: &'a str, offset: u32) -> Self {
17        Self { rest: src, offset }
18    }
19
20    pub fn rest(&self) -> &'a str {
21        self.rest
22    }
23
24    pub fn offset(&self) -> u32 {
25        self.offset
26    }
27
28    pub fn is_empty(&self) -> bool {
29        self.rest.is_empty()
30    }
31
32    pub fn first(&self) -> Option<char> {
33        self.rest.chars().next()
34    }
35
36    pub fn starts_with(&self, s: &str) -> bool {
37        self.rest.starts_with(s)
38    }
39
40    pub fn span(&self) -> Span {
41        fallback::Span::new(self.offset, self.offset + 1).into()
42    }
43
44    /// Create a fallback::Span from this cursor to another.
45    pub fn span_to(&self, end: &Cursor<'_>) -> Span {
46        fallback::Span::new(self.offset, end.offset).into()
47    }
48
49    /// Get a slice of text from the current cursor to the provided end.
50    pub fn slice_to(&self, end: Cursor<'_>) -> &'a str {
51        let len = (end.offset - self.offset) as usize;
52        &self.rest[..len]
53    }
54
55    /// Create an error at the current span
56    pub fn error(&self) -> LexError {
57        LexError::new(self.span())
58    }
59
60    /// Advance by 1 byte, counting characters for the offset.
61    pub fn advance(&self) -> Self {
62        self.advance_by(1)
63    }
64
65    /// Advance by `n` bytes, counting characters for the offset.
66    pub fn advance_by(&self, n: usize) -> Self {
67        Self {
68            rest: &self.rest[n..],
69            offset: self.offset + n as u32,
70        }
71    }
72
73    /// Advance while predicate holds on chars.
74    pub fn skip_while(&self, mut pred: impl FnMut(char) -> bool) -> Self {
75        let mut bytes = 0;
76
77        for ch in self.rest.chars() {
78            if !pred(ch) {
79                break;
80            }
81
82            bytes += ch.len_utf8();
83        }
84
85        self.advance_by(bytes)
86    }
87
88    pub fn skip_whitespace(mut self) -> Self {
89        loop {
90            // Whitespace
91            let next = self.skip_while(|ch| ch.is_whitespace());
92
93            if next.offset() != self.offset() {
94                self = next;
95                continue;
96            }
97
98            // Line comment — skip plain `//` and `////+`, but NOT doc `///`/`//!`.
99            if self.starts_with("//") && !self.is_line_doc() {
100                self = self.skip_while(|ch| ch != '\n');
101
102                if self.starts_with("\n") {
103                    self = self.advance();
104                }
105
106                continue;
107            }
108
109            // Block comment (nested) — skip plain `/*`, but NOT doc `/**`/`/*!`.
110            if self.starts_with("/*") && !self.is_block_doc() {
111                match self.skip_comment() {
112                    None => break, // unterminated — let the main parser deal with it
113                    Some(next) => {
114                        self = next;
115                        continue;
116                    }
117                }
118            }
119
120            break;
121        }
122
123        self
124    }
125
126    /// True at a line doc comment: `///...` (but not `////...`) or `//!...`.
127    pub fn is_line_doc(&self) -> bool {
128        (self.starts_with("///") && !self.starts_with("////")) || self.starts_with("//!")
129    }
130
131    /// True at a block doc comment: `/**...` (but not `/***`/`/**/`) or `/*!...`.
132    pub fn is_block_doc(&self) -> bool {
133        self.starts_with("/*!") || (self.starts_with("/**") && !self.starts_with("/***") && !self.starts_with("/**/"))
134    }
135
136    /// If positioned at a doc comment, return `(cursor after it, is_inner, text)`.
137    pub fn doc_comment(&self) -> Option<(Self, bool, String)> {
138        if self.is_line_doc() {
139            let inner = self.starts_with("//!");
140            let body = self.advance_by(3); // skip /// or //!
141            let end = body.skip_while(|ch| ch != '\n');
142            let text: String = body.rest()[..(end.offset() - body.offset()) as usize].to_string();
143            let next = if end.starts_with("\n") { end.advance() } else { end };
144            return Some((next, inner, text.trim().to_string()));
145        }
146
147        if self.is_block_doc() {
148            let inner = self.starts_with("/*!");
149            let body = self.advance_by(3); // skip /** or /*!
150            let close = body.skip_comment_to_close()?;
151            // close is positioned just after `*/`; text is between body and `*/`.
152            let len = (close.offset() - body.offset()) as usize - 2;
153            let text: String = body.rest()[..len].to_string();
154            return Some((close, inner, text.trim().to_string()));
155        }
156
157        None
158    }
159
160    pub fn skip_comment(&self) -> Option<Self> {
161        let mut cur = self.advance_by(2); // skip /*
162        let mut depth = 1u32;
163
164        while !cur.is_empty() {
165            if cur.starts_with("/*") {
166                depth += 1;
167                cur = cur.advance_by(2);
168            } else if cur.starts_with("*/") {
169                depth -= 1;
170                cur = cur.advance_by(2);
171
172                if depth == 0 {
173                    return Some(cur);
174                }
175            } else {
176                let ch = cur.first().unwrap();
177                cur = cur.advance_by(ch.len_utf8());
178            }
179        }
180
181        None
182    }
183
184    /// Skip to just past the matching `*/` of a (non-nested) block comment body.
185    fn skip_comment_to_close(&self) -> Option<Self> {
186        let mut cur = *self;
187
188        while !cur.is_empty() {
189            if cur.starts_with("*/") {
190                return Some(cur.advance_by(2));
191            }
192
193            let ch = cur.first().unwrap();
194            cur = cur.advance_by(ch.len_utf8());
195        }
196
197        None
198    }
199}