Skip to main content

moxy_token/lex/
cursor.rs

1use super::LexError;
2use crate::Span;
3use crate::span::fallback;
4
5/// Zero-copy immutable cursor over source text.
6/// Each parse step returns a new advanced cursor.
7#[derive(Copy, Clone)]
8pub struct Cursor<'a> {
9    rest: &'a str,
10    off: u32,
11}
12
13impl<'a> Cursor<'a> {
14    pub fn new(src: &'a str, offset: u32) -> Self {
15        Self { rest: src, off: offset }
16    }
17
18    pub fn rest(&self) -> &'a str {
19        self.rest
20    }
21
22    pub fn offset(&self) -> u32 {
23        self.off
24    }
25
26    pub fn is_empty(&self) -> bool {
27        self.rest.is_empty()
28    }
29
30    pub fn first(&self) -> Option<char> {
31        self.rest.chars().next()
32    }
33
34    pub fn starts_with(&self, s: &str) -> bool {
35        self.rest.starts_with(s)
36    }
37
38    /// Create a fallback::Span from this cursor to another.
39    pub fn span_to(&self, end: &Cursor<'_>) -> Span {
40        fallback::Span::new(self.off, end.off).into()
41    }
42
43    pub fn span(&self) -> Span {
44        fallback::Span::new(self.off, self.off + 1).into()
45    }
46
47    pub fn error(&self) -> LexError {
48        LexError::new(self.span())
49    }
50
51    /// Advance by `n` bytes, counting characters for the offset.
52    pub fn advance(&self, n: usize) -> Self {
53        Self {
54            rest: &self.rest[n..],
55            off: self.off + n as u32,
56        }
57    }
58
59    /// Advance while predicate holds on chars.
60    pub fn skip_while(&self, mut pred: impl FnMut(char) -> bool) -> Self {
61        let mut bytes = 0;
62
63        for ch in self.rest.chars() {
64            if !pred(ch) {
65                break;
66            }
67            bytes += ch.len_utf8();
68        }
69
70        self.advance(bytes)
71    }
72
73    pub fn skip_whitespace(mut self) -> Self {
74        loop {
75            // Whitespace
76            let next = self.skip_while(|ch| ch.is_whitespace());
77
78            if next.offset() != self.offset() {
79                self = next;
80                continue;
81            }
82
83            // Line comment — skip plain `//` and `////+`, but NOT doc `///`/`//!`.
84            if self.starts_with("//") && !self.is_line_doc() {
85                self = self.skip_while(|ch| ch != '\n');
86
87                if self.starts_with("\n") {
88                    self = self.advance(1);
89                }
90
91                continue;
92            }
93
94            // Block comment (nested) — skip plain `/*`, but NOT doc `/**`/`/*!`.
95            if self.starts_with("/*") && !self.is_block_doc() {
96                match self.skip_comment() {
97                    None => break, // unterminated — let the main parser deal with it
98                    Some(next) => {
99                        self = next;
100                        continue;
101                    }
102                }
103            }
104
105            break;
106        }
107
108        self
109    }
110
111    /// True at a line doc comment: `///...` (but not `////...`) or `//!...`.
112    pub fn is_line_doc(&self) -> bool {
113        (self.starts_with("///") && !self.starts_with("////")) || self.starts_with("//!")
114    }
115
116    /// True at a block doc comment: `/**...` (but not `/***`/`/**/`) or `/*!...`.
117    pub fn is_block_doc(&self) -> bool {
118        self.starts_with("/*!") || (self.starts_with("/**") && !self.starts_with("/***") && !self.starts_with("/**/"))
119    }
120
121    /// If positioned at a doc comment, return `(cursor after it, is_inner, text)`.
122    pub fn doc_comment(&self) -> Option<(Self, bool, String)> {
123        if self.is_line_doc() {
124            let inner = self.starts_with("//!");
125            let body = self.advance(3); // skip /// or //!
126            let end = body.skip_while(|ch| ch != '\n');
127            let text: String = body.rest()[..(end.offset() - body.offset()) as usize].to_string();
128            let next = if end.starts_with("\n") { end.advance(1) } else { end };
129            return Some((next, inner, text.trim().to_string()));
130        }
131
132        if self.is_block_doc() {
133            let inner = self.starts_with("/*!");
134            let body = self.advance(3); // skip /** or /*!
135            let close = body.skip_comment_to_close()?;
136            // close is positioned just after `*/`; text is between body and `*/`.
137            let len = (close.offset() - body.offset()) as usize - 2;
138            let text: String = body.rest()[..len].to_string();
139            return Some((close, inner, text.trim().to_string()));
140        }
141
142        None
143    }
144
145    /// Skip to just past the matching `*/` of a (non-nested) block comment body.
146    fn skip_comment_to_close(&self) -> Option<Self> {
147        let mut cur = *self;
148
149        while !cur.is_empty() {
150            if cur.starts_with("*/") {
151                return Some(cur.advance(2));
152            }
153            let ch = cur.first().unwrap();
154            cur = cur.advance(ch.len_utf8());
155        }
156
157        None
158    }
159
160    pub fn skip_comment(&self) -> Option<Self> {
161        let mut cur = self.advance(2); // skip /*
162        let mut depth = 1u32;
163
164        while !cur.is_empty() {
165            if cur.starts_with("/*") {
166                depth += 1;
167                cur = cur.advance(2);
168            } else if cur.starts_with("*/") {
169                depth -= 1;
170                cur = cur.advance(2);
171
172                if depth == 0 {
173                    return Some(cur);
174                }
175            } else {
176                let ch = cur.first().unwrap();
177                cur = cur.advance(ch.len_utf8());
178            }
179        }
180
181        None
182    }
183}