Skip to main content

kcl_lib/parsing/token/
mod.rs

1// Clippy does not agree with rustc here for some reason.
2#![allow(clippy::needless_lifetimes)]
3
4use std::env;
5use std::fmt;
6use std::iter::Enumerate;
7use std::num::NonZeroUsize;
8use std::str::FromStr;
9
10use anyhow::Result;
11use kcl_error::KclErrorDetails;
12use parse_display::Display;
13use serde::Deserialize;
14use serde::Serialize;
15use tower_lsp::lsp_types::SemanticTokenType;
16use winnow::stream::ContainsToken;
17use winnow::stream::Stream;
18use winnow::{self};
19
20use crate::CompilationIssue;
21use crate::ModuleId;
22use crate::RuntimeFlag;
23use crate::SourceRange;
24use crate::errors::KclError;
25use crate::kcl_runtime_flags;
26use crate::parsing::ast::types::ItemVisibility;
27use crate::parsing::ast::types::VariableKind;
28use crate::runtime_flags::RuntimeFlagResolve;
29use crate::runtime_flags::resolve_from_sources;
30
31mod tokeniser;
32
33#[doc(hidden)]
34pub mod adapter;
35
36#[cfg(test)]
37mod compat_tests;
38
39#[cfg(test)]
40mod error_matrix_tests;
41
42pub(crate) use tokeniser::RESERVED_SKETCH_BLOCK_WORDS;
43pub use tokeniser::RESERVED_WORDS;
44
45// Note the ordering, it's important that `m` comes after `mm` and `cm`.
46pub const NUM_SUFFIXES: [&str; 10] = ["mm", "cm", "m", "inch", "in", "ft", "yd", "deg", "rad", "?"];
47
48#[derive(Clone, Copy, Debug, Eq, PartialEq, Serialize, Deserialize, ts_rs::TS)]
49#[repr(u32)]
50pub enum NumericSuffix {
51    None,
52    Count,
53    Length,
54    Angle,
55    Mm,
56    Cm,
57    M,
58    Inch,
59    Ft,
60    Yd,
61    Deg,
62    Rad,
63    Unknown,
64}
65
66impl NumericSuffix {
67    #[allow(dead_code)]
68    pub fn is_none(self) -> bool {
69        self == Self::None
70    }
71
72    pub fn is_some(self) -> bool {
73        self != Self::None
74    }
75
76    pub fn digestable_id(&self) -> &[u8] {
77        match self {
78            NumericSuffix::None => &[],
79            NumericSuffix::Count => b"_",
80            NumericSuffix::Unknown => b"?",
81            NumericSuffix::Length => b"Length",
82            NumericSuffix::Angle => b"Angle",
83            NumericSuffix::Mm => b"mm",
84            NumericSuffix::Cm => b"cm",
85            NumericSuffix::M => b"m",
86            NumericSuffix::Inch => b"in",
87            NumericSuffix::Ft => b"ft",
88            NumericSuffix::Yd => b"yd",
89            NumericSuffix::Deg => b"deg",
90            NumericSuffix::Rad => b"rad",
91        }
92    }
93}
94
95impl FromStr for NumericSuffix {
96    type Err = CompilationIssue;
97
98    fn from_str(s: &str) -> Result<Self, Self::Err> {
99        match s {
100            "_" | "Count" => Ok(NumericSuffix::Count),
101            "Length" => Ok(NumericSuffix::Length),
102            "Angle" => Ok(NumericSuffix::Angle),
103            "mm" | "millimeters" => Ok(NumericSuffix::Mm),
104            "cm" | "centimeters" => Ok(NumericSuffix::Cm),
105            "m" | "meters" => Ok(NumericSuffix::M),
106            "inch" | "in" => Ok(NumericSuffix::Inch),
107            "ft" | "feet" => Ok(NumericSuffix::Ft),
108            "yd" | "yards" => Ok(NumericSuffix::Yd),
109            "deg" | "degrees" => Ok(NumericSuffix::Deg),
110            "rad" | "radians" => Ok(NumericSuffix::Rad),
111            "?" => Ok(NumericSuffix::Unknown),
112            _ => Err(CompilationIssue::err(SourceRange::default(), "invalid unit of measure")),
113        }
114    }
115}
116
117impl fmt::Display for NumericSuffix {
118    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
119        match self {
120            NumericSuffix::None => Ok(()),
121            NumericSuffix::Count => write!(f, "_"),
122            NumericSuffix::Unknown => write!(f, "_?"),
123            NumericSuffix::Length => write!(f, "Length"),
124            NumericSuffix::Angle => write!(f, "Angle"),
125            NumericSuffix::Mm => write!(f, "mm"),
126            NumericSuffix::Cm => write!(f, "cm"),
127            NumericSuffix::M => write!(f, "m"),
128            NumericSuffix::Inch => write!(f, "in"),
129            NumericSuffix::Ft => write!(f, "ft"),
130            NumericSuffix::Yd => write!(f, "yd"),
131            NumericSuffix::Deg => write!(f, "deg"),
132            NumericSuffix::Rad => write!(f, "rad"),
133        }
134    }
135}
136
137#[derive(Clone, Debug, PartialEq)]
138pub struct TokenStream {
139    tokens: Vec<Token>,
140}
141
142impl TokenStream {
143    fn new(tokens: Vec<Token>) -> Self {
144        Self { tokens }
145    }
146
147    /// Allow the parser to read `use` as an identifier until the module's KCL
148    /// version is known. Return the original keyword ranges for validation.
149    pub(super) fn allow_use_identifiers(&mut self) -> Vec<SourceRange> {
150        self.tokens
151            .iter_mut()
152            .filter_map(|token| {
153                if token.token_type == TokenType::Keyword && token.value == "use" {
154                    token.token_type = TokenType::Word;
155                    Some(token.as_source_range())
156                } else {
157                    None
158                }
159            })
160            .collect()
161    }
162
163    pub(super) fn remove_unknown(&mut self) -> Vec<Token> {
164        let tokens = std::mem::take(&mut self.tokens);
165        let (tokens, unknown_tokens): (Vec<Token>, Vec<Token>) = tokens
166            .into_iter()
167            .partition(|token| token.token_type != TokenType::Unknown);
168        self.tokens = tokens;
169        unknown_tokens
170    }
171
172    pub fn iter(&self) -> impl Iterator<Item = &Token> {
173        self.tokens.iter()
174    }
175
176    pub fn is_empty(&self) -> bool {
177        self.tokens.is_empty()
178    }
179
180    pub fn as_slice(&self) -> TokenSlice<'_> {
181        TokenSlice::from(self)
182    }
183}
184
185impl<'a> From<&'a TokenStream> for TokenSlice<'a> {
186    fn from(stream: &'a TokenStream) -> Self {
187        TokenSlice {
188            start: 0,
189            end: stream.tokens.len(),
190            stream,
191        }
192    }
193}
194
195impl IntoIterator for TokenStream {
196    type Item = Token;
197
198    type IntoIter = std::vec::IntoIter<Token>;
199
200    fn into_iter(self) -> Self::IntoIter {
201        self.tokens.into_iter()
202    }
203}
204
205#[derive(Debug, Clone)]
206pub struct TokenSlice<'a> {
207    stream: &'a TokenStream,
208    /// Current position of the leading Token in the stream
209    start: usize,
210    /// The number of total Tokens in the stream
211    end: usize,
212}
213
214impl<'a> std::ops::Deref for TokenSlice<'a> {
215    type Target = [Token];
216
217    fn deref(&self) -> &Self::Target {
218        &self.stream.tokens[self.start..self.end]
219    }
220}
221
222impl<'a> TokenSlice<'a> {
223    pub fn token(&self, i: usize) -> &Token {
224        &self.stream.tokens[i + self.start]
225    }
226
227    pub fn iter(&self) -> impl Iterator<Item = &Token> {
228        (**self).iter()
229    }
230
231    pub fn without_ends(&self) -> Self {
232        Self {
233            start: self.start + 1,
234            end: self.end - 1,
235            stream: self.stream,
236        }
237    }
238
239    pub fn as_source_range(&self) -> SourceRange {
240        let stream_len = self.stream.tokens.len();
241        let first_token = if stream_len == self.start {
242            &self.stream.tokens[self.start - 1]
243        } else {
244            self.token(0)
245        };
246        let last_token = if stream_len == self.end {
247            &self.stream.tokens[stream_len - 1]
248        } else {
249            self.token(self.end - self.start)
250        };
251        SourceRange::new(first_token.start, last_token.end, last_token.module_id)
252    }
253}
254
255impl<'a> IntoIterator for TokenSlice<'a> {
256    type Item = &'a Token;
257
258    type IntoIter = std::slice::Iter<'a, Token>;
259
260    fn into_iter(self) -> Self::IntoIter {
261        self.stream.tokens[self.start..self.end].iter()
262    }
263}
264
265impl<'a> Stream for TokenSlice<'a> {
266    type Token = Token;
267    type Slice = Self;
268    type IterOffsets = Enumerate<std::vec::IntoIter<Token>>;
269    type Checkpoint = Checkpoint;
270
271    fn iter_offsets(&self) -> Self::IterOffsets {
272        #[allow(clippy::unnecessary_to_owned)]
273        self.to_vec().into_iter().enumerate()
274    }
275
276    fn eof_offset(&self) -> usize {
277        self.len()
278    }
279
280    fn next_token(&mut self) -> Option<Self::Token> {
281        let token = self.first()?.clone();
282        self.start += 1;
283        Some(token)
284    }
285
286    /// Split off the next token from the input
287    fn peek_token(&self) -> Option<Self::Token> {
288        Some(self.first()?.clone())
289    }
290
291    fn offset_for<P>(&self, predicate: P) -> Option<usize>
292    where
293        P: Fn(Self::Token) -> bool,
294    {
295        self.iter().position(|b| predicate(b.clone()))
296    }
297
298    fn offset_at(&self, tokens: usize) -> Result<usize, winnow::error::Needed> {
299        if let Some(needed) = tokens.checked_sub(self.len()).and_then(NonZeroUsize::new) {
300            Err(winnow::error::Needed::Size(needed))
301        } else {
302            Ok(tokens)
303        }
304    }
305
306    fn next_slice(&mut self, offset: usize) -> Self::Slice {
307        assert!(self.start + offset <= self.end);
308
309        let next = TokenSlice {
310            stream: self.stream,
311            start: self.start,
312            end: self.start + offset,
313        };
314        self.start += offset;
315        next
316    }
317
318    /// Split off a slice of tokens from the input
319    fn peek_slice(&self, offset: usize) -> Self::Slice {
320        assert!(self.start + offset <= self.end);
321
322        TokenSlice {
323            stream: self.stream,
324            start: self.start,
325            end: self.start + offset,
326        }
327    }
328
329    fn checkpoint(&self) -> Self::Checkpoint {
330        Checkpoint(self.start, self.end)
331    }
332
333    fn reset(&mut self, checkpoint: &Self::Checkpoint) {
334        self.start = checkpoint.0;
335        self.end = checkpoint.1;
336    }
337
338    fn trace(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result {
339        write!(f, "{self:?}")
340    }
341}
342
343impl<'a> winnow::stream::Offset for TokenSlice<'a> {
344    fn offset_from(&self, start: &Self) -> usize {
345        self.start - start.start
346    }
347}
348
349impl<'a> winnow::stream::Offset<Checkpoint> for TokenSlice<'a> {
350    fn offset_from(&self, start: &Checkpoint) -> usize {
351        self.start - start.0
352    }
353}
354
355impl winnow::stream::Offset for Checkpoint {
356    fn offset_from(&self, start: &Self) -> usize {
357        self.0 - start.0
358    }
359}
360
361impl<'a> winnow::stream::StreamIsPartial for TokenSlice<'a> {
362    type PartialState = ();
363
364    fn complete(&mut self) -> Self::PartialState {}
365
366    fn restore_partial(&mut self, _: Self::PartialState) {}
367
368    fn is_partial_supported() -> bool {
369        false
370    }
371}
372
373impl<'a> winnow::stream::FindSlice<&str> for TokenSlice<'a> {
374    fn find_slice(&self, substr: &str) -> Option<std::ops::Range<usize>> {
375        self.iter()
376            .enumerate()
377            .find_map(|(i, b)| if b.value == substr { Some(i..self.end) } else { None })
378    }
379}
380
381#[derive(Clone, Debug)]
382pub struct Checkpoint(usize, usize);
383
384/// The types of tokens.
385#[derive(Debug, PartialEq, Eq, Copy, Clone, Display)]
386#[display(style = "camelCase")]
387pub enum TokenType {
388    /// A number.
389    Number,
390    /// A word.
391    Word,
392    /// An operator.
393    Operator,
394    /// A string.
395    String,
396    /// A keyword.
397    Keyword,
398    /// A type.
399    Type,
400    /// A brace.
401    Brace,
402    /// A hash.
403    Hash,
404    /// A bang.
405    Bang,
406    /// A dollar sign.
407    Dollar,
408    /// Whitespace.
409    Whitespace,
410    /// A comma.
411    Comma,
412    /// A colon.
413    Colon,
414    /// A double colon: `::`
415    DoubleColon,
416    /// A period.
417    Period,
418    /// A double period: `..`.
419    DoublePeriod,
420    /// A double period and a less than: `..<`.
421    DoublePeriodLessThan,
422    /// A line comment.
423    LineComment,
424    /// A block comment.
425    BlockComment,
426    /// A function name.
427    Function,
428    /// Unknown lexemes.
429    Unknown,
430    /// The ? symbol, used for optional values.
431    QuestionMark,
432    /// The @ symbol.
433    At,
434    /// `;`
435    SemiColon,
436}
437
438/// Most KCL tokens correspond to LSP semantic tokens (but not all).
439impl TryFrom<TokenType> for SemanticTokenType {
440    type Error = anyhow::Error;
441    fn try_from(token_type: TokenType) -> Result<Self> {
442        // If you return a new kind of `SemanticTokenType`, make sure to update `SEMANTIC_TOKEN_TYPES`
443        // in the LSP implementation.
444        Ok(match token_type {
445            TokenType::Number => Self::NUMBER,
446            TokenType::Word => Self::VARIABLE,
447            TokenType::Keyword => Self::KEYWORD,
448            TokenType::Type => Self::TYPE,
449            TokenType::Operator => Self::OPERATOR,
450            TokenType::QuestionMark => Self::OPERATOR,
451            TokenType::String => Self::STRING,
452            TokenType::Bang => Self::OPERATOR,
453            TokenType::LineComment => Self::COMMENT,
454            TokenType::BlockComment => Self::COMMENT,
455            TokenType::Function => Self::FUNCTION,
456            TokenType::Whitespace
457            | TokenType::Brace
458            | TokenType::Comma
459            | TokenType::Colon
460            | TokenType::DoubleColon
461            | TokenType::Period
462            | TokenType::DoublePeriod
463            | TokenType::DoublePeriodLessThan
464            | TokenType::Hash
465            | TokenType::Dollar
466            | TokenType::At
467            | TokenType::SemiColon
468            | TokenType::Unknown => {
469                anyhow::bail!("unsupported token type: {:?}", token_type)
470            }
471        })
472    }
473}
474
475impl TokenType {
476    pub fn is_whitespace(&self) -> bool {
477        matches!(self, Self::Whitespace)
478    }
479
480    pub fn is_comment(&self) -> bool {
481        matches!(self, Self::LineComment | Self::BlockComment)
482    }
483}
484
485#[derive(Debug, PartialEq, Eq, Clone)]
486pub struct Token {
487    pub token_type: TokenType,
488    /// Offset in the source code where this token begins.
489    pub start: usize,
490    /// Offset in the source code where this token ends.
491    pub end: usize,
492    pub(super) module_id: ModuleId,
493    pub(super) value: String,
494}
495
496impl ContainsToken<Token> for (TokenType, &str) {
497    fn contains_token(&self, token: Token) -> bool {
498        self.0 == token.token_type && self.1 == token.value
499    }
500}
501
502impl ContainsToken<Token> for TokenType {
503    fn contains_token(&self, token: Token) -> bool {
504        *self == token.token_type
505    }
506}
507
508impl Token {
509    pub fn from_range(
510        range: std::ops::Range<usize>,
511        module_id: ModuleId,
512        token_type: TokenType,
513        value: String,
514    ) -> Self {
515        Self {
516            start: range.start,
517            end: range.end,
518            module_id,
519            value,
520            token_type,
521        }
522    }
523    pub fn is_code_token(&self) -> bool {
524        !matches!(
525            self.token_type,
526            TokenType::Whitespace | TokenType::LineComment | TokenType::BlockComment
527        )
528    }
529
530    pub fn as_source_range(&self) -> SourceRange {
531        SourceRange::new(self.start, self.end, self.module_id)
532    }
533
534    pub fn as_source_ranges(&self) -> Vec<SourceRange> {
535        vec![self.as_source_range()]
536    }
537
538    pub fn visibility_keyword(&self) -> Option<ItemVisibility> {
539        if !matches!(self.token_type, TokenType::Keyword) {
540            return None;
541        }
542        match self.value.as_str() {
543            "export" => Some(ItemVisibility::Export),
544            _ => None,
545        }
546    }
547
548    pub fn numeric_value(&self) -> Option<f64> {
549        if self.token_type != TokenType::Number {
550            return None;
551        }
552        let value = &self.value;
553        let value = value
554            .split_once(|c: char| c == '_' || c.is_ascii_alphabetic())
555            .map(|(s, _)| s)
556            .unwrap_or(value);
557        value.parse().ok()
558    }
559
560    pub fn uint_value(&self) -> Option<u32> {
561        if self.token_type != TokenType::Number {
562            return None;
563        }
564        let value = &self.value;
565        let value = value
566            .split_once(|c: char| c == '_' || c.is_ascii_alphabetic())
567            .map(|(s, _)| s)
568            .unwrap_or(value);
569        value.parse().ok()
570    }
571
572    pub fn numeric_suffix(&self) -> NumericSuffix {
573        if self.token_type != TokenType::Number {
574            return NumericSuffix::None;
575        }
576
577        if self.value.ends_with('_') {
578            return NumericSuffix::Count;
579        }
580
581        for suffix in NUM_SUFFIXES {
582            if self.value.ends_with(suffix) {
583                return suffix.parse().unwrap();
584            }
585        }
586
587        NumericSuffix::None
588    }
589
590    /// Is this token the beginning of a variable/function declaration?
591    /// If so, what kind?
592    /// If not, returns None.
593    pub fn declaration_keyword(&self) -> Option<VariableKind> {
594        if !matches!(self.token_type, TokenType::Keyword) {
595            return None;
596        }
597        Some(match self.value.as_str() {
598            "fn" => VariableKind::Fn,
599            "var" | "let" | "const" => VariableKind::Const,
600            _ => return None,
601        })
602    }
603}
604
605impl From<Token> for SourceRange {
606    fn from(token: Token) -> Self {
607        Self::new(token.start, token.end, token.module_id)
608    }
609}
610
611impl From<&Token> for SourceRange {
612    fn from(token: &Token) -> Self {
613        Self::new(token.start, token.end, token.module_id)
614    }
615}
616
617/// Environment variable selecting which lexer implementation [`lex`] uses.
618pub(crate) const KCL_LEXER_ENV_VAR: &str = "KCL_LEXER";
619
620/// Which lexer implementation [`lex`] uses: the old winnow `tokeniser` (`Old`) or
621/// the new `kcl-syntax` logos lexer (`New`). Selected at runtime via the
622/// `KCL_LEXER` environment variable, so a process can pick either lexer without a
623/// rebuild.
624///
625/// Precedence: runtime flags > test override > `KCL_LEXER` >
626/// [`LexerMode::DEFAULT`].
627#[derive(Debug, Clone, Copy, PartialEq, Eq)]
628pub enum LexerMode {
629    Old,
630    New,
631}
632
633impl RuntimeFlagResolve for LexerMode {
634    fn on() -> Self {
635        Self::New
636    }
637
638    fn off() -> Self {
639        Self::Old
640    }
641
642    fn resolve_default() -> Self {
643        Self::DEFAULT
644    }
645
646    fn parse_env_var(value: &str) -> Self {
647        Self::parse(value)
648    }
649}
650
651impl LexerMode {
652    /// The mode used when `KCL_LEXER` is unset.
653    const DEFAULT: Self = Self::New;
654
655    /// Resolve the active lexer mode (see precedence on [`LexerMode`]).
656    pub fn resolve() -> Self {
657        let env_value = match env::var(KCL_LEXER_ENV_VAR) {
658            Ok(value) => Some(value),
659            Err(env::VarError::NotPresent) => None,
660            Err(env::VarError::NotUnicode(value)) => {
661                // Invalid-unicode env var: warn and fall back rather than crash.
662                Self::warn_once(|| {
663                    format!(
664                        "{KCL_LEXER_ENV_VAR} must be valid unicode; got `{}`. Defaulting to `new`.",
665                        value.to_string_lossy()
666                    )
667                });
668                None
669            }
670        };
671
672        Self::resolve_from_sources(
673            kcl_runtime_flags().use_new_lexer_parser,
674            Self::test_override_for_resolve(),
675            env_value.as_deref(),
676        )
677    }
678
679    fn resolve_from_sources(runtime_flag: RuntimeFlag, test_override: Option<Self>, env_value: Option<&str>) -> Self {
680        resolve_from_sources(runtime_flag, test_override, env_value)
681    }
682
683    #[cfg(any(test, feature = "lsp-test-util"))]
684    fn test_override_for_resolve() -> Option<Self> {
685        Self::test_override()
686    }
687
688    #[cfg(not(any(test, feature = "lsp-test-util")))]
689    fn test_override_for_resolve() -> Option<Self> {
690        None
691    }
692
693    fn parse(value: &str) -> Self {
694        let value = value.trim();
695        if value.eq_ignore_ascii_case("old") {
696            return Self::Old;
697        }
698        if value.eq_ignore_ascii_case("new") {
699            return Self::New;
700        }
701
702        // A mistyped `KCL_LEXER` should not crash the process: warn and fall back
703        // to the new lexer (the conservative choice for a misconfiguration).
704        Self::warn_once(|| {
705            format!("Unsupported {KCL_LEXER_ENV_VAR} value `{value}`; expected `old` or `new`. Defaulting to `new`.")
706        });
707        Self::New
708    }
709
710    /// Emit a one-time configuration warning through `crate::log` (gated on
711    /// `ZOO_LOG`). `resolve`/`parse` run on every `lex`, so a misconfigured
712    /// `KCL_LEXER` must not warn -- or allocate the message -- on every call. One
713    /// guard suffices: only one kind of misconfiguration can occur per process,
714    /// since the env var holds a single value.
715    fn warn_once(make_message: impl FnOnce() -> String) {
716        static WARNED: std::sync::Once = std::sync::Once::new();
717        WARNED.call_once(|| crate::log::log(make_message()));
718    }
719
720    #[cfg(any(test, feature = "lsp-test-util"))]
721    fn test_override_value(self) -> u8 {
722        match self {
723            Self::Old => 1,
724            Self::New => 2,
725        }
726    }
727
728    #[cfg(any(test, feature = "lsp-test-util"))]
729    fn test_override() -> Option<Self> {
730        match TEST_LEXER_MODE_OVERRIDE.load(std::sync::atomic::Ordering::SeqCst) {
731            1 => Some(Self::Old),
732            2 => Some(Self::New),
733            _ => None,
734        }
735    }
736
737    /// Override the lexer mode for the lifetime of the returned guard.
738    ///
739    /// This uses a process-global atomic, so it is only race-free under test
740    /// runners that isolate tests in separate processes (e.g. `cargo nextest`).
741    /// Under in-process parallel `cargo test`, prefer driving the lexer with an
742    /// explicit mode; reserve this guard for dispatch/integration tests.
743    #[cfg(any(test, feature = "lsp-test-util"))]
744    pub fn override_for_test(mode: Self) -> LexerModeOverrideGuard {
745        let previous = TEST_LEXER_MODE_OVERRIDE.swap(mode.test_override_value(), std::sync::atomic::Ordering::SeqCst);
746        LexerModeOverrideGuard { previous }
747    }
748}
749
750#[cfg(any(test, feature = "lsp-test-util"))]
751static TEST_LEXER_MODE_OVERRIDE: std::sync::atomic::AtomicU8 = std::sync::atomic::AtomicU8::new(0);
752
753#[cfg(any(test, feature = "lsp-test-util"))]
754pub struct LexerModeOverrideGuard {
755    previous: u8,
756}
757
758#[cfg(any(test, feature = "lsp-test-util"))]
759impl Drop for LexerModeOverrideGuard {
760    fn drop(&mut self) {
761        TEST_LEXER_MODE_OVERRIDE.store(self.previous, std::sync::atomic::Ordering::SeqCst);
762    }
763}
764
765// `lex` dispatches on the runtime `LexerMode`. `Old` runs the winnow
766// `tokeniser`; `New` runs the `kcl-syntax` adapter and folds any fatal lexical
767// diagnostics into a single lexical `KclError`, preserving the public `Result`
768// contract. (The LSP consumes the richer `LexResult` directly so it can keep
769// tokens for highlighting while reporting diagnostics.)
770pub fn lex(s: &str, module_id: ModuleId) -> Result<TokenStream, KclError> {
771    match LexerMode::resolve() {
772        LexerMode::Old => lex_legacy(s, module_id),
773        LexerMode::New => {
774            let result = adapter::lex_with_diagnostics(s, module_id);
775            match result.to_lexical_error() {
776                Some(err) => Err(err),
777                None => Ok(result.tokens),
778            }
779        }
780    }
781}
782
783fn lex_legacy(s: &str, module_id: ModuleId) -> Result<TokenStream, KclError> {
784    tokeniser::lex(s, module_id).map_err(|err| {
785        let (input, offset): (Vec<char>, usize) = (err.input().chars().collect(), err.offset());
786        let module_id = err.input().state.module_id;
787
788        if offset >= input.len() {
789            // From the winnow docs:
790            //
791            // This is an offset, not an index, and may point to
792            // the end of input (input.len()) on eof errors.
793
794            return KclError::new_lexical(KclErrorDetails::new(
795                "unexpected EOF while parsing".to_owned(),
796                vec![SourceRange::new(offset, offset, module_id)],
797            ));
798        }
799
800        // TODO: Add the Winnow tokenizer context to the error.
801        // See https://github.com/KittyCAD/modeling-app/issues/784
802        let bad_token = &input[offset];
803        // TODO: Add the Winnow parser context to the error.
804        // See https://github.com/KittyCAD/modeling-app/issues/784
805        KclError::new_lexical(KclErrorDetails::new(
806            format!("found unknown token '{bad_token}'"),
807            vec![SourceRange::new(offset, offset + 1, module_id)],
808        ))
809    })
810}
811
812#[cfg(test)]
813mod lexer_mode_tests {
814    use super::LexerMode;
815    use super::lex;
816    use crate::KclRuntimeFlags;
817    use crate::ModuleId;
818    use crate::RuntimeFlag;
819
820    fn set_runtime_lexer_flag(flag: RuntimeFlag) {
821        crate::set_kcl_runtime_flags(KclRuntimeFlags {
822            use_new_lexer_parser: flag,
823            ..Default::default()
824        });
825    }
826
827    fn reset_runtime_lexer_flags() {
828        crate::set_kcl_runtime_flags(KclRuntimeFlags::DEFAULT);
829    }
830
831    #[test]
832    fn default_mode_is_new() {
833        reset_runtime_lexer_flags();
834        assert_eq!(LexerMode::DEFAULT, LexerMode::New);
835    }
836
837    #[test]
838    fn parse_accepts_known_values_case_insensitively() {
839        assert_eq!(LexerMode::parse("old"), LexerMode::Old);
840        assert_eq!(LexerMode::parse("  NEW  "), LexerMode::New);
841    }
842
843    #[test]
844    fn parse_falls_back_to_new_on_unknown_value() {
845        // An unknown value warns and defaults to the new lexer instead of panicking.
846        assert_eq!(LexerMode::parse("rowan"), LexerMode::New);
847    }
848
849    #[test]
850    fn override_guard_sets_and_restores_mode() {
851        reset_runtime_lexer_flags();
852        // Reserved for dispatch/integration tests; relies on the process-global
853        // atomic, which is race-free under nextest's process isolation.
854        {
855            let _guard = LexerMode::override_for_test(LexerMode::New);
856            assert_eq!(LexerMode::resolve(), LexerMode::New);
857        }
858        let _guard = LexerMode::override_for_test(LexerMode::Old);
859        assert_eq!(LexerMode::resolve(), LexerMode::Old);
860    }
861
862    #[test]
863    fn runtime_flags_default_to_unset() {
864        reset_runtime_lexer_flags();
865        assert_eq!(
866            crate::kcl_runtime_flags(),
867            KclRuntimeFlags {
868                use_new_lexer_parser: RuntimeFlag::Unset,
869                ..Default::default()
870            }
871        );
872    }
873
874    #[test]
875    fn runtime_flag_on_selects_new_lexer() {
876        reset_runtime_lexer_flags();
877        set_runtime_lexer_flag(RuntimeFlag::On);
878        assert_eq!(LexerMode::resolve(), LexerMode::New);
879    }
880
881    #[test]
882    fn runtime_flag_off_selects_old_lexer() {
883        reset_runtime_lexer_flags();
884        set_runtime_lexer_flag(RuntimeFlag::Off);
885        assert_eq!(LexerMode::resolve(), LexerMode::Old);
886    }
887
888    #[test]
889    fn runtime_flag_takes_priority_over_test_override_and_env() {
890        assert_eq!(
891            LexerMode::resolve_from_sources(RuntimeFlag::Off, Some(LexerMode::New), Some("new")),
892            LexerMode::Old
893        );
894        assert_eq!(
895            LexerMode::resolve_from_sources(RuntimeFlag::On, Some(LexerMode::Old), Some("old")),
896            LexerMode::New
897        );
898    }
899
900    #[test]
901    fn unset_runtime_flag_allows_env_to_select_lexer() {
902        assert_eq!(
903            LexerMode::resolve_from_sources(RuntimeFlag::Unset, None, Some("new")),
904            LexerMode::New
905        );
906        assert_eq!(
907            LexerMode::resolve_from_sources(RuntimeFlag::Unset, None, Some("old")),
908            LexerMode::Old
909        );
910    }
911
912    #[test]
913    fn unset_runtime_flag_and_missing_env_selects_default_lexer() {
914        assert_eq!(
915            LexerMode::resolve_from_sources(RuntimeFlag::Unset, None, None),
916            LexerMode::DEFAULT
917        );
918    }
919
920    #[test]
921    fn test_override_takes_priority_over_env() {
922        assert_eq!(
923            LexerMode::resolve_from_sources(RuntimeFlag::Unset, Some(LexerMode::Old), Some("new")),
924            LexerMode::Old
925        );
926        assert_eq!(
927            LexerMode::resolve_from_sources(RuntimeFlag::Unset, Some(LexerMode::New), Some("old")),
928            LexerMode::New
929        );
930    }
931
932    /// Exercises the `New` arm of `lex` in default CI: no `KCL_LEXER` env var is
933    /// set; the new lexer is selected via the process-global test override (which
934    /// is race-free under nextest's process-per-test isolation).
935    ///
936    /// The unterminated-string assertion is deliberately a *distinguishing* one:
937    /// the new lexer folds the recovery token into the message "unterminated
938    /// string literal", whereas the old lexer reports `found unknown token '"'`.
939    /// Asserting the new-lexer-only message proves `lex` took the `New` arm --
940    /// not merely that some lexer ran.
941    #[test]
942    fn lex_dispatches_to_new_lexer() {
943        reset_runtime_lexer_flags();
944        let _guard = LexerMode::override_for_test(LexerMode::New);
945        assert_eq!(LexerMode::resolve(), LexerMode::New);
946
947        let module_id = ModuleId::default();
948
949        // Valid input flows through the New arm and yields a token stream.
950        let tokens = lex("x = 1", module_id).expect("new lexer should tokenize valid input");
951        assert!(!tokens.is_empty(), "expected a non-empty token stream");
952
953        // Unterminated string: the new-lexer-only message (see doc comment).
954        let err = lex("\"abc", module_id).expect_err("unterminated string is a lexical error");
955        assert_eq!(err.error_type(), "lexical");
956        assert_eq!(err.message(), "unterminated string literal");
957    }
958}