Skip to main content

codehelion_frontend_c/
lexer.rs

1//! Error-tolerant lexer for the C language family.
2//!
3//! Whitespace and comments are dropped. Preprocessor directives are dropped
4//! whole (through their `\` line continuations): Fast mode does not
5//! preprocess, so both sides of an `#if` stay in the stream as ordinary
6//! tokens while the directive lines themselves never pollute clone content.
7//! Every other lexeme becomes a token carrying its preprocessor-normalized
8//! text and a reporting-only source span. Malformed spans (unterminated strings,
9//! characters and block comments) are recorded as diagnostics and lexing
10//! resumes, so a single broken construct never discards the rest of the file.
11//! Macros are not expanded: an invocation's name and delimiters are ordinary
12//! tokens.
13//!
14//! The lexer is parameterized by a [`Dialect`], which supplies the keyword
15//! set, the operator inventory and the dialect-only literal forms (raw
16//! strings, digit separators), so the same machinery lexes both C and C++.
17
18use codehelion_core::conditional::{ArmPath, ArmTracker, StaticCondition};
19use codehelion_core::frontend::{
20    Diagnostic, DiagnosticKind, LexemeInterner, LiteralKind, SourceSpan, Token, TokenKind,
21};
22
23use crate::dialect::Dialect;
24
25/// Raw-string prefixes, longest first (C++ only; gated by the dialect).
26const RAW_STRING_PREFIXES: &[&str] = &["u8R", "LR", "uR", "UR", "R"];
27
28/// Encoding prefixes of ordinary string and character literals, longest first.
29const TEXT_PREFIXES: &[&str] = &["u8", "L", "u", "U"];
30
31/// C/C++ digraph spellings and their single-token meanings.
32const DIGRAPHS: &[(&str, &str)] = &[
33    ("%:%:", "##"),
34    ("<:", "["),
35    (":>", "]"),
36    ("<%", "{"),
37    ("%>", "}"),
38    ("%:", "#"),
39];
40
41/// C/C++ trigraph spellings other than `??/`, which may form a line splice.
42const TRIGRAPHS: &[(&str, &str)] = &[
43    ("??=", "#"),
44    ("??(", "["),
45    ("??)", "]"),
46    ("??<", "{"),
47    ("??>", "}"),
48    ("??!", "|"),
49    ("??'", "^"),
50    ("??-", "~"),
51];
52
53fn is_ident_start(c: char) -> bool {
54    c.is_alphabetic() || c == '_'
55}
56
57fn is_ident_continue(c: char) -> bool {
58    c.is_alphanumeric() || c == '_'
59}
60
61#[cfg(test)]
62thread_local! {
63    /// Lexer runs performed on the calling thread. Each test owns its thread,
64    /// so this counts only the work that test asked for.
65    static LEXES_RUN: std::cell::Cell<usize> = const { std::cell::Cell::new(0) };
66}
67
68/// Number of lexer runs performed on the calling thread so far.
69#[cfg(test)]
70fn lexes_run() -> usize {
71    LEXES_RUN.with(std::cell::Cell::get)
72}
73
74struct Lexer<'d, 's> {
75    dialect: &'d Dialect,
76    source: &'s str,
77    chars: Vec<char>,
78    byte_at: Vec<usize>,
79    i: usize,
80    line: u32,
81    column: u32,
82    /// Whether a token has been emitted on the current line; a `#` may only
83    /// start a preprocessor directive when nothing but whitespace and
84    /// comments precede it on its line.
85    line_has_token: bool,
86    interner: LexemeInterner,
87    tokens: Vec<Token>,
88    diagnostics: Vec<Diagnostic>,
89    conditional_directives: Vec<(usize, ConditionalDirective)>,
90}
91
92/// A position captured at the start of a token.
93#[derive(Clone, Copy)]
94struct Mark {
95    index: usize,
96    line: u32,
97    column: u32,
98}
99
100impl<'d, 's> Lexer<'d, 's> {
101    fn new(source: &'s str, dialect: &'d Dialect) -> Self {
102        let chars: Vec<char> = source.chars().collect();
103        let mut byte_at = Vec::with_capacity(chars.len() + 1);
104        let mut byte = 0;
105        for c in &chars {
106            byte_at.push(byte);
107            byte += c.len_utf8();
108        }
109        byte_at.push(source.len());
110        Self {
111            dialect,
112            source,
113            chars,
114            byte_at,
115            i: 0,
116            line: 1,
117            column: 1,
118            line_has_token: false,
119            interner: LexemeInterner::new(),
120            tokens: Vec::new(),
121            diagnostics: Vec::new(),
122            conditional_directives: Vec::new(),
123        }
124    }
125
126    /// The source text between a mark and the current position.
127    fn text_from(&self, start: Mark) -> &'s str {
128        &self.source[self.byte_at[start.index]..self.byte_at[self.i]]
129    }
130
131    fn peek(&self, ahead: usize) -> Option<char> {
132        self.chars.get(self.i + ahead).copied()
133    }
134
135    const fn mark(&self) -> Mark {
136        Mark {
137            index: self.i,
138            line: self.line,
139            column: self.column,
140        }
141    }
142
143    /// Consume the current character, tracking line and column.
144    fn bump(&mut self) {
145        if let Some(c) = self.chars.get(self.i) {
146            if *c == '\n' {
147                self.line += 1;
148                self.column = 1;
149            } else {
150                self.column += 1;
151            }
152            self.i += 1;
153        }
154    }
155
156    fn span_from(&self, start: Mark) -> SourceSpan {
157        SourceSpan {
158            start_byte: self.byte_at[start.index],
159            end_byte: self.byte_at[self.i],
160            start_line: start.line,
161            start_column: start.column,
162        }
163    }
164
165    fn push(&mut self, kind: TokenKind, start: Mark) {
166        let text = self.interner.intern(self.text_from(start));
167        self.tokens.push(Token {
168            kind,
169            text,
170            span: self.span_from(start),
171        });
172    }
173
174    /// Emit a token with preprocessor-normalized spelling but source span.
175    fn push_normalized(&mut self, kind: TokenKind, start: Mark, text: &str) {
176        let text = self.interner.intern(text);
177        self.tokens.push(Token {
178            kind,
179            text,
180            span: self.span_from(start),
181        });
182    }
183
184    fn diagnose(&mut self, kind: DiagnosticKind, start: Mark) {
185        let span = self.span_from(start);
186        self.diagnostics.push(Diagnostic { kind, span });
187    }
188
189    /// Whether a `\` at the current position splices the line; if so consume
190    /// it together with its line break.
191    fn try_line_splice(&mut self) -> bool {
192        let width = self.splice_width_at(0);
193        if width == 0 {
194            return false;
195        }
196        for _ in 0..width {
197            self.bump();
198        }
199        true
200    }
201
202    /// Width, in characters, of the line continuations starting `ahead`
203    /// characters past the current position; zero when there is none.
204    ///
205    /// Line splicing is translation phase 2 and tokenisation is phase 3, so a
206    /// continuation is transparent *inside* a token. Every lookahead that
207    /// decides a token's extent therefore measures across it rather than
208    /// stopping at the backslash. Consecutive continuations count as one run.
209    fn splice_width_at(&self, ahead: usize) -> usize {
210        let mut width = 0;
211        loop {
212            let at = ahead + width;
213            let marker = if self.peek(at) == Some('\\') {
214                1
215            } else if self.matches_ahead_at(at, "??/") {
216                3
217            } else {
218                return width;
219            };
220            match (self.peek(at + marker), self.peek(at + marker + 1)) {
221                (Some('\n'), _) => width += marker + 1,
222                (Some('\r'), Some('\n')) => width += marker + 2,
223                _ => return width,
224            }
225        }
226    }
227
228    /// The character `ahead` positions past the current one, seeing through a
229    /// line continuation that precedes it.
230    fn spliced_char_at(&self, ahead: usize) -> Option<char> {
231        self.peek(ahead + self.splice_width_at(ahead))
232    }
233
234    /// Match `text` against the upcoming characters, seeing through line
235    /// continuations, and return how many raw characters it spans.
236    fn spliced_match(&self, text: &str) -> Option<usize> {
237        let mut ahead = 0usize;
238        for expected in text.chars() {
239            ahead += self.splice_width_at(ahead);
240            if self.peek(ahead) != Some(expected) {
241                return None;
242            }
243            ahead += 1;
244        }
245        Some(ahead)
246    }
247
248    fn run(self) -> (Vec<Token>, Vec<Diagnostic>) {
249        let (tokens, diagnostics, _) = self.run_with_directives();
250        (tokens, diagnostics)
251    }
252
253    fn run_with_directives(
254        mut self,
255    ) -> (
256        Vec<Token>,
257        Vec<Diagnostic>,
258        Vec<(usize, ConditionalDirective)>,
259    ) {
260        #[cfg(test)]
261        LEXES_RUN.with(|count| count.set(count.get() + 1));
262        while let Some(c) = self.peek(0) {
263            // A UTF-8 BOM is an encoding marker, not source text. Consume it
264            // only at byte zero. Keep its byte width in spans without letting
265            // it shift the source column visible to users.
266            if self.i == 0 && c == '\u{feff}' {
267                self.i += 1;
268                continue;
269            }
270            if c.is_whitespace() {
271                if c == '\n' {
272                    self.line_has_token = false;
273                }
274                self.bump();
275                continue;
276            }
277            // The comment introducers are matched across line continuations:
278            // `/\<newline>/` opens a line comment, exactly as it would after
279            // the continuations were removed.
280            if let Some(width) = self.spliced_match("//") {
281                self.consume_line_comment(width);
282                continue;
283            }
284            if let Some(width) = self.spliced_match("/*") {
285                self.consume_block_comment(width);
286                continue;
287            }
288            if self.is_directive_marker() && !self.line_has_token {
289                self.consume_directive();
290                continue;
291            }
292            if self.try_line_splice() {
293                continue;
294            }
295            self.line_has_token = true;
296            if self.try_prefixed_literal() {
297                continue;
298            }
299            if c == '"' {
300                let start = self.mark();
301                self.consume_string_from(start);
302                continue;
303            }
304            if c == '\'' {
305                let start = self.mark();
306                self.consume_char_from(start);
307                continue;
308            }
309            if c.is_ascii_digit()
310                || (c == '.' && self.spliced_char_at(1).is_some_and(|d| d.is_ascii_digit()))
311            {
312                self.consume_number();
313                continue;
314            }
315            if is_ident_start(c) {
316                self.consume_ident();
317                continue;
318            }
319            self.consume_punct();
320        }
321        // Streams are long-lived (the whole scan holds every file's tokens),
322        // so growth slack is returned to the allocator.
323        self.tokens.shrink_to_fit();
324        self.conditional_directives.shrink_to_fit();
325        (self.tokens, self.diagnostics, self.conditional_directives)
326    }
327
328    /// Consume a `// ...` comment whose introducer spans `open_width`
329    /// characters (more than two when a line continuation splits it).
330    fn consume_line_comment(&mut self, open_width: usize) {
331        for _ in 0..open_width {
332            self.bump();
333        }
334        while let Some(c) = self.peek(0) {
335            if self.try_line_splice() {
336                // A spliced line continues the comment.
337                continue;
338            }
339            if c == '\n' {
340                break;
341            }
342            self.bump();
343        }
344    }
345
346    /// Consume a `/* ... */` comment whose introducer spans `open_width`
347    /// characters. C-family block comments do not nest.
348    fn consume_block_comment(&mut self, open_width: usize) {
349        let start = self.mark();
350        for _ in 0..open_width {
351            self.bump();
352        }
353        loop {
354            if let Some(width) = self.spliced_match("*/") {
355                for _ in 0..width {
356                    self.bump();
357                }
358                break;
359            }
360            if self.peek(0).is_none() {
361                self.diagnose(DiagnosticKind::UnterminatedBlockComment, start);
362                break;
363            }
364            self.bump();
365        }
366        // A comment that crossed onto a new line leaves that line still
367        // "empty": a `#` after it is that line's first token and may start a
368        // directive.
369        if self.line > start.line {
370            self.line_has_token = false;
371        }
372    }
373
374    /// Consume a preprocessor directive from its `#` through the end of the
375    /// logical line, honouring `\` line continuations and embedded comments.
376    fn consume_directive(&mut self) {
377        let start = self.mark();
378        while let Some(c) = self.peek(0) {
379            if self.try_line_splice() {
380                continue;
381            }
382            if c == '\\' {
383                self.bump();
384                continue;
385            }
386            if c == '\n' {
387                // Leave the newline for the main loop, which resets the
388                // line state.
389                break;
390            }
391            if let Some(width) = self.spliced_match("/*") {
392                self.consume_block_comment(width);
393                continue;
394            }
395            if let Some(width) = self.spliced_match("//") {
396                self.consume_line_comment(width);
397                break;
398            }
399            self.bump();
400        }
401        if let Some(directive) = directive(self.text_from(start)) {
402            self.conditional_directives
403                .push((self.byte_at[start.index], directive));
404        }
405    }
406
407    fn matches_ahead(&self, text: &str) -> bool {
408        self.matches_ahead_at(0, text)
409    }
410
411    fn matches_ahead_at(&self, ahead: usize, text: &str) -> bool {
412        text.chars()
413            .enumerate()
414            .all(|(k, ch)| self.peek(ahead + k) == Some(ch))
415    }
416
417    /// Handle encoding-prefixed and raw string/character literals (`L"..."`,
418    /// `u8'...'`, `R"(...)"`, ...). Returns `true` if a token was produced.
419    fn try_prefixed_literal(&mut self) -> bool {
420        if self.dialect.raw_strings {
421            for prefix in RAW_STRING_PREFIXES {
422                if self.matches_ahead(prefix) && self.peek(prefix.len()) == Some('"') {
423                    let start = self.mark();
424                    for _ in 0..prefix.len() {
425                        self.bump();
426                    }
427                    self.consume_raw_string_body(start);
428                    return true;
429                }
430            }
431        }
432        for prefix in TEXT_PREFIXES {
433            if !self.matches_ahead(prefix) {
434                continue;
435            }
436            match self.peek(prefix.len()) {
437                Some('"') => {
438                    let start = self.mark();
439                    for _ in 0..prefix.len() {
440                        self.bump();
441                    }
442                    self.consume_string_from(start);
443                    return true;
444                }
445                Some('\'') => {
446                    let start = self.mark();
447                    for _ in 0..prefix.len() {
448                        self.bump();
449                    }
450                    self.consume_char_from(start);
451                    return true;
452                }
453                _ => {}
454            }
455        }
456        false
457    }
458
459    /// Consume a string literal; the current character is the opening `"` and
460    /// `start` marks the beginning of the whole literal (including any
461    /// encoding prefix). An unescaped line break ends the literal with a
462    /// diagnostic, so a missing quote never swallows the rest of the file.
463    fn consume_string_from(&mut self, start: Mark) {
464        self.consume_quoted(
465            start,
466            '"',
467            LiteralKind::String,
468            DiagnosticKind::UnterminatedString,
469        );
470    }
471
472    /// Consume a character literal (multi-character constants included); the
473    /// current character is the opening `'`.
474    fn consume_char_from(&mut self, start: Mark) {
475        self.consume_quoted(
476            start,
477            '\'',
478            LiteralKind::Char,
479            DiagnosticKind::UnterminatedChar,
480        );
481    }
482
483    /// Consume a quoted literal whose opening `quote` is the current
484    /// character. Line continuations are transparent, so the token carries the
485    /// spliced spelling, and an escape applies to the character that follows
486    /// the continuations rather than to the backslash of one.
487    fn consume_quoted(
488        &mut self,
489        start: Mark,
490        quote: char,
491        kind: LiteralKind,
492        unterminated: DiagnosticKind,
493    ) {
494        let mut normalized = String::from(self.text_from(start));
495        self.bump();
496        normalized.push(quote);
497        loop {
498            if self.try_line_splice() {
499                continue;
500            }
501            match self.peek(0) {
502                None | Some('\n') => {
503                    self.push_normalized(TokenKind::Literal(kind), start, &normalized);
504                    self.diagnose(unterminated, start);
505                    return;
506                }
507                Some('\\') => {
508                    normalized.push('\\');
509                    self.bump();
510                    while self.try_line_splice() {}
511                    match self.peek(0) {
512                        None | Some('\n') => {}
513                        Some(escaped) => {
514                            normalized.push(escaped);
515                            self.bump();
516                        }
517                    }
518                }
519                Some(c) if c == quote => {
520                    normalized.push(c);
521                    self.bump();
522                    self.push_normalized(TokenKind::Literal(kind), start, &normalized);
523                    return;
524                }
525                Some(c) => {
526                    normalized.push(c);
527                    self.bump();
528                }
529            }
530        }
531    }
532
533    /// Consume a raw string body; the current character is the opening `"`
534    /// and `start` marks the whole literal including its `R` prefix.
535    fn consume_raw_string_body(&mut self, start: Mark) {
536        // Opening `"`.
537        self.bump();
538        let mut delim: Vec<char> = Vec::new();
539        loop {
540            match self.peek(0) {
541                Some('(') => {
542                    self.bump();
543                    break;
544                }
545                Some(c)
546                    if c != '"'
547                        && c != ')'
548                        && c != '\\'
549                        && !c.is_whitespace()
550                        && delim.len() < 16 =>
551                {
552                    delim.push(c);
553                    self.bump();
554                }
555                _ => {
556                    // Malformed delimiter: emit what we have as a broken string.
557                    self.push(TokenKind::Literal(LiteralKind::String), start);
558                    self.diagnose(DiagnosticKind::UnterminatedString, start);
559                    return;
560                }
561            }
562        }
563        loop {
564            match self.peek(0) {
565                None => {
566                    self.push(TokenKind::Literal(LiteralKind::String), start);
567                    self.diagnose(DiagnosticKind::UnterminatedString, start);
568                    return;
569                }
570                Some(')') => {
571                    let closes = delim
572                        .iter()
573                        .enumerate()
574                        .all(|(k, &dc)| self.peek(1 + k) == Some(dc))
575                        && self.peek(1 + delim.len()) == Some('"');
576                    if closes {
577                        for _ in 0..(delim.len() + 2) {
578                            self.bump();
579                        }
580                        self.push(TokenKind::Literal(LiteralKind::String), start);
581                        return;
582                    }
583                    self.bump();
584                }
585                Some(_) => self.bump(),
586            }
587        }
588    }
589
590    /// Consume a numeric literal. Like an identifier, a number is one token
591    /// across line continuations, so the spelling it carries is the spliced
592    /// one rather than the raw source slice.
593    fn consume_number(&mut self) {
594        let start = self.mark();
595        let hex = self.spliced_match("0x").is_some() || self.spliced_match("0X").is_some();
596        let mut is_float = false;
597        // A user-defined suffix starts at `_` and is not numeric: an `e` or a
598        // `.` in it says nothing about the value.
599        let mut in_suffix = false;
600        let mut normalized = String::new();
601        loop {
602            if self.try_line_splice() {
603                continue;
604            }
605            let Some(ch) = self.peek(0) else {
606                break;
607            };
608            if ch == '_' {
609                in_suffix = true;
610            }
611            if in_suffix {
612                if !is_ident_continue(ch) {
613                    break;
614                }
615                normalized.push(ch);
616                self.bump();
617            } else if ch == '.' {
618                // Do not swallow a `...` (GNU case ranges like `1 ... 5`).
619                if self.spliced_char_at(1) == Some('.') {
620                    break;
621                }
622                is_float = true;
623                normalized.push(ch);
624                self.bump();
625            } else if (hex && matches!(ch, 'p' | 'P')) || (!hex && matches!(ch, 'e' | 'E')) {
626                // Hexadecimal floats use a `p` exponent, decimal ones `e`.
627                is_float = true;
628                normalized.push(ch);
629                self.bump();
630                while self.try_line_splice() {}
631                if let Some(sign @ ('+' | '-')) = self.peek(0) {
632                    normalized.push(sign);
633                    self.bump();
634                }
635            } else if ch == '\''
636                && self.dialect.digit_separators
637                && self
638                    .spliced_char_at(1)
639                    .is_some_and(|c| c.is_ascii_alphanumeric())
640            {
641                // Digit separator, kept in the token text.
642                normalized.push(ch);
643                self.bump();
644            } else if is_ident_continue(ch) {
645                normalized.push(ch);
646                self.bump();
647            } else {
648                break;
649            }
650        }
651        let kind = if is_float {
652            LiteralKind::Float
653        } else {
654            LiteralKind::Integer
655        };
656        self.push_normalized(TokenKind::Literal(kind), start, &normalized);
657    }
658
659    fn consume_ident(&mut self) {
660        let start = self.mark();
661        let mut normalized = String::new();
662        loop {
663            if self.try_line_splice() {
664                continue;
665            }
666            let Some(ch) = self.peek(0) else {
667                break;
668            };
669            if !is_ident_continue(ch) {
670                break;
671            }
672            normalized.push(ch);
673            self.bump();
674        }
675        let kind = if self.dialect.keywords.contains(&normalized.as_str()) {
676            TokenKind::Keyword
677        } else {
678            TokenKind::Identifier
679        };
680        self.push_normalized(kind, start, &normalized);
681    }
682
683    /// Consume a punctuator. Multi-character operators, digraphs and trigraphs
684    /// are matched across line continuations and carry their canonical
685    /// spelling, so `+\<newline>=` is the one `+=` token it becomes in
686    /// translation phase 2.
687    fn consume_punct(&mut self) {
688        let start = self.mark();
689        for &(spelling, normalized) in DIGRAPHS.iter().chain(TRIGRAPHS) {
690            // C++ gives `<::` a dedicated maximal-munch exception: unless the
691            // fourth character is `:` or `>`, the `<` is its own token and
692            // the following `::` is scope resolution, not the `<:` digraph
693            // followed by `:`. The C dialect has no `::` operator and keeps
694            // the ordinary digraph rule.
695            if spelling == "<:"
696                && self.dialect.multi_punct.contains(&"::")
697                && let Some(width) = self.spliced_match("<::")
698                && !matches!(self.spliced_char_at(width), Some(':' | '>'))
699            {
700                continue;
701            }
702            if let Some(width) = self.spliced_match(spelling) {
703                for _ in 0..width {
704                    self.bump();
705                }
706                self.push_normalized(TokenKind::Punctuation, start, normalized);
707                return;
708            }
709        }
710        for op in self.dialect.multi_punct {
711            if let Some(width) = self.spliced_match(op) {
712                for _ in 0..width {
713                    self.bump();
714                }
715                self.push_normalized(TokenKind::Punctuation, start, op);
716                return;
717            }
718        }
719        // Single character. ASCII punctuation is punctuation; anything else
720        // that reached here is not lexable.
721        let c = self.peek(0).unwrap_or('\0');
722        self.bump();
723        if c.is_ascii() && !c.is_alphanumeric() {
724            self.push(TokenKind::Punctuation, start);
725        } else {
726            self.push(TokenKind::Unknown, start);
727            self.diagnose(DiagnosticKind::UnexpectedCharacter, start);
728        }
729    }
730
731    fn is_directive_marker(&self) -> bool {
732        self.peek(0) == Some('#') || self.matches_ahead("%:") || self.matches_ahead("??=")
733    }
734}
735
736/// Lex `source` under `dialect` into tokens and diagnostics.
737#[must_use]
738pub fn lex(source: &str, dialect: &Dialect) -> (Vec<Token>, Vec<Diagnostic>) {
739    Lexer::new(source, dialect).run()
740}
741
742/// Lex `source` under `dialect`, keeping the conditional directives it passed.
743///
744/// The directives are a by-product of the same scan that produces the tokens,
745/// so a caller that needs both takes them from here rather than lexing the
746/// text a second time.
747#[must_use]
748pub fn lex_with_directives(
749    source: &str,
750    dialect: &Dialect,
751) -> (Vec<Token>, Vec<Diagnostic>, ConditionalDirectives) {
752    let (tokens, diagnostics, directives) = Lexer::new(source, dialect).run_with_directives();
753    (tokens, diagnostics, ConditionalDirectives(directives))
754}
755
756/// The conditional-compilation directives one lex passed, in source order.
757///
758/// Produced by [`lex_with_directives`] and consumed by [`conditional_paths`];
759/// the arm paths of a file are derived from the lex that already ran, never
760/// from a fresh one over the same text.
761#[derive(Debug, Clone, Default)]
762pub struct ConditionalDirectives(Vec<(usize, ConditionalDirective)>);
763
764/// Record the preprocessor arm active at each token.
765///
766/// Directives never become clone content, but their nesting determines whether
767/// two otherwise-equal C-family snippets can coexist in one build. This pass
768/// intentionally recognises only literal `0` / `1` conditions; macro and
769/// expression evaluation belongs to a compiler frontend, not Fast mode.
770///
771/// `tokens` and `directives` must come from the same lex, which is what makes
772/// the byte offsets of the two comparable.
773#[must_use]
774pub fn conditional_paths(tokens: &[Token], directives: &ConditionalDirectives) -> Vec<ArmPath> {
775    let directives = &directives.0;
776    if !directives_are_balanced(directives) {
777        return vec![ArmPath::default(); tokens.len()];
778    }
779    let mut next_directive = 0usize;
780    let mut tracker = ArmTracker::default();
781    let mut paths = Vec::with_capacity(tokens.len());
782    for token in tokens {
783        while directives
784            .get(next_directive)
785            .is_some_and(|(offset, _)| *offset < token.span.start_byte)
786        {
787            apply_directive(&mut tracker, directives[next_directive].1);
788            next_directive += 1;
789        }
790        paths.push(tracker.current());
791    }
792    paths
793}
794
795/// Refuse to infer exclusions from an unterminated or otherwise unbalanced
796/// directive sequence. Missing a suppression is safer than hiding a clone.
797fn directives_are_balanced(directives: &[(usize, ConditionalDirective)]) -> bool {
798    let mut depth = 0usize;
799    for (_, directive) in directives {
800        match directive {
801            ConditionalDirective::Begin(_) => depth = depth.saturating_add(1),
802            ConditionalDirective::Next(_) | ConditionalDirective::End if depth == 0 => {
803                return false;
804            }
805            ConditionalDirective::Next(_) => {}
806            ConditionalDirective::End => depth -= 1,
807        }
808    }
809    depth == 0
810}
811
812/// One directive that changes the active conditional arm.
813#[derive(Debug, Clone, Copy)]
814enum ConditionalDirective {
815    /// Enter an `#if`-style condition.
816    Begin(StaticCondition),
817    /// Enter an `#elif` or `#else` arm.
818    Next(StaticCondition),
819    /// Leave an `#endif`.
820    End,
821}
822
823/// Classify one lexically recognised directive line.
824fn directive(line: &str) -> Option<ConditionalDirective> {
825    let line = line.trim_start_matches([' ', '\t', '\r']);
826    let line = line
827        .strip_prefix('#')
828        .or_else(|| line.strip_prefix("%:"))
829        .or_else(|| line.strip_prefix("??="))?
830        .trim_start();
831    let word_end = line
832        .find(|ch: char| !ch.is_ascii_alphabetic())
833        .unwrap_or(line.len());
834    let (word, tail) = line.split_at(word_end);
835    let condition = static_condition(tail);
836    match word {
837        "if" => Some(ConditionalDirective::Begin(condition)),
838        "ifdef" | "ifndef" => Some(ConditionalDirective::Begin(StaticCondition::Unknown)),
839        "elif" => Some(ConditionalDirective::Next(condition)),
840        "elifdef" | "elifndef" | "else" => {
841            Some(ConditionalDirective::Next(StaticCondition::Unknown))
842        }
843        "endif" => Some(ConditionalDirective::End),
844        _ => None,
845    }
846}
847
848/// Recognise only a whole literal condition; anything richer stays unknown.
849///
850/// A trailing comment is not part of the condition, in either spelling: `#if 0
851/// // off` and `#if 0 /* off */` say the same thing and must classify the same
852/// way.
853fn static_condition(tail: &str) -> StaticCondition {
854    let tail = tail.split_once("//").map_or(tail, |(before, _)| before);
855    let mut condition = String::new();
856    let mut rest = tail;
857    while let Some((before, after)) = rest.split_once("/*") {
858        condition.push_str(before);
859        // The comment stood between two lexemes; keep them apart.
860        condition.push(' ');
861        let Some((_, remainder)) = after.split_once("*/") else {
862            // An unterminated comment leaves nothing readable behind it.
863            rest = "";
864            break;
865        };
866        rest = remainder;
867    }
868    condition.push_str(rest);
869    match condition.trim() {
870        "0" => StaticCondition::False,
871        "1" => StaticCondition::True,
872        _ => StaticCondition::Unknown,
873    }
874}
875
876/// Apply a recognised directive to the current lexical arm.
877fn apply_directive(tracker: &mut ArmTracker, directive: ConditionalDirective) {
878    match directive {
879        ConditionalDirective::Begin(condition) => tracker.begin(condition),
880        ConditionalDirective::Next(condition) => tracker.next_arm(condition),
881        ConditionalDirective::End => tracker.end(),
882    }
883}
884
885#[cfg(test)]
886#[allow(clippy::expect_used, clippy::unwrap_used, clippy::panic)]
887mod tests {
888    use super::*;
889    use crate::dialect;
890
891    fn lex_c(source: &str) -> (Vec<Token>, Vec<Diagnostic>) {
892        lex(source, &dialect::C)
893    }
894
895    fn texts(source: &str) -> Vec<String> {
896        lex_c(source).0.iter().map(|t| t.text.to_string()).collect()
897    }
898
899    #[test]
900    fn splits_keywords_identifiers_and_operators() {
901        let (tokens, diags) = lex_c("static int add(int a, struct pair *p) { return a + p->x; }");
902        assert!(diags.is_empty(), "diags: {diags:?}");
903        let pairs: Vec<_> = tokens.iter().map(|t| (t.kind, t.text.as_str())).collect();
904        assert_eq!(pairs[0], (TokenKind::Keyword, "static"));
905        assert_eq!(pairs[1], (TokenKind::Keyword, "int"));
906        assert_eq!(pairs[2], (TokenKind::Identifier, "add"));
907        assert!(pairs.contains(&(TokenKind::Punctuation, "->")));
908        assert!(pairs.contains(&(TokenKind::Keyword, "struct")));
909    }
910
911    #[test]
912    fn drops_comments_and_whitespace() {
913        let src = "int x; // trailing\n/* block\nspanning lines */ int y;";
914        let texts = texts(src);
915        assert!(!texts.iter().any(|t| t.contains("trailing")));
916        assert!(!texts.iter().any(|t| t.contains("spanning")));
917        assert!(texts.contains(&"x".to_string()));
918        assert!(texts.contains(&"y".to_string()));
919    }
920
921    #[test]
922    fn preprocessor_directives_are_dropped_whole() {
923        let src = "#include <stdio.h>\n#define TWICE(x) \\\n    ((x) + (x))\nint y;\n";
924        let (tokens, diags) = lex_c(src);
925        assert!(diags.is_empty(), "diags: {diags:?}");
926        let texts: Vec<_> = tokens.iter().map(|t| t.text.as_str()).collect();
927        assert_eq!(texts, vec!["int", "y", ";"]);
928    }
929
930    #[test]
931    fn directive_after_a_multiline_comment_is_still_a_directive() {
932        let src = "int x; /* comment\nspanning */ #define GONE 1\nint y;";
933        let texts = texts(src);
934        assert_eq!(texts, vec!["int", "x", ";", "int", "y", ";"]);
935    }
936
937    #[test]
938    fn a_hash_after_code_on_the_same_line_is_ordinary_punctuation() {
939        // Not a directive: `#` is not the first token of its line.
940        let (tokens, _) = lex_c("int a; # 1\nint b;");
941        assert!(
942            tokens
943                .iter()
944                .any(|t| t.kind == TokenKind::Punctuation && t.text == "#")
945        );
946    }
947
948    #[test]
949    fn conditional_compilation_keeps_both_branches() {
950        let src = "#if FLAG\nint a;\n#else\nint b;\n#endif\n";
951        let texts = texts(src);
952        assert_eq!(texts, vec!["int", "a", ";", "int", "b", ";"]);
953    }
954
955    #[test]
956    fn conditional_paths_separate_alternative_arms_and_literal_dead_code() {
957        let src = "#ifdef _WIN32\nint windows_value;\n#else\nint unix_value;\n#endif\n#if 0\nint dead_value;\n#else\nint live_value;\n#endif\n";
958        let (tokens, diagnostics, directives) = lex_with_directives(src, &dialect::C);
959        assert!(diagnostics.is_empty(), "diagnostics: {diagnostics:?}");
960        let paths = conditional_paths(&tokens, &directives);
961        let path_for = |name: &str| {
962            let index = tokens
963                .iter()
964                .position(|token| token.text == name)
965                .unwrap_or_else(|| panic!("missing {name}"));
966            &paths[index]
967        };
968
969        assert!(path_for("windows_value").excludes(path_for("unix_value")));
970        assert!(path_for("dead_value").is_unreachable());
971        assert!(!path_for("live_value").is_unreachable());
972    }
973
974    #[test]
975    fn unclosed_conditionals_do_not_invent_an_exclusion() {
976        let src = "#ifdef MAYBE\nint first_value;\n#else\nint second_value;\n";
977        let (tokens, diagnostics, directives) = lex_with_directives(src, &dialect::C);
978        assert!(diagnostics.is_empty(), "diagnostics: {diagnostics:?}");
979        let paths = conditional_paths(&tokens, &directives);
980        let path_for = |name: &str| {
981            let index = tokens
982                .iter()
983                .position(|token| token.text == name)
984                .unwrap_or_else(|| panic!("missing {name}"));
985            &paths[index]
986        };
987
988        assert!(
989            !path_for("first_value").excludes(path_for("second_value")),
990            "malformed directives must not hide a Fast finding"
991        );
992    }
993
994    #[test]
995    fn comment_pseudo_directives_do_not_make_code_unreachable() {
996        let src = "// #if 0\nint still_live;\n// #endif\n";
997        let (tokens, diagnostics, directives) = lex_with_directives(src, &dialect::C);
998        assert!(diagnostics.is_empty(), "diagnostics: {diagnostics:?}");
999        let paths = conditional_paths(&tokens, &directives);
1000        let index = tokens
1001            .iter()
1002            .position(|token| token.text == "still_live")
1003            .unwrap_or_else(|| panic!("missing still_live"));
1004        assert!(!paths[index].is_unreachable());
1005    }
1006
1007    #[test]
1008    fn a_fast_pass_over_one_source_lexes_it_once() {
1009        let src = "#if 0\nint dead_value;\n#else\nint live_value;\n#endif\n";
1010        let before = lexes_run();
1011        let (tokens, diagnostics, directives) = lex_with_directives(src, &dialect::C);
1012        let paths = conditional_paths(&tokens, &directives);
1013        assert_eq!(
1014            lexes_run() - before,
1015            1,
1016            "arm paths come from the lex that already ran, not from a second one"
1017        );
1018
1019        assert!(diagnostics.is_empty(), "diagnostics: {diagnostics:?}");
1020        assert_eq!(paths.len(), tokens.len());
1021        let path_for = |name: &str| {
1022            let index = tokens
1023                .iter()
1024                .position(|token| token.text == name)
1025                .unwrap_or_else(|| panic!("missing {name}"));
1026            &paths[index]
1027        };
1028        assert!(path_for("dead_value").is_unreachable());
1029        assert!(!path_for("live_value").is_unreachable());
1030    }
1031
1032    #[test]
1033    fn strings_and_chars_lex_with_escapes_and_prefixes() {
1034        let (tokens, diags) = lex_c("char *s = \"a \\\"q\\\" b\"; char c = 'x'; int m = 'ab';");
1035        assert!(diags.is_empty(), "diags: {diags:?}");
1036        let strings: Vec<_> = tokens
1037            .iter()
1038            .filter(|t| t.kind == TokenKind::Literal(LiteralKind::String))
1039            .map(|t| t.text.as_str())
1040            .collect();
1041        assert_eq!(strings, vec!["\"a \\\"q\\\" b\""]);
1042        let chars: Vec<_> = tokens
1043            .iter()
1044            .filter(|t| t.kind == TokenKind::Literal(LiteralKind::Char))
1045            .map(|t| t.text.as_str())
1046            .collect();
1047        assert_eq!(chars, vec!["'x'", "'ab'"]);
1048
1049        let (tokens, diags) = lex_c("const wchar_t *w = L\"wide\"; int u = u8\"n\"[0];");
1050        assert!(diags.is_empty(), "diags: {diags:?}");
1051        let strings: Vec<_> = tokens
1052            .iter()
1053            .filter(|t| t.kind == TokenKind::Literal(LiteralKind::String))
1054            .map(|t| t.text.as_str())
1055            .collect();
1056        assert_eq!(strings, vec!["L\"wide\"", "u8\"n\""]);
1057    }
1058
1059    #[test]
1060    fn unterminated_string_recovers_at_the_line_break() {
1061        let (tokens, diags) = lex_c("char *s = \"open;\nint next;");
1062        assert_eq!(diags.len(), 1);
1063        assert_eq!(diags[0].kind, DiagnosticKind::UnterminatedString);
1064        assert!(
1065            tokens
1066                .iter()
1067                .any(|t| t.kind == TokenKind::Keyword && t.text == "int"),
1068            "lexing must continue on the next line"
1069        );
1070    }
1071
1072    #[test]
1073    fn unterminated_char_and_block_comment_are_diagnosed() {
1074        let (_, diags) = lex_c("char c = 'x\nint y;");
1075        assert!(
1076            diags
1077                .iter()
1078                .any(|d| d.kind == DiagnosticKind::UnterminatedChar)
1079        );
1080        let (_, diags) = lex_c("int x; /* open");
1081        assert!(
1082            diags
1083                .iter()
1084                .any(|d| d.kind == DiagnosticKind::UnterminatedBlockComment)
1085        );
1086    }
1087
1088    #[test]
1089    fn numbers_cover_hex_float_and_suffix_forms() {
1090        let (tokens, diags) = lex_c(
1091            "int a = 0xFF; double b = 1.5e3; double c = 0x1.8p3; long d = 100UL; float e = .5f; float f = 1.f;",
1092        );
1093        assert!(diags.is_empty(), "diags: {diags:?}");
1094        let by_text = |needle: &str| {
1095            tokens
1096                .iter()
1097                .find(|t| t.text == needle)
1098                .unwrap_or_else(|| panic!("token {needle} missing"))
1099                .kind
1100        };
1101        assert_eq!(by_text("0xFF"), TokenKind::Literal(LiteralKind::Integer));
1102        assert_eq!(by_text("1.5e3"), TokenKind::Literal(LiteralKind::Float));
1103        assert_eq!(by_text("0x1.8p3"), TokenKind::Literal(LiteralKind::Float));
1104        assert_eq!(by_text("100UL"), TokenKind::Literal(LiteralKind::Integer));
1105        assert_eq!(by_text(".5f"), TokenKind::Literal(LiteralKind::Float));
1106        assert_eq!(by_text("1.f"), TokenKind::Literal(LiteralKind::Float));
1107    }
1108
1109    #[test]
1110    fn recovers_after_an_unexpected_character() {
1111        let (tokens, diags) = lex_c("int x = \u{20ac}; int next;");
1112        assert_eq!(diags.len(), 1);
1113        assert_eq!(diags[0].kind, DiagnosticKind::UnexpectedCharacter);
1114        assert!(
1115            tokens
1116                .iter()
1117                .any(|t| t.kind == TokenKind::Identifier && t.text == "next")
1118        );
1119    }
1120
1121    #[test]
1122    fn line_splices_join_code_lines() {
1123        // A backslash-newline inside ordinary code is consumed as whitespace.
1124        let texts = texts("int a \\\n= 1;");
1125        assert_eq!(texts, vec!["int", "a", "=", "1", ";"]);
1126    }
1127
1128    #[test]
1129    fn line_splices_inside_identifiers_preserve_one_normalized_token() {
1130        let texts = texts("int spl\\\nit = 1; int mo??/\nre = split;");
1131        assert_eq!(
1132            texts,
1133            vec![
1134                "int", "split", "=", "1", ";", "int", "more", "=", "split", ";"
1135            ]
1136        );
1137    }
1138
1139    #[test]
1140    fn a_line_splice_is_transparent_inside_every_kind_of_token() {
1141        // Line splicing is translation phase 2 and tokenisation is phase 3,
1142        // so a stream lexed with continuations in place must equal the one
1143        // lexed from the same text with them removed.
1144        for (spliced, joined) in [
1145            ("int a = 1; a +\\\n= 1;", "int a = 1; a += 1;"),
1146            ("int a = 12\\\n34;", "int a = 1234;"),
1147            ("double d = 1.5e\\\n3;", "double d = 1.5e3;"),
1148            ("double d = 0x1.8p\\\n3;", "double d = 0x1.8p3;"),
1149            ("int a = 0\\\nx1F;", "int a = 0x1F;"),
1150            (
1151                "int a = 1; /\\\n/ comment\nint b;",
1152                "int a = 1; // comment\nint b;",
1153            ),
1154            (
1155                "int a = 1; /\\\n* comment *\\\n/ int b;",
1156                "int a = 1; /* comment */ int b;",
1157            ),
1158            ("int a = 1 <\\\n< 2;", "int a = 1 << 2;"),
1159            ("int values ?\\\n?(2:> = 0;", "int values <:2:> = 0;"),
1160        ] {
1161            let observed: Vec<_> = lex_c(spliced)
1162                .0
1163                .iter()
1164                .map(|token| (token.kind, token.text.to_string()))
1165                .collect();
1166            let expected: Vec<_> = lex_c(joined)
1167                .0
1168                .iter()
1169                .map(|token| (token.kind, token.text.to_string()))
1170                .collect();
1171            assert_eq!(observed, expected, "{spliced:?}");
1172        }
1173    }
1174
1175    #[test]
1176    fn a_line_continuation_inside_a_quoted_literal_is_transparent() {
1177        for (spliced, joined) in [
1178            ("char *s = \"abc\\\ndef\";", "char *s = \"abcdef\";"),
1179            ("char *s = \"abc\\\r\ndef\";", "char *s = \"abcdef\";"),
1180            ("char *s = \"abc??/\ndef\";", "char *s = \"abcdef\";"),
1181            ("char *s = \"abc\\\\\ndef\";", "char *s = \"abc\\def\";"),
1182            ("char *s = \"a\\\\\\\ndef\";", "char *s = \"a\\\\def\";"),
1183            ("char *s = \"a\\\n\\\nb\";", "char *s = \"ab\";"),
1184            ("char *s = \"a\\\\\\\"\";", "char *s = \"a\\\\\\\"\";"),
1185            ("int c = '\\\n\\n';", "int c = '\\n';"),
1186            ("char *s = L\"w\\\nx\";", "char *s = L\"wx\";"),
1187        ] {
1188            let (observed_tokens, observed_diags) = lex_c(spliced);
1189            let (expected_tokens, expected_diags) = lex_c(joined);
1190            let observed: Vec<_> = observed_tokens
1191                .iter()
1192                .map(|token| (token.kind, token.text.to_string()))
1193                .collect();
1194            let expected: Vec<_> = expected_tokens
1195                .iter()
1196                .map(|token| (token.kind, token.text.to_string()))
1197                .collect();
1198            assert_eq!(observed, expected, "{spliced:?}");
1199            assert_eq!(observed_diags.len(), expected_diags.len(), "{spliced:?}");
1200            assert!(observed_diags.is_empty(), "{spliced:?}");
1201        }
1202    }
1203
1204    #[test]
1205    fn a_user_defined_suffix_does_not_decide_the_numeric_kind() {
1206        let kind_of = |source: &str| lex_c(source).0[0].kind;
1207        for source in ["5_sec", "7_min", "3_e", "0x1F_x", "10_f", "2_pm"] {
1208            assert_eq!(
1209                kind_of(source),
1210                TokenKind::Literal(LiteralKind::Integer),
1211                "{source}"
1212            );
1213        }
1214        for source in ["1.5_deg", "1e3_x", "2.5e-3_e", "0x1p3_x"] {
1215            assert_eq!(
1216                kind_of(source),
1217                TokenKind::Literal(LiteralKind::Float),
1218                "{source}"
1219            );
1220        }
1221    }
1222
1223    #[test]
1224    fn a_condition_classifies_the_same_whichever_comment_syntax_trails_it() {
1225        for source in [
1226            "#if 0\nint dead_value;\n#endif\nint live_value;\n",
1227            "#if 0 // off\nint dead_value;\n#endif\nint live_value;\n",
1228            "#if 0 /* off */\nint dead_value;\n#endif\nint live_value;\n",
1229            "#if /* why */ 0\nint dead_value;\n#endif\nint live_value;\n",
1230        ] {
1231            let (tokens, _, directives) = lex_with_directives(source, &dialect::C);
1232            let paths = conditional_paths(&tokens, &directives);
1233            let path_for = |name: &str| {
1234                let index = tokens
1235                    .iter()
1236                    .position(|token| token.text == name)
1237                    .unwrap_or_else(|| panic!("missing {name} in {source:?}"));
1238                &paths[index]
1239            };
1240            assert!(path_for("dead_value").is_unreachable(), "{source:?}");
1241            assert!(!path_for("live_value").is_unreachable(), "{source:?}");
1242        }
1243
1244        // A condition the pass cannot read stays unknown, comment or not.
1245        for source in [
1246            "#if FLAG\nint maybe_value;\n#endif\n",
1247            "#if FLAG /* x */\nint maybe_value;\n#endif\n",
1248        ] {
1249            let (tokens, _, directives) = lex_with_directives(source, &dialect::C);
1250            let paths = conditional_paths(&tokens, &directives);
1251            let index = tokens
1252                .iter()
1253                .position(|token| token.text == "maybe_value")
1254                .unwrap_or_else(|| panic!("missing maybe_value"));
1255            assert!(!paths[index].is_unreachable(), "{source:?}");
1256        }
1257    }
1258
1259    #[test]
1260    fn digraphs_and_trigraphs_use_their_canonical_punctuation() {
1261        let texts = texts("%:define COUNT 2\nint values<:COUNT:> = <% 1, 2 %>; int flag ??= 1;");
1262        assert_eq!(
1263            texts,
1264            vec![
1265                "int", "values", "[", "COUNT", "]", "=", "{", "1", ",", "2", "}", ";", "int",
1266                "flag", "#", "1", ";"
1267            ]
1268        );
1269    }
1270
1271    #[test]
1272    fn raw_strings_do_not_exist_in_c() {
1273        // `R"(x)"` in C is the identifier `R` followed by an ordinary string.
1274        let (tokens, diags) = lex_c("R\"(x)\"");
1275        assert!(diags.is_empty(), "diags: {diags:?}");
1276        assert_eq!(tokens[0].kind, TokenKind::Identifier);
1277        assert_eq!(tokens[0].text, "R");
1278        assert_eq!(tokens[1].kind, TokenKind::Literal(LiteralKind::String));
1279    }
1280
1281    #[test]
1282    fn spans_are_byte_accurate() {
1283        let (tokens, _) = lex_c("int x;");
1284        let x = tokens.iter().find(|t| t.text == "x").expect("x token");
1285        assert_eq!(x.span.start_byte, 4);
1286        assert_eq!(x.span.end_byte, 5);
1287        assert_eq!(x.span.start_line, 1);
1288    }
1289
1290    #[test]
1291    fn skips_a_leading_utf8_bom_without_shifting_source_columns() {
1292        let (tokens, diagnostics) = lex_c("\u{feff}int value;");
1293        assert!(diagnostics.is_empty(), "diagnostics: {diagnostics:?}");
1294
1295        let keyword = &tokens[0];
1296        assert_eq!(keyword.kind, TokenKind::Keyword);
1297        assert_eq!(keyword.text, "int");
1298        assert_eq!(keyword.span.start_byte, 3);
1299        assert_eq!(keyword.span.start_line, 1);
1300        assert_eq!(keyword.span.start_column, 1);
1301    }
1302}