Skip to main content

cinrs_core/
lex.rs

1//! A complete C99 lexer working on the captured source text.
2//!
3//! The lexer is byte-offset based: every [`Token`] carries a
4//! [`SourceRange`] into the [`SourceMap`](crate::capture::SourceMap), which is
5//! what makes exact diagnostics possible. Constants are decoded here (value,
6//! base, suffix, escape sequences) so that neither the parser nor sema has to
7//! look at raw text again.
8//!
9//! Two flags on each token — [`Token::bol`] and [`Token::preceded_by_space`] —
10//! are what the [preprocessor](crate::pp) recognises directives with and what
11//! its stringification operator reproduces: directive recognition needs "first
12//! token on a line", and `#` needs to know where whitespace was.
13//!
14//! # The lexer never reports anything
15//!
16//! Lexical errors never stop the lexer, and they never reach [`Diagnostics`]
17//! either: each one is attached to the token it was found in, in
18//! [`Token::errors`], and the preprocessor reports the ones whose token
19//! survives into its output. That is not a detail — a group skipped by
20//! `#if 0` may legally hold text that is not C at all, and a macro that is
21//! never invoked may hold anything its author liked:
22//!
23//! ```c
24//! #if 0
25//! this is not C: 08, 'unterminated, @@@
26//! #endif
27//! ```
28//!
29//! Text that is not a token at all becomes a [`TokenKind::Error`] token
30//! carrying its own spelling, so that it too can be skipped rather than
31//! reported. The preprocessor drops those tokens after reporting them, so
32//! nothing downstream ever sees one.
33//!
34//! [`Diagnostics`]: crate::diag::Diagnostics
35
36use crate::capture::{Pos, Source, SourceRange};
37use crate::diag::Diagnostic;
38use crate::{Dialect, Options, Standard};
39
40/// A C99 keyword.
41#[derive(Clone, Copy, PartialEq, Eq, Debug, Hash)]
42#[allow(missing_docs)]
43pub enum Keyword {
44    Auto,
45    Break,
46    Case,
47    Char,
48    Const,
49    Continue,
50    Default,
51    Do,
52    Double,
53    Else,
54    Enum,
55    Extern,
56    Float,
57    For,
58    Goto,
59    If,
60    Inline,
61    Int,
62    Long,
63    Register,
64    Restrict,
65    Return,
66    Short,
67    Signed,
68    Sizeof,
69    Static,
70    Struct,
71    Switch,
72    Typedef,
73    Union,
74    Unsigned,
75    Void,
76    Volatile,
77    While,
78    Bool,
79    Complex,
80    Imaginary,
81    // C11. Every one of these is spelled with a leading underscore, which C99
82    // already reserves, so they are recognised in every mode and the parser
83    // reports the ones a C99 block may not use — a friendlier answer than
84    // "expected a declaration, found identifier '_Static_assert'".
85    Alignas,
86    Alignof,
87    Atomic,
88    Generic,
89    Noreturn,
90    StaticAssert,
91    ThreadLocal,
92    BitInt,
93    // C23. These are ordinary identifiers before C23 — `<stdbool.h>` writes
94    // `#define bool _Bool`, and a C99 program may have a variable called
95    // `typeof` — so they are keywords only in a `c23!` block.
96    BoolName,
97    True,
98    False,
99    Nullptr,
100    Typeof,
101    TypeofUnqual,
102    Constexpr,
103    StaticAssertName,
104    AlignofName,
105    AlignasName,
106    ThreadLocalName,
107    // The GNU keywords. Every one of them is spelled with a leading double
108    // underscore, which C reserves, so they are available in every entry point
109    // — exactly as they are in GCC's own strict modes. They are *not* produced
110    // by [`Keyword::from_str`]: the [preprocessor](crate::pp) turns the
111    // identifiers into them on the way out, after macro replacement, so that
112    // `#define __attribute__(x)` — which portability headers really do write —
113    // still defines and expands a macro of that name.
114    /// `__attribute__`, `__attribute`
115    Attribute,
116    /// `__extension__`
117    Extension,
118    /// `__alignof__`, `__alignof`
119    AlignofGnu,
120    /// `__typeof__`, `__typeof`, and `typeof` in a GNU dialect
121    TypeofGnu,
122    /// `__typeof_unqual__`
123    TypeofUnqualGnu,
124    /// `__asm__`, `__asm`, and `asm` in a GNU dialect
125    Asm,
126    /// `__label__`
127    Label,
128    /// `__auto_type`
129    AutoType,
130    /// `__thread`
131    ThreadGnu,
132    /// `__int128`
133    ///
134    /// A type specifier of its own, which `signed` and `unsigned` combine
135    /// with; the `__int128_t` and `__uint128_t` spellings are `typedef` names
136    /// the compiler owns rather than keywords, exactly as they are in GCC.
137    Int128,
138    /// `__real__`, `__real`
139    RealGnu,
140    /// `__imag__`, `__imag`
141    ImagGnu,
142    /// `__inline__`, `__inline`
143    ///
144    /// A variant of its own because the plain spelling is C99's and `c89!`
145    /// gates it, while this one — being reserved — is available everywhere.
146    InlineGnu,
147    /// `__restrict__`, `__restrict`; see [`Keyword::InlineGnu`].
148    RestrictGnu,
149}
150
151impl Keyword {
152    /// The spelling of this keyword in C source.
153    pub fn as_str(self) -> &'static str {
154        use Keyword::*;
155        match self {
156            Auto => "auto",
157            Break => "break",
158            Case => "case",
159            Char => "char",
160            Const => "const",
161            Continue => "continue",
162            Default => "default",
163            Do => "do",
164            Double => "double",
165            Else => "else",
166            Enum => "enum",
167            Extern => "extern",
168            Float => "float",
169            For => "for",
170            Goto => "goto",
171            If => "if",
172            Inline => "inline",
173            Int => "int",
174            Long => "long",
175            Register => "register",
176            Restrict => "restrict",
177            Return => "return",
178            Short => "short",
179            Signed => "signed",
180            Sizeof => "sizeof",
181            Static => "static",
182            Struct => "struct",
183            Switch => "switch",
184            Typedef => "typedef",
185            Union => "union",
186            Unsigned => "unsigned",
187            Void => "void",
188            Volatile => "volatile",
189            While => "while",
190            Bool => "_Bool",
191            Complex => "_Complex",
192            Imaginary => "_Imaginary",
193            Alignas => "_Alignas",
194            Alignof => "_Alignof",
195            Atomic => "_Atomic",
196            Generic => "_Generic",
197            Noreturn => "_Noreturn",
198            StaticAssert => "_Static_assert",
199            ThreadLocal => "_Thread_local",
200            BitInt => "_BitInt",
201            BoolName => "bool",
202            True => "true",
203            False => "false",
204            Nullptr => "nullptr",
205            Typeof => "typeof",
206            TypeofUnqual => "typeof_unqual",
207            Constexpr => "constexpr",
208            StaticAssertName => "static_assert",
209            AlignofName => "alignof",
210            AlignasName => "alignas",
211            ThreadLocalName => "thread_local",
212            Attribute => "__attribute__",
213            Extension => "__extension__",
214            AlignofGnu => "__alignof__",
215            TypeofGnu => "__typeof__",
216            TypeofUnqualGnu => "__typeof_unqual__",
217            Asm => "__asm__",
218            Label => "__label__",
219            AutoType => "__auto_type",
220            ThreadGnu => "__thread",
221            Int128 => "__int128",
222            RealGnu => "__real__",
223            ImagGnu => "__imag__",
224            InlineGnu => "__inline__",
225            RestrictGnu => "__restrict__",
226        }
227    }
228
229    /// Whether this keyword is one of the GNU spellings the preprocessor
230    /// introduces; see the variants' own documentation.
231    pub fn is_gnu(self) -> bool {
232        use Keyword::*;
233        matches!(
234            self,
235            Attribute
236                | Extension
237                | AlignofGnu
238                | TypeofGnu
239                | TypeofUnqualGnu
240                | Asm
241                | Label
242                | AutoType
243                | ThreadGnu
244                | Int128
245                | RealGnu
246                | ImagGnu
247                | InlineGnu
248                | RestrictGnu
249        )
250    }
251
252    /// The revision that made this spelling a keyword.
253    ///
254    /// The C11 keywords are recognised in every mode — they are reserved
255    /// identifiers in C99, so nothing legal can be broken by it, and the
256    /// parser's "requires C11 or later" is a better answer than a syntax
257    /// error. The C23 ones are *not*: they are ordinary identifiers before
258    /// C23, and `<stdbool.h>`'s `#define bool _Bool` depends on it.
259    ///
260    /// The four C99 added — `inline`, `restrict`, `_Bool` and `_Complex` (with
261    /// `_Imaginary` beside it) — are recognised in every mode for the same
262    /// reason the C11 ones are, and gated where they are parsed; everything
263    /// else has been a keyword since C89.
264    pub fn since(self) -> Standard {
265        use Keyword::*;
266        match self {
267            Alignas | Alignof | Atomic | Generic | Noreturn | StaticAssert | ThreadLocal => {
268                Standard::C11
269            }
270            BitInt | BoolName | True | False | Nullptr | Typeof | TypeofUnqual | Constexpr
271            | StaticAssertName | AlignofName | AlignasName | ThreadLocalName => Standard::C23,
272            Inline | Restrict | Bool | Complex | Imaginary => Standard::C99,
273            _ => Standard::C89,
274        }
275    }
276
277    /// Looks a keyword up by spelling, honouring the language standard.
278    pub fn from_str(s: &str, standard: Standard) -> Option<Keyword> {
279        use Keyword::*;
280        if standard >= Standard::C23 {
281            let c23 = match s {
282                "bool" => Some(BoolName),
283                "true" => Some(True),
284                "false" => Some(False),
285                "nullptr" => Some(Nullptr),
286                "typeof" => Some(Typeof),
287                "typeof_unqual" => Some(TypeofUnqual),
288                "constexpr" => Some(Constexpr),
289                "static_assert" => Some(StaticAssertName),
290                "alignof" => Some(AlignofName),
291                "alignas" => Some(AlignasName),
292                "thread_local" => Some(ThreadLocalName),
293                _ => None,
294            };
295            if c23.is_some() {
296                return c23;
297            }
298        }
299        Some(match s {
300            "_Alignas" => Alignas,
301            "_Alignof" => Alignof,
302            "_Atomic" => Atomic,
303            "_Generic" => Generic,
304            "_Noreturn" => Noreturn,
305            "_Static_assert" => StaticAssert,
306            "_Thread_local" => ThreadLocal,
307            "_BitInt" => BitInt,
308            "auto" => Auto,
309            "break" => Break,
310            "case" => Case,
311            "char" => Char,
312            "const" => Const,
313            "continue" => Continue,
314            "default" => Default,
315            "do" => Do,
316            "double" => Double,
317            "else" => Else,
318            "enum" => Enum,
319            "extern" => Extern,
320            "float" => Float,
321            "for" => For,
322            "goto" => Goto,
323            "if" => If,
324            "inline" => Inline,
325            "int" => Int,
326            "long" => Long,
327            "register" => Register,
328            "restrict" => Restrict,
329            "return" => Return,
330            "short" => Short,
331            "signed" => Signed,
332            "sizeof" => Sizeof,
333            "static" => Static,
334            "struct" => Struct,
335            "switch" => Switch,
336            "typedef" => Typedef,
337            "union" => Union,
338            "unsigned" => Unsigned,
339            "void" => Void,
340            "volatile" => Volatile,
341            "while" => While,
342            "_Bool" => Bool,
343            "_Complex" => Complex,
344            "_Imaginary" => Imaginary,
345            _ => return None,
346        })
347    }
348}
349
350/// A C99 punctuator.
351///
352/// Digraphs are folded into the token they stand for: `<:` lexes as
353/// [`Punct::LBracket`], `%:%:` as [`Punct::HashHash`], and so on.
354#[derive(Clone, Copy, PartialEq, Eq, Debug, Hash)]
355#[allow(missing_docs)]
356pub enum Punct {
357    LBracket,
358    RBracket,
359    LParen,
360    RParen,
361    LBrace,
362    RBrace,
363    Dot,
364    Arrow,
365    PlusPlus,
366    MinusMinus,
367    Amp,
368    Star,
369    Plus,
370    Minus,
371    Tilde,
372    Bang,
373    Slash,
374    Percent,
375    Shl,
376    Shr,
377    Lt,
378    Gt,
379    Le,
380    Ge,
381    EqEq,
382    Ne,
383    Caret,
384    Pipe,
385    AmpAmp,
386    PipePipe,
387    Question,
388    Colon,
389    Semi,
390    Ellipsis,
391    Assign,
392    StarAssign,
393    SlashAssign,
394    PercentAssign,
395    PlusAssign,
396    MinusAssign,
397    ShlAssign,
398    ShrAssign,
399    AmpAssign,
400    CaretAssign,
401    PipeAssign,
402    Comma,
403    Hash,
404    HashHash,
405}
406
407impl Punct {
408    /// The canonical spelling of this punctuator (never the digraph form).
409    pub fn as_str(self) -> &'static str {
410        use Punct::*;
411        match self {
412            LBracket => "[",
413            RBracket => "]",
414            LParen => "(",
415            RParen => ")",
416            LBrace => "{",
417            RBrace => "}",
418            Dot => ".",
419            Arrow => "->",
420            PlusPlus => "++",
421            MinusMinus => "--",
422            Amp => "&",
423            Star => "*",
424            Plus => "+",
425            Minus => "-",
426            Tilde => "~",
427            Bang => "!",
428            Slash => "/",
429            Percent => "%",
430            Shl => "<<",
431            Shr => ">>",
432            Lt => "<",
433            Gt => ">",
434            Le => "<=",
435            Ge => ">=",
436            EqEq => "==",
437            Ne => "!=",
438            Caret => "^",
439            Pipe => "|",
440            AmpAmp => "&&",
441            PipePipe => "||",
442            Question => "?",
443            Colon => ":",
444            Semi => ";",
445            Ellipsis => "...",
446            Assign => "=",
447            StarAssign => "*=",
448            SlashAssign => "/=",
449            PercentAssign => "%=",
450            PlusAssign => "+=",
451            MinusAssign => "-=",
452            ShlAssign => "<<=",
453            ShrAssign => ">>=",
454            AmpAssign => "&=",
455            CaretAssign => "^=",
456            PipeAssign => "|=",
457            Comma => ",",
458            Hash => "#",
459            HashHash => "##",
460        }
461    }
462}
463
464/// Punctuator spellings, longest first so that a greedy match is also the
465/// maximal munch the standard asks for.
466const PUNCTUATORS: &[(&str, Punct)] = &[
467    ("%:%:", Punct::HashHash),
468    ("...", Punct::Ellipsis),
469    ("<<=", Punct::ShlAssign),
470    (">>=", Punct::ShrAssign),
471    ("->", Punct::Arrow),
472    ("++", Punct::PlusPlus),
473    ("--", Punct::MinusMinus),
474    ("<<", Punct::Shl),
475    (">>", Punct::Shr),
476    ("<=", Punct::Le),
477    (">=", Punct::Ge),
478    ("==", Punct::EqEq),
479    ("!=", Punct::Ne),
480    ("&&", Punct::AmpAmp),
481    ("||", Punct::PipePipe),
482    ("*=", Punct::StarAssign),
483    ("/=", Punct::SlashAssign),
484    ("%=", Punct::PercentAssign),
485    ("+=", Punct::PlusAssign),
486    ("-=", Punct::MinusAssign),
487    ("&=", Punct::AmpAssign),
488    ("^=", Punct::CaretAssign),
489    ("|=", Punct::PipeAssign),
490    ("##", Punct::HashHash),
491    ("<:", Punct::LBracket),
492    (":>", Punct::RBracket),
493    ("<%", Punct::LBrace),
494    ("%>", Punct::RBrace),
495    ("%:", Punct::Hash),
496    ("[", Punct::LBracket),
497    ("]", Punct::RBracket),
498    ("(", Punct::LParen),
499    (")", Punct::RParen),
500    ("{", Punct::LBrace),
501    ("}", Punct::RBrace),
502    (".", Punct::Dot),
503    ("&", Punct::Amp),
504    ("*", Punct::Star),
505    ("+", Punct::Plus),
506    ("-", Punct::Minus),
507    ("~", Punct::Tilde),
508    ("!", Punct::Bang),
509    ("/", Punct::Slash),
510    ("%", Punct::Percent),
511    ("<", Punct::Lt),
512    (">", Punct::Gt),
513    ("^", Punct::Caret),
514    ("|", Punct::Pipe),
515    ("?", Punct::Question),
516    (":", Punct::Colon),
517    (";", Punct::Semi),
518    ("=", Punct::Assign),
519    (",", Punct::Comma),
520    ("#", Punct::Hash),
521];
522
523/// The `l`/`ll` part of an integer suffix.
524#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
525pub enum LongKind {
526    /// No `l` suffix.
527    #[default]
528    None,
529    /// `l` or `L`.
530    Long,
531    /// `ll` or `LL`.
532    LongLong,
533}
534
535/// The base an integer constant was written in.
536#[derive(Clone, Copy, PartialEq, Eq, Debug)]
537pub enum NumBase {
538    /// `0b101` — C23.
539    Binary,
540    /// `0777`
541    Octal,
542    /// `42`
543    Decimal,
544    /// `0x2a`
545    Hex,
546}
547
548impl NumBase {
549    /// The radix.
550    pub fn radix(self) -> u32 {
551        match self {
552            NumBase::Binary => 2,
553            NumBase::Octal => 8,
554            NumBase::Decimal => 10,
555            NumBase::Hex => 16,
556        }
557    }
558
559    /// How a diagnostic names this base.
560    pub fn as_str(self) -> &'static str {
561        match self {
562            NumBase::Binary => "binary",
563            NumBase::Octal => "octal",
564            NumBase::Decimal => "decimal",
565            NumBase::Hex => "hexadecimal",
566        }
567    }
568}
569
570/// A decoded integer constant.
571#[derive(Clone, PartialEq, Eq, Debug)]
572pub struct IntLit {
573    /// The value, before any type is chosen for it.
574    pub value: u128,
575    /// How it was written.
576    pub base: NumBase,
577    /// Whether a `u`/`U` suffix was present.
578    pub unsigned: bool,
579    /// Whether an `l`/`ll` suffix was present.
580    pub long: LongKind,
581    /// The exact source spelling.
582    pub text: String,
583}
584
585/// The suffix of a floating constant.
586#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
587pub enum FloatSuffix {
588    /// No suffix: the constant has type `double`.
589    #[default]
590    None,
591    /// `f`/`F`: `float`.
592    Float,
593    /// `l`/`L`: `long double`.
594    LongDouble,
595}
596
597/// A decoded floating constant.
598#[derive(Clone, PartialEq, Debug)]
599pub struct FloatLit {
600    /// The value, rounded to `f64`.
601    pub value: f64,
602    /// The suffix, which fixes the constant's type.
603    pub suffix: FloatSuffix,
604    /// Whether the constant carried GNU's imaginary suffix, `i` or `j`.
605    ///
606    /// `2.0i` is `(0, 2)` of the complex type [`FloatLit::suffix`] names the
607    /// real part of, which is what makes `1.0 + 2.0i` read the way every C
608    /// program writes a complex constant. It is a GNU extension that C11
609    /// blessed by giving `<complex.h>` a `CMPLX` built on it; the *standard*
610    /// spelling of the same thing is `_Complex_I`.
611    pub imaginary: bool,
612    /// Whether the constant was written in hexadecimal form.
613    pub hex: bool,
614    /// The exact source spelling.
615    pub text: String,
616}
617
618/// A decoded character constant.
619#[derive(Clone, PartialEq, Eq, Debug)]
620pub struct CharLit {
621    /// The value of the constant.
622    ///
623    /// For a plain single-character constant this is the unsigned value of the
624    /// execution character (so `'\xff'` is `255`); whether that is later
625    /// interpreted as `-1` is up to sema, which knows the signedness of
626    /// `char`. Multi-character constants are packed big-endian, as GCC does.
627    pub value: i64,
628    /// Which prefix the constant was written with, which is what fixes its
629    /// type: `'x'` is an `int`, `L'x'` a `wchar_t`, `u'x'` a `char16_t`,
630    /// `U'x'` a `char32_t` and `u8'x'` a `char8_t`.
631    pub kind: StrKind,
632    /// The exact source spelling, including quotes.
633    pub text: String,
634}
635
636/// Which prefix a character constant or string literal was written with.
637///
638/// The five of them are the same set for both, which is why one enum serves
639/// both: `'x'`/`"…"`, `u8'x'`/`u8"…"`, `u'x'`/`u"…"`, `U'x'`/`U"…"` and
640/// `L'x'`/`L"…"`.
641#[derive(Clone, Copy, PartialEq, Eq, Debug)]
642pub enum StrKind {
643    /// `"…"` — bytes of the execution character set, which is UTF-8.
644    Narrow,
645    /// `u8"…"` (C11) and `u8'x'` (C23) — UTF-8 bytes, of type `char` before
646    /// C23 and `char8_t` (an `unsigned char`) from C23 on.
647    Utf8,
648    /// `u"…"` and `u'x'` (C11) — UTF-16 code units, of type `char16_t`.
649    Utf16,
650    /// `U"…"` and `U'x'` (C11) — UTF-32 code units, of type `char32_t`.
651    Utf32,
652    /// `L"…"` — `wchar_t`.
653    Wide,
654}
655
656impl StrKind {
657    /// The prefix a literal of this kind is written with.
658    pub fn prefix(self) -> &'static str {
659        match self {
660            StrKind::Narrow => "",
661            StrKind::Utf8 => "u8",
662            StrKind::Utf16 => "u",
663            StrKind::Utf32 => "U",
664            StrKind::Wide => "L",
665        }
666    }
667
668    /// The revision that introduced the prefix, in the form the constant is
669    /// being used in.
670    ///
671    /// `u8"…"` is C11 (N1488) and `u8'x'` is C23 (N2418); the other three
672    /// prefixes are the same revision either way.
673    pub fn since(self, character: bool) -> Standard {
674        match self {
675            StrKind::Narrow | StrKind::Wide => Standard::C89,
676            StrKind::Utf8 if character => Standard::C23,
677            StrKind::Utf8 | StrKind::Utf16 | StrKind::Utf32 => Standard::C11,
678        }
679    }
680
681    /// Whether the elements are the bytes of the source's UTF-8, rather than
682    /// character values.
683    fn is_bytes(self) -> bool {
684        matches!(self, StrKind::Narrow | StrKind::Utf8)
685    }
686
687    /// The largest value one element of this kind can hold.
688    ///
689    /// `wide_bits` is how wide the target's `wchar_t` is, which is the one
690    /// answer here that is not fixed by the language: 32 bits on the Unix
691    /// platforms and 16 on Windows, where `L"…"` is UTF-16 and a character
692    /// outside the basic multilingual plane takes a surrogate pair, exactly as
693    /// `u"…"` does.
694    fn max_element(self, wide_bits: u32) -> u32 {
695        match self {
696            StrKind::Narrow | StrKind::Utf8 => 0xff,
697            StrKind::Utf16 => 0xffff,
698            StrKind::Wide if wide_bits <= 16 => 0xffff,
699            StrKind::Utf32 | StrKind::Wide => u32::MAX,
700        }
701    }
702
703    /// Whether one element holds a UTF-16 code unit, so that a character
704    /// beyond the basic multilingual plane becomes a surrogate pair.
705    fn is_utf16(self, wide_bits: u32) -> bool {
706        self == StrKind::Utf16 || (self == StrKind::Wide && wide_bits <= 16)
707    }
708}
709
710/// A decoded string literal.
711#[derive(Clone, PartialEq, Eq, Debug)]
712pub struct StrLit {
713    /// Which prefix it was written with.
714    pub kind: StrKind,
715    /// The decoded elements: bytes (`0..=255`) for a narrow or UTF-8 literal,
716    /// UTF-16 code units (surrogate pairs and all) for a `u"…"` one, and
717    /// character values for `U"…"` and `L"…"`. The terminating NUL is *not*
718    /// included.
719    pub values: Vec<u32>,
720    /// The exact source spelling, including quotes.
721    pub text: String,
722}
723
724impl StrLit {
725    /// The literal's bytes, if it is an ordinary narrow one.
726    pub fn as_bytes(&self) -> Option<Vec<u8>> {
727        (self.kind == StrKind::Narrow).then(|| self.values.iter().map(|v| *v as u8).collect())
728    }
729}
730
731/// What a [`Token`] is.
732#[derive(Clone, PartialEq, Debug)]
733pub enum TokenKind {
734    /// End of the token list. Always present, exactly once, last.
735    Eof,
736    /// An identifier that is not a keyword.
737    Ident(String),
738    /// A keyword.
739    Keyword(Keyword),
740    /// An integer constant.
741    Int(IntLit),
742    /// A floating constant.
743    Float(FloatLit),
744    /// A character constant.
745    Char(CharLit),
746    /// A string literal.
747    Str(StrLit),
748    /// A punctuator.
749    Punct(Punct),
750    /// Text that is not a C token at all, kept so that a skipped group may
751    /// contain it. Holds its own spelling.
752    Error(String),
753}
754
755impl TokenKind {
756    /// A short description used in "expected …, found …" messages.
757    pub fn describe(&self) -> String {
758        match self {
759            TokenKind::Eof => "end of input".to_owned(),
760            TokenKind::Ident(name) => format!("identifier '{name}'"),
761            TokenKind::Keyword(k) => format!("keyword '{}'", k.as_str()),
762            TokenKind::Int(_) => "integer constant".to_owned(),
763            TokenKind::Float(_) => "floating constant".to_owned(),
764            TokenKind::Char(_) => "character constant".to_owned(),
765            TokenKind::Str(_) => "string literal".to_owned(),
766            TokenKind::Punct(p) => format!("'{}'", p.as_str()),
767            TokenKind::Error(text) => format!("'{text}'"),
768        }
769    }
770
771    /// The exact source spelling of this token.
772    ///
773    /// This is what `#` stringifies and what `##` pastes, so it has to be the
774    /// text as written — `0x1f` rather than `31`, `'\n'` rather than a
775    /// newline. [`TokenKind::Eof`] has no spelling and gives `""`.
776    pub fn spelling(&self) -> &str {
777        match self {
778            TokenKind::Eof => "",
779            TokenKind::Ident(name) => name,
780            TokenKind::Keyword(k) => k.as_str(),
781            TokenKind::Int(lit) => &lit.text,
782            TokenKind::Float(lit) => &lit.text,
783            TokenKind::Char(lit) => &lit.text,
784            TokenKind::Str(lit) => &lit.text,
785            TokenKind::Punct(p) => p.as_str(),
786            TokenKind::Error(text) => text,
787        }
788    }
789
790    /// The name this token has when it is used as a macro name.
791    ///
792    /// Keywords are ordinary identifiers during translation phase 4 — our
793    /// lexer classifies them early, so `#define restrict` and `#ifdef inline`
794    /// would otherwise be unusable — which is why a keyword answers with its
795    /// spelling here.
796    pub fn macro_name(&self) -> Option<&str> {
797        match self {
798            TokenKind::Ident(name) => Some(name),
799            TokenKind::Keyword(k) => Some(k.as_str()),
800            _ => None,
801        }
802    }
803}
804
805/// A lexed C token.
806#[derive(Clone, PartialEq, Debug)]
807pub struct Token {
808    /// What the token is.
809    pub kind: TokenKind,
810    /// Where it is.
811    pub range: SourceRange,
812    /// Whether it is the first token on its logical line (needed by the
813    /// preprocessor to spot directives).
814    pub bol: bool,
815    /// Whether whitespace or a comment preceded it (needed by the
816    /// preprocessor's stringification and macro replacement).
817    pub preceded_by_space: bool,
818    /// Everything wrong with this token, and with the text between it and the
819    /// token before it.
820    ///
821    /// The [preprocessor](crate::pp) reports what is wrong with the token
822    /// *itself* if and only if the token reaches its output. What stands
823    /// whatever becomes of the token — its spelling, and the comments that
824    /// preceded it — is reported as soon as the token is read from the file,
825    /// which is what keeps it from being lost when the token opens a directive,
826    /// names a macro or is an argument the macro drops; see
827    /// [`Diagnostic::lexical`].
828    pub errors: Vec<Diagnostic>,
829}
830
831impl Token {
832    /// The keyword this token is, if any.
833    pub fn keyword(&self) -> Option<Keyword> {
834        match &self.kind {
835            TokenKind::Keyword(k) => Some(*k),
836            _ => None,
837        }
838    }
839
840    /// Whether this token is the given punctuator.
841    pub fn is_punct(&self, p: Punct) -> bool {
842        self.kind == TokenKind::Punct(p)
843    }
844
845    /// Whether this token is the given keyword.
846    pub fn is_keyword(&self, k: Keyword) -> bool {
847        self.kind == TokenKind::Keyword(k)
848    }
849
850    /// Whether this token ends the input.
851    pub fn is_eof(&self) -> bool {
852        self.kind == TokenKind::Eof
853    }
854
855    /// The identifier this token is, if any.
856    pub fn ident(&self) -> Option<&str> {
857        match &self.kind {
858            TokenKind::Ident(name) => Some(name),
859            _ => None,
860        }
861    }
862}
863
864/// Knobs for the lexer.
865#[derive(Clone, Copy, Debug)]
866pub struct LexOptions {
867    /// Which standard's lexical rules to apply.
868    pub standard: Standard,
869    /// How a constant form a newer revision introduced is gated.
870    pub gating: crate::Gating,
871    /// Accept `$` in identifiers, like GCC's `-fdollars-in-identifiers`, which
872    /// is on by default; see [`crate::Options::dollar_in_identifiers`].
873    pub dollar_in_identifiers: bool,
874    /// Whether translation phase 1 replaces the nine trigraphs.
875    ///
876    /// See [`trigraphs_enabled`] for who has them and why.
877    pub trigraphs: bool,
878    /// How wide the target's `wchar_t` is.
879    ///
880    /// The one target property the *lexer* has an opinion about: it decides
881    /// what an `L'\xffff'` escape may hold and whether `L"😀"` is one element
882    /// or a surrogate pair, since Windows makes `wchar_t` 16 bits.
883    pub wchar_bits: u32,
884    /// Whether the complex types are available, which is what decides whether
885    /// an imaginary constant (`2.0i`) has a type; see
886    /// [`crate::Options::complex`].
887    pub complex: bool,
888}
889
890impl LexOptions {
891    /// The lexical rules of `standard`, in the strict ISO dialect.
892    pub fn new(standard: Standard) -> Self {
893        Self {
894            standard,
895            gating: crate::Gating {
896                standard,
897                dialect: crate::Dialect::Iso,
898            },
899            dollar_in_identifiers: true,
900            trigraphs: trigraphs_enabled(standard, crate::Dialect::Iso),
901            wchar_bits: crate::TargetModel::host().wchar_bits,
902            complex: crate::COMPLEX_SUPPORTED,
903        }
904    }
905}
906
907impl From<&Options> for LexOptions {
908    fn from(o: &Options) -> Self {
909        Self {
910            standard: o.standard,
911            gating: o.gating(),
912            dollar_in_identifiers: o.dollar_in_identifiers,
913            trigraphs: trigraphs_enabled(o.standard, o.dialect),
914            wchar_bits: o.target.wchar_bits,
915            complex: o.complex,
916        }
917    }
918}
919
920/// Whether translation phase 1 replaces trigraphs in this entry point.
921///
922/// The nine of them were in C from the beginning and C23 removed them
923/// (N2940), so every strict entry point below `c23!` has them and `c23!` does
924/// not. No *GNU* dialect has them: `gcc -std=gnu99` switches them off, because
925/// `"what??!"` in a string is far more likely to be an exclamation than a
926/// pipe, and that is the line Clang draws too.
927pub fn trigraphs_enabled(standard: Standard, dialect: crate::Dialect) -> bool {
928    standard < Standard::C23 && !dialect.is_gnu()
929}
930
931/// The nine trigraphs of C 5.2.1.1, as `(third character, replacement)`.
932const TRIGRAPHS: &[(u8, u8)] = &[
933    (b'=', b'#'),
934    (b'(', b'['),
935    (b'/', b'\\'),
936    (b')', b']'),
937    (b'\'', b'^'),
938    (b'<', b'{'),
939    (b'!', b'|'),
940    (b'>', b'}'),
941    (b'-', b'~'),
942];
943
944/// Lexes the root file of `source`.
945///
946/// The returned vector always ends with a [`TokenKind::Eof`] token. Problems
947/// are attached to the tokens they were found in ([`Token::errors`]) rather
948/// than reported; scanning always runs to the end of the input.
949pub fn lex(source: &Source, options: &Options) -> Vec<Token> {
950    let file = source.map.file(source.root);
951    lex_file(file.text(), file.base(), &options.into())
952}
953
954/// The UTF-8 byte order mark, which an editor on Windows may write at the start
955/// of a file.
956const BYTE_ORDER_MARK: char = '\u{feff}';
957
958/// Lexes the whole of a file's `text` — a unit's own or an `#include`d one —
959/// whose first byte lives at global offset `base`.
960///
961/// A byte order mark at the very start is skipped, as GCC and Clang skip it.
962/// It is skipped rather than removed: scanning simply starts three bytes in,
963/// so every token's offset, and every column the source map computes from one,
964/// is still the file's own. A byte order mark anywhere else is an ordinary
965/// stray character and an error, which [`lex_text`] reports.
966pub fn lex_file(text: &str, base: Pos, options: &LexOptions) -> Vec<Token> {
967    let start = if text.starts_with(BYTE_ORDER_MARK) {
968        BYTE_ORDER_MARK.len_utf8()
969    } else {
970        0
971    };
972    lex_from(text, base, start, options)
973}
974
975/// Lexes `text`, whose first byte lives at global offset `base`.
976pub fn lex_text(text: &str, base: Pos, options: &LexOptions) -> Vec<Token> {
977    lex_from(text, base, 0, options)
978}
979
980/// Lexes `text` from byte `start` on; the tokens' offsets count from the
981/// beginning of `text` all the same.
982fn lex_from(text: &str, base: Pos, start: usize, options: &LexOptions) -> Vec<Token> {
983    Lexer {
984        text,
985        bytes: text.as_bytes(),
986        base,
987        pos: start,
988        options: *options,
989        pending: Vec::new(),
990    }
991    .run()
992}
993
994struct Lexer<'a> {
995    text: &'a str,
996    bytes: &'a [u8],
997    base: Pos,
998    pos: usize,
999    options: LexOptions,
1000    /// Problems found since the last token was finished; they belong to the
1001    /// token currently being scanned.
1002    pending: Vec<Diagnostic>,
1003}
1004
1005impl<'a> Lexer<'a> {
1006    fn range(&self, start: usize, end: usize) -> SourceRange {
1007        SourceRange::new(self.base + start as Pos, self.base + end as Pos)
1008    }
1009
1010    /// Records a problem with the token being scanned.
1011    fn error(&mut self, range: SourceRange, message: impl Into<String>) {
1012        self.pending.push(Diagnostic::error(range, message));
1013    }
1014
1015    /// Records a problem that stands whether or not the token being scanned
1016    /// ever reaches the parser: one with its *spelling*, or one in the text
1017    /// between it and the token before it. See [`Diagnostic::lexical`].
1018    fn lexical_error(&mut self, range: SourceRange, message: impl Into<String>) {
1019        self.pending
1020            .push(Diagnostic::error(range, message).at_lexing());
1021    }
1022
1023    /// Records an advisory remark about the token being scanned.
1024    fn warning(&mut self, range: SourceRange, message: impl Into<String>) {
1025        self.pending.push(Diagnostic::warning(range, message));
1026    }
1027
1028    fn peek(&self) -> Option<u8> {
1029        self.bytes.get(self.pos).copied()
1030    }
1031
1032    fn peek_at(&self, n: usize) -> Option<u8> {
1033        self.bytes.get(self.pos + n).copied()
1034    }
1035
1036    fn eof(&self) -> bool {
1037        self.pos >= self.bytes.len()
1038    }
1039
1040    fn run(mut self) -> Vec<Token> {
1041        let mut tokens = Vec::new();
1042        let mut bol = true;
1043        let mut space = false;
1044        loop {
1045            let (saw_newline, saw_space) = self.skip_whitespace();
1046            bol |= saw_newline;
1047            space |= saw_space || saw_newline;
1048            if self.eof() {
1049                let end = self.bytes.len();
1050                tokens.push(Token {
1051                    kind: TokenKind::Eof,
1052                    range: self.range(end, end),
1053                    bol,
1054                    preceded_by_space: space,
1055                    errors: std::mem::take(&mut self.pending),
1056                });
1057                break;
1058            }
1059            let start = self.pos;
1060            let kind = self.scan_token();
1061            tokens.push(Token {
1062                kind,
1063                range: self.range(start, self.pos),
1064                bol,
1065                preceded_by_space: space,
1066                errors: std::mem::take(&mut self.pending),
1067            });
1068            bol = false;
1069            space = false;
1070        }
1071        tokens
1072    }
1073
1074    /// Skips whitespace, comments and line splices.
1075    ///
1076    /// Returns `(saw_newline, saw_space)`.
1077    fn skip_whitespace(&mut self) -> (bool, bool) {
1078        let mut newline = false;
1079        let mut space = false;
1080        loop {
1081            match self.peek() {
1082                Some(b'\n') => {
1083                    self.pos += 1;
1084                    newline = true;
1085                }
1086                Some(b' ' | b'\t' | b'\r' | 0x0b | 0x0c) => {
1087                    self.pos += 1;
1088                    space = true;
1089                }
1090                // Translation phase 2: a backslash immediately followed by a
1091                // newline splices the two lines, so it is *not* a line break.
1092                // `??/` is that backslash where trigraphs are on.
1093                Some(b'\\' | b'?') if self.is_line_splice(self.pos) => {
1094                    self.pos += self.line_splice_len(self.pos);
1095                    space = true;
1096                }
1097                Some(b'/') if self.peek_at(1) == Some(b'*') => {
1098                    // Translation phase 3 replaces the whole comment with one
1099                    // space, so a newline *inside* it is not a line break at
1100                    // all: a directive may span one, and a `#` after one does
1101                    // not start a directive. Both are what GCC does, and the
1102                    // standard's own `FUNC_LIKE` example in 6.10.3 depends on
1103                    // the first.
1104                    let start = self.pos;
1105                    self.pos += 2;
1106                    let mut closed = false;
1107                    while let Some(c) = self.peek() {
1108                        if c == b'*' && self.peek_at(1) == Some(b'/') {
1109                            self.pos += 2;
1110                            closed = true;
1111                            break;
1112                        }
1113                        self.pos += 1;
1114                    }
1115                    if !closed {
1116                        // Both of the problems a *comment* can have belong to
1117                        // the token that follows it for want of anywhere else
1118                        // to put them, and neither is about that token: whether
1119                        // a comment is terminated, and whether this revision
1120                        // has `//`, is settled in translation phase 3 by the
1121                        // text alone. So they are recorded as lexical, and
1122                        // reported wherever the token ends up — including
1123                        // nowhere, which is what a `#` opening a directive and
1124                        // a macro name do with it.
1125                        let range = self.range(start, self.bytes.len());
1126                        self.lexical_error(range, "unterminated comment");
1127                    }
1128                    space = true;
1129                }
1130                Some(b'/') if self.peek_at(1) == Some(b'/') => {
1131                    let start = self.pos;
1132                    self.pos += 2;
1133                    while let Some(c) = self.peek() {
1134                        if c == b'\n' {
1135                            break;
1136                        }
1137                        if matches!(c, b'\\' | b'?') && self.is_line_splice(self.pos) {
1138                            self.pos += self.line_splice_len(self.pos);
1139                            continue;
1140                        }
1141                        self.pos += 1;
1142                    }
1143                    // C99 took the `//` comment from C++ (N644); before that
1144                    // `a //* b */ c` was a division, which is why the gate is
1145                    // here rather than being a warning. Lexical for the reason
1146                    // above: a `c89!` block that writes one has to be told so
1147                    // whatever follows it.
1148                    if let Some(message) = self
1149                        .options
1150                        .gating
1151                        .requires("a '//' comment", Standard::C99)
1152                    {
1153                        let range = self.range(start, self.pos);
1154                        self.lexical_error(range, message);
1155                    }
1156                    space = true;
1157                }
1158                _ => return (newline, space),
1159            }
1160        }
1161    }
1162
1163    fn is_line_splice(&self, at: usize) -> bool {
1164        self.line_splice_len(at) > 0
1165    }
1166
1167    /// Length of a `\`-newline splice starting at `at`, or 0.
1168    ///
1169    /// Translation phase 1 runs *before* phase 2, so `??/` at the end of a
1170    /// line splices it exactly as a written backslash does — which is the one
1171    /// trigraph whose replacement is not a character the lexer can simply hand
1172    /// on.
1173    fn line_splice_len(&self, at: usize) -> usize {
1174        let lead = match self.trigraph_at(at) {
1175            Some(b'\\') => 3,
1176            Some(_) => return 0,
1177            None if self.bytes.get(at) == Some(&b'\\') => 1,
1178            None => return 0,
1179        };
1180        match (self.bytes.get(at + lead), self.bytes.get(at + lead + 1)) {
1181            (Some(b'\n'), _) => lead + 1,
1182            (Some(b'\r'), Some(b'\n')) => lead + 2,
1183            _ => 0,
1184        }
1185    }
1186
1187    /// The character a trigraph at `at` stands for, if there is one there.
1188    ///
1189    /// C 5.2.1.1: the nine three-character sequences beginning `??` are
1190    /// replaced in translation phase 1, before line splicing and before the
1191    /// source is split into tokens — so this is consulted from everywhere the
1192    /// lexer looks at a raw byte, rather than the text being rewritten. Not
1193    /// rewriting it is what keeps every [`SourceRange`] a range of the source
1194    /// the user really wrote: a diagnostic about `??=` points at all three
1195    /// characters.
1196    fn trigraph_at(&self, at: usize) -> Option<u8> {
1197        if !self.options.trigraphs
1198            || self.bytes.get(at) != Some(&b'?')
1199            || self.bytes.get(at + 1) != Some(&b'?')
1200        {
1201            return None;
1202        }
1203        let third = *self.bytes.get(at + 2)?;
1204        TRIGRAPHS
1205            .iter()
1206            .find(|(c, _)| *c == third)
1207            .map(|(_, replacement)| *replacement)
1208    }
1209
1210    /// The length in source bytes of the character at `at`, which is three for
1211    /// a trigraph and one otherwise.
1212    fn trigraph_len(&self, at: usize) -> usize {
1213        if self.trigraph_at(at).is_some() { 3 } else { 1 }
1214    }
1215
1216    /// The character-constant or string-literal prefix starting here, with its
1217    /// length in bytes.
1218    ///
1219    /// A prefix is only one when a quote follows it, which is what keeps
1220    /// `unsigned`, `u8x` and a variable called `U` ordinary identifiers.
1221    fn literal_prefix(&self, first: u8) -> Option<(StrKind, usize)> {
1222        let (kind, len) = match first {
1223            b'L' => (StrKind::Wide, 1),
1224            b'U' => (StrKind::Utf32, 1),
1225            b'u' if self.peek_at(1) == Some(b'8') => (StrKind::Utf8, 2),
1226            b'u' => (StrKind::Utf16, 1),
1227            _ => return None,
1228        };
1229        matches!(self.peek_at(len), Some(b'"' | b'\'')).then_some((kind, len))
1230    }
1231
1232    /// Whether an identifier that begins with an extended character starts
1233    /// here.
1234    fn extended_ident_start(&self) -> bool {
1235        match self.peek() {
1236            Some(b'\\') if matches!(self.peek_at(1), Some(b'u' | b'U')) => {
1237                let want = if self.peek_at(1) == Some(b'u') { 4 } else { 8 };
1238                (0..want).all(|i| self.peek_at(2 + i).is_some_and(|c| c.is_ascii_hexdigit()))
1239            }
1240            Some(c) if c >= 0x80 => self.text[self.pos..]
1241                .chars()
1242                .next()
1243                .is_some_and(|ch| is_extended_ident_char(ch, true)),
1244            _ => false,
1245        }
1246    }
1247
1248    /// Scans one token, which is [`TokenKind::Error`] for text that is not a C
1249    /// token at all.
1250    fn scan_token(&mut self) -> TokenKind {
1251        let Some(c) = self.peek() else {
1252            return TokenKind::Eof;
1253        };
1254        if is_ident_start(c, self.options.dollar_in_identifiers) {
1255            // `L'x'`, `u8"…"`, `u'x'` and the rest are constants rather than an
1256            // identifier followed by one. The prefix is recognised in every
1257            // entry point and *gated* rather than not recognised at all: a
1258            // `c99!` block that writes `u"x"` is told which macro has it,
1259            // instead of being told that `u` is undeclared.
1260            if let Some((kind, len)) = self.literal_prefix(c) {
1261                let start = self.pos;
1262                self.pos += len;
1263                let character = self.peek() == Some(b'\'');
1264                if let Some(message) = self.options.gating.requires(
1265                    &format!("a '{}' literal", kind.prefix()),
1266                    kind.since(character),
1267                ) {
1268                    let range = self.range(start, self.pos);
1269                    self.error(range, message);
1270                }
1271                return if character {
1272                    self.scan_char_constant(kind)
1273                } else {
1274                    self.scan_string_literal(kind)
1275                };
1276            }
1277            return self.scan_ident();
1278        }
1279        // An extended identifier, written either as the character itself —
1280        // which GCC and Clang have taken since GCC 10 — or as the universal
1281        // character name C99 6.4.2.1 introduced for it.
1282        if self.extended_ident_start() {
1283            return self.scan_ident();
1284        }
1285        if c.is_ascii_digit() || (c == b'.' && self.peek_at(1).is_some_and(|d| d.is_ascii_digit()))
1286        {
1287            return self.scan_number();
1288        }
1289        if c == b'\'' {
1290            return self.scan_char_constant(StrKind::Narrow);
1291        }
1292        if c == b'"' {
1293            return self.scan_string_literal(StrKind::Narrow);
1294        }
1295        if let Some(p) = self.scan_punctuator() {
1296            return TokenKind::Punct(p);
1297        }
1298
1299        // Anything else is not a C token at all.
1300        let start = self.pos;
1301        // `??/` that does not splice a line is the stray backslash a written
1302        // one would be, and is reported as one rather than as two question
1303        // marks.
1304        let ch = match self.trigraph_at(start) {
1305            Some(c) => {
1306                self.pos += 3;
1307                c as char
1308            }
1309            None => {
1310                let ch = self.text[start..].chars().next().unwrap_or('\u{fffd}');
1311                self.pos += ch.len_utf8();
1312                ch
1313            }
1314        };
1315        let range = self.range(start, self.pos);
1316        self.error(
1317            range,
1318            format!("unexpected character '{}' in program", ch.escape_debug()),
1319        );
1320        TokenKind::Error(ch.to_string())
1321    }
1322
1323    /// Scans an identifier or a keyword.
1324    ///
1325    /// Translation phase 2 deletes a backslash-newline *before* the source is
1326    /// split into tokens, so one may sit in the middle of an identifier:
1327    /// `__LI\<newline>NE__` is `__LINE__`, and Clang's own `drs/dr464.c`
1328    /// writes exactly that. Almost no identifier has one, so the spelling
1329    /// stays a slice of the source until the first splice is found and only
1330    /// then becomes a `String`.
1331    fn scan_ident(&mut self) -> TokenKind {
1332        let start = self.pos;
1333        // The text before the current splice or universal character name, when
1334        // there has been one.
1335        let mut spliced: Option<String> = None;
1336        // Where the run of characters that is still a slice begins.
1337        let mut segment = start;
1338        // Whether anything outside the basic character set was written, which
1339        // is the only case that has to be checked for normalization.
1340        let mut extended = false;
1341        loop {
1342            match self.peek() {
1343                Some(c) if c < 0x80 && is_ident_continue(c, self.options.dollar_in_identifiers) => {
1344                    self.pos += 1;
1345                }
1346                // Only a splice that the identifier *continues* over: one at
1347                // the end of it is whitespace, and belongs to whatever comes
1348                // next.
1349                Some(b'\\' | b'?')
1350                    if self.line_splice_len(self.pos) > 0
1351                        && self
1352                            .bytes
1353                            .get(self.pos + self.line_splice_len(self.pos))
1354                            .is_some_and(|c| {
1355                                is_ident_continue(*c, self.options.dollar_in_identifiers)
1356                            }) =>
1357                {
1358                    let text = spliced.get_or_insert_with(String::new);
1359                    text.push_str(&self.text[segment..self.pos]);
1360                    self.pos += self.line_splice_len(self.pos);
1361                    segment = self.pos;
1362                }
1363                // A universal character name spells one extended character:
1364                // `café` and `café` are the same identifier, which is
1365                // exactly what C99 6.4.2.1 says.
1366                Some(b'\\') if matches!(self.peek_at(1), Some(b'u' | b'U')) => {
1367                    let at = self.pos;
1368                    let Some(ch) = self.scan_ident_ucn(at == start) else {
1369                        break;
1370                    };
1371                    let text = spliced.get_or_insert_with(String::new);
1372                    text.push_str(&self.text[segment..at]);
1373                    text.push(ch);
1374                    segment = self.pos;
1375                    extended = true;
1376                }
1377                Some(c) if c >= 0x80 => {
1378                    let ch = self.text[self.pos..].chars().next().unwrap_or('\u{fffd}');
1379                    if !is_extended_ident_char(ch, self.pos == start) {
1380                        break;
1381                    }
1382                    self.pos += ch.len_utf8();
1383                    extended = true;
1384                }
1385                _ => break,
1386            }
1387        }
1388        let text = match spliced {
1389            Some(mut text) => {
1390                text.push_str(&self.text[segment..self.pos]);
1391                text
1392            }
1393            None => self.text[start..self.pos].to_owned(),
1394        };
1395        // Nothing was an identifier character after all — the whole of it was
1396        // one bad universal character name, which has been reported. Something
1397        // has to be consumed, or the scanner would sit here forever.
1398        if text.is_empty() {
1399            if self.pos == start {
1400                let ch = self.text[start..].chars().next().unwrap_or('\u{fffd}');
1401                self.pos += ch.len_utf8();
1402            }
1403            return TokenKind::Error(self.text[start..self.pos].to_owned());
1404        }
1405        // Rust identifiers have to be in Normalization Form C, and `rustc`
1406        // *normalises* the ones a procedural macro hands it rather than
1407        // refusing them — so two C identifiers that differ only by
1408        // normalization would silently become one Rust item. C23 (N2836) asks
1409        // for NFC as well, so refusing is both the safe answer and the
1410        // conforming one.
1411        if extended && !unicode_normalization::is_nfc(&text) {
1412            let range = self.range(start, self.pos);
1413            self.error(
1414                range,
1415                format!(
1416                    "identifier '{text}' is not in Unicode Normalization Form C; \
1417                     write the composed form"
1418                ),
1419            );
1420        }
1421        match Keyword::from_str(&text, self.options.standard) {
1422            Some(k) => TokenKind::Keyword(k),
1423            None => TokenKind::Ident(text),
1424        }
1425    }
1426
1427    /// Reads a `\uXXXX` or `\UXXXXXXXX` written inside an identifier.
1428    ///
1429    /// A name that is too short to be one leaves the position where it was, so
1430    /// that the identifier simply ends there and the backslash is reported by
1431    /// [`Lexer::scan_token`] as the stray character it is. One that is the
1432    /// right shape but names something an identifier may not hold is *always*
1433    /// consumed, and reported: leaving it would be a second diagnostic about
1434    /// the same text, and — where it is the first character of the identifier
1435    /// — a token that consumed nothing at all.
1436    fn scan_ident_ucn(&mut self, start: bool) -> Option<char> {
1437        let at = self.pos;
1438        let want = if self.peek_at(1) == Some(b'u') { 4 } else { 8 };
1439        let mut value: u32 = 0;
1440        for i in 0..want {
1441            let digit = self.peek_at(2 + i).and_then(|c| (c as char).to_digit(16))?;
1442            value = value * 16 + digit;
1443        }
1444        let spelling = if want == 4 { 'u' } else { 'U' };
1445        let ch = char::from_u32(value);
1446        if !ch.is_some_and(|ch| is_extended_ident_char(ch, start)) {
1447            self.pos += 2 + want;
1448            let range = self.range(at, self.pos);
1449            let digits = if want == 4 {
1450                format!("{value:04X}")
1451            } else {
1452                format!("{value:08X}")
1453            };
1454            let message =
1455                format!("'\\{spelling}{digits}' is not a valid character in an identifier");
1456            // Two different refusals wear the same words. Naming a character
1457            // below U+00A0, or a surrogate, is what C99 6.4.3p2 forbids of the
1458            // *name* — C23 puts `$`, `@` and `` ` `` in the basic character
1459            // set, and Clang refuses those in every mode — so it is ill formed
1460            // where it is written and is reported even where the token goes
1461            // nowhere: Clang's own `C99/n717.c` is a file of them, each
1462            // written as the argument of a macro that expands to nothing.
1463            // Everything else here — a value outside Unicode, a character
1464            // that simply is not `XID_Continue` — is about the *identifier*,
1465            // and a preprocessing token that never reaches the parser is
1466            // allowed to be one C has no other use for (6.4p3).
1467            if value < 0xA0 || (0xD800..=0xDFFF).contains(&value) {
1468                self.lexical_error(range, message);
1469            } else {
1470                self.error(range, message);
1471            }
1472            return None;
1473        }
1474        if let Some(message) = self
1475            .options
1476            .gating
1477            .requires("a universal character name", Standard::C99)
1478        {
1479            let range = self.range(at, at + 2 + want);
1480            self.error(range, message);
1481        }
1482        self.pos += 2 + want;
1483        ch
1484    }
1485
1486    fn scan_punctuator(&mut self) -> Option<Punct> {
1487        if self.trigraph_at(self.pos).is_some() {
1488            return self.scan_trigraph_punctuator();
1489        }
1490        let rest = &self.text[self.pos..];
1491        for (spelling, punct) in PUNCTUATORS {
1492            if rest.starts_with(spelling) {
1493                // `<:` is a digraph for `[`, but `1<::x` must not be mangled;
1494                // C99 has no `::`, so a plain greedy match is correct here.
1495                self.pos += spelling.len();
1496                return Some(*punct);
1497            }
1498        }
1499        None
1500    }
1501
1502    /// A punctuator that begins with a trigraph.
1503    ///
1504    /// Phase 1 happens before tokens exist, so a punctuator may be spelled
1505    /// partly or wholly with trigraphs — `??!??!` is `||`, `??'=` is `^=`,
1506    /// `??=??=` is `##` — and maximal munch applies to the *replaced*
1507    /// characters. Up to the length of the longest punctuator is replaced into
1508    /// a small buffer, matched there, and the source position advanced by what
1509    /// the matched characters really cost.
1510    fn scan_trigraph_punctuator(&mut self) -> Option<Punct> {
1511        const LONGEST: usize = 4;
1512        let mut logical = [0u8; LONGEST];
1513        let mut widths = [0usize; LONGEST];
1514        let mut count = 0;
1515        let mut at = self.pos;
1516        while count < LONGEST {
1517            let (c, width) = match self.trigraph_at(at) {
1518                Some(c) => (c, 3),
1519                None => match self.bytes.get(at) {
1520                    Some(c) => (*c, 1),
1521                    None => break,
1522                },
1523            };
1524            // A backslash is not part of any punctuator, and neither is
1525            // anything outside the basic character set.
1526            if c == b'\\' || !c.is_ascii() {
1527                break;
1528            }
1529            logical[count] = c;
1530            widths[count] = width;
1531            count += 1;
1532            at += width;
1533        }
1534        let text = std::str::from_utf8(&logical[..count]).ok()?;
1535        for (spelling, punct) in PUNCTUATORS {
1536            if text.starts_with(spelling) {
1537                self.pos += widths[..spelling.len()].iter().sum::<usize>();
1538                return Some(*punct);
1539            }
1540        }
1541        None
1542    }
1543
1544    // -- numbers ------------------------------------------------------------
1545
1546    /// Scans a preprocessing number and classifies it as integer or float.
1547    ///
1548    /// Scanning the whole pp-number first (rather than stopping at the first
1549    /// character that does not fit) is what lets `08` or `1.0q` be reported as
1550    /// one bad token instead of two good ones.
1551    fn scan_number(&mut self) -> TokenKind {
1552        let start = self.pos;
1553        if self.peek() == Some(b'.') {
1554            self.pos += 1;
1555        }
1556        self.pos += 1;
1557        let mut separators = false;
1558        while let Some(c) = self.peek() {
1559            if matches!(c, b'e' | b'E' | b'p' | b'P')
1560                && matches!(self.peek_at(1), Some(b'+') | Some(b'-'))
1561            {
1562                self.pos += 2;
1563                continue;
1564            }
1565            // C23's digit separator. It is part of the pp-number when a digit
1566            // or a nondigit follows it, which is what keeps `1'a'` from
1567            // swallowing the character constant that follows a constant.
1568            if c == b'\''
1569                && self
1570                    .peek_at(1)
1571                    .is_some_and(|d| d.is_ascii_alphanumeric() || d == b'_')
1572            {
1573                separators = true;
1574                self.pos += 1;
1575                continue;
1576            }
1577            if c.is_ascii_alphanumeric()
1578                || c == b'_'
1579                || c == b'.'
1580                || (c == b'$' && self.options.dollar_in_identifiers)
1581            {
1582                self.pos += 1;
1583                continue;
1584            }
1585            break;
1586        }
1587        let text = &self.text[start..self.pos];
1588        let range = self.range(start, self.pos);
1589        if separators
1590            && let Some(message) = self
1591                .options
1592                .gating
1593                .requires("a digit separator", Standard::C23)
1594        {
1595            self.error(range, message);
1596        }
1597        // Everything below reads the digits; the separators are not part of
1598        // the value, while `text` keeps the spelling `#` has to reproduce.
1599        let stripped: String;
1600        let digits = if separators {
1601            stripped = text.replace('\'', "");
1602            stripped.as_str()
1603        } else {
1604            text
1605        };
1606        let lower = digits.to_ascii_lowercase();
1607        let hex = lower.starts_with("0x");
1608        let is_float = if hex {
1609            digits.contains('.') || lower[2..].contains('p')
1610        } else {
1611            digits.contains('.') || (!lower.starts_with("0b") && lower.contains('e'))
1612        };
1613        if is_float {
1614            TokenKind::Float(self.decode_float(digits, text, range, hex))
1615        } else {
1616            TokenKind::Int(self.decode_int(digits, text, range))
1617        }
1618    }
1619
1620    /// Decodes an integer constant from its separator-free `digits`; `text` is
1621    /// the spelling as written, which is what diagnostics quote and what `#`
1622    /// reproduces.
1623    fn decode_int(&mut self, digits: &str, text: &str, range: SourceRange) -> IntLit {
1624        let bytes = digits.as_bytes();
1625        let (base, digits_start) =
1626            if digits.len() >= 2 && (bytes[1] | 0x20) == b'x' && bytes[0] == b'0' {
1627                (NumBase::Hex, 2)
1628            } else if digits.len() >= 2 && (bytes[1] | 0x20) == b'b' && bytes[0] == b'0' {
1629                (NumBase::Binary, 2)
1630            } else if bytes[0] == b'0' && digits.len() > 1 {
1631                (NumBase::Octal, 1)
1632            } else {
1633                (NumBase::Decimal, 0)
1634            };
1635        if base == NumBase::Binary
1636            && let Some(message) = self
1637                .options
1638                .gating
1639                .requires("a binary integer constant", Standard::C23)
1640        {
1641            self.error(range, message);
1642        }
1643
1644        let radix = base.radix();
1645        // Accept every decimal digit in an octal or binary constant so that
1646        // the whole constant is consumed and reported once, the way C
1647        // compilers do.
1648        let scan_radix = if radix < 10 { 10 } else { radix };
1649
1650        let mut i = digits_start;
1651        let mut value: u128 = 0;
1652        let mut overflow = false;
1653        let mut bad_digit: Option<char> = None;
1654        while i < bytes.len() {
1655            let c = bytes[i] as char;
1656            let digit = match c.to_digit(scan_radix) {
1657                Some(d) => d,
1658                None => break,
1659            };
1660            if digit >= radix && bad_digit.is_none() {
1661                bad_digit = Some(c);
1662            }
1663            match value
1664                .checked_mul(radix as u128)
1665                .and_then(|v| v.checked_add(digit as u128))
1666            {
1667                Some(v) => value = v,
1668                None => overflow = true,
1669            }
1670            i += 1;
1671        }
1672
1673        if i == digits_start && matches!(base, NumBase::Hex | NumBase::Binary) {
1674            let prefix = &digits[..2];
1675            self.error(
1676                range,
1677                format!(
1678                    "expected digits after '{prefix}' in {} constant",
1679                    base.as_str()
1680                ),
1681            );
1682        }
1683        if let Some(c) = bad_digit {
1684            self.error(
1685                range,
1686                format!("invalid digit '{c}' in {} constant '{text}'", base.as_str()),
1687            );
1688        }
1689        if overflow {
1690            self.error(range, format!("integer constant '{text}' is too large"));
1691        }
1692
1693        let suffix = &digits[i..];
1694        let (unsigned, long) = match parse_int_suffix(suffix) {
1695            Some(v) => v,
1696            None => {
1697                // `3i` is GCC's `_Complex int`, one of the two complex
1698                // *integer* types that are a GNU extension of their own and
1699                // that cinrs does not have. Saying what to write instead is
1700                // more use than "invalid suffix".
1701                if matches!(suffix, "i" | "j" | "I" | "J") {
1702                    let digits = text.strip_suffix(suffix).unwrap_or(text);
1703                    self.error(
1704                        range,
1705                        format!(
1706                            "'{text}' is a complex integer constant, which is a GNU extension \
1707                             cinrs does not support; write '{digits}.0{suffix}' for the \
1708                             complex floating constant"
1709                        ),
1710                    );
1711                } else {
1712                    self.error(
1713                        range,
1714                        format!("invalid suffix '{suffix}' on integer constant '{text}'"),
1715                    );
1716                }
1717                (false, LongKind::None)
1718            }
1719        };
1720
1721        IntLit {
1722            value,
1723            base,
1724            unsigned,
1725            long,
1726            text: text.to_owned(),
1727        }
1728    }
1729
1730    fn decode_float(
1731        &mut self,
1732        digits: &str,
1733        text: &str,
1734        range: SourceRange,
1735        hex: bool,
1736    ) -> FloatLit {
1737        let (body, suffix) = split_float_suffix(digits, hex);
1738        let (suffix_kind, imaginary) = self.float_suffix(suffix, text, range);
1739
1740        if hex
1741            && let Some(message) = self
1742                .options
1743                .gating
1744                .requires("a hexadecimal floating constant", Standard::C99)
1745        {
1746            self.error(range, message);
1747        }
1748        let value = if hex {
1749            match parse_hex_float(body) {
1750                Some(v) => v,
1751                None => {
1752                    self.error(
1753                        range,
1754                        format!(
1755                            "invalid hexadecimal floating constant '{text}'; \
1756                             a 'p' exponent is required"
1757                        ),
1758                    );
1759                    0.0
1760                }
1761            }
1762        } else {
1763            match parse_decimal_float(body) {
1764                Some(v) => v,
1765                None => {
1766                    self.error(range, format!("invalid floating constant '{text}'"));
1767                    0.0
1768                }
1769            }
1770        };
1771
1772        FloatLit {
1773            value,
1774            suffix: suffix_kind,
1775            imaginary,
1776            hex,
1777            text: text.to_owned(),
1778        }
1779    }
1780
1781    /// The type a floating constant's suffix gives it.
1782    ///
1783    /// C has three: none, `f` and `l`. GCC has a dozen more, and they fall
1784    /// into three groups here.
1785    ///
1786    /// * **The ones that name a format wider than `double`** — `d`, `w`
1787    ///   (`__float80`), `q` (`__float128`), and the `_FloatN` and `_FloatNx`
1788    ///   suffixes `f64`, `f64x`, `f32x` and `f128`. Every one of them is
1789    ///   `double` in this implementation, exactly as `long double` is, so each
1790    ///   is accepted in a GNU dialect and **loses precision** where GCC would
1791    ///   not; `doc/gnu-extensions.md` records that. `f32` is `float`.
1792    /// * **The decimal floating suffixes** `df`, `dd` and `dl`, whose types
1793    ///   are radix-10 and have no Rust counterpart at all.
1794    /// * **The imaginary suffixes** `i` and `j`, which make the constant an
1795    ///   imaginary one — `2.0i` is `(0, 2)` — and so need the complex types.
1796    ///
1797    /// The decimal ones are refused with the reason, and so are the imaginary
1798    /// ones when the complex types are switched off. A strict entry point
1799    /// refuses the first group too, naming the GNU entry point that has it —
1800    /// the suffixes are spelled without underscores, which is the line
1801    /// [`crate::Dialect`] draws.
1802    ///
1803    /// The second half of the answer is whether the constant is imaginary; see
1804    /// [`FloatLit::imaginary`].
1805    fn float_suffix(
1806        &mut self,
1807        suffix: &str,
1808        text: &str,
1809        range: SourceRange,
1810    ) -> (FloatSuffix, bool) {
1811        let lower = suffix.to_ascii_lowercase();
1812        match lower.as_str() {
1813            "" => return (FloatSuffix::None, false),
1814            "f" => return (FloatSuffix::Float, false),
1815            "l" => return (FloatSuffix::LongDouble, false),
1816            _ => {}
1817        }
1818        // GNU's imaginary suffix, on its own or beside `f` or `l`. Both
1819        // spellings are the same thing: `j` is what Fortran and engineering
1820        // habit write, and GCC takes either.
1821        let imaginary = match lower.as_str() {
1822            "i" | "j" => Some(FloatSuffix::None),
1823            "if" | "fi" | "jf" | "fj" => Some(FloatSuffix::Float),
1824            "il" | "li" | "jl" | "lj" => Some(FloatSuffix::LongDouble),
1825            _ => None,
1826        };
1827        if let Some(kind) = imaginary {
1828            if !self.options.complex {
1829                self.error(
1830                    range,
1831                    format!(
1832                        "invalid suffix '{suffix}' on floating constant '{text}': an \
1833                         imaginary constant needs _Complex. {}",
1834                        crate::COMPLEX_UNSUPPORTED
1835                    ),
1836                );
1837                return (FloatSuffix::None, false);
1838            }
1839            if let Some(message) = self
1840                .options
1841                .gating
1842                .requires("an imaginary constant", Standard::C99)
1843            {
1844                self.error(range, message);
1845                return (FloatSuffix::None, false);
1846            }
1847            return (kind, true);
1848        }
1849        // A decimal constant is refused whatever the entry point: it has no
1850        // type this crate can give it.
1851        let refusal = match lower.as_str() {
1852            "df" | "dd" | "dl" => Some(
1853                "the decimal floating types (_Decimal32, _Decimal64, _Decimal128) are not \
1854                 supported: they are radix-10 and no Rust type is",
1855            ),
1856            "f16" | "f16x" | "bf16" => Some(
1857                "'_Float16' is not supported: Rust's `f16` is unstable, and rounding the \
1858                 constant to a wider type would change what the program computes",
1859            ),
1860            _ => None,
1861        };
1862        if let Some(reason) = refusal {
1863            self.error(
1864                range,
1865                format!("invalid suffix '{suffix}' on floating constant '{text}': {reason}"),
1866            );
1867            return (FloatSuffix::None, false);
1868        }
1869        // The rest are the GNU widths. `f32` is `float`; every other one names
1870        // a format this implementation makes a `double`.
1871        let wider = matches!(
1872            lower.as_str(),
1873            "d" | "w" | "q" | "f64" | "f64x" | "f32x" | "f128" | "f128x"
1874        );
1875        if !wider && lower != "f32" {
1876            self.error(
1877                range,
1878                format!("invalid suffix '{suffix}' on floating constant '{text}'"),
1879            );
1880            return (FloatSuffix::None, false);
1881        }
1882        if !self.options.gating.dialect.is_gnu() {
1883            let gnu = self.options.gating.standard.macro_name_in(Dialect::Gnu);
1884            let here = self
1885                .options
1886                .gating
1887                .standard
1888                .macro_name_in(self.options.gating.dialect);
1889            self.error(
1890                range,
1891                format!(
1892                    "the suffix '{suffix}' on a floating constant is a GNU extension, and \
1893                     requires a GNU dialect ({gnu}) (this block is {here})"
1894                ),
1895            );
1896            return (FloatSuffix::None, false);
1897        }
1898        if lower == "f32" {
1899            return (FloatSuffix::Float, false);
1900        }
1901        // `LongDouble` is `double`, which is what all of these come to.
1902        (FloatSuffix::LongDouble, false)
1903    }
1904
1905    // -- character and string constants -------------------------------------
1906
1907    fn scan_char_constant(&mut self, kind: StrKind) -> TokenKind {
1908        let start = self.pos;
1909        debug_assert_eq!(self.peek(), Some(b'\''));
1910        self.pos += 1;
1911        let mut values: Vec<u32> = Vec::new();
1912        let mut terminated = false;
1913        // How many *characters* were written, which is not how many elements
1914        // they came to: one `\U0001F600` is one character and two UTF-16 code
1915        // units, and the two say different things about what is wrong.
1916        let mut characters = 0usize;
1917        while let Some(c) = self.peek() {
1918            if c == b'\'' {
1919                self.pos += 1;
1920                terminated = true;
1921                break;
1922            }
1923            if c == b'\n' {
1924                break;
1925            }
1926            let before = values.len();
1927            self.read_char_element(kind, &mut values);
1928            if values.len() > before {
1929                characters += 1;
1930            }
1931        }
1932        let range = self.range(start, self.pos);
1933        if !terminated {
1934            self.error(range, "missing terminating \' character");
1935        }
1936        if values.is_empty() {
1937            self.error(range, "empty character constant");
1938        }
1939        // Only `'ab'` and `L'ab'` have an implementation-defined meaning; C11
1940        // 6.4.4.4p2 makes more than one character in a `u8`, `u` or `U`
1941        // constant a constraint violation, because there is no room for a
1942        // second one in the type — and so is one character that needs more
1943        // than one code unit, which is what `u8'é'` and `u'😀'` are.
1944        if values.len() > 1 {
1945            match kind {
1946                StrKind::Narrow | StrKind::Wide => {
1947                    self.warning(range, "multi-character character constant");
1948                }
1949                _ if characters > 1 => {
1950                    self.error(
1951                        range,
1952                        format!(
1953                            "a '{}' character constant holds exactly one character",
1954                            kind.prefix()
1955                        ),
1956                    );
1957                }
1958                _ => {
1959                    self.error(
1960                        range,
1961                        format!(
1962                            "the character in a '{}' character constant must fit in a \
1963                             single code unit",
1964                            kind.prefix()
1965                        ),
1966                    );
1967                }
1968            }
1969        }
1970
1971        let value = if kind != StrKind::Narrow {
1972            values.last().copied().unwrap_or(0) as i64
1973        } else if values.len() <= 1 {
1974            values.first().copied().unwrap_or(0) as i64
1975        } else {
1976            // GCC packs the bytes big-endian into an `int`.
1977            let mut v: u32 = 0;
1978            for b in &values {
1979                v = (v << 8) | (*b & 0xff);
1980            }
1981            v as i32 as i64
1982        };
1983
1984        TokenKind::Char(CharLit {
1985            value,
1986            kind,
1987            text: self.text[start..self.pos].to_owned(),
1988        })
1989    }
1990
1991    fn scan_string_literal(&mut self, kind: StrKind) -> TokenKind {
1992        let start = self.pos;
1993        debug_assert_eq!(self.peek(), Some(b'"'));
1994        self.pos += 1;
1995        let mut values: Vec<u32> = Vec::new();
1996        let mut terminated = false;
1997        while let Some(c) = self.peek() {
1998            if c == b'"' {
1999                self.pos += 1;
2000                terminated = true;
2001                break;
2002            }
2003            if c == b'\n' {
2004                break;
2005            }
2006            self.read_char_element(kind, &mut values);
2007        }
2008        let range = self.range(start, self.pos);
2009        if !terminated {
2010            self.error(range, "missing terminating \" character");
2011        }
2012        TokenKind::Str(StrLit {
2013            kind,
2014            values,
2015            text: self.text[start..self.pos].to_owned(),
2016        })
2017    }
2018
2019    /// Reads one element of a character constant or string literal, appending
2020    /// its decoded value(s) to `out`.
2021    fn read_char_element(&mut self, kind: StrKind, out: &mut Vec<u32>) {
2022        if self.is_line_splice(self.pos) {
2023            self.pos += self.line_splice_len(self.pos);
2024            return;
2025        }
2026        let start = self.pos;
2027        // Phase 1 replaces trigraphs inside literals too: `"??!"` is `"|"`,
2028        // and `"??/n"` is `"\n"`.
2029        let trigraph = self.trigraph_at(self.pos);
2030        if trigraph != Some(b'\\') && self.peek() != Some(b'\\') {
2031            match trigraph {
2032                Some(c) => {
2033                    self.pos += 3;
2034                    out.push(u32::from(c));
2035                }
2036                None if kind.is_bytes() => {
2037                    // A narrow or UTF-8 literal keeps the raw
2038                    // execution-charset bytes, so UTF-8 text in one survives
2039                    // byte for byte.
2040                    self.pos += 1;
2041                    out.push(self.bytes[start] as u32);
2042                }
2043                None => {
2044                    let ch = self.text[start..].chars().next().unwrap_or('\u{fffd}');
2045                    self.pos += ch.len_utf8();
2046                    push_character(ch as u32, kind, self.options.wchar_bits, out);
2047                }
2048            }
2049            return;
2050        }
2051
2052        self.pos += self.trigraph_len(self.pos);
2053        let e = match self.trigraph_at(self.pos) {
2054            Some(c) => c,
2055            None => match self.peek() {
2056                Some(c) => c,
2057                None => {
2058                    let range = self.range(start, self.pos);
2059                    self.error(range, "incomplete escape sequence");
2060                    return;
2061                }
2062            },
2063        };
2064        self.pos += self.trigraph_len(self.pos);
2065        let simple = match e {
2066            b'\'' => Some(0x27),
2067            b'"' => Some(0x22),
2068            b'?' => Some(0x3f),
2069            b'\\' => Some(0x5c),
2070            b'a' => Some(0x07),
2071            b'b' => Some(0x08),
2072            // `\e` is GNU's escape for ESC. GCC accepts it in every mode (with
2073            // a pedantic warning), and `"\e[0m"` is how a program writes a
2074            // terminal colour; refusing it would be refusing the extension in
2075            // the one place a strict mode still has it.
2076            b'e' => Some(0x1b),
2077            b'f' => Some(0x0c),
2078            b'n' => Some(0x0a),
2079            b'r' => Some(0x0d),
2080            b't' => Some(0x09),
2081            b'v' => Some(0x0b),
2082            _ => None,
2083        };
2084        if let Some(v) = simple {
2085            out.push(v);
2086            return;
2087        }
2088        match e {
2089            b'0'..=b'7' => {
2090                let mut v: u32 = (e - b'0') as u32;
2091                for _ in 0..2 {
2092                    match self.peek() {
2093                        Some(d @ b'0'..=b'7') => {
2094                            v = v * 8 + (d - b'0') as u32;
2095                            self.pos += 1;
2096                        }
2097                        _ => break,
2098                    }
2099                }
2100                self.push_escape_value(v, kind, start, out);
2101            }
2102            b'x' => {
2103                let mut v: u32 = 0;
2104                let mut any = false;
2105                let mut overflow = false;
2106                while let Some(d) = self.peek().and_then(|c| (c as char).to_digit(16)) {
2107                    any = true;
2108                    v = match v.checked_mul(16).and_then(|v| v.checked_add(d)) {
2109                        Some(v) => v,
2110                        None => {
2111                            overflow = true;
2112                            v
2113                        }
2114                    };
2115                    self.pos += 1;
2116                }
2117                let range = self.range(start, self.pos);
2118                if !any {
2119                    self.error(range, "'\\x' used with no following hex digits");
2120                } else if overflow {
2121                    self.error(range, "hex escape sequence out of range");
2122                }
2123                self.push_escape_value(v, kind, start, out);
2124            }
2125            b'u' | b'U' => {
2126                if let Some(message) = self
2127                    .options
2128                    .gating
2129                    .requires("a universal character name", Standard::C99)
2130                {
2131                    let range = self.range(start, self.pos);
2132                    self.error(range, message);
2133                }
2134                let want = if e == b'u' { 4 } else { 8 };
2135                let mut v: u32 = 0;
2136                let mut count = 0;
2137                while count < want {
2138                    match self.peek().and_then(|c| (c as char).to_digit(16)) {
2139                        Some(d) => {
2140                            v = v.wrapping_mul(16).wrapping_add(d);
2141                            self.pos += 1;
2142                            count += 1;
2143                        }
2144                        None => break,
2145                    }
2146                }
2147                let range = self.range(start, self.pos);
2148                if count != want {
2149                    self.error(
2150                        range,
2151                        format!("incomplete universal character name; expected {want} hex digits"),
2152                    );
2153                    return;
2154                }
2155                match char::from_u32(v) {
2156                    Some(ch) if kind.is_bytes() => {
2157                        // The execution character set is UTF-8.
2158                        let mut buf = [0u8; 4];
2159                        for b in ch.encode_utf8(&mut buf).as_bytes() {
2160                            out.push(*b as u32);
2161                        }
2162                    }
2163                    Some(ch) => push_character(ch as u32, kind, self.options.wchar_bits, out),
2164                    None => {
2165                        // Outside Unicode, or a surrogate: 6.4.3p2 again, and
2166                        // the *literal token's* spelling is what is wrong, so
2167                        // this stands wherever the token ends up.
2168                        self.lexical_error(
2169                            range,
2170                            format!("'\\u{v:04X}' is not a valid universal character name"),
2171                        );
2172                    }
2173                }
2174            }
2175            _ => {
2176                let range = self.range(start, self.pos);
2177                self.error(
2178                    range,
2179                    format!("unknown escape sequence '\\{}'", (e as char).escape_debug()),
2180                );
2181                out.push(e as u32);
2182            }
2183        }
2184    }
2185
2186    /// Pushes the value of a numeric escape sequence, which unlike a character
2187    /// is *not* re-encoded: `u"\xd83d"` is that one code unit.
2188    fn push_escape_value(&mut self, v: u32, kind: StrKind, start: usize, out: &mut Vec<u32>) {
2189        let max = kind.max_element(self.options.wchar_bits);
2190        if v > max {
2191            let range = self.range(start, self.pos);
2192            let ty = match kind {
2193                StrKind::Narrow => "char",
2194                StrKind::Utf8 => "char8_t",
2195                StrKind::Utf16 => "char16_t",
2196                StrKind::Utf32 => "char32_t",
2197                StrKind::Wide => "wchar_t",
2198            };
2199            self.error(
2200                range,
2201                format!("escape sequence out of range for type '{ty}'"),
2202            );
2203            out.push(v & max);
2204        } else {
2205            out.push(v);
2206        }
2207    }
2208}
2209
2210/// Appends one character, encoded the way `kind` stores its elements.
2211///
2212/// A `u"…"` literal holds UTF-16 code units, so a character outside the basic
2213/// multilingual plane becomes the two halves of a surrogate pair — which is
2214/// what makes `sizeof(u"\U0001F600")` six rather than four. On a target whose
2215/// `wchar_t` is 16 bits wide, `L"…"` is UTF-16 too and does the same.
2216fn push_character(value: u32, kind: StrKind, wchar_bits: u32, out: &mut Vec<u32>) {
2217    if !kind.is_utf16(wchar_bits) || value <= 0xffff {
2218        out.push(value);
2219        return;
2220    }
2221    let v = value - 0x1_0000;
2222    out.push(0xd800 + (v >> 10));
2223    out.push(0xdc00 + (v & 0x3ff));
2224}
2225
2226fn is_ident_start(c: u8, dollar: bool) -> bool {
2227    c.is_ascii_alphabetic() || c == b'_' || (dollar && c == b'$')
2228}
2229
2230fn is_ident_continue(c: u8, dollar: bool) -> bool {
2231    c.is_ascii_alphanumeric() || c == b'_' || (dollar && c == b'$')
2232}
2233
2234/// Whether an extended character may appear in an identifier.
2235///
2236/// C99 Annex D listed the ranges by hand, C11 revised the list, and C23 (N2836,
2237/// N2939) replaced all of it with Unicode Annex #31's `XID_Start` and
2238/// `XID_Continue` — which is also what Rust's own identifiers are, and what
2239/// makes a C name usable as a Rust one. The one list is used in every entry
2240/// point: the earlier annexes are approximations of the same intent, and a
2241/// program that uses a character C11 left out is one this would otherwise
2242/// refuse for no reason a user could act on.
2243///
2244/// A character of the basic character set is never one of these: the ASCII
2245/// path has already decided about it, and a universal character name is not
2246/// allowed to spell one (6.4.3p2).
2247fn is_extended_ident_char(ch: char, start: bool) -> bool {
2248    if ch.is_ascii() {
2249        return false;
2250    }
2251    if start {
2252        unicode_ident::is_xid_start(ch)
2253    } else {
2254        unicode_ident::is_xid_continue(ch)
2255    }
2256}
2257
2258/// Validates an integer suffix, returning `(unsigned, long_kind)`.
2259fn parse_int_suffix(s: &str) -> Option<(bool, LongKind)> {
2260    if s.is_empty() {
2261        return Some((false, LongKind::None));
2262    }
2263    let b = s.as_bytes();
2264    let mut i = 0;
2265    let mut unsigned = false;
2266    let mut long = LongKind::None;
2267
2268    if b[i] == b'u' || b[i] == b'U' {
2269        unsigned = true;
2270        i += 1;
2271    }
2272    if i < b.len() && (b[i] == b'l' || b[i] == b'L') {
2273        // `ll` and `LL` must not be mixed as `lL` or `Ll`.
2274        if i + 1 < b.len() && b[i + 1] == b[i] {
2275            long = LongKind::LongLong;
2276            i += 2;
2277        } else {
2278            long = LongKind::Long;
2279            i += 1;
2280        }
2281    }
2282    if !unsigned && i < b.len() && (b[i] == b'u' || b[i] == b'U') {
2283        unsigned = true;
2284        i += 1;
2285    }
2286    (i == b.len()).then_some((unsigned, long))
2287}
2288
2289/// Splits a floating constant into its numeric body and its suffix.
2290///
2291/// The *body* is scanned forward rather than the suffix backwards, because a
2292/// suffix may hold digits of its own: `1.0f128` names `_Float128` and the
2293/// `128` is no part of the number. What is left after the digits, the point
2294/// and the exponent is the suffix, whatever it looks like; naming it is
2295/// [`Lexer::float_suffix`]'s business.
2296fn split_float_suffix(text: &str, hex: bool) -> (&str, &str) {
2297    let b = text.as_bytes();
2298    let mut i = 0;
2299    let (exponent, digit): (u8, fn(u8) -> bool) = if hex {
2300        i = 2; // the `0x` the caller has already recognised
2301        (b'p', |c| c.is_ascii_hexdigit())
2302    } else {
2303        (b'e', |c| c.is_ascii_digit())
2304    };
2305    while i < b.len() && (digit(b[i]) || b[i] == b'.') {
2306        i += 1;
2307    }
2308    if i < b.len() && b[i] | 0x20 == exponent {
2309        i += 1;
2310        if i < b.len() && (b[i] == b'+' || b[i] == b'-') {
2311            i += 1;
2312        }
2313        while i < b.len() && b[i].is_ascii_digit() {
2314            i += 1;
2315        }
2316    }
2317    (&text[..i], &text[i..])
2318}
2319
2320/// Parses the numeric body of a decimal floating constant.
2321fn parse_decimal_float(body: &str) -> Option<f64> {
2322    if body.is_empty() {
2323        return None;
2324    }
2325    // `None` is "no exponent at all", which is a different thing from an `e`
2326    // with nothing after it — `1e` is not a constant.
2327    let (mantissa, exponent) = match body.find(['e', 'E']) {
2328        Some(i) => (&body[..i], Some(&body[i + 1..])),
2329        None => (body, None),
2330    };
2331    let (int_part, frac_part) = match mantissa.find('.') {
2332        Some(i) => (&mantissa[..i], &mantissa[i + 1..]),
2333        None => (mantissa, ""),
2334    };
2335    if int_part.is_empty() && frac_part.is_empty() {
2336        return None;
2337    }
2338    if !int_part.bytes().all(|c| c.is_ascii_digit())
2339        || !frac_part.bytes().all(|c| c.is_ascii_digit())
2340    {
2341        return None;
2342    }
2343    let exponent = match exponent {
2344        None => 0i32,
2345        Some(exponent) => {
2346            let (sign, digits) = match exponent.as_bytes().first() {
2347                Some(b'+') => (1, &exponent[1..]),
2348                Some(b'-') => (-1, &exponent[1..]),
2349                _ => (1, exponent),
2350            };
2351            if digits.is_empty() || !digits.bytes().all(|c| c.is_ascii_digit()) {
2352                return None;
2353            }
2354            // Saturate: an absurd exponent simply becomes 0 or infinity.
2355            sign * digits.parse::<i32>().unwrap_or(i32::MAX / 2)
2356        }
2357    };
2358    let normalized = format!(
2359        "{}.{}e{}",
2360        if int_part.is_empty() { "0" } else { int_part },
2361        if frac_part.is_empty() { "0" } else { frac_part },
2362        exponent
2363    );
2364    normalized.parse::<f64>().ok()
2365}
2366
2367/// Parses the numeric body of a hexadecimal floating constant (`0x1.8p3`).
2368fn parse_hex_float(body: &str) -> Option<f64> {
2369    let rest = body
2370        .strip_prefix("0x")
2371        .or_else(|| body.strip_prefix("0X"))?;
2372    let p = rest.find(['p', 'P'])?;
2373    let (mantissa, exponent) = (&rest[..p], &rest[p + 1..]);
2374    let (int_part, frac_part) = match mantissa.find('.') {
2375        Some(i) => (&mantissa[..i], &mantissa[i + 1..]),
2376        None => (mantissa, ""),
2377    };
2378    if int_part.is_empty() && frac_part.is_empty() {
2379        return None;
2380    }
2381    let mut value = 0f64;
2382    for c in int_part.chars() {
2383        value = value * 16.0 + c.to_digit(16)? as f64;
2384    }
2385    let mut scale = 1.0 / 16.0;
2386    for c in frac_part.chars() {
2387        value += c.to_digit(16)? as f64 * scale;
2388        scale /= 16.0;
2389    }
2390    let (sign, digits) = match exponent.as_bytes().first() {
2391        Some(b'+') => (1i32, &exponent[1..]),
2392        Some(b'-') => (-1i32, &exponent[1..]),
2393        _ => (1i32, exponent),
2394    };
2395    if digits.is_empty() || !digits.bytes().all(|c| c.is_ascii_digit()) {
2396        return None;
2397    }
2398    let exp = sign * digits.parse::<i32>().unwrap_or(i32::MAX / 2);
2399    Some(value * 2f64.powi(exp))
2400}