Skip to main content

cinrs_core/
lex.rs

1//! A complete C99 lexer working on the captured source text.
2//!
3//! The lexer is byte-offset based: every [`Token`] carries a
4//! [`SourceRange`] into the [`SourceMap`](crate::capture::SourceMap), which is
5//! what makes exact diagnostics possible. Constants are decoded here (value,
6//! base, suffix, escape sequences) so that neither the parser nor sema has to
7//! look at raw text again.
8//!
9//! Two flags on each token — [`Token::bol`] and [`Token::preceded_by_space`] —
10//! are what the [preprocessor](crate::pp) recognises directives with and what
11//! its stringification operator reproduces: directive recognition needs "first
12//! token on a line", and `#` needs to know where whitespace was.
13//!
14//! # The lexer never reports anything
15//!
16//! Lexical errors never stop the lexer, and they never reach [`Diagnostics`]
17//! either: each one is attached to the token it was found in, in
18//! [`Token::errors`], and the preprocessor reports the ones whose token
19//! survives into its output. That is not a detail — a group skipped by
20//! `#if 0` may legally hold text that is not C at all, and a macro that is
21//! never invoked may hold anything its author liked:
22//!
23//! ```c
24//! #if 0
25//! this is not C: 08, 'unterminated, @@@
26//! #endif
27//! ```
28//!
29//! Text that is not a token at all becomes a [`TokenKind::Error`] token
30//! carrying its own spelling, so that it too can be skipped rather than
31//! reported. The preprocessor drops those tokens after reporting them, so
32//! nothing downstream ever sees one.
33//!
34//! [`Diagnostics`]: crate::diag::Diagnostics
35
36use crate::capture::{Pos, Source, SourceRange};
37use crate::diag::Diagnostic;
38use crate::{Dialect, Options, Standard};
39
40/// A C99 keyword.
41#[derive(Clone, Copy, PartialEq, Eq, Debug, Hash)]
42#[allow(missing_docs)]
43pub enum Keyword {
44    Auto,
45    Break,
46    Case,
47    Char,
48    Const,
49    Continue,
50    Default,
51    Do,
52    Double,
53    Else,
54    Enum,
55    Extern,
56    Float,
57    For,
58    Goto,
59    If,
60    Inline,
61    Int,
62    Long,
63    Register,
64    Restrict,
65    Return,
66    Short,
67    Signed,
68    Sizeof,
69    Static,
70    Struct,
71    Switch,
72    Typedef,
73    Union,
74    Unsigned,
75    Void,
76    Volatile,
77    While,
78    Bool,
79    Complex,
80    Imaginary,
81    // C11. Every one of these is spelled with a leading underscore, which C99
82    // already reserves, so they are recognised in every mode and the parser
83    // reports the ones a C99 block may not use — a friendlier answer than
84    // "expected a declaration, found identifier '_Static_assert'".
85    Alignas,
86    Alignof,
87    Atomic,
88    Generic,
89    Noreturn,
90    StaticAssert,
91    ThreadLocal,
92    BitInt,
93    // C23. These are ordinary identifiers before C23 — `<stdbool.h>` writes
94    // `#define bool _Bool`, and a C99 program may have a variable called
95    // `typeof` — so they are keywords only in a `c23!` block.
96    BoolName,
97    True,
98    False,
99    Nullptr,
100    Typeof,
101    TypeofUnqual,
102    Constexpr,
103    StaticAssertName,
104    AlignofName,
105    AlignasName,
106    ThreadLocalName,
107    // The GNU keywords. Every one of them is spelled with a leading double
108    // underscore, which C reserves, so they are available in every entry point
109    // — exactly as they are in GCC's own strict modes. They are *not* produced
110    // by [`Keyword::from_str`]: the [preprocessor](crate::pp) turns the
111    // identifiers into them on the way out, after macro replacement, so that
112    // `#define __attribute__(x)` — which portability headers really do write —
113    // still defines and expands a macro of that name.
114    /// `__attribute__`, `__attribute`
115    Attribute,
116    /// `__extension__`
117    Extension,
118    /// `__alignof__`, `__alignof`
119    AlignofGnu,
120    /// `__typeof__`, `__typeof`, and `typeof` in a GNU dialect
121    TypeofGnu,
122    /// `__typeof_unqual__`
123    TypeofUnqualGnu,
124    /// `__asm__`, `__asm`, and `asm` in a GNU dialect
125    Asm,
126    /// `__label__`
127    Label,
128    /// `__auto_type`
129    AutoType,
130    /// `__thread`
131    ThreadGnu,
132    /// `__int128`
133    ///
134    /// A type specifier of its own, which `signed` and `unsigned` combine
135    /// with; the `__int128_t` and `__uint128_t` spellings are `typedef` names
136    /// the compiler owns rather than keywords, exactly as they are in GCC.
137    Int128,
138    /// `__real__`, `__real`
139    RealGnu,
140    /// `__imag__`, `__imag`
141    ImagGnu,
142    /// `__inline__`, `__inline`
143    ///
144    /// A variant of its own because the plain spelling is C99's and `c89!`
145    /// gates it, while this one — being reserved — is available everywhere.
146    InlineGnu,
147    /// `__restrict__`, `__restrict`; see [`Keyword::InlineGnu`].
148    RestrictGnu,
149}
150
151impl Keyword {
152    /// The spelling of this keyword in C source.
153    pub fn as_str(self) -> &'static str {
154        use Keyword::*;
155        match self {
156            Auto => "auto",
157            Break => "break",
158            Case => "case",
159            Char => "char",
160            Const => "const",
161            Continue => "continue",
162            Default => "default",
163            Do => "do",
164            Double => "double",
165            Else => "else",
166            Enum => "enum",
167            Extern => "extern",
168            Float => "float",
169            For => "for",
170            Goto => "goto",
171            If => "if",
172            Inline => "inline",
173            Int => "int",
174            Long => "long",
175            Register => "register",
176            Restrict => "restrict",
177            Return => "return",
178            Short => "short",
179            Signed => "signed",
180            Sizeof => "sizeof",
181            Static => "static",
182            Struct => "struct",
183            Switch => "switch",
184            Typedef => "typedef",
185            Union => "union",
186            Unsigned => "unsigned",
187            Void => "void",
188            Volatile => "volatile",
189            While => "while",
190            Bool => "_Bool",
191            Complex => "_Complex",
192            Imaginary => "_Imaginary",
193            Alignas => "_Alignas",
194            Alignof => "_Alignof",
195            Atomic => "_Atomic",
196            Generic => "_Generic",
197            Noreturn => "_Noreturn",
198            StaticAssert => "_Static_assert",
199            ThreadLocal => "_Thread_local",
200            BitInt => "_BitInt",
201            BoolName => "bool",
202            True => "true",
203            False => "false",
204            Nullptr => "nullptr",
205            Typeof => "typeof",
206            TypeofUnqual => "typeof_unqual",
207            Constexpr => "constexpr",
208            StaticAssertName => "static_assert",
209            AlignofName => "alignof",
210            AlignasName => "alignas",
211            ThreadLocalName => "thread_local",
212            Attribute => "__attribute__",
213            Extension => "__extension__",
214            AlignofGnu => "__alignof__",
215            TypeofGnu => "__typeof__",
216            TypeofUnqualGnu => "__typeof_unqual__",
217            Asm => "__asm__",
218            Label => "__label__",
219            AutoType => "__auto_type",
220            ThreadGnu => "__thread",
221            Int128 => "__int128",
222            RealGnu => "__real__",
223            ImagGnu => "__imag__",
224            InlineGnu => "__inline__",
225            RestrictGnu => "__restrict__",
226        }
227    }
228
229    /// Whether this keyword is one of the GNU spellings the preprocessor
230    /// introduces; see the variants' own documentation.
231    pub fn is_gnu(self) -> bool {
232        use Keyword::*;
233        matches!(
234            self,
235            Attribute
236                | Extension
237                | AlignofGnu
238                | TypeofGnu
239                | TypeofUnqualGnu
240                | Asm
241                | Label
242                | AutoType
243                | ThreadGnu
244                | Int128
245                | RealGnu
246                | ImagGnu
247                | InlineGnu
248                | RestrictGnu
249        )
250    }
251
252    /// The revision that made this spelling a keyword.
253    ///
254    /// The C11 keywords are recognised in every mode — they are reserved
255    /// identifiers in C99, so nothing legal can be broken by it, and the
256    /// parser's "requires C11 or later" is a better answer than a syntax
257    /// error. The C23 ones are *not*: they are ordinary identifiers before
258    /// C23, and `<stdbool.h>`'s `#define bool _Bool` depends on it.
259    ///
260    /// The four C99 added — `inline`, `restrict`, `_Bool` and `_Complex` (with
261    /// `_Imaginary` beside it) — are recognised in every mode for the same
262    /// reason the C11 ones are, and gated where they are parsed; everything
263    /// else has been a keyword since C89.
264    pub fn since(self) -> Standard {
265        use Keyword::*;
266        match self {
267            Alignas | Alignof | Atomic | Generic | Noreturn | StaticAssert | ThreadLocal => {
268                Standard::C11
269            }
270            BitInt | BoolName | True | False | Nullptr | Typeof | TypeofUnqual | Constexpr
271            | StaticAssertName | AlignofName | AlignasName | ThreadLocalName => Standard::C23,
272            Inline | Restrict | Bool | Complex | Imaginary => Standard::C99,
273            _ => Standard::C89,
274        }
275    }
276
277    /// Looks a keyword up by spelling, honouring the language standard.
278    pub fn from_str(s: &str, standard: Standard) -> Option<Keyword> {
279        use Keyword::*;
280        if standard >= Standard::C23 {
281            let c23 = match s {
282                "bool" => Some(BoolName),
283                "true" => Some(True),
284                "false" => Some(False),
285                "nullptr" => Some(Nullptr),
286                "typeof" => Some(Typeof),
287                "typeof_unqual" => Some(TypeofUnqual),
288                "constexpr" => Some(Constexpr),
289                "static_assert" => Some(StaticAssertName),
290                "alignof" => Some(AlignofName),
291                "alignas" => Some(AlignasName),
292                "thread_local" => Some(ThreadLocalName),
293                _ => None,
294            };
295            if c23.is_some() {
296                return c23;
297            }
298        }
299        Some(match s {
300            "_Alignas" => Alignas,
301            "_Alignof" => Alignof,
302            "_Atomic" => Atomic,
303            "_Generic" => Generic,
304            "_Noreturn" => Noreturn,
305            "_Static_assert" => StaticAssert,
306            "_Thread_local" => ThreadLocal,
307            "_BitInt" => BitInt,
308            "auto" => Auto,
309            "break" => Break,
310            "case" => Case,
311            "char" => Char,
312            "const" => Const,
313            "continue" => Continue,
314            "default" => Default,
315            "do" => Do,
316            "double" => Double,
317            "else" => Else,
318            "enum" => Enum,
319            "extern" => Extern,
320            "float" => Float,
321            "for" => For,
322            "goto" => Goto,
323            "if" => If,
324            "inline" => Inline,
325            "int" => Int,
326            "long" => Long,
327            "register" => Register,
328            "restrict" => Restrict,
329            "return" => Return,
330            "short" => Short,
331            "signed" => Signed,
332            "sizeof" => Sizeof,
333            "static" => Static,
334            "struct" => Struct,
335            "switch" => Switch,
336            "typedef" => Typedef,
337            "union" => Union,
338            "unsigned" => Unsigned,
339            "void" => Void,
340            "volatile" => Volatile,
341            "while" => While,
342            "_Bool" => Bool,
343            "_Complex" => Complex,
344            "_Imaginary" => Imaginary,
345            _ => return None,
346        })
347    }
348}
349
350/// A C99 punctuator.
351///
352/// Digraphs are folded into the token they stand for: `<:` lexes as
353/// [`Punct::LBracket`], `%:%:` as [`Punct::HashHash`], and so on.
354#[derive(Clone, Copy, PartialEq, Eq, Debug, Hash)]
355#[allow(missing_docs)]
356pub enum Punct {
357    LBracket,
358    RBracket,
359    LParen,
360    RParen,
361    LBrace,
362    RBrace,
363    Dot,
364    Arrow,
365    PlusPlus,
366    MinusMinus,
367    Amp,
368    Star,
369    Plus,
370    Minus,
371    Tilde,
372    Bang,
373    Slash,
374    Percent,
375    Shl,
376    Shr,
377    Lt,
378    Gt,
379    Le,
380    Ge,
381    EqEq,
382    Ne,
383    Caret,
384    Pipe,
385    AmpAmp,
386    PipePipe,
387    Question,
388    Colon,
389    Semi,
390    Ellipsis,
391    Assign,
392    StarAssign,
393    SlashAssign,
394    PercentAssign,
395    PlusAssign,
396    MinusAssign,
397    ShlAssign,
398    ShrAssign,
399    AmpAssign,
400    CaretAssign,
401    PipeAssign,
402    Comma,
403    Hash,
404    HashHash,
405}
406
407impl Punct {
408    /// The canonical spelling of this punctuator (never the digraph form).
409    pub fn as_str(self) -> &'static str {
410        use Punct::*;
411        match self {
412            LBracket => "[",
413            RBracket => "]",
414            LParen => "(",
415            RParen => ")",
416            LBrace => "{",
417            RBrace => "}",
418            Dot => ".",
419            Arrow => "->",
420            PlusPlus => "++",
421            MinusMinus => "--",
422            Amp => "&",
423            Star => "*",
424            Plus => "+",
425            Minus => "-",
426            Tilde => "~",
427            Bang => "!",
428            Slash => "/",
429            Percent => "%",
430            Shl => "<<",
431            Shr => ">>",
432            Lt => "<",
433            Gt => ">",
434            Le => "<=",
435            Ge => ">=",
436            EqEq => "==",
437            Ne => "!=",
438            Caret => "^",
439            Pipe => "|",
440            AmpAmp => "&&",
441            PipePipe => "||",
442            Question => "?",
443            Colon => ":",
444            Semi => ";",
445            Ellipsis => "...",
446            Assign => "=",
447            StarAssign => "*=",
448            SlashAssign => "/=",
449            PercentAssign => "%=",
450            PlusAssign => "+=",
451            MinusAssign => "-=",
452            ShlAssign => "<<=",
453            ShrAssign => ">>=",
454            AmpAssign => "&=",
455            CaretAssign => "^=",
456            PipeAssign => "|=",
457            Comma => ",",
458            Hash => "#",
459            HashHash => "##",
460        }
461    }
462}
463
464/// Punctuator spellings, longest first so that a greedy match is also the
465/// maximal munch the standard asks for.
466const PUNCTUATORS: &[(&str, Punct)] = &[
467    ("%:%:", Punct::HashHash),
468    ("...", Punct::Ellipsis),
469    ("<<=", Punct::ShlAssign),
470    (">>=", Punct::ShrAssign),
471    ("->", Punct::Arrow),
472    ("++", Punct::PlusPlus),
473    ("--", Punct::MinusMinus),
474    ("<<", Punct::Shl),
475    (">>", Punct::Shr),
476    ("<=", Punct::Le),
477    (">=", Punct::Ge),
478    ("==", Punct::EqEq),
479    ("!=", Punct::Ne),
480    ("&&", Punct::AmpAmp),
481    ("||", Punct::PipePipe),
482    ("*=", Punct::StarAssign),
483    ("/=", Punct::SlashAssign),
484    ("%=", Punct::PercentAssign),
485    ("+=", Punct::PlusAssign),
486    ("-=", Punct::MinusAssign),
487    ("&=", Punct::AmpAssign),
488    ("^=", Punct::CaretAssign),
489    ("|=", Punct::PipeAssign),
490    ("##", Punct::HashHash),
491    ("<:", Punct::LBracket),
492    (":>", Punct::RBracket),
493    ("<%", Punct::LBrace),
494    ("%>", Punct::RBrace),
495    ("%:", Punct::Hash),
496    ("[", Punct::LBracket),
497    ("]", Punct::RBracket),
498    ("(", Punct::LParen),
499    (")", Punct::RParen),
500    ("{", Punct::LBrace),
501    ("}", Punct::RBrace),
502    (".", Punct::Dot),
503    ("&", Punct::Amp),
504    ("*", Punct::Star),
505    ("+", Punct::Plus),
506    ("-", Punct::Minus),
507    ("~", Punct::Tilde),
508    ("!", Punct::Bang),
509    ("/", Punct::Slash),
510    ("%", Punct::Percent),
511    ("<", Punct::Lt),
512    (">", Punct::Gt),
513    ("^", Punct::Caret),
514    ("|", Punct::Pipe),
515    ("?", Punct::Question),
516    (":", Punct::Colon),
517    (";", Punct::Semi),
518    ("=", Punct::Assign),
519    (",", Punct::Comma),
520    ("#", Punct::Hash),
521];
522
523/// The `l`/`ll` part of an integer suffix.
524#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
525pub enum LongKind {
526    /// No `l` suffix.
527    #[default]
528    None,
529    /// `l` or `L`.
530    Long,
531    /// `ll` or `LL`.
532    LongLong,
533}
534
535/// The base an integer constant was written in.
536#[derive(Clone, Copy, PartialEq, Eq, Debug)]
537pub enum NumBase {
538    /// `0b101` — C23.
539    Binary,
540    /// `0777`
541    Octal,
542    /// `42`
543    Decimal,
544    /// `0x2a`
545    Hex,
546}
547
548impl NumBase {
549    /// The radix.
550    pub fn radix(self) -> u32 {
551        match self {
552            NumBase::Binary => 2,
553            NumBase::Octal => 8,
554            NumBase::Decimal => 10,
555            NumBase::Hex => 16,
556        }
557    }
558
559    /// How a diagnostic names this base.
560    pub fn as_str(self) -> &'static str {
561        match self {
562            NumBase::Binary => "binary",
563            NumBase::Octal => "octal",
564            NumBase::Decimal => "decimal",
565            NumBase::Hex => "hexadecimal",
566        }
567    }
568}
569
570/// A decoded integer constant.
571#[derive(Clone, PartialEq, Eq, Debug)]
572pub struct IntLit {
573    /// The value, before any type is chosen for it.
574    pub value: u128,
575    /// How it was written.
576    pub base: NumBase,
577    /// Whether a `u`/`U` suffix was present.
578    pub unsigned: bool,
579    /// Whether an `l`/`ll` suffix was present.
580    pub long: LongKind,
581    /// The exact source spelling.
582    pub text: String,
583}
584
585/// The suffix of a floating constant.
586#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
587pub enum FloatSuffix {
588    /// No suffix: the constant has type `double`.
589    #[default]
590    None,
591    /// `f`/`F`: `float`.
592    Float,
593    /// `l`/`L`: `long double`.
594    LongDouble,
595}
596
597/// A decoded floating constant.
598#[derive(Clone, PartialEq, Debug)]
599pub struct FloatLit {
600    /// The value, rounded to `f64`.
601    pub value: f64,
602    /// The suffix, which fixes the constant's type.
603    pub suffix: FloatSuffix,
604    /// Whether the constant carried GNU's imaginary suffix, `i` or `j`.
605    ///
606    /// `2.0i` is `(0, 2)` of the complex type [`FloatLit::suffix`] names the
607    /// real part of, which is what makes `1.0 + 2.0i` read the way every C
608    /// program writes a complex constant. It is a GNU extension that C11
609    /// blessed by giving `<complex.h>` a `CMPLX` built on it; the *standard*
610    /// spelling of the same thing is `_Complex_I`.
611    pub imaginary: bool,
612    /// Whether the constant was written in hexadecimal form.
613    pub hex: bool,
614    /// The exact source spelling.
615    pub text: String,
616}
617
618/// A decoded character constant.
619#[derive(Clone, PartialEq, Eq, Debug)]
620pub struct CharLit {
621    /// The value of the constant.
622    ///
623    /// For a plain single-character constant this is the unsigned value of the
624    /// execution character (so `'\xff'` is `255`); whether that is later
625    /// interpreted as `-1` is up to sema, which knows the signedness of
626    /// `char`. Multi-character constants are packed big-endian, as GCC does.
627    pub value: i64,
628    /// Which prefix the constant was written with, which is what fixes its
629    /// type: `'x'` is an `int`, `L'x'` a `wchar_t`, `u'x'` a `char16_t`,
630    /// `U'x'` a `char32_t` and `u8'x'` a `char8_t`.
631    pub kind: StrKind,
632    /// The exact source spelling, including quotes.
633    pub text: String,
634}
635
636/// Which prefix a character constant or string literal was written with.
637///
638/// The five of them are the same set for both, which is why one enum serves
639/// both: `'x'`/`"…"`, `u8'x'`/`u8"…"`, `u'x'`/`u"…"`, `U'x'`/`U"…"` and
640/// `L'x'`/`L"…"`.
641#[derive(Clone, Copy, PartialEq, Eq, Debug)]
642pub enum StrKind {
643    /// `"…"` — bytes of the execution character set, which is UTF-8.
644    Narrow,
645    /// `u8"…"` (C11) and `u8'x'` (C23) — UTF-8 bytes, of type `char` before
646    /// C23 and `char8_t` (an `unsigned char`) from C23 on.
647    Utf8,
648    /// `u"…"` and `u'x'` (C11) — UTF-16 code units, of type `char16_t`.
649    Utf16,
650    /// `U"…"` and `U'x'` (C11) — UTF-32 code units, of type `char32_t`.
651    Utf32,
652    /// `L"…"` — `wchar_t`.
653    Wide,
654}
655
656impl StrKind {
657    /// The prefix a literal of this kind is written with.
658    pub fn prefix(self) -> &'static str {
659        match self {
660            StrKind::Narrow => "",
661            StrKind::Utf8 => "u8",
662            StrKind::Utf16 => "u",
663            StrKind::Utf32 => "U",
664            StrKind::Wide => "L",
665        }
666    }
667
668    /// The revision that introduced the prefix, in the form the constant is
669    /// being used in.
670    ///
671    /// `u8"…"` is C11 (N1488) and `u8'x'` is C23 (N2418); the other three
672    /// prefixes are the same revision either way.
673    pub fn since(self, character: bool) -> Standard {
674        match self {
675            StrKind::Narrow | StrKind::Wide => Standard::C89,
676            StrKind::Utf8 if character => Standard::C23,
677            StrKind::Utf8 | StrKind::Utf16 | StrKind::Utf32 => Standard::C11,
678        }
679    }
680
681    /// Whether the elements are the bytes of the source's UTF-8, rather than
682    /// character values.
683    fn is_bytes(self) -> bool {
684        matches!(self, StrKind::Narrow | StrKind::Utf8)
685    }
686
687    /// The largest value one element of this kind can hold.
688    ///
689    /// `wide_bits` is how wide the target's `wchar_t` is, which is the one
690    /// answer here that is not fixed by the language: 32 bits on the Unix
691    /// platforms and 16 on Windows, where `L"…"` is UTF-16 and a character
692    /// outside the basic multilingual plane takes a surrogate pair, exactly as
693    /// `u"…"` does.
694    fn max_element(self, wide_bits: u32) -> u32 {
695        match self {
696            StrKind::Narrow | StrKind::Utf8 => 0xff,
697            StrKind::Utf16 => 0xffff,
698            StrKind::Wide if wide_bits <= 16 => 0xffff,
699            StrKind::Utf32 | StrKind::Wide => u32::MAX,
700        }
701    }
702
703    /// Whether one element holds a UTF-16 code unit, so that a character
704    /// beyond the basic multilingual plane becomes a surrogate pair.
705    fn is_utf16(self, wide_bits: u32) -> bool {
706        self == StrKind::Utf16 || (self == StrKind::Wide && wide_bits <= 16)
707    }
708}
709
710/// A decoded string literal.
711#[derive(Clone, PartialEq, Eq, Debug)]
712pub struct StrLit {
713    /// Which prefix it was written with.
714    pub kind: StrKind,
715    /// The decoded elements: bytes (`0..=255`) for a narrow or UTF-8 literal,
716    /// UTF-16 code units (surrogate pairs and all) for a `u"…"` one, and
717    /// character values for `U"…"` and `L"…"`. The terminating NUL is *not*
718    /// included.
719    pub values: Vec<u32>,
720    /// The exact source spelling, including quotes.
721    pub text: String,
722}
723
724impl StrLit {
725    /// The literal's bytes, if it is an ordinary narrow one.
726    pub fn as_bytes(&self) -> Option<Vec<u8>> {
727        (self.kind == StrKind::Narrow).then(|| self.values.iter().map(|v| *v as u8).collect())
728    }
729}
730
731/// What a [`Token`] is.
732#[derive(Clone, PartialEq, Debug)]
733pub enum TokenKind {
734    /// End of the token list. Always present, exactly once, last.
735    Eof,
736    /// An identifier that is not a keyword.
737    Ident(String),
738    /// A keyword.
739    Keyword(Keyword),
740    /// An integer constant.
741    Int(IntLit),
742    /// A floating constant.
743    Float(FloatLit),
744    /// A character constant.
745    Char(CharLit),
746    /// A string literal.
747    Str(StrLit),
748    /// A punctuator.
749    Punct(Punct),
750    /// Text that is not a C token at all, kept so that a skipped group may
751    /// contain it. Holds its own spelling.
752    Error(String),
753}
754
755impl TokenKind {
756    /// A short description used in "expected …, found …" messages.
757    pub fn describe(&self) -> String {
758        match self {
759            TokenKind::Eof => "end of input".to_owned(),
760            TokenKind::Ident(name) => format!("identifier '{name}'"),
761            TokenKind::Keyword(k) => format!("keyword '{}'", k.as_str()),
762            TokenKind::Int(_) => "integer constant".to_owned(),
763            TokenKind::Float(_) => "floating constant".to_owned(),
764            TokenKind::Char(_) => "character constant".to_owned(),
765            TokenKind::Str(_) => "string literal".to_owned(),
766            TokenKind::Punct(p) => format!("'{}'", p.as_str()),
767            TokenKind::Error(text) => format!("'{text}'"),
768        }
769    }
770
771    /// The exact source spelling of this token.
772    ///
773    /// This is what `#` stringifies and what `##` pastes, so it has to be the
774    /// text as written — `0x1f` rather than `31`, `'\n'` rather than a
775    /// newline. [`TokenKind::Eof`] has no spelling and gives `""`.
776    pub fn spelling(&self) -> &str {
777        match self {
778            TokenKind::Eof => "",
779            TokenKind::Ident(name) => name,
780            TokenKind::Keyword(k) => k.as_str(),
781            TokenKind::Int(lit) => &lit.text,
782            TokenKind::Float(lit) => &lit.text,
783            TokenKind::Char(lit) => &lit.text,
784            TokenKind::Str(lit) => &lit.text,
785            TokenKind::Punct(p) => p.as_str(),
786            TokenKind::Error(text) => text,
787        }
788    }
789
790    /// The name this token has when it is used as a macro name.
791    ///
792    /// Keywords are ordinary identifiers during translation phase 4 — our
793    /// lexer classifies them early, so `#define restrict` and `#ifdef inline`
794    /// would otherwise be unusable — which is why a keyword answers with its
795    /// spelling here.
796    pub fn macro_name(&self) -> Option<&str> {
797        match self {
798            TokenKind::Ident(name) => Some(name),
799            TokenKind::Keyword(k) => Some(k.as_str()),
800            _ => None,
801        }
802    }
803}
804
805/// A lexed C token.
806#[derive(Clone, PartialEq, Debug)]
807pub struct Token {
808    /// What the token is.
809    pub kind: TokenKind,
810    /// Where it is.
811    pub range: SourceRange,
812    /// Whether it is the first token on its logical line (needed by the
813    /// preprocessor to spot directives).
814    pub bol: bool,
815    /// Whether whitespace or a comment preceded it (needed by the
816    /// preprocessor's stringification and macro replacement).
817    pub preceded_by_space: bool,
818    /// Everything wrong with this token, and with the text between it and the
819    /// token before it.
820    ///
821    /// The [preprocessor](crate::pp) reports what is wrong with the token
822    /// *itself* if and only if the token reaches its output. What stands
823    /// whatever becomes of the token — its spelling, and the comments that
824    /// preceded it — is reported as soon as the token is read from the file,
825    /// which is what keeps it from being lost when the token opens a directive,
826    /// names a macro or is an argument the macro drops; see
827    /// [`Diagnostic::lexical`].
828    pub errors: Vec<Diagnostic>,
829}
830
831impl Token {
832    /// The keyword this token is, if any.
833    pub fn keyword(&self) -> Option<Keyword> {
834        match &self.kind {
835            TokenKind::Keyword(k) => Some(*k),
836            _ => None,
837        }
838    }
839
840    /// Whether this token is the given punctuator.
841    pub fn is_punct(&self, p: Punct) -> bool {
842        self.kind == TokenKind::Punct(p)
843    }
844
845    /// Whether this token is the given keyword.
846    pub fn is_keyword(&self, k: Keyword) -> bool {
847        self.kind == TokenKind::Keyword(k)
848    }
849
850    /// Whether this token ends the input.
851    pub fn is_eof(&self) -> bool {
852        self.kind == TokenKind::Eof
853    }
854
855    /// The identifier this token is, if any.
856    pub fn ident(&self) -> Option<&str> {
857        match &self.kind {
858            TokenKind::Ident(name) => Some(name),
859            _ => None,
860        }
861    }
862}
863
864/// Knobs for the lexer.
865#[derive(Clone, Copy, Debug)]
866pub struct LexOptions {
867    /// Which standard's lexical rules to apply.
868    pub standard: Standard,
869    /// How a constant form a newer revision introduced is gated.
870    pub gating: crate::Gating,
871    /// Accept `$` in identifiers, like GCC's `-fdollars-in-identifiers`, which
872    /// is on by default; see [`crate::Options::dollar_in_identifiers`].
873    pub dollar_in_identifiers: bool,
874    /// Whether translation phase 1 replaces the nine trigraphs.
875    ///
876    /// See [`trigraphs_enabled`] for who has them and why.
877    pub trigraphs: bool,
878    /// How wide the target's `wchar_t` is.
879    ///
880    /// The one target property the *lexer* has an opinion about: it decides
881    /// what an `L'\xffff'` escape may hold and whether `L"😀"` is one element
882    /// or a surrogate pair, since Windows makes `wchar_t` 16 bits.
883    pub wchar_bits: u32,
884    /// Whether the complex types are available, which is what decides whether
885    /// an imaginary constant (`2.0i`) has a type; see
886    /// [`crate::Options::complex`].
887    pub complex: bool,
888}
889
890impl LexOptions {
891    /// The lexical rules of `standard`, in the strict ISO dialect.
892    pub fn new(standard: Standard) -> Self {
893        Self {
894            standard,
895            gating: crate::Gating {
896                standard,
897                dialect: crate::Dialect::Iso,
898            },
899            dollar_in_identifiers: true,
900            trigraphs: trigraphs_enabled(standard, crate::Dialect::Iso),
901            wchar_bits: crate::TargetModel::host().wchar_bits,
902            complex: crate::COMPLEX_SUPPORTED,
903        }
904    }
905}
906
907impl From<&Options> for LexOptions {
908    fn from(o: &Options) -> Self {
909        Self {
910            standard: o.standard,
911            gating: o.gating(),
912            dollar_in_identifiers: o.dollar_in_identifiers,
913            trigraphs: trigraphs_enabled(o.standard, o.dialect),
914            wchar_bits: o.target.wchar_bits,
915            complex: o.complex,
916        }
917    }
918}
919
920/// Whether translation phase 1 replaces trigraphs in this entry point.
921///
922/// The nine of them were in C from the beginning and C23 removed them
923/// (N2940), so every strict entry point below `c23!` has them and `c23!` does
924/// not. No *GNU* dialect has them: `gcc -std=gnu99` switches them off, because
925/// `"what??!"` in a string is far more likely to be an exclamation than a
926/// pipe, and that is the line Clang draws too.
927pub fn trigraphs_enabled(standard: Standard, dialect: crate::Dialect) -> bool {
928    standard < Standard::C23 && !dialect.is_gnu()
929}
930
931/// The nine trigraphs of C 5.2.1.1, as `(third character, replacement)`.
932const TRIGRAPHS: &[(u8, u8)] = &[
933    (b'=', b'#'),
934    (b'(', b'['),
935    (b'/', b'\\'),
936    (b')', b']'),
937    (b'\'', b'^'),
938    (b'<', b'{'),
939    (b'!', b'|'),
940    (b'>', b'}'),
941    (b'-', b'~'),
942];
943
944/// Lexes the root file of `source`.
945///
946/// The returned vector always ends with a [`TokenKind::Eof`] token. Problems
947/// are attached to the tokens they were found in ([`Token::errors`]) rather
948/// than reported; scanning always runs to the end of the input.
949pub fn lex(source: &Source, options: &Options) -> Vec<Token> {
950    let file = source.map.file(source.root);
951    lex_text(file.text(), file.base(), &options.into())
952}
953
954/// Lexes `text`, whose first byte lives at global offset `base`.
955pub fn lex_text(text: &str, base: Pos, options: &LexOptions) -> Vec<Token> {
956    Lexer {
957        text,
958        bytes: text.as_bytes(),
959        base,
960        pos: 0,
961        options: *options,
962        pending: Vec::new(),
963    }
964    .run()
965}
966
967struct Lexer<'a> {
968    text: &'a str,
969    bytes: &'a [u8],
970    base: Pos,
971    pos: usize,
972    options: LexOptions,
973    /// Problems found since the last token was finished; they belong to the
974    /// token currently being scanned.
975    pending: Vec<Diagnostic>,
976}
977
978impl<'a> Lexer<'a> {
979    fn range(&self, start: usize, end: usize) -> SourceRange {
980        SourceRange::new(self.base + start as Pos, self.base + end as Pos)
981    }
982
983    /// Records a problem with the token being scanned.
984    fn error(&mut self, range: SourceRange, message: impl Into<String>) {
985        self.pending.push(Diagnostic::error(range, message));
986    }
987
988    /// Records a problem that stands whether or not the token being scanned
989    /// ever reaches the parser: one with its *spelling*, or one in the text
990    /// between it and the token before it. See [`Diagnostic::lexical`].
991    fn lexical_error(&mut self, range: SourceRange, message: impl Into<String>) {
992        self.pending
993            .push(Diagnostic::error(range, message).at_lexing());
994    }
995
996    /// Records an advisory remark about the token being scanned.
997    fn warning(&mut self, range: SourceRange, message: impl Into<String>) {
998        self.pending.push(Diagnostic::warning(range, message));
999    }
1000
1001    fn peek(&self) -> Option<u8> {
1002        self.bytes.get(self.pos).copied()
1003    }
1004
1005    fn peek_at(&self, n: usize) -> Option<u8> {
1006        self.bytes.get(self.pos + n).copied()
1007    }
1008
1009    fn eof(&self) -> bool {
1010        self.pos >= self.bytes.len()
1011    }
1012
1013    fn run(mut self) -> Vec<Token> {
1014        let mut tokens = Vec::new();
1015        let mut bol = true;
1016        let mut space = false;
1017        loop {
1018            let (saw_newline, saw_space) = self.skip_whitespace();
1019            bol |= saw_newline;
1020            space |= saw_space || saw_newline;
1021            if self.eof() {
1022                let end = self.bytes.len();
1023                tokens.push(Token {
1024                    kind: TokenKind::Eof,
1025                    range: self.range(end, end),
1026                    bol,
1027                    preceded_by_space: space,
1028                    errors: std::mem::take(&mut self.pending),
1029                });
1030                break;
1031            }
1032            let start = self.pos;
1033            let kind = self.scan_token();
1034            tokens.push(Token {
1035                kind,
1036                range: self.range(start, self.pos),
1037                bol,
1038                preceded_by_space: space,
1039                errors: std::mem::take(&mut self.pending),
1040            });
1041            bol = false;
1042            space = false;
1043        }
1044        tokens
1045    }
1046
1047    /// Skips whitespace, comments and line splices.
1048    ///
1049    /// Returns `(saw_newline, saw_space)`.
1050    fn skip_whitespace(&mut self) -> (bool, bool) {
1051        let mut newline = false;
1052        let mut space = false;
1053        loop {
1054            match self.peek() {
1055                Some(b'\n') => {
1056                    self.pos += 1;
1057                    newline = true;
1058                }
1059                Some(b' ' | b'\t' | b'\r' | 0x0b | 0x0c) => {
1060                    self.pos += 1;
1061                    space = true;
1062                }
1063                // Translation phase 2: a backslash immediately followed by a
1064                // newline splices the two lines, so it is *not* a line break.
1065                // `??/` is that backslash where trigraphs are on.
1066                Some(b'\\' | b'?') if self.is_line_splice(self.pos) => {
1067                    self.pos += self.line_splice_len(self.pos);
1068                    space = true;
1069                }
1070                Some(b'/') if self.peek_at(1) == Some(b'*') => {
1071                    // Translation phase 3 replaces the whole comment with one
1072                    // space, so a newline *inside* it is not a line break at
1073                    // all: a directive may span one, and a `#` after one does
1074                    // not start a directive. Both are what GCC does, and the
1075                    // standard's own `FUNC_LIKE` example in 6.10.3 depends on
1076                    // the first.
1077                    let start = self.pos;
1078                    self.pos += 2;
1079                    let mut closed = false;
1080                    while let Some(c) = self.peek() {
1081                        if c == b'*' && self.peek_at(1) == Some(b'/') {
1082                            self.pos += 2;
1083                            closed = true;
1084                            break;
1085                        }
1086                        self.pos += 1;
1087                    }
1088                    if !closed {
1089                        // Both of the problems a *comment* can have belong to
1090                        // the token that follows it for want of anywhere else
1091                        // to put them, and neither is about that token: whether
1092                        // a comment is terminated, and whether this revision
1093                        // has `//`, is settled in translation phase 3 by the
1094                        // text alone. So they are recorded as lexical, and
1095                        // reported wherever the token ends up — including
1096                        // nowhere, which is what a `#` opening a directive and
1097                        // a macro name do with it.
1098                        let range = self.range(start, self.bytes.len());
1099                        self.lexical_error(range, "unterminated comment");
1100                    }
1101                    space = true;
1102                }
1103                Some(b'/') if self.peek_at(1) == Some(b'/') => {
1104                    let start = self.pos;
1105                    self.pos += 2;
1106                    while let Some(c) = self.peek() {
1107                        if c == b'\n' {
1108                            break;
1109                        }
1110                        if matches!(c, b'\\' | b'?') && self.is_line_splice(self.pos) {
1111                            self.pos += self.line_splice_len(self.pos);
1112                            continue;
1113                        }
1114                        self.pos += 1;
1115                    }
1116                    // C99 took the `//` comment from C++ (N644); before that
1117                    // `a //* b */ c` was a division, which is why the gate is
1118                    // here rather than being a warning. Lexical for the reason
1119                    // above: a `c89!` block that writes one has to be told so
1120                    // whatever follows it.
1121                    if let Some(message) = self
1122                        .options
1123                        .gating
1124                        .requires("a '//' comment", Standard::C99)
1125                    {
1126                        let range = self.range(start, self.pos);
1127                        self.lexical_error(range, message);
1128                    }
1129                    space = true;
1130                }
1131                _ => return (newline, space),
1132            }
1133        }
1134    }
1135
1136    fn is_line_splice(&self, at: usize) -> bool {
1137        self.line_splice_len(at) > 0
1138    }
1139
1140    /// Length of a `\`-newline splice starting at `at`, or 0.
1141    ///
1142    /// Translation phase 1 runs *before* phase 2, so `??/` at the end of a
1143    /// line splices it exactly as a written backslash does — which is the one
1144    /// trigraph whose replacement is not a character the lexer can simply hand
1145    /// on.
1146    fn line_splice_len(&self, at: usize) -> usize {
1147        let lead = match self.trigraph_at(at) {
1148            Some(b'\\') => 3,
1149            Some(_) => return 0,
1150            None if self.bytes.get(at) == Some(&b'\\') => 1,
1151            None => return 0,
1152        };
1153        match (self.bytes.get(at + lead), self.bytes.get(at + lead + 1)) {
1154            (Some(b'\n'), _) => lead + 1,
1155            (Some(b'\r'), Some(b'\n')) => lead + 2,
1156            _ => 0,
1157        }
1158    }
1159
1160    /// The character a trigraph at `at` stands for, if there is one there.
1161    ///
1162    /// C 5.2.1.1: the nine three-character sequences beginning `??` are
1163    /// replaced in translation phase 1, before line splicing and before the
1164    /// source is split into tokens — so this is consulted from everywhere the
1165    /// lexer looks at a raw byte, rather than the text being rewritten. Not
1166    /// rewriting it is what keeps every [`SourceRange`] a range of the source
1167    /// the user really wrote: a diagnostic about `??=` points at all three
1168    /// characters.
1169    fn trigraph_at(&self, at: usize) -> Option<u8> {
1170        if !self.options.trigraphs
1171            || self.bytes.get(at) != Some(&b'?')
1172            || self.bytes.get(at + 1) != Some(&b'?')
1173        {
1174            return None;
1175        }
1176        let third = *self.bytes.get(at + 2)?;
1177        TRIGRAPHS
1178            .iter()
1179            .find(|(c, _)| *c == third)
1180            .map(|(_, replacement)| *replacement)
1181    }
1182
1183    /// The length in source bytes of the character at `at`, which is three for
1184    /// a trigraph and one otherwise.
1185    fn trigraph_len(&self, at: usize) -> usize {
1186        if self.trigraph_at(at).is_some() { 3 } else { 1 }
1187    }
1188
1189    /// The character-constant or string-literal prefix starting here, with its
1190    /// length in bytes.
1191    ///
1192    /// A prefix is only one when a quote follows it, which is what keeps
1193    /// `unsigned`, `u8x` and a variable called `U` ordinary identifiers.
1194    fn literal_prefix(&self, first: u8) -> Option<(StrKind, usize)> {
1195        let (kind, len) = match first {
1196            b'L' => (StrKind::Wide, 1),
1197            b'U' => (StrKind::Utf32, 1),
1198            b'u' if self.peek_at(1) == Some(b'8') => (StrKind::Utf8, 2),
1199            b'u' => (StrKind::Utf16, 1),
1200            _ => return None,
1201        };
1202        matches!(self.peek_at(len), Some(b'"' | b'\'')).then_some((kind, len))
1203    }
1204
1205    /// Whether an identifier that begins with an extended character starts
1206    /// here.
1207    fn extended_ident_start(&self) -> bool {
1208        match self.peek() {
1209            Some(b'\\') if matches!(self.peek_at(1), Some(b'u' | b'U')) => {
1210                let want = if self.peek_at(1) == Some(b'u') { 4 } else { 8 };
1211                (0..want).all(|i| self.peek_at(2 + i).is_some_and(|c| c.is_ascii_hexdigit()))
1212            }
1213            Some(c) if c >= 0x80 => self.text[self.pos..]
1214                .chars()
1215                .next()
1216                .is_some_and(|ch| is_extended_ident_char(ch, true)),
1217            _ => false,
1218        }
1219    }
1220
1221    /// Scans one token, which is [`TokenKind::Error`] for text that is not a C
1222    /// token at all.
1223    fn scan_token(&mut self) -> TokenKind {
1224        let Some(c) = self.peek() else {
1225            return TokenKind::Eof;
1226        };
1227        if is_ident_start(c, self.options.dollar_in_identifiers) {
1228            // `L'x'`, `u8"…"`, `u'x'` and the rest are constants rather than an
1229            // identifier followed by one. The prefix is recognised in every
1230            // entry point and *gated* rather than not recognised at all: a
1231            // `c99!` block that writes `u"x"` is told which macro has it,
1232            // instead of being told that `u` is undeclared.
1233            if let Some((kind, len)) = self.literal_prefix(c) {
1234                let start = self.pos;
1235                self.pos += len;
1236                let character = self.peek() == Some(b'\'');
1237                if let Some(message) = self.options.gating.requires(
1238                    &format!("a '{}' literal", kind.prefix()),
1239                    kind.since(character),
1240                ) {
1241                    let range = self.range(start, self.pos);
1242                    self.error(range, message);
1243                }
1244                return if character {
1245                    self.scan_char_constant(kind)
1246                } else {
1247                    self.scan_string_literal(kind)
1248                };
1249            }
1250            return self.scan_ident();
1251        }
1252        // An extended identifier, written either as the character itself —
1253        // which GCC and Clang have taken since GCC 10 — or as the universal
1254        // character name C99 6.4.2.1 introduced for it.
1255        if self.extended_ident_start() {
1256            return self.scan_ident();
1257        }
1258        if c.is_ascii_digit() || (c == b'.' && self.peek_at(1).is_some_and(|d| d.is_ascii_digit()))
1259        {
1260            return self.scan_number();
1261        }
1262        if c == b'\'' {
1263            return self.scan_char_constant(StrKind::Narrow);
1264        }
1265        if c == b'"' {
1266            return self.scan_string_literal(StrKind::Narrow);
1267        }
1268        if let Some(p) = self.scan_punctuator() {
1269            return TokenKind::Punct(p);
1270        }
1271
1272        // Anything else is not a C token at all.
1273        let start = self.pos;
1274        // `??/` that does not splice a line is the stray backslash a written
1275        // one would be, and is reported as one rather than as two question
1276        // marks.
1277        let ch = match self.trigraph_at(start) {
1278            Some(c) => {
1279                self.pos += 3;
1280                c as char
1281            }
1282            None => {
1283                let ch = self.text[start..].chars().next().unwrap_or('\u{fffd}');
1284                self.pos += ch.len_utf8();
1285                ch
1286            }
1287        };
1288        let range = self.range(start, self.pos);
1289        self.error(
1290            range,
1291            format!("unexpected character '{}' in program", ch.escape_debug()),
1292        );
1293        TokenKind::Error(ch.to_string())
1294    }
1295
1296    /// Scans an identifier or a keyword.
1297    ///
1298    /// Translation phase 2 deletes a backslash-newline *before* the source is
1299    /// split into tokens, so one may sit in the middle of an identifier:
1300    /// `__LI\<newline>NE__` is `__LINE__`, and Clang's own `drs/dr464.c`
1301    /// writes exactly that. Almost no identifier has one, so the spelling
1302    /// stays a slice of the source until the first splice is found and only
1303    /// then becomes a `String`.
1304    fn scan_ident(&mut self) -> TokenKind {
1305        let start = self.pos;
1306        // The text before the current splice or universal character name, when
1307        // there has been one.
1308        let mut spliced: Option<String> = None;
1309        // Where the run of characters that is still a slice begins.
1310        let mut segment = start;
1311        // Whether anything outside the basic character set was written, which
1312        // is the only case that has to be checked for normalization.
1313        let mut extended = false;
1314        loop {
1315            match self.peek() {
1316                Some(c) if c < 0x80 && is_ident_continue(c, self.options.dollar_in_identifiers) => {
1317                    self.pos += 1;
1318                }
1319                // Only a splice that the identifier *continues* over: one at
1320                // the end of it is whitespace, and belongs to whatever comes
1321                // next.
1322                Some(b'\\' | b'?')
1323                    if self.line_splice_len(self.pos) > 0
1324                        && self
1325                            .bytes
1326                            .get(self.pos + self.line_splice_len(self.pos))
1327                            .is_some_and(|c| {
1328                                is_ident_continue(*c, self.options.dollar_in_identifiers)
1329                            }) =>
1330                {
1331                    let text = spliced.get_or_insert_with(String::new);
1332                    text.push_str(&self.text[segment..self.pos]);
1333                    self.pos += self.line_splice_len(self.pos);
1334                    segment = self.pos;
1335                }
1336                // A universal character name spells one extended character:
1337                // `café` and `café` are the same identifier, which is
1338                // exactly what C99 6.4.2.1 says.
1339                Some(b'\\') if matches!(self.peek_at(1), Some(b'u' | b'U')) => {
1340                    let at = self.pos;
1341                    let Some(ch) = self.scan_ident_ucn(at == start) else {
1342                        break;
1343                    };
1344                    let text = spliced.get_or_insert_with(String::new);
1345                    text.push_str(&self.text[segment..at]);
1346                    text.push(ch);
1347                    segment = self.pos;
1348                    extended = true;
1349                }
1350                Some(c) if c >= 0x80 => {
1351                    let ch = self.text[self.pos..].chars().next().unwrap_or('\u{fffd}');
1352                    if !is_extended_ident_char(ch, self.pos == start) {
1353                        break;
1354                    }
1355                    self.pos += ch.len_utf8();
1356                    extended = true;
1357                }
1358                _ => break,
1359            }
1360        }
1361        let text = match spliced {
1362            Some(mut text) => {
1363                text.push_str(&self.text[segment..self.pos]);
1364                text
1365            }
1366            None => self.text[start..self.pos].to_owned(),
1367        };
1368        // Nothing was an identifier character after all — the whole of it was
1369        // one bad universal character name, which has been reported. Something
1370        // has to be consumed, or the scanner would sit here forever.
1371        if text.is_empty() {
1372            if self.pos == start {
1373                let ch = self.text[start..].chars().next().unwrap_or('\u{fffd}');
1374                self.pos += ch.len_utf8();
1375            }
1376            return TokenKind::Error(self.text[start..self.pos].to_owned());
1377        }
1378        // Rust identifiers have to be in Normalization Form C, and `rustc`
1379        // *normalises* the ones a procedural macro hands it rather than
1380        // refusing them — so two C identifiers that differ only by
1381        // normalization would silently become one Rust item. C23 (N2836) asks
1382        // for NFC as well, so refusing is both the safe answer and the
1383        // conforming one.
1384        if extended && !unicode_normalization::is_nfc(&text) {
1385            let range = self.range(start, self.pos);
1386            self.error(
1387                range,
1388                format!(
1389                    "identifier '{text}' is not in Unicode Normalization Form C; \
1390                     write the composed form"
1391                ),
1392            );
1393        }
1394        match Keyword::from_str(&text, self.options.standard) {
1395            Some(k) => TokenKind::Keyword(k),
1396            None => TokenKind::Ident(text),
1397        }
1398    }
1399
1400    /// Reads a `\uXXXX` or `\UXXXXXXXX` written inside an identifier.
1401    ///
1402    /// A name that is too short to be one leaves the position where it was, so
1403    /// that the identifier simply ends there and the backslash is reported by
1404    /// [`Lexer::scan_token`] as the stray character it is. One that is the
1405    /// right shape but names something an identifier may not hold is *always*
1406    /// consumed, and reported: leaving it would be a second diagnostic about
1407    /// the same text, and — where it is the first character of the identifier
1408    /// — a token that consumed nothing at all.
1409    fn scan_ident_ucn(&mut self, start: bool) -> Option<char> {
1410        let at = self.pos;
1411        let want = if self.peek_at(1) == Some(b'u') { 4 } else { 8 };
1412        let mut value: u32 = 0;
1413        for i in 0..want {
1414            let digit = self.peek_at(2 + i).and_then(|c| (c as char).to_digit(16))?;
1415            value = value * 16 + digit;
1416        }
1417        let spelling = if want == 4 { 'u' } else { 'U' };
1418        let ch = char::from_u32(value);
1419        if !ch.is_some_and(|ch| is_extended_ident_char(ch, start)) {
1420            self.pos += 2 + want;
1421            let range = self.range(at, self.pos);
1422            let digits = if want == 4 {
1423                format!("{value:04X}")
1424            } else {
1425                format!("{value:08X}")
1426            };
1427            let message =
1428                format!("'\\{spelling}{digits}' is not a valid character in an identifier");
1429            // Two different refusals wear the same words. Naming a character
1430            // below U+00A0, or a surrogate, is what C99 6.4.3p2 forbids of the
1431            // *name* — C23 puts `$`, `@` and `` ` `` in the basic character
1432            // set, and Clang refuses those in every mode — so it is ill formed
1433            // where it is written and is reported even where the token goes
1434            // nowhere: Clang's own `C99/n717.c` is a file of them, each
1435            // written as the argument of a macro that expands to nothing.
1436            // Everything else here — a value outside Unicode, a character
1437            // that simply is not `XID_Continue` — is about the *identifier*,
1438            // and a preprocessing token that never reaches the parser is
1439            // allowed to be one C has no other use for (6.4p3).
1440            if value < 0xA0 || (0xD800..=0xDFFF).contains(&value) {
1441                self.lexical_error(range, message);
1442            } else {
1443                self.error(range, message);
1444            }
1445            return None;
1446        }
1447        if let Some(message) = self
1448            .options
1449            .gating
1450            .requires("a universal character name", Standard::C99)
1451        {
1452            let range = self.range(at, at + 2 + want);
1453            self.error(range, message);
1454        }
1455        self.pos += 2 + want;
1456        ch
1457    }
1458
1459    fn scan_punctuator(&mut self) -> Option<Punct> {
1460        if self.trigraph_at(self.pos).is_some() {
1461            return self.scan_trigraph_punctuator();
1462        }
1463        let rest = &self.text[self.pos..];
1464        for (spelling, punct) in PUNCTUATORS {
1465            if rest.starts_with(spelling) {
1466                // `<:` is a digraph for `[`, but `1<::x` must not be mangled;
1467                // C99 has no `::`, so a plain greedy match is correct here.
1468                self.pos += spelling.len();
1469                return Some(*punct);
1470            }
1471        }
1472        None
1473    }
1474
1475    /// A punctuator that begins with a trigraph.
1476    ///
1477    /// Phase 1 happens before tokens exist, so a punctuator may be spelled
1478    /// partly or wholly with trigraphs — `??!??!` is `||`, `??'=` is `^=`,
1479    /// `??=??=` is `##` — and maximal munch applies to the *replaced*
1480    /// characters. Up to the length of the longest punctuator is replaced into
1481    /// a small buffer, matched there, and the source position advanced by what
1482    /// the matched characters really cost.
1483    fn scan_trigraph_punctuator(&mut self) -> Option<Punct> {
1484        const LONGEST: usize = 4;
1485        let mut logical = [0u8; LONGEST];
1486        let mut widths = [0usize; LONGEST];
1487        let mut count = 0;
1488        let mut at = self.pos;
1489        while count < LONGEST {
1490            let (c, width) = match self.trigraph_at(at) {
1491                Some(c) => (c, 3),
1492                None => match self.bytes.get(at) {
1493                    Some(c) => (*c, 1),
1494                    None => break,
1495                },
1496            };
1497            // A backslash is not part of any punctuator, and neither is
1498            // anything outside the basic character set.
1499            if c == b'\\' || !c.is_ascii() {
1500                break;
1501            }
1502            logical[count] = c;
1503            widths[count] = width;
1504            count += 1;
1505            at += width;
1506        }
1507        let text = std::str::from_utf8(&logical[..count]).ok()?;
1508        for (spelling, punct) in PUNCTUATORS {
1509            if text.starts_with(spelling) {
1510                self.pos += widths[..spelling.len()].iter().sum::<usize>();
1511                return Some(*punct);
1512            }
1513        }
1514        None
1515    }
1516
1517    // -- numbers ------------------------------------------------------------
1518
1519    /// Scans a preprocessing number and classifies it as integer or float.
1520    ///
1521    /// Scanning the whole pp-number first (rather than stopping at the first
1522    /// character that does not fit) is what lets `08` or `1.0q` be reported as
1523    /// one bad token instead of two good ones.
1524    fn scan_number(&mut self) -> TokenKind {
1525        let start = self.pos;
1526        if self.peek() == Some(b'.') {
1527            self.pos += 1;
1528        }
1529        self.pos += 1;
1530        let mut separators = false;
1531        while let Some(c) = self.peek() {
1532            if matches!(c, b'e' | b'E' | b'p' | b'P')
1533                && matches!(self.peek_at(1), Some(b'+') | Some(b'-'))
1534            {
1535                self.pos += 2;
1536                continue;
1537            }
1538            // C23's digit separator. It is part of the pp-number when a digit
1539            // or a nondigit follows it, which is what keeps `1'a'` from
1540            // swallowing the character constant that follows a constant.
1541            if c == b'\''
1542                && self
1543                    .peek_at(1)
1544                    .is_some_and(|d| d.is_ascii_alphanumeric() || d == b'_')
1545            {
1546                separators = true;
1547                self.pos += 1;
1548                continue;
1549            }
1550            if c.is_ascii_alphanumeric()
1551                || c == b'_'
1552                || c == b'.'
1553                || (c == b'$' && self.options.dollar_in_identifiers)
1554            {
1555                self.pos += 1;
1556                continue;
1557            }
1558            break;
1559        }
1560        let text = &self.text[start..self.pos];
1561        let range = self.range(start, self.pos);
1562        if separators
1563            && let Some(message) = self
1564                .options
1565                .gating
1566                .requires("a digit separator", Standard::C23)
1567        {
1568            self.error(range, message);
1569        }
1570        // Everything below reads the digits; the separators are not part of
1571        // the value, while `text` keeps the spelling `#` has to reproduce.
1572        let stripped: String;
1573        let digits = if separators {
1574            stripped = text.replace('\'', "");
1575            stripped.as_str()
1576        } else {
1577            text
1578        };
1579        let lower = digits.to_ascii_lowercase();
1580        let hex = lower.starts_with("0x");
1581        let is_float = if hex {
1582            digits.contains('.') || lower[2..].contains('p')
1583        } else {
1584            digits.contains('.') || (!lower.starts_with("0b") && lower.contains('e'))
1585        };
1586        if is_float {
1587            TokenKind::Float(self.decode_float(digits, text, range, hex))
1588        } else {
1589            TokenKind::Int(self.decode_int(digits, text, range))
1590        }
1591    }
1592
1593    /// Decodes an integer constant from its separator-free `digits`; `text` is
1594    /// the spelling as written, which is what diagnostics quote and what `#`
1595    /// reproduces.
1596    fn decode_int(&mut self, digits: &str, text: &str, range: SourceRange) -> IntLit {
1597        let bytes = digits.as_bytes();
1598        let (base, digits_start) =
1599            if digits.len() >= 2 && (bytes[1] | 0x20) == b'x' && bytes[0] == b'0' {
1600                (NumBase::Hex, 2)
1601            } else if digits.len() >= 2 && (bytes[1] | 0x20) == b'b' && bytes[0] == b'0' {
1602                (NumBase::Binary, 2)
1603            } else if bytes[0] == b'0' && digits.len() > 1 {
1604                (NumBase::Octal, 1)
1605            } else {
1606                (NumBase::Decimal, 0)
1607            };
1608        if base == NumBase::Binary
1609            && let Some(message) = self
1610                .options
1611                .gating
1612                .requires("a binary integer constant", Standard::C23)
1613        {
1614            self.error(range, message);
1615        }
1616
1617        let radix = base.radix();
1618        // Accept every decimal digit in an octal or binary constant so that
1619        // the whole constant is consumed and reported once, the way C
1620        // compilers do.
1621        let scan_radix = if radix < 10 { 10 } else { radix };
1622
1623        let mut i = digits_start;
1624        let mut value: u128 = 0;
1625        let mut overflow = false;
1626        let mut bad_digit: Option<char> = None;
1627        while i < bytes.len() {
1628            let c = bytes[i] as char;
1629            let digit = match c.to_digit(scan_radix) {
1630                Some(d) => d,
1631                None => break,
1632            };
1633            if digit >= radix && bad_digit.is_none() {
1634                bad_digit = Some(c);
1635            }
1636            match value
1637                .checked_mul(radix as u128)
1638                .and_then(|v| v.checked_add(digit as u128))
1639            {
1640                Some(v) => value = v,
1641                None => overflow = true,
1642            }
1643            i += 1;
1644        }
1645
1646        if i == digits_start && matches!(base, NumBase::Hex | NumBase::Binary) {
1647            let prefix = &digits[..2];
1648            self.error(
1649                range,
1650                format!(
1651                    "expected digits after '{prefix}' in {} constant",
1652                    base.as_str()
1653                ),
1654            );
1655        }
1656        if let Some(c) = bad_digit {
1657            self.error(
1658                range,
1659                format!("invalid digit '{c}' in {} constant '{text}'", base.as_str()),
1660            );
1661        }
1662        if overflow {
1663            self.error(range, format!("integer constant '{text}' is too large"));
1664        }
1665
1666        let suffix = &digits[i..];
1667        let (unsigned, long) = match parse_int_suffix(suffix) {
1668            Some(v) => v,
1669            None => {
1670                // `3i` is GCC's `_Complex int`, one of the two complex
1671                // *integer* types that are a GNU extension of their own and
1672                // that cinrs does not have. Saying what to write instead is
1673                // more use than "invalid suffix".
1674                if matches!(suffix, "i" | "j" | "I" | "J") {
1675                    let digits = text.strip_suffix(suffix).unwrap_or(text);
1676                    self.error(
1677                        range,
1678                        format!(
1679                            "'{text}' is a complex integer constant, which is a GNU extension \
1680                             cinrs does not support; write '{digits}.0{suffix}' for the \
1681                             complex floating constant"
1682                        ),
1683                    );
1684                } else {
1685                    self.error(
1686                        range,
1687                        format!("invalid suffix '{suffix}' on integer constant '{text}'"),
1688                    );
1689                }
1690                (false, LongKind::None)
1691            }
1692        };
1693
1694        IntLit {
1695            value,
1696            base,
1697            unsigned,
1698            long,
1699            text: text.to_owned(),
1700        }
1701    }
1702
1703    fn decode_float(
1704        &mut self,
1705        digits: &str,
1706        text: &str,
1707        range: SourceRange,
1708        hex: bool,
1709    ) -> FloatLit {
1710        let (body, suffix) = split_float_suffix(digits, hex);
1711        let (suffix_kind, imaginary) = self.float_suffix(suffix, text, range);
1712
1713        if hex
1714            && let Some(message) = self
1715                .options
1716                .gating
1717                .requires("a hexadecimal floating constant", Standard::C99)
1718        {
1719            self.error(range, message);
1720        }
1721        let value = if hex {
1722            match parse_hex_float(body) {
1723                Some(v) => v,
1724                None => {
1725                    self.error(
1726                        range,
1727                        format!(
1728                            "invalid hexadecimal floating constant '{text}'; \
1729                             a 'p' exponent is required"
1730                        ),
1731                    );
1732                    0.0
1733                }
1734            }
1735        } else {
1736            match parse_decimal_float(body) {
1737                Some(v) => v,
1738                None => {
1739                    self.error(range, format!("invalid floating constant '{text}'"));
1740                    0.0
1741                }
1742            }
1743        };
1744
1745        FloatLit {
1746            value,
1747            suffix: suffix_kind,
1748            imaginary,
1749            hex,
1750            text: text.to_owned(),
1751        }
1752    }
1753
1754    /// The type a floating constant's suffix gives it.
1755    ///
1756    /// C has three: none, `f` and `l`. GCC has a dozen more, and they fall
1757    /// into three groups here.
1758    ///
1759    /// * **The ones that name a format wider than `double`** — `d`, `w`
1760    ///   (`__float80`), `q` (`__float128`), and the `_FloatN` and `_FloatNx`
1761    ///   suffixes `f64`, `f64x`, `f32x` and `f128`. Every one of them is
1762    ///   `double` in this implementation, exactly as `long double` is, so each
1763    ///   is accepted in a GNU dialect and **loses precision** where GCC would
1764    ///   not; `doc/gnu-extensions.md` records that. `f32` is `float`.
1765    /// * **The decimal floating suffixes** `df`, `dd` and `dl`, whose types
1766    ///   are radix-10 and have no Rust counterpart at all.
1767    /// * **The imaginary suffixes** `i` and `j`, which make the constant an
1768    ///   imaginary one — `2.0i` is `(0, 2)` — and so need the complex types.
1769    ///
1770    /// The decimal ones are refused with the reason, and so are the imaginary
1771    /// ones when the complex types are switched off. A strict entry point
1772    /// refuses the first group too, naming the GNU entry point that has it —
1773    /// the suffixes are spelled without underscores, which is the line
1774    /// [`crate::Dialect`] draws.
1775    ///
1776    /// The second half of the answer is whether the constant is imaginary; see
1777    /// [`FloatLit::imaginary`].
1778    fn float_suffix(
1779        &mut self,
1780        suffix: &str,
1781        text: &str,
1782        range: SourceRange,
1783    ) -> (FloatSuffix, bool) {
1784        let lower = suffix.to_ascii_lowercase();
1785        match lower.as_str() {
1786            "" => return (FloatSuffix::None, false),
1787            "f" => return (FloatSuffix::Float, false),
1788            "l" => return (FloatSuffix::LongDouble, false),
1789            _ => {}
1790        }
1791        // GNU's imaginary suffix, on its own or beside `f` or `l`. Both
1792        // spellings are the same thing: `j` is what Fortran and engineering
1793        // habit write, and GCC takes either.
1794        let imaginary = match lower.as_str() {
1795            "i" | "j" => Some(FloatSuffix::None),
1796            "if" | "fi" | "jf" | "fj" => Some(FloatSuffix::Float),
1797            "il" | "li" | "jl" | "lj" => Some(FloatSuffix::LongDouble),
1798            _ => None,
1799        };
1800        if let Some(kind) = imaginary {
1801            if !self.options.complex {
1802                self.error(
1803                    range,
1804                    format!(
1805                        "invalid suffix '{suffix}' on floating constant '{text}': an \
1806                         imaginary constant needs _Complex. {}",
1807                        crate::COMPLEX_UNSUPPORTED
1808                    ),
1809                );
1810                return (FloatSuffix::None, false);
1811            }
1812            if let Some(message) = self
1813                .options
1814                .gating
1815                .requires("an imaginary constant", Standard::C99)
1816            {
1817                self.error(range, message);
1818                return (FloatSuffix::None, false);
1819            }
1820            return (kind, true);
1821        }
1822        // A decimal constant is refused whatever the entry point: it has no
1823        // type this crate can give it.
1824        let refusal = match lower.as_str() {
1825            "df" | "dd" | "dl" => Some(
1826                "the decimal floating types (_Decimal32, _Decimal64, _Decimal128) are not \
1827                 supported: they are radix-10 and no Rust type is",
1828            ),
1829            "f16" | "f16x" | "bf16" => Some(
1830                "'_Float16' is not supported: Rust's `f16` is unstable, and rounding the \
1831                 constant to a wider type would change what the program computes",
1832            ),
1833            _ => None,
1834        };
1835        if let Some(reason) = refusal {
1836            self.error(
1837                range,
1838                format!("invalid suffix '{suffix}' on floating constant '{text}': {reason}"),
1839            );
1840            return (FloatSuffix::None, false);
1841        }
1842        // The rest are the GNU widths. `f32` is `float`; every other one names
1843        // a format this implementation makes a `double`.
1844        let wider = matches!(
1845            lower.as_str(),
1846            "d" | "w" | "q" | "f64" | "f64x" | "f32x" | "f128" | "f128x"
1847        );
1848        if !wider && lower != "f32" {
1849            self.error(
1850                range,
1851                format!("invalid suffix '{suffix}' on floating constant '{text}'"),
1852            );
1853            return (FloatSuffix::None, false);
1854        }
1855        if !self.options.gating.dialect.is_gnu() {
1856            let gnu = self.options.gating.standard.macro_name_in(Dialect::Gnu);
1857            let here = self
1858                .options
1859                .gating
1860                .standard
1861                .macro_name_in(self.options.gating.dialect);
1862            self.error(
1863                range,
1864                format!(
1865                    "the suffix '{suffix}' on a floating constant is a GNU extension, and \
1866                     requires a GNU dialect ({gnu}) (this block is {here})"
1867                ),
1868            );
1869            return (FloatSuffix::None, false);
1870        }
1871        if lower == "f32" {
1872            return (FloatSuffix::Float, false);
1873        }
1874        // `LongDouble` is `double`, which is what all of these come to.
1875        (FloatSuffix::LongDouble, false)
1876    }
1877
1878    // -- character and string constants -------------------------------------
1879
1880    fn scan_char_constant(&mut self, kind: StrKind) -> TokenKind {
1881        let start = self.pos;
1882        debug_assert_eq!(self.peek(), Some(b'\''));
1883        self.pos += 1;
1884        let mut values: Vec<u32> = Vec::new();
1885        let mut terminated = false;
1886        // How many *characters* were written, which is not how many elements
1887        // they came to: one `\U0001F600` is one character and two UTF-16 code
1888        // units, and the two say different things about what is wrong.
1889        let mut characters = 0usize;
1890        while let Some(c) = self.peek() {
1891            if c == b'\'' {
1892                self.pos += 1;
1893                terminated = true;
1894                break;
1895            }
1896            if c == b'\n' {
1897                break;
1898            }
1899            let before = values.len();
1900            self.read_char_element(kind, &mut values);
1901            if values.len() > before {
1902                characters += 1;
1903            }
1904        }
1905        let range = self.range(start, self.pos);
1906        if !terminated {
1907            self.error(range, "missing terminating \' character");
1908        }
1909        if values.is_empty() {
1910            self.error(range, "empty character constant");
1911        }
1912        // Only `'ab'` and `L'ab'` have an implementation-defined meaning; C11
1913        // 6.4.4.4p2 makes more than one character in a `u8`, `u` or `U`
1914        // constant a constraint violation, because there is no room for a
1915        // second one in the type — and so is one character that needs more
1916        // than one code unit, which is what `u8'é'` and `u'😀'` are.
1917        if values.len() > 1 {
1918            match kind {
1919                StrKind::Narrow | StrKind::Wide => {
1920                    self.warning(range, "multi-character character constant");
1921                }
1922                _ if characters > 1 => {
1923                    self.error(
1924                        range,
1925                        format!(
1926                            "a '{}' character constant holds exactly one character",
1927                            kind.prefix()
1928                        ),
1929                    );
1930                }
1931                _ => {
1932                    self.error(
1933                        range,
1934                        format!(
1935                            "the character in a '{}' character constant must fit in a \
1936                             single code unit",
1937                            kind.prefix()
1938                        ),
1939                    );
1940                }
1941            }
1942        }
1943
1944        let value = if kind != StrKind::Narrow {
1945            values.last().copied().unwrap_or(0) as i64
1946        } else if values.len() <= 1 {
1947            values.first().copied().unwrap_or(0) as i64
1948        } else {
1949            // GCC packs the bytes big-endian into an `int`.
1950            let mut v: u32 = 0;
1951            for b in &values {
1952                v = (v << 8) | (*b & 0xff);
1953            }
1954            v as i32 as i64
1955        };
1956
1957        TokenKind::Char(CharLit {
1958            value,
1959            kind,
1960            text: self.text[start..self.pos].to_owned(),
1961        })
1962    }
1963
1964    fn scan_string_literal(&mut self, kind: StrKind) -> TokenKind {
1965        let start = self.pos;
1966        debug_assert_eq!(self.peek(), Some(b'"'));
1967        self.pos += 1;
1968        let mut values: Vec<u32> = Vec::new();
1969        let mut terminated = false;
1970        while let Some(c) = self.peek() {
1971            if c == b'"' {
1972                self.pos += 1;
1973                terminated = true;
1974                break;
1975            }
1976            if c == b'\n' {
1977                break;
1978            }
1979            self.read_char_element(kind, &mut values);
1980        }
1981        let range = self.range(start, self.pos);
1982        if !terminated {
1983            self.error(range, "missing terminating \" character");
1984        }
1985        TokenKind::Str(StrLit {
1986            kind,
1987            values,
1988            text: self.text[start..self.pos].to_owned(),
1989        })
1990    }
1991
1992    /// Reads one element of a character constant or string literal, appending
1993    /// its decoded value(s) to `out`.
1994    fn read_char_element(&mut self, kind: StrKind, out: &mut Vec<u32>) {
1995        if self.is_line_splice(self.pos) {
1996            self.pos += self.line_splice_len(self.pos);
1997            return;
1998        }
1999        let start = self.pos;
2000        // Phase 1 replaces trigraphs inside literals too: `"??!"` is `"|"`,
2001        // and `"??/n"` is `"\n"`.
2002        let trigraph = self.trigraph_at(self.pos);
2003        if trigraph != Some(b'\\') && self.peek() != Some(b'\\') {
2004            match trigraph {
2005                Some(c) => {
2006                    self.pos += 3;
2007                    out.push(u32::from(c));
2008                }
2009                None if kind.is_bytes() => {
2010                    // A narrow or UTF-8 literal keeps the raw
2011                    // execution-charset bytes, so UTF-8 text in one survives
2012                    // byte for byte.
2013                    self.pos += 1;
2014                    out.push(self.bytes[start] as u32);
2015                }
2016                None => {
2017                    let ch = self.text[start..].chars().next().unwrap_or('\u{fffd}');
2018                    self.pos += ch.len_utf8();
2019                    push_character(ch as u32, kind, self.options.wchar_bits, out);
2020                }
2021            }
2022            return;
2023        }
2024
2025        self.pos += self.trigraph_len(self.pos);
2026        let e = match self.trigraph_at(self.pos) {
2027            Some(c) => c,
2028            None => match self.peek() {
2029                Some(c) => c,
2030                None => {
2031                    let range = self.range(start, self.pos);
2032                    self.error(range, "incomplete escape sequence");
2033                    return;
2034                }
2035            },
2036        };
2037        self.pos += self.trigraph_len(self.pos);
2038        let simple = match e {
2039            b'\'' => Some(0x27),
2040            b'"' => Some(0x22),
2041            b'?' => Some(0x3f),
2042            b'\\' => Some(0x5c),
2043            b'a' => Some(0x07),
2044            b'b' => Some(0x08),
2045            // `\e` is GNU's escape for ESC. GCC accepts it in every mode (with
2046            // a pedantic warning), and `"\e[0m"` is how a program writes a
2047            // terminal colour; refusing it would be refusing the extension in
2048            // the one place a strict mode still has it.
2049            b'e' => Some(0x1b),
2050            b'f' => Some(0x0c),
2051            b'n' => Some(0x0a),
2052            b'r' => Some(0x0d),
2053            b't' => Some(0x09),
2054            b'v' => Some(0x0b),
2055            _ => None,
2056        };
2057        if let Some(v) = simple {
2058            out.push(v);
2059            return;
2060        }
2061        match e {
2062            b'0'..=b'7' => {
2063                let mut v: u32 = (e - b'0') as u32;
2064                for _ in 0..2 {
2065                    match self.peek() {
2066                        Some(d @ b'0'..=b'7') => {
2067                            v = v * 8 + (d - b'0') as u32;
2068                            self.pos += 1;
2069                        }
2070                        _ => break,
2071                    }
2072                }
2073                self.push_escape_value(v, kind, start, out);
2074            }
2075            b'x' => {
2076                let mut v: u32 = 0;
2077                let mut any = false;
2078                let mut overflow = false;
2079                while let Some(d) = self.peek().and_then(|c| (c as char).to_digit(16)) {
2080                    any = true;
2081                    v = match v.checked_mul(16).and_then(|v| v.checked_add(d)) {
2082                        Some(v) => v,
2083                        None => {
2084                            overflow = true;
2085                            v
2086                        }
2087                    };
2088                    self.pos += 1;
2089                }
2090                let range = self.range(start, self.pos);
2091                if !any {
2092                    self.error(range, "'\\x' used with no following hex digits");
2093                } else if overflow {
2094                    self.error(range, "hex escape sequence out of range");
2095                }
2096                self.push_escape_value(v, kind, start, out);
2097            }
2098            b'u' | b'U' => {
2099                if let Some(message) = self
2100                    .options
2101                    .gating
2102                    .requires("a universal character name", Standard::C99)
2103                {
2104                    let range = self.range(start, self.pos);
2105                    self.error(range, message);
2106                }
2107                let want = if e == b'u' { 4 } else { 8 };
2108                let mut v: u32 = 0;
2109                let mut count = 0;
2110                while count < want {
2111                    match self.peek().and_then(|c| (c as char).to_digit(16)) {
2112                        Some(d) => {
2113                            v = v.wrapping_mul(16).wrapping_add(d);
2114                            self.pos += 1;
2115                            count += 1;
2116                        }
2117                        None => break,
2118                    }
2119                }
2120                let range = self.range(start, self.pos);
2121                if count != want {
2122                    self.error(
2123                        range,
2124                        format!("incomplete universal character name; expected {want} hex digits"),
2125                    );
2126                    return;
2127                }
2128                match char::from_u32(v) {
2129                    Some(ch) if kind.is_bytes() => {
2130                        // The execution character set is UTF-8.
2131                        let mut buf = [0u8; 4];
2132                        for b in ch.encode_utf8(&mut buf).as_bytes() {
2133                            out.push(*b as u32);
2134                        }
2135                    }
2136                    Some(ch) => push_character(ch as u32, kind, self.options.wchar_bits, out),
2137                    None => {
2138                        // Outside Unicode, or a surrogate: 6.4.3p2 again, and
2139                        // the *literal token's* spelling is what is wrong, so
2140                        // this stands wherever the token ends up.
2141                        self.lexical_error(
2142                            range,
2143                            format!("'\\u{v:04X}' is not a valid universal character name"),
2144                        );
2145                    }
2146                }
2147            }
2148            _ => {
2149                let range = self.range(start, self.pos);
2150                self.error(
2151                    range,
2152                    format!("unknown escape sequence '\\{}'", (e as char).escape_debug()),
2153                );
2154                out.push(e as u32);
2155            }
2156        }
2157    }
2158
2159    /// Pushes the value of a numeric escape sequence, which unlike a character
2160    /// is *not* re-encoded: `u"\xd83d"` is that one code unit.
2161    fn push_escape_value(&mut self, v: u32, kind: StrKind, start: usize, out: &mut Vec<u32>) {
2162        let max = kind.max_element(self.options.wchar_bits);
2163        if v > max {
2164            let range = self.range(start, self.pos);
2165            let ty = match kind {
2166                StrKind::Narrow => "char",
2167                StrKind::Utf8 => "char8_t",
2168                StrKind::Utf16 => "char16_t",
2169                StrKind::Utf32 => "char32_t",
2170                StrKind::Wide => "wchar_t",
2171            };
2172            self.error(
2173                range,
2174                format!("escape sequence out of range for type '{ty}'"),
2175            );
2176            out.push(v & max);
2177        } else {
2178            out.push(v);
2179        }
2180    }
2181}
2182
2183/// Appends one character, encoded the way `kind` stores its elements.
2184///
2185/// A `u"…"` literal holds UTF-16 code units, so a character outside the basic
2186/// multilingual plane becomes the two halves of a surrogate pair — which is
2187/// what makes `sizeof(u"\U0001F600")` six rather than four. On a target whose
2188/// `wchar_t` is 16 bits wide, `L"…"` is UTF-16 too and does the same.
2189fn push_character(value: u32, kind: StrKind, wchar_bits: u32, out: &mut Vec<u32>) {
2190    if !kind.is_utf16(wchar_bits) || value <= 0xffff {
2191        out.push(value);
2192        return;
2193    }
2194    let v = value - 0x1_0000;
2195    out.push(0xd800 + (v >> 10));
2196    out.push(0xdc00 + (v & 0x3ff));
2197}
2198
2199fn is_ident_start(c: u8, dollar: bool) -> bool {
2200    c.is_ascii_alphabetic() || c == b'_' || (dollar && c == b'$')
2201}
2202
2203fn is_ident_continue(c: u8, dollar: bool) -> bool {
2204    c.is_ascii_alphanumeric() || c == b'_' || (dollar && c == b'$')
2205}
2206
2207/// Whether an extended character may appear in an identifier.
2208///
2209/// C99 Annex D listed the ranges by hand, C11 revised the list, and C23 (N2836,
2210/// N2939) replaced all of it with Unicode Annex #31's `XID_Start` and
2211/// `XID_Continue` — which is also what Rust's own identifiers are, and what
2212/// makes a C name usable as a Rust one. The one list is used in every entry
2213/// point: the earlier annexes are approximations of the same intent, and a
2214/// program that uses a character C11 left out is one this would otherwise
2215/// refuse for no reason a user could act on.
2216///
2217/// A character of the basic character set is never one of these: the ASCII
2218/// path has already decided about it, and a universal character name is not
2219/// allowed to spell one (6.4.3p2).
2220fn is_extended_ident_char(ch: char, start: bool) -> bool {
2221    if ch.is_ascii() {
2222        return false;
2223    }
2224    if start {
2225        unicode_ident::is_xid_start(ch)
2226    } else {
2227        unicode_ident::is_xid_continue(ch)
2228    }
2229}
2230
2231/// Validates an integer suffix, returning `(unsigned, long_kind)`.
2232fn parse_int_suffix(s: &str) -> Option<(bool, LongKind)> {
2233    if s.is_empty() {
2234        return Some((false, LongKind::None));
2235    }
2236    let b = s.as_bytes();
2237    let mut i = 0;
2238    let mut unsigned = false;
2239    let mut long = LongKind::None;
2240
2241    if b[i] == b'u' || b[i] == b'U' {
2242        unsigned = true;
2243        i += 1;
2244    }
2245    if i < b.len() && (b[i] == b'l' || b[i] == b'L') {
2246        // `ll` and `LL` must not be mixed as `lL` or `Ll`.
2247        if i + 1 < b.len() && b[i + 1] == b[i] {
2248            long = LongKind::LongLong;
2249            i += 2;
2250        } else {
2251            long = LongKind::Long;
2252            i += 1;
2253        }
2254    }
2255    if !unsigned && i < b.len() && (b[i] == b'u' || b[i] == b'U') {
2256        unsigned = true;
2257        i += 1;
2258    }
2259    (i == b.len()).then_some((unsigned, long))
2260}
2261
2262/// Splits a floating constant into its numeric body and its suffix.
2263///
2264/// The *body* is scanned forward rather than the suffix backwards, because a
2265/// suffix may hold digits of its own: `1.0f128` names `_Float128` and the
2266/// `128` is no part of the number. What is left after the digits, the point
2267/// and the exponent is the suffix, whatever it looks like; naming it is
2268/// [`Lexer::float_suffix`]'s business.
2269fn split_float_suffix(text: &str, hex: bool) -> (&str, &str) {
2270    let b = text.as_bytes();
2271    let mut i = 0;
2272    let (exponent, digit): (u8, fn(u8) -> bool) = if hex {
2273        i = 2; // the `0x` the caller has already recognised
2274        (b'p', |c| c.is_ascii_hexdigit())
2275    } else {
2276        (b'e', |c| c.is_ascii_digit())
2277    };
2278    while i < b.len() && (digit(b[i]) || b[i] == b'.') {
2279        i += 1;
2280    }
2281    if i < b.len() && b[i] | 0x20 == exponent {
2282        i += 1;
2283        if i < b.len() && (b[i] == b'+' || b[i] == b'-') {
2284            i += 1;
2285        }
2286        while i < b.len() && b[i].is_ascii_digit() {
2287            i += 1;
2288        }
2289    }
2290    (&text[..i], &text[i..])
2291}
2292
2293/// Parses the numeric body of a decimal floating constant.
2294fn parse_decimal_float(body: &str) -> Option<f64> {
2295    if body.is_empty() {
2296        return None;
2297    }
2298    // `None` is "no exponent at all", which is a different thing from an `e`
2299    // with nothing after it — `1e` is not a constant.
2300    let (mantissa, exponent) = match body.find(['e', 'E']) {
2301        Some(i) => (&body[..i], Some(&body[i + 1..])),
2302        None => (body, None),
2303    };
2304    let (int_part, frac_part) = match mantissa.find('.') {
2305        Some(i) => (&mantissa[..i], &mantissa[i + 1..]),
2306        None => (mantissa, ""),
2307    };
2308    if int_part.is_empty() && frac_part.is_empty() {
2309        return None;
2310    }
2311    if !int_part.bytes().all(|c| c.is_ascii_digit())
2312        || !frac_part.bytes().all(|c| c.is_ascii_digit())
2313    {
2314        return None;
2315    }
2316    let exponent = match exponent {
2317        None => 0i32,
2318        Some(exponent) => {
2319            let (sign, digits) = match exponent.as_bytes().first() {
2320                Some(b'+') => (1, &exponent[1..]),
2321                Some(b'-') => (-1, &exponent[1..]),
2322                _ => (1, exponent),
2323            };
2324            if digits.is_empty() || !digits.bytes().all(|c| c.is_ascii_digit()) {
2325                return None;
2326            }
2327            // Saturate: an absurd exponent simply becomes 0 or infinity.
2328            sign * digits.parse::<i32>().unwrap_or(i32::MAX / 2)
2329        }
2330    };
2331    let normalized = format!(
2332        "{}.{}e{}",
2333        if int_part.is_empty() { "0" } else { int_part },
2334        if frac_part.is_empty() { "0" } else { frac_part },
2335        exponent
2336    );
2337    normalized.parse::<f64>().ok()
2338}
2339
2340/// Parses the numeric body of a hexadecimal floating constant (`0x1.8p3`).
2341fn parse_hex_float(body: &str) -> Option<f64> {
2342    let rest = body
2343        .strip_prefix("0x")
2344        .or_else(|| body.strip_prefix("0X"))?;
2345    let p = rest.find(['p', 'P'])?;
2346    let (mantissa, exponent) = (&rest[..p], &rest[p + 1..]);
2347    let (int_part, frac_part) = match mantissa.find('.') {
2348        Some(i) => (&mantissa[..i], &mantissa[i + 1..]),
2349        None => (mantissa, ""),
2350    };
2351    if int_part.is_empty() && frac_part.is_empty() {
2352        return None;
2353    }
2354    let mut value = 0f64;
2355    for c in int_part.chars() {
2356        value = value * 16.0 + c.to_digit(16)? as f64;
2357    }
2358    let mut scale = 1.0 / 16.0;
2359    for c in frac_part.chars() {
2360        value += c.to_digit(16)? as f64 * scale;
2361        scale /= 16.0;
2362    }
2363    let (sign, digits) = match exponent.as_bytes().first() {
2364        Some(b'+') => (1i32, &exponent[1..]),
2365        Some(b'-') => (-1i32, &exponent[1..]),
2366        _ => (1i32, exponent),
2367    };
2368    if digits.is_empty() || !digits.bytes().all(|c| c.is_ascii_digit()) {
2369        return None;
2370    }
2371    let exp = sign * digits.parse::<i32>().unwrap_or(i32::MAX / 2);
2372    Some(value * 2f64.powi(exp))
2373}