cinrs_core/lex.rs
1//! A complete C99 lexer working on the captured source text.
2//!
3//! The lexer is byte-offset based: every [`Token`] carries a
4//! [`SourceRange`] into the [`SourceMap`](crate::capture::SourceMap), which is
5//! what makes exact diagnostics possible. Constants are decoded here (value,
6//! base, suffix, escape sequences) so that neither the parser nor sema has to
7//! look at raw text again.
8//!
9//! Two flags on each token — [`Token::bol`] and [`Token::preceded_by_space`] —
10//! are what the [preprocessor](crate::pp) recognises directives with and what
11//! its stringification operator reproduces: directive recognition needs "first
12//! token on a line", and `#` needs to know where whitespace was.
13//!
14//! # The lexer never reports anything
15//!
16//! Lexical errors never stop the lexer, and they never reach [`Diagnostics`]
17//! either: each one is attached to the token it was found in, in
18//! [`Token::errors`], and the preprocessor reports the ones whose token
19//! survives into its output. That is not a detail — a group skipped by
20//! `#if 0` may legally hold text that is not C at all, and a macro that is
21//! never invoked may hold anything its author liked:
22//!
23//! ```c
24//! #if 0
25//! this is not C: 08, 'unterminated, @@@
26//! #endif
27//! ```
28//!
29//! Text that is not a token at all becomes a [`TokenKind::Error`] token
30//! carrying its own spelling, so that it too can be skipped rather than
31//! reported. The preprocessor drops those tokens after reporting them, so
32//! nothing downstream ever sees one.
33//!
34//! [`Diagnostics`]: crate::diag::Diagnostics
35
36use crate::capture::{Pos, Source, SourceRange};
37use crate::diag::Diagnostic;
38use crate::{Dialect, Options, Standard};
39
40/// A C99 keyword.
41#[derive(Clone, Copy, PartialEq, Eq, Debug, Hash)]
42#[allow(missing_docs)]
43pub enum Keyword {
44 Auto,
45 Break,
46 Case,
47 Char,
48 Const,
49 Continue,
50 Default,
51 Do,
52 Double,
53 Else,
54 Enum,
55 Extern,
56 Float,
57 For,
58 Goto,
59 If,
60 Inline,
61 Int,
62 Long,
63 Register,
64 Restrict,
65 Return,
66 Short,
67 Signed,
68 Sizeof,
69 Static,
70 Struct,
71 Switch,
72 Typedef,
73 Union,
74 Unsigned,
75 Void,
76 Volatile,
77 While,
78 Bool,
79 Complex,
80 Imaginary,
81 // C11. Every one of these is spelled with a leading underscore, which C99
82 // already reserves, so they are recognised in every mode and the parser
83 // reports the ones a C99 block may not use — a friendlier answer than
84 // "expected a declaration, found identifier '_Static_assert'".
85 Alignas,
86 Alignof,
87 Atomic,
88 Generic,
89 Noreturn,
90 StaticAssert,
91 ThreadLocal,
92 BitInt,
93 // C23. These are ordinary identifiers before C23 — `<stdbool.h>` writes
94 // `#define bool _Bool`, and a C99 program may have a variable called
95 // `typeof` — so they are keywords only in a `c23!` block.
96 BoolName,
97 True,
98 False,
99 Nullptr,
100 Typeof,
101 TypeofUnqual,
102 Constexpr,
103 StaticAssertName,
104 AlignofName,
105 AlignasName,
106 ThreadLocalName,
107 // The GNU keywords. Every one of them is spelled with a leading double
108 // underscore, which C reserves, so they are available in every entry point
109 // — exactly as they are in GCC's own strict modes. They are *not* produced
110 // by [`Keyword::from_str`]: the [preprocessor](crate::pp) turns the
111 // identifiers into them on the way out, after macro replacement, so that
112 // `#define __attribute__(x)` — which portability headers really do write —
113 // still defines and expands a macro of that name.
114 /// `__attribute__`, `__attribute`
115 Attribute,
116 /// `__extension__`
117 Extension,
118 /// `__alignof__`, `__alignof`
119 AlignofGnu,
120 /// `__typeof__`, `__typeof`, and `typeof` in a GNU dialect
121 TypeofGnu,
122 /// `__typeof_unqual__`
123 TypeofUnqualGnu,
124 /// `__asm__`, `__asm`, and `asm` in a GNU dialect
125 Asm,
126 /// `__label__`
127 Label,
128 /// `__auto_type`
129 AutoType,
130 /// `__thread`
131 ThreadGnu,
132 /// `__int128`
133 ///
134 /// A type specifier of its own, which `signed` and `unsigned` combine
135 /// with; the `__int128_t` and `__uint128_t` spellings are `typedef` names
136 /// the compiler owns rather than keywords, exactly as they are in GCC.
137 Int128,
138 /// `__real__`, `__real`
139 RealGnu,
140 /// `__imag__`, `__imag`
141 ImagGnu,
142 /// `__inline__`, `__inline`
143 ///
144 /// A variant of its own because the plain spelling is C99's and `c89!`
145 /// gates it, while this one — being reserved — is available everywhere.
146 InlineGnu,
147 /// `__restrict__`, `__restrict`; see [`Keyword::InlineGnu`].
148 RestrictGnu,
149}
150
151impl Keyword {
152 /// The spelling of this keyword in C source.
153 pub fn as_str(self) -> &'static str {
154 use Keyword::*;
155 match self {
156 Auto => "auto",
157 Break => "break",
158 Case => "case",
159 Char => "char",
160 Const => "const",
161 Continue => "continue",
162 Default => "default",
163 Do => "do",
164 Double => "double",
165 Else => "else",
166 Enum => "enum",
167 Extern => "extern",
168 Float => "float",
169 For => "for",
170 Goto => "goto",
171 If => "if",
172 Inline => "inline",
173 Int => "int",
174 Long => "long",
175 Register => "register",
176 Restrict => "restrict",
177 Return => "return",
178 Short => "short",
179 Signed => "signed",
180 Sizeof => "sizeof",
181 Static => "static",
182 Struct => "struct",
183 Switch => "switch",
184 Typedef => "typedef",
185 Union => "union",
186 Unsigned => "unsigned",
187 Void => "void",
188 Volatile => "volatile",
189 While => "while",
190 Bool => "_Bool",
191 Complex => "_Complex",
192 Imaginary => "_Imaginary",
193 Alignas => "_Alignas",
194 Alignof => "_Alignof",
195 Atomic => "_Atomic",
196 Generic => "_Generic",
197 Noreturn => "_Noreturn",
198 StaticAssert => "_Static_assert",
199 ThreadLocal => "_Thread_local",
200 BitInt => "_BitInt",
201 BoolName => "bool",
202 True => "true",
203 False => "false",
204 Nullptr => "nullptr",
205 Typeof => "typeof",
206 TypeofUnqual => "typeof_unqual",
207 Constexpr => "constexpr",
208 StaticAssertName => "static_assert",
209 AlignofName => "alignof",
210 AlignasName => "alignas",
211 ThreadLocalName => "thread_local",
212 Attribute => "__attribute__",
213 Extension => "__extension__",
214 AlignofGnu => "__alignof__",
215 TypeofGnu => "__typeof__",
216 TypeofUnqualGnu => "__typeof_unqual__",
217 Asm => "__asm__",
218 Label => "__label__",
219 AutoType => "__auto_type",
220 ThreadGnu => "__thread",
221 Int128 => "__int128",
222 RealGnu => "__real__",
223 ImagGnu => "__imag__",
224 InlineGnu => "__inline__",
225 RestrictGnu => "__restrict__",
226 }
227 }
228
229 /// Whether this keyword is one of the GNU spellings the preprocessor
230 /// introduces; see the variants' own documentation.
231 pub fn is_gnu(self) -> bool {
232 use Keyword::*;
233 matches!(
234 self,
235 Attribute
236 | Extension
237 | AlignofGnu
238 | TypeofGnu
239 | TypeofUnqualGnu
240 | Asm
241 | Label
242 | AutoType
243 | ThreadGnu
244 | Int128
245 | RealGnu
246 | ImagGnu
247 | InlineGnu
248 | RestrictGnu
249 )
250 }
251
252 /// The revision that made this spelling a keyword.
253 ///
254 /// The C11 keywords are recognised in every mode — they are reserved
255 /// identifiers in C99, so nothing legal can be broken by it, and the
256 /// parser's "requires C11 or later" is a better answer than a syntax
257 /// error. The C23 ones are *not*: they are ordinary identifiers before
258 /// C23, and `<stdbool.h>`'s `#define bool _Bool` depends on it.
259 ///
260 /// The four C99 added — `inline`, `restrict`, `_Bool` and `_Complex` (with
261 /// `_Imaginary` beside it) — are recognised in every mode for the same
262 /// reason the C11 ones are, and gated where they are parsed; everything
263 /// else has been a keyword since C89.
264 pub fn since(self) -> Standard {
265 use Keyword::*;
266 match self {
267 Alignas | Alignof | Atomic | Generic | Noreturn | StaticAssert | ThreadLocal => {
268 Standard::C11
269 }
270 BitInt | BoolName | True | False | Nullptr | Typeof | TypeofUnqual | Constexpr
271 | StaticAssertName | AlignofName | AlignasName | ThreadLocalName => Standard::C23,
272 Inline | Restrict | Bool | Complex | Imaginary => Standard::C99,
273 _ => Standard::C89,
274 }
275 }
276
277 /// Looks a keyword up by spelling, honouring the language standard.
278 pub fn from_str(s: &str, standard: Standard) -> Option<Keyword> {
279 use Keyword::*;
280 if standard >= Standard::C23 {
281 let c23 = match s {
282 "bool" => Some(BoolName),
283 "true" => Some(True),
284 "false" => Some(False),
285 "nullptr" => Some(Nullptr),
286 "typeof" => Some(Typeof),
287 "typeof_unqual" => Some(TypeofUnqual),
288 "constexpr" => Some(Constexpr),
289 "static_assert" => Some(StaticAssertName),
290 "alignof" => Some(AlignofName),
291 "alignas" => Some(AlignasName),
292 "thread_local" => Some(ThreadLocalName),
293 _ => None,
294 };
295 if c23.is_some() {
296 return c23;
297 }
298 }
299 Some(match s {
300 "_Alignas" => Alignas,
301 "_Alignof" => Alignof,
302 "_Atomic" => Atomic,
303 "_Generic" => Generic,
304 "_Noreturn" => Noreturn,
305 "_Static_assert" => StaticAssert,
306 "_Thread_local" => ThreadLocal,
307 "_BitInt" => BitInt,
308 "auto" => Auto,
309 "break" => Break,
310 "case" => Case,
311 "char" => Char,
312 "const" => Const,
313 "continue" => Continue,
314 "default" => Default,
315 "do" => Do,
316 "double" => Double,
317 "else" => Else,
318 "enum" => Enum,
319 "extern" => Extern,
320 "float" => Float,
321 "for" => For,
322 "goto" => Goto,
323 "if" => If,
324 "inline" => Inline,
325 "int" => Int,
326 "long" => Long,
327 "register" => Register,
328 "restrict" => Restrict,
329 "return" => Return,
330 "short" => Short,
331 "signed" => Signed,
332 "sizeof" => Sizeof,
333 "static" => Static,
334 "struct" => Struct,
335 "switch" => Switch,
336 "typedef" => Typedef,
337 "union" => Union,
338 "unsigned" => Unsigned,
339 "void" => Void,
340 "volatile" => Volatile,
341 "while" => While,
342 "_Bool" => Bool,
343 "_Complex" => Complex,
344 "_Imaginary" => Imaginary,
345 _ => return None,
346 })
347 }
348}
349
350/// A C99 punctuator.
351///
352/// Digraphs are folded into the token they stand for: `<:` lexes as
353/// [`Punct::LBracket`], `%:%:` as [`Punct::HashHash`], and so on.
354#[derive(Clone, Copy, PartialEq, Eq, Debug, Hash)]
355#[allow(missing_docs)]
356pub enum Punct {
357 LBracket,
358 RBracket,
359 LParen,
360 RParen,
361 LBrace,
362 RBrace,
363 Dot,
364 Arrow,
365 PlusPlus,
366 MinusMinus,
367 Amp,
368 Star,
369 Plus,
370 Minus,
371 Tilde,
372 Bang,
373 Slash,
374 Percent,
375 Shl,
376 Shr,
377 Lt,
378 Gt,
379 Le,
380 Ge,
381 EqEq,
382 Ne,
383 Caret,
384 Pipe,
385 AmpAmp,
386 PipePipe,
387 Question,
388 Colon,
389 Semi,
390 Ellipsis,
391 Assign,
392 StarAssign,
393 SlashAssign,
394 PercentAssign,
395 PlusAssign,
396 MinusAssign,
397 ShlAssign,
398 ShrAssign,
399 AmpAssign,
400 CaretAssign,
401 PipeAssign,
402 Comma,
403 Hash,
404 HashHash,
405}
406
407impl Punct {
408 /// The canonical spelling of this punctuator (never the digraph form).
409 pub fn as_str(self) -> &'static str {
410 use Punct::*;
411 match self {
412 LBracket => "[",
413 RBracket => "]",
414 LParen => "(",
415 RParen => ")",
416 LBrace => "{",
417 RBrace => "}",
418 Dot => ".",
419 Arrow => "->",
420 PlusPlus => "++",
421 MinusMinus => "--",
422 Amp => "&",
423 Star => "*",
424 Plus => "+",
425 Minus => "-",
426 Tilde => "~",
427 Bang => "!",
428 Slash => "/",
429 Percent => "%",
430 Shl => "<<",
431 Shr => ">>",
432 Lt => "<",
433 Gt => ">",
434 Le => "<=",
435 Ge => ">=",
436 EqEq => "==",
437 Ne => "!=",
438 Caret => "^",
439 Pipe => "|",
440 AmpAmp => "&&",
441 PipePipe => "||",
442 Question => "?",
443 Colon => ":",
444 Semi => ";",
445 Ellipsis => "...",
446 Assign => "=",
447 StarAssign => "*=",
448 SlashAssign => "/=",
449 PercentAssign => "%=",
450 PlusAssign => "+=",
451 MinusAssign => "-=",
452 ShlAssign => "<<=",
453 ShrAssign => ">>=",
454 AmpAssign => "&=",
455 CaretAssign => "^=",
456 PipeAssign => "|=",
457 Comma => ",",
458 Hash => "#",
459 HashHash => "##",
460 }
461 }
462}
463
464/// Punctuator spellings, longest first so that a greedy match is also the
465/// maximal munch the standard asks for.
466const PUNCTUATORS: &[(&str, Punct)] = &[
467 ("%:%:", Punct::HashHash),
468 ("...", Punct::Ellipsis),
469 ("<<=", Punct::ShlAssign),
470 (">>=", Punct::ShrAssign),
471 ("->", Punct::Arrow),
472 ("++", Punct::PlusPlus),
473 ("--", Punct::MinusMinus),
474 ("<<", Punct::Shl),
475 (">>", Punct::Shr),
476 ("<=", Punct::Le),
477 (">=", Punct::Ge),
478 ("==", Punct::EqEq),
479 ("!=", Punct::Ne),
480 ("&&", Punct::AmpAmp),
481 ("||", Punct::PipePipe),
482 ("*=", Punct::StarAssign),
483 ("/=", Punct::SlashAssign),
484 ("%=", Punct::PercentAssign),
485 ("+=", Punct::PlusAssign),
486 ("-=", Punct::MinusAssign),
487 ("&=", Punct::AmpAssign),
488 ("^=", Punct::CaretAssign),
489 ("|=", Punct::PipeAssign),
490 ("##", Punct::HashHash),
491 ("<:", Punct::LBracket),
492 (":>", Punct::RBracket),
493 ("<%", Punct::LBrace),
494 ("%>", Punct::RBrace),
495 ("%:", Punct::Hash),
496 ("[", Punct::LBracket),
497 ("]", Punct::RBracket),
498 ("(", Punct::LParen),
499 (")", Punct::RParen),
500 ("{", Punct::LBrace),
501 ("}", Punct::RBrace),
502 (".", Punct::Dot),
503 ("&", Punct::Amp),
504 ("*", Punct::Star),
505 ("+", Punct::Plus),
506 ("-", Punct::Minus),
507 ("~", Punct::Tilde),
508 ("!", Punct::Bang),
509 ("/", Punct::Slash),
510 ("%", Punct::Percent),
511 ("<", Punct::Lt),
512 (">", Punct::Gt),
513 ("^", Punct::Caret),
514 ("|", Punct::Pipe),
515 ("?", Punct::Question),
516 (":", Punct::Colon),
517 (";", Punct::Semi),
518 ("=", Punct::Assign),
519 (",", Punct::Comma),
520 ("#", Punct::Hash),
521];
522
523/// The `l`/`ll` part of an integer suffix.
524#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
525pub enum LongKind {
526 /// No `l` suffix.
527 #[default]
528 None,
529 /// `l` or `L`.
530 Long,
531 /// `ll` or `LL`.
532 LongLong,
533}
534
535/// The base an integer constant was written in.
536#[derive(Clone, Copy, PartialEq, Eq, Debug)]
537pub enum NumBase {
538 /// `0b101` — C23.
539 Binary,
540 /// `0777`
541 Octal,
542 /// `42`
543 Decimal,
544 /// `0x2a`
545 Hex,
546}
547
548impl NumBase {
549 /// The radix.
550 pub fn radix(self) -> u32 {
551 match self {
552 NumBase::Binary => 2,
553 NumBase::Octal => 8,
554 NumBase::Decimal => 10,
555 NumBase::Hex => 16,
556 }
557 }
558
559 /// How a diagnostic names this base.
560 pub fn as_str(self) -> &'static str {
561 match self {
562 NumBase::Binary => "binary",
563 NumBase::Octal => "octal",
564 NumBase::Decimal => "decimal",
565 NumBase::Hex => "hexadecimal",
566 }
567 }
568}
569
570/// A decoded integer constant.
571#[derive(Clone, PartialEq, Eq, Debug)]
572pub struct IntLit {
573 /// The value, before any type is chosen for it.
574 pub value: u128,
575 /// How it was written.
576 pub base: NumBase,
577 /// Whether a `u`/`U` suffix was present.
578 pub unsigned: bool,
579 /// Whether an `l`/`ll` suffix was present.
580 pub long: LongKind,
581 /// The exact source spelling.
582 pub text: String,
583}
584
585/// The suffix of a floating constant.
586#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
587pub enum FloatSuffix {
588 /// No suffix: the constant has type `double`.
589 #[default]
590 None,
591 /// `f`/`F`: `float`.
592 Float,
593 /// `l`/`L`: `long double`.
594 LongDouble,
595}
596
597/// A decoded floating constant.
598#[derive(Clone, PartialEq, Debug)]
599pub struct FloatLit {
600 /// The value, rounded to `f64`.
601 pub value: f64,
602 /// The suffix, which fixes the constant's type.
603 pub suffix: FloatSuffix,
604 /// Whether the constant carried GNU's imaginary suffix, `i` or `j`.
605 ///
606 /// `2.0i` is `(0, 2)` of the complex type [`FloatLit::suffix`] names the
607 /// real part of, which is what makes `1.0 + 2.0i` read the way every C
608 /// program writes a complex constant. It is a GNU extension that C11
609 /// blessed by giving `<complex.h>` a `CMPLX` built on it; the *standard*
610 /// spelling of the same thing is `_Complex_I`.
611 pub imaginary: bool,
612 /// Whether the constant was written in hexadecimal form.
613 pub hex: bool,
614 /// The exact source spelling.
615 pub text: String,
616}
617
618/// A decoded character constant.
619#[derive(Clone, PartialEq, Eq, Debug)]
620pub struct CharLit {
621 /// The value of the constant.
622 ///
623 /// For a plain single-character constant this is the unsigned value of the
624 /// execution character (so `'\xff'` is `255`); whether that is later
625 /// interpreted as `-1` is up to sema, which knows the signedness of
626 /// `char`. Multi-character constants are packed big-endian, as GCC does.
627 pub value: i64,
628 /// Which prefix the constant was written with, which is what fixes its
629 /// type: `'x'` is an `int`, `L'x'` a `wchar_t`, `u'x'` a `char16_t`,
630 /// `U'x'` a `char32_t` and `u8'x'` a `char8_t`.
631 pub kind: StrKind,
632 /// The exact source spelling, including quotes.
633 pub text: String,
634}
635
636/// Which prefix a character constant or string literal was written with.
637///
638/// The five of them are the same set for both, which is why one enum serves
639/// both: `'x'`/`"…"`, `u8'x'`/`u8"…"`, `u'x'`/`u"…"`, `U'x'`/`U"…"` and
640/// `L'x'`/`L"…"`.
641#[derive(Clone, Copy, PartialEq, Eq, Debug)]
642pub enum StrKind {
643 /// `"…"` — bytes of the execution character set, which is UTF-8.
644 Narrow,
645 /// `u8"…"` (C11) and `u8'x'` (C23) — UTF-8 bytes, of type `char` before
646 /// C23 and `char8_t` (an `unsigned char`) from C23 on.
647 Utf8,
648 /// `u"…"` and `u'x'` (C11) — UTF-16 code units, of type `char16_t`.
649 Utf16,
650 /// `U"…"` and `U'x'` (C11) — UTF-32 code units, of type `char32_t`.
651 Utf32,
652 /// `L"…"` — `wchar_t`.
653 Wide,
654}
655
656impl StrKind {
657 /// The prefix a literal of this kind is written with.
658 pub fn prefix(self) -> &'static str {
659 match self {
660 StrKind::Narrow => "",
661 StrKind::Utf8 => "u8",
662 StrKind::Utf16 => "u",
663 StrKind::Utf32 => "U",
664 StrKind::Wide => "L",
665 }
666 }
667
668 /// The revision that introduced the prefix, in the form the constant is
669 /// being used in.
670 ///
671 /// `u8"…"` is C11 (N1488) and `u8'x'` is C23 (N2418); the other three
672 /// prefixes are the same revision either way.
673 pub fn since(self, character: bool) -> Standard {
674 match self {
675 StrKind::Narrow | StrKind::Wide => Standard::C89,
676 StrKind::Utf8 if character => Standard::C23,
677 StrKind::Utf8 | StrKind::Utf16 | StrKind::Utf32 => Standard::C11,
678 }
679 }
680
681 /// Whether the elements are the bytes of the source's UTF-8, rather than
682 /// character values.
683 fn is_bytes(self) -> bool {
684 matches!(self, StrKind::Narrow | StrKind::Utf8)
685 }
686
687 /// The largest value one element of this kind can hold.
688 ///
689 /// `wide_bits` is how wide the target's `wchar_t` is, which is the one
690 /// answer here that is not fixed by the language: 32 bits on the Unix
691 /// platforms and 16 on Windows, where `L"…"` is UTF-16 and a character
692 /// outside the basic multilingual plane takes a surrogate pair, exactly as
693 /// `u"…"` does.
694 fn max_element(self, wide_bits: u32) -> u32 {
695 match self {
696 StrKind::Narrow | StrKind::Utf8 => 0xff,
697 StrKind::Utf16 => 0xffff,
698 StrKind::Wide if wide_bits <= 16 => 0xffff,
699 StrKind::Utf32 | StrKind::Wide => u32::MAX,
700 }
701 }
702
703 /// Whether one element holds a UTF-16 code unit, so that a character
704 /// beyond the basic multilingual plane becomes a surrogate pair.
705 fn is_utf16(self, wide_bits: u32) -> bool {
706 self == StrKind::Utf16 || (self == StrKind::Wide && wide_bits <= 16)
707 }
708}
709
710/// A decoded string literal.
711#[derive(Clone, PartialEq, Eq, Debug)]
712pub struct StrLit {
713 /// Which prefix it was written with.
714 pub kind: StrKind,
715 /// The decoded elements: bytes (`0..=255`) for a narrow or UTF-8 literal,
716 /// UTF-16 code units (surrogate pairs and all) for a `u"…"` one, and
717 /// character values for `U"…"` and `L"…"`. The terminating NUL is *not*
718 /// included.
719 pub values: Vec<u32>,
720 /// The exact source spelling, including quotes.
721 pub text: String,
722}
723
724impl StrLit {
725 /// The literal's bytes, if it is an ordinary narrow one.
726 pub fn as_bytes(&self) -> Option<Vec<u8>> {
727 (self.kind == StrKind::Narrow).then(|| self.values.iter().map(|v| *v as u8).collect())
728 }
729}
730
731/// What a [`Token`] is.
732#[derive(Clone, PartialEq, Debug)]
733pub enum TokenKind {
734 /// End of the token list. Always present, exactly once, last.
735 Eof,
736 /// An identifier that is not a keyword.
737 Ident(String),
738 /// A keyword.
739 Keyword(Keyword),
740 /// An integer constant.
741 Int(IntLit),
742 /// A floating constant.
743 Float(FloatLit),
744 /// A character constant.
745 Char(CharLit),
746 /// A string literal.
747 Str(StrLit),
748 /// A punctuator.
749 Punct(Punct),
750 /// Text that is not a C token at all, kept so that a skipped group may
751 /// contain it. Holds its own spelling.
752 Error(String),
753}
754
755impl TokenKind {
756 /// A short description used in "expected …, found …" messages.
757 pub fn describe(&self) -> String {
758 match self {
759 TokenKind::Eof => "end of input".to_owned(),
760 TokenKind::Ident(name) => format!("identifier '{name}'"),
761 TokenKind::Keyword(k) => format!("keyword '{}'", k.as_str()),
762 TokenKind::Int(_) => "integer constant".to_owned(),
763 TokenKind::Float(_) => "floating constant".to_owned(),
764 TokenKind::Char(_) => "character constant".to_owned(),
765 TokenKind::Str(_) => "string literal".to_owned(),
766 TokenKind::Punct(p) => format!("'{}'", p.as_str()),
767 TokenKind::Error(text) => format!("'{text}'"),
768 }
769 }
770
771 /// The exact source spelling of this token.
772 ///
773 /// This is what `#` stringifies and what `##` pastes, so it has to be the
774 /// text as written — `0x1f` rather than `31`, `'\n'` rather than a
775 /// newline. [`TokenKind::Eof`] has no spelling and gives `""`.
776 pub fn spelling(&self) -> &str {
777 match self {
778 TokenKind::Eof => "",
779 TokenKind::Ident(name) => name,
780 TokenKind::Keyword(k) => k.as_str(),
781 TokenKind::Int(lit) => &lit.text,
782 TokenKind::Float(lit) => &lit.text,
783 TokenKind::Char(lit) => &lit.text,
784 TokenKind::Str(lit) => &lit.text,
785 TokenKind::Punct(p) => p.as_str(),
786 TokenKind::Error(text) => text,
787 }
788 }
789
790 /// The name this token has when it is used as a macro name.
791 ///
792 /// Keywords are ordinary identifiers during translation phase 4 — our
793 /// lexer classifies them early, so `#define restrict` and `#ifdef inline`
794 /// would otherwise be unusable — which is why a keyword answers with its
795 /// spelling here.
796 pub fn macro_name(&self) -> Option<&str> {
797 match self {
798 TokenKind::Ident(name) => Some(name),
799 TokenKind::Keyword(k) => Some(k.as_str()),
800 _ => None,
801 }
802 }
803}
804
805/// A lexed C token.
806#[derive(Clone, PartialEq, Debug)]
807pub struct Token {
808 /// What the token is.
809 pub kind: TokenKind,
810 /// Where it is.
811 pub range: SourceRange,
812 /// Whether it is the first token on its logical line (needed by the
813 /// preprocessor to spot directives).
814 pub bol: bool,
815 /// Whether whitespace or a comment preceded it (needed by the
816 /// preprocessor's stringification and macro replacement).
817 pub preceded_by_space: bool,
818 /// Everything wrong with this token, and with the text between it and the
819 /// token before it.
820 ///
821 /// The [preprocessor](crate::pp) reports what is wrong with the token
822 /// *itself* if and only if the token reaches its output. What stands
823 /// whatever becomes of the token — its spelling, and the comments that
824 /// preceded it — is reported as soon as the token is read from the file,
825 /// which is what keeps it from being lost when the token opens a directive,
826 /// names a macro or is an argument the macro drops; see
827 /// [`Diagnostic::lexical`].
828 pub errors: Vec<Diagnostic>,
829}
830
831impl Token {
832 /// The keyword this token is, if any.
833 pub fn keyword(&self) -> Option<Keyword> {
834 match &self.kind {
835 TokenKind::Keyword(k) => Some(*k),
836 _ => None,
837 }
838 }
839
840 /// Whether this token is the given punctuator.
841 pub fn is_punct(&self, p: Punct) -> bool {
842 self.kind == TokenKind::Punct(p)
843 }
844
845 /// Whether this token is the given keyword.
846 pub fn is_keyword(&self, k: Keyword) -> bool {
847 self.kind == TokenKind::Keyword(k)
848 }
849
850 /// Whether this token ends the input.
851 pub fn is_eof(&self) -> bool {
852 self.kind == TokenKind::Eof
853 }
854
855 /// The identifier this token is, if any.
856 pub fn ident(&self) -> Option<&str> {
857 match &self.kind {
858 TokenKind::Ident(name) => Some(name),
859 _ => None,
860 }
861 }
862}
863
864/// Knobs for the lexer.
865#[derive(Clone, Copy, Debug)]
866pub struct LexOptions {
867 /// Which standard's lexical rules to apply.
868 pub standard: Standard,
869 /// How a constant form a newer revision introduced is gated.
870 pub gating: crate::Gating,
871 /// Accept `$` in identifiers, like GCC's `-fdollars-in-identifiers`, which
872 /// is on by default; see [`crate::Options::dollar_in_identifiers`].
873 pub dollar_in_identifiers: bool,
874 /// Whether translation phase 1 replaces the nine trigraphs.
875 ///
876 /// See [`trigraphs_enabled`] for who has them and why.
877 pub trigraphs: bool,
878 /// How wide the target's `wchar_t` is.
879 ///
880 /// The one target property the *lexer* has an opinion about: it decides
881 /// what an `L'\xffff'` escape may hold and whether `L"😀"` is one element
882 /// or a surrogate pair, since Windows makes `wchar_t` 16 bits.
883 pub wchar_bits: u32,
884 /// Whether the complex types are available, which is what decides whether
885 /// an imaginary constant (`2.0i`) has a type; see
886 /// [`crate::Options::complex`].
887 pub complex: bool,
888}
889
890impl LexOptions {
891 /// The lexical rules of `standard`, in the strict ISO dialect.
892 pub fn new(standard: Standard) -> Self {
893 Self {
894 standard,
895 gating: crate::Gating {
896 standard,
897 dialect: crate::Dialect::Iso,
898 },
899 dollar_in_identifiers: true,
900 trigraphs: trigraphs_enabled(standard, crate::Dialect::Iso),
901 wchar_bits: crate::TargetModel::host().wchar_bits,
902 complex: crate::COMPLEX_SUPPORTED,
903 }
904 }
905}
906
907impl From<&Options> for LexOptions {
908 fn from(o: &Options) -> Self {
909 Self {
910 standard: o.standard,
911 gating: o.gating(),
912 dollar_in_identifiers: o.dollar_in_identifiers,
913 trigraphs: trigraphs_enabled(o.standard, o.dialect),
914 wchar_bits: o.target.wchar_bits,
915 complex: o.complex,
916 }
917 }
918}
919
920/// Whether translation phase 1 replaces trigraphs in this entry point.
921///
922/// The nine of them were in C from the beginning and C23 removed them
923/// (N2940), so every strict entry point below `c23!` has them and `c23!` does
924/// not. No *GNU* dialect has them: `gcc -std=gnu99` switches them off, because
925/// `"what??!"` in a string is far more likely to be an exclamation than a
926/// pipe, and that is the line Clang draws too.
927pub fn trigraphs_enabled(standard: Standard, dialect: crate::Dialect) -> bool {
928 standard < Standard::C23 && !dialect.is_gnu()
929}
930
931/// The nine trigraphs of C 5.2.1.1, as `(third character, replacement)`.
932const TRIGRAPHS: &[(u8, u8)] = &[
933 (b'=', b'#'),
934 (b'(', b'['),
935 (b'/', b'\\'),
936 (b')', b']'),
937 (b'\'', b'^'),
938 (b'<', b'{'),
939 (b'!', b'|'),
940 (b'>', b'}'),
941 (b'-', b'~'),
942];
943
944/// Lexes the root file of `source`.
945///
946/// The returned vector always ends with a [`TokenKind::Eof`] token. Problems
947/// are attached to the tokens they were found in ([`Token::errors`]) rather
948/// than reported; scanning always runs to the end of the input.
949pub fn lex(source: &Source, options: &Options) -> Vec<Token> {
950 let file = source.map.file(source.root);
951 lex_text(file.text(), file.base(), &options.into())
952}
953
954/// Lexes `text`, whose first byte lives at global offset `base`.
955pub fn lex_text(text: &str, base: Pos, options: &LexOptions) -> Vec<Token> {
956 Lexer {
957 text,
958 bytes: text.as_bytes(),
959 base,
960 pos: 0,
961 options: *options,
962 pending: Vec::new(),
963 }
964 .run()
965}
966
967struct Lexer<'a> {
968 text: &'a str,
969 bytes: &'a [u8],
970 base: Pos,
971 pos: usize,
972 options: LexOptions,
973 /// Problems found since the last token was finished; they belong to the
974 /// token currently being scanned.
975 pending: Vec<Diagnostic>,
976}
977
978impl<'a> Lexer<'a> {
979 fn range(&self, start: usize, end: usize) -> SourceRange {
980 SourceRange::new(self.base + start as Pos, self.base + end as Pos)
981 }
982
983 /// Records a problem with the token being scanned.
984 fn error(&mut self, range: SourceRange, message: impl Into<String>) {
985 self.pending.push(Diagnostic::error(range, message));
986 }
987
988 /// Records a problem that stands whether or not the token being scanned
989 /// ever reaches the parser: one with its *spelling*, or one in the text
990 /// between it and the token before it. See [`Diagnostic::lexical`].
991 fn lexical_error(&mut self, range: SourceRange, message: impl Into<String>) {
992 self.pending
993 .push(Diagnostic::error(range, message).at_lexing());
994 }
995
996 /// Records an advisory remark about the token being scanned.
997 fn warning(&mut self, range: SourceRange, message: impl Into<String>) {
998 self.pending.push(Diagnostic::warning(range, message));
999 }
1000
1001 fn peek(&self) -> Option<u8> {
1002 self.bytes.get(self.pos).copied()
1003 }
1004
1005 fn peek_at(&self, n: usize) -> Option<u8> {
1006 self.bytes.get(self.pos + n).copied()
1007 }
1008
1009 fn eof(&self) -> bool {
1010 self.pos >= self.bytes.len()
1011 }
1012
1013 fn run(mut self) -> Vec<Token> {
1014 let mut tokens = Vec::new();
1015 let mut bol = true;
1016 let mut space = false;
1017 loop {
1018 let (saw_newline, saw_space) = self.skip_whitespace();
1019 bol |= saw_newline;
1020 space |= saw_space || saw_newline;
1021 if self.eof() {
1022 let end = self.bytes.len();
1023 tokens.push(Token {
1024 kind: TokenKind::Eof,
1025 range: self.range(end, end),
1026 bol,
1027 preceded_by_space: space,
1028 errors: std::mem::take(&mut self.pending),
1029 });
1030 break;
1031 }
1032 let start = self.pos;
1033 let kind = self.scan_token();
1034 tokens.push(Token {
1035 kind,
1036 range: self.range(start, self.pos),
1037 bol,
1038 preceded_by_space: space,
1039 errors: std::mem::take(&mut self.pending),
1040 });
1041 bol = false;
1042 space = false;
1043 }
1044 tokens
1045 }
1046
1047 /// Skips whitespace, comments and line splices.
1048 ///
1049 /// Returns `(saw_newline, saw_space)`.
1050 fn skip_whitespace(&mut self) -> (bool, bool) {
1051 let mut newline = false;
1052 let mut space = false;
1053 loop {
1054 match self.peek() {
1055 Some(b'\n') => {
1056 self.pos += 1;
1057 newline = true;
1058 }
1059 Some(b' ' | b'\t' | b'\r' | 0x0b | 0x0c) => {
1060 self.pos += 1;
1061 space = true;
1062 }
1063 // Translation phase 2: a backslash immediately followed by a
1064 // newline splices the two lines, so it is *not* a line break.
1065 // `??/` is that backslash where trigraphs are on.
1066 Some(b'\\' | b'?') if self.is_line_splice(self.pos) => {
1067 self.pos += self.line_splice_len(self.pos);
1068 space = true;
1069 }
1070 Some(b'/') if self.peek_at(1) == Some(b'*') => {
1071 // Translation phase 3 replaces the whole comment with one
1072 // space, so a newline *inside* it is not a line break at
1073 // all: a directive may span one, and a `#` after one does
1074 // not start a directive. Both are what GCC does, and the
1075 // standard's own `FUNC_LIKE` example in 6.10.3 depends on
1076 // the first.
1077 let start = self.pos;
1078 self.pos += 2;
1079 let mut closed = false;
1080 while let Some(c) = self.peek() {
1081 if c == b'*' && self.peek_at(1) == Some(b'/') {
1082 self.pos += 2;
1083 closed = true;
1084 break;
1085 }
1086 self.pos += 1;
1087 }
1088 if !closed {
1089 // Both of the problems a *comment* can have belong to
1090 // the token that follows it for want of anywhere else
1091 // to put them, and neither is about that token: whether
1092 // a comment is terminated, and whether this revision
1093 // has `//`, is settled in translation phase 3 by the
1094 // text alone. So they are recorded as lexical, and
1095 // reported wherever the token ends up — including
1096 // nowhere, which is what a `#` opening a directive and
1097 // a macro name do with it.
1098 let range = self.range(start, self.bytes.len());
1099 self.lexical_error(range, "unterminated comment");
1100 }
1101 space = true;
1102 }
1103 Some(b'/') if self.peek_at(1) == Some(b'/') => {
1104 let start = self.pos;
1105 self.pos += 2;
1106 while let Some(c) = self.peek() {
1107 if c == b'\n' {
1108 break;
1109 }
1110 if matches!(c, b'\\' | b'?') && self.is_line_splice(self.pos) {
1111 self.pos += self.line_splice_len(self.pos);
1112 continue;
1113 }
1114 self.pos += 1;
1115 }
1116 // C99 took the `//` comment from C++ (N644); before that
1117 // `a //* b */ c` was a division, which is why the gate is
1118 // here rather than being a warning. Lexical for the reason
1119 // above: a `c89!` block that writes one has to be told so
1120 // whatever follows it.
1121 if let Some(message) = self
1122 .options
1123 .gating
1124 .requires("a '//' comment", Standard::C99)
1125 {
1126 let range = self.range(start, self.pos);
1127 self.lexical_error(range, message);
1128 }
1129 space = true;
1130 }
1131 _ => return (newline, space),
1132 }
1133 }
1134 }
1135
1136 fn is_line_splice(&self, at: usize) -> bool {
1137 self.line_splice_len(at) > 0
1138 }
1139
1140 /// Length of a `\`-newline splice starting at `at`, or 0.
1141 ///
1142 /// Translation phase 1 runs *before* phase 2, so `??/` at the end of a
1143 /// line splices it exactly as a written backslash does — which is the one
1144 /// trigraph whose replacement is not a character the lexer can simply hand
1145 /// on.
1146 fn line_splice_len(&self, at: usize) -> usize {
1147 let lead = match self.trigraph_at(at) {
1148 Some(b'\\') => 3,
1149 Some(_) => return 0,
1150 None if self.bytes.get(at) == Some(&b'\\') => 1,
1151 None => return 0,
1152 };
1153 match (self.bytes.get(at + lead), self.bytes.get(at + lead + 1)) {
1154 (Some(b'\n'), _) => lead + 1,
1155 (Some(b'\r'), Some(b'\n')) => lead + 2,
1156 _ => 0,
1157 }
1158 }
1159
1160 /// The character a trigraph at `at` stands for, if there is one there.
1161 ///
1162 /// C 5.2.1.1: the nine three-character sequences beginning `??` are
1163 /// replaced in translation phase 1, before line splicing and before the
1164 /// source is split into tokens — so this is consulted from everywhere the
1165 /// lexer looks at a raw byte, rather than the text being rewritten. Not
1166 /// rewriting it is what keeps every [`SourceRange`] a range of the source
1167 /// the user really wrote: a diagnostic about `??=` points at all three
1168 /// characters.
1169 fn trigraph_at(&self, at: usize) -> Option<u8> {
1170 if !self.options.trigraphs
1171 || self.bytes.get(at) != Some(&b'?')
1172 || self.bytes.get(at + 1) != Some(&b'?')
1173 {
1174 return None;
1175 }
1176 let third = *self.bytes.get(at + 2)?;
1177 TRIGRAPHS
1178 .iter()
1179 .find(|(c, _)| *c == third)
1180 .map(|(_, replacement)| *replacement)
1181 }
1182
1183 /// The length in source bytes of the character at `at`, which is three for
1184 /// a trigraph and one otherwise.
1185 fn trigraph_len(&self, at: usize) -> usize {
1186 if self.trigraph_at(at).is_some() { 3 } else { 1 }
1187 }
1188
1189 /// The character-constant or string-literal prefix starting here, with its
1190 /// length in bytes.
1191 ///
1192 /// A prefix is only one when a quote follows it, which is what keeps
1193 /// `unsigned`, `u8x` and a variable called `U` ordinary identifiers.
1194 fn literal_prefix(&self, first: u8) -> Option<(StrKind, usize)> {
1195 let (kind, len) = match first {
1196 b'L' => (StrKind::Wide, 1),
1197 b'U' => (StrKind::Utf32, 1),
1198 b'u' if self.peek_at(1) == Some(b'8') => (StrKind::Utf8, 2),
1199 b'u' => (StrKind::Utf16, 1),
1200 _ => return None,
1201 };
1202 matches!(self.peek_at(len), Some(b'"' | b'\'')).then_some((kind, len))
1203 }
1204
1205 /// Whether an identifier that begins with an extended character starts
1206 /// here.
1207 fn extended_ident_start(&self) -> bool {
1208 match self.peek() {
1209 Some(b'\\') if matches!(self.peek_at(1), Some(b'u' | b'U')) => {
1210 let want = if self.peek_at(1) == Some(b'u') { 4 } else { 8 };
1211 (0..want).all(|i| self.peek_at(2 + i).is_some_and(|c| c.is_ascii_hexdigit()))
1212 }
1213 Some(c) if c >= 0x80 => self.text[self.pos..]
1214 .chars()
1215 .next()
1216 .is_some_and(|ch| is_extended_ident_char(ch, true)),
1217 _ => false,
1218 }
1219 }
1220
1221 /// Scans one token, which is [`TokenKind::Error`] for text that is not a C
1222 /// token at all.
1223 fn scan_token(&mut self) -> TokenKind {
1224 let Some(c) = self.peek() else {
1225 return TokenKind::Eof;
1226 };
1227 if is_ident_start(c, self.options.dollar_in_identifiers) {
1228 // `L'x'`, `u8"…"`, `u'x'` and the rest are constants rather than an
1229 // identifier followed by one. The prefix is recognised in every
1230 // entry point and *gated* rather than not recognised at all: a
1231 // `c99!` block that writes `u"x"` is told which macro has it,
1232 // instead of being told that `u` is undeclared.
1233 if let Some((kind, len)) = self.literal_prefix(c) {
1234 let start = self.pos;
1235 self.pos += len;
1236 let character = self.peek() == Some(b'\'');
1237 if let Some(message) = self.options.gating.requires(
1238 &format!("a '{}' literal", kind.prefix()),
1239 kind.since(character),
1240 ) {
1241 let range = self.range(start, self.pos);
1242 self.error(range, message);
1243 }
1244 return if character {
1245 self.scan_char_constant(kind)
1246 } else {
1247 self.scan_string_literal(kind)
1248 };
1249 }
1250 return self.scan_ident();
1251 }
1252 // An extended identifier, written either as the character itself —
1253 // which GCC and Clang have taken since GCC 10 — or as the universal
1254 // character name C99 6.4.2.1 introduced for it.
1255 if self.extended_ident_start() {
1256 return self.scan_ident();
1257 }
1258 if c.is_ascii_digit() || (c == b'.' && self.peek_at(1).is_some_and(|d| d.is_ascii_digit()))
1259 {
1260 return self.scan_number();
1261 }
1262 if c == b'\'' {
1263 return self.scan_char_constant(StrKind::Narrow);
1264 }
1265 if c == b'"' {
1266 return self.scan_string_literal(StrKind::Narrow);
1267 }
1268 if let Some(p) = self.scan_punctuator() {
1269 return TokenKind::Punct(p);
1270 }
1271
1272 // Anything else is not a C token at all.
1273 let start = self.pos;
1274 // `??/` that does not splice a line is the stray backslash a written
1275 // one would be, and is reported as one rather than as two question
1276 // marks.
1277 let ch = match self.trigraph_at(start) {
1278 Some(c) => {
1279 self.pos += 3;
1280 c as char
1281 }
1282 None => {
1283 let ch = self.text[start..].chars().next().unwrap_or('\u{fffd}');
1284 self.pos += ch.len_utf8();
1285 ch
1286 }
1287 };
1288 let range = self.range(start, self.pos);
1289 self.error(
1290 range,
1291 format!("unexpected character '{}' in program", ch.escape_debug()),
1292 );
1293 TokenKind::Error(ch.to_string())
1294 }
1295
1296 /// Scans an identifier or a keyword.
1297 ///
1298 /// Translation phase 2 deletes a backslash-newline *before* the source is
1299 /// split into tokens, so one may sit in the middle of an identifier:
1300 /// `__LI\<newline>NE__` is `__LINE__`, and Clang's own `drs/dr464.c`
1301 /// writes exactly that. Almost no identifier has one, so the spelling
1302 /// stays a slice of the source until the first splice is found and only
1303 /// then becomes a `String`.
1304 fn scan_ident(&mut self) -> TokenKind {
1305 let start = self.pos;
1306 // The text before the current splice or universal character name, when
1307 // there has been one.
1308 let mut spliced: Option<String> = None;
1309 // Where the run of characters that is still a slice begins.
1310 let mut segment = start;
1311 // Whether anything outside the basic character set was written, which
1312 // is the only case that has to be checked for normalization.
1313 let mut extended = false;
1314 loop {
1315 match self.peek() {
1316 Some(c) if c < 0x80 && is_ident_continue(c, self.options.dollar_in_identifiers) => {
1317 self.pos += 1;
1318 }
1319 // Only a splice that the identifier *continues* over: one at
1320 // the end of it is whitespace, and belongs to whatever comes
1321 // next.
1322 Some(b'\\' | b'?')
1323 if self.line_splice_len(self.pos) > 0
1324 && self
1325 .bytes
1326 .get(self.pos + self.line_splice_len(self.pos))
1327 .is_some_and(|c| {
1328 is_ident_continue(*c, self.options.dollar_in_identifiers)
1329 }) =>
1330 {
1331 let text = spliced.get_or_insert_with(String::new);
1332 text.push_str(&self.text[segment..self.pos]);
1333 self.pos += self.line_splice_len(self.pos);
1334 segment = self.pos;
1335 }
1336 // A universal character name spells one extended character:
1337 // `café` and `café` are the same identifier, which is
1338 // exactly what C99 6.4.2.1 says.
1339 Some(b'\\') if matches!(self.peek_at(1), Some(b'u' | b'U')) => {
1340 let at = self.pos;
1341 let Some(ch) = self.scan_ident_ucn(at == start) else {
1342 break;
1343 };
1344 let text = spliced.get_or_insert_with(String::new);
1345 text.push_str(&self.text[segment..at]);
1346 text.push(ch);
1347 segment = self.pos;
1348 extended = true;
1349 }
1350 Some(c) if c >= 0x80 => {
1351 let ch = self.text[self.pos..].chars().next().unwrap_or('\u{fffd}');
1352 if !is_extended_ident_char(ch, self.pos == start) {
1353 break;
1354 }
1355 self.pos += ch.len_utf8();
1356 extended = true;
1357 }
1358 _ => break,
1359 }
1360 }
1361 let text = match spliced {
1362 Some(mut text) => {
1363 text.push_str(&self.text[segment..self.pos]);
1364 text
1365 }
1366 None => self.text[start..self.pos].to_owned(),
1367 };
1368 // Nothing was an identifier character after all — the whole of it was
1369 // one bad universal character name, which has been reported. Something
1370 // has to be consumed, or the scanner would sit here forever.
1371 if text.is_empty() {
1372 if self.pos == start {
1373 let ch = self.text[start..].chars().next().unwrap_or('\u{fffd}');
1374 self.pos += ch.len_utf8();
1375 }
1376 return TokenKind::Error(self.text[start..self.pos].to_owned());
1377 }
1378 // Rust identifiers have to be in Normalization Form C, and `rustc`
1379 // *normalises* the ones a procedural macro hands it rather than
1380 // refusing them — so two C identifiers that differ only by
1381 // normalization would silently become one Rust item. C23 (N2836) asks
1382 // for NFC as well, so refusing is both the safe answer and the
1383 // conforming one.
1384 if extended && !unicode_normalization::is_nfc(&text) {
1385 let range = self.range(start, self.pos);
1386 self.error(
1387 range,
1388 format!(
1389 "identifier '{text}' is not in Unicode Normalization Form C; \
1390 write the composed form"
1391 ),
1392 );
1393 }
1394 match Keyword::from_str(&text, self.options.standard) {
1395 Some(k) => TokenKind::Keyword(k),
1396 None => TokenKind::Ident(text),
1397 }
1398 }
1399
1400 /// Reads a `\uXXXX` or `\UXXXXXXXX` written inside an identifier.
1401 ///
1402 /// A name that is too short to be one leaves the position where it was, so
1403 /// that the identifier simply ends there and the backslash is reported by
1404 /// [`Lexer::scan_token`] as the stray character it is. One that is the
1405 /// right shape but names something an identifier may not hold is *always*
1406 /// consumed, and reported: leaving it would be a second diagnostic about
1407 /// the same text, and — where it is the first character of the identifier
1408 /// — a token that consumed nothing at all.
1409 fn scan_ident_ucn(&mut self, start: bool) -> Option<char> {
1410 let at = self.pos;
1411 let want = if self.peek_at(1) == Some(b'u') { 4 } else { 8 };
1412 let mut value: u32 = 0;
1413 for i in 0..want {
1414 let digit = self.peek_at(2 + i).and_then(|c| (c as char).to_digit(16))?;
1415 value = value * 16 + digit;
1416 }
1417 let spelling = if want == 4 { 'u' } else { 'U' };
1418 let ch = char::from_u32(value);
1419 if !ch.is_some_and(|ch| is_extended_ident_char(ch, start)) {
1420 self.pos += 2 + want;
1421 let range = self.range(at, self.pos);
1422 let digits = if want == 4 {
1423 format!("{value:04X}")
1424 } else {
1425 format!("{value:08X}")
1426 };
1427 let message =
1428 format!("'\\{spelling}{digits}' is not a valid character in an identifier");
1429 // Two different refusals wear the same words. Naming a character
1430 // below U+00A0, or a surrogate, is what C99 6.4.3p2 forbids of the
1431 // *name* — C23 puts `$`, `@` and `` ` `` in the basic character
1432 // set, and Clang refuses those in every mode — so it is ill formed
1433 // where it is written and is reported even where the token goes
1434 // nowhere: Clang's own `C99/n717.c` is a file of them, each
1435 // written as the argument of a macro that expands to nothing.
1436 // Everything else here — a value outside Unicode, a character
1437 // that simply is not `XID_Continue` — is about the *identifier*,
1438 // and a preprocessing token that never reaches the parser is
1439 // allowed to be one C has no other use for (6.4p3).
1440 if value < 0xA0 || (0xD800..=0xDFFF).contains(&value) {
1441 self.lexical_error(range, message);
1442 } else {
1443 self.error(range, message);
1444 }
1445 return None;
1446 }
1447 if let Some(message) = self
1448 .options
1449 .gating
1450 .requires("a universal character name", Standard::C99)
1451 {
1452 let range = self.range(at, at + 2 + want);
1453 self.error(range, message);
1454 }
1455 self.pos += 2 + want;
1456 ch
1457 }
1458
1459 fn scan_punctuator(&mut self) -> Option<Punct> {
1460 if self.trigraph_at(self.pos).is_some() {
1461 return self.scan_trigraph_punctuator();
1462 }
1463 let rest = &self.text[self.pos..];
1464 for (spelling, punct) in PUNCTUATORS {
1465 if rest.starts_with(spelling) {
1466 // `<:` is a digraph for `[`, but `1<::x` must not be mangled;
1467 // C99 has no `::`, so a plain greedy match is correct here.
1468 self.pos += spelling.len();
1469 return Some(*punct);
1470 }
1471 }
1472 None
1473 }
1474
1475 /// A punctuator that begins with a trigraph.
1476 ///
1477 /// Phase 1 happens before tokens exist, so a punctuator may be spelled
1478 /// partly or wholly with trigraphs — `??!??!` is `||`, `??'=` is `^=`,
1479 /// `??=??=` is `##` — and maximal munch applies to the *replaced*
1480 /// characters. Up to the length of the longest punctuator is replaced into
1481 /// a small buffer, matched there, and the source position advanced by what
1482 /// the matched characters really cost.
1483 fn scan_trigraph_punctuator(&mut self) -> Option<Punct> {
1484 const LONGEST: usize = 4;
1485 let mut logical = [0u8; LONGEST];
1486 let mut widths = [0usize; LONGEST];
1487 let mut count = 0;
1488 let mut at = self.pos;
1489 while count < LONGEST {
1490 let (c, width) = match self.trigraph_at(at) {
1491 Some(c) => (c, 3),
1492 None => match self.bytes.get(at) {
1493 Some(c) => (*c, 1),
1494 None => break,
1495 },
1496 };
1497 // A backslash is not part of any punctuator, and neither is
1498 // anything outside the basic character set.
1499 if c == b'\\' || !c.is_ascii() {
1500 break;
1501 }
1502 logical[count] = c;
1503 widths[count] = width;
1504 count += 1;
1505 at += width;
1506 }
1507 let text = std::str::from_utf8(&logical[..count]).ok()?;
1508 for (spelling, punct) in PUNCTUATORS {
1509 if text.starts_with(spelling) {
1510 self.pos += widths[..spelling.len()].iter().sum::<usize>();
1511 return Some(*punct);
1512 }
1513 }
1514 None
1515 }
1516
1517 // -- numbers ------------------------------------------------------------
1518
1519 /// Scans a preprocessing number and classifies it as integer or float.
1520 ///
1521 /// Scanning the whole pp-number first (rather than stopping at the first
1522 /// character that does not fit) is what lets `08` or `1.0q` be reported as
1523 /// one bad token instead of two good ones.
1524 fn scan_number(&mut self) -> TokenKind {
1525 let start = self.pos;
1526 if self.peek() == Some(b'.') {
1527 self.pos += 1;
1528 }
1529 self.pos += 1;
1530 let mut separators = false;
1531 while let Some(c) = self.peek() {
1532 if matches!(c, b'e' | b'E' | b'p' | b'P')
1533 && matches!(self.peek_at(1), Some(b'+') | Some(b'-'))
1534 {
1535 self.pos += 2;
1536 continue;
1537 }
1538 // C23's digit separator. It is part of the pp-number when a digit
1539 // or a nondigit follows it, which is what keeps `1'a'` from
1540 // swallowing the character constant that follows a constant.
1541 if c == b'\''
1542 && self
1543 .peek_at(1)
1544 .is_some_and(|d| d.is_ascii_alphanumeric() || d == b'_')
1545 {
1546 separators = true;
1547 self.pos += 1;
1548 continue;
1549 }
1550 if c.is_ascii_alphanumeric()
1551 || c == b'_'
1552 || c == b'.'
1553 || (c == b'$' && self.options.dollar_in_identifiers)
1554 {
1555 self.pos += 1;
1556 continue;
1557 }
1558 break;
1559 }
1560 let text = &self.text[start..self.pos];
1561 let range = self.range(start, self.pos);
1562 if separators
1563 && let Some(message) = self
1564 .options
1565 .gating
1566 .requires("a digit separator", Standard::C23)
1567 {
1568 self.error(range, message);
1569 }
1570 // Everything below reads the digits; the separators are not part of
1571 // the value, while `text` keeps the spelling `#` has to reproduce.
1572 let stripped: String;
1573 let digits = if separators {
1574 stripped = text.replace('\'', "");
1575 stripped.as_str()
1576 } else {
1577 text
1578 };
1579 let lower = digits.to_ascii_lowercase();
1580 let hex = lower.starts_with("0x");
1581 let is_float = if hex {
1582 digits.contains('.') || lower[2..].contains('p')
1583 } else {
1584 digits.contains('.') || (!lower.starts_with("0b") && lower.contains('e'))
1585 };
1586 if is_float {
1587 TokenKind::Float(self.decode_float(digits, text, range, hex))
1588 } else {
1589 TokenKind::Int(self.decode_int(digits, text, range))
1590 }
1591 }
1592
1593 /// Decodes an integer constant from its separator-free `digits`; `text` is
1594 /// the spelling as written, which is what diagnostics quote and what `#`
1595 /// reproduces.
1596 fn decode_int(&mut self, digits: &str, text: &str, range: SourceRange) -> IntLit {
1597 let bytes = digits.as_bytes();
1598 let (base, digits_start) =
1599 if digits.len() >= 2 && (bytes[1] | 0x20) == b'x' && bytes[0] == b'0' {
1600 (NumBase::Hex, 2)
1601 } else if digits.len() >= 2 && (bytes[1] | 0x20) == b'b' && bytes[0] == b'0' {
1602 (NumBase::Binary, 2)
1603 } else if bytes[0] == b'0' && digits.len() > 1 {
1604 (NumBase::Octal, 1)
1605 } else {
1606 (NumBase::Decimal, 0)
1607 };
1608 if base == NumBase::Binary
1609 && let Some(message) = self
1610 .options
1611 .gating
1612 .requires("a binary integer constant", Standard::C23)
1613 {
1614 self.error(range, message);
1615 }
1616
1617 let radix = base.radix();
1618 // Accept every decimal digit in an octal or binary constant so that
1619 // the whole constant is consumed and reported once, the way C
1620 // compilers do.
1621 let scan_radix = if radix < 10 { 10 } else { radix };
1622
1623 let mut i = digits_start;
1624 let mut value: u128 = 0;
1625 let mut overflow = false;
1626 let mut bad_digit: Option<char> = None;
1627 while i < bytes.len() {
1628 let c = bytes[i] as char;
1629 let digit = match c.to_digit(scan_radix) {
1630 Some(d) => d,
1631 None => break,
1632 };
1633 if digit >= radix && bad_digit.is_none() {
1634 bad_digit = Some(c);
1635 }
1636 match value
1637 .checked_mul(radix as u128)
1638 .and_then(|v| v.checked_add(digit as u128))
1639 {
1640 Some(v) => value = v,
1641 None => overflow = true,
1642 }
1643 i += 1;
1644 }
1645
1646 if i == digits_start && matches!(base, NumBase::Hex | NumBase::Binary) {
1647 let prefix = &digits[..2];
1648 self.error(
1649 range,
1650 format!(
1651 "expected digits after '{prefix}' in {} constant",
1652 base.as_str()
1653 ),
1654 );
1655 }
1656 if let Some(c) = bad_digit {
1657 self.error(
1658 range,
1659 format!("invalid digit '{c}' in {} constant '{text}'", base.as_str()),
1660 );
1661 }
1662 if overflow {
1663 self.error(range, format!("integer constant '{text}' is too large"));
1664 }
1665
1666 let suffix = &digits[i..];
1667 let (unsigned, long) = match parse_int_suffix(suffix) {
1668 Some(v) => v,
1669 None => {
1670 // `3i` is GCC's `_Complex int`, one of the two complex
1671 // *integer* types that are a GNU extension of their own and
1672 // that cinrs does not have. Saying what to write instead is
1673 // more use than "invalid suffix".
1674 if matches!(suffix, "i" | "j" | "I" | "J") {
1675 let digits = text.strip_suffix(suffix).unwrap_or(text);
1676 self.error(
1677 range,
1678 format!(
1679 "'{text}' is a complex integer constant, which is a GNU extension \
1680 cinrs does not support; write '{digits}.0{suffix}' for the \
1681 complex floating constant"
1682 ),
1683 );
1684 } else {
1685 self.error(
1686 range,
1687 format!("invalid suffix '{suffix}' on integer constant '{text}'"),
1688 );
1689 }
1690 (false, LongKind::None)
1691 }
1692 };
1693
1694 IntLit {
1695 value,
1696 base,
1697 unsigned,
1698 long,
1699 text: text.to_owned(),
1700 }
1701 }
1702
1703 fn decode_float(
1704 &mut self,
1705 digits: &str,
1706 text: &str,
1707 range: SourceRange,
1708 hex: bool,
1709 ) -> FloatLit {
1710 let (body, suffix) = split_float_suffix(digits, hex);
1711 let (suffix_kind, imaginary) = self.float_suffix(suffix, text, range);
1712
1713 if hex
1714 && let Some(message) = self
1715 .options
1716 .gating
1717 .requires("a hexadecimal floating constant", Standard::C99)
1718 {
1719 self.error(range, message);
1720 }
1721 let value = if hex {
1722 match parse_hex_float(body) {
1723 Some(v) => v,
1724 None => {
1725 self.error(
1726 range,
1727 format!(
1728 "invalid hexadecimal floating constant '{text}'; \
1729 a 'p' exponent is required"
1730 ),
1731 );
1732 0.0
1733 }
1734 }
1735 } else {
1736 match parse_decimal_float(body) {
1737 Some(v) => v,
1738 None => {
1739 self.error(range, format!("invalid floating constant '{text}'"));
1740 0.0
1741 }
1742 }
1743 };
1744
1745 FloatLit {
1746 value,
1747 suffix: suffix_kind,
1748 imaginary,
1749 hex,
1750 text: text.to_owned(),
1751 }
1752 }
1753
1754 /// The type a floating constant's suffix gives it.
1755 ///
1756 /// C has three: none, `f` and `l`. GCC has a dozen more, and they fall
1757 /// into three groups here.
1758 ///
1759 /// * **The ones that name a format wider than `double`** — `d`, `w`
1760 /// (`__float80`), `q` (`__float128`), and the `_FloatN` and `_FloatNx`
1761 /// suffixes `f64`, `f64x`, `f32x` and `f128`. Every one of them is
1762 /// `double` in this implementation, exactly as `long double` is, so each
1763 /// is accepted in a GNU dialect and **loses precision** where GCC would
1764 /// not; `doc/gnu-extensions.md` records that. `f32` is `float`.
1765 /// * **The decimal floating suffixes** `df`, `dd` and `dl`, whose types
1766 /// are radix-10 and have no Rust counterpart at all.
1767 /// * **The imaginary suffixes** `i` and `j`, which make the constant an
1768 /// imaginary one — `2.0i` is `(0, 2)` — and so need the complex types.
1769 ///
1770 /// The decimal ones are refused with the reason, and so are the imaginary
1771 /// ones when the complex types are switched off. A strict entry point
1772 /// refuses the first group too, naming the GNU entry point that has it —
1773 /// the suffixes are spelled without underscores, which is the line
1774 /// [`crate::Dialect`] draws.
1775 ///
1776 /// The second half of the answer is whether the constant is imaginary; see
1777 /// [`FloatLit::imaginary`].
1778 fn float_suffix(
1779 &mut self,
1780 suffix: &str,
1781 text: &str,
1782 range: SourceRange,
1783 ) -> (FloatSuffix, bool) {
1784 let lower = suffix.to_ascii_lowercase();
1785 match lower.as_str() {
1786 "" => return (FloatSuffix::None, false),
1787 "f" => return (FloatSuffix::Float, false),
1788 "l" => return (FloatSuffix::LongDouble, false),
1789 _ => {}
1790 }
1791 // GNU's imaginary suffix, on its own or beside `f` or `l`. Both
1792 // spellings are the same thing: `j` is what Fortran and engineering
1793 // habit write, and GCC takes either.
1794 let imaginary = match lower.as_str() {
1795 "i" | "j" => Some(FloatSuffix::None),
1796 "if" | "fi" | "jf" | "fj" => Some(FloatSuffix::Float),
1797 "il" | "li" | "jl" | "lj" => Some(FloatSuffix::LongDouble),
1798 _ => None,
1799 };
1800 if let Some(kind) = imaginary {
1801 if !self.options.complex {
1802 self.error(
1803 range,
1804 format!(
1805 "invalid suffix '{suffix}' on floating constant '{text}': an \
1806 imaginary constant needs _Complex. {}",
1807 crate::COMPLEX_UNSUPPORTED
1808 ),
1809 );
1810 return (FloatSuffix::None, false);
1811 }
1812 if let Some(message) = self
1813 .options
1814 .gating
1815 .requires("an imaginary constant", Standard::C99)
1816 {
1817 self.error(range, message);
1818 return (FloatSuffix::None, false);
1819 }
1820 return (kind, true);
1821 }
1822 // A decimal constant is refused whatever the entry point: it has no
1823 // type this crate can give it.
1824 let refusal = match lower.as_str() {
1825 "df" | "dd" | "dl" => Some(
1826 "the decimal floating types (_Decimal32, _Decimal64, _Decimal128) are not \
1827 supported: they are radix-10 and no Rust type is",
1828 ),
1829 "f16" | "f16x" | "bf16" => Some(
1830 "'_Float16' is not supported: Rust's `f16` is unstable, and rounding the \
1831 constant to a wider type would change what the program computes",
1832 ),
1833 _ => None,
1834 };
1835 if let Some(reason) = refusal {
1836 self.error(
1837 range,
1838 format!("invalid suffix '{suffix}' on floating constant '{text}': {reason}"),
1839 );
1840 return (FloatSuffix::None, false);
1841 }
1842 // The rest are the GNU widths. `f32` is `float`; every other one names
1843 // a format this implementation makes a `double`.
1844 let wider = matches!(
1845 lower.as_str(),
1846 "d" | "w" | "q" | "f64" | "f64x" | "f32x" | "f128" | "f128x"
1847 );
1848 if !wider && lower != "f32" {
1849 self.error(
1850 range,
1851 format!("invalid suffix '{suffix}' on floating constant '{text}'"),
1852 );
1853 return (FloatSuffix::None, false);
1854 }
1855 if !self.options.gating.dialect.is_gnu() {
1856 let gnu = self.options.gating.standard.macro_name_in(Dialect::Gnu);
1857 let here = self
1858 .options
1859 .gating
1860 .standard
1861 .macro_name_in(self.options.gating.dialect);
1862 self.error(
1863 range,
1864 format!(
1865 "the suffix '{suffix}' on a floating constant is a GNU extension, and \
1866 requires a GNU dialect ({gnu}) (this block is {here})"
1867 ),
1868 );
1869 return (FloatSuffix::None, false);
1870 }
1871 if lower == "f32" {
1872 return (FloatSuffix::Float, false);
1873 }
1874 // `LongDouble` is `double`, which is what all of these come to.
1875 (FloatSuffix::LongDouble, false)
1876 }
1877
1878 // -- character and string constants -------------------------------------
1879
1880 fn scan_char_constant(&mut self, kind: StrKind) -> TokenKind {
1881 let start = self.pos;
1882 debug_assert_eq!(self.peek(), Some(b'\''));
1883 self.pos += 1;
1884 let mut values: Vec<u32> = Vec::new();
1885 let mut terminated = false;
1886 // How many *characters* were written, which is not how many elements
1887 // they came to: one `\U0001F600` is one character and two UTF-16 code
1888 // units, and the two say different things about what is wrong.
1889 let mut characters = 0usize;
1890 while let Some(c) = self.peek() {
1891 if c == b'\'' {
1892 self.pos += 1;
1893 terminated = true;
1894 break;
1895 }
1896 if c == b'\n' {
1897 break;
1898 }
1899 let before = values.len();
1900 self.read_char_element(kind, &mut values);
1901 if values.len() > before {
1902 characters += 1;
1903 }
1904 }
1905 let range = self.range(start, self.pos);
1906 if !terminated {
1907 self.error(range, "missing terminating \' character");
1908 }
1909 if values.is_empty() {
1910 self.error(range, "empty character constant");
1911 }
1912 // Only `'ab'` and `L'ab'` have an implementation-defined meaning; C11
1913 // 6.4.4.4p2 makes more than one character in a `u8`, `u` or `U`
1914 // constant a constraint violation, because there is no room for a
1915 // second one in the type — and so is one character that needs more
1916 // than one code unit, which is what `u8'é'` and `u'😀'` are.
1917 if values.len() > 1 {
1918 match kind {
1919 StrKind::Narrow | StrKind::Wide => {
1920 self.warning(range, "multi-character character constant");
1921 }
1922 _ if characters > 1 => {
1923 self.error(
1924 range,
1925 format!(
1926 "a '{}' character constant holds exactly one character",
1927 kind.prefix()
1928 ),
1929 );
1930 }
1931 _ => {
1932 self.error(
1933 range,
1934 format!(
1935 "the character in a '{}' character constant must fit in a \
1936 single code unit",
1937 kind.prefix()
1938 ),
1939 );
1940 }
1941 }
1942 }
1943
1944 let value = if kind != StrKind::Narrow {
1945 values.last().copied().unwrap_or(0) as i64
1946 } else if values.len() <= 1 {
1947 values.first().copied().unwrap_or(0) as i64
1948 } else {
1949 // GCC packs the bytes big-endian into an `int`.
1950 let mut v: u32 = 0;
1951 for b in &values {
1952 v = (v << 8) | (*b & 0xff);
1953 }
1954 v as i32 as i64
1955 };
1956
1957 TokenKind::Char(CharLit {
1958 value,
1959 kind,
1960 text: self.text[start..self.pos].to_owned(),
1961 })
1962 }
1963
1964 fn scan_string_literal(&mut self, kind: StrKind) -> TokenKind {
1965 let start = self.pos;
1966 debug_assert_eq!(self.peek(), Some(b'"'));
1967 self.pos += 1;
1968 let mut values: Vec<u32> = Vec::new();
1969 let mut terminated = false;
1970 while let Some(c) = self.peek() {
1971 if c == b'"' {
1972 self.pos += 1;
1973 terminated = true;
1974 break;
1975 }
1976 if c == b'\n' {
1977 break;
1978 }
1979 self.read_char_element(kind, &mut values);
1980 }
1981 let range = self.range(start, self.pos);
1982 if !terminated {
1983 self.error(range, "missing terminating \" character");
1984 }
1985 TokenKind::Str(StrLit {
1986 kind,
1987 values,
1988 text: self.text[start..self.pos].to_owned(),
1989 })
1990 }
1991
1992 /// Reads one element of a character constant or string literal, appending
1993 /// its decoded value(s) to `out`.
1994 fn read_char_element(&mut self, kind: StrKind, out: &mut Vec<u32>) {
1995 if self.is_line_splice(self.pos) {
1996 self.pos += self.line_splice_len(self.pos);
1997 return;
1998 }
1999 let start = self.pos;
2000 // Phase 1 replaces trigraphs inside literals too: `"??!"` is `"|"`,
2001 // and `"??/n"` is `"\n"`.
2002 let trigraph = self.trigraph_at(self.pos);
2003 if trigraph != Some(b'\\') && self.peek() != Some(b'\\') {
2004 match trigraph {
2005 Some(c) => {
2006 self.pos += 3;
2007 out.push(u32::from(c));
2008 }
2009 None if kind.is_bytes() => {
2010 // A narrow or UTF-8 literal keeps the raw
2011 // execution-charset bytes, so UTF-8 text in one survives
2012 // byte for byte.
2013 self.pos += 1;
2014 out.push(self.bytes[start] as u32);
2015 }
2016 None => {
2017 let ch = self.text[start..].chars().next().unwrap_or('\u{fffd}');
2018 self.pos += ch.len_utf8();
2019 push_character(ch as u32, kind, self.options.wchar_bits, out);
2020 }
2021 }
2022 return;
2023 }
2024
2025 self.pos += self.trigraph_len(self.pos);
2026 let e = match self.trigraph_at(self.pos) {
2027 Some(c) => c,
2028 None => match self.peek() {
2029 Some(c) => c,
2030 None => {
2031 let range = self.range(start, self.pos);
2032 self.error(range, "incomplete escape sequence");
2033 return;
2034 }
2035 },
2036 };
2037 self.pos += self.trigraph_len(self.pos);
2038 let simple = match e {
2039 b'\'' => Some(0x27),
2040 b'"' => Some(0x22),
2041 b'?' => Some(0x3f),
2042 b'\\' => Some(0x5c),
2043 b'a' => Some(0x07),
2044 b'b' => Some(0x08),
2045 // `\e` is GNU's escape for ESC. GCC accepts it in every mode (with
2046 // a pedantic warning), and `"\e[0m"` is how a program writes a
2047 // terminal colour; refusing it would be refusing the extension in
2048 // the one place a strict mode still has it.
2049 b'e' => Some(0x1b),
2050 b'f' => Some(0x0c),
2051 b'n' => Some(0x0a),
2052 b'r' => Some(0x0d),
2053 b't' => Some(0x09),
2054 b'v' => Some(0x0b),
2055 _ => None,
2056 };
2057 if let Some(v) = simple {
2058 out.push(v);
2059 return;
2060 }
2061 match e {
2062 b'0'..=b'7' => {
2063 let mut v: u32 = (e - b'0') as u32;
2064 for _ in 0..2 {
2065 match self.peek() {
2066 Some(d @ b'0'..=b'7') => {
2067 v = v * 8 + (d - b'0') as u32;
2068 self.pos += 1;
2069 }
2070 _ => break,
2071 }
2072 }
2073 self.push_escape_value(v, kind, start, out);
2074 }
2075 b'x' => {
2076 let mut v: u32 = 0;
2077 let mut any = false;
2078 let mut overflow = false;
2079 while let Some(d) = self.peek().and_then(|c| (c as char).to_digit(16)) {
2080 any = true;
2081 v = match v.checked_mul(16).and_then(|v| v.checked_add(d)) {
2082 Some(v) => v,
2083 None => {
2084 overflow = true;
2085 v
2086 }
2087 };
2088 self.pos += 1;
2089 }
2090 let range = self.range(start, self.pos);
2091 if !any {
2092 self.error(range, "'\\x' used with no following hex digits");
2093 } else if overflow {
2094 self.error(range, "hex escape sequence out of range");
2095 }
2096 self.push_escape_value(v, kind, start, out);
2097 }
2098 b'u' | b'U' => {
2099 if let Some(message) = self
2100 .options
2101 .gating
2102 .requires("a universal character name", Standard::C99)
2103 {
2104 let range = self.range(start, self.pos);
2105 self.error(range, message);
2106 }
2107 let want = if e == b'u' { 4 } else { 8 };
2108 let mut v: u32 = 0;
2109 let mut count = 0;
2110 while count < want {
2111 match self.peek().and_then(|c| (c as char).to_digit(16)) {
2112 Some(d) => {
2113 v = v.wrapping_mul(16).wrapping_add(d);
2114 self.pos += 1;
2115 count += 1;
2116 }
2117 None => break,
2118 }
2119 }
2120 let range = self.range(start, self.pos);
2121 if count != want {
2122 self.error(
2123 range,
2124 format!("incomplete universal character name; expected {want} hex digits"),
2125 );
2126 return;
2127 }
2128 match char::from_u32(v) {
2129 Some(ch) if kind.is_bytes() => {
2130 // The execution character set is UTF-8.
2131 let mut buf = [0u8; 4];
2132 for b in ch.encode_utf8(&mut buf).as_bytes() {
2133 out.push(*b as u32);
2134 }
2135 }
2136 Some(ch) => push_character(ch as u32, kind, self.options.wchar_bits, out),
2137 None => {
2138 // Outside Unicode, or a surrogate: 6.4.3p2 again, and
2139 // the *literal token's* spelling is what is wrong, so
2140 // this stands wherever the token ends up.
2141 self.lexical_error(
2142 range,
2143 format!("'\\u{v:04X}' is not a valid universal character name"),
2144 );
2145 }
2146 }
2147 }
2148 _ => {
2149 let range = self.range(start, self.pos);
2150 self.error(
2151 range,
2152 format!("unknown escape sequence '\\{}'", (e as char).escape_debug()),
2153 );
2154 out.push(e as u32);
2155 }
2156 }
2157 }
2158
2159 /// Pushes the value of a numeric escape sequence, which unlike a character
2160 /// is *not* re-encoded: `u"\xd83d"` is that one code unit.
2161 fn push_escape_value(&mut self, v: u32, kind: StrKind, start: usize, out: &mut Vec<u32>) {
2162 let max = kind.max_element(self.options.wchar_bits);
2163 if v > max {
2164 let range = self.range(start, self.pos);
2165 let ty = match kind {
2166 StrKind::Narrow => "char",
2167 StrKind::Utf8 => "char8_t",
2168 StrKind::Utf16 => "char16_t",
2169 StrKind::Utf32 => "char32_t",
2170 StrKind::Wide => "wchar_t",
2171 };
2172 self.error(
2173 range,
2174 format!("escape sequence out of range for type '{ty}'"),
2175 );
2176 out.push(v & max);
2177 } else {
2178 out.push(v);
2179 }
2180 }
2181}
2182
2183/// Appends one character, encoded the way `kind` stores its elements.
2184///
2185/// A `u"…"` literal holds UTF-16 code units, so a character outside the basic
2186/// multilingual plane becomes the two halves of a surrogate pair — which is
2187/// what makes `sizeof(u"\U0001F600")` six rather than four. On a target whose
2188/// `wchar_t` is 16 bits wide, `L"…"` is UTF-16 too and does the same.
2189fn push_character(value: u32, kind: StrKind, wchar_bits: u32, out: &mut Vec<u32>) {
2190 if !kind.is_utf16(wchar_bits) || value <= 0xffff {
2191 out.push(value);
2192 return;
2193 }
2194 let v = value - 0x1_0000;
2195 out.push(0xd800 + (v >> 10));
2196 out.push(0xdc00 + (v & 0x3ff));
2197}
2198
2199fn is_ident_start(c: u8, dollar: bool) -> bool {
2200 c.is_ascii_alphabetic() || c == b'_' || (dollar && c == b'$')
2201}
2202
2203fn is_ident_continue(c: u8, dollar: bool) -> bool {
2204 c.is_ascii_alphanumeric() || c == b'_' || (dollar && c == b'$')
2205}
2206
2207/// Whether an extended character may appear in an identifier.
2208///
2209/// C99 Annex D listed the ranges by hand, C11 revised the list, and C23 (N2836,
2210/// N2939) replaced all of it with Unicode Annex #31's `XID_Start` and
2211/// `XID_Continue` — which is also what Rust's own identifiers are, and what
2212/// makes a C name usable as a Rust one. The one list is used in every entry
2213/// point: the earlier annexes are approximations of the same intent, and a
2214/// program that uses a character C11 left out is one this would otherwise
2215/// refuse for no reason a user could act on.
2216///
2217/// A character of the basic character set is never one of these: the ASCII
2218/// path has already decided about it, and a universal character name is not
2219/// allowed to spell one (6.4.3p2).
2220fn is_extended_ident_char(ch: char, start: bool) -> bool {
2221 if ch.is_ascii() {
2222 return false;
2223 }
2224 if start {
2225 unicode_ident::is_xid_start(ch)
2226 } else {
2227 unicode_ident::is_xid_continue(ch)
2228 }
2229}
2230
2231/// Validates an integer suffix, returning `(unsigned, long_kind)`.
2232fn parse_int_suffix(s: &str) -> Option<(bool, LongKind)> {
2233 if s.is_empty() {
2234 return Some((false, LongKind::None));
2235 }
2236 let b = s.as_bytes();
2237 let mut i = 0;
2238 let mut unsigned = false;
2239 let mut long = LongKind::None;
2240
2241 if b[i] == b'u' || b[i] == b'U' {
2242 unsigned = true;
2243 i += 1;
2244 }
2245 if i < b.len() && (b[i] == b'l' || b[i] == b'L') {
2246 // `ll` and `LL` must not be mixed as `lL` or `Ll`.
2247 if i + 1 < b.len() && b[i + 1] == b[i] {
2248 long = LongKind::LongLong;
2249 i += 2;
2250 } else {
2251 long = LongKind::Long;
2252 i += 1;
2253 }
2254 }
2255 if !unsigned && i < b.len() && (b[i] == b'u' || b[i] == b'U') {
2256 unsigned = true;
2257 i += 1;
2258 }
2259 (i == b.len()).then_some((unsigned, long))
2260}
2261
2262/// Splits a floating constant into its numeric body and its suffix.
2263///
2264/// The *body* is scanned forward rather than the suffix backwards, because a
2265/// suffix may hold digits of its own: `1.0f128` names `_Float128` and the
2266/// `128` is no part of the number. What is left after the digits, the point
2267/// and the exponent is the suffix, whatever it looks like; naming it is
2268/// [`Lexer::float_suffix`]'s business.
2269fn split_float_suffix(text: &str, hex: bool) -> (&str, &str) {
2270 let b = text.as_bytes();
2271 let mut i = 0;
2272 let (exponent, digit): (u8, fn(u8) -> bool) = if hex {
2273 i = 2; // the `0x` the caller has already recognised
2274 (b'p', |c| c.is_ascii_hexdigit())
2275 } else {
2276 (b'e', |c| c.is_ascii_digit())
2277 };
2278 while i < b.len() && (digit(b[i]) || b[i] == b'.') {
2279 i += 1;
2280 }
2281 if i < b.len() && b[i] | 0x20 == exponent {
2282 i += 1;
2283 if i < b.len() && (b[i] == b'+' || b[i] == b'-') {
2284 i += 1;
2285 }
2286 while i < b.len() && b[i].is_ascii_digit() {
2287 i += 1;
2288 }
2289 }
2290 (&text[..i], &text[i..])
2291}
2292
2293/// Parses the numeric body of a decimal floating constant.
2294fn parse_decimal_float(body: &str) -> Option<f64> {
2295 if body.is_empty() {
2296 return None;
2297 }
2298 // `None` is "no exponent at all", which is a different thing from an `e`
2299 // with nothing after it — `1e` is not a constant.
2300 let (mantissa, exponent) = match body.find(['e', 'E']) {
2301 Some(i) => (&body[..i], Some(&body[i + 1..])),
2302 None => (body, None),
2303 };
2304 let (int_part, frac_part) = match mantissa.find('.') {
2305 Some(i) => (&mantissa[..i], &mantissa[i + 1..]),
2306 None => (mantissa, ""),
2307 };
2308 if int_part.is_empty() && frac_part.is_empty() {
2309 return None;
2310 }
2311 if !int_part.bytes().all(|c| c.is_ascii_digit())
2312 || !frac_part.bytes().all(|c| c.is_ascii_digit())
2313 {
2314 return None;
2315 }
2316 let exponent = match exponent {
2317 None => 0i32,
2318 Some(exponent) => {
2319 let (sign, digits) = match exponent.as_bytes().first() {
2320 Some(b'+') => (1, &exponent[1..]),
2321 Some(b'-') => (-1, &exponent[1..]),
2322 _ => (1, exponent),
2323 };
2324 if digits.is_empty() || !digits.bytes().all(|c| c.is_ascii_digit()) {
2325 return None;
2326 }
2327 // Saturate: an absurd exponent simply becomes 0 or infinity.
2328 sign * digits.parse::<i32>().unwrap_or(i32::MAX / 2)
2329 }
2330 };
2331 let normalized = format!(
2332 "{}.{}e{}",
2333 if int_part.is_empty() { "0" } else { int_part },
2334 if frac_part.is_empty() { "0" } else { frac_part },
2335 exponent
2336 );
2337 normalized.parse::<f64>().ok()
2338}
2339
2340/// Parses the numeric body of a hexadecimal floating constant (`0x1.8p3`).
2341fn parse_hex_float(body: &str) -> Option<f64> {
2342 let rest = body
2343 .strip_prefix("0x")
2344 .or_else(|| body.strip_prefix("0X"))?;
2345 let p = rest.find(['p', 'P'])?;
2346 let (mantissa, exponent) = (&rest[..p], &rest[p + 1..]);
2347 let (int_part, frac_part) = match mantissa.find('.') {
2348 Some(i) => (&mantissa[..i], &mantissa[i + 1..]),
2349 None => (mantissa, ""),
2350 };
2351 if int_part.is_empty() && frac_part.is_empty() {
2352 return None;
2353 }
2354 let mut value = 0f64;
2355 for c in int_part.chars() {
2356 value = value * 16.0 + c.to_digit(16)? as f64;
2357 }
2358 let mut scale = 1.0 / 16.0;
2359 for c in frac_part.chars() {
2360 value += c.to_digit(16)? as f64 * scale;
2361 scale /= 16.0;
2362 }
2363 let (sign, digits) = match exponent.as_bytes().first() {
2364 Some(b'+') => (1i32, &exponent[1..]),
2365 Some(b'-') => (-1i32, &exponent[1..]),
2366 _ => (1i32, exponent),
2367 };
2368 if digits.is_empty() || !digits.bytes().all(|c| c.is_ascii_digit()) {
2369 return None;
2370 }
2371 let exp = sign * digits.parse::<i32>().unwrap_or(i32::MAX / 2);
2372 Some(value * 2f64.powi(exp))
2373}