cinrs_core/lex.rs
1//! A complete C99 lexer working on the captured source text.
2//!
3//! The lexer is byte-offset based: every [`Token`] carries a
4//! [`SourceRange`] into the [`SourceMap`](crate::capture::SourceMap), which is
5//! what makes exact diagnostics possible. Constants are decoded here (value,
6//! base, suffix, escape sequences) so that neither the parser nor sema has to
7//! look at raw text again.
8//!
9//! Two flags on each token — [`Token::bol`] and [`Token::preceded_by_space`] —
10//! are what the [preprocessor](crate::pp) recognises directives with and what
11//! its stringification operator reproduces: directive recognition needs "first
12//! token on a line", and `#` needs to know where whitespace was.
13//!
14//! # The lexer never reports anything
15//!
16//! Lexical errors never stop the lexer, and they never reach [`Diagnostics`]
17//! either: each one is attached to the token it was found in, in
18//! [`Token::errors`], and the preprocessor reports the ones whose token
19//! survives into its output. That is not a detail — a group skipped by
20//! `#if 0` may legally hold text that is not C at all, and a macro that is
21//! never invoked may hold anything its author liked:
22//!
23//! ```c
24//! #if 0
25//! this is not C: 08, 'unterminated, @@@
26//! #endif
27//! ```
28//!
29//! Text that is not a token at all becomes a [`TokenKind::Error`] token
30//! carrying its own spelling, so that it too can be skipped rather than
31//! reported. The preprocessor drops those tokens after reporting them, so
32//! nothing downstream ever sees one.
33//!
34//! [`Diagnostics`]: crate::diag::Diagnostics
35
36use crate::capture::{Pos, Source, SourceRange};
37use crate::diag::Diagnostic;
38use crate::{Dialect, Options, Standard};
39
40/// A C99 keyword.
41#[derive(Clone, Copy, PartialEq, Eq, Debug, Hash)]
42#[allow(missing_docs)]
43pub enum Keyword {
44 Auto,
45 Break,
46 Case,
47 Char,
48 Const,
49 Continue,
50 Default,
51 Do,
52 Double,
53 Else,
54 Enum,
55 Extern,
56 Float,
57 For,
58 Goto,
59 If,
60 Inline,
61 Int,
62 Long,
63 Register,
64 Restrict,
65 Return,
66 Short,
67 Signed,
68 Sizeof,
69 Static,
70 Struct,
71 Switch,
72 Typedef,
73 Union,
74 Unsigned,
75 Void,
76 Volatile,
77 While,
78 Bool,
79 Complex,
80 Imaginary,
81 // C11. Every one of these is spelled with a leading underscore, which C99
82 // already reserves, so they are recognised in every mode and the parser
83 // reports the ones a C99 block may not use — a friendlier answer than
84 // "expected a declaration, found identifier '_Static_assert'".
85 Alignas,
86 Alignof,
87 Atomic,
88 Generic,
89 Noreturn,
90 StaticAssert,
91 ThreadLocal,
92 BitInt,
93 // C23. These are ordinary identifiers before C23 — `<stdbool.h>` writes
94 // `#define bool _Bool`, and a C99 program may have a variable called
95 // `typeof` — so they are keywords only in a `c23!` block.
96 BoolName,
97 True,
98 False,
99 Nullptr,
100 Typeof,
101 TypeofUnqual,
102 Constexpr,
103 StaticAssertName,
104 AlignofName,
105 AlignasName,
106 ThreadLocalName,
107 // The GNU keywords. Every one of them is spelled with a leading double
108 // underscore, which C reserves, so they are available in every entry point
109 // — exactly as they are in GCC's own strict modes. They are *not* produced
110 // by [`Keyword::from_str`]: the [preprocessor](crate::pp) turns the
111 // identifiers into them on the way out, after macro replacement, so that
112 // `#define __attribute__(x)` — which portability headers really do write —
113 // still defines and expands a macro of that name.
114 /// `__attribute__`, `__attribute`
115 Attribute,
116 /// `__extension__`
117 Extension,
118 /// `__alignof__`, `__alignof`
119 AlignofGnu,
120 /// `__typeof__`, `__typeof`, and `typeof` in a GNU dialect
121 TypeofGnu,
122 /// `__typeof_unqual__`
123 TypeofUnqualGnu,
124 /// `__asm__`, `__asm`, and `asm` in a GNU dialect
125 Asm,
126 /// `__label__`
127 Label,
128 /// `__auto_type`
129 AutoType,
130 /// `__thread`
131 ThreadGnu,
132 /// `__int128`
133 ///
134 /// A type specifier of its own, which `signed` and `unsigned` combine
135 /// with; the `__int128_t` and `__uint128_t` spellings are `typedef` names
136 /// the compiler owns rather than keywords, exactly as they are in GCC.
137 Int128,
138 /// `__real__`, `__real`
139 RealGnu,
140 /// `__imag__`, `__imag`
141 ImagGnu,
142 /// `__inline__`, `__inline`
143 ///
144 /// A variant of its own because the plain spelling is C99's and `c89!`
145 /// gates it, while this one — being reserved — is available everywhere.
146 InlineGnu,
147 /// `__restrict__`, `__restrict`; see [`Keyword::InlineGnu`].
148 RestrictGnu,
149}
150
151impl Keyword {
152 /// The spelling of this keyword in C source.
153 pub fn as_str(self) -> &'static str {
154 use Keyword::*;
155 match self {
156 Auto => "auto",
157 Break => "break",
158 Case => "case",
159 Char => "char",
160 Const => "const",
161 Continue => "continue",
162 Default => "default",
163 Do => "do",
164 Double => "double",
165 Else => "else",
166 Enum => "enum",
167 Extern => "extern",
168 Float => "float",
169 For => "for",
170 Goto => "goto",
171 If => "if",
172 Inline => "inline",
173 Int => "int",
174 Long => "long",
175 Register => "register",
176 Restrict => "restrict",
177 Return => "return",
178 Short => "short",
179 Signed => "signed",
180 Sizeof => "sizeof",
181 Static => "static",
182 Struct => "struct",
183 Switch => "switch",
184 Typedef => "typedef",
185 Union => "union",
186 Unsigned => "unsigned",
187 Void => "void",
188 Volatile => "volatile",
189 While => "while",
190 Bool => "_Bool",
191 Complex => "_Complex",
192 Imaginary => "_Imaginary",
193 Alignas => "_Alignas",
194 Alignof => "_Alignof",
195 Atomic => "_Atomic",
196 Generic => "_Generic",
197 Noreturn => "_Noreturn",
198 StaticAssert => "_Static_assert",
199 ThreadLocal => "_Thread_local",
200 BitInt => "_BitInt",
201 BoolName => "bool",
202 True => "true",
203 False => "false",
204 Nullptr => "nullptr",
205 Typeof => "typeof",
206 TypeofUnqual => "typeof_unqual",
207 Constexpr => "constexpr",
208 StaticAssertName => "static_assert",
209 AlignofName => "alignof",
210 AlignasName => "alignas",
211 ThreadLocalName => "thread_local",
212 Attribute => "__attribute__",
213 Extension => "__extension__",
214 AlignofGnu => "__alignof__",
215 TypeofGnu => "__typeof__",
216 TypeofUnqualGnu => "__typeof_unqual__",
217 Asm => "__asm__",
218 Label => "__label__",
219 AutoType => "__auto_type",
220 ThreadGnu => "__thread",
221 Int128 => "__int128",
222 RealGnu => "__real__",
223 ImagGnu => "__imag__",
224 InlineGnu => "__inline__",
225 RestrictGnu => "__restrict__",
226 }
227 }
228
229 /// Whether this keyword is one of the GNU spellings the preprocessor
230 /// introduces; see the variants' own documentation.
231 pub fn is_gnu(self) -> bool {
232 use Keyword::*;
233 matches!(
234 self,
235 Attribute
236 | Extension
237 | AlignofGnu
238 | TypeofGnu
239 | TypeofUnqualGnu
240 | Asm
241 | Label
242 | AutoType
243 | ThreadGnu
244 | Int128
245 | RealGnu
246 | ImagGnu
247 | InlineGnu
248 | RestrictGnu
249 )
250 }
251
252 /// The revision that made this spelling a keyword.
253 ///
254 /// The C11 keywords are recognised in every mode — they are reserved
255 /// identifiers in C99, so nothing legal can be broken by it, and the
256 /// parser's "requires C11 or later" is a better answer than a syntax
257 /// error. The C23 ones are *not*: they are ordinary identifiers before
258 /// C23, and `<stdbool.h>`'s `#define bool _Bool` depends on it.
259 ///
260 /// The four C99 added — `inline`, `restrict`, `_Bool` and `_Complex` (with
261 /// `_Imaginary` beside it) — are recognised in every mode for the same
262 /// reason the C11 ones are, and gated where they are parsed; everything
263 /// else has been a keyword since C89.
264 pub fn since(self) -> Standard {
265 use Keyword::*;
266 match self {
267 Alignas | Alignof | Atomic | Generic | Noreturn | StaticAssert | ThreadLocal => {
268 Standard::C11
269 }
270 BitInt | BoolName | True | False | Nullptr | Typeof | TypeofUnqual | Constexpr
271 | StaticAssertName | AlignofName | AlignasName | ThreadLocalName => Standard::C23,
272 Inline | Restrict | Bool | Complex | Imaginary => Standard::C99,
273 _ => Standard::C89,
274 }
275 }
276
277 /// Looks a keyword up by spelling, honouring the language standard.
278 pub fn from_str(s: &str, standard: Standard) -> Option<Keyword> {
279 use Keyword::*;
280 if standard >= Standard::C23 {
281 let c23 = match s {
282 "bool" => Some(BoolName),
283 "true" => Some(True),
284 "false" => Some(False),
285 "nullptr" => Some(Nullptr),
286 "typeof" => Some(Typeof),
287 "typeof_unqual" => Some(TypeofUnqual),
288 "constexpr" => Some(Constexpr),
289 "static_assert" => Some(StaticAssertName),
290 "alignof" => Some(AlignofName),
291 "alignas" => Some(AlignasName),
292 "thread_local" => Some(ThreadLocalName),
293 _ => None,
294 };
295 if c23.is_some() {
296 return c23;
297 }
298 }
299 Some(match s {
300 "_Alignas" => Alignas,
301 "_Alignof" => Alignof,
302 "_Atomic" => Atomic,
303 "_Generic" => Generic,
304 "_Noreturn" => Noreturn,
305 "_Static_assert" => StaticAssert,
306 "_Thread_local" => ThreadLocal,
307 "_BitInt" => BitInt,
308 "auto" => Auto,
309 "break" => Break,
310 "case" => Case,
311 "char" => Char,
312 "const" => Const,
313 "continue" => Continue,
314 "default" => Default,
315 "do" => Do,
316 "double" => Double,
317 "else" => Else,
318 "enum" => Enum,
319 "extern" => Extern,
320 "float" => Float,
321 "for" => For,
322 "goto" => Goto,
323 "if" => If,
324 "inline" => Inline,
325 "int" => Int,
326 "long" => Long,
327 "register" => Register,
328 "restrict" => Restrict,
329 "return" => Return,
330 "short" => Short,
331 "signed" => Signed,
332 "sizeof" => Sizeof,
333 "static" => Static,
334 "struct" => Struct,
335 "switch" => Switch,
336 "typedef" => Typedef,
337 "union" => Union,
338 "unsigned" => Unsigned,
339 "void" => Void,
340 "volatile" => Volatile,
341 "while" => While,
342 "_Bool" => Bool,
343 "_Complex" => Complex,
344 "_Imaginary" => Imaginary,
345 _ => return None,
346 })
347 }
348}
349
350/// A C99 punctuator.
351///
352/// Digraphs are folded into the token they stand for: `<:` lexes as
353/// [`Punct::LBracket`], `%:%:` as [`Punct::HashHash`], and so on.
354#[derive(Clone, Copy, PartialEq, Eq, Debug, Hash)]
355#[allow(missing_docs)]
356pub enum Punct {
357 LBracket,
358 RBracket,
359 LParen,
360 RParen,
361 LBrace,
362 RBrace,
363 Dot,
364 Arrow,
365 PlusPlus,
366 MinusMinus,
367 Amp,
368 Star,
369 Plus,
370 Minus,
371 Tilde,
372 Bang,
373 Slash,
374 Percent,
375 Shl,
376 Shr,
377 Lt,
378 Gt,
379 Le,
380 Ge,
381 EqEq,
382 Ne,
383 Caret,
384 Pipe,
385 AmpAmp,
386 PipePipe,
387 Question,
388 Colon,
389 Semi,
390 Ellipsis,
391 Assign,
392 StarAssign,
393 SlashAssign,
394 PercentAssign,
395 PlusAssign,
396 MinusAssign,
397 ShlAssign,
398 ShrAssign,
399 AmpAssign,
400 CaretAssign,
401 PipeAssign,
402 Comma,
403 Hash,
404 HashHash,
405}
406
407impl Punct {
408 /// The canonical spelling of this punctuator (never the digraph form).
409 pub fn as_str(self) -> &'static str {
410 use Punct::*;
411 match self {
412 LBracket => "[",
413 RBracket => "]",
414 LParen => "(",
415 RParen => ")",
416 LBrace => "{",
417 RBrace => "}",
418 Dot => ".",
419 Arrow => "->",
420 PlusPlus => "++",
421 MinusMinus => "--",
422 Amp => "&",
423 Star => "*",
424 Plus => "+",
425 Minus => "-",
426 Tilde => "~",
427 Bang => "!",
428 Slash => "/",
429 Percent => "%",
430 Shl => "<<",
431 Shr => ">>",
432 Lt => "<",
433 Gt => ">",
434 Le => "<=",
435 Ge => ">=",
436 EqEq => "==",
437 Ne => "!=",
438 Caret => "^",
439 Pipe => "|",
440 AmpAmp => "&&",
441 PipePipe => "||",
442 Question => "?",
443 Colon => ":",
444 Semi => ";",
445 Ellipsis => "...",
446 Assign => "=",
447 StarAssign => "*=",
448 SlashAssign => "/=",
449 PercentAssign => "%=",
450 PlusAssign => "+=",
451 MinusAssign => "-=",
452 ShlAssign => "<<=",
453 ShrAssign => ">>=",
454 AmpAssign => "&=",
455 CaretAssign => "^=",
456 PipeAssign => "|=",
457 Comma => ",",
458 Hash => "#",
459 HashHash => "##",
460 }
461 }
462}
463
464/// Punctuator spellings, longest first so that a greedy match is also the
465/// maximal munch the standard asks for.
466const PUNCTUATORS: &[(&str, Punct)] = &[
467 ("%:%:", Punct::HashHash),
468 ("...", Punct::Ellipsis),
469 ("<<=", Punct::ShlAssign),
470 (">>=", Punct::ShrAssign),
471 ("->", Punct::Arrow),
472 ("++", Punct::PlusPlus),
473 ("--", Punct::MinusMinus),
474 ("<<", Punct::Shl),
475 (">>", Punct::Shr),
476 ("<=", Punct::Le),
477 (">=", Punct::Ge),
478 ("==", Punct::EqEq),
479 ("!=", Punct::Ne),
480 ("&&", Punct::AmpAmp),
481 ("||", Punct::PipePipe),
482 ("*=", Punct::StarAssign),
483 ("/=", Punct::SlashAssign),
484 ("%=", Punct::PercentAssign),
485 ("+=", Punct::PlusAssign),
486 ("-=", Punct::MinusAssign),
487 ("&=", Punct::AmpAssign),
488 ("^=", Punct::CaretAssign),
489 ("|=", Punct::PipeAssign),
490 ("##", Punct::HashHash),
491 ("<:", Punct::LBracket),
492 (":>", Punct::RBracket),
493 ("<%", Punct::LBrace),
494 ("%>", Punct::RBrace),
495 ("%:", Punct::Hash),
496 ("[", Punct::LBracket),
497 ("]", Punct::RBracket),
498 ("(", Punct::LParen),
499 (")", Punct::RParen),
500 ("{", Punct::LBrace),
501 ("}", Punct::RBrace),
502 (".", Punct::Dot),
503 ("&", Punct::Amp),
504 ("*", Punct::Star),
505 ("+", Punct::Plus),
506 ("-", Punct::Minus),
507 ("~", Punct::Tilde),
508 ("!", Punct::Bang),
509 ("/", Punct::Slash),
510 ("%", Punct::Percent),
511 ("<", Punct::Lt),
512 (">", Punct::Gt),
513 ("^", Punct::Caret),
514 ("|", Punct::Pipe),
515 ("?", Punct::Question),
516 (":", Punct::Colon),
517 (";", Punct::Semi),
518 ("=", Punct::Assign),
519 (",", Punct::Comma),
520 ("#", Punct::Hash),
521];
522
523/// The `l`/`ll` part of an integer suffix.
524#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
525pub enum LongKind {
526 /// No `l` suffix.
527 #[default]
528 None,
529 /// `l` or `L`.
530 Long,
531 /// `ll` or `LL`.
532 LongLong,
533}
534
535/// The base an integer constant was written in.
536#[derive(Clone, Copy, PartialEq, Eq, Debug)]
537pub enum NumBase {
538 /// `0b101` — C23.
539 Binary,
540 /// `0777`
541 Octal,
542 /// `42`
543 Decimal,
544 /// `0x2a`
545 Hex,
546}
547
548impl NumBase {
549 /// The radix.
550 pub fn radix(self) -> u32 {
551 match self {
552 NumBase::Binary => 2,
553 NumBase::Octal => 8,
554 NumBase::Decimal => 10,
555 NumBase::Hex => 16,
556 }
557 }
558
559 /// How a diagnostic names this base.
560 pub fn as_str(self) -> &'static str {
561 match self {
562 NumBase::Binary => "binary",
563 NumBase::Octal => "octal",
564 NumBase::Decimal => "decimal",
565 NumBase::Hex => "hexadecimal",
566 }
567 }
568}
569
570/// A decoded integer constant.
571#[derive(Clone, PartialEq, Eq, Debug)]
572pub struct IntLit {
573 /// The value, before any type is chosen for it.
574 pub value: u128,
575 /// How it was written.
576 pub base: NumBase,
577 /// Whether a `u`/`U` suffix was present.
578 pub unsigned: bool,
579 /// Whether an `l`/`ll` suffix was present.
580 pub long: LongKind,
581 /// The exact source spelling.
582 pub text: String,
583}
584
585/// The suffix of a floating constant.
586#[derive(Clone, Copy, PartialEq, Eq, Debug, Default)]
587pub enum FloatSuffix {
588 /// No suffix: the constant has type `double`.
589 #[default]
590 None,
591 /// `f`/`F`: `float`.
592 Float,
593 /// `l`/`L`: `long double`.
594 LongDouble,
595}
596
597/// A decoded floating constant.
598#[derive(Clone, PartialEq, Debug)]
599pub struct FloatLit {
600 /// The value, rounded to `f64`.
601 pub value: f64,
602 /// The suffix, which fixes the constant's type.
603 pub suffix: FloatSuffix,
604 /// Whether the constant carried GNU's imaginary suffix, `i` or `j`.
605 ///
606 /// `2.0i` is `(0, 2)` of the complex type [`FloatLit::suffix`] names the
607 /// real part of, which is what makes `1.0 + 2.0i` read the way every C
608 /// program writes a complex constant. It is a GNU extension that C11
609 /// blessed by giving `<complex.h>` a `CMPLX` built on it; the *standard*
610 /// spelling of the same thing is `_Complex_I`.
611 pub imaginary: bool,
612 /// Whether the constant was written in hexadecimal form.
613 pub hex: bool,
614 /// The exact source spelling.
615 pub text: String,
616}
617
618/// A decoded character constant.
619#[derive(Clone, PartialEq, Eq, Debug)]
620pub struct CharLit {
621 /// The value of the constant.
622 ///
623 /// For a plain single-character constant this is the unsigned value of the
624 /// execution character (so `'\xff'` is `255`); whether that is later
625 /// interpreted as `-1` is up to sema, which knows the signedness of
626 /// `char`. Multi-character constants are packed big-endian, as GCC does.
627 pub value: i64,
628 /// Which prefix the constant was written with, which is what fixes its
629 /// type: `'x'` is an `int`, `L'x'` a `wchar_t`, `u'x'` a `char16_t`,
630 /// `U'x'` a `char32_t` and `u8'x'` a `char8_t`.
631 pub kind: StrKind,
632 /// The exact source spelling, including quotes.
633 pub text: String,
634}
635
636/// Which prefix a character constant or string literal was written with.
637///
638/// The five of them are the same set for both, which is why one enum serves
639/// both: `'x'`/`"…"`, `u8'x'`/`u8"…"`, `u'x'`/`u"…"`, `U'x'`/`U"…"` and
640/// `L'x'`/`L"…"`.
641#[derive(Clone, Copy, PartialEq, Eq, Debug)]
642pub enum StrKind {
643 /// `"…"` — bytes of the execution character set, which is UTF-8.
644 Narrow,
645 /// `u8"…"` (C11) and `u8'x'` (C23) — UTF-8 bytes, of type `char` before
646 /// C23 and `char8_t` (an `unsigned char`) from C23 on.
647 Utf8,
648 /// `u"…"` and `u'x'` (C11) — UTF-16 code units, of type `char16_t`.
649 Utf16,
650 /// `U"…"` and `U'x'` (C11) — UTF-32 code units, of type `char32_t`.
651 Utf32,
652 /// `L"…"` — `wchar_t`.
653 Wide,
654}
655
656impl StrKind {
657 /// The prefix a literal of this kind is written with.
658 pub fn prefix(self) -> &'static str {
659 match self {
660 StrKind::Narrow => "",
661 StrKind::Utf8 => "u8",
662 StrKind::Utf16 => "u",
663 StrKind::Utf32 => "U",
664 StrKind::Wide => "L",
665 }
666 }
667
668 /// The revision that introduced the prefix, in the form the constant is
669 /// being used in.
670 ///
671 /// `u8"…"` is C11 (N1488) and `u8'x'` is C23 (N2418); the other three
672 /// prefixes are the same revision either way.
673 pub fn since(self, character: bool) -> Standard {
674 match self {
675 StrKind::Narrow | StrKind::Wide => Standard::C89,
676 StrKind::Utf8 if character => Standard::C23,
677 StrKind::Utf8 | StrKind::Utf16 | StrKind::Utf32 => Standard::C11,
678 }
679 }
680
681 /// Whether the elements are the bytes of the source's UTF-8, rather than
682 /// character values.
683 fn is_bytes(self) -> bool {
684 matches!(self, StrKind::Narrow | StrKind::Utf8)
685 }
686
687 /// The largest value one element of this kind can hold.
688 ///
689 /// `wide_bits` is how wide the target's `wchar_t` is, which is the one
690 /// answer here that is not fixed by the language: 32 bits on the Unix
691 /// platforms and 16 on Windows, where `L"…"` is UTF-16 and a character
692 /// outside the basic multilingual plane takes a surrogate pair, exactly as
693 /// `u"…"` does.
694 fn max_element(self, wide_bits: u32) -> u32 {
695 match self {
696 StrKind::Narrow | StrKind::Utf8 => 0xff,
697 StrKind::Utf16 => 0xffff,
698 StrKind::Wide if wide_bits <= 16 => 0xffff,
699 StrKind::Utf32 | StrKind::Wide => u32::MAX,
700 }
701 }
702
703 /// Whether one element holds a UTF-16 code unit, so that a character
704 /// beyond the basic multilingual plane becomes a surrogate pair.
705 fn is_utf16(self, wide_bits: u32) -> bool {
706 self == StrKind::Utf16 || (self == StrKind::Wide && wide_bits <= 16)
707 }
708}
709
710/// A decoded string literal.
711#[derive(Clone, PartialEq, Eq, Debug)]
712pub struct StrLit {
713 /// Which prefix it was written with.
714 pub kind: StrKind,
715 /// The decoded elements: bytes (`0..=255`) for a narrow or UTF-8 literal,
716 /// UTF-16 code units (surrogate pairs and all) for a `u"…"` one, and
717 /// character values for `U"…"` and `L"…"`. The terminating NUL is *not*
718 /// included.
719 pub values: Vec<u32>,
720 /// The exact source spelling, including quotes.
721 pub text: String,
722}
723
724impl StrLit {
725 /// The literal's bytes, if it is an ordinary narrow one.
726 pub fn as_bytes(&self) -> Option<Vec<u8>> {
727 (self.kind == StrKind::Narrow).then(|| self.values.iter().map(|v| *v as u8).collect())
728 }
729}
730
731/// What a [`Token`] is.
732#[derive(Clone, PartialEq, Debug)]
733pub enum TokenKind {
734 /// End of the token list. Always present, exactly once, last.
735 Eof,
736 /// An identifier that is not a keyword.
737 Ident(String),
738 /// A keyword.
739 Keyword(Keyword),
740 /// An integer constant.
741 Int(IntLit),
742 /// A floating constant.
743 Float(FloatLit),
744 /// A character constant.
745 Char(CharLit),
746 /// A string literal.
747 Str(StrLit),
748 /// A punctuator.
749 Punct(Punct),
750 /// Text that is not a C token at all, kept so that a skipped group may
751 /// contain it. Holds its own spelling.
752 Error(String),
753}
754
755impl TokenKind {
756 /// A short description used in "expected …, found …" messages.
757 pub fn describe(&self) -> String {
758 match self {
759 TokenKind::Eof => "end of input".to_owned(),
760 TokenKind::Ident(name) => format!("identifier '{name}'"),
761 TokenKind::Keyword(k) => format!("keyword '{}'", k.as_str()),
762 TokenKind::Int(_) => "integer constant".to_owned(),
763 TokenKind::Float(_) => "floating constant".to_owned(),
764 TokenKind::Char(_) => "character constant".to_owned(),
765 TokenKind::Str(_) => "string literal".to_owned(),
766 TokenKind::Punct(p) => format!("'{}'", p.as_str()),
767 TokenKind::Error(text) => format!("'{text}'"),
768 }
769 }
770
771 /// The exact source spelling of this token.
772 ///
773 /// This is what `#` stringifies and what `##` pastes, so it has to be the
774 /// text as written — `0x1f` rather than `31`, `'\n'` rather than a
775 /// newline. [`TokenKind::Eof`] has no spelling and gives `""`.
776 pub fn spelling(&self) -> &str {
777 match self {
778 TokenKind::Eof => "",
779 TokenKind::Ident(name) => name,
780 TokenKind::Keyword(k) => k.as_str(),
781 TokenKind::Int(lit) => &lit.text,
782 TokenKind::Float(lit) => &lit.text,
783 TokenKind::Char(lit) => &lit.text,
784 TokenKind::Str(lit) => &lit.text,
785 TokenKind::Punct(p) => p.as_str(),
786 TokenKind::Error(text) => text,
787 }
788 }
789
790 /// The name this token has when it is used as a macro name.
791 ///
792 /// Keywords are ordinary identifiers during translation phase 4 — our
793 /// lexer classifies them early, so `#define restrict` and `#ifdef inline`
794 /// would otherwise be unusable — which is why a keyword answers with its
795 /// spelling here.
796 pub fn macro_name(&self) -> Option<&str> {
797 match self {
798 TokenKind::Ident(name) => Some(name),
799 TokenKind::Keyword(k) => Some(k.as_str()),
800 _ => None,
801 }
802 }
803}
804
805/// A lexed C token.
806#[derive(Clone, PartialEq, Debug)]
807pub struct Token {
808 /// What the token is.
809 pub kind: TokenKind,
810 /// Where it is.
811 pub range: SourceRange,
812 /// Whether it is the first token on its logical line (needed by the
813 /// preprocessor to spot directives).
814 pub bol: bool,
815 /// Whether whitespace or a comment preceded it (needed by the
816 /// preprocessor's stringification and macro replacement).
817 pub preceded_by_space: bool,
818 /// Everything wrong with this token, and with the text between it and the
819 /// token before it.
820 ///
821 /// The [preprocessor](crate::pp) reports what is wrong with the token
822 /// *itself* if and only if the token reaches its output. What stands
823 /// whatever becomes of the token — its spelling, and the comments that
824 /// preceded it — is reported as soon as the token is read from the file,
825 /// which is what keeps it from being lost when the token opens a directive,
826 /// names a macro or is an argument the macro drops; see
827 /// [`Diagnostic::lexical`].
828 pub errors: Vec<Diagnostic>,
829}
830
831impl Token {
832 /// The keyword this token is, if any.
833 pub fn keyword(&self) -> Option<Keyword> {
834 match &self.kind {
835 TokenKind::Keyword(k) => Some(*k),
836 _ => None,
837 }
838 }
839
840 /// Whether this token is the given punctuator.
841 pub fn is_punct(&self, p: Punct) -> bool {
842 self.kind == TokenKind::Punct(p)
843 }
844
845 /// Whether this token is the given keyword.
846 pub fn is_keyword(&self, k: Keyword) -> bool {
847 self.kind == TokenKind::Keyword(k)
848 }
849
850 /// Whether this token ends the input.
851 pub fn is_eof(&self) -> bool {
852 self.kind == TokenKind::Eof
853 }
854
855 /// The identifier this token is, if any.
856 pub fn ident(&self) -> Option<&str> {
857 match &self.kind {
858 TokenKind::Ident(name) => Some(name),
859 _ => None,
860 }
861 }
862}
863
864/// Knobs for the lexer.
865#[derive(Clone, Copy, Debug)]
866pub struct LexOptions {
867 /// Which standard's lexical rules to apply.
868 pub standard: Standard,
869 /// How a constant form a newer revision introduced is gated.
870 pub gating: crate::Gating,
871 /// Accept `$` in identifiers, like GCC's `-fdollars-in-identifiers`, which
872 /// is on by default; see [`crate::Options::dollar_in_identifiers`].
873 pub dollar_in_identifiers: bool,
874 /// Whether translation phase 1 replaces the nine trigraphs.
875 ///
876 /// See [`trigraphs_enabled`] for who has them and why.
877 pub trigraphs: bool,
878 /// How wide the target's `wchar_t` is.
879 ///
880 /// The one target property the *lexer* has an opinion about: it decides
881 /// what an `L'\xffff'` escape may hold and whether `L"😀"` is one element
882 /// or a surrogate pair, since Windows makes `wchar_t` 16 bits.
883 pub wchar_bits: u32,
884 /// Whether the complex types are available, which is what decides whether
885 /// an imaginary constant (`2.0i`) has a type; see
886 /// [`crate::Options::complex`].
887 pub complex: bool,
888}
889
890impl LexOptions {
891 /// The lexical rules of `standard`, in the strict ISO dialect.
892 pub fn new(standard: Standard) -> Self {
893 Self {
894 standard,
895 gating: crate::Gating {
896 standard,
897 dialect: crate::Dialect::Iso,
898 },
899 dollar_in_identifiers: true,
900 trigraphs: trigraphs_enabled(standard, crate::Dialect::Iso),
901 wchar_bits: crate::TargetModel::host().wchar_bits,
902 complex: crate::COMPLEX_SUPPORTED,
903 }
904 }
905}
906
907impl From<&Options> for LexOptions {
908 fn from(o: &Options) -> Self {
909 Self {
910 standard: o.standard,
911 gating: o.gating(),
912 dollar_in_identifiers: o.dollar_in_identifiers,
913 trigraphs: trigraphs_enabled(o.standard, o.dialect),
914 wchar_bits: o.target.wchar_bits,
915 complex: o.complex,
916 }
917 }
918}
919
920/// Whether translation phase 1 replaces trigraphs in this entry point.
921///
922/// The nine of them were in C from the beginning and C23 removed them
923/// (N2940), so every strict entry point below `c23!` has them and `c23!` does
924/// not. No *GNU* dialect has them: `gcc -std=gnu99` switches them off, because
925/// `"what??!"` in a string is far more likely to be an exclamation than a
926/// pipe, and that is the line Clang draws too.
927pub fn trigraphs_enabled(standard: Standard, dialect: crate::Dialect) -> bool {
928 standard < Standard::C23 && !dialect.is_gnu()
929}
930
931/// The nine trigraphs of C 5.2.1.1, as `(third character, replacement)`.
932const TRIGRAPHS: &[(u8, u8)] = &[
933 (b'=', b'#'),
934 (b'(', b'['),
935 (b'/', b'\\'),
936 (b')', b']'),
937 (b'\'', b'^'),
938 (b'<', b'{'),
939 (b'!', b'|'),
940 (b'>', b'}'),
941 (b'-', b'~'),
942];
943
944/// Lexes the root file of `source`.
945///
946/// The returned vector always ends with a [`TokenKind::Eof`] token. Problems
947/// are attached to the tokens they were found in ([`Token::errors`]) rather
948/// than reported; scanning always runs to the end of the input.
949pub fn lex(source: &Source, options: &Options) -> Vec<Token> {
950 let file = source.map.file(source.root);
951 lex_file(file.text(), file.base(), &options.into())
952}
953
954/// The UTF-8 byte order mark, which an editor on Windows may write at the start
955/// of a file.
956const BYTE_ORDER_MARK: char = '\u{feff}';
957
958/// Lexes the whole of a file's `text` — a unit's own or an `#include`d one —
959/// whose first byte lives at global offset `base`.
960///
961/// A byte order mark at the very start is skipped, as GCC and Clang skip it.
962/// It is skipped rather than removed: scanning simply starts three bytes in,
963/// so every token's offset, and every column the source map computes from one,
964/// is still the file's own. A byte order mark anywhere else is an ordinary
965/// stray character and an error, which [`lex_text`] reports.
966pub fn lex_file(text: &str, base: Pos, options: &LexOptions) -> Vec<Token> {
967 let start = if text.starts_with(BYTE_ORDER_MARK) {
968 BYTE_ORDER_MARK.len_utf8()
969 } else {
970 0
971 };
972 lex_from(text, base, start, options)
973}
974
975/// Lexes `text`, whose first byte lives at global offset `base`.
976pub fn lex_text(text: &str, base: Pos, options: &LexOptions) -> Vec<Token> {
977 lex_from(text, base, 0, options)
978}
979
980/// Lexes `text` from byte `start` on; the tokens' offsets count from the
981/// beginning of `text` all the same.
982fn lex_from(text: &str, base: Pos, start: usize, options: &LexOptions) -> Vec<Token> {
983 Lexer {
984 text,
985 bytes: text.as_bytes(),
986 base,
987 pos: start,
988 options: *options,
989 pending: Vec::new(),
990 }
991 .run()
992}
993
994struct Lexer<'a> {
995 text: &'a str,
996 bytes: &'a [u8],
997 base: Pos,
998 pos: usize,
999 options: LexOptions,
1000 /// Problems found since the last token was finished; they belong to the
1001 /// token currently being scanned.
1002 pending: Vec<Diagnostic>,
1003}
1004
1005impl<'a> Lexer<'a> {
1006 fn range(&self, start: usize, end: usize) -> SourceRange {
1007 SourceRange::new(self.base + start as Pos, self.base + end as Pos)
1008 }
1009
1010 /// Records a problem with the token being scanned.
1011 fn error(&mut self, range: SourceRange, message: impl Into<String>) {
1012 self.pending.push(Diagnostic::error(range, message));
1013 }
1014
1015 /// Records a problem that stands whether or not the token being scanned
1016 /// ever reaches the parser: one with its *spelling*, or one in the text
1017 /// between it and the token before it. See [`Diagnostic::lexical`].
1018 fn lexical_error(&mut self, range: SourceRange, message: impl Into<String>) {
1019 self.pending
1020 .push(Diagnostic::error(range, message).at_lexing());
1021 }
1022
1023 /// Records an advisory remark about the token being scanned.
1024 fn warning(&mut self, range: SourceRange, message: impl Into<String>) {
1025 self.pending.push(Diagnostic::warning(range, message));
1026 }
1027
1028 fn peek(&self) -> Option<u8> {
1029 self.bytes.get(self.pos).copied()
1030 }
1031
1032 fn peek_at(&self, n: usize) -> Option<u8> {
1033 self.bytes.get(self.pos + n).copied()
1034 }
1035
1036 fn eof(&self) -> bool {
1037 self.pos >= self.bytes.len()
1038 }
1039
1040 fn run(mut self) -> Vec<Token> {
1041 let mut tokens = Vec::new();
1042 let mut bol = true;
1043 let mut space = false;
1044 loop {
1045 let (saw_newline, saw_space) = self.skip_whitespace();
1046 bol |= saw_newline;
1047 space |= saw_space || saw_newline;
1048 if self.eof() {
1049 let end = self.bytes.len();
1050 tokens.push(Token {
1051 kind: TokenKind::Eof,
1052 range: self.range(end, end),
1053 bol,
1054 preceded_by_space: space,
1055 errors: std::mem::take(&mut self.pending),
1056 });
1057 break;
1058 }
1059 let start = self.pos;
1060 let kind = self.scan_token();
1061 tokens.push(Token {
1062 kind,
1063 range: self.range(start, self.pos),
1064 bol,
1065 preceded_by_space: space,
1066 errors: std::mem::take(&mut self.pending),
1067 });
1068 bol = false;
1069 space = false;
1070 }
1071 tokens
1072 }
1073
1074 /// Skips whitespace, comments and line splices.
1075 ///
1076 /// Returns `(saw_newline, saw_space)`.
1077 fn skip_whitespace(&mut self) -> (bool, bool) {
1078 let mut newline = false;
1079 let mut space = false;
1080 loop {
1081 match self.peek() {
1082 Some(b'\n') => {
1083 self.pos += 1;
1084 newline = true;
1085 }
1086 Some(b' ' | b'\t' | b'\r' | 0x0b | 0x0c) => {
1087 self.pos += 1;
1088 space = true;
1089 }
1090 // Translation phase 2: a backslash immediately followed by a
1091 // newline splices the two lines, so it is *not* a line break.
1092 // `??/` is that backslash where trigraphs are on.
1093 Some(b'\\' | b'?') if self.is_line_splice(self.pos) => {
1094 self.pos += self.line_splice_len(self.pos);
1095 space = true;
1096 }
1097 Some(b'/') if self.peek_at(1) == Some(b'*') => {
1098 // Translation phase 3 replaces the whole comment with one
1099 // space, so a newline *inside* it is not a line break at
1100 // all: a directive may span one, and a `#` after one does
1101 // not start a directive. Both are what GCC does, and the
1102 // standard's own `FUNC_LIKE` example in 6.10.3 depends on
1103 // the first.
1104 let start = self.pos;
1105 self.pos += 2;
1106 let mut closed = false;
1107 while let Some(c) = self.peek() {
1108 if c == b'*' && self.peek_at(1) == Some(b'/') {
1109 self.pos += 2;
1110 closed = true;
1111 break;
1112 }
1113 self.pos += 1;
1114 }
1115 if !closed {
1116 // Both of the problems a *comment* can have belong to
1117 // the token that follows it for want of anywhere else
1118 // to put them, and neither is about that token: whether
1119 // a comment is terminated, and whether this revision
1120 // has `//`, is settled in translation phase 3 by the
1121 // text alone. So they are recorded as lexical, and
1122 // reported wherever the token ends up — including
1123 // nowhere, which is what a `#` opening a directive and
1124 // a macro name do with it.
1125 let range = self.range(start, self.bytes.len());
1126 self.lexical_error(range, "unterminated comment");
1127 }
1128 space = true;
1129 }
1130 Some(b'/') if self.peek_at(1) == Some(b'/') => {
1131 let start = self.pos;
1132 self.pos += 2;
1133 while let Some(c) = self.peek() {
1134 if c == b'\n' {
1135 break;
1136 }
1137 if matches!(c, b'\\' | b'?') && self.is_line_splice(self.pos) {
1138 self.pos += self.line_splice_len(self.pos);
1139 continue;
1140 }
1141 self.pos += 1;
1142 }
1143 // C99 took the `//` comment from C++ (N644); before that
1144 // `a //* b */ c` was a division, which is why the gate is
1145 // here rather than being a warning. Lexical for the reason
1146 // above: a `c89!` block that writes one has to be told so
1147 // whatever follows it.
1148 if let Some(message) = self
1149 .options
1150 .gating
1151 .requires("a '//' comment", Standard::C99)
1152 {
1153 let range = self.range(start, self.pos);
1154 self.lexical_error(range, message);
1155 }
1156 space = true;
1157 }
1158 _ => return (newline, space),
1159 }
1160 }
1161 }
1162
1163 fn is_line_splice(&self, at: usize) -> bool {
1164 self.line_splice_len(at) > 0
1165 }
1166
1167 /// Length of a `\`-newline splice starting at `at`, or 0.
1168 ///
1169 /// Translation phase 1 runs *before* phase 2, so `??/` at the end of a
1170 /// line splices it exactly as a written backslash does — which is the one
1171 /// trigraph whose replacement is not a character the lexer can simply hand
1172 /// on.
1173 fn line_splice_len(&self, at: usize) -> usize {
1174 let lead = match self.trigraph_at(at) {
1175 Some(b'\\') => 3,
1176 Some(_) => return 0,
1177 None if self.bytes.get(at) == Some(&b'\\') => 1,
1178 None => return 0,
1179 };
1180 match (self.bytes.get(at + lead), self.bytes.get(at + lead + 1)) {
1181 (Some(b'\n'), _) => lead + 1,
1182 (Some(b'\r'), Some(b'\n')) => lead + 2,
1183 _ => 0,
1184 }
1185 }
1186
1187 /// The character a trigraph at `at` stands for, if there is one there.
1188 ///
1189 /// C 5.2.1.1: the nine three-character sequences beginning `??` are
1190 /// replaced in translation phase 1, before line splicing and before the
1191 /// source is split into tokens — so this is consulted from everywhere the
1192 /// lexer looks at a raw byte, rather than the text being rewritten. Not
1193 /// rewriting it is what keeps every [`SourceRange`] a range of the source
1194 /// the user really wrote: a diagnostic about `??=` points at all three
1195 /// characters.
1196 fn trigraph_at(&self, at: usize) -> Option<u8> {
1197 if !self.options.trigraphs
1198 || self.bytes.get(at) != Some(&b'?')
1199 || self.bytes.get(at + 1) != Some(&b'?')
1200 {
1201 return None;
1202 }
1203 let third = *self.bytes.get(at + 2)?;
1204 TRIGRAPHS
1205 .iter()
1206 .find(|(c, _)| *c == third)
1207 .map(|(_, replacement)| *replacement)
1208 }
1209
1210 /// The length in source bytes of the character at `at`, which is three for
1211 /// a trigraph and one otherwise.
1212 fn trigraph_len(&self, at: usize) -> usize {
1213 if self.trigraph_at(at).is_some() { 3 } else { 1 }
1214 }
1215
1216 /// The character-constant or string-literal prefix starting here, with its
1217 /// length in bytes.
1218 ///
1219 /// A prefix is only one when a quote follows it, which is what keeps
1220 /// `unsigned`, `u8x` and a variable called `U` ordinary identifiers.
1221 fn literal_prefix(&self, first: u8) -> Option<(StrKind, usize)> {
1222 let (kind, len) = match first {
1223 b'L' => (StrKind::Wide, 1),
1224 b'U' => (StrKind::Utf32, 1),
1225 b'u' if self.peek_at(1) == Some(b'8') => (StrKind::Utf8, 2),
1226 b'u' => (StrKind::Utf16, 1),
1227 _ => return None,
1228 };
1229 matches!(self.peek_at(len), Some(b'"' | b'\'')).then_some((kind, len))
1230 }
1231
1232 /// Whether an identifier that begins with an extended character starts
1233 /// here.
1234 fn extended_ident_start(&self) -> bool {
1235 match self.peek() {
1236 Some(b'\\') if matches!(self.peek_at(1), Some(b'u' | b'U')) => {
1237 let want = if self.peek_at(1) == Some(b'u') { 4 } else { 8 };
1238 (0..want).all(|i| self.peek_at(2 + i).is_some_and(|c| c.is_ascii_hexdigit()))
1239 }
1240 Some(c) if c >= 0x80 => self.text[self.pos..]
1241 .chars()
1242 .next()
1243 .is_some_and(|ch| is_extended_ident_char(ch, true)),
1244 _ => false,
1245 }
1246 }
1247
1248 /// Scans one token, which is [`TokenKind::Error`] for text that is not a C
1249 /// token at all.
1250 fn scan_token(&mut self) -> TokenKind {
1251 let Some(c) = self.peek() else {
1252 return TokenKind::Eof;
1253 };
1254 if is_ident_start(c, self.options.dollar_in_identifiers) {
1255 // `L'x'`, `u8"…"`, `u'x'` and the rest are constants rather than an
1256 // identifier followed by one. The prefix is recognised in every
1257 // entry point and *gated* rather than not recognised at all: a
1258 // `c99!` block that writes `u"x"` is told which macro has it,
1259 // instead of being told that `u` is undeclared.
1260 if let Some((kind, len)) = self.literal_prefix(c) {
1261 let start = self.pos;
1262 self.pos += len;
1263 let character = self.peek() == Some(b'\'');
1264 if let Some(message) = self.options.gating.requires(
1265 &format!("a '{}' literal", kind.prefix()),
1266 kind.since(character),
1267 ) {
1268 let range = self.range(start, self.pos);
1269 self.error(range, message);
1270 }
1271 return if character {
1272 self.scan_char_constant(kind)
1273 } else {
1274 self.scan_string_literal(kind)
1275 };
1276 }
1277 return self.scan_ident();
1278 }
1279 // An extended identifier, written either as the character itself —
1280 // which GCC and Clang have taken since GCC 10 — or as the universal
1281 // character name C99 6.4.2.1 introduced for it.
1282 if self.extended_ident_start() {
1283 return self.scan_ident();
1284 }
1285 if c.is_ascii_digit() || (c == b'.' && self.peek_at(1).is_some_and(|d| d.is_ascii_digit()))
1286 {
1287 return self.scan_number();
1288 }
1289 if c == b'\'' {
1290 return self.scan_char_constant(StrKind::Narrow);
1291 }
1292 if c == b'"' {
1293 return self.scan_string_literal(StrKind::Narrow);
1294 }
1295 if let Some(p) = self.scan_punctuator() {
1296 return TokenKind::Punct(p);
1297 }
1298
1299 // Anything else is not a C token at all.
1300 let start = self.pos;
1301 // `??/` that does not splice a line is the stray backslash a written
1302 // one would be, and is reported as one rather than as two question
1303 // marks.
1304 let ch = match self.trigraph_at(start) {
1305 Some(c) => {
1306 self.pos += 3;
1307 c as char
1308 }
1309 None => {
1310 let ch = self.text[start..].chars().next().unwrap_or('\u{fffd}');
1311 self.pos += ch.len_utf8();
1312 ch
1313 }
1314 };
1315 let range = self.range(start, self.pos);
1316 self.error(
1317 range,
1318 format!("unexpected character '{}' in program", ch.escape_debug()),
1319 );
1320 TokenKind::Error(ch.to_string())
1321 }
1322
1323 /// Scans an identifier or a keyword.
1324 ///
1325 /// Translation phase 2 deletes a backslash-newline *before* the source is
1326 /// split into tokens, so one may sit in the middle of an identifier:
1327 /// `__LI\<newline>NE__` is `__LINE__`, and Clang's own `drs/dr464.c`
1328 /// writes exactly that. Almost no identifier has one, so the spelling
1329 /// stays a slice of the source until the first splice is found and only
1330 /// then becomes a `String`.
1331 fn scan_ident(&mut self) -> TokenKind {
1332 let start = self.pos;
1333 // The text before the current splice or universal character name, when
1334 // there has been one.
1335 let mut spliced: Option<String> = None;
1336 // Where the run of characters that is still a slice begins.
1337 let mut segment = start;
1338 // Whether anything outside the basic character set was written, which
1339 // is the only case that has to be checked for normalization.
1340 let mut extended = false;
1341 loop {
1342 match self.peek() {
1343 Some(c) if c < 0x80 && is_ident_continue(c, self.options.dollar_in_identifiers) => {
1344 self.pos += 1;
1345 }
1346 // Only a splice that the identifier *continues* over: one at
1347 // the end of it is whitespace, and belongs to whatever comes
1348 // next.
1349 Some(b'\\' | b'?')
1350 if self.line_splice_len(self.pos) > 0
1351 && self
1352 .bytes
1353 .get(self.pos + self.line_splice_len(self.pos))
1354 .is_some_and(|c| {
1355 is_ident_continue(*c, self.options.dollar_in_identifiers)
1356 }) =>
1357 {
1358 let text = spliced.get_or_insert_with(String::new);
1359 text.push_str(&self.text[segment..self.pos]);
1360 self.pos += self.line_splice_len(self.pos);
1361 segment = self.pos;
1362 }
1363 // A universal character name spells one extended character:
1364 // `café` and `café` are the same identifier, which is
1365 // exactly what C99 6.4.2.1 says.
1366 Some(b'\\') if matches!(self.peek_at(1), Some(b'u' | b'U')) => {
1367 let at = self.pos;
1368 let Some(ch) = self.scan_ident_ucn(at == start) else {
1369 break;
1370 };
1371 let text = spliced.get_or_insert_with(String::new);
1372 text.push_str(&self.text[segment..at]);
1373 text.push(ch);
1374 segment = self.pos;
1375 extended = true;
1376 }
1377 Some(c) if c >= 0x80 => {
1378 let ch = self.text[self.pos..].chars().next().unwrap_or('\u{fffd}');
1379 if !is_extended_ident_char(ch, self.pos == start) {
1380 break;
1381 }
1382 self.pos += ch.len_utf8();
1383 extended = true;
1384 }
1385 _ => break,
1386 }
1387 }
1388 let text = match spliced {
1389 Some(mut text) => {
1390 text.push_str(&self.text[segment..self.pos]);
1391 text
1392 }
1393 None => self.text[start..self.pos].to_owned(),
1394 };
1395 // Nothing was an identifier character after all — the whole of it was
1396 // one bad universal character name, which has been reported. Something
1397 // has to be consumed, or the scanner would sit here forever.
1398 if text.is_empty() {
1399 if self.pos == start {
1400 let ch = self.text[start..].chars().next().unwrap_or('\u{fffd}');
1401 self.pos += ch.len_utf8();
1402 }
1403 return TokenKind::Error(self.text[start..self.pos].to_owned());
1404 }
1405 // Rust identifiers have to be in Normalization Form C, and `rustc`
1406 // *normalises* the ones a procedural macro hands it rather than
1407 // refusing them — so two C identifiers that differ only by
1408 // normalization would silently become one Rust item. C23 (N2836) asks
1409 // for NFC as well, so refusing is both the safe answer and the
1410 // conforming one.
1411 if extended && !unicode_normalization::is_nfc(&text) {
1412 let range = self.range(start, self.pos);
1413 self.error(
1414 range,
1415 format!(
1416 "identifier '{text}' is not in Unicode Normalization Form C; \
1417 write the composed form"
1418 ),
1419 );
1420 }
1421 match Keyword::from_str(&text, self.options.standard) {
1422 Some(k) => TokenKind::Keyword(k),
1423 None => TokenKind::Ident(text),
1424 }
1425 }
1426
1427 /// Reads a `\uXXXX` or `\UXXXXXXXX` written inside an identifier.
1428 ///
1429 /// A name that is too short to be one leaves the position where it was, so
1430 /// that the identifier simply ends there and the backslash is reported by
1431 /// [`Lexer::scan_token`] as the stray character it is. One that is the
1432 /// right shape but names something an identifier may not hold is *always*
1433 /// consumed, and reported: leaving it would be a second diagnostic about
1434 /// the same text, and — where it is the first character of the identifier
1435 /// — a token that consumed nothing at all.
1436 fn scan_ident_ucn(&mut self, start: bool) -> Option<char> {
1437 let at = self.pos;
1438 let want = if self.peek_at(1) == Some(b'u') { 4 } else { 8 };
1439 let mut value: u32 = 0;
1440 for i in 0..want {
1441 let digit = self.peek_at(2 + i).and_then(|c| (c as char).to_digit(16))?;
1442 value = value * 16 + digit;
1443 }
1444 let spelling = if want == 4 { 'u' } else { 'U' };
1445 let ch = char::from_u32(value);
1446 if !ch.is_some_and(|ch| is_extended_ident_char(ch, start)) {
1447 self.pos += 2 + want;
1448 let range = self.range(at, self.pos);
1449 let digits = if want == 4 {
1450 format!("{value:04X}")
1451 } else {
1452 format!("{value:08X}")
1453 };
1454 let message =
1455 format!("'\\{spelling}{digits}' is not a valid character in an identifier");
1456 // Two different refusals wear the same words. Naming a character
1457 // below U+00A0, or a surrogate, is what C99 6.4.3p2 forbids of the
1458 // *name* — C23 puts `$`, `@` and `` ` `` in the basic character
1459 // set, and Clang refuses those in every mode — so it is ill formed
1460 // where it is written and is reported even where the token goes
1461 // nowhere: Clang's own `C99/n717.c` is a file of them, each
1462 // written as the argument of a macro that expands to nothing.
1463 // Everything else here — a value outside Unicode, a character
1464 // that simply is not `XID_Continue` — is about the *identifier*,
1465 // and a preprocessing token that never reaches the parser is
1466 // allowed to be one C has no other use for (6.4p3).
1467 if value < 0xA0 || (0xD800..=0xDFFF).contains(&value) {
1468 self.lexical_error(range, message);
1469 } else {
1470 self.error(range, message);
1471 }
1472 return None;
1473 }
1474 if let Some(message) = self
1475 .options
1476 .gating
1477 .requires("a universal character name", Standard::C99)
1478 {
1479 let range = self.range(at, at + 2 + want);
1480 self.error(range, message);
1481 }
1482 self.pos += 2 + want;
1483 ch
1484 }
1485
1486 fn scan_punctuator(&mut self) -> Option<Punct> {
1487 if self.trigraph_at(self.pos).is_some() {
1488 return self.scan_trigraph_punctuator();
1489 }
1490 let rest = &self.text[self.pos..];
1491 for (spelling, punct) in PUNCTUATORS {
1492 if rest.starts_with(spelling) {
1493 // `<:` is a digraph for `[`, but `1<::x` must not be mangled;
1494 // C99 has no `::`, so a plain greedy match is correct here.
1495 self.pos += spelling.len();
1496 return Some(*punct);
1497 }
1498 }
1499 None
1500 }
1501
1502 /// A punctuator that begins with a trigraph.
1503 ///
1504 /// Phase 1 happens before tokens exist, so a punctuator may be spelled
1505 /// partly or wholly with trigraphs — `??!??!` is `||`, `??'=` is `^=`,
1506 /// `??=??=` is `##` — and maximal munch applies to the *replaced*
1507 /// characters. Up to the length of the longest punctuator is replaced into
1508 /// a small buffer, matched there, and the source position advanced by what
1509 /// the matched characters really cost.
1510 fn scan_trigraph_punctuator(&mut self) -> Option<Punct> {
1511 const LONGEST: usize = 4;
1512 let mut logical = [0u8; LONGEST];
1513 let mut widths = [0usize; LONGEST];
1514 let mut count = 0;
1515 let mut at = self.pos;
1516 while count < LONGEST {
1517 let (c, width) = match self.trigraph_at(at) {
1518 Some(c) => (c, 3),
1519 None => match self.bytes.get(at) {
1520 Some(c) => (*c, 1),
1521 None => break,
1522 },
1523 };
1524 // A backslash is not part of any punctuator, and neither is
1525 // anything outside the basic character set.
1526 if c == b'\\' || !c.is_ascii() {
1527 break;
1528 }
1529 logical[count] = c;
1530 widths[count] = width;
1531 count += 1;
1532 at += width;
1533 }
1534 let text = std::str::from_utf8(&logical[..count]).ok()?;
1535 for (spelling, punct) in PUNCTUATORS {
1536 if text.starts_with(spelling) {
1537 self.pos += widths[..spelling.len()].iter().sum::<usize>();
1538 return Some(*punct);
1539 }
1540 }
1541 None
1542 }
1543
1544 // -- numbers ------------------------------------------------------------
1545
1546 /// Scans a preprocessing number and classifies it as integer or float.
1547 ///
1548 /// Scanning the whole pp-number first (rather than stopping at the first
1549 /// character that does not fit) is what lets `08` or `1.0q` be reported as
1550 /// one bad token instead of two good ones.
1551 fn scan_number(&mut self) -> TokenKind {
1552 let start = self.pos;
1553 if self.peek() == Some(b'.') {
1554 self.pos += 1;
1555 }
1556 self.pos += 1;
1557 let mut separators = false;
1558 while let Some(c) = self.peek() {
1559 if matches!(c, b'e' | b'E' | b'p' | b'P')
1560 && matches!(self.peek_at(1), Some(b'+') | Some(b'-'))
1561 {
1562 self.pos += 2;
1563 continue;
1564 }
1565 // C23's digit separator. It is part of the pp-number when a digit
1566 // or a nondigit follows it, which is what keeps `1'a'` from
1567 // swallowing the character constant that follows a constant.
1568 if c == b'\''
1569 && self
1570 .peek_at(1)
1571 .is_some_and(|d| d.is_ascii_alphanumeric() || d == b'_')
1572 {
1573 separators = true;
1574 self.pos += 1;
1575 continue;
1576 }
1577 if c.is_ascii_alphanumeric()
1578 || c == b'_'
1579 || c == b'.'
1580 || (c == b'$' && self.options.dollar_in_identifiers)
1581 {
1582 self.pos += 1;
1583 continue;
1584 }
1585 break;
1586 }
1587 let text = &self.text[start..self.pos];
1588 let range = self.range(start, self.pos);
1589 if separators
1590 && let Some(message) = self
1591 .options
1592 .gating
1593 .requires("a digit separator", Standard::C23)
1594 {
1595 self.error(range, message);
1596 }
1597 // Everything below reads the digits; the separators are not part of
1598 // the value, while `text` keeps the spelling `#` has to reproduce.
1599 let stripped: String;
1600 let digits = if separators {
1601 stripped = text.replace('\'', "");
1602 stripped.as_str()
1603 } else {
1604 text
1605 };
1606 let lower = digits.to_ascii_lowercase();
1607 let hex = lower.starts_with("0x");
1608 let is_float = if hex {
1609 digits.contains('.') || lower[2..].contains('p')
1610 } else {
1611 digits.contains('.') || (!lower.starts_with("0b") && lower.contains('e'))
1612 };
1613 if is_float {
1614 TokenKind::Float(self.decode_float(digits, text, range, hex))
1615 } else {
1616 TokenKind::Int(self.decode_int(digits, text, range))
1617 }
1618 }
1619
1620 /// Decodes an integer constant from its separator-free `digits`; `text` is
1621 /// the spelling as written, which is what diagnostics quote and what `#`
1622 /// reproduces.
1623 fn decode_int(&mut self, digits: &str, text: &str, range: SourceRange) -> IntLit {
1624 let bytes = digits.as_bytes();
1625 let (base, digits_start) =
1626 if digits.len() >= 2 && (bytes[1] | 0x20) == b'x' && bytes[0] == b'0' {
1627 (NumBase::Hex, 2)
1628 } else if digits.len() >= 2 && (bytes[1] | 0x20) == b'b' && bytes[0] == b'0' {
1629 (NumBase::Binary, 2)
1630 } else if bytes[0] == b'0' && digits.len() > 1 {
1631 (NumBase::Octal, 1)
1632 } else {
1633 (NumBase::Decimal, 0)
1634 };
1635 if base == NumBase::Binary
1636 && let Some(message) = self
1637 .options
1638 .gating
1639 .requires("a binary integer constant", Standard::C23)
1640 {
1641 self.error(range, message);
1642 }
1643
1644 let radix = base.radix();
1645 // Accept every decimal digit in an octal or binary constant so that
1646 // the whole constant is consumed and reported once, the way C
1647 // compilers do.
1648 let scan_radix = if radix < 10 { 10 } else { radix };
1649
1650 let mut i = digits_start;
1651 let mut value: u128 = 0;
1652 let mut overflow = false;
1653 let mut bad_digit: Option<char> = None;
1654 while i < bytes.len() {
1655 let c = bytes[i] as char;
1656 let digit = match c.to_digit(scan_radix) {
1657 Some(d) => d,
1658 None => break,
1659 };
1660 if digit >= radix && bad_digit.is_none() {
1661 bad_digit = Some(c);
1662 }
1663 match value
1664 .checked_mul(radix as u128)
1665 .and_then(|v| v.checked_add(digit as u128))
1666 {
1667 Some(v) => value = v,
1668 None => overflow = true,
1669 }
1670 i += 1;
1671 }
1672
1673 if i == digits_start && matches!(base, NumBase::Hex | NumBase::Binary) {
1674 let prefix = &digits[..2];
1675 self.error(
1676 range,
1677 format!(
1678 "expected digits after '{prefix}' in {} constant",
1679 base.as_str()
1680 ),
1681 );
1682 }
1683 if let Some(c) = bad_digit {
1684 self.error(
1685 range,
1686 format!("invalid digit '{c}' in {} constant '{text}'", base.as_str()),
1687 );
1688 }
1689 if overflow {
1690 self.error(range, format!("integer constant '{text}' is too large"));
1691 }
1692
1693 let suffix = &digits[i..];
1694 let (unsigned, long) = match parse_int_suffix(suffix) {
1695 Some(v) => v,
1696 None => {
1697 // `3i` is GCC's `_Complex int`, one of the two complex
1698 // *integer* types that are a GNU extension of their own and
1699 // that cinrs does not have. Saying what to write instead is
1700 // more use than "invalid suffix".
1701 if matches!(suffix, "i" | "j" | "I" | "J") {
1702 let digits = text.strip_suffix(suffix).unwrap_or(text);
1703 self.error(
1704 range,
1705 format!(
1706 "'{text}' is a complex integer constant, which is a GNU extension \
1707 cinrs does not support; write '{digits}.0{suffix}' for the \
1708 complex floating constant"
1709 ),
1710 );
1711 } else {
1712 self.error(
1713 range,
1714 format!("invalid suffix '{suffix}' on integer constant '{text}'"),
1715 );
1716 }
1717 (false, LongKind::None)
1718 }
1719 };
1720
1721 IntLit {
1722 value,
1723 base,
1724 unsigned,
1725 long,
1726 text: text.to_owned(),
1727 }
1728 }
1729
1730 fn decode_float(
1731 &mut self,
1732 digits: &str,
1733 text: &str,
1734 range: SourceRange,
1735 hex: bool,
1736 ) -> FloatLit {
1737 let (body, suffix) = split_float_suffix(digits, hex);
1738 let (suffix_kind, imaginary) = self.float_suffix(suffix, text, range);
1739
1740 if hex
1741 && let Some(message) = self
1742 .options
1743 .gating
1744 .requires("a hexadecimal floating constant", Standard::C99)
1745 {
1746 self.error(range, message);
1747 }
1748 let value = if hex {
1749 match parse_hex_float(body) {
1750 Some(v) => v,
1751 None => {
1752 self.error(
1753 range,
1754 format!(
1755 "invalid hexadecimal floating constant '{text}'; \
1756 a 'p' exponent is required"
1757 ),
1758 );
1759 0.0
1760 }
1761 }
1762 } else {
1763 match parse_decimal_float(body) {
1764 Some(v) => v,
1765 None => {
1766 self.error(range, format!("invalid floating constant '{text}'"));
1767 0.0
1768 }
1769 }
1770 };
1771
1772 FloatLit {
1773 value,
1774 suffix: suffix_kind,
1775 imaginary,
1776 hex,
1777 text: text.to_owned(),
1778 }
1779 }
1780
1781 /// The type a floating constant's suffix gives it.
1782 ///
1783 /// C has three: none, `f` and `l`. GCC has a dozen more, and they fall
1784 /// into three groups here.
1785 ///
1786 /// * **The ones that name a format wider than `double`** — `d`, `w`
1787 /// (`__float80`), `q` (`__float128`), and the `_FloatN` and `_FloatNx`
1788 /// suffixes `f64`, `f64x`, `f32x` and `f128`. Every one of them is
1789 /// `double` in this implementation, exactly as `long double` is, so each
1790 /// is accepted in a GNU dialect and **loses precision** where GCC would
1791 /// not; `doc/gnu-extensions.md` records that. `f32` is `float`.
1792 /// * **The decimal floating suffixes** `df`, `dd` and `dl`, whose types
1793 /// are radix-10 and have no Rust counterpart at all.
1794 /// * **The imaginary suffixes** `i` and `j`, which make the constant an
1795 /// imaginary one — `2.0i` is `(0, 2)` — and so need the complex types.
1796 ///
1797 /// The decimal ones are refused with the reason, and so are the imaginary
1798 /// ones when the complex types are switched off. A strict entry point
1799 /// refuses the first group too, naming the GNU entry point that has it —
1800 /// the suffixes are spelled without underscores, which is the line
1801 /// [`crate::Dialect`] draws.
1802 ///
1803 /// The second half of the answer is whether the constant is imaginary; see
1804 /// [`FloatLit::imaginary`].
1805 fn float_suffix(
1806 &mut self,
1807 suffix: &str,
1808 text: &str,
1809 range: SourceRange,
1810 ) -> (FloatSuffix, bool) {
1811 let lower = suffix.to_ascii_lowercase();
1812 match lower.as_str() {
1813 "" => return (FloatSuffix::None, false),
1814 "f" => return (FloatSuffix::Float, false),
1815 "l" => return (FloatSuffix::LongDouble, false),
1816 _ => {}
1817 }
1818 // GNU's imaginary suffix, on its own or beside `f` or `l`. Both
1819 // spellings are the same thing: `j` is what Fortran and engineering
1820 // habit write, and GCC takes either.
1821 let imaginary = match lower.as_str() {
1822 "i" | "j" => Some(FloatSuffix::None),
1823 "if" | "fi" | "jf" | "fj" => Some(FloatSuffix::Float),
1824 "il" | "li" | "jl" | "lj" => Some(FloatSuffix::LongDouble),
1825 _ => None,
1826 };
1827 if let Some(kind) = imaginary {
1828 if !self.options.complex {
1829 self.error(
1830 range,
1831 format!(
1832 "invalid suffix '{suffix}' on floating constant '{text}': an \
1833 imaginary constant needs _Complex. {}",
1834 crate::COMPLEX_UNSUPPORTED
1835 ),
1836 );
1837 return (FloatSuffix::None, false);
1838 }
1839 if let Some(message) = self
1840 .options
1841 .gating
1842 .requires("an imaginary constant", Standard::C99)
1843 {
1844 self.error(range, message);
1845 return (FloatSuffix::None, false);
1846 }
1847 return (kind, true);
1848 }
1849 // A decimal constant is refused whatever the entry point: it has no
1850 // type this crate can give it.
1851 let refusal = match lower.as_str() {
1852 "df" | "dd" | "dl" => Some(
1853 "the decimal floating types (_Decimal32, _Decimal64, _Decimal128) are not \
1854 supported: they are radix-10 and no Rust type is",
1855 ),
1856 "f16" | "f16x" | "bf16" => Some(
1857 "'_Float16' is not supported: Rust's `f16` is unstable, and rounding the \
1858 constant to a wider type would change what the program computes",
1859 ),
1860 _ => None,
1861 };
1862 if let Some(reason) = refusal {
1863 self.error(
1864 range,
1865 format!("invalid suffix '{suffix}' on floating constant '{text}': {reason}"),
1866 );
1867 return (FloatSuffix::None, false);
1868 }
1869 // The rest are the GNU widths. `f32` is `float`; every other one names
1870 // a format this implementation makes a `double`.
1871 let wider = matches!(
1872 lower.as_str(),
1873 "d" | "w" | "q" | "f64" | "f64x" | "f32x" | "f128" | "f128x"
1874 );
1875 if !wider && lower != "f32" {
1876 self.error(
1877 range,
1878 format!("invalid suffix '{suffix}' on floating constant '{text}'"),
1879 );
1880 return (FloatSuffix::None, false);
1881 }
1882 if !self.options.gating.dialect.is_gnu() {
1883 let gnu = self.options.gating.standard.macro_name_in(Dialect::Gnu);
1884 let here = self
1885 .options
1886 .gating
1887 .standard
1888 .macro_name_in(self.options.gating.dialect);
1889 self.error(
1890 range,
1891 format!(
1892 "the suffix '{suffix}' on a floating constant is a GNU extension, and \
1893 requires a GNU dialect ({gnu}) (this block is {here})"
1894 ),
1895 );
1896 return (FloatSuffix::None, false);
1897 }
1898 if lower == "f32" {
1899 return (FloatSuffix::Float, false);
1900 }
1901 // `LongDouble` is `double`, which is what all of these come to.
1902 (FloatSuffix::LongDouble, false)
1903 }
1904
1905 // -- character and string constants -------------------------------------
1906
1907 fn scan_char_constant(&mut self, kind: StrKind) -> TokenKind {
1908 let start = self.pos;
1909 debug_assert_eq!(self.peek(), Some(b'\''));
1910 self.pos += 1;
1911 let mut values: Vec<u32> = Vec::new();
1912 let mut terminated = false;
1913 // How many *characters* were written, which is not how many elements
1914 // they came to: one `\U0001F600` is one character and two UTF-16 code
1915 // units, and the two say different things about what is wrong.
1916 let mut characters = 0usize;
1917 while let Some(c) = self.peek() {
1918 if c == b'\'' {
1919 self.pos += 1;
1920 terminated = true;
1921 break;
1922 }
1923 if c == b'\n' {
1924 break;
1925 }
1926 let before = values.len();
1927 self.read_char_element(kind, &mut values);
1928 if values.len() > before {
1929 characters += 1;
1930 }
1931 }
1932 let range = self.range(start, self.pos);
1933 if !terminated {
1934 self.error(range, "missing terminating \' character");
1935 }
1936 if values.is_empty() {
1937 self.error(range, "empty character constant");
1938 }
1939 // Only `'ab'` and `L'ab'` have an implementation-defined meaning; C11
1940 // 6.4.4.4p2 makes more than one character in a `u8`, `u` or `U`
1941 // constant a constraint violation, because there is no room for a
1942 // second one in the type — and so is one character that needs more
1943 // than one code unit, which is what `u8'é'` and `u'😀'` are.
1944 if values.len() > 1 {
1945 match kind {
1946 StrKind::Narrow | StrKind::Wide => {
1947 self.warning(range, "multi-character character constant");
1948 }
1949 _ if characters > 1 => {
1950 self.error(
1951 range,
1952 format!(
1953 "a '{}' character constant holds exactly one character",
1954 kind.prefix()
1955 ),
1956 );
1957 }
1958 _ => {
1959 self.error(
1960 range,
1961 format!(
1962 "the character in a '{}' character constant must fit in a \
1963 single code unit",
1964 kind.prefix()
1965 ),
1966 );
1967 }
1968 }
1969 }
1970
1971 let value = if kind != StrKind::Narrow {
1972 values.last().copied().unwrap_or(0) as i64
1973 } else if values.len() <= 1 {
1974 values.first().copied().unwrap_or(0) as i64
1975 } else {
1976 // GCC packs the bytes big-endian into an `int`.
1977 let mut v: u32 = 0;
1978 for b in &values {
1979 v = (v << 8) | (*b & 0xff);
1980 }
1981 v as i32 as i64
1982 };
1983
1984 TokenKind::Char(CharLit {
1985 value,
1986 kind,
1987 text: self.text[start..self.pos].to_owned(),
1988 })
1989 }
1990
1991 fn scan_string_literal(&mut self, kind: StrKind) -> TokenKind {
1992 let start = self.pos;
1993 debug_assert_eq!(self.peek(), Some(b'"'));
1994 self.pos += 1;
1995 let mut values: Vec<u32> = Vec::new();
1996 let mut terminated = false;
1997 while let Some(c) = self.peek() {
1998 if c == b'"' {
1999 self.pos += 1;
2000 terminated = true;
2001 break;
2002 }
2003 if c == b'\n' {
2004 break;
2005 }
2006 self.read_char_element(kind, &mut values);
2007 }
2008 let range = self.range(start, self.pos);
2009 if !terminated {
2010 self.error(range, "missing terminating \" character");
2011 }
2012 TokenKind::Str(StrLit {
2013 kind,
2014 values,
2015 text: self.text[start..self.pos].to_owned(),
2016 })
2017 }
2018
2019 /// Reads one element of a character constant or string literal, appending
2020 /// its decoded value(s) to `out`.
2021 fn read_char_element(&mut self, kind: StrKind, out: &mut Vec<u32>) {
2022 if self.is_line_splice(self.pos) {
2023 self.pos += self.line_splice_len(self.pos);
2024 return;
2025 }
2026 let start = self.pos;
2027 // Phase 1 replaces trigraphs inside literals too: `"??!"` is `"|"`,
2028 // and `"??/n"` is `"\n"`.
2029 let trigraph = self.trigraph_at(self.pos);
2030 if trigraph != Some(b'\\') && self.peek() != Some(b'\\') {
2031 match trigraph {
2032 Some(c) => {
2033 self.pos += 3;
2034 out.push(u32::from(c));
2035 }
2036 None if kind.is_bytes() => {
2037 // A narrow or UTF-8 literal keeps the raw
2038 // execution-charset bytes, so UTF-8 text in one survives
2039 // byte for byte.
2040 self.pos += 1;
2041 out.push(self.bytes[start] as u32);
2042 }
2043 None => {
2044 let ch = self.text[start..].chars().next().unwrap_or('\u{fffd}');
2045 self.pos += ch.len_utf8();
2046 push_character(ch as u32, kind, self.options.wchar_bits, out);
2047 }
2048 }
2049 return;
2050 }
2051
2052 self.pos += self.trigraph_len(self.pos);
2053 let e = match self.trigraph_at(self.pos) {
2054 Some(c) => c,
2055 None => match self.peek() {
2056 Some(c) => c,
2057 None => {
2058 let range = self.range(start, self.pos);
2059 self.error(range, "incomplete escape sequence");
2060 return;
2061 }
2062 },
2063 };
2064 self.pos += self.trigraph_len(self.pos);
2065 let simple = match e {
2066 b'\'' => Some(0x27),
2067 b'"' => Some(0x22),
2068 b'?' => Some(0x3f),
2069 b'\\' => Some(0x5c),
2070 b'a' => Some(0x07),
2071 b'b' => Some(0x08),
2072 // `\e` is GNU's escape for ESC. GCC accepts it in every mode (with
2073 // a pedantic warning), and `"\e[0m"` is how a program writes a
2074 // terminal colour; refusing it would be refusing the extension in
2075 // the one place a strict mode still has it.
2076 b'e' => Some(0x1b),
2077 b'f' => Some(0x0c),
2078 b'n' => Some(0x0a),
2079 b'r' => Some(0x0d),
2080 b't' => Some(0x09),
2081 b'v' => Some(0x0b),
2082 _ => None,
2083 };
2084 if let Some(v) = simple {
2085 out.push(v);
2086 return;
2087 }
2088 match e {
2089 b'0'..=b'7' => {
2090 let mut v: u32 = (e - b'0') as u32;
2091 for _ in 0..2 {
2092 match self.peek() {
2093 Some(d @ b'0'..=b'7') => {
2094 v = v * 8 + (d - b'0') as u32;
2095 self.pos += 1;
2096 }
2097 _ => break,
2098 }
2099 }
2100 self.push_escape_value(v, kind, start, out);
2101 }
2102 b'x' => {
2103 let mut v: u32 = 0;
2104 let mut any = false;
2105 let mut overflow = false;
2106 while let Some(d) = self.peek().and_then(|c| (c as char).to_digit(16)) {
2107 any = true;
2108 v = match v.checked_mul(16).and_then(|v| v.checked_add(d)) {
2109 Some(v) => v,
2110 None => {
2111 overflow = true;
2112 v
2113 }
2114 };
2115 self.pos += 1;
2116 }
2117 let range = self.range(start, self.pos);
2118 if !any {
2119 self.error(range, "'\\x' used with no following hex digits");
2120 } else if overflow {
2121 self.error(range, "hex escape sequence out of range");
2122 }
2123 self.push_escape_value(v, kind, start, out);
2124 }
2125 b'u' | b'U' => {
2126 if let Some(message) = self
2127 .options
2128 .gating
2129 .requires("a universal character name", Standard::C99)
2130 {
2131 let range = self.range(start, self.pos);
2132 self.error(range, message);
2133 }
2134 let want = if e == b'u' { 4 } else { 8 };
2135 let mut v: u32 = 0;
2136 let mut count = 0;
2137 while count < want {
2138 match self.peek().and_then(|c| (c as char).to_digit(16)) {
2139 Some(d) => {
2140 v = v.wrapping_mul(16).wrapping_add(d);
2141 self.pos += 1;
2142 count += 1;
2143 }
2144 None => break,
2145 }
2146 }
2147 let range = self.range(start, self.pos);
2148 if count != want {
2149 self.error(
2150 range,
2151 format!("incomplete universal character name; expected {want} hex digits"),
2152 );
2153 return;
2154 }
2155 match char::from_u32(v) {
2156 Some(ch) if kind.is_bytes() => {
2157 // The execution character set is UTF-8.
2158 let mut buf = [0u8; 4];
2159 for b in ch.encode_utf8(&mut buf).as_bytes() {
2160 out.push(*b as u32);
2161 }
2162 }
2163 Some(ch) => push_character(ch as u32, kind, self.options.wchar_bits, out),
2164 None => {
2165 // Outside Unicode, or a surrogate: 6.4.3p2 again, and
2166 // the *literal token's* spelling is what is wrong, so
2167 // this stands wherever the token ends up.
2168 self.lexical_error(
2169 range,
2170 format!("'\\u{v:04X}' is not a valid universal character name"),
2171 );
2172 }
2173 }
2174 }
2175 _ => {
2176 let range = self.range(start, self.pos);
2177 self.error(
2178 range,
2179 format!("unknown escape sequence '\\{}'", (e as char).escape_debug()),
2180 );
2181 out.push(e as u32);
2182 }
2183 }
2184 }
2185
2186 /// Pushes the value of a numeric escape sequence, which unlike a character
2187 /// is *not* re-encoded: `u"\xd83d"` is that one code unit.
2188 fn push_escape_value(&mut self, v: u32, kind: StrKind, start: usize, out: &mut Vec<u32>) {
2189 let max = kind.max_element(self.options.wchar_bits);
2190 if v > max {
2191 let range = self.range(start, self.pos);
2192 let ty = match kind {
2193 StrKind::Narrow => "char",
2194 StrKind::Utf8 => "char8_t",
2195 StrKind::Utf16 => "char16_t",
2196 StrKind::Utf32 => "char32_t",
2197 StrKind::Wide => "wchar_t",
2198 };
2199 self.error(
2200 range,
2201 format!("escape sequence out of range for type '{ty}'"),
2202 );
2203 out.push(v & max);
2204 } else {
2205 out.push(v);
2206 }
2207 }
2208}
2209
2210/// Appends one character, encoded the way `kind` stores its elements.
2211///
2212/// A `u"…"` literal holds UTF-16 code units, so a character outside the basic
2213/// multilingual plane becomes the two halves of a surrogate pair — which is
2214/// what makes `sizeof(u"\U0001F600")` six rather than four. On a target whose
2215/// `wchar_t` is 16 bits wide, `L"…"` is UTF-16 too and does the same.
2216fn push_character(value: u32, kind: StrKind, wchar_bits: u32, out: &mut Vec<u32>) {
2217 if !kind.is_utf16(wchar_bits) || value <= 0xffff {
2218 out.push(value);
2219 return;
2220 }
2221 let v = value - 0x1_0000;
2222 out.push(0xd800 + (v >> 10));
2223 out.push(0xdc00 + (v & 0x3ff));
2224}
2225
2226fn is_ident_start(c: u8, dollar: bool) -> bool {
2227 c.is_ascii_alphabetic() || c == b'_' || (dollar && c == b'$')
2228}
2229
2230fn is_ident_continue(c: u8, dollar: bool) -> bool {
2231 c.is_ascii_alphanumeric() || c == b'_' || (dollar && c == b'$')
2232}
2233
2234/// Whether an extended character may appear in an identifier.
2235///
2236/// C99 Annex D listed the ranges by hand, C11 revised the list, and C23 (N2836,
2237/// N2939) replaced all of it with Unicode Annex #31's `XID_Start` and
2238/// `XID_Continue` — which is also what Rust's own identifiers are, and what
2239/// makes a C name usable as a Rust one. The one list is used in every entry
2240/// point: the earlier annexes are approximations of the same intent, and a
2241/// program that uses a character C11 left out is one this would otherwise
2242/// refuse for no reason a user could act on.
2243///
2244/// A character of the basic character set is never one of these: the ASCII
2245/// path has already decided about it, and a universal character name is not
2246/// allowed to spell one (6.4.3p2).
2247fn is_extended_ident_char(ch: char, start: bool) -> bool {
2248 if ch.is_ascii() {
2249 return false;
2250 }
2251 if start {
2252 unicode_ident::is_xid_start(ch)
2253 } else {
2254 unicode_ident::is_xid_continue(ch)
2255 }
2256}
2257
2258/// Validates an integer suffix, returning `(unsigned, long_kind)`.
2259fn parse_int_suffix(s: &str) -> Option<(bool, LongKind)> {
2260 if s.is_empty() {
2261 return Some((false, LongKind::None));
2262 }
2263 let b = s.as_bytes();
2264 let mut i = 0;
2265 let mut unsigned = false;
2266 let mut long = LongKind::None;
2267
2268 if b[i] == b'u' || b[i] == b'U' {
2269 unsigned = true;
2270 i += 1;
2271 }
2272 if i < b.len() && (b[i] == b'l' || b[i] == b'L') {
2273 // `ll` and `LL` must not be mixed as `lL` or `Ll`.
2274 if i + 1 < b.len() && b[i + 1] == b[i] {
2275 long = LongKind::LongLong;
2276 i += 2;
2277 } else {
2278 long = LongKind::Long;
2279 i += 1;
2280 }
2281 }
2282 if !unsigned && i < b.len() && (b[i] == b'u' || b[i] == b'U') {
2283 unsigned = true;
2284 i += 1;
2285 }
2286 (i == b.len()).then_some((unsigned, long))
2287}
2288
2289/// Splits a floating constant into its numeric body and its suffix.
2290///
2291/// The *body* is scanned forward rather than the suffix backwards, because a
2292/// suffix may hold digits of its own: `1.0f128` names `_Float128` and the
2293/// `128` is no part of the number. What is left after the digits, the point
2294/// and the exponent is the suffix, whatever it looks like; naming it is
2295/// [`Lexer::float_suffix`]'s business.
2296fn split_float_suffix(text: &str, hex: bool) -> (&str, &str) {
2297 let b = text.as_bytes();
2298 let mut i = 0;
2299 let (exponent, digit): (u8, fn(u8) -> bool) = if hex {
2300 i = 2; // the `0x` the caller has already recognised
2301 (b'p', |c| c.is_ascii_hexdigit())
2302 } else {
2303 (b'e', |c| c.is_ascii_digit())
2304 };
2305 while i < b.len() && (digit(b[i]) || b[i] == b'.') {
2306 i += 1;
2307 }
2308 if i < b.len() && b[i] | 0x20 == exponent {
2309 i += 1;
2310 if i < b.len() && (b[i] == b'+' || b[i] == b'-') {
2311 i += 1;
2312 }
2313 while i < b.len() && b[i].is_ascii_digit() {
2314 i += 1;
2315 }
2316 }
2317 (&text[..i], &text[i..])
2318}
2319
2320/// Parses the numeric body of a decimal floating constant.
2321fn parse_decimal_float(body: &str) -> Option<f64> {
2322 if body.is_empty() {
2323 return None;
2324 }
2325 // `None` is "no exponent at all", which is a different thing from an `e`
2326 // with nothing after it — `1e` is not a constant.
2327 let (mantissa, exponent) = match body.find(['e', 'E']) {
2328 Some(i) => (&body[..i], Some(&body[i + 1..])),
2329 None => (body, None),
2330 };
2331 let (int_part, frac_part) = match mantissa.find('.') {
2332 Some(i) => (&mantissa[..i], &mantissa[i + 1..]),
2333 None => (mantissa, ""),
2334 };
2335 if int_part.is_empty() && frac_part.is_empty() {
2336 return None;
2337 }
2338 if !int_part.bytes().all(|c| c.is_ascii_digit())
2339 || !frac_part.bytes().all(|c| c.is_ascii_digit())
2340 {
2341 return None;
2342 }
2343 let exponent = match exponent {
2344 None => 0i32,
2345 Some(exponent) => {
2346 let (sign, digits) = match exponent.as_bytes().first() {
2347 Some(b'+') => (1, &exponent[1..]),
2348 Some(b'-') => (-1, &exponent[1..]),
2349 _ => (1, exponent),
2350 };
2351 if digits.is_empty() || !digits.bytes().all(|c| c.is_ascii_digit()) {
2352 return None;
2353 }
2354 // Saturate: an absurd exponent simply becomes 0 or infinity.
2355 sign * digits.parse::<i32>().unwrap_or(i32::MAX / 2)
2356 }
2357 };
2358 let normalized = format!(
2359 "{}.{}e{}",
2360 if int_part.is_empty() { "0" } else { int_part },
2361 if frac_part.is_empty() { "0" } else { frac_part },
2362 exponent
2363 );
2364 normalized.parse::<f64>().ok()
2365}
2366
2367/// Parses the numeric body of a hexadecimal floating constant (`0x1.8p3`).
2368fn parse_hex_float(body: &str) -> Option<f64> {
2369 let rest = body
2370 .strip_prefix("0x")
2371 .or_else(|| body.strip_prefix("0X"))?;
2372 let p = rest.find(['p', 'P'])?;
2373 let (mantissa, exponent) = (&rest[..p], &rest[p + 1..]);
2374 let (int_part, frac_part) = match mantissa.find('.') {
2375 Some(i) => (&mantissa[..i], &mantissa[i + 1..]),
2376 None => (mantissa, ""),
2377 };
2378 if int_part.is_empty() && frac_part.is_empty() {
2379 return None;
2380 }
2381 let mut value = 0f64;
2382 for c in int_part.chars() {
2383 value = value * 16.0 + c.to_digit(16)? as f64;
2384 }
2385 let mut scale = 1.0 / 16.0;
2386 for c in frac_part.chars() {
2387 value += c.to_digit(16)? as f64 * scale;
2388 scale /= 16.0;
2389 }
2390 let (sign, digits) = match exponent.as_bytes().first() {
2391 Some(b'+') => (1i32, &exponent[1..]),
2392 Some(b'-') => (-1i32, &exponent[1..]),
2393 _ => (1i32, exponent),
2394 };
2395 if digits.is_empty() || !digits.bytes().all(|c| c.is_ascii_digit()) {
2396 return None;
2397 }
2398 let exp = sign * digits.parse::<i32>().unwrap_or(i32::MAX / 2);
2399 Some(value * 2f64.powi(exp))
2400}