Skip to main content

rucc_lex/
literal.rs

1//! Character constants and string literals: the escapes, the encoding prefixes, and what the
2//! elements end up being.
3//!
4//! Design: `spec/06-lexer-and-parser.md` section 6.1.
5//!
6//! This is the last piece of phase 7 that is about spellings. A literal arrives here as the
7//! bytes the user wrote, prefix and quotes included, and leaves as elements: one value per
8//! element of the array a string is, or one number for the character constant. What an element
9//! is depends on the prefix and on the target, which is most of the work.
10//!
11//! The execution character set is UTF-8 and the wide one is UTF-32, or UTF-16 where `wchar_t`
12//! is sixteen bits, which is Windows. That is what both compilers default to and it is the only
13//! choice that makes a UTF-8 source file mean what it looks like. `-fexec-charset` is not
14//! implemented and would change these answers if it ever is.
15//!
16//! # What an escape is worth
17//!
18//! There are two kinds of escape and the difference matters more than it looks. `é` names
19//! a character, so it is encoded in whatever the literal's encoding is, and in a plain string it
20//! becomes the two bytes `c3 a9`. `\xe9` is a value, not a character, so it is that one element
21//! and nothing encodes it. gcc 13.3 agrees on both: `"é"` is three bytes long and
22//! `"\xe9"` is two.
23//!
24//! A value escape that does not fit its element is truncated with a warning, and the element is
25//! what decides, not the type: `u8'\xff'` is fine and `'\x1ff'` is not, and both are eight bits
26//! wide. Octal runs to three digits and stops, so `"\1234"` is `S4` and not one escape, while
27//! hexadecimal runs as far as there are hexadecimal digits.
28//!
29//! An escape whose letter means nothing is that letter, with a warning, so `'\q'` is `'q'`.
30//! `\e` is the escape character in both compilers and in no standard. `\N{NAME}` is refused,
31//! because gcc 13.3 has it in C++23 alone and inventing our own answer would be worse.
32//!
33//! # Universal character names
34//!
35//! A UCN may not name a character in the basic character set, so a UCN that spells out the
36//! letter `A` is an error even though `'A'` is a constant. The exception is the three characters
37//! `$`, `@` and the backquote, which are below a space in the table and allowed anyway.
38//! Surrogates are refused. gcc still reports the basic character case in C23, where the standard
39//! relaxed it, so this follows gcc and not the paper. Before C99 a UCN is converted with a remark
40//! rather than refused, which is also what gcc does.
41//!
42//! A code point above `\U0010ffff` is an error here and a warning in gcc, which then encodes
43//! the value as though UTF-8 went that far. clang refuses it, this refuses it, and it is the
44//! one deliberate difference in this module.
45//!
46//! # Character constants
47//!
48//! A plain character constant is an `int` and not a `char`, and its value is the character
49//! converted to `int`, so `'\xff'` is minus one where plain `char` is signed and 255 where it
50//! is not. More than one character is implementation defined and both compilers shift them
51//! together, so `'ab'` is `0x6162`, with a warning. Past the width of the type the ones at the
52//! front fall off, so `'abcde'` is `0x62636465`, and gcc says "too long" there instead of
53//! "multi-character" rather than as well as it. A wide constant has room for exactly one, so
54//! `L'ab'` is `'b'` with the same warning, and a `u8` constant with two characters is an error
55//! rather than a warning. All measured on gcc 13.3, x86-64 Linux.
56//!
57//! # The prefixes and the dialects
58//!
59//! `L` is C89, `u` and `U` are C11, `u8` on a string is C11 and `u8` on a character constant is
60//! C23. In an older dialect gcc lexes `u8'a'` as the identifier `u8` followed by a character
61//! constant, which is a different token stream rather than a different constant. The scanner
62//! here reads it as one token in every dialect, which is the simpler rule and a divergence in
63//! the token stream that no real program can see, so the dialect is checked at this point
64//! instead and the constant is refused with a message that names the dialect it needs.
65
66use rucc_session::Std;
67use rucc_target::TargetInfo;
68
69use crate::remarks::Remarks;
70
71/// The encoding prefix of a character constant or a string literal.
72#[derive(Debug, Clone, Copy, PartialEq, Eq)]
73pub enum Encoding {
74    /// No prefix, whose element is a `char`.
75    Plain,
76    /// `L`, whose element is a `wchar_t` and so is a fact about the target.
77    Wide,
78    /// `u8`, whose element is a `char8_t`, which is an `unsigned char`.
79    Utf8,
80    /// `u`, whose element is a `char16_t`, which is a `uint_least16_t`.
81    Utf16,
82    /// `U`, whose element is a `char32_t`, which is a `uint_least32_t`.
83    Utf32,
84}
85
86impl Encoding {
87    /// The width of one element in bits.
88    #[must_use]
89    pub fn element_width(self, target: &TargetInfo) -> u32 {
90        match self {
91            Encoding::Plain | Encoding::Utf8 => 8,
92            Encoding::Wide => target.wchar_width,
93            Encoding::Utf16 => 16,
94            Encoding::Utf32 => 32,
95        }
96    }
97
98    /// Whether the element type is signed, which only `char` and `wchar_t` can be.
99    #[must_use]
100    pub fn is_signed(self, target: &TargetInfo) -> bool {
101        match self {
102            Encoding::Plain => target.char_is_signed,
103            Encoding::Wide => target.wchar_is_signed,
104            Encoding::Utf8 | Encoding::Utf16 | Encoding::Utf32 => false,
105        }
106    }
107
108    /// The prefix at the front of a spelling, and how many bytes it took.
109    fn read(bytes: &[u8]) -> (Encoding, usize) {
110        match bytes {
111            [b'u', b'8', ..] => (Encoding::Utf8, 2),
112            [b'u', ..] => (Encoding::Utf16, 1),
113            [b'U', ..] => (Encoding::Utf32, 1),
114            [b'L', ..] => (Encoding::Wide, 1),
115            _ => (Encoding::Plain, 0),
116        }
117    }
118
119    /// The first dialect that has this prefix, which is not the same for a character constant
120    /// as for a string literal.
121    fn since(self, character: bool) -> Std {
122        match self {
123            Encoding::Plain | Encoding::Wide => Std::C89,
124            Encoding::Utf8 if character => Std::C23,
125            Encoding::Utf8 | Encoding::Utf16 | Encoding::Utf32 => Std::C11,
126        }
127    }
128}
129
130/// A converted character constant.
131#[derive(Debug, Clone, Copy, PartialEq, Eq)]
132pub struct CharConstant {
133    /// The value, already converted to the constant's type, which is why it can be negative:
134    /// `'\xff'` is minus one where plain `char` is signed.
135    pub value: i64,
136    /// The prefix it was written with.
137    pub encoding: Encoding,
138    /// What is worth saying about it, for the caller that holds the span.
139    pub remarks: Remarks,
140}
141
142/// A converted string literal.
143#[derive(Debug, Clone, PartialEq, Eq)]
144pub struct StringLiteral {
145    /// One value per element of the array, without the terminating zero, since the zero belongs
146    /// to the type and not to the spelling. An element is a byte in a plain or `u8` literal, a
147    /// UTF-16 code unit in a `u` one, and a code point in a `U` one.
148    pub elements: Vec<u32>,
149    /// The prefix it was written with.
150    pub encoding: Encoding,
151    /// What is worth saying about it, for the caller that holds the span.
152    pub remarks: Remarks,
153}
154
155impl StringLiteral {
156    /// The bytes this literal becomes in the object, terminator included, in the target's byte
157    /// order.
158    #[must_use]
159    pub fn bytes(&self, target: &TargetInfo) -> Vec<u8> {
160        let width = self.encoding.element_width(target) / 8;
161        let mut bytes = Vec::with_capacity((self.elements.len() + 1) * width as usize);
162        for element in self.elements.iter().copied().chain([0]) {
163            let taken = &element.to_le_bytes()[..width as usize];
164            if target.little_endian {
165                bytes.extend_from_slice(taken);
166            } else {
167                bytes.extend(taken.iter().rev());
168            }
169        }
170        bytes
171    }
172}
173
174/// Why a spelling is not a literal.
175#[derive(Debug, Clone, Copy, PartialEq, Eq)]
176pub enum LiteralError {
177    /// The spelling is not a character constant or a string literal at all, which the caller
178    /// cannot reach from a token the scanner produced.
179    NotALiteral,
180    /// `''`, which has no character in it to have a value.
181    Empty,
182    /// More characters than the type has room for, in the one encoding where that is an error
183    /// rather than a warning.
184    TooLong,
185    /// `\x` with no hexadecimal digit after it.
186    NoHexDigits,
187    /// A universal character name that stops before its digits are done.
188    IncompleteUcn,
189    /// A universal character name that may not name what it names: a character in the basic
190    /// character set, a surrogate, or a code point past the end of Unicode.
191    InvalidUcn,
192    /// `\N{NAME}`, which gcc 13.3 has in C++23 and in no C dialect.
193    NamedUcn,
194    /// A byte in the source that is not part of a character, in a literal whose encoding has to
195    /// know what the characters are.
196    InvalidUtf8,
197    /// An encoding prefix the dialect does not have.
198    PrefixNotInDialect,
199}
200
201impl LiteralError {
202    /// What to print, in GCC's words where GCC has any.
203    #[must_use]
204    pub const fn message(self) -> &'static str {
205        match self {
206            LiteralError::NotALiteral => "not a character constant or a string literal",
207            LiteralError::Empty => "empty character constant",
208            LiteralError::TooLong => "character constant too long for its type",
209            LiteralError::NoHexDigits => "\\x used with no following hex digits",
210            LiteralError::IncompleteUcn => "incomplete universal character name",
211            LiteralError::InvalidUcn => "not a valid universal character",
212            LiteralError::NamedUcn => "named universal character escapes are not supported yet",
213            LiteralError::InvalidUtf8 => "failure to convert the source to the execution charset",
214            LiteralError::PrefixNotInDialect => {
215                "this encoding prefix is not available in this dialect"
216            }
217        }
218    }
219}
220
221/// Converts the spelling of a character constant into a value.
222///
223/// # Errors
224///
225/// [`LiteralError`], for a spelling that is not a character constant or that holds an escape
226/// that is not one.
227pub fn character(text: &str, std: Std, target: &TargetInfo) -> Result<CharConstant, LiteralError> {
228    let (encoding, body) = open(text, b'\'', std, true)?;
229    let width = encoding.element_width(target);
230    let mut reader = Reader { bytes: body, index: 0, std, remarks: Remarks::NONE };
231
232    // The elements are shifted together into one number, which is what both compilers do with a
233    // constant holding more than one, and the ones that fall off the top are the ones the
234    // warning is about.
235    let mut value: u64 = 0;
236    let mut count = 0u32;
237    while let Some(piece) = reader.next(width)? {
238        for element in piece.elements(width) {
239            value = (value << width) | u64::from(element);
240            count += 1;
241        }
242    }
243    let mut remarks = reader.remarks;
244
245    // A plain constant is an `int` however many characters it holds, and every other kind is
246    // exactly one element wide, so that is how many characters fit.
247    let type_width = if encoding == Encoding::Plain { 32 } else { width };
248    let capacity = type_width / width;
249    match count {
250        0 => return Err(LiteralError::Empty),
251        1 => {}
252        _ if encoding == Encoding::Utf8 => return Err(LiteralError::TooLong),
253        _ if count > capacity => remarks = remarks.with(Remarks::TOO_LONG),
254        _ => remarks = remarks.with(Remarks::MULTICHARACTER),
255    }
256
257    // One character is converted to the constant's type from the element's, which is where the
258    // sign of a plain `char` gets in. More than one is already a number of the constant's type
259    // and nothing sign extends it from any narrower width.
260    let (bits, signed) = if count == 1 {
261        (width, encoding.is_signed(target))
262    } else {
263        (type_width, encoding == Encoding::Plain || encoding.is_signed(target))
264    };
265    Ok(CharConstant { value: narrow(value, bits, signed), encoding, remarks })
266}
267
268/// Converts the spelling of a string literal into its elements.
269///
270/// # Errors
271///
272/// [`LiteralError`], for a spelling that is not a string literal or that holds an escape that
273/// is not one.
274pub fn string(text: &str, std: Std, target: &TargetInfo) -> Result<StringLiteral, LiteralError> {
275    let (encoding, body) = open(text, b'"', std, false)?;
276    let width = encoding.element_width(target);
277    let mut reader = Reader { bytes: body, index: 0, std, remarks: Remarks::NONE };
278    let mut elements = Vec::new();
279    while let Some(piece) = reader.next(width)? {
280        elements.extend(piece.elements(width));
281    }
282    Ok(StringLiteral { elements, encoding, remarks: reader.remarks })
283}
284
285/// Reads the prefix and the quotes, and hands back the encoding and what is between them.
286fn open(
287    text: &str,
288    quote: u8,
289    std: Std,
290    character: bool,
291) -> Result<(Encoding, &[u8]), LiteralError> {
292    let bytes = text.as_bytes();
293    let (encoding, prefix) = Encoding::read(bytes);
294    if std < encoding.since(character) {
295        return Err(LiteralError::PrefixNotInDialect);
296    }
297    let rest = &bytes[prefix..];
298    match rest {
299        [first, .., last] if *first == quote && *last == quote => {
300            Ok((encoding, &rest[1..rest.len() - 1]))
301        }
302        _ => Err(LiteralError::NotALiteral),
303    }
304}
305
306/// Cuts a value down to `bits` and reads it back as signed or unsigned.
307fn narrow(value: u64, bits: u32, signed: bool) -> i64 {
308    let masked = if bits >= 64 { value } else { value & ((1u64 << bits) - 1) };
309    if signed && bits < 64 && masked >> (bits - 1) & 1 == 1 {
310        // The bits above the width are the sign, which is what makes `'\xff'` minus one.
311        (masked | !((1u64 << bits) - 1)) as i64
312    } else {
313        masked as i64
314    }
315}
316
317/// One piece of a literal, before it is turned into elements.
318#[derive(Debug, Clone, Copy, PartialEq, Eq)]
319enum Piece {
320    /// A character, which the encoding has to encode: a character from the source or one named
321    /// by a universal character name.
322    Char(u32),
323    /// A value written as a value, with `\x` or with octal digits, which is one element as
324    /// written and is not encoded.
325    Value(u32),
326}
327
328impl Piece {
329    /// The elements this piece becomes in an encoding of the given element width.
330    fn elements(self, width: u32) -> Vec<u32> {
331        let code = match self {
332            Piece::Value(value) => return vec![value],
333            Piece::Char(code) => code,
334        };
335        match width {
336            8 => {
337                let mut buffer = [0u8; 4];
338                let text = char::from_u32(code)
339                    .map(|character| character.encode_utf8(&mut buffer).len())
340                    .unwrap_or(0);
341                buffer[..text].iter().map(|&byte| u32::from(byte)).collect()
342            }
343            // UTF-16 is the one encoding where a character can take two elements, which is why
344            // a wide string is not the same length on Windows as it is anywhere else.
345            16 if code > 0xffff => {
346                let value = code - 0x1_0000;
347                vec![0xd800 + (value >> 10), 0xdc00 + (value & 0x3ff)]
348            }
349            _ => vec![code],
350        }
351    }
352}
353
354/// Reads the body of a literal one character at a time.
355struct Reader<'a> {
356    /// What is between the quotes.
357    bytes: &'a [u8],
358    /// How far in the reader is.
359    index: usize,
360    /// The dialect, which decides what is worth a remark.
361    std: Std,
362    /// What the literal has earned so far.
363    remarks: Remarks,
364}
365
366impl Reader<'_> {
367    /// The next piece, or [`None`] at the end of the literal.
368    fn next(&mut self, width: u32) -> Result<Option<Piece>, LiteralError> {
369        let Some(&byte) = self.bytes.get(self.index) else {
370            return Ok(None);
371        };
372        self.index += 1;
373        if byte == b'\\' {
374            return self.escape(width).map(Some);
375        }
376        if byte < 0x80 {
377            return Ok(Some(Piece::Char(u32::from(byte))));
378        }
379        // A byte above ASCII begins a character in the source, which is UTF-8. A narrow literal
380        // is UTF-8 too, so its bytes go through untouched and nothing has to be able to decode
381        // them; a wide one has to know which character this is before it can encode it again.
382        if width == 8 {
383            return Ok(Some(Piece::Value(u32::from(byte))));
384        }
385        // Only this one character is decoded, and not the rest of the literal, because what
386        // comes after it may be an escape holding a byte that no character begins with.
387        let length = utf8_length(byte).ok_or(LiteralError::InvalidUtf8)?;
388        let end = self.index - 1 + length;
389        let text = self
390            .bytes
391            .get(self.index - 1..end)
392            .and_then(|slice| std::str::from_utf8(slice).ok())
393            .ok_or(LiteralError::InvalidUtf8)?;
394        let character = text.chars().next().ok_or(LiteralError::InvalidUtf8)?;
395        self.index = end;
396        Ok(Some(Piece::Char(character as u32)))
397    }
398
399    /// The piece an escape sequence is worth, with the backslash already read.
400    fn escape(&mut self, width: u32) -> Result<Piece, LiteralError> {
401        let Some(&byte) = self.bytes.get(self.index) else {
402            // The scanner reports the missing quote, and there is nothing here to convert.
403            return Err(LiteralError::NotALiteral);
404        };
405        self.index += 1;
406        let simple = match byte {
407            b'n' => Some(0x0a),
408            b't' => Some(0x09),
409            b'r' => Some(0x0d),
410            b'a' => Some(0x07),
411            b'b' => Some(0x08),
412            b'f' => Some(0x0c),
413            b'v' => Some(0x0b),
414            b'\\' | b'\'' | b'"' | b'?' => Some(u32::from(byte)),
415            _ => None,
416        };
417        if let Some(value) = simple {
418            return Ok(Piece::Value(value));
419        }
420        match byte {
421            // The escape character, which both compilers have and no standard does.
422            b'e' | b'E' => {
423                self.remarks = self.remarks.with(Remarks::NON_ISO_ESCAPE);
424                Ok(Piece::Value(0x1b))
425            }
426            b'0'..=b'7' => Ok(Piece::Value(self.octal(byte, width))),
427            b'x' => self.hex(width).map(Piece::Value),
428            b'u' | b'U' => self.ucn(byte).map(Piece::Char),
429            b'N' => Err(LiteralError::NamedUcn),
430            // An escape that means nothing is the character itself, which both compilers do
431            // after a warning rather than refusing the program.
432            _ => {
433                self.remarks = self.remarks.with(Remarks::UNKNOWN_ESCAPE);
434                Ok(Piece::Value(u32::from(byte)))
435            }
436        }
437    }
438
439    /// An octal escape, which is at most three digits however many follow, so that `"\1234"` is
440    /// two characters.
441    fn octal(&mut self, first: u8, width: u32) -> u32 {
442        let mut value = u32::from(first - b'0');
443        for _ in 0..2 {
444            match self.bytes.get(self.index) {
445                Some(&byte @ b'0'..=b'7') => {
446                    value = value * 8 + u32::from(byte - b'0');
447                    self.index += 1;
448                }
449                _ => break,
450            }
451        }
452        self.fit(value, width)
453    }
454
455    /// A hexadecimal escape, which runs as far as there are hexadecimal digits.
456    fn hex(&mut self, width: u32) -> Result<u32, LiteralError> {
457        let mut value: u64 = 0;
458        let mut digits = 0;
459        while let Some(digit) = self.bytes.get(self.index).and_then(|&byte| hex_digit(byte)) {
460            // A value far past the width is truncated anyway, so the accumulator stops growing
461            // rather than overflowing, and the remark still gets made.
462            value = value.saturating_mul(16).saturating_add(u64::from(digit));
463            digits += 1;
464            self.index += 1;
465        }
466        if digits == 0 {
467            return Err(LiteralError::NoHexDigits);
468        }
469        Ok(self.fit(u32::try_from(value).unwrap_or(u32::MAX), width))
470    }
471
472    /// A universal character name, with its `u` or `U` already read.
473    fn ucn(&mut self, marker: u8) -> Result<u32, LiteralError> {
474        if self.bytes.get(self.index) == Some(&b'{') {
475            return Err(LiteralError::NamedUcn);
476        }
477        let digits = if marker == b'u' { 4 } else { 8 };
478        let mut value: u32 = 0;
479        for _ in 0..digits {
480            let Some(digit) = self.bytes.get(self.index).and_then(|&byte| hex_digit(byte)) else {
481                return Err(LiteralError::IncompleteUcn);
482            };
483            value = value * 16 + digit;
484            self.index += 1;
485        }
486        // The basic character set is off limits, and so is everything below ` ` except the
487        // three characters the standard lets through. gcc still says so in C23, where the
488        // wording was relaxed, so this follows the compiler rather than the paper.
489        let allowed_low = matches!(value, 0x24 | 0x40 | 0x60);
490        if (value < 0xa0 && !allowed_low) || (0xd800..=0xdfff).contains(&value) || value > 0x10ffff
491        {
492            return Err(LiteralError::InvalidUcn);
493        }
494        if self.std < Std::C99 {
495            self.remarks = self.remarks.with(Remarks::UCN);
496        }
497        Ok(value)
498    }
499
500    /// Cuts an escape down to the element it is written in, and says so when that loses
501    /// something.
502    fn fit(&mut self, value: u32, width: u32) -> u32 {
503        if width >= 32 {
504            return value;
505        }
506        let mask = (1u32 << width) - 1;
507        if value & !mask != 0 {
508            self.remarks = self.remarks.with(Remarks::ESCAPE_OUT_OF_RANGE);
509        }
510        value & mask
511    }
512}
513
514/// The value of a hexadecimal digit, and [`None`] when the byte is not one.
515fn hex_digit(byte: u8) -> Option<u32> {
516    char::from(byte).to_digit(16)
517}
518
519/// How many bytes the character starting with this one takes, and [`None`] when no character
520/// starts with it.
521fn utf8_length(byte: u8) -> Option<usize> {
522    match byte {
523        0x00..=0x7f => Some(1),
524        0xc2..=0xdf => Some(2),
525        0xe0..=0xef => Some(3),
526        0xf0..=0xf4 => Some(4),
527        _ => None,
528    }
529}
530
531#[cfg(test)]
532mod tests {
533    use rucc_target::Triple;
534
535    use super::*;
536
537    fn linux() -> TargetInfo {
538        TargetInfo::new("x86_64-unknown-linux-gnu".parse::<Triple>().expect("a known triple"))
539    }
540
541    fn windows() -> TargetInfo {
542        TargetInfo::new("x86_64-pc-windows-msvc".parse::<Triple>().expect("a known triple"))
543    }
544
545    fn arm() -> TargetInfo {
546        TargetInfo::new("aarch64-unknown-linux-gnu".parse::<Triple>().expect("a known triple"))
547    }
548
549    /// The value of a character constant on x86-64 Linux, in C23.
550    fn ch(text: &str) -> i64 {
551        character(text, Std::C23, &linux()).expect("a character constant").value
552    }
553
554    /// The remarks a character constant earns on x86-64 Linux, in C23.
555    fn ch_remarks(text: &str) -> Remarks {
556        character(text, Std::C23, &linux()).expect("a character constant").remarks
557    }
558
559    /// What a character constant goes wrong with.
560    fn ch_error(text: &str) -> LiteralError {
561        character(text, Std::C23, &linux()).expect_err("not a character constant")
562    }
563
564    /// The elements of a string literal on x86-64 Linux, in C23.
565    fn str_elements(text: &str) -> Vec<u32> {
566        string(text, Std::C23, &linux()).expect("a string literal").elements
567    }
568
569    /// The bytes a string literal becomes on x86-64 Linux, terminator included.
570    fn str_bytes(text: &str) -> Vec<u8> {
571        string(text, Std::C23, &linux()).expect("a string literal").bytes(&linux())
572    }
573
574    #[test]
575    fn the_ordinary_cases_are_the_characters_they_look_like() {
576        assert_eq!(ch("'a'"), 0x61);
577        assert_eq!(ch(r"'\n'"), 0x0a);
578        assert_eq!(ch(r"'\0'"), 0);
579        assert_eq!(ch(r"'\\'"), 0x5c);
580        assert_eq!(ch(r"'\''"), 0x27);
581        assert_eq!(ch(r#"'\"'"#), 0x22);
582        assert_eq!(ch(r"'\?'"), 0x3f);
583        assert_eq!(str_elements(r#""hi""#), vec![0x68, 0x69]);
584    }
585
586    /// A single character in a plain constant goes through plain `char` on the way to `int`,
587    /// which is the whole reason `'\xff'` is a negative number on one target and a positive
588    /// one on another.
589    #[test]
590    fn a_high_character_takes_the_sign_of_plain_char() {
591        assert_eq!(ch(r"'\xff'"), -1);
592        assert_eq!(ch(r"'\377'"), -1);
593        assert_eq!(character(r"'\xff'", Std::C23, &arm()).expect("a constant").value, 255);
594        // Not a `char`, so nothing sign extends it.
595        assert_eq!(ch(r"u8'\xff'"), 255);
596    }
597
598    /// Measured on GCC 13.3, x86-64 Linux. A value escape is truncated to its element and the
599    /// warning is about the truncation, not about the type.
600    #[test]
601    fn an_escape_too_big_for_its_element_is_truncated_and_says_so() {
602        let out = character(r"'\x1ff'", Std::C23, &linux()).expect("a constant");
603        assert_eq!(out.value, -1);
604        assert!(out.remarks.has(Remarks::ESCAPE_OUT_OF_RANGE));
605        let out = character(r"'\400'", Std::C23, &linux()).expect("a constant");
606        assert_eq!(out.value, 0);
607        assert!(out.remarks.has(Remarks::ESCAPE_OUT_OF_RANGE));
608        // Wide enough to hold it, so there is nothing to say.
609        assert!(!ch_remarks(r"L'\x1ff'").has(Remarks::ESCAPE_OUT_OF_RANGE));
610        assert_eq!(ch(r"L'\x1ff'"), 0x1ff);
611    }
612
613    /// Also measured on GCC 13.3. The characters are shifted together, the ones past the width
614    /// of the type fall off the front, and past that point GCC says "too long" instead of
615    /// "multi-character" rather than as well as it.
616    #[test]
617    fn more_than_one_character_shifts_them_together() {
618        assert_eq!(ch("'ab'"), 0x6162);
619        assert_eq!(ch("'abc'"), 0x616263);
620        assert_eq!(ch("'abcd'"), 0x61626364);
621        assert_eq!(ch("'abcde'"), 0x62636465);
622        assert_eq!(ch(r"'\xff\xfe'"), 0xfffe);
623        assert_eq!(ch(r"'\xff\xff\xff\xff'"), -1);
624        assert_eq!(ch(r"'\x80\x00'"), 0x8000);
625
626        assert!(ch_remarks("'ab'").has(Remarks::MULTICHARACTER));
627        assert!(ch_remarks("'abcd'").has(Remarks::MULTICHARACTER));
628        assert!(ch_remarks("'abcde'").has(Remarks::TOO_LONG));
629        assert!(!ch_remarks("'abcde'").has(Remarks::MULTICHARACTER));
630        assert!(!ch_remarks("'a'").has(Remarks::MULTICHARACTER));
631    }
632
633    /// A wide constant has room for exactly one character, so two is already too many and the
634    /// last one is what survives. `u8` is the one encoding where GCC makes this an error.
635    #[test]
636    fn a_prefixed_constant_holds_one_character_and_keeps_the_last() {
637        for text in [r"L'ab'", r"u'ab'", r"U'ab'"] {
638            let out = character(text, Std::C23, &linux()).expect("a constant");
639            assert_eq!(out.value, 0x62, "{text}");
640            assert!(out.remarks.has(Remarks::TOO_LONG), "{text}");
641        }
642        assert_eq!(ch_error("u8'ab'"), LiteralError::TooLong);
643        assert_eq!(ch_error("u8'é'"), LiteralError::TooLong);
644    }
645
646    #[test]
647    fn the_empty_constant_has_no_value_to_have() {
648        assert_eq!(ch_error("''"), LiteralError::Empty);
649        assert_eq!(ch_error("L''"), LiteralError::Empty);
650        // The empty string is fine, and is one element long once the terminator is there.
651        assert_eq!(str_elements(r#""""#), Vec::<u32>::new());
652        assert_eq!(str_bytes(r#""""#), vec![0]);
653    }
654
655    /// A character from the source is encoded in the literal's encoding, so the same `é` is
656    /// two bytes in one constant and one code point in another.
657    #[test]
658    fn a_source_character_is_encoded_and_an_escape_is_not() {
659        assert_eq!(ch("'é'"), 0xc3a9);
660        assert_eq!(ch("L'é'"), 0xe9);
661        assert_eq!(ch("u'€'"), 0x20ac);
662        assert_eq!(ch(r"U'\U0001F600'"), 0x1f600);
663        // The same character in a plain constant is its UTF-8 bytes, which makes it a
664        // multi-character constant and a negative number.
665        assert_eq!(ch(r"'\U0001F600'"), i64::from(0xf09f_9880u32 as i32));
666        assert!(ch_remarks(r"'\U0001F600'").has(Remarks::MULTICHARACTER));
667    }
668
669    /// Both compilers have `\e` and neither standard does, and an escape that means nothing is
670    /// the letter itself after a warning rather than an error.
671    #[test]
672    fn the_escapes_outside_the_standard_still_have_values() {
673        assert_eq!(ch(r"'\e'"), 0x1b);
674        assert!(ch_remarks(r"'\e'").has(Remarks::NON_ISO_ESCAPE));
675        assert_eq!(ch(r"'\q'"), 0x71);
676        assert!(ch_remarks(r"'\q'").has(Remarks::UNKNOWN_ESCAPE));
677        assert_eq!(ch_error(r"'\x'"), LiteralError::NoHexDigits);
678        assert_eq!(ch_error(r"'\N{LATIN SMALL LETTER A}'"), LiteralError::NamedUcn);
679    }
680
681    /// A universal character name may not name a character in the basic character set, which
682    /// GCC still enforces in C23 where the wording was relaxed, and the three characters below
683    /// a space that are allowed anyway are allowed here too.
684    #[test]
685    fn a_universal_character_name_may_not_name_just_anything() {
686        assert_eq!(ch("'\\u0024'"), 0x24);
687        assert_eq!(ch("'\\u00e9'"), 0xc3a9);
688        assert_eq!(ch_error("'\\u0041'"), LiteralError::InvalidUcn);
689        assert_eq!(ch_error(r"'\ud800'"), LiteralError::InvalidUcn);
690        assert_eq!(ch_error(r"'\u00'"), LiteralError::IncompleteUcn);
691        // GCC warns here and encodes the value anyway. clang refuses it and so does this.
692        assert_eq!(ch_error(r"'\U00110000'"), LiteralError::InvalidUcn);
693    }
694
695    #[test]
696    fn a_universal_character_name_before_c99_is_worth_a_remark() {
697        let out = character("'\\u00e9'", Std::C89, &linux()).expect("a constant");
698        assert!(out.remarks.has(Remarks::UCN));
699        let out = character("'\\u00e9'", Std::C99, &linux()).expect("a constant");
700        assert!(!out.remarks.has(Remarks::UCN));
701    }
702
703    /// Octal runs to three digits and stops, and hexadecimal runs as far as the digits go, so
704    /// `"\1234"` is two characters and `"\x41z"` is two as well.
705    #[test]
706    fn an_octal_escape_ends_and_a_hex_escape_does_not() {
707        assert_eq!(str_elements(r#""\1234""#), vec![0x53, 0x34]);
708        assert_eq!(str_elements(r#""\x41z""#), vec![0x41, 0x7a]);
709        assert_eq!(str_elements(r#""\x41""#), vec![0x41]);
710    }
711
712    /// The sizes GCC reports for these, which is the elements plus the terminator times the
713    /// width of one.
714    #[test]
715    fn a_string_is_as_many_bytes_as_its_encoding_makes_it() {
716        assert_eq!(str_bytes(r#""abc""#).len(), 4);
717        assert_eq!(str_bytes(r#"L"abc""#).len(), 16);
718        assert_eq!(str_bytes(r#"u"abc""#).len(), 8);
719        assert_eq!(str_bytes(r#"U"abc""#).len(), 16);
720        assert_eq!(str_bytes(r#"u8"abc""#).len(), 4);
721        // A zero in the middle is an element like any other, and the terminator is still added.
722        assert_eq!(str_bytes(r#""a\0b""#), vec![0x61, 0x00, 0x62, 0x00]);
723        assert_eq!(str_bytes(r#""é""#), vec![0xc3, 0xa9, 0x00]);
724    }
725
726    /// The one encoding where a character can take two elements, which is why a wide string is
727    /// not the same length on Windows as it is anywhere else.
728    #[test]
729    fn utf16_splits_the_characters_that_do_not_fit_into_a_surrogate_pair() {
730        assert_eq!(
731            str_elements(r#"u8"é€😀""#),
732            vec![0xc3, 0xa9, 0xe2, 0x82, 0xac, 0xf0, 0x9f, 0x98, 0x80]
733        );
734        assert_eq!(str_elements(r#"u"€😀""#), vec![0x20ac, 0xd83d, 0xde00]);
735        assert_eq!(str_elements(r#"U"€😀""#), vec![0x20ac, 0x1f600]);
736    }
737
738    /// A wide literal is UTF-16 on Windows and UTF-32 everywhere else, so the same three
739    /// characters are four elements on one target and three on the other.
740    #[test]
741    fn a_wide_literal_is_whatever_the_target_makes_wchar_t() {
742        let text = r#"L"a😀""#;
743        let here = string(text, Std::C23, &linux()).expect("a string");
744        assert_eq!(here.elements, vec![0x61, 0x1f600]);
745        assert_eq!(here.bytes(&linux()).len(), 12);
746        let there = string(text, Std::C23, &windows()).expect("a string");
747        assert_eq!(there.elements, vec![0x61, 0xd83d, 0xde00]);
748        assert_eq!(there.bytes(&windows()).len(), 8);
749        // And a wide character constant takes the sign of `wchar_t`, which is not the same on
750        // every target either.
751        assert_eq!(character(r"L'\xffffffff'", Std::C23, &linux()).expect("a constant").value, -1);
752        assert_eq!(
753            character(r"L'\xffffffff'", Std::C23, &arm()).expect("a constant").value,
754            0xffff_ffff
755        );
756    }
757
758    /// Every target the compiler has is little-endian, so the other order is checked by
759    /// flipping the field rather than by naming a target, and this is the test that fails on
760    /// the day a big-endian one arrives with the layout still assuming otherwise.
761    #[test]
762    fn the_bytes_come_out_in_the_targets_order() {
763        let mut big = linux();
764        big.little_endian = false;
765        let literal = string(r#"u"ab""#, Std::C23, &big).expect("a string");
766        assert_eq!(literal.bytes(&big), vec![0x00, 0x61, 0x00, 0x62, 0x00, 0x00]);
767        assert_eq!(literal.bytes(&linux()), vec![0x61, 0x00, 0x62, 0x00, 0x00, 0x00]);
768    }
769
770    /// `L` is C89, `u` and `U` are C11, and `u8` is C11 on a string and C23 on a character
771    /// constant, which is the one place the two differ.
772    #[test]
773    fn a_prefix_is_only_available_in_the_dialect_that_has_it() {
774        assert!(character("L'a'", Std::C89, &linux()).is_ok());
775        assert_eq!(
776            character("u'a'", Std::C99, &linux()).expect_err("not in C99"),
777            LiteralError::PrefixNotInDialect
778        );
779        assert!(character("u'a'", Std::C11, &linux()).is_ok());
780        assert!(string(r#"u8"a""#, Std::C11, &linux()).is_ok());
781        assert_eq!(
782            character("u8'a'", Std::C11, &linux()).expect_err("not in C11"),
783            LiteralError::PrefixNotInDialect
784        );
785        assert!(character("u8'a'", Std::C23, &linux()).is_ok());
786    }
787
788    /// The widths and signs the elements have, which is what the parser will turn into the
789    /// type of the literal.
790    #[test]
791    fn an_element_is_as_wide_as_the_encoding_and_the_target_agree() {
792        let target = linux();
793        assert_eq!(Encoding::Plain.element_width(&target), 8);
794        assert_eq!(Encoding::Utf8.element_width(&target), 8);
795        assert_eq!(Encoding::Utf16.element_width(&target), 16);
796        assert_eq!(Encoding::Utf32.element_width(&target), 32);
797        assert_eq!(Encoding::Wide.element_width(&target), 32);
798        assert_eq!(Encoding::Wide.element_width(&windows()), 16);
799
800        assert!(Encoding::Plain.is_signed(&target));
801        assert!(!Encoding::Plain.is_signed(&arm()));
802        assert!(Encoding::Wide.is_signed(&target));
803        assert!(!Encoding::Wide.is_signed(&arm()));
804        assert!(!Encoding::Utf8.is_signed(&target));
805        assert!(!Encoding::Utf16.is_signed(&target));
806        assert!(!Encoding::Utf32.is_signed(&target));
807    }
808
809    #[test]
810    fn a_spelling_that_is_not_a_literal_is_refused_rather_than_guessed_at() {
811        assert_eq!(ch_error("a"), LiteralError::NotALiteral);
812        assert_eq!(ch_error("'a"), LiteralError::NotALiteral);
813        assert_eq!(
814            string("'a'", Std::C23, &linux()).expect_err("not a string"),
815            LiteralError::NotALiteral
816        );
817        assert_eq!(ch_error("'"), LiteralError::NotALiteral);
818    }
819
820    #[test]
821    fn every_error_has_something_to_print() {
822        for error in [
823            LiteralError::NotALiteral,
824            LiteralError::Empty,
825            LiteralError::TooLong,
826            LiteralError::NoHexDigits,
827            LiteralError::IncompleteUcn,
828            LiteralError::InvalidUcn,
829            LiteralError::NamedUcn,
830            LiteralError::InvalidUtf8,
831            LiteralError::PrefixNotInDialect,
832        ] {
833            assert!(!error.message().is_empty());
834        }
835    }
836}