Skip to main content

rucc_lex/
literal.rs

1//! Character constants and string literals: the escapes, the encoding prefixes, and what the
2//! elements end up being.
3//!
4//! Design: `spec/06-lexer-and-parser.md` section 6.1.
5//!
6//! This is the last piece of phase 7 that is about spellings. A literal arrives here as the
7//! bytes the user wrote, prefix and quotes included, and leaves as elements: one value per
8//! element of the array a string is, or one number for the character constant. What an element
9//! is depends on the prefix and on the target, which is most of the work.
10//!
11//! The execution character set is UTF-8 and the wide one is UTF-32, or UTF-16 where `wchar_t`
12//! is sixteen bits, which is Windows. That is what both compilers default to and it is the only
13//! choice that makes a UTF-8 source file mean what it looks like. `-fexec-charset` is not
14//! implemented and would change these answers if it ever is.
15//!
16//! # What an escape is worth
17//!
18//! There are two kinds of escape and the difference matters more than it looks. `é` names
19//! a character, so it is encoded in whatever the literal's encoding is, and in a plain string it
20//! becomes the two bytes `c3 a9`. `\xe9` is a value, not a character, so it is that one element
21//! and nothing encodes it. gcc 13.3 agrees on both: `"é"` is three bytes long and
22//! `"\xe9"` is two.
23//!
24//! A value escape that does not fit its element is truncated with a warning, and the element is
25//! what decides, not the type: `u8'\xff'` is fine and `'\x1ff'` is not, and both are eight bits
26//! wide. Octal runs to three digits and stops, so `"\1234"` is `S4` and not one escape, while
27//! hexadecimal runs as far as there are hexadecimal digits.
28//!
29//! An escape whose letter means nothing is that letter, with a warning, so `'\q'` is `'q'`.
30//! `\e` is the escape character in both compilers and in no standard. `\N{NAME}` is refused,
31//! because gcc 13.3 has it in C++23 alone and inventing our own answer would be worse.
32//!
33//! # Universal character names
34//!
35//! A UCN may not name a character in the basic character set, so a UCN that spells out the
36//! letter `A` is an error even though `'A'` is a constant. The exception is the three characters
37//! `$`, `@` and the backquote, which are below a space in the table and allowed anyway.
38//! Surrogates are refused. gcc still reports the basic character case in C23, where the standard
39//! relaxed it, so this follows gcc and not the paper. Before C99 a UCN is converted with a remark
40//! rather than refused, which is also what gcc does.
41//!
42//! A code point above `\U0010ffff` is an error here and a warning in gcc, which then encodes
43//! the value as though UTF-8 went that far. clang refuses it, this refuses it, and it is the
44//! one deliberate difference in this module.
45//!
46//! # Character constants
47//!
48//! A plain character constant is an `int` and not a `char`, and its value is the character
49//! converted to `int`, so `'\xff'` is minus one where plain `char` is signed and 255 where it
50//! is not. More than one character is implementation defined and both compilers shift them
51//! together, so `'ab'` is `0x6162`, with a warning. Past the width of the type the ones at the
52//! front fall off, so `'abcde'` is `0x62636465`, and gcc says "too long" there instead of
53//! "multi-character" rather than as well as it. A wide constant has room for exactly one, so
54//! `L'ab'` is `'b'` with the same warning, and a `u8` constant with two characters is an error
55//! rather than a warning. All measured on gcc 13.3, x86-64 Linux.
56//!
57//! # The prefixes and the dialects
58//!
59//! `L` is C89, `u` and `U` are C11, `u8` on a string is C11 and `u8` on a character constant is
60//! C23. In an older dialect gcc lexes `u8'a'` as the identifier `u8` followed by a character
61//! constant, which is a different token stream rather than a different constant. The scanner
62//! here reads it as one token in every dialect, which is the simpler rule and a divergence in
63//! the token stream that no real program can see, so the dialect is checked at this point
64//! instead and the constant is refused with a message that names the dialect it needs.
65
66use rucc_session::Std;
67use rucc_target::TargetInfo;
68
69use crate::remarks::Remarks;
70
71/// The encoding prefix of a character constant or a string literal.
72#[derive(Debug, Clone, Copy, PartialEq, Eq)]
73pub enum Encoding {
74    /// No prefix, whose element is a `char`.
75    Plain,
76    /// `L`, whose element is a `wchar_t` and so is a fact about the target.
77    Wide,
78    /// `u8`, whose element is a `char8_t`, which is an `unsigned char`.
79    Utf8,
80    /// `u`, whose element is a `char16_t`, which is a `uint_least16_t`.
81    Utf16,
82    /// `U`, whose element is a `char32_t`, which is a `uint_least32_t`.
83    Utf32,
84}
85
86impl Encoding {
87    /// The width of one element in bits.
88    #[must_use]
89    pub fn element_width(self, target: &TargetInfo) -> u32 {
90        match self {
91            Encoding::Plain | Encoding::Utf8 => 8,
92            Encoding::Wide => target.wchar_width,
93            Encoding::Utf16 => 16,
94            Encoding::Utf32 => 32,
95        }
96    }
97
98    /// Whether the element type is signed, which only `char` and `wchar_t` can be.
99    #[must_use]
100    pub fn is_signed(self, target: &TargetInfo) -> bool {
101        match self {
102            Encoding::Plain => target.char_is_signed,
103            Encoding::Wide => target.wchar_is_signed,
104            Encoding::Utf8 | Encoding::Utf16 | Encoding::Utf32 => false,
105        }
106    }
107
108    /// The prefix this encoding is written with, which is empty for a plain literal.
109    #[must_use]
110    pub const fn prefix(self) -> &'static str {
111        match self {
112            Encoding::Plain => "",
113            Encoding::Wide => "L",
114            Encoding::Utf8 => "u8",
115            Encoding::Utf16 => "u",
116            Encoding::Utf32 => "U",
117        }
118    }
119
120    /// The prefix a spelling was written with, for a caller that needs the encoding of a
121    /// literal it could not convert.
122    #[must_use]
123    pub fn read_prefix(text: &str) -> Encoding {
124        Encoding::read(text.as_bytes()).0
125    }
126
127    /// The prefix at the front of a spelling, and how many bytes it took.
128    fn read(bytes: &[u8]) -> (Encoding, usize) {
129        match bytes {
130            [b'u', b'8', ..] => (Encoding::Utf8, 2),
131            [b'u', ..] => (Encoding::Utf16, 1),
132            [b'U', ..] => (Encoding::Utf32, 1),
133            [b'L', ..] => (Encoding::Wide, 1),
134            _ => (Encoding::Plain, 0),
135        }
136    }
137
138    /// The first dialect that has this prefix, which is not the same for a character constant
139    /// as for a string literal.
140    fn since(self, character: bool) -> Std {
141        match self {
142            Encoding::Plain | Encoding::Wide => Std::C89,
143            Encoding::Utf8 if character => Std::C23,
144            Encoding::Utf8 | Encoding::Utf16 | Encoding::Utf32 => Std::C11,
145        }
146    }
147}
148
149/// A converted character constant.
150#[derive(Debug, Clone, Copy, PartialEq, Eq)]
151pub struct CharConstant {
152    /// The value, already converted to the constant's type, which is why it can be negative:
153    /// `'\xff'` is minus one where plain `char` is signed.
154    pub value: i64,
155    /// The prefix it was written with.
156    pub encoding: Encoding,
157    /// What is worth saying about it, for the caller that holds the span.
158    pub remarks: Remarks,
159}
160
161impl CharConstant {
162    /// The spelling this constant would be written with, quotes and prefix included.
163    ///
164    /// It is a spelling and not the spelling: the one the author wrote is gone by the time
165    /// there is a value here. What it guarantees is that reading it back gives this constant,
166    /// which is what a printer downstream needs and what a message quoting a constant wants.
167    ///
168    /// A constant holding more than one character is written as the bytes it was shifted
169    /// together from, most significant first, which is how it is read back.
170    #[must_use]
171    pub fn spell(self) -> String {
172        let mut out = String::from(self.encoding.prefix());
173        out.push('\'');
174        match self.encoding {
175            Encoding::Plain | Encoding::Utf8 if !(-128..=255).contains(&self.value) => {
176                let bits = self.value as u32;
177                let mut writing = false;
178                for shift in [24, 16, 8, 0] {
179                    let byte = (bits >> shift) as u8;
180                    writing |= byte != 0;
181                    if writing {
182                        out.push_str(&format!("\\x{byte:02x}"));
183                    }
184                }
185            }
186            Encoding::Plain | Encoding::Utf8 => {
187                let byte = self.value as u8;
188                escape(u32::from(byte), '\'', &mut out);
189            }
190            _ => escape(self.value as u32, '\'', &mut out),
191        }
192        out.push('\'');
193        out
194    }
195}
196
197/// A converted string literal.
198#[derive(Debug, Clone, PartialEq, Eq)]
199pub struct StringLiteral {
200    /// One value per element of the array, without the terminating zero, since the zero belongs
201    /// to the type and not to the spelling. An element is a byte in a plain or `u8` literal, a
202    /// UTF-16 code unit in a `u` one, and a code point in a `U` one.
203    pub elements: Vec<u32>,
204    /// The prefix it was written with.
205    pub encoding: Encoding,
206    /// What is worth saying about it, for the caller that holds the span.
207    pub remarks: Remarks,
208}
209
210impl StringLiteral {
211    /// The bytes this literal becomes in the object, terminator included, in the target's byte
212    /// order.
213    #[must_use]
214    pub fn bytes(&self, target: &TargetInfo) -> Vec<u8> {
215        let width = self.encoding.element_width(target) / 8;
216        let mut bytes = Vec::with_capacity((self.elements.len() + 1) * width as usize);
217        for element in self.elements.iter().copied().chain([0]) {
218            let taken = &element.to_le_bytes()[..width as usize];
219            if target.little_endian {
220                bytes.extend_from_slice(taken);
221            } else {
222                bytes.extend(taken.iter().rev());
223            }
224        }
225        bytes
226    }
227
228    /// The spelling this literal would be written with, quotes and prefix included.
229    ///
230    /// A byte escape takes three octal digits rather than two hexadecimal ones, because an
231    /// octal escape stops after three digits and a hexadecimal one runs on until the digits do.
232    /// Where the elements are too wide for octal there is no such spelling, so the literal is
233    /// closed and reopened instead, which phase 7 joins back into the one literal it came from.
234    #[must_use]
235    pub fn spell(&self) -> String {
236        let prefix = self.encoding.prefix();
237        let wide = !matches!(self.encoding, Encoding::Plain | Encoding::Utf8);
238        let mut out = String::from(prefix);
239        out.push('"');
240        let mut ran_on = false;
241        for &element in &self.elements {
242            match printable(element) {
243                Some(ch) => {
244                    if ran_on && ch.is_ascii_hexdigit() {
245                        out.push('"');
246                        out.push(' ');
247                        out.push_str(prefix);
248                        out.push('"');
249                    }
250                    escape(element, '"', &mut out);
251                    ran_on = false;
252                }
253                None if wide => {
254                    out.push_str(&format!("\\x{element:x}"));
255                    ran_on = true;
256                }
257                None => {
258                    out.push_str(&format!("\\{element:03o}"));
259                    ran_on = false;
260                }
261            }
262        }
263        out.push('"');
264        out
265    }
266}
267
268/// The character an element is, where it is one that can be written as itself.
269///
270/// Everything outside printable ASCII is escaped, whatever the element means, because the
271/// output has no encoding of its own to be right about and an escape is right in all of them.
272fn printable(element: u32) -> Option<char> {
273    match element {
274        0x20..=0x7e => char::from_u32(element),
275        _ => None,
276    }
277}
278
279/// Writes one element of a literal, escaped if it has to be.
280fn escape(element: u32, quote: char, out: &mut String) {
281    match printable(element) {
282        Some(ch) if ch == quote || ch == '\\' => {
283            out.push('\\');
284            out.push(ch);
285        }
286        // Two question marks in a row are a trigraph where a compiler is told to look for one,
287        // so the second is written as an escape and the first never needs to be.
288        Some('?') if out.ends_with('?') => out.push_str("\\?"),
289        Some(ch) => out.push(ch),
290        None => out.push_str(&format!("\\x{element:x}")),
291    }
292}
293
294/// Why a spelling is not a literal.
295#[derive(Debug, Clone, Copy, PartialEq, Eq)]
296pub enum LiteralError {
297    /// The spelling is not a character constant or a string literal at all, which the caller
298    /// cannot reach from a token the scanner produced.
299    NotALiteral,
300    /// `''`, which has no character in it to have a value.
301    Empty,
302    /// More characters than the type has room for, in the one encoding where that is an error
303    /// rather than a warning.
304    TooLong,
305    /// `\x` with no hexadecimal digit after it.
306    NoHexDigits,
307    /// A universal character name that stops before its digits are done.
308    IncompleteUcn,
309    /// A universal character name that may not name what it names: a character in the basic
310    /// character set, a surrogate, or a code point past the end of Unicode.
311    InvalidUcn,
312    /// `\N{NAME}`, which gcc 13.3 has in C++23 and in no C dialect.
313    NamedUcn,
314    /// A byte in the source that is not part of a character, in a literal whose encoding has to
315    /// know what the characters are.
316    InvalidUtf8,
317    /// An encoding prefix the dialect does not have.
318    PrefixNotInDialect,
319    /// A run of adjacent string literals written with two different prefixes, which neither
320    /// compiler has an answer for.
321    MixedEncodings,
322}
323
324impl LiteralError {
325    /// What to print, in GCC's words where GCC has any.
326    #[must_use]
327    pub const fn message(self) -> &'static str {
328        match self {
329            LiteralError::NotALiteral => "not a character constant or a string literal",
330            LiteralError::Empty => "empty character constant",
331            LiteralError::TooLong => "character constant too long for its type",
332            LiteralError::NoHexDigits => "\\x used with no following hex digits",
333            LiteralError::IncompleteUcn => "incomplete universal character name",
334            LiteralError::InvalidUcn => "not a valid universal character",
335            LiteralError::NamedUcn => "named universal character escapes are not supported yet",
336            LiteralError::InvalidUtf8 => "failure to convert the source to the execution charset",
337            LiteralError::PrefixNotInDialect => {
338                "this encoding prefix is not available in this dialect"
339            }
340            LiteralError::MixedEncodings => {
341                "unsupported non-standard concatenation of string literals"
342            }
343        }
344    }
345}
346
347/// Converts the spelling of a character constant into a value.
348///
349/// # Errors
350///
351/// [`LiteralError`], for a spelling that is not a character constant or that holds an escape
352/// that is not one.
353pub fn character(text: &str, std: Std, target: &TargetInfo) -> Result<CharConstant, LiteralError> {
354    let (encoding, body) = open(text, b'\'', std, true)?;
355    let width = encoding.element_width(target);
356    let mut reader = Reader { bytes: body, index: 0, std, remarks: Remarks::NONE };
357
358    // The elements are shifted together into one number, which is what both compilers do with a
359    // constant holding more than one, and the ones that fall off the top are the ones the
360    // warning is about.
361    let mut value: u64 = 0;
362    let mut count = 0u32;
363    while let Some(piece) = reader.next(width)? {
364        for element in piece.elements(width) {
365            value = (value << width) | u64::from(element);
366            count += 1;
367        }
368    }
369    let mut remarks = reader.remarks;
370
371    // A plain constant is an `int` however many characters it holds, and every other kind is
372    // exactly one element wide, so that is how many characters fit.
373    let type_width = if encoding == Encoding::Plain { 32 } else { width };
374    let capacity = type_width / width;
375    match count {
376        0 => return Err(LiteralError::Empty),
377        1 => {}
378        _ if encoding == Encoding::Utf8 => return Err(LiteralError::TooLong),
379        _ if count > capacity => remarks = remarks.with(Remarks::TOO_LONG),
380        _ => remarks = remarks.with(Remarks::MULTICHARACTER),
381    }
382
383    // One character is converted to the constant's type from the element's, which is where the
384    // sign of a plain `char` gets in. More than one is already a number of the constant's type
385    // and nothing sign extends it from any narrower width.
386    let (bits, signed) = if count == 1 {
387        (width, encoding.is_signed(target))
388    } else {
389        (type_width, encoding == Encoding::Plain || encoding.is_signed(target))
390    };
391    Ok(CharConstant { value: narrow(value, bits, signed), encoding, remarks })
392}
393
394/// Converts the spelling of a string literal into its elements.
395///
396/// # Errors
397///
398/// [`LiteralError`], for a spelling that is not a string literal or that holds an escape that
399/// is not one.
400pub fn string(text: &str, std: Std, target: &TargetInfo) -> Result<StringLiteral, LiteralError> {
401    strings(std::slice::from_ref(&text), std, target)
402}
403
404/// Converts a run of adjacent string literals into the one literal they are.
405///
406/// The encoding of the result is the prefixed one when any of them is prefixed, so `L"a" "b"`
407/// and `"a" L"b"` are both wide, and two different prefixes in one run is an error rather than
408/// a choice. That is measured on gcc 13.3, which puts it exactly that way: "unsupported
409/// non-standard concatenation of string literals".
410///
411/// The bodies are read in the encoding of the whole run rather than each in its own, which
412/// matters for a character rather than an escape: the accented letter in `L"a" "e-acute"` is
413/// one wide element and not the two bytes it would have been on its own.
414///
415/// # Errors
416///
417/// [`LiteralError`], for a spelling that is not a string literal, a run mixing two prefixes, or
418/// an escape that is not one.
419pub fn strings(
420    texts: &[&str],
421    std: Std,
422    target: &TargetInfo,
423) -> Result<StringLiteral, LiteralError> {
424    let mut bodies = Vec::with_capacity(texts.len());
425    let mut encoding = Encoding::Plain;
426    for text in texts {
427        let (found, body) = open(text, b'"', std, false)?;
428        if found != Encoding::Plain {
429            if encoding != Encoding::Plain && encoding != found {
430                return Err(LiteralError::MixedEncodings);
431            }
432            encoding = found;
433        }
434        bodies.push(body);
435    }
436
437    let width = encoding.element_width(target);
438    let mut elements = Vec::new();
439    let mut remarks = Remarks::NONE;
440    for body in bodies {
441        let mut reader = Reader { bytes: body, index: 0, std, remarks: Remarks::NONE };
442        while let Some(piece) = reader.next(width)? {
443            elements.extend(piece.elements(width));
444        }
445        remarks = remarks.with(reader.remarks);
446    }
447    Ok(StringLiteral { elements, encoding, remarks })
448}
449
450/// Reads the prefix and the quotes, and hands back the encoding and what is between them.
451fn open(
452    text: &str,
453    quote: u8,
454    std: Std,
455    character: bool,
456) -> Result<(Encoding, &[u8]), LiteralError> {
457    let bytes = text.as_bytes();
458    let (encoding, prefix) = Encoding::read(bytes);
459    if std < encoding.since(character) {
460        return Err(LiteralError::PrefixNotInDialect);
461    }
462    let rest = &bytes[prefix..];
463    match rest {
464        [first, .., last] if *first == quote && *last == quote => {
465            Ok((encoding, &rest[1..rest.len() - 1]))
466        }
467        _ => Err(LiteralError::NotALiteral),
468    }
469}
470
471/// Cuts a value down to `bits` and reads it back as signed or unsigned.
472fn narrow(value: u64, bits: u32, signed: bool) -> i64 {
473    let masked = if bits >= 64 { value } else { value & ((1u64 << bits) - 1) };
474    if signed && bits < 64 && masked >> (bits - 1) & 1 == 1 {
475        // The bits above the width are the sign, which is what makes `'\xff'` minus one.
476        (masked | !((1u64 << bits) - 1)) as i64
477    } else {
478        masked as i64
479    }
480}
481
482/// One piece of a literal, before it is turned into elements.
483#[derive(Debug, Clone, Copy, PartialEq, Eq)]
484enum Piece {
485    /// A character, which the encoding has to encode: a character from the source or one named
486    /// by a universal character name.
487    Char(u32),
488    /// A value written as a value, with `\x` or with octal digits, which is one element as
489    /// written and is not encoded.
490    Value(u32),
491}
492
493impl Piece {
494    /// The elements this piece becomes in an encoding of the given element width.
495    fn elements(self, width: u32) -> Vec<u32> {
496        let code = match self {
497            Piece::Value(value) => return vec![value],
498            Piece::Char(code) => code,
499        };
500        match width {
501            8 => {
502                let mut buffer = [0u8; 4];
503                let text = char::from_u32(code)
504                    .map(|character| character.encode_utf8(&mut buffer).len())
505                    .unwrap_or(0);
506                buffer[..text].iter().map(|&byte| u32::from(byte)).collect()
507            }
508            // UTF-16 is the one encoding where a character can take two elements, which is why
509            // a wide string is not the same length on Windows as it is anywhere else.
510            16 if code > 0xffff => {
511                let value = code - 0x1_0000;
512                vec![0xd800 + (value >> 10), 0xdc00 + (value & 0x3ff)]
513            }
514            _ => vec![code],
515        }
516    }
517}
518
519/// Reads the body of a literal one character at a time.
520struct Reader<'a> {
521    /// What is between the quotes.
522    bytes: &'a [u8],
523    /// How far in the reader is.
524    index: usize,
525    /// The dialect, which decides what is worth a remark.
526    std: Std,
527    /// What the literal has earned so far.
528    remarks: Remarks,
529}
530
531impl Reader<'_> {
532    /// The next piece, or [`None`] at the end of the literal.
533    fn next(&mut self, width: u32) -> Result<Option<Piece>, LiteralError> {
534        let Some(&byte) = self.bytes.get(self.index) else {
535            return Ok(None);
536        };
537        self.index += 1;
538        if byte == b'\\' {
539            return self.escape(width).map(Some);
540        }
541        if byte < 0x80 {
542            return Ok(Some(Piece::Char(u32::from(byte))));
543        }
544        // A byte above ASCII begins a character in the source, which is UTF-8. A narrow literal
545        // is UTF-8 too, so its bytes go through untouched and nothing has to be able to decode
546        // them; a wide one has to know which character this is before it can encode it again.
547        if width == 8 {
548            return Ok(Some(Piece::Value(u32::from(byte))));
549        }
550        // Only this one character is decoded, and not the rest of the literal, because what
551        // comes after it may be an escape holding a byte that no character begins with.
552        let length = utf8_length(byte).ok_or(LiteralError::InvalidUtf8)?;
553        let end = self.index - 1 + length;
554        let text = self
555            .bytes
556            .get(self.index - 1..end)
557            .and_then(|slice| std::str::from_utf8(slice).ok())
558            .ok_or(LiteralError::InvalidUtf8)?;
559        let character = text.chars().next().ok_or(LiteralError::InvalidUtf8)?;
560        self.index = end;
561        Ok(Some(Piece::Char(character as u32)))
562    }
563
564    /// The piece an escape sequence is worth, with the backslash already read.
565    fn escape(&mut self, width: u32) -> Result<Piece, LiteralError> {
566        let Some(&byte) = self.bytes.get(self.index) else {
567            // The scanner reports the missing quote, and there is nothing here to convert.
568            return Err(LiteralError::NotALiteral);
569        };
570        self.index += 1;
571        let simple = match byte {
572            b'n' => Some(0x0a),
573            b't' => Some(0x09),
574            b'r' => Some(0x0d),
575            b'a' => Some(0x07),
576            b'b' => Some(0x08),
577            b'f' => Some(0x0c),
578            b'v' => Some(0x0b),
579            b'\\' | b'\'' | b'"' | b'?' => Some(u32::from(byte)),
580            _ => None,
581        };
582        if let Some(value) = simple {
583            return Ok(Piece::Value(value));
584        }
585        match byte {
586            // The escape character, which both compilers have and no standard does.
587            b'e' | b'E' => {
588                self.remarks = self.remarks.with(Remarks::NON_ISO_ESCAPE);
589                Ok(Piece::Value(0x1b))
590            }
591            b'0'..=b'7' => Ok(Piece::Value(self.octal(byte, width))),
592            b'x' => self.hex(width).map(Piece::Value),
593            b'u' | b'U' => self.ucn(byte).map(Piece::Char),
594            b'N' => Err(LiteralError::NamedUcn),
595            // An escape that means nothing is the character itself, which both compilers do
596            // after a warning rather than refusing the program.
597            _ => {
598                self.remarks = self.remarks.with(Remarks::UNKNOWN_ESCAPE);
599                Ok(Piece::Value(u32::from(byte)))
600            }
601        }
602    }
603
604    /// An octal escape, which is at most three digits however many follow, so that `"\1234"` is
605    /// two characters.
606    fn octal(&mut self, first: u8, width: u32) -> u32 {
607        let mut value = u32::from(first - b'0');
608        for _ in 0..2 {
609            match self.bytes.get(self.index) {
610                Some(&byte @ b'0'..=b'7') => {
611                    value = value * 8 + u32::from(byte - b'0');
612                    self.index += 1;
613                }
614                _ => break,
615            }
616        }
617        self.fit(value, width, Remarks::OCTAL_ESCAPE_OUT_OF_RANGE)
618    }
619
620    /// A hexadecimal escape, which runs as far as there are hexadecimal digits.
621    fn hex(&mut self, width: u32) -> Result<u32, LiteralError> {
622        let mut value: u64 = 0;
623        let mut digits = 0;
624        while let Some(digit) = self.bytes.get(self.index).and_then(|&byte| hex_digit(byte)) {
625            // A value far past the width is truncated anyway, so the accumulator stops growing
626            // rather than overflowing, and the remark still gets made.
627            value = value.saturating_mul(16).saturating_add(u64::from(digit));
628            digits += 1;
629            self.index += 1;
630        }
631        if digits == 0 {
632            return Err(LiteralError::NoHexDigits);
633        }
634        Ok(self.fit(
635            u32::try_from(value).unwrap_or(u32::MAX),
636            width,
637            Remarks::HEX_ESCAPE_OUT_OF_RANGE,
638        ))
639    }
640
641    /// A universal character name, with its `u` or `U` already read.
642    fn ucn(&mut self, marker: u8) -> Result<u32, LiteralError> {
643        if self.bytes.get(self.index) == Some(&b'{') {
644            return Err(LiteralError::NamedUcn);
645        }
646        let digits = if marker == b'u' { 4 } else { 8 };
647        let mut value: u32 = 0;
648        for _ in 0..digits {
649            let Some(digit) = self.bytes.get(self.index).and_then(|&byte| hex_digit(byte)) else {
650                return Err(LiteralError::IncompleteUcn);
651            };
652            value = value * 16 + digit;
653            self.index += 1;
654        }
655        // The basic character set is off limits, and so is everything below ` ` except the
656        // three characters the standard lets through. gcc still says so in C23, where the
657        // wording was relaxed, so this follows the compiler rather than the paper.
658        let allowed_low = matches!(value, 0x24 | 0x40 | 0x60);
659        if (value < 0xa0 && !allowed_low) || (0xd800..=0xdfff).contains(&value) || value > 0x10ffff
660        {
661            return Err(LiteralError::InvalidUcn);
662        }
663        if self.std < Std::C99 {
664            self.remarks = self.remarks.with(Remarks::UCN);
665        }
666        Ok(value)
667    }
668
669    /// Cuts an escape down to the element it is written in, and says so when that loses
670    /// something. Which remark that is depends on how the escape was written, because GCC words
671    /// the hexadecimal case and the octal one differently.
672    fn fit(&mut self, value: u32, width: u32, out_of_range: Remarks) -> u32 {
673        if width >= 32 {
674            return value;
675        }
676        let mask = (1u32 << width) - 1;
677        if value & !mask != 0 {
678            self.remarks = self.remarks.with(out_of_range);
679        }
680        value & mask
681    }
682}
683
684/// The value of a hexadecimal digit, and [`None`] when the byte is not one.
685fn hex_digit(byte: u8) -> Option<u32> {
686    char::from(byte).to_digit(16)
687}
688
689/// How many bytes the character starting with this one takes, and [`None`] when no character
690/// starts with it.
691fn utf8_length(byte: u8) -> Option<usize> {
692    match byte {
693        0x00..=0x7f => Some(1),
694        0xc2..=0xdf => Some(2),
695        0xe0..=0xef => Some(3),
696        0xf0..=0xf4 => Some(4),
697        _ => None,
698    }
699}
700
701#[cfg(test)]
702mod tests {
703    use rucc_target::Triple;
704
705    use super::*;
706
707    fn linux() -> TargetInfo {
708        TargetInfo::new("x86_64-unknown-linux-gnu".parse::<Triple>().expect("a known triple"))
709    }
710
711    fn windows() -> TargetInfo {
712        TargetInfo::new("x86_64-pc-windows-msvc".parse::<Triple>().expect("a known triple"))
713    }
714
715    fn arm() -> TargetInfo {
716        TargetInfo::new("aarch64-unknown-linux-gnu".parse::<Triple>().expect("a known triple"))
717    }
718
719    /// The value of a character constant on x86-64 Linux, in C23.
720    fn ch(text: &str) -> i64 {
721        character(text, Std::C23, &linux()).expect("a character constant").value
722    }
723
724    /// The remarks a character constant earns on x86-64 Linux, in C23.
725    fn ch_remarks(text: &str) -> Remarks {
726        character(text, Std::C23, &linux()).expect("a character constant").remarks
727    }
728
729    /// What a character constant goes wrong with.
730    fn ch_error(text: &str) -> LiteralError {
731        character(text, Std::C23, &linux()).expect_err("not a character constant")
732    }
733
734    /// The elements of a string literal on x86-64 Linux, in C23.
735    fn str_elements(text: &str) -> Vec<u32> {
736        string(text, Std::C23, &linux()).expect("a string literal").elements
737    }
738
739    /// The bytes a string literal becomes on x86-64 Linux, terminator included.
740    fn str_bytes(text: &str) -> Vec<u8> {
741        string(text, Std::C23, &linux()).expect("a string literal").bytes(&linux())
742    }
743
744    #[test]
745    fn the_ordinary_cases_are_the_characters_they_look_like() {
746        assert_eq!(ch("'a'"), 0x61);
747        assert_eq!(ch(r"'\n'"), 0x0a);
748        assert_eq!(ch(r"'\0'"), 0);
749        assert_eq!(ch(r"'\\'"), 0x5c);
750        assert_eq!(ch(r"'\''"), 0x27);
751        assert_eq!(ch(r#"'\"'"#), 0x22);
752        assert_eq!(ch(r"'\?'"), 0x3f);
753        assert_eq!(str_elements(r#""hi""#), vec![0x68, 0x69]);
754    }
755
756    /// A single character in a plain constant goes through plain `char` on the way to `int`,
757    /// which is the whole reason `'\xff'` is a negative number on one target and a positive
758    /// one on another.
759    #[test]
760    fn a_high_character_takes_the_sign_of_plain_char() {
761        assert_eq!(ch(r"'\xff'"), -1);
762        assert_eq!(ch(r"'\377'"), -1);
763        assert_eq!(character(r"'\xff'", Std::C23, &arm()).expect("a constant").value, 255);
764        // Not a `char`, so nothing sign extends it.
765        assert_eq!(ch(r"u8'\xff'"), 255);
766    }
767
768    /// Measured on GCC 13.3, x86-64 Linux. A value escape is truncated to its element and the
769    /// warning is about the truncation, not about the type. The two spellings get two remarks
770    /// because GCC gives them two wordings.
771    #[test]
772    fn an_escape_too_big_for_its_element_is_truncated_and_says_so() {
773        let out = character(r"'\x1ff'", Std::C23, &linux()).expect("a constant");
774        assert_eq!(out.value, -1);
775        assert!(out.remarks.has(Remarks::HEX_ESCAPE_OUT_OF_RANGE));
776        let out = character(r"'\400'", Std::C23, &linux()).expect("a constant");
777        assert_eq!(out.value, 0);
778        assert!(out.remarks.has(Remarks::OCTAL_ESCAPE_OUT_OF_RANGE));
779        assert!(!out.remarks.has(Remarks::HEX_ESCAPE_OUT_OF_RANGE));
780        // Wide enough to hold it, so there is nothing to say.
781        assert!(!ch_remarks(r"L'\x1ff'").has(Remarks::HEX_ESCAPE_OUT_OF_RANGE));
782        assert_eq!(ch(r"L'\x1ff'"), 0x1ff);
783    }
784
785    /// Measured on GCC 13.3: a plain literal in a run takes the prefix of its neighbour, two
786    /// different prefixes are an error, and the bodies are read in the encoding of the run, so
787    /// a character in the plain part is one wide element rather than its UTF-8 bytes.
788    #[test]
789    fn adjacent_literals_agree_on_one_encoding_or_none_at_all() {
790        let target = linux();
791        let wide = strings(&[r#"L"a""#, r#""b""#], Std::C23, &target).expect("a string");
792        assert_eq!(wide.encoding, Encoding::Wide);
793        assert_eq!(wide.elements, vec![0x61, 0x62]);
794        assert_eq!(wide.bytes(&target).len(), 12);
795        let other_way = strings(&[r#""a""#, r#"L"b""#], Std::C23, &target).expect("a string");
796        assert_eq!(other_way.encoding, Encoding::Wide);
797        assert_eq!(other_way.bytes(&target).len(), 12);
798
799        let u8_run = strings(&[r#"u8"a""#, r#""b""#], Std::C23, &target).expect("a string");
800        assert_eq!(u8_run.encoding, Encoding::Utf8);
801        assert_eq!(u8_run.bytes(&target).len(), 3);
802
803        // The plain part is read as wide, so the accented letter is one element and not two.
804        let mixed = strings(&[r#"L"a""#, r#""é""#], Std::C23, &target).expect("a string");
805        assert_eq!(mixed.elements, vec![0x61, 0xe9]);
806
807        for run in [[r#"u8"a""#, r#"u"b""#], [r#"u8"a""#, r#"L"b""#], [r#"u"a""#, r#"L"b""#]] {
808            assert_eq!(
809                strings(&run, Std::C23, &target).expect_err("two prefixes in one run"),
810                LiteralError::MixedEncodings
811            );
812        }
813
814        // A run of one is the same thing as the literal on its own.
815        assert_eq!(
816            strings(&[r#""hi""#], Std::C23, &target).expect("a string").elements,
817            vec![0x68, 0x69]
818        );
819    }
820
821    /// Also measured on GCC 13.3. The characters are shifted together, the ones past the width
822    /// of the type fall off the front, and past that point GCC says "too long" instead of
823    /// "multi-character" rather than as well as it.
824    #[test]
825    fn more_than_one_character_shifts_them_together() {
826        assert_eq!(ch("'ab'"), 0x6162);
827        assert_eq!(ch("'abc'"), 0x616263);
828        assert_eq!(ch("'abcd'"), 0x61626364);
829        assert_eq!(ch("'abcde'"), 0x62636465);
830        assert_eq!(ch(r"'\xff\xfe'"), 0xfffe);
831        assert_eq!(ch(r"'\xff\xff\xff\xff'"), -1);
832        assert_eq!(ch(r"'\x80\x00'"), 0x8000);
833
834        assert!(ch_remarks("'ab'").has(Remarks::MULTICHARACTER));
835        assert!(ch_remarks("'abcd'").has(Remarks::MULTICHARACTER));
836        assert!(ch_remarks("'abcde'").has(Remarks::TOO_LONG));
837        assert!(!ch_remarks("'abcde'").has(Remarks::MULTICHARACTER));
838        assert!(!ch_remarks("'a'").has(Remarks::MULTICHARACTER));
839    }
840
841    /// A wide constant has room for exactly one character, so two is already too many and the
842    /// last one is what survives. `u8` is the one encoding where GCC makes this an error.
843    #[test]
844    fn a_prefixed_constant_holds_one_character_and_keeps_the_last() {
845        for text in [r"L'ab'", r"u'ab'", r"U'ab'"] {
846            let out = character(text, Std::C23, &linux()).expect("a constant");
847            assert_eq!(out.value, 0x62, "{text}");
848            assert!(out.remarks.has(Remarks::TOO_LONG), "{text}");
849        }
850        assert_eq!(ch_error("u8'ab'"), LiteralError::TooLong);
851        assert_eq!(ch_error("u8'é'"), LiteralError::TooLong);
852    }
853
854    #[test]
855    fn the_empty_constant_has_no_value_to_have() {
856        assert_eq!(ch_error("''"), LiteralError::Empty);
857        assert_eq!(ch_error("L''"), LiteralError::Empty);
858        // The empty string is fine, and is one element long once the terminator is there.
859        assert_eq!(str_elements(r#""""#), Vec::<u32>::new());
860        assert_eq!(str_bytes(r#""""#), vec![0]);
861    }
862
863    /// A character from the source is encoded in the literal's encoding, so the same `é` is
864    /// two bytes in one constant and one code point in another.
865    #[test]
866    fn a_source_character_is_encoded_and_an_escape_is_not() {
867        assert_eq!(ch("'é'"), 0xc3a9);
868        assert_eq!(ch("L'é'"), 0xe9);
869        assert_eq!(ch("u'€'"), 0x20ac);
870        assert_eq!(ch(r"U'\U0001F600'"), 0x1f600);
871        // The same character in a plain constant is its UTF-8 bytes, which makes it a
872        // multi-character constant and a negative number.
873        assert_eq!(ch(r"'\U0001F600'"), i64::from(0xf09f_9880u32 as i32));
874        assert!(ch_remarks(r"'\U0001F600'").has(Remarks::MULTICHARACTER));
875    }
876
877    /// Both compilers have `\e` and neither standard does, and an escape that means nothing is
878    /// the letter itself after a warning rather than an error.
879    #[test]
880    fn the_escapes_outside_the_standard_still_have_values() {
881        assert_eq!(ch(r"'\e'"), 0x1b);
882        assert!(ch_remarks(r"'\e'").has(Remarks::NON_ISO_ESCAPE));
883        assert_eq!(ch(r"'\q'"), 0x71);
884        assert!(ch_remarks(r"'\q'").has(Remarks::UNKNOWN_ESCAPE));
885        assert_eq!(ch_error(r"'\x'"), LiteralError::NoHexDigits);
886        assert_eq!(ch_error(r"'\N{LATIN SMALL LETTER A}'"), LiteralError::NamedUcn);
887    }
888
889    /// A universal character name may not name a character in the basic character set, which
890    /// GCC still enforces in C23 where the wording was relaxed, and the three characters below
891    /// a space that are allowed anyway are allowed here too.
892    #[test]
893    fn a_universal_character_name_may_not_name_just_anything() {
894        assert_eq!(ch("'\\u0024'"), 0x24);
895        assert_eq!(ch("'\\u00e9'"), 0xc3a9);
896        assert_eq!(ch_error("'\\u0041'"), LiteralError::InvalidUcn);
897        assert_eq!(ch_error(r"'\ud800'"), LiteralError::InvalidUcn);
898        assert_eq!(ch_error(r"'\u00'"), LiteralError::IncompleteUcn);
899        // GCC warns here and encodes the value anyway. clang refuses it and so does this.
900        assert_eq!(ch_error(r"'\U00110000'"), LiteralError::InvalidUcn);
901    }
902
903    #[test]
904    fn a_universal_character_name_before_c99_is_worth_a_remark() {
905        let out = character("'\\u00e9'", Std::C89, &linux()).expect("a constant");
906        assert!(out.remarks.has(Remarks::UCN));
907        let out = character("'\\u00e9'", Std::C99, &linux()).expect("a constant");
908        assert!(!out.remarks.has(Remarks::UCN));
909    }
910
911    /// Octal runs to three digits and stops, and hexadecimal runs as far as the digits go, so
912    /// `"\1234"` is two characters and `"\x41z"` is two as well.
913    #[test]
914    fn an_octal_escape_ends_and_a_hex_escape_does_not() {
915        assert_eq!(str_elements(r#""\1234""#), vec![0x53, 0x34]);
916        assert_eq!(str_elements(r#""\x41z""#), vec![0x41, 0x7a]);
917        assert_eq!(str_elements(r#""\x41""#), vec![0x41]);
918    }
919
920    /// The sizes GCC reports for these, which is the elements plus the terminator times the
921    /// width of one.
922    #[test]
923    fn a_string_is_as_many_bytes_as_its_encoding_makes_it() {
924        assert_eq!(str_bytes(r#""abc""#).len(), 4);
925        assert_eq!(str_bytes(r#"L"abc""#).len(), 16);
926        assert_eq!(str_bytes(r#"u"abc""#).len(), 8);
927        assert_eq!(str_bytes(r#"U"abc""#).len(), 16);
928        assert_eq!(str_bytes(r#"u8"abc""#).len(), 4);
929        // A zero in the middle is an element like any other, and the terminator is still added.
930        assert_eq!(str_bytes(r#""a\0b""#), vec![0x61, 0x00, 0x62, 0x00]);
931        assert_eq!(str_bytes(r#""é""#), vec![0xc3, 0xa9, 0x00]);
932    }
933
934    /// The one encoding where a character can take two elements, which is why a wide string is
935    /// not the same length on Windows as it is anywhere else.
936    #[test]
937    fn utf16_splits_the_characters_that_do_not_fit_into_a_surrogate_pair() {
938        assert_eq!(
939            str_elements(r#"u8"é€😀""#),
940            vec![0xc3, 0xa9, 0xe2, 0x82, 0xac, 0xf0, 0x9f, 0x98, 0x80]
941        );
942        assert_eq!(str_elements(r#"u"€😀""#), vec![0x20ac, 0xd83d, 0xde00]);
943        assert_eq!(str_elements(r#"U"€😀""#), vec![0x20ac, 0x1f600]);
944    }
945
946    /// A wide literal is UTF-16 on Windows and UTF-32 everywhere else, so the same three
947    /// characters are four elements on one target and three on the other.
948    #[test]
949    fn a_wide_literal_is_whatever_the_target_makes_wchar_t() {
950        let text = r#"L"a😀""#;
951        let here = string(text, Std::C23, &linux()).expect("a string");
952        assert_eq!(here.elements, vec![0x61, 0x1f600]);
953        assert_eq!(here.bytes(&linux()).len(), 12);
954        let there = string(text, Std::C23, &windows()).expect("a string");
955        assert_eq!(there.elements, vec![0x61, 0xd83d, 0xde00]);
956        assert_eq!(there.bytes(&windows()).len(), 8);
957        // And a wide character constant takes the sign of `wchar_t`, which is not the same on
958        // every target either.
959        assert_eq!(character(r"L'\xffffffff'", Std::C23, &linux()).expect("a constant").value, -1);
960        assert_eq!(
961            character(r"L'\xffffffff'", Std::C23, &arm()).expect("a constant").value,
962            0xffff_ffff
963        );
964    }
965
966    /// Every target the compiler has is little-endian, so the other order is checked by
967    /// flipping the field rather than by naming a target, and this is the test that fails on
968    /// the day a big-endian one arrives with the layout still assuming otherwise.
969    #[test]
970    fn the_bytes_come_out_in_the_targets_order() {
971        let mut big = linux();
972        big.little_endian = false;
973        let literal = string(r#"u"ab""#, Std::C23, &big).expect("a string");
974        assert_eq!(literal.bytes(&big), vec![0x00, 0x61, 0x00, 0x62, 0x00, 0x00]);
975        assert_eq!(literal.bytes(&linux()), vec![0x61, 0x00, 0x62, 0x00, 0x00, 0x00]);
976    }
977
978    /// `L` is C89, `u` and `U` are C11, and `u8` is C11 on a string and C23 on a character
979    /// constant, which is the one place the two differ.
980    #[test]
981    fn a_prefix_is_only_available_in_the_dialect_that_has_it() {
982        assert!(character("L'a'", Std::C89, &linux()).is_ok());
983        assert_eq!(
984            character("u'a'", Std::C99, &linux()).expect_err("not in C99"),
985            LiteralError::PrefixNotInDialect
986        );
987        assert!(character("u'a'", Std::C11, &linux()).is_ok());
988        assert!(string(r#"u8"a""#, Std::C11, &linux()).is_ok());
989        assert_eq!(
990            character("u8'a'", Std::C11, &linux()).expect_err("not in C11"),
991            LiteralError::PrefixNotInDialect
992        );
993        assert!(character("u8'a'", Std::C23, &linux()).is_ok());
994    }
995
996    /// The widths and signs the elements have, which is what the parser will turn into the
997    /// type of the literal.
998    #[test]
999    fn an_element_is_as_wide_as_the_encoding_and_the_target_agree() {
1000        let target = linux();
1001        assert_eq!(Encoding::Plain.element_width(&target), 8);
1002        assert_eq!(Encoding::Utf8.element_width(&target), 8);
1003        assert_eq!(Encoding::Utf16.element_width(&target), 16);
1004        assert_eq!(Encoding::Utf32.element_width(&target), 32);
1005        assert_eq!(Encoding::Wide.element_width(&target), 32);
1006        assert_eq!(Encoding::Wide.element_width(&windows()), 16);
1007
1008        assert!(Encoding::Plain.is_signed(&target));
1009        assert!(!Encoding::Plain.is_signed(&arm()));
1010        assert!(Encoding::Wide.is_signed(&target));
1011        assert!(!Encoding::Wide.is_signed(&arm()));
1012        assert!(!Encoding::Utf8.is_signed(&target));
1013        assert!(!Encoding::Utf16.is_signed(&target));
1014        assert!(!Encoding::Utf32.is_signed(&target));
1015    }
1016
1017    #[test]
1018    fn a_spelling_that_is_not_a_literal_is_refused_rather_than_guessed_at() {
1019        assert_eq!(ch_error("a"), LiteralError::NotALiteral);
1020        assert_eq!(ch_error("'a"), LiteralError::NotALiteral);
1021        assert_eq!(
1022            string("'a'", Std::C23, &linux()).expect_err("not a string"),
1023            LiteralError::NotALiteral
1024        );
1025        assert_eq!(ch_error("'"), LiteralError::NotALiteral);
1026    }
1027
1028    #[test]
1029    fn every_error_has_something_to_print() {
1030        for error in [
1031            LiteralError::NotALiteral,
1032            LiteralError::Empty,
1033            LiteralError::TooLong,
1034            LiteralError::NoHexDigits,
1035            LiteralError::IncompleteUcn,
1036            LiteralError::InvalidUcn,
1037            LiteralError::NamedUcn,
1038            LiteralError::InvalidUtf8,
1039            LiteralError::PrefixNotInDialect,
1040            LiteralError::MixedEncodings,
1041        ] {
1042            assert!(!error.message().is_empty());
1043        }
1044    }
1045}