Skip to main content

rucc_types/
kind.rs

1//! What a C type is made of, before any of it has been interned.
2//!
3//! Design: `spec/07-types-and-semantics.md` section 7.1.
4//!
5//! Everything here is `Copy` and small, because [`TypeKind`] is the interning key and a key
6//! that owns a heap allocation cannot be hashed cheaply or compared cheaply. The two parts of
7//! a type that are genuinely variable length, a function's parameter list and a record's
8//! members, live in side tables and are referred to by index.
9
10use std::num::NonZeroU32;
11
12use rucc_base::Symbol;
13
14use crate::TypeId;
15
16/// The qualifiers a type can carry.
17///
18/// A bitmask in the interning key rather than a chain of wrapper nodes, so `const int` is one
19/// entry in the table beside `int` rather than a node pointing at it. That makes stripping
20/// qualifiers a field read instead of a walk, which matters because almost every semantic rule
21/// in C is stated on the unqualified type.
22///
23/// `_Atomic` is deliberately not here. C lets it be written in the same position as a
24/// qualifier, but `_Atomic(T)` is a different type from `T` with its own size and alignment,
25/// so it is a type constructor, [`TypeKind::Atomic`], and the parser is what maps the
26/// qualifier spelling onto it.
27#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Hash, PartialOrd, Ord)]
28pub struct Qualifiers(u8);
29
30impl Qualifiers {
31    /// No qualifiers.
32    pub const NONE: Qualifiers = Qualifiers(0);
33    /// `const`.
34    pub const CONST: Qualifiers = Qualifiers(1);
35    /// `volatile`.
36    pub const VOLATILE: Qualifiers = Qualifiers(2);
37    /// `restrict`.
38    pub const RESTRICT: Qualifiers = Qualifiers(4);
39
40    /// Whether every qualifier in `other` is present here.
41    #[inline]
42    #[must_use]
43    pub const fn has(self, other: Qualifiers) -> bool {
44        self.0 & other.0 == other.0
45    }
46
47    /// This set with `other` added.
48    #[inline]
49    #[must_use]
50    pub const fn with(self, other: Qualifiers) -> Qualifiers {
51        Qualifiers(self.0 | other.0)
52    }
53
54    /// This set with `other` removed.
55    #[inline]
56    #[must_use]
57    pub const fn without(self, other: Qualifiers) -> Qualifiers {
58        Qualifiers(self.0 & !other.0)
59    }
60
61    /// Whether there are no qualifiers at all.
62    #[inline]
63    #[must_use]
64    pub const fn is_none(self) -> bool {
65        self.0 == 0
66    }
67}
68
69/// The standard integer types, the character types kept apart from them, and `__int128`.
70///
71/// `Char` is its own kind rather than an alias for one of the other two. The standard makes
72/// plain `char` a third type distinct from both `signed char` and `unsigned char` even though
73/// it has the same range as one of them, and a compiler that folds it into whichever one the
74/// target picked gets `char *` and `signed char *` wrongly deemed compatible.
75///
76/// `__int128` is here rather than modelled as a `_BitInt(128)`, because the two are different
77/// types with different layouts: `__int128` is sixteen bytes aligned to sixteen on every
78/// target we have, and `_BitInt(128)` is aligned to its granule, which is eight on x86-64. It
79/// is available everywhere for us, since all three architectures are 64-bit, and GCC has it
80/// on every 64-bit target. It is deliberately not an extended integer type in the sense the
81/// standard means, which is what keeps `intmax_t` sixty four bits wide the way GCC has it.
82#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
83pub enum IntKind {
84    /// `char`, whose signedness is a target property.
85    Char,
86    /// `signed char`.
87    SChar,
88    /// `unsigned char`.
89    UChar,
90    /// `short`.
91    Short,
92    /// `unsigned short`.
93    UShort,
94    /// `int`.
95    Int,
96    /// `unsigned int`.
97    UInt,
98    /// `long`, the width that separates LP64 from Windows LLP64.
99    Long,
100    /// `unsigned long`.
101    ULong,
102    /// `long long`.
103    LongLong,
104    /// `unsigned long long`.
105    ULongLong,
106    /// `__int128`.
107    Int128,
108    /// `unsigned __int128`.
109    UInt128,
110}
111
112impl IntKind {
113    /// Every integer kind, in rank order, with `__int128` last.
114    ///
115    /// The order is what the internal index agrees with, and it is also the order the standard
116    /// walks when it picks the type of an integer constant, so a table walk over the candidate
117    /// list for a suffix is a walk over a slice of this. `__int128` is at the end because that
118    /// is where GCC reaches for it: after every standard type has been tried and none of them
119    /// was wide enough.
120    pub const ALL: [IntKind; 13] = [
121        IntKind::Char,
122        IntKind::SChar,
123        IntKind::UChar,
124        IntKind::Short,
125        IntKind::UShort,
126        IntKind::Int,
127        IntKind::UInt,
128        IntKind::Long,
129        IntKind::ULong,
130        IntKind::LongLong,
131        IntKind::ULongLong,
132        IntKind::Int128,
133        IntKind::UInt128,
134    ];
135
136    /// A dense index, so that one of these can select a slot in a fixed size array.
137    pub(crate) const fn index(self) -> usize {
138        match self {
139            IntKind::Char => 0,
140            IntKind::SChar => 1,
141            IntKind::UChar => 2,
142            IntKind::Short => 3,
143            IntKind::UShort => 4,
144            IntKind::Int => 5,
145            IntKind::UInt => 6,
146            IntKind::Long => 7,
147            IntKind::ULong => 8,
148            IntKind::LongLong => 9,
149            IntKind::ULongLong => 10,
150            IntKind::Int128 => 11,
151            IntKind::UInt128 => 12,
152        }
153    }
154
155    /// Whether this type is signed, given what the target says about plain `char`.
156    ///
157    /// The argument is there because `char` is the one integer type whose signedness is not
158    /// in the standard. It is signed on x86-64 and unsigned on AArch64 Linux, and a compiler
159    /// that assumes either one is the source of a whole genre of bug report.
160    #[must_use]
161    pub const fn is_signed(self, char_is_signed: bool) -> bool {
162        match self {
163            IntKind::Char => char_is_signed,
164            IntKind::SChar
165            | IntKind::Short
166            | IntKind::Int
167            | IntKind::Long
168            | IntKind::LongLong
169            | IntKind::Int128 => true,
170            IntKind::UChar
171            | IntKind::UShort
172            | IntKind::UInt
173            | IntKind::ULong
174            | IntKind::ULongLong
175            | IntKind::UInt128 => false,
176        }
177    }
178
179    /// The integer conversion rank, as an ordering rather than as a number from the standard.
180    ///
181    /// The standard gives no values, only a set of relations, and every one of them is a
182    /// comparison between two ranks. Signed and unsigned of the same width share a rank, which
183    /// is what makes the usual arithmetic conversions between them pick the unsigned type
184    /// rather than the wider one.
185    #[must_use]
186    pub const fn rank(self) -> u8 {
187        match self {
188            IntKind::Char | IntKind::SChar | IntKind::UChar => 1,
189            IntKind::Short | IntKind::UShort => 2,
190            IntKind::Int | IntKind::UInt => 3,
191            IntKind::Long | IntKind::ULong => 4,
192            IntKind::LongLong | IntKind::ULongLong => 5,
193            // Above `long long`, which is what makes `__int128 + unsigned long long` an
194            // `__int128` rather than an unsigned type. Both compilers agree.
195            IntKind::Int128 | IntKind::UInt128 => 6,
196        }
197    }
198
199    /// The same width with the other signedness.
200    ///
201    /// `char` maps to `unsigned char` and back to `signed char`, which is the mapping the
202    /// usual arithmetic conversions need and is not a round trip. That asymmetry is the type
203    /// system telling the truth: there is no way back to plain `char` from either of the
204    /// other two.
205    #[must_use]
206    pub const fn flip_sign(self) -> IntKind {
207        match self {
208            IntKind::Char | IntKind::SChar => IntKind::UChar,
209            IntKind::UChar => IntKind::SChar,
210            IntKind::Short => IntKind::UShort,
211            IntKind::UShort => IntKind::Short,
212            IntKind::Int => IntKind::UInt,
213            IntKind::UInt => IntKind::Int,
214            IntKind::Long => IntKind::ULong,
215            IntKind::ULong => IntKind::Long,
216            IntKind::LongLong => IntKind::ULongLong,
217            IntKind::ULongLong => IntKind::LongLong,
218            IntKind::Int128 => IntKind::UInt128,
219            IntKind::UInt128 => IntKind::Int128,
220        }
221    }
222
223    /// How the type is spelled in a diagnostic.
224    #[must_use]
225    pub const fn as_str(self) -> &'static str {
226        match self {
227            IntKind::Char => "char",
228            IntKind::SChar => "signed char",
229            IntKind::UChar => "unsigned char",
230            IntKind::Short => "short",
231            IntKind::UShort => "unsigned short",
232            IntKind::Int => "int",
233            IntKind::UInt => "unsigned int",
234            IntKind::Long => "long",
235            IntKind::ULong => "unsigned long",
236            IntKind::LongLong => "long long",
237            IntKind::ULongLong => "unsigned long long",
238            IntKind::Int128 => "__int128",
239            IntKind::UInt128 => "unsigned __int128",
240        }
241    }
242}
243
244/// The real floating types.
245///
246/// Nine of them, which is three standard ones and six from C23 Annex H. The interchange types
247/// `_Float16`, `_Float32`, `_Float64` and `_Float128` name an IEEE format outright, and the
248/// extended types `_Float32x` and `_Float64x` name whatever the target has that is wider than
249/// the interchange type they are named after, which makes `_Float64x` the x87 format on x86 and
250/// quad precision on AArch64. None of them is the standard type it shares a format with:
251/// `_Float64` and `double` are both binary64 and are two types, which `_Generic` can tell apart
252/// and which decides what `_Float64 + double` is.
253///
254/// `_Float128x` is a type no target gcc supports has, so it is not here. The three decimal
255/// floating types from C23 are, and they are real floating types like the others in every way
256/// except the one that matters most: a decimal and a binary type never meet in an operation, so
257/// the usual arithmetic conversions have no answer for the pair and the program is refused.
258#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
259pub enum FloatKind {
260    /// `_Float16`, always the binary16 format.
261    Float16,
262    /// `float`, always the binary32 format.
263    Float,
264    /// `_Float32`, always the binary32 format, and not the same type as `float`.
265    Float32,
266    /// `double`, always the binary64 format.
267    Double,
268    /// `_Float32x`, the format the target has that is wider than `_Float32`, which is binary64
269    /// everywhere this compiles for.
270    Float32x,
271    /// `_Float64`, always the binary64 format, and not the same type as `double`.
272    Float64,
273    /// `long double`, whose format is a target property and is not always distinct from
274    /// `double`. It is 80 bits of x87 on SysV x86-64, quad precision on AArch64 Linux, and
275    /// the same as `double` on Apple and Windows.
276    LongDouble,
277    /// `_Float64x`, the format the target has that is wider than `_Float64`. That is the x87
278    /// eighty bit format on x86-64 and quad precision on AArch64 and RISC-V, and unlike
279    /// `long double` it does not become a `double` on Apple or on Windows.
280    Float64x,
281    /// `_Float128`, always the binary128 format.
282    Float128,
283    /// `_Decimal32`, the decimal32 format in the binary integer encoding.
284    Decimal32,
285    /// `_Decimal64`, the decimal64 format in the binary integer encoding.
286    Decimal64,
287    /// `_Decimal128`, the decimal128 format in the binary integer encoding.
288    Decimal128,
289}
290
291impl FloatKind {
292    /// Every real floating type, in the order they are written above.
293    ///
294    /// Not in rank order, because there is no such order to put them in: which of `long double`
295    /// and `_Float64x` is the wider one is a question about the target, and on Apple the answer
296    /// is the second.
297    pub const ALL: [FloatKind; 12] = [
298        FloatKind::Float16,
299        FloatKind::Float,
300        FloatKind::Float32,
301        FloatKind::Double,
302        FloatKind::Float32x,
303        FloatKind::Float64,
304        FloatKind::LongDouble,
305        FloatKind::Float64x,
306        FloatKind::Float128,
307        FloatKind::Decimal32,
308        FloatKind::Decimal64,
309        FloatKind::Decimal128,
310    ];
311
312    /// A dense index, so that one of these can select a slot in a fixed size array.
313    pub(crate) const fn index(self) -> usize {
314        match self {
315            FloatKind::Float16 => 0,
316            FloatKind::Float => 1,
317            FloatKind::Float32 => 2,
318            FloatKind::Double => 3,
319            FloatKind::Float32x => 4,
320            FloatKind::Float64 => 5,
321            FloatKind::LongDouble => 6,
322            FloatKind::Float64x => 7,
323            FloatKind::Float128 => 8,
324            FloatKind::Decimal32 => 9,
325            FloatKind::Decimal64 => 10,
326            FloatKind::Decimal128 => 11,
327        }
328    }
329
330    /// Whether this is one of the three decimal types.
331    #[must_use]
332    pub const fn is_decimal(self) -> bool {
333        matches!(self, FloatKind::Decimal32 | FloatKind::Decimal64 | FloatKind::Decimal128)
334    }
335
336    /// What decides between two of these when they have the same format.
337    ///
338    /// Two real floating types can be the same format and still be two types, and then the
339    /// format cannot say which of them an operation on both of them produces. C23 answers with
340    /// the family first: an interchange type wins over the standard type it shares a format
341    /// with, and the standard type wins over an extended one, so `double + _Float64` is a
342    /// `_Float64` and `double + _Float32x` is a `double`. Inside a family it is the usual order,
343    /// which only ever comes up between `double` and `long double` on the targets where the
344    /// second one is the first one.
345    ///
346    /// Higher wins. This is not an ordering on the types on its own, because it says nothing
347    /// about the formats: `_Float32` sits above `long double` here and loses to it everywhere it
348    /// meets it.
349    #[must_use]
350    pub const fn tie_break(self) -> u8 {
351        match self {
352            FloatKind::Float32x => 0,
353            FloatKind::Float64x => 1,
354            FloatKind::Float => 4,
355            FloatKind::Double => 5,
356            FloatKind::LongDouble => 6,
357            FloatKind::Float16 => 8,
358            FloatKind::Float32 => 9,
359            FloatKind::Float64 => 10,
360            FloatKind::Float128 => 11,
361            // Never compared with a binary type, and each decimal is its own format, so these
362            // only have to be distinct.
363            FloatKind::Decimal32 => 12,
364            FloatKind::Decimal64 => 13,
365            FloatKind::Decimal128 => 14,
366        }
367    }
368
369    /// How the type is spelled in a diagnostic.
370    #[must_use]
371    pub const fn as_str(self) -> &'static str {
372        match self {
373            FloatKind::Float16 => "_Float16",
374            FloatKind::Float => "float",
375            FloatKind::Float32 => "_Float32",
376            FloatKind::Double => "double",
377            FloatKind::Float32x => "_Float32x",
378            FloatKind::Float64 => "_Float64",
379            FloatKind::LongDouble => "long double",
380            FloatKind::Float64x => "_Float64x",
381            FloatKind::Float128 => "_Float128",
382            FloatKind::Decimal32 => "_Decimal32",
383            FloatKind::Decimal64 => "_Decimal64",
384            FloatKind::Decimal128 => "_Decimal128",
385        }
386    }
387}
388
389/// How many elements an array has, which is four different answers in C.
390#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
391pub enum ArrayLen {
392    /// `int a[4]`. The count of elements, not the size in bytes.
393    Fixed(u64),
394    /// `int a[]`, an incomplete array type. It has an element type and no size, and it is
395    /// completed by an initializer or by a later declaration.
396    Unknown,
397    /// `int a[*]`, a variably modified type in a prototype, where the size exists but is not
398    /// available to the declaration that mentions it.
399    Star,
400    /// `int a[n]`, a variable length array. The size expression stays in the AST, and the
401    /// type carries only the identity of the one that made it, because two variable length
402    /// arrays written with the same element type are still distinct types.
403    Variable(VlaId),
404}
405
406/// The identity of one variable length array's size expression.
407///
408/// An opaque number handed out by whoever is building the type, which in practice is
409/// semantic analysis walking a declarator. This crate never looks inside it; it is here so
410/// that interning two variable length arrays does not accidentally make them the same type.
411#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
412pub struct VlaId(pub u32);
413
414/// Whether a record is a `struct` or a `union`.
415#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
416pub enum RecordKind {
417    /// `struct`, whose members are laid out one after another.
418    Struct,
419    /// `union`, whose members all start at offset zero.
420    Union,
421}
422
423impl RecordKind {
424    /// How the keyword is spelled in a diagnostic.
425    #[must_use]
426    pub const fn as_str(self) -> &'static str {
427        match self {
428            RecordKind::Struct => "struct",
429            RecordKind::Union => "union",
430        }
431    }
432}
433
434/// What a type is, with its qualifiers stripped off into [`Type::quals`].
435///
436/// This is `Copy` and sixteen bytes, which is what lets it be the interning key directly.
437/// Function types and record types are the two that carry a variable amount of information,
438/// and both of them are an index into a table this crate owns.
439#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
440pub enum TypeKind {
441    /// `void`.
442    Void,
443    /// `bool`, which C23 spells without an underscore and which is one byte with two values.
444    Bool,
445    /// One of the standard integer types.
446    Int(IntKind),
447    /// One of the real floating types.
448    Float(FloatKind),
449    /// `_Complex T`, holding the type of each half.
450    ///
451    /// `T` is a real floating type in C and may also be an integer one, which is a GNU
452    /// extension gcc has always had and which `_Complex int` is. The half's own type is held
453    /// rather than a floating kind, because the two spellings are the same type in every way
454    /// but what a half is, and a kind could only say the floating half of that.
455    Complex(TypeId),
456    /// `_BitInt(N)` and `unsigned _BitInt(N)`.
457    ///
458    /// A distinct kind rather than an integer type with a width, because these do not take
459    /// part in the integer promotions and folding them in with the standard types is how
460    /// that rule gets forgotten.
461    BitInt {
462        /// Whether the type is signed. A signed `_BitInt(1)` is legal and holds `0` and `-1`.
463        signed: bool,
464        /// The declared width in bits, which is what the standard calls `N`.
465        width: u32,
466    },
467    /// A pointer to the given type.
468    Pointer(TypeId),
469    /// `_Atomic(T)`, which is a type and not a qualifier. See [`Qualifiers`].
470    Atomic(TypeId),
471    /// An array of the given element type.
472    Array {
473        /// The element type.
474        elem: TypeId,
475        /// How many of them there are, which may be unknown.
476        len: ArrayLen,
477    },
478    /// A function type, whose parameter list is in this crate's side table.
479    Function(FunctionId),
480    /// A GNU vector type, `__attribute__((vector_size(n)))`.
481    Vector {
482        /// The element type, which must be a scalar.
483        elem: TypeId,
484        /// How many elements there are.
485        len: u32,
486    },
487    /// A `struct` or `union`, identified by its declaration rather than by its members.
488    Record(RecordId),
489    /// An `enum`, identified by its declaration.
490    Enum(EnumId),
491    /// A typedef name, which is sugar over whatever it was declared as.
492    ///
493    /// Every semantic decision reads [`Types::canonical`](crate::Types::canonical) and never
494    /// sees this; every diagnostic reads the type as written and sees nothing else, so the
495    /// error says `size_t` rather than `unsigned long`. Compilers that drop the sugar produce
496    /// messages nobody can act on, and compilers that decide on the sugar produce wrong
497    /// answers, and both are common.
498    Typedef {
499        /// The name, for printing.
500        name: Symbol,
501        /// What it was declared as.
502        underlying: TypeId,
503        /// What `__attribute__((aligned(n)))` on the typedef asked an object of it to be
504        /// aligned to, and [`None`] when it asked for nothing.
505        ///
506        /// The one thing a typedef changes about the type behind it, and the reason the
507        /// alignment is on this node rather than in a table beside it: two typedefs of one
508        /// underlying type that ask for different alignments are two types, so the alignment
509        /// has to be part of what the table interns them by.
510        ///
511        /// It is what the type is aligned to and not a floor on it. Written on a declaration
512        /// the attribute only ever raises, and written on a typedef GCC lets it lower as well,
513        /// so `typedef int L __attribute__((aligned(2)))` really is an `int` at a multiple of
514        /// two and `struct { char c; L x; }` really is six bytes.
515        align: Option<NonZeroU32>,
516    },
517}
518
519/// A type with its qualifiers, which together are one entry in the type table.
520#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
521pub struct Type {
522    /// What the type is.
523    pub kind: TypeKind,
524    /// What it is qualified with.
525    pub quals: Qualifiers,
526}
527
528impl Type {
529    /// An unqualified type of the given kind.
530    #[must_use]
531    pub const fn new(kind: TypeKind) -> Type {
532        Type { kind, quals: Qualifiers::NONE }
533    }
534}
535
536/// The identity of a function type in [`Types`](crate::Types).
537///
538/// Deduplicated by content, so two declarations written with the same return type, the same
539/// parameters and the same variadic flag share one of these and therefore one [`TypeId`].
540#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
541pub struct FunctionId(pub(crate) u32);
542
543/// The identity of a `struct` or `union` declaration in [`Types`](crate::Types).
544///
545/// Not deduplicated by content, because record types in C are nominal. Two `struct` types
546/// written with the same members in the same translation unit are different types, and the
547/// looser relation that does hold between them is compatibility rather than identity.
548#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
549pub struct RecordId(pub(crate) u32);
550
551/// The identity of an `enum` declaration in [`Types`](crate::Types).
552#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
553pub struct EnumId(pub(crate) u32);
554
555/// A function type.
556#[derive(Debug, Clone, PartialEq, Eq, Hash)]
557pub struct FunctionType {
558    /// What it returns.
559    pub ret: TypeId,
560    /// The parameter types, after the adjustments a parameter declaration gets: an array
561    /// parameter has already decayed to a pointer and a function parameter to a function
562    /// pointer, because those adjustments are part of forming the type and not part of
563    /// calling it.
564    pub params: Vec<TypeId>,
565    /// Whether the list ends in `...`.
566    pub variadic: bool,
567    /// Whether there was a prototype at all.
568    ///
569    /// `int f()` declares an unprototyped function before C23 and a function taking no
570    /// arguments from C23 onwards, and the difference is visible in what calls are checked
571    /// and in what the composite type of a redeclaration is. The dialect decides which
572    /// meaning `()` gets, and this records the decision rather than repeating it.
573    pub prototyped: bool,
574}