rucc_types/kind.rs
1//! What a C type is made of, before any of it has been interned.
2//!
3//! Design: `spec/07-types-and-semantics.md` section 7.1.
4//!
5//! Everything here is `Copy` and small, because [`TypeKind`] is the interning key and a key
6//! that owns a heap allocation cannot be hashed cheaply or compared cheaply. The two parts of
7//! a type that are genuinely variable length, a function's parameter list and a record's
8//! members, live in side tables and are referred to by index.
9
10use std::num::NonZeroU32;
11
12use rucc_base::Symbol;
13use rucc_target::Convention;
14
15use crate::TypeId;
16
17/// The qualifiers a type can carry.
18///
19/// A bitmask in the interning key rather than a chain of wrapper nodes, so `const int` is one
20/// entry in the table beside `int` rather than a node pointing at it. That makes stripping
21/// qualifiers a field read instead of a walk, which matters because almost every semantic rule
22/// in C is stated on the unqualified type.
23///
24/// `_Atomic` is deliberately not here. C lets it be written in the same position as a
25/// qualifier, but `_Atomic(T)` is a different type from `T` with its own size and alignment,
26/// so it is a type constructor, [`TypeKind::Atomic`], and the parser is what maps the
27/// qualifier spelling onto it.
28#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Hash, PartialOrd, Ord)]
29pub struct Qualifiers(u8);
30
31impl Qualifiers {
32 /// No qualifiers.
33 pub const NONE: Qualifiers = Qualifiers(0);
34 /// `const`.
35 pub const CONST: Qualifiers = Qualifiers(1);
36 /// `volatile`.
37 pub const VOLATILE: Qualifiers = Qualifiers(2);
38 /// `restrict`.
39 pub const RESTRICT: Qualifiers = Qualifiers(4);
40
41 /// Whether every qualifier in `other` is present here.
42 #[inline]
43 #[must_use]
44 pub const fn has(self, other: Qualifiers) -> bool {
45 self.0 & other.0 == other.0
46 }
47
48 /// This set with `other` added.
49 #[inline]
50 #[must_use]
51 pub const fn with(self, other: Qualifiers) -> Qualifiers {
52 Qualifiers(self.0 | other.0)
53 }
54
55 /// This set with `other` removed.
56 #[inline]
57 #[must_use]
58 pub const fn without(self, other: Qualifiers) -> Qualifiers {
59 Qualifiers(self.0 & !other.0)
60 }
61
62 /// Whether there are no qualifiers at all.
63 #[inline]
64 #[must_use]
65 pub const fn is_none(self) -> bool {
66 self.0 == 0
67 }
68}
69
70/// The standard integer types, the character types kept apart from them, and `__int128`.
71///
72/// `Char` is its own kind rather than an alias for one of the other two. The standard makes
73/// plain `char` a third type distinct from both `signed char` and `unsigned char` even though
74/// it has the same range as one of them, and a compiler that folds it into whichever one the
75/// target picked gets `char *` and `signed char *` wrongly deemed compatible.
76///
77/// `__int128` is here rather than modelled as a `_BitInt(128)`, because the two are different
78/// types with different layouts: `__int128` is sixteen bytes aligned to sixteen on every
79/// target we have, and `_BitInt(128)` is aligned to its granule, which is eight on x86-64. It
80/// is available everywhere for us, since all three architectures are 64-bit, and GCC has it
81/// on every 64-bit target. It is deliberately not an extended integer type in the sense the
82/// standard means, which is what keeps `intmax_t` sixty four bits wide the way GCC has it.
83#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
84pub enum IntKind {
85 /// `char`, whose signedness is a target property.
86 Char,
87 /// `signed char`.
88 SChar,
89 /// `unsigned char`.
90 UChar,
91 /// `short`.
92 Short,
93 /// `unsigned short`.
94 UShort,
95 /// `int`.
96 Int,
97 /// `unsigned int`.
98 UInt,
99 /// `long`, the width that separates LP64 from Windows LLP64.
100 Long,
101 /// `unsigned long`.
102 ULong,
103 /// `long long`.
104 LongLong,
105 /// `unsigned long long`.
106 ULongLong,
107 /// `__int128`.
108 Int128,
109 /// `unsigned __int128`.
110 UInt128,
111}
112
113impl IntKind {
114 /// Every integer kind, in rank order, with `__int128` last.
115 ///
116 /// The order is what the internal index agrees with, and it is also the order the standard
117 /// walks when it picks the type of an integer constant, so a table walk over the candidate
118 /// list for a suffix is a walk over a slice of this. `__int128` is at the end because that
119 /// is where GCC reaches for it: after every standard type has been tried and none of them
120 /// was wide enough.
121 pub const ALL: [IntKind; 13] = [
122 IntKind::Char,
123 IntKind::SChar,
124 IntKind::UChar,
125 IntKind::Short,
126 IntKind::UShort,
127 IntKind::Int,
128 IntKind::UInt,
129 IntKind::Long,
130 IntKind::ULong,
131 IntKind::LongLong,
132 IntKind::ULongLong,
133 IntKind::Int128,
134 IntKind::UInt128,
135 ];
136
137 /// A dense index, so that one of these can select a slot in a fixed size array.
138 pub(crate) const fn index(self) -> usize {
139 match self {
140 IntKind::Char => 0,
141 IntKind::SChar => 1,
142 IntKind::UChar => 2,
143 IntKind::Short => 3,
144 IntKind::UShort => 4,
145 IntKind::Int => 5,
146 IntKind::UInt => 6,
147 IntKind::Long => 7,
148 IntKind::ULong => 8,
149 IntKind::LongLong => 9,
150 IntKind::ULongLong => 10,
151 IntKind::Int128 => 11,
152 IntKind::UInt128 => 12,
153 }
154 }
155
156 /// Whether this type is signed, given what the target says about plain `char`.
157 ///
158 /// The argument is there because `char` is the one integer type whose signedness is not
159 /// in the standard. It is signed on x86-64 and unsigned on AArch64 Linux, and a compiler
160 /// that assumes either one is the source of a whole genre of bug report.
161 #[must_use]
162 pub const fn is_signed(self, char_is_signed: bool) -> bool {
163 match self {
164 IntKind::Char => char_is_signed,
165 IntKind::SChar
166 | IntKind::Short
167 | IntKind::Int
168 | IntKind::Long
169 | IntKind::LongLong
170 | IntKind::Int128 => true,
171 IntKind::UChar
172 | IntKind::UShort
173 | IntKind::UInt
174 | IntKind::ULong
175 | IntKind::ULongLong
176 | IntKind::UInt128 => false,
177 }
178 }
179
180 /// The integer conversion rank, as an ordering rather than as a number from the standard.
181 ///
182 /// The standard gives no values, only a set of relations, and every one of them is a
183 /// comparison between two ranks. Signed and unsigned of the same width share a rank, which
184 /// is what makes the usual arithmetic conversions between them pick the unsigned type
185 /// rather than the wider one.
186 #[must_use]
187 pub const fn rank(self) -> u8 {
188 match self {
189 IntKind::Char | IntKind::SChar | IntKind::UChar => 1,
190 IntKind::Short | IntKind::UShort => 2,
191 IntKind::Int | IntKind::UInt => 3,
192 IntKind::Long | IntKind::ULong => 4,
193 IntKind::LongLong | IntKind::ULongLong => 5,
194 // Above `long long`, which is what makes `__int128 + unsigned long long` an
195 // `__int128` rather than an unsigned type. Both compilers agree.
196 IntKind::Int128 | IntKind::UInt128 => 6,
197 }
198 }
199
200 /// The same width with the other signedness.
201 ///
202 /// `char` maps to `unsigned char` and back to `signed char`, which is the mapping the
203 /// usual arithmetic conversions need and is not a round trip. That asymmetry is the type
204 /// system telling the truth: there is no way back to plain `char` from either of the
205 /// other two.
206 #[must_use]
207 pub const fn flip_sign(self) -> IntKind {
208 match self {
209 IntKind::Char | IntKind::SChar => IntKind::UChar,
210 IntKind::UChar => IntKind::SChar,
211 IntKind::Short => IntKind::UShort,
212 IntKind::UShort => IntKind::Short,
213 IntKind::Int => IntKind::UInt,
214 IntKind::UInt => IntKind::Int,
215 IntKind::Long => IntKind::ULong,
216 IntKind::ULong => IntKind::Long,
217 IntKind::LongLong => IntKind::ULongLong,
218 IntKind::ULongLong => IntKind::LongLong,
219 IntKind::Int128 => IntKind::UInt128,
220 IntKind::UInt128 => IntKind::Int128,
221 }
222 }
223
224 /// How the type is spelled in a diagnostic.
225 #[must_use]
226 pub const fn as_str(self) -> &'static str {
227 match self {
228 IntKind::Char => "char",
229 IntKind::SChar => "signed char",
230 IntKind::UChar => "unsigned char",
231 IntKind::Short => "short",
232 IntKind::UShort => "unsigned short",
233 IntKind::Int => "int",
234 IntKind::UInt => "unsigned int",
235 IntKind::Long => "long",
236 IntKind::ULong => "unsigned long",
237 IntKind::LongLong => "long long",
238 IntKind::ULongLong => "unsigned long long",
239 IntKind::Int128 => "__int128",
240 IntKind::UInt128 => "unsigned __int128",
241 }
242 }
243}
244
245/// The real floating types.
246///
247/// Nine of them, which is three standard ones and six from C23 Annex H. The interchange types
248/// `_Float16`, `_Float32`, `_Float64` and `_Float128` name an IEEE format outright, and the
249/// extended types `_Float32x` and `_Float64x` name whatever the target has that is wider than
250/// the interchange type they are named after, which makes `_Float64x` the x87 format on x86 and
251/// quad precision on AArch64. None of them is the standard type it shares a format with:
252/// `_Float64` and `double` are both binary64 and are two types, which `_Generic` can tell apart
253/// and which decides what `_Float64 + double` is.
254///
255/// `_Float128x` is a type no target gcc supports has, so it is not here. The three decimal
256/// floating types from C23 are, and they are real floating types like the others in every way
257/// except the one that matters most: a decimal and a binary type never meet in an operation, so
258/// the usual arithmetic conversions have no answer for the pair and the program is refused.
259#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
260pub enum FloatKind {
261 /// `_Float16`, always the binary16 format.
262 Float16,
263 /// `float`, always the binary32 format.
264 Float,
265 /// `_Float32`, always the binary32 format, and not the same type as `float`.
266 Float32,
267 /// `double`, always the binary64 format.
268 Double,
269 /// `_Float32x`, the format the target has that is wider than `_Float32`, which is binary64
270 /// everywhere this compiles for.
271 Float32x,
272 /// `_Float64`, always the binary64 format, and not the same type as `double`.
273 Float64,
274 /// `long double`, whose format is a target property and is not always distinct from
275 /// `double`. It is 80 bits of x87 on SysV x86-64, quad precision on AArch64 Linux, and
276 /// the same as `double` on Apple and Windows.
277 LongDouble,
278 /// `_Float64x`, the format the target has that is wider than `_Float64`. That is the x87
279 /// eighty bit format on x86-64 and quad precision on AArch64 and RISC-V, and unlike
280 /// `long double` it does not become a `double` on Apple or on Windows.
281 Float64x,
282 /// `_Float128`, always the binary128 format.
283 Float128,
284 /// `_Decimal32`, the decimal32 format in the binary integer encoding.
285 Decimal32,
286 /// `_Decimal64`, the decimal64 format in the binary integer encoding.
287 Decimal64,
288 /// `_Decimal128`, the decimal128 format in the binary integer encoding.
289 Decimal128,
290}
291
292impl FloatKind {
293 /// Every real floating type, in the order they are written above.
294 ///
295 /// Not in rank order, because there is no such order to put them in: which of `long double`
296 /// and `_Float64x` is the wider one is a question about the target, and on Apple the answer
297 /// is the second.
298 pub const ALL: [FloatKind; 12] = [
299 FloatKind::Float16,
300 FloatKind::Float,
301 FloatKind::Float32,
302 FloatKind::Double,
303 FloatKind::Float32x,
304 FloatKind::Float64,
305 FloatKind::LongDouble,
306 FloatKind::Float64x,
307 FloatKind::Float128,
308 FloatKind::Decimal32,
309 FloatKind::Decimal64,
310 FloatKind::Decimal128,
311 ];
312
313 /// A dense index, so that one of these can select a slot in a fixed size array.
314 pub(crate) const fn index(self) -> usize {
315 match self {
316 FloatKind::Float16 => 0,
317 FloatKind::Float => 1,
318 FloatKind::Float32 => 2,
319 FloatKind::Double => 3,
320 FloatKind::Float32x => 4,
321 FloatKind::Float64 => 5,
322 FloatKind::LongDouble => 6,
323 FloatKind::Float64x => 7,
324 FloatKind::Float128 => 8,
325 FloatKind::Decimal32 => 9,
326 FloatKind::Decimal64 => 10,
327 FloatKind::Decimal128 => 11,
328 }
329 }
330
331 /// Whether this is one of the three decimal types.
332 #[must_use]
333 pub const fn is_decimal(self) -> bool {
334 matches!(self, FloatKind::Decimal32 | FloatKind::Decimal64 | FloatKind::Decimal128)
335 }
336
337 /// What decides between two of these when they have the same format.
338 ///
339 /// Two real floating types can be the same format and still be two types, and then the
340 /// format cannot say which of them an operation on both of them produces. C23 answers with
341 /// the family first: an interchange type wins over the standard type it shares a format
342 /// with, and the standard type wins over an extended one, so `double + _Float64` is a
343 /// `_Float64` and `double + _Float32x` is a `double`. Inside a family it is the usual order,
344 /// which only ever comes up between `double` and `long double` on the targets where the
345 /// second one is the first one.
346 ///
347 /// Higher wins. This is not an ordering on the types on its own, because it says nothing
348 /// about the formats: `_Float32` sits above `long double` here and loses to it everywhere it
349 /// meets it.
350 #[must_use]
351 pub const fn tie_break(self) -> u8 {
352 match self {
353 FloatKind::Float32x => 0,
354 FloatKind::Float64x => 1,
355 FloatKind::Float => 4,
356 FloatKind::Double => 5,
357 FloatKind::LongDouble => 6,
358 FloatKind::Float16 => 8,
359 FloatKind::Float32 => 9,
360 FloatKind::Float64 => 10,
361 FloatKind::Float128 => 11,
362 // Never compared with a binary type, and each decimal is its own format, so these
363 // only have to be distinct.
364 FloatKind::Decimal32 => 12,
365 FloatKind::Decimal64 => 13,
366 FloatKind::Decimal128 => 14,
367 }
368 }
369
370 /// How the type is spelled in a diagnostic.
371 #[must_use]
372 pub const fn as_str(self) -> &'static str {
373 match self {
374 FloatKind::Float16 => "_Float16",
375 FloatKind::Float => "float",
376 FloatKind::Float32 => "_Float32",
377 FloatKind::Double => "double",
378 FloatKind::Float32x => "_Float32x",
379 FloatKind::Float64 => "_Float64",
380 FloatKind::LongDouble => "long double",
381 FloatKind::Float64x => "_Float64x",
382 FloatKind::Float128 => "_Float128",
383 FloatKind::Decimal32 => "_Decimal32",
384 FloatKind::Decimal64 => "_Decimal64",
385 FloatKind::Decimal128 => "_Decimal128",
386 }
387 }
388}
389
390/// How many elements an array has, which is four different answers in C.
391#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
392pub enum ArrayLen {
393 /// `int a[4]`. The count of elements, not the size in bytes.
394 Fixed(u64),
395 /// `int a[]`, an incomplete array type. It has an element type and no size, and it is
396 /// completed by an initializer or by a later declaration.
397 Unknown,
398 /// `int a[*]`, a variably modified type in a prototype, where the size exists but is not
399 /// available to the declaration that mentions it.
400 Star,
401 /// `int a[n]`, a variable length array. The size expression stays in the AST, and the
402 /// type carries only the identity of the one that made it, because two variable length
403 /// arrays written with the same element type are still distinct types.
404 Variable(VlaId),
405}
406
407/// The identity of one variable length array's size expression.
408///
409/// An opaque number handed out by whoever is building the type, which in practice is
410/// semantic analysis walking a declarator. This crate never looks inside it; it is here so
411/// that interning two variable length arrays does not accidentally make them the same type.
412#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
413pub struct VlaId(pub u32);
414
415/// Whether a record is a `struct` or a `union`.
416#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
417pub enum RecordKind {
418 /// `struct`, whose members are laid out one after another.
419 Struct,
420 /// `union`, whose members all start at offset zero.
421 Union,
422}
423
424impl RecordKind {
425 /// How the keyword is spelled in a diagnostic.
426 #[must_use]
427 pub const fn as_str(self) -> &'static str {
428 match self {
429 RecordKind::Struct => "struct",
430 RecordKind::Union => "union",
431 }
432 }
433}
434
435/// What a type is, with its qualifiers stripped off into [`Type::quals`].
436///
437/// This is `Copy` and sixteen bytes, which is what lets it be the interning key directly.
438/// Function types and record types are the two that carry a variable amount of information,
439/// and both of them are an index into a table this crate owns.
440#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
441pub enum TypeKind {
442 /// `void`.
443 Void,
444 /// `bool`, which C23 spells without an underscore and which is one byte with two values.
445 Bool,
446 /// One of the standard integer types.
447 Int(IntKind),
448 /// One of the real floating types.
449 Float(FloatKind),
450 /// `_Complex T`, holding the type of each half.
451 ///
452 /// `T` is a real floating type in C and may also be an integer one, which is a GNU
453 /// extension gcc has always had and which `_Complex int` is. The half's own type is held
454 /// rather than a floating kind, because the two spellings are the same type in every way
455 /// but what a half is, and a kind could only say the floating half of that.
456 Complex(TypeId),
457 /// `_BitInt(N)` and `unsigned _BitInt(N)`.
458 ///
459 /// A distinct kind rather than an integer type with a width, because these do not take
460 /// part in the integer promotions and folding them in with the standard types is how
461 /// that rule gets forgotten.
462 BitInt {
463 /// Whether the type is signed. A signed `_BitInt(1)` is legal and holds `0` and `-1`.
464 signed: bool,
465 /// The declared width in bits, which is what the standard calls `N`.
466 width: u32,
467 },
468 /// A pointer to the given type.
469 Pointer(TypeId),
470 /// `_Atomic(T)`, which is a type and not a qualifier. See [`Qualifiers`].
471 Atomic(TypeId),
472 /// An array of the given element type.
473 Array {
474 /// The element type.
475 elem: TypeId,
476 /// How many of them there are, which may be unknown.
477 len: ArrayLen,
478 },
479 /// A function type, whose parameter list is in this crate's side table.
480 Function(FunctionId),
481 /// A GNU vector type, `__attribute__((vector_size(n)))`.
482 Vector {
483 /// The element type, which must be a scalar.
484 elem: TypeId,
485 /// How many elements there are.
486 len: u32,
487 },
488 /// A `struct` or `union`, identified by its declaration rather than by its members.
489 Record(RecordId),
490 /// An `enum`, identified by its declaration.
491 Enum(EnumId),
492 /// A typedef name, which is sugar over whatever it was declared as.
493 ///
494 /// Every semantic decision reads [`Types::canonical`](crate::Types::canonical) and never
495 /// sees this; every diagnostic reads the type as written and sees nothing else, so the
496 /// error says `size_t` rather than `unsigned long`. Compilers that drop the sugar produce
497 /// messages nobody can act on, and compilers that decide on the sugar produce wrong
498 /// answers, and both are common.
499 Typedef {
500 /// The name, for printing.
501 name: Symbol,
502 /// What it was declared as.
503 underlying: TypeId,
504 /// What `__attribute__((aligned(n)))` on the typedef asked an object of it to be
505 /// aligned to, and [`None`] when it asked for nothing.
506 ///
507 /// The one thing a typedef changes about the type behind it, and the reason the
508 /// alignment is on this node rather than in a table beside it: two typedefs of one
509 /// underlying type that ask for different alignments are two types, so the alignment
510 /// has to be part of what the table interns them by.
511 ///
512 /// It is what the type is aligned to and not a floor on it. Written on a declaration
513 /// the attribute only ever raises, and written on a typedef GCC lets it lower as well,
514 /// so `typedef int L __attribute__((aligned(2)))` really is an `int` at a multiple of
515 /// two and `struct { char c; L x; }` really is six bytes.
516 align: Option<NonZeroU32>,
517 },
518}
519
520/// A type with its qualifiers, which together are one entry in the type table.
521#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
522pub struct Type {
523 /// What the type is.
524 pub kind: TypeKind,
525 /// What it is qualified with.
526 pub quals: Qualifiers,
527}
528
529impl Type {
530 /// An unqualified type of the given kind.
531 #[must_use]
532 pub const fn new(kind: TypeKind) -> Type {
533 Type { kind, quals: Qualifiers::NONE }
534 }
535}
536
537/// The identity of a function type in [`Types`](crate::Types).
538///
539/// Deduplicated by content, so two declarations written with the same return type, the same
540/// parameters and the same variadic flag share one of these and therefore one [`TypeId`].
541#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
542pub struct FunctionId(pub(crate) u32);
543
544/// The identity of a `struct` or `union` declaration in [`Types`](crate::Types).
545///
546/// Not deduplicated by content, because record types in C are nominal. Two `struct` types
547/// written with the same members in the same translation unit are different types, and the
548/// looser relation that does hold between them is compatibility rather than identity.
549#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
550pub struct RecordId(pub(crate) u32);
551
552/// The identity of an `enum` declaration in [`Types`](crate::Types).
553#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
554pub struct EnumId(pub(crate) u32);
555
556/// A function type.
557#[derive(Debug, Clone, PartialEq, Eq, Hash)]
558pub struct FunctionType {
559 /// What it returns.
560 pub ret: TypeId,
561 /// The parameter types, after the adjustments a parameter declaration gets: an array
562 /// parameter has already decayed to a pointer and a function parameter to a function
563 /// pointer, because those adjustments are part of forming the type and not part of
564 /// calling it.
565 pub params: Vec<TypeId>,
566 /// Whether the list ends in `...`.
567 pub variadic: bool,
568 /// Whether there was a prototype at all.
569 ///
570 /// `int f()` declares an unprototyped function before C23 and a function taking no
571 /// arguments from C23 onwards, and the difference is visible in what calls are checked
572 /// and in what the composite type of a redeclaration is. The dialect decides which
573 /// meaning `()` gets, and this records the decision rather than repeating it.
574 pub prototyped: bool,
575 /// Which calling convention a call to it and its own body use.
576 ///
577 /// Part of the type because gcc makes it part of the type: a pointer to an `ms_abi` function
578 /// and a pointer to an ordinary one are pointers to different types on Linux, and a function
579 /// declared once each way has conflicting types. [`Convention::Target`] is the target's own
580 /// whichever attribute named it, so `ms_abi` on Windows changes nothing about a type and
581 /// neither does `sysv_abi` anywhere else, and a type that never met an attribute is the same
582 /// type as one that met the attribute naming the target's convention.
583 pub convention: Convention,
584}