Skip to main content

rucc_types/
layout.rs

1//! How large a type is and what it has to be aligned to, computed from the target description.
2//!
3//! Design: `spec/07-types-and-semantics.md` section 7.1 and `spec/18-package-layout.md`
4//! section 18.2, which is the rule that none of this may be a `#[cfg]`.
5//!
6//! Every number here comes out of [`TargetInfo`] rather than out of the host. That is not
7//! pedantry: `long` is four bytes on Windows and eight on Linux, `long double` is eight bytes
8//! on Apple and sixteen on SysV x86-64, and a cross compiler that asks its own platform gets
9//! both of them wrong. The widths were checked against GCC 13 on x86-64 Linux and against
10//! clang on AArch64 Darwin rather than recalled.
11//!
12//! [`integer_info`] is here for the same reason and answers a neighbouring question: not how
13//! large the object is but how wide the value in it is, which is not the same number for `bool`
14//! or for a `_BitInt` and is what folding a constant depends on.
15//!
16//! Records are the one thing not computed here. Their layout depends on their members, on
17//! bit-field packing and on attributes, so it is computed by whoever walks the members and
18//! recorded with [`Types::complete_record`](crate::Types::complete_record); this module reads
19//! it back.
20
21use rucc_base::float::Format;
22use rucc_target::TargetInfo;
23
24use crate::classify::{bare, element_as_written};
25use crate::kind::{ArrayLen, FloatKind, IntKind, TypeKind};
26use crate::types::{TypeId, Types};
27
28/// The size and alignment of a complete object type.
29#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
30pub struct Layout {
31    /// The size in bytes, which is what `sizeof` answers.
32    pub size: u64,
33    /// The alignment in bytes, which is what `_Alignof` answers. Always a power of two.
34    pub align: u64,
35}
36
37impl Layout {
38    /// A layout with the given size and alignment.
39    #[must_use]
40    pub const fn new(size: u64, align: u64) -> Layout {
41        Layout { size, align }
42    }
43
44    /// A scalar that is as aligned as it is large, which every one on a 64-bit target is.
45    #[must_use]
46    const fn scalar(size: u64) -> Layout {
47        Layout { size, align: size }
48    }
49}
50
51/// Why a type has no layout.
52#[derive(Debug, Clone, Copy, PartialEq, Eq)]
53pub enum LayoutError {
54    /// The type is incomplete: `void`, an array with no size, or a record or enumeration whose
55    /// definition has not been seen. GNU C gives `sizeof(void)` the value one, and that is a
56    /// dialect decision made where there is a warning to emit, not here.
57    Incomplete,
58    /// The type is a function type, which has no size at all. GNU C gives it the value one for
59    /// the same reason it does for `void`.
60    Function,
61    /// The type is complete and how large it is depends on something the program computes, which
62    /// is a variable length array or a record with one among its members.
63    ///
64    /// Not a diagnostic on its own. Every one of these has a size, worked out where the
65    /// declaration carrying it was reached, and this is what tells a caller to go and ask for it
66    /// rather than to report that there is none. [`align`] still answers for one.
67    Variable,
68    /// The type is complete and describes an object larger than one may be, which an array
69    /// declaration can ask for by multiplying two innocent looking numbers.
70    ///
71    /// The limit is
72    /// [`TargetInfo::max_object_size`](rucc_target::TargetInfo::max_object_size), which is
73    /// `PTRDIFF_MAX` and not the address space: an object of every byte there is would have a
74    /// pointer subtraction across it with no answer.
75    TooLarge,
76}
77
78impl std::fmt::Display for LayoutError {
79    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
80        let text = match self {
81            LayoutError::Incomplete => "the type is incomplete",
82            LayoutError::Function => "a function type has no size",
83            LayoutError::Variable => "the size of the type is not known until the program runs",
84            LayoutError::TooLarge => "the type is larger than an object may be",
85        };
86        f.write_str(text)
87    }
88}
89
90impl std::error::Error for LayoutError {}
91
92/// The size and alignment of `id` on `target`.
93///
94/// # Errors
95///
96/// [`LayoutError`] when the type has no layout, which is a normal answer rather than a bug:
97/// `sizeof` an incomplete type is a diagnostic, and the caller is the one holding the span.
98pub fn layout(types: &Types, id: TypeId, target: &TargetInfo) -> Result<Layout, LayoutError> {
99    // Sugar has whatever layout the type behind it has, with the one exception that a typedef
100    // may say what an object of it is aligned to. That is asked before the sugar is resolved
101    // because resolving it is what throws the answer away, and it replaces the alignment rather
102    // than raising it: `typedef int L __attribute__((aligned(2)))` really is an `int` at a
103    // multiple of two. The size is untouched, which is GCC's answer and not an omission, so
104    // `sizeof` an over aligned typedef is the size of what it stands for and an array of one is
105    // a thing GCC refuses rather than pads.
106    let asked = types.align_override(id);
107    let plain = unaligned_layout(types, id, target);
108    match asked {
109        Some(align) => plain.map(|layout| Layout::new(layout.size, u64::from(align.get()))),
110        None => plain,
111    }
112}
113
114/// What an object of `id` has to be aligned to, whether or not it has a size here.
115///
116/// The number [`layout`] answers with wherever there is one. Where there is not, which is a
117/// variable length array or a record with one among its members, there is still an alignment,
118/// because an alignment never depends on a length: an array is as aligned as its element however
119/// long it turns out to be, and a record's alignment is decided by its members rather than by
120/// where they land.
121///
122/// # Errors
123///
124/// [`LayoutError`] when the type has no alignment either, which is every reason [`layout`] has
125/// for having no size but the one this is here for.
126pub fn align(types: &Types, id: TypeId, target: &TargetInfo) -> Result<u64, LayoutError> {
127    let natural = match layout(types, id, target) {
128        Ok(laid_out) => return Ok(laid_out.align),
129        Err(LayoutError::Variable) => variable_align(types, id, target)?,
130        Err(error) => return Err(error),
131    };
132    // A typedef that asked for an alignment replaces the one the type has, the same way it does
133    // in [`layout`], and it is asked here as well because a `typedef int T[n]` may carry one.
134    match types.align_override(id) {
135        Some(asked) => Ok(u64::from(asked.get())),
136        None => Ok(natural),
137    }
138}
139
140/// The alignment of a type whose size is not known here.
141fn variable_align(types: &Types, id: TypeId, target: &TargetInfo) -> Result<u64, LayoutError> {
142    if let Some(elem) = element_as_written(types, id) {
143        return align(types, elem, target);
144    }
145    match types.kind(types.canonical(id)) {
146        // The alignment is in the layout beside the recipe, where the size is zero and this is
147        // the part of it that means something.
148        TypeKind::Record(record) => {
149            let info = types.record_info(record);
150            Ok(info.layout.ok_or(LayoutError::Incomplete)?.align)
151        }
152        _ => Err(LayoutError::Variable),
153    }
154}
155
156/// The same, before any typedef in the sugar has had its say about the alignment.
157fn unaligned_layout(types: &Types, id: TypeId, target: &TargetInfo) -> Result<Layout, LayoutError> {
158    // A typedef of an array of a typedef is common enough that resolving it once here beats
159    // resolving it at every arm below.
160    let original = id;
161    let id = types.canonical(id);
162    match types.kind(id) {
163        TypeKind::Void => Err(LayoutError::Incomplete),
164        TypeKind::Bool => Ok(Layout::scalar(1)),
165        TypeKind::Int(kind) => Ok(int_layout(kind, target)),
166        TypeKind::Float(kind) => Ok(float_layout(kind, target)),
167        TypeKind::Complex(part) => {
168            // Two of the component, adjacent, with the component's own alignment rather than
169            // the pair's. `_Complex long double` on SysV x86-64 is thirty two bytes aligned to
170            // sixteen, which is what both GCC and clang report.
171            let part = layout(types, part, target)?;
172            Ok(Layout::new(part.size * 2, part.align))
173        }
174        TypeKind::BitInt { width, .. } => Ok(bit_int_layout(width, target)),
175        TypeKind::Pointer(_) => {
176            Ok(Layout::new(target.scalars.pointer_size, target.scalars.pointer_align))
177        }
178        TypeKind::Function(_) => Err(LayoutError::Function),
179        TypeKind::Atomic(inner) => {
180            let inner = layout(types, inner, target)?;
181            Ok(atomic_layout(inner))
182        }
183        TypeKind::Array { elem, len } => {
184            let ArrayLen::Fixed(count) = len else {
185                // A length the program computes is a size that exists and is not a number here.
186                // A length left out and a `[*]` in a prototype are neither, so those two stay
187                // what they have always been, which is incomplete.
188                if matches!(len, ArrayLen::Variable(_)) {
189                    return Err(LayoutError::Variable);
190                }
191                return Err(LayoutError::Incomplete);
192            };
193            // The element as it was written and not the canonical one, which has lost any
194            // alignment a typedef in it asked for.
195            let elem = element_as_written(types, original).unwrap_or(elem);
196            let elem = layout(types, elem, target)?;
197            let size = elem.size.checked_mul(count).ok_or(LayoutError::TooLarge)?;
198            if size > target.max_object_size() {
199                return Err(LayoutError::TooLarge);
200            }
201            Ok(Layout::new(size, elem.align))
202        }
203        TypeKind::Vector { elem, len } => {
204            let elem = layout(types, elem, target)?;
205            let raw = elem.size.checked_mul(u64::from(len)).ok_or(LayoutError::TooLarge)?;
206            Ok(vector_layout(raw))
207        }
208        TypeKind::Record(record) => {
209            let info = types.record_info(record);
210            if info.variable.is_some() {
211                return Err(LayoutError::Variable);
212            }
213            info.layout.ok_or(LayoutError::Incomplete)
214        }
215        TypeKind::Enum(id) => {
216            let underlying = types.enum_info(id).underlying.ok_or(LayoutError::Incomplete)?;
217            layout(types, underlying, target)
218        }
219        // Unreachable in practice: the id was canonicalised on the way in. Answering rather
220        // than panicking, because a wrong size is easier to find than a crash in a compiler.
221        TypeKind::Typedef { underlying, .. } => layout(types, underlying, target),
222    }
223}
224
225/// The width of a standard integer type in bits.
226#[must_use]
227pub fn int_width(kind: IntKind, target: &TargetInfo) -> u32 {
228    match kind {
229        IntKind::Char | IntKind::SChar | IntKind::UChar => 8,
230        IntKind::Short | IntKind::UShort => 16,
231        IntKind::Int | IntKind::UInt => 32,
232        IntKind::Long | IntKind::ULong => target.long_width,
233        IntKind::LongLong | IntKind::ULongLong => 64,
234        IntKind::Int128 | IntKind::UInt128 => 128,
235    }
236}
237
238/// What an integer type is once it no longer matters how it was spelled.
239///
240/// A width and a signedness, which between them are everything the value of an integer constant
241/// depends on. `int`, an enumeration represented in `int`, and `_BitInt(32)` are three different
242/// types with one [`IntegerInfo`], and every question about what a constant of any of them holds
243/// has the same answer for all three.
244///
245/// The width is the value's and not the object's. `bool` is one bit here and one byte in
246/// [`layout`], and `_BitInt(37)` is thirty seven bits here and eight bytes there.
247#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
248pub struct IntegerInfo {
249    /// Whether the type can hold a negative value.
250    pub signed: bool,
251    /// How many bits of a value the type keeps.
252    pub width: u32,
253}
254
255impl IntegerInfo {
256    /// An integer type of the given signedness and width.
257    #[must_use]
258    pub const fn new(signed: bool, width: u32) -> IntegerInfo {
259        IntegerInfo { signed, width }
260    }
261
262    /// The value `raw` becomes once it is stored in a type of this shape.
263    ///
264    /// The low `width` bits of it, extended into the rest by the signedness. That is the form a
265    /// folded constant is held in, so `300` wrapped by a `char` is `44`, `-1` wrapped by an
266    /// `unsigned int` is `4294967295`, and a value of a hundred and twenty eight bit type is
267    /// itself, because there is nothing wider left to extend it into.
268    #[must_use]
269    pub const fn wrap(self, raw: i128) -> i128 {
270        if self.width == 0 {
271            return 0;
272        }
273        if self.width >= 128 {
274            return raw;
275        }
276        let unused = 128 - self.width;
277        if self.signed {
278            (raw << unused) >> unused
279        } else {
280            (((raw as u128) << unused) >> unused) as i128
281        }
282    }
283
284    /// Whether `raw` is a value a type of this shape can hold.
285    ///
286    /// Every hundred and twenty eight bit pattern is a value of a hundred and twenty eight bit
287    /// type, of either signedness, which is why this is a question about the width rather than
288    /// a comparison against a pair of bounds: `unsigned __int128` has a greatest value that no
289    /// [`i128`] can be handed to ask about.
290    #[must_use]
291    pub const fn holds(self, raw: i128) -> bool {
292        self.wrap(raw) == raw
293    }
294}
295
296/// The signedness and width of an integer type, and [`None`] when `id` is not one.
297///
298/// Every integer type C has. `bool` is one bit and unsigned, an enumeration answers as whatever
299/// it is represented in, a `_BitInt` answers with the width it was written with, and `_Atomic`
300/// and a typedef name answer as the type underneath. The coverage is the point: the shape used
301/// by the conversion ranks in `convert.rs` deliberately covers only the two the ranks are
302/// defined over, and folding a constant with that one would get `bool` and every enumeration
303/// wrong rather than refusing them.
304#[must_use]
305pub fn integer_info(types: &Types, id: TypeId, target: &TargetInfo) -> Option<IntegerInfo> {
306    match bare(types, id) {
307        TypeKind::Bool => Some(IntegerInfo::new(false, 1)),
308        TypeKind::Int(kind) => {
309            Some(IntegerInfo::new(kind.is_signed(target.char_is_signed), int_width(kind, target)))
310        }
311        TypeKind::BitInt { signed, width } => Some(IntegerInfo::new(signed, width)),
312        // An enumeration is represented in some integer type, and until its definition has been
313        // seen there is no answer to give. Saying so beats picking `int`, because a caller that
314        // folds a constant in a width the type does not have folds it wrongly and silently.
315        TypeKind::Enum(id) => {
316            let underlying = types.enum_info(id).underlying?;
317            integer_info(types, underlying, target)
318        }
319        _ => None,
320    }
321}
322
323/// The size and alignment of a standard integer type.
324///
325/// The alignment is the size on all but two rows of the target table, and the two are the reason
326/// this is not written as one. System V i386 aligns an eight byte integer to four, and s390x caps
327/// every scalar at eight, so `__int128` there is sixteen bytes aligned to eight.
328fn int_layout(kind: IntKind, target: &TargetInfo) -> Layout {
329    let size = u64::from(int_width(kind, target) / 8);
330    let align = match kind {
331        IntKind::LongLong | IntKind::ULongLong => target.scalars.long_long_align,
332        IntKind::Int128 | IntKind::UInt128 => capped(size, target),
333        _ => size,
334    };
335    Layout::new(size, align)
336}
337
338/// The size and alignment of a real floating type.
339fn float_layout(kind: FloatKind, target: &TargetInfo) -> Layout {
340    let size = u64::from(float_width(kind, target) / 8);
341    let align = match kind {
342        // A `double` is aligned to four on System V i386 and to eight everywhere else, including
343        // under mingw on the same architecture, which is why the number comes from the ABI
344        // description rather than from the size.
345        FloatKind::Double | FloatKind::Float64 | FloatKind::Float32x => target.scalars.double.align,
346        // Twelve bytes aligned to four on i386, sixteen aligned to sixteen on x86-64 and sixteen
347        // aligned to eight on s390x, all of them a `long double`.
348        FloatKind::LongDouble => target.scalars.long_double.align,
349        FloatKind::Float64x | FloatKind::Float128 => capped(size, target),
350        // Each decimal is aligned to its size on every target gcc has them for, the sixteen
351        // byte one included.
352        FloatKind::Float16
353        | FloatKind::Float
354        | FloatKind::Float32
355        | FloatKind::Decimal32
356        | FloatKind::Decimal64
357        | FloatKind::Decimal128 => size,
358    };
359    Layout::new(size, align)
360}
361
362/// A natural alignment of `size` with the target's cap on scalar alignment applied.
363fn capped(size: u64, target: &TargetInfo) -> u64 {
364    match target.scalars.max_field_align {
365        Some(cap) => size.min(cap),
366        None => size,
367    }
368}
369
370/// The width of a real floating type in bits, including the padding `long double` carries.
371///
372/// The number for `long double` is storage rather than precision. Eighty bits of x87 occupy
373/// sixteen bytes on SysV x86-64, and it is the sixteen that `sizeof` answers with.
374#[must_use]
375pub fn float_width(kind: FloatKind, target: &TargetInfo) -> u32 {
376    match kind {
377        FloatKind::Float16 => 16,
378        FloatKind::Float | FloatKind::Float32 | FloatKind::Decimal32 => 32,
379        FloatKind::Double | FloatKind::Float32x | FloatKind::Float64 | FloatKind::Decimal64 => 64,
380        FloatKind::Decimal128 => 128,
381        FloatKind::LongDouble => target.long_double_width,
382        // The same sixteen bytes whichever of the two formats it is, for the same reason
383        // `long double` is sixteen on x86-64: the x87 eighty bits are stored padded.
384        FloatKind::Float64x | FloatKind::Float128 => 128,
385    }
386}
387
388/// The binary format a real floating type has on `target`.
389///
390/// Not derivable from [`float_width`], which is why it is a separate question: the width of a
391/// `long double` on SysV x86-64 is a hundred and twenty eight bits and its format is the eighty
392/// bit x87 one, and a compiler that picked the format by the size would fold every `long double`
393/// constant on that target with seventeen decimal digits too many.
394#[must_use]
395pub fn float_format(kind: FloatKind, target: &TargetInfo) -> Format {
396    match kind {
397        FloatKind::Float16 => Format::Half,
398        FloatKind::Float | FloatKind::Float32 => Format::Single,
399        FloatKind::Double | FloatKind::Float32x | FloatKind::Float64 => Format::Double,
400        FloatKind::LongDouble => target.long_double_format,
401        // A target whose widest format is a `double` has no `_Float64x` and the front end should
402        // never have built one, so the answer here is the widest format the target does have
403        // rather than a panic in a compiler.
404        FloatKind::Float64x => target.float64x_format.unwrap_or(target.long_double_format),
405        FloatKind::Float128 => Format::Quad,
406        FloatKind::Decimal32 => Format::Decimal32,
407        FloatKind::Decimal64 => Format::Decimal64,
408        FloatKind::Decimal128 => Format::Decimal128,
409    }
410}
411
412/// The layout of `_BitInt(width)`.
413///
414/// Up to 64 bits a `_BitInt` is laid out like the smallest standard integer type that holds
415/// it, so the size is the byte count rounded up to a power of two and the alignment is the
416/// size. Above that the psABIs treat it as an array of a granule instead, and the granule is
417/// not the same everywhere: it is 64 bits on x86-64 and RISC-V and 128 on AArch64, which is
418/// why `_BitInt(65)` is sixteen bytes aligned to eight on the first and sixteen bytes aligned
419/// to sixteen on the second. Measured with clang 18 on x86-64 Linux and clang on AArch64
420/// Darwin, including the cases above 128 bits where the size keeps growing by a granule.
421fn bit_int_layout(width: u32, target: &TargetInfo) -> Layout {
422    let bytes = u64::from(width).div_ceil(8);
423    if bytes <= 8 {
424        let size = bytes.max(1).next_power_of_two();
425        // Like the standard type it is laid out as, which on System V i386 means an eight byte
426        // one is aligned to four rather than to eight.
427        return Layout::new(size, size.min(target.scalars.long_long_align));
428    }
429    let granule = u64::from(target.bit_int_granule / 8);
430    Layout::new(bytes.next_multiple_of(granule), granule)
431}
432
433/// The layout of `_Atomic(T)` given the layout of `T`.
434///
435/// Same size, and an alignment raised to the size when the size is one of the widths the
436/// target can do a lock free access at. That is why `_Atomic` is a type and not a qualifier:
437/// a sixteen byte structure is aligned to eight and `_Atomic` of it is aligned to sixteen, and
438/// a type system that treated the two as one type would silently disagree with itself about
439/// where the object goes. Checked against GCC 13 on x86-64 Linux and clang on AArch64 Darwin,
440/// which report exactly that.
441fn atomic_layout(inner: Layout) -> Layout {
442    if inner.size.is_power_of_two() && inner.size <= 16 {
443        return Layout::new(inner.size, inner.align.max(inner.size));
444    }
445    inner
446}
447
448/// The layout of a GNU vector whose elements occupy `raw` bytes in total.
449///
450/// Rounded up to a power of two and aligned to the whole thing, which is what GCC does with a
451/// `vector_size` that is not already one. GCC rejects an element count that is not a power of
452/// two and clang rounds instead, so this rounds and leaves the rejecting to whoever is holding
453/// the attribute and the dialect.
454fn vector_layout(raw: u64) -> Layout {
455    let size = raw.max(1).next_power_of_two();
456    Layout::scalar(size)
457}