rucc_target/lib.rs
1//! Target descriptions: triples, and the facts about a target that the rest of the
2//! compiler reads rather than hard-codes.
3//!
4//! Design: `spec/12-abi-and-runtime.md`. Layer rank 2, see `spec/18-package-layout.md`.
5//!
6//! The rule from `spec/18-package-layout.md` section 18.2 is that there is no
7//! target-specific code outside this crate, `rucc-tuple`, `rucc-abi`, `rucc-sysroot` and the
8//! per-target rule sets. Those four are one group rather than four exceptions: the tuple names
9//! a machine, `rucc-abi` says what its types look like and how its calls are made,
10//! `rucc-sysroot` says where its headers and libraries are, and this crate is what the rest of
11//! the compiler reads all of it through. Everything a pass
12//! needs to know about a target is a field it can read here. That rule is what makes the
13//! claim in `spec/10-backend.md` testable, namely that a new target is a rule set and a few
14//! data files, and `M10` brings up a fourth target specifically to put a number on it.
15//!
16//! [`TargetInfo::call`] is the other half of that rule and the one with teeth. How a structure
17//! travels between a caller and a callee is the target's answer rather than C's, so the walk to
18//! the IR flattens a C type into a [`Shape`] and asks here what form it takes. Every psABI rule
19//! is behind [`Call`] and nothing outside this crate matches on an architecture to find one.
20//! The rules themselves are `rucc-abi`'s, as data rather than as code, and this crate hands the
21//! question over to them. It answers [`None`] on a target whose ABI is not written down yet,
22//! which today is AArch64 on Windows and nothing else.
23//!
24//! # Status
25//!
26//! Triple parsing and the basic data model are real, which is what `rucc --print-config`
27//! reports, and so is the argument classification of every psABI in
28//! `spec/12-abi-and-runtime.md` sections 12.2 to 12.5, which `rucc-abi` describes as data and
29//! this crate selects between. x86-64's register file is written down,
30//! in [`x86_64`], along with what each of the two conventions over it does with each register,
31//! what each of its machine instructions does with its operands, and which instructions a frame
32//! is made of, which is [`FrameInsts`]. AArch64's and RISC-V's arrive with their backends.
33//! Machine models land in `M6`.
34//!
35//! This crate is tier 3 in `spec/18-package-layout.md` section 18.5: its Rust API is
36//! explicitly unstable and will change without a major version bump.
37
38#![doc(html_root_url = "https://docs.rs/rucc-target/0.10.42")]
39
40use std::fmt;
41use std::str::FromStr;
42
43use rucc_abi::DataLayout;
44use rucc_base::float::Format;
45use rucc_tuple::{self as tuple, TargetTuple};
46
47mod abi;
48mod bits;
49mod branch;
50mod flags;
51mod frame;
52mod machine;
53mod operand;
54mod regs;
55pub mod x86_64;
56
57pub use crate::abi::{Arg, Call, Kind, Pass, Piece, Scalar, Shape, Slot};
58pub use crate::bits::BitInsts;
59pub use crate::branch::{BranchInsts, Fusion};
60pub use crate::flags::{Compare, FlagInsts, Reader, Reads, Zeroing};
61pub use crate::frame::{ClassMoves, FrameInsts, Probe};
62pub use crate::machine::MachineInsts;
63pub use crate::operand::{Constraint, OperandDesc, Role};
64pub use crate::regs::{
65 CallRegs, ClassInfo, Guard, PhysReg, Places, RegClass, RegFile, Segment, Trace, Where,
66};
67
68/// A target architecture.
69#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash)]
70// Deliberately not `#[non_exhaustive]`. Adding a variant here has to break every
71// match that needs to change, in this workspace and in anyone else's code. That is
72// the property `spec/10-backend.md` section 10.8 is claiming when it says adding a
73// target is a data change: the compiler tells you every place the data is read.
74pub enum Arch {
75 /// x86-64, the first target and the one `M3` brings up.
76 X86_64,
77 /// AArch64, the second target, `M6`.
78 Aarch64,
79 /// 64-bit RISC-V. `spec/10-backend.md` calls this the middle-end canary, because it has
80 /// no condition codes and no complex addressing modes, so anything the middle end got
81 /// away with on x86-64 shows up here.
82 Riscv64,
83}
84
85impl Arch {
86 /// Pointer width in bits.
87 pub const fn pointer_width(self) -> u32 {
88 match self {
89 Arch::X86_64 | Arch::Aarch64 | Arch::Riscv64 => 64,
90 }
91 }
92
93 /// Whether the target is little-endian.
94 pub const fn is_little_endian(self) -> bool {
95 match self {
96 Arch::X86_64 | Arch::Aarch64 | Arch::Riscv64 => true,
97 }
98 }
99
100 /// The name as it appears in a triple.
101 pub const fn as_str(self) -> &'static str {
102 match self {
103 Arch::X86_64 => "x86_64",
104 Arch::Aarch64 => "aarch64",
105 Arch::Riscv64 => "riscv64",
106 }
107 }
108}
109
110/// The operating system a target runs on.
111#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash)]
112// Deliberately not `#[non_exhaustive]`. Adding a variant here has to break every
113// match that needs to change, in this workspace and in anyone else's code. That is
114// the property `spec/10-backend.md` section 10.8 is claiming when it says adding a
115// target is a data change: the compiler tells you every place the data is read.
116pub enum Os {
117 /// Linux, hosted or freestanding.
118 Linux,
119 /// Apple platforms. `spec/12-abi-and-runtime.md` section 12.3 lists the four places
120 /// Apple diverges from AAPCS64, and every one of them is a real bug if missed.
121 Darwin,
122 /// Windows.
123 Windows,
124 /// No operating system, which is what `-ffreestanding` kernel work looks like.
125 None,
126}
127
128impl Os {
129 /// The name as it appears in a triple.
130 pub const fn as_str(self) -> &'static str {
131 match self {
132 Os::Linux => "linux",
133 Os::Darwin => "darwin",
134 Os::Windows => "windows",
135 Os::None => "none",
136 }
137 }
138
139 /// The object file format this operating system uses.
140 pub const fn object_format(self) -> ObjectFormat {
141 match self {
142 Os::Linux | Os::None => ObjectFormat::Elf,
143 Os::Darwin => ObjectFormat::MachO,
144 Os::Windows => ObjectFormat::Coff,
145 }
146 }
147}
148
149/// The C runtime and ABI variant.
150#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash)]
151// Deliberately not `#[non_exhaustive]`. Adding a variant here has to break every
152// match that needs to change, in this workspace and in anyone else's code. That is
153// the property `spec/10-backend.md` section 10.8 is claiming when it says adding a
154// target is a data change: the compiler tells you every place the data is read.
155pub enum Env {
156 /// The default for the operating system.
157 None,
158 /// glibc.
159 Gnu,
160 /// musl.
161 Musl,
162 /// The MSVC ABI.
163 Msvc,
164}
165
166impl Env {
167 /// The name as it appears in a triple, if it appears at all.
168 pub const fn as_str(self) -> &'static str {
169 match self {
170 Env::None => "none",
171 Env::Gnu => "gnu",
172 Env::Musl => "musl",
173 Env::Msvc => "msvc",
174 }
175 }
176}
177
178/// The object file format to emit.
179#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash)]
180// Deliberately not `#[non_exhaustive]`. Adding a variant here has to break every
181// match that needs to change, in this workspace and in anyone else's code. That is
182// the property `spec/10-backend.md` section 10.8 is claiming when it says adding a
183// target is a data change: the compiler tells you every place the data is read.
184pub enum ObjectFormat {
185 /// ELF.
186 Elf,
187 /// Mach-O.
188 MachO,
189 /// COFF.
190 Coff,
191 /// WebAssembly, which is a format for a module rather than for a machine's object file and
192 /// is in this list because the target table has two rows that emit one.
193 Wasm,
194}
195
196impl ObjectFormat {
197 /// The name used in diagnostics and in `--print-config`.
198 pub const fn as_str(self) -> &'static str {
199 match self {
200 ObjectFormat::Elf => "elf",
201 ObjectFormat::MachO => "macho",
202 ObjectFormat::Coff => "coff",
203 ObjectFormat::Wasm => "wasm",
204 }
205 }
206
207 /// The same format as [`rucc_tuple::ObjectFormat`] names it.
208 ///
209 /// The two enumerations exist because the tuple describes forty two targets and this crate
210 /// describes what the compiler emits for one, and they will stay separate for as long as that
211 /// is true. This is the one place they are put side by side.
212 #[must_use]
213 pub const fn from_tuple(format: tuple::ObjectFormat) -> Self {
214 match format {
215 tuple::ObjectFormat::Elf => ObjectFormat::Elf,
216 tuple::ObjectFormat::MachO => ObjectFormat::MachO,
217 tuple::ObjectFormat::Coff => ObjectFormat::Coff,
218 tuple::ObjectFormat::Wasm => ObjectFormat::Wasm,
219 }
220 }
221}
222
223/// A target triple.
224///
225/// We accept the LLVM-style `arch-vendor-os-env` form because that is what build systems
226/// pass, and we normalise it to the three fields we actually branch on. The vendor field is
227/// parsed and discarded: no decision in the compiler depends on it, and keeping it would
228/// invite one.
229#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash)]
230pub struct Triple {
231 /// The architecture.
232 pub arch: Arch,
233 /// The operating system.
234 pub os: Os,
235 /// The runtime and ABI variant.
236 pub env: Env,
237}
238
239impl Triple {
240 /// A triple from its three parts.
241 pub const fn new(arch: Arch, os: Os, env: Env) -> Self {
242 Self { arch, os, env }
243 }
244
245 /// The same machine as a [`TargetTuple`], which is what the layout and ABI descriptions are
246 /// written over.
247 ///
248 /// The tuple carries ten fields and this carries three, so this fills the other seven in from
249 /// their defaults, and every one of those defaults is the answer for the targets this type can
250 /// spell. There is no `x32` here and no big-endian AArch64, so the data model and the byte
251 /// order follow the architecture, and the sub-architecture, the versions and the float ABI have
252 /// nothing to say about any of the combinations.
253 ///
254 /// The environment is narrowed rather than copied across. This type will hold
255 /// `Triple { os: Darwin, env: Gnu }`, because its parser takes the fields by content and
256 /// `aarch64-apple-darwin-gnu` is a string somebody can type, and that is not a machine: a
257 /// Darwin target has one libc and it is not glibc. A tuple refuses to describe one, so the
258 /// pairs that are not machines are mapped to the environment the operating system actually
259 /// has.
260 ///
261 /// # Panics
262 ///
263 /// Never, for a triple this type can hold, which `every_triple_describes_a_machine` checks by
264 /// building all forty eight of them.
265 #[must_use]
266 pub fn tuple(self) -> TargetTuple {
267 let arch = match self.arch {
268 Arch::X86_64 => tuple::Arch::X86_64,
269 Arch::Aarch64 => tuple::Arch::Aarch64,
270 Arch::Riscv64 => tuple::Arch::Riscv64,
271 };
272 let os = match self.os {
273 Os::Linux => tuple::Os::Linux,
274 // macOS rather than iOS, because the three field triple cannot tell them apart and
275 // this compiler is hosted on the one and not on the other.
276 Os::Darwin => tuple::Os::MacOs,
277 Os::Windows => tuple::Os::Windows,
278 Os::None => tuple::Os::None,
279 };
280 let env = match (self.os, self.env) {
281 (Os::Linux, Env::Musl) => tuple::Env::Musl,
282 (Os::Linux, _) => tuple::Env::Gnu,
283 // mingw-w64 is a real Windows environment and the one place `gnu` survives the
284 // narrowing, because it has a different `long double` from MSVC on the same OS.
285 (Os::Windows, Env::Gnu) => tuple::Env::Gnu,
286 (Os::Windows, _) => tuple::Env::Msvc,
287 // Darwin and freestanding have no libc to name.
288 (Os::Darwin | Os::None, _) => tuple::Env::None,
289 };
290 TargetTuple::builder(arch, os)
291 .env(env)
292 .build()
293 .expect("every triple this type can hold describes a machine")
294 }
295
296 /// The triple that describes the same machine as `target`, if this type can spell it.
297 ///
298 /// The inverse of [`Triple::tuple`], and computed by running that function over every triple
299 /// there is rather than by writing the narrowing out a second time. A second table would be a
300 /// second thing to keep in step, and the failure it invites is not a compile error: it is one
301 /// row of the matrix quietly answering as a neighbour.
302 ///
303 /// It returns `None` for most of the target table, and that is the honest answer rather than a
304 /// gap to be papered over. `rucc-abi` describes the scalar layout of all forty two rows, and
305 /// this type holds three fields with three architectures in the first, so seventeen of those
306 /// rows have a [`TargetInfo`] and the other twenty five do not. Anything that needs to lay a
307 /// record out for `s390x-linux-gnu` needs that gap closed rather than an approximation of it.
308 ///
309 /// The environment of the answer is the narrowed one, so the triple this gives back is the
310 /// canonical spelling of that machine: `Env::None` on Darwin and on a freestanding target,
311 /// never the `Env::Gnu` that a parser will accept from a string somebody typed.
312 #[must_use]
313 pub fn from_tuple(target: TargetTuple) -> Option<Triple> {
314 // Four triples narrow onto `x86_64-linux-gnu`, because a Darwin triple claiming glibc is
315 // a string somebody can type and not a machine. So a match is not enough on its own: the
316 // answer is the candidate whose environment came through the narrowing unchanged, and
317 // anything else is only a fallback for the day a narrowing loses a spelling entirely.
318 let mut fallback = None;
319 for arch in [Arch::X86_64, Arch::Aarch64, Arch::Riscv64] {
320 for os in [Os::Linux, Os::Darwin, Os::Windows, Os::None] {
321 for env in [Env::None, Env::Gnu, Env::Musl, Env::Msvc] {
322 let candidate = Triple::new(arch, os, env);
323 if candidate.tuple() != target {
324 continue;
325 }
326 // By name rather than by a match on the pair, so that an environment added to
327 // either enumeration does not need a line here. The one name the two spell
328 // differently is the absent one, which the tuple writes as nothing.
329 let survived = match env {
330 Env::None => target.env() == tuple::Env::None,
331 _ => env.as_str() == target.env().as_str(),
332 };
333 if survived {
334 return Some(candidate);
335 }
336 fallback.get_or_insert(candidate);
337 }
338 }
339 }
340 fallback
341 }
342
343 /// The triple of the machine this compiler is running on.
344 ///
345 /// Used as the default target, which is what makes `rucc hello.c` work with no flags.
346 /// Unknown host combinations are not an error here: they are reported by the driver,
347 /// where there is somewhere to report them to.
348 pub fn host() -> Option<Self> {
349 let arch = match std::env::consts::ARCH {
350 "x86_64" => Arch::X86_64,
351 "aarch64" => Arch::Aarch64,
352 "riscv64" => Arch::Riscv64,
353 _ => return None,
354 };
355 // Which libc this is matters, and `std::env::consts` does not say. A compiler built on
356 // Alpine and defaulting to `x86_64-unknown-linux-gnu` describes a machine it is not
357 // running on: musl and glibc disagree about `int_fast16_t` among other things, and a
358 // header that is written out of the predefined type names picks the disagreement up.
359 // The libc rucc itself was linked against is the best evidence available about the one
360 // the code it compiles will be linked against, and it is right on every machine where
361 // rucc was built for the machine it runs on.
362 let linux = if cfg!(target_env = "musl") { Env::Musl } else { Env::Gnu };
363 let (os, env) = match std::env::consts::OS {
364 "linux" => (Os::Linux, linux),
365 "macos" => (Os::Darwin, Env::None),
366 "windows" => (Os::Windows, Env::Msvc),
367 _ => return None,
368 };
369 Some(Self::new(arch, os, env))
370 }
371}
372
373impl fmt::Display for Triple {
374 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
375 // Always four fields, always the same spelling, because this string ends up in
376 // `--print-config` output that people diff.
377 write!(f, "{}-unknown-{}-{}", self.arch.as_str(), self.os.as_str(), self.env.as_str())
378 }
379}
380
381/// Why a triple failed to parse.
382#[derive(Debug, Clone, PartialEq, Eq)]
383pub struct ParseTripleError {
384 /// The triple as given.
385 pub input: String,
386 /// What specifically was not recognised.
387 pub reason: &'static str,
388}
389
390impl fmt::Display for ParseTripleError {
391 fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
392 write!(f, "unsupported target triple `{}`: {}", self.input, self.reason)
393 }
394}
395
396impl std::error::Error for ParseTripleError {}
397
398impl FromStr for Triple {
399 type Err = ParseTripleError;
400
401 fn from_str(s: &str) -> Result<Self, Self::Err> {
402 let err = |reason| ParseTripleError { input: s.to_owned(), reason };
403 let mut parts = s.split('-');
404
405 let arch = match parts.next() {
406 Some("x86_64" | "amd64") => Arch::X86_64,
407 Some("aarch64" | "arm64") => Arch::Aarch64,
408 Some("riscv64") => Arch::Riscv64,
409 _ => return Err(err("unknown architecture")),
410 };
411
412 // The vendor field is optional in practice. `x86_64-linux-gnu` and
413 // `x86_64-unknown-linux-gnu` both occur in the wild and mean the same thing, so the
414 // remaining fields are matched by content rather than by position.
415 let rest: Vec<&str> = parts.collect();
416 let mut os = None;
417 let mut env = None;
418 for part in &rest {
419 match *part {
420 "linux" => os = Some(Os::Linux),
421 "darwin" | "macos" | "macosx" | "ios" => os = Some(Os::Darwin),
422 "windows" | "win32" => os = Some(Os::Windows),
423 // `none` is the one token that means different things in the two positions.
424 // In `x86_64-unknown-none-elf` it is the operating system; in
425 // `aarch64-apple-darwin-none` it is the environment. Which one it is depends
426 // on whether an operating system has already been seen, and that rule is what
427 // makes `Display` round-trip through `FromStr`.
428 "none" if os.is_none() => os = Some(Os::None),
429 "none" => env = Some(Env::None),
430 "elf" => os = os.or(Some(Os::None)),
431 "gnu" | "gnueabi" | "gnueabihf" => env = Some(Env::Gnu),
432 "musl" | "musleabi" | "musleabihf" => env = Some(Env::Musl),
433 "msvc" => env = Some(Env::Msvc),
434 _ => {}
435 }
436 }
437
438 let os = os.ok_or_else(|| err("unknown operating system"))?;
439 let env = env.unwrap_or(match os {
440 Os::Linux => Env::Gnu,
441 Os::Windows => Env::Msvc,
442 Os::Darwin | Os::None => Env::None,
443 });
444 Ok(Self::new(arch, os, env))
445 }
446}
447
448/// The facts about a target that the compiler reads instead of hard-coding.
449///
450/// This is the whole of what a pass is allowed to know about where its output will run.
451/// It grows, and every field added here is one fewer `#[cfg]` somewhere it should not be.
452#[derive(Debug, Clone, PartialEq, Eq)]
453#[non_exhaustive]
454pub struct TargetInfo {
455 /// The machine this describes, as the ten field tuple rather than as a three field triple.
456 ///
457 /// It is the tuple because a record layout is a question every row of the target table has an
458 /// answer to, and a triple can spell fifteen of the forty two. Nothing else in this type had
459 /// to change to widen it: every field below is already derived from `rucc-abi`'s description
460 /// of this tuple, and the ones that were not were the bugs.
461 pub tuple: TargetTuple,
462 /// The sizes, the alignments and the signedness this target's headers were written against.
463 ///
464 /// The widths below are views of this and the alignments are not, which is the reason it is
465 /// kept whole. A `long long` is eight bytes on every row of the table and is aligned to four
466 /// on System V i386 and to eight everywhere else, and no width can say that.
467 pub scalars: DataLayout,
468 /// Width of a pointer in bits.
469 pub pointer_width: u32,
470 /// Whether bytes are ordered little end first.
471 pub little_endian: bool,
472 /// Whether a bare `char` is signed.
473 ///
474 /// Signed on x86-64 and unsigned on AArch64 Linux, which is the classic source of code
475 /// that works on one and not the other, so it is data rather than an assumption.
476 pub char_is_signed: bool,
477 /// Width of `long` in bits. This is the field that separates the LP64 world from
478 /// Windows LLP64.
479 pub long_width: u32,
480 /// Width of `long double` in bits: 80 bits of x87 stored in 128 on every x86-64 target but
481 /// MSVC, 128 of true quad precision on AArch64 Linux and RISC-V, and 64 on Apple's AArch64 and
482 /// under MSVC.
483 ///
484 /// Apple's x86-64 is not one of the 64-bit ones, which is the trap. The change to a `double`
485 /// came with AArch64 and the Intel answer stayed as it was, so `x86_64-apple-darwin` and
486 /// `x86_64-unknown-linux-gnu` agree here and `aarch64-apple-darwin` is the odd one.
487 pub long_double_width: u32,
488 /// The format `long double` actually is, which the width does not say.
489 ///
490 /// It is 128 bits wide on SysV x86-64 and on AArch64 Linux and the two are not the same
491 /// type: one is the x87 eighty bit format padded out to sixteen bytes and the other is
492 /// true quad precision with a hundred and thirteen bits of significand. Anything that
493 /// converts a constant or folds one has to know which, and the width alone cannot say.
494 pub long_double_format: Format,
495 /// The format `_Float64x` is, which is the widest format the target has short of a software
496 /// one.
497 ///
498 /// It follows the architecture and not the operating system, which is what makes it worth a
499 /// field of its own next to `long double`. Apple and Windows define `long double` as a
500 /// `double` and neither of them takes `_Float64x` down with it: the type has to be wider
501 /// than a `_Float64`, so it is the x87 eighty bit format on x86-64 and quad precision on
502 /// AArch64 and RISC-V wherever it is written.
503 ///
504 /// [`None`] on a machine whose widest format is a `double`, which is 32-bit ARM and wasm32.
505 /// The type does not exist there and neither reference defines the macros that describe it,
506 /// so the honest answer is that there is no format rather than a `double` in its place.
507 pub float64x_format: Option<Format>,
508 /// Whether the target has `_Float16`.
509 ///
510 /// The named types are not all universal the way `_Float32` and `_Float64` are. gcc 13 has
511 /// this one on x86-64, AArch64 and RISC-V and does not have it on i686, armv7, ppc64le or
512 /// s390x, which was measured by compiling a declaration of it with each of those cross
513 /// compilers. The `__FLT16_*__` macros and the `f16` suffix are defined on exactly the rows
514 /// where the type is, so all three ask this one field.
515 ///
516 /// i686 is the row worth explaining. gcc aims at the baseline of the target rather than at
517 /// whatever chip is under it, and half precision on x86 needs SSE2, which is in the baseline
518 /// of x86-64 and not in the baseline of i686. So the two x86 rows disagree, and a `-msse2`
519 /// on the command line would move the 32-bit one, which is a thing this compiler has no
520 /// place to say yet.
521 pub has_float16: bool,
522 /// Whether the target has `_Float128`.
523 ///
524 /// Every row but 32-bit ARM among the seven measured against gcc 13. x86-64 and i686 have it
525 /// in software, and AArch64, RISC-V, s390x and ppc64le have it because quad precision is
526 /// already the format of something on those machines. armv7 has no format wider than a
527 /// `double` at all, so the type is not there and gcc says so.
528 ///
529 /// This is the ISO spelling. gcc's `__float128` is a narrower thing and is not this field:
530 /// that name exists on x86 and PowerPC only, and on AArch64, RISC-V and s390x gcc offers
531 /// `_Float128` in its place when a program writes it. `__SIZEOF_FLOAT128__` follows the
532 /// vendor name rather than the type, which is why it is missing on rows where the type is
533 /// there.
534 pub has_float128: bool,
535 /// Width of `wchar_t` in bits, which decides what a wide literal is encoded in.
536 ///
537 /// It is 16 on Windows, so a wide string there is UTF-16 and a character outside the basic
538 /// plane takes two elements, and 32 everywhere else, where a wide string is UTF-32 and no
539 /// character takes more than one.
540 pub wchar_width: u32,
541 /// Whether `wchar_t` is signed.
542 ///
543 /// x86-64 Linux makes it a signed `int` and AArch64 Linux makes it an `unsigned int`,
544 /// following the psABI's rule for plain `char`, so `L'\xffffffff'` is minus one on one of
545 /// them and four billion on the other.
546 pub wchar_is_signed: bool,
547 /// The granule a `_BitInt` wider than 64 bits is laid out in, in bits.
548 ///
549 /// Above 64 bits the psABIs stop treating a `_BitInt` like a standard integer type and
550 /// start treating it like an array of these, so its size is rounded up to a multiple of
551 /// this and its alignment is this. It is 64 on x86-64 and RISC-V and 128 on AArch64, which
552 /// is why `_BitInt(65)` is sixteen bytes aligned to eight on one and sixteen bytes aligned
553 /// to sixteen on the other. Measured with clang 18 on x86-64 Linux and clang on AArch64
554 /// Darwin rather than read off the documents.
555 pub bit_int_granule: u32,
556 /// The widest access, in bits, this machine performs atomically without taking a lock.
557 ///
558 /// It is what `__atomic_always_lock_free` and `__atomic_is_lock_free` answer from, and it is
559 /// a claim about what this compiler emits rather than about what the processor is capable of.
560 /// Sixty four on every target here. x86-64 does sixteen bytes atomically with `cmpxchg16b`,
561 /// which is not in the baseline the psABI names and which nothing in this compiler writes, and
562 /// AArch64 does the same with its pair instructions, which nothing writes either. A target
563 /// that answered yes for sixteen bytes and then called a library that has to take a lock for
564 /// them would have two answers to one question, and the wrong one is the one in the header.
565 pub lock_free_width: u32,
566 /// The object format to emit.
567 pub object_format: ObjectFormat,
568 /// How bit-fields are allocated into storage, which is the one record layout question where
569 /// two targets in this table run different algorithms rather than the same one over different
570 /// numbers.
571 pub bit_field_style: BitFieldStyle,
572 /// Whether an unnamed bit-field raises the record's alignment the way a named one does.
573 ///
574 /// Almost everywhere it does not, which is why `struct { char c; int :20; }` is four bytes
575 /// aligned to one on x86-64 and four aligned to four with the field named. AAPCS64 says
576 /// otherwise and says it for the zero width member too, so `struct { unsigned :0; }` is
577 /// aligned to four on AArch64 Linux and to one on Apple's AArch64, on Windows on AArch64, on
578 /// x86-64 and on RISC-V. Measured with the pinned reference across every row that has one,
579 /// because it is neither an architecture rule nor an operating system rule: it is the ABI, and
580 /// Apple and Microsoft each dropped it.
581 ///
582 /// Windows says yes as well, and there it is not AAPCS64 but Microsoft's own rule, which is
583 /// why the two facts are separate fields rather than one. In a `union` the Microsoft rule goes
584 /// further and no bit-field contributes alignment at all, named or not, so this field is only
585 /// half the answer there and [`BitFieldStyle`] carries the other half.
586 pub unnamed_bit_field_aligns: bool,
587 /// How large a record with no storage in it is, in bytes, before its alignment is applied.
588 ///
589 /// Zero everywhere but MSVC, where it is four. A `struct` with no members is not C at all, it
590 /// is a GNU extension, and C++ gives it a size of one, so there is no standard to read the
591 /// answer out of and the number has to come from whatever else compiles for the target. On
592 /// mingw that is GCC and the answer is zero. On MSVC it is clang, because MSVC itself rejects
593 /// the declaration outright, and clang's Microsoft record layout gives it four bytes and gives
594 /// an array of three of them twelve. So this is a fact about the environment and not about the
595 /// operating system, which is the one place in this type where those two come apart in that
596 /// direction.
597 ///
598 /// It covers a record with no members and a record whose only members occupy nothing, which is
599 /// the zero width bit-field, the zero length array and the flexible array member. All four
600 /// were measured and all four agree.
601 pub empty_record_size: u64,
602 /// What `__builtin_va_list` is, which is the type every `va_list` in every header is a
603 /// typedef of.
604 ///
605 /// [`None`] on a target whose answer is a type this crate does not build yet. 32-bit ARM's is
606 /// a structure of one pointer and s390x's is a structure of four members, and neither is any
607 /// of the four below. A target with no backend cannot compile a call to `va_arg` in any case,
608 /// so saying so beats naming a neighbour's type and having a header believe it.
609 pub va_list: Option<VaList>,
610 /// The registers the machine has, which is [`RegFile::EMPTY`] for an architecture nothing
611 /// has described yet.
612 pub regs: &'static RegFile,
613 /// Which registers the calling convention gives which job, or `None` while the
614 /// architecture has no register file to name them out of.
615 pub call_regs: Option<&'static CallRegs>,
616}
617
618/// The type a target's `__builtin_va_list` is.
619///
620/// A variable argument list is the one place a psABI dictates a C type rather than how a type
621/// travels, and the four answers below are not four spellings of one thing: `sizeof(va_list)` is
622/// eight bytes on Apple's AArch64 and thirty two on Linux's, and on SysV x86-64 a `va_list` is an
623/// array, so a `va_list` passed to a function is passed as a pointer and one assigned to another
624/// is a constraint violation rather than a copy. Code in the wild depends on all of that.
625#[derive(Debug, Clone, Copy, PartialEq, Eq)]
626// Deliberately not `#[non_exhaustive]`, for the reason [`Arch`] is not: a fifth answer here is
627// a fifth type to build, and every place that builds one should stop compiling until it does.
628pub enum VaList {
629 /// `char *`, which is what a target whose arguments are all passed in one place needs: the
630 /// address of the next argument and nothing else. Apple's AArch64 and both Windows targets.
631 CharPointer,
632 /// `void *`, which is the RISC-V psABI's spelling of the same thing.
633 VoidPointer,
634 /// `struct __va_list_tag { unsigned gp_offset, fp_offset; void *overflow_arg_area,
635 /// *reg_save_area; } [1]`, the SysV x86-64 one. Arguments arrive in two register files and
636 /// on the stack, so the list is a cursor into each, and the array of one is what makes
637 /// passing it to `vfprintf` pass its address.
638 SysV,
639 /// `struct __va_list { void *__stack, *__gr_top, *__vr_top; int __gr_offs, __vr_offs; }`,
640 /// the AAPCS64 one. The same idea as SysV's, counting down from the top of each save area
641 /// rather than up from the bottom, and not an array.
642 Aapcs,
643}
644
645impl VaList {
646 /// The name used in `--print-config`.
647 #[must_use]
648 pub const fn as_str(self) -> &'static str {
649 match self {
650 VaList::CharPointer => "char-pointer",
651 VaList::VoidPointer => "void-pointer",
652 VaList::SysV => "sysv",
653 VaList::Aapcs => "aapcs",
654 }
655 }
656}
657
658/// How a target allocates bit-fields into storage.
659///
660/// Everything else about laying a record out is one algorithm reading different sizes and
661/// alignments per target. This is not: the two answers below place the same members at different
662/// offsets and give the same struct different sizes, and no amount of changing what an `int` is
663/// turns one into the other. `struct { unsigned m:3; char c; }` is four bytes with the `char` at
664/// offset one under the first and eight bytes with it at offset four under the second.
665#[derive(Debug, Clone, Copy, PartialEq, Eq)]
666// Deliberately not `#[non_exhaustive]`, for the reason [`Arch`] is not: a third answer here is a
667// third algorithm to write, and every place that chooses between them should stop compiling until
668// it does.
669pub enum BitFieldStyle {
670 /// The Itanium C++ ABI's rule, which every psABI in this table except Windows follows. A
671 /// bit-field goes at the next free bit unless that would make it span more storage than its
672 /// own type occupies, in which case it starts at the next boundary of its alignment. Storage
673 /// is shared between members of different types freely, so `struct { char a:3; unsigned b:3; }`
674 /// is four bytes with both fields in the first one.
675 Itanium,
676 /// Microsoft's rule, which both Windows environments follow and not only MSVC. A run of
677 /// bit-fields is allocated into a unit the size and alignment of the declared type, and the
678 /// unit is closed both when the next member's declared type has a different size and when the
679 /// field does not fit in what is left. An ordinary member closes a unit too, and the closed
680 /// unit occupies its whole declared size whether or not the bits were used. So the same struct
681 /// is eight bytes: a one byte unit for the `char` and a four byte one for the `unsigned`,
682 /// aligned to four.
683 Microsoft,
684}
685
686impl BitFieldStyle {
687 /// The name used in `--print-config`.
688 #[must_use]
689 pub const fn as_str(self) -> &'static str {
690 match self {
691 BitFieldStyle::Itanium => "itanium",
692 BitFieldStyle::Microsoft => "microsoft",
693 }
694 }
695}
696
697/// A width in bits, from a size in bytes.
698///
699/// The fields here are widths because that is what a predefined macro and a diagnostic say, and a
700/// layout is sizes because that is what `sizeof` says. The conversion belongs at the one boundary
701/// between them rather than at every reader of one of these fields.
702fn bits(bytes: u64) -> u32 {
703 u32::try_from(bytes * 8).expect("no standard type is four billion bits wide")
704}
705
706impl TargetInfo {
707 /// The description of `triple`.
708 ///
709 /// The three field triple spells fifteen of the forty two rows of the target table, which is
710 /// every row with a backend and every row a driver will be handed today, so this is what the
711 /// compiler proper calls. [`TargetInfo::for_tuple`] is the one that answers for the whole
712 /// table.
713 #[must_use]
714 pub fn new(triple: Triple) -> Self {
715 Self::for_tuple(triple.tuple())
716 }
717
718 /// The description of `target`.
719 ///
720 /// Every row of the target table has one of these, whether or not there is a backend that can
721 /// emit code for it, because laying a record out and reading a header are questions that do
722 /// not need a backend. The fields that genuinely need one say so: [`TargetInfo::regs`] is
723 /// empty and [`TargetInfo::call_regs`] is [`None`] for an architecture whose register file is
724 /// not written down.
725 #[must_use]
726 pub fn for_tuple(target: TargetTuple) -> Self {
727 // Every size, alignment and signedness below is `rucc-abi`'s answer over the ten field
728 // tuple rather than a match written out here. They were written out here, and the copy was
729 // wrong about `x86_64-apple-darwin`, whose `long double` is the eighty bit x87 format in
730 // sixteen bytes and not a `double`: Apple made that change on AArch64 and left the Intel
731 // answer alone, and a rule keyed on the operating system takes both.
732 let layout = DataLayout::for_target(target);
733 // AArch64, RISC-V and everything else with a row and no backend have register files and
734 // this crate has not written them down yet. They arrive with the backends that need them,
735 // in M6 and M7.
736 let regs = match target.arch() {
737 tuple::Arch::X86_64 => &x86_64::REGS,
738 _ => &RegFile::EMPTY,
739 };
740 let call_regs = match (target.arch(), target.os()) {
741 (tuple::Arch::X86_64, tuple::Os::Windows) => Some(&x86_64::WIN64),
742 // Apple's x86-64 follows SysV, and its divergences from it are on AArch64.
743 (tuple::Arch::X86_64, _) => Some(&x86_64::SYSV),
744 _ => None,
745 };
746 Self {
747 tuple: target,
748 scalars: layout,
749 pointer_width: bits(layout.pointer_size),
750 little_endian: target.is_little_endian(),
751 char_is_signed: layout.char_is_signed,
752 long_width: bits(layout.long_size),
753 long_double_width: bits(layout.long_double.size),
754 long_double_format: layout.long_double.format,
755 float64x_format: float64x_format(target),
756 has_float16: has_float16(target),
757 has_float128: has_float128(target),
758 wchar_width: bits(layout.wchar_size),
759 wchar_is_signed: layout.wchar_is_signed,
760 bit_int_granule: bit_int_granule(target),
761 // Eight bytes everywhere, for the reason the field gives: it is the widest access this
762 // compiler writes an instruction for, and every one of these machines has a wider one
763 // that nothing here reaches. It is a claim about the code this compiler emits, so the
764 // day a backend emits a sixteen byte atomic is the day this stops being one number.
765 lock_free_width: 64,
766 object_format: ObjectFormat::from_tuple(target.object_format()),
767 bit_field_style: bit_field_style(target),
768 unnamed_bit_field_aligns: unnamed_bit_field_aligns(target),
769 // The environment and not the operating system, so `x86_64-windows-gnu` keeps GCC's
770 // zero while `x86_64-windows-msvc` takes clang's four.
771 empty_record_size: match target.env() {
772 tuple::Env::Msvc => 4,
773 _ => 0,
774 },
775 va_list: va_list(target),
776 regs,
777 call_regs,
778 }
779 }
780
781 /// The largest an object may be on this target, in bytes.
782 ///
783 /// `PTRDIFF_MAX`, which is what C 6.5.6 needs it to be: subtracting two pointers into one
784 /// object has to have an answer, and the answer has a `ptrdiff_t` to fit in. So an object
785 /// of exactly this many bytes is allowed and one byte more is not, which is the line GCC
786 /// draws too. It is the only size limit in the compiler and every layout question that has
787 /// one asks here rather than at whatever its own arithmetic happens to overflow at.
788 #[must_use]
789 pub const fn max_object_size(&self) -> u64 {
790 (1u64 << (self.pointer_width - 1)) - 1
791 }
792}
793
794/// The format `_Float64x` is, where the target has one.
795fn float64x_format(target: TargetTuple) -> Option<Format> {
796 match target.arch() {
797 // The x87 unit is on the machine whatever the operating system says a `long double` is,
798 // so `x86_64-apple-darwin` and `x86_64-windows-msvc` both have an eighty bit `_Float64x`
799 // and an eight byte `long double`.
800 tuple::Arch::X86_64 | tuple::Arch::X86 => Some(Format::X87Extended),
801 tuple::Arch::Aarch64
802 | tuple::Arch::Riscv64
803 | tuple::Arch::Riscv32
804 | tuple::Arch::LoongArch64
805 | tuple::Arch::S390x
806 | tuple::Arch::PowerPc64 => Some(Format::Quad),
807 // Nothing on these machines is wider than a `double`, so there is no type here to
808 // describe and neither reference defines the macros that would describe it.
809 tuple::Arch::Arm | tuple::Arch::Arm64Ec | tuple::Arch::Wasm32 => None,
810 }
811}
812
813/// Whether the target has `_Float16`.
814fn has_float16(target: TargetTuple) -> bool {
815 match target.arch() {
816 // Half precision is in the baseline of these: SSE2 on x86-64, the FP16 storage format
817 // every ARMv8 has, and RISC-V, where gcc gives the type whether or not the hardware has
818 // the instructions to go with it.
819 tuple::Arch::X86_64
820 | tuple::Arch::Aarch64
821 | tuple::Arch::Arm64Ec
822 | tuple::Arch::Riscv64
823 | tuple::Arch::Riscv32 => true,
824 // i686 for the reason the field gives, which is the baseline and not the chip, and the
825 // rest are machines gcc 13 has not written the type for.
826 tuple::Arch::X86
827 | tuple::Arch::Arm
828 | tuple::Arch::LoongArch64
829 | tuple::Arch::PowerPc64
830 | tuple::Arch::S390x
831 | tuple::Arch::Wasm32 => false,
832 }
833}
834
835/// Whether the target has `_Float128`.
836fn has_float128(target: TargetTuple) -> bool {
837 match target.arch() {
838 // Either the machine already has quad precision, which is the AArch64, RISC-V, s390x and
839 // PowerPC answer, or the compiler provides it in software, which is what x86 does.
840 tuple::Arch::X86_64
841 | tuple::Arch::X86
842 | tuple::Arch::Aarch64
843 | tuple::Arch::Arm64Ec
844 | tuple::Arch::Riscv64
845 | tuple::Arch::Riscv32
846 | tuple::Arch::LoongArch64
847 | tuple::Arch::PowerPc64
848 | tuple::Arch::S390x => true,
849 // The same two rows that have no `_Float64x`, and for the same reason: nothing on the
850 // machine is wider than a `double` and neither reference offers a type that is.
851 tuple::Arch::Arm | tuple::Arch::Wasm32 => false,
852 }
853}
854
855/// The granule a `_BitInt` wider than 64 bits is laid out in, in bits.
856fn bit_int_granule(target: TargetTuple) -> u32 {
857 match target.arch() {
858 // AAPCS64 says a `_BitInt` above sixty four bits is an array of `__int128`, which is the
859 // one psABI that departs from the register width here.
860 tuple::Arch::Aarch64 | tuple::Arch::Arm64Ec => 128,
861 // Everywhere else it is the width of a general purpose register, which is what the psABIs
862 // that have written the rule down all say and what both references do on the rows that
863 // have not.
864 tuple::Arch::X86 | tuple::Arch::Arm | tuple::Arch::Riscv32 => 32,
865 tuple::Arch::X86_64
866 | tuple::Arch::Riscv64
867 | tuple::Arch::LoongArch64
868 | tuple::Arch::PowerPc64
869 | tuple::Arch::S390x
870 | tuple::Arch::Wasm32 => 64,
871 }
872}
873
874/// How this target allocates bit-fields into storage.
875///
876/// Keyed on the operating system rather than the environment, because mingw's answer here is
877/// Microsoft's and not GCC's. That is the whole reason it is not a guess: a rule keyed on
878/// `Env::Msvc` gets `x86_64-windows-gnu` wrong by four bytes on a struct of an `unsigned :3` and a
879/// `char`, and gets it wrong quietly.
880fn bit_field_style(target: TargetTuple) -> BitFieldStyle {
881 match target.os() {
882 tuple::Os::Windows => BitFieldStyle::Microsoft,
883 _ => BitFieldStyle::Itanium,
884 }
885}
886
887/// Whether an unnamed bit-field raises the record's alignment the way a named one does.
888///
889/// AAPCS says it does, on both widths of ARM, and Apple and Microsoft each dropped that rule.
890/// Microsoft then put its own rule in the same place for a `struct`, so Windows says yes again by
891/// a different route, and says something else entirely for a `union`, which [`BitFieldStyle`]
892/// carries rather than this.
893fn unnamed_bit_field_aligns(target: TargetTuple) -> bool {
894 match (target.arch(), target.os()) {
895 (_, tuple::Os::Windows) => true,
896 // A freestanding ARM target is AAPCS proper, so it says yes: there is no operating system
897 // there to have dropped it.
898 (tuple::Arch::Aarch64 | tuple::Arch::Arm | tuple::Arch::Arm64Ec, os) => !os.is_darwin(),
899 _ => false,
900 }
901}
902
903/// What `__builtin_va_list` is on this target, where this crate can build the type.
904fn va_list(target: TargetTuple) -> Option<VaList> {
905 match (target.arch(), target.os()) {
906 // Windows passes every argument in one place and spills the register ones next to the
907 // stack ones, so the list is an address, and Apple does the same on AArch64.
908 (_, tuple::Os::Windows) => Some(VaList::CharPointer),
909 (tuple::Arch::Aarch64, os) if os.is_darwin() => Some(VaList::CharPointer),
910 (tuple::Arch::Aarch64, _) => Some(VaList::Aapcs),
911 // The x32 ABI's list is the same structure with four byte pointers in it, which is what
912 // building it out of this target's pointer type gives, so it is the same answer.
913 (tuple::Arch::X86_64, _) => Some(VaList::SysV),
914 (tuple::Arch::X86, _) => Some(VaList::CharPointer),
915 (tuple::Arch::Riscv64 | tuple::Arch::Riscv32 | tuple::Arch::LoongArch64, _)
916 | (tuple::Arch::Wasm32, _) => Some(VaList::VoidPointer),
917 // 32-bit ARM's is a structure of one pointer, s390x's is a structure of four members, and
918 // PowerPC's is a structure of five. None of them is any of the four types above and this
919 // crate does not build them, so it says so rather than naming a neighbour's.
920 (
921 tuple::Arch::Arm | tuple::Arch::S390x | tuple::Arch::PowerPc64 | tuple::Arch::Arm64Ec,
922 _,
923 ) => None,
924 }
925}
926
927#[cfg(test)]
928mod tests {
929 use super::*;
930
931 #[test]
932 fn parses_a_four_field_triple() {
933 let t: Triple = "x86_64-unknown-linux-gnu".parse().unwrap();
934 assert_eq!(t, Triple::new(Arch::X86_64, Os::Linux, Env::Gnu));
935 }
936
937 #[test]
938 fn parses_a_triple_with_no_vendor() {
939 let t: Triple = "aarch64-linux-musl".parse().unwrap();
940 assert_eq!(t, Triple::new(Arch::Aarch64, Os::Linux, Env::Musl));
941 }
942
943 #[test]
944 fn accepts_the_common_aliases() {
945 let a: Triple = "arm64-apple-darwin".parse().unwrap();
946 let b: Triple = "aarch64-apple-darwin".parse().unwrap();
947 assert_eq!(a, b);
948 assert_eq!(a.env, Env::None);
949 }
950
951 #[test]
952 fn fills_in_the_default_environment() {
953 let t: Triple = "x86_64-unknown-linux".parse().unwrap();
954 assert_eq!(t.env, Env::Gnu);
955 let w: Triple = "x86_64-pc-windows".parse().unwrap();
956 assert_eq!(w.env, Env::Msvc);
957 }
958
959 #[test]
960 fn rejects_what_it_does_not_support() {
961 let e = "sparc64-unknown-linux-gnu".parse::<Triple>().unwrap_err();
962 assert_eq!(e.reason, "unknown architecture");
963 let e = "x86_64-unknown-plan9".parse::<Triple>().unwrap_err();
964 assert_eq!(e.reason, "unknown operating system");
965 }
966
967 #[test]
968 fn displays_in_a_normalised_form() {
969 let t: Triple = "amd64-linux-gnu".parse().unwrap();
970 assert_eq!(t.to_string(), "x86_64-unknown-linux-gnu");
971 }
972
973 #[test]
974 fn display_round_trips_through_parse() {
975 for s in [
976 "x86_64-unknown-linux-gnu",
977 "aarch64-unknown-darwin-none",
978 "riscv64-unknown-linux-musl",
979 ] {
980 let t: Triple = s.parse().unwrap();
981 assert_eq!(t.to_string().parse::<Triple>().unwrap(), t);
982 }
983 }
984
985 #[test]
986 fn char_signedness_follows_the_psabi() {
987 let x86 = TargetInfo::new("x86_64-unknown-linux-gnu".parse().unwrap());
988 let arm = TargetInfo::new("aarch64-unknown-linux-gnu".parse().unwrap());
989 let mac = TargetInfo::new("aarch64-apple-darwin".parse().unwrap());
990 assert!(x86.char_is_signed);
991 assert!(!arm.char_is_signed);
992 assert!(mac.char_is_signed, "Apple overrides AAPCS64 back to a signed char");
993 }
994
995 #[test]
996 fn windows_is_llp64() {
997 let win = TargetInfo::new("x86_64-pc-windows-msvc".parse().unwrap());
998 assert_eq!(win.pointer_width, 64);
999 assert_eq!(win.long_width, 32);
1000 }
1001
1002 #[test]
1003 fn the_largest_object_is_ptrdiff_max() {
1004 // Half the address space less one, which is what a pointer subtraction across the whole
1005 // of one object has to fit in. gcc 16 on x86-64 prints this same number when it refuses
1006 // an array, and takes an object of exactly this many bytes.
1007 for triple in ["x86_64-unknown-linux-gnu", "aarch64-apple-darwin", "x86_64-pc-windows-msvc"]
1008 {
1009 let target = TargetInfo::new(triple.parse().unwrap());
1010 assert_eq!(target.max_object_size(), 9_223_372_036_854_775_807, "{triple}");
1011 }
1012 }
1013
1014 #[test]
1015 fn apple_long_double_is_double() {
1016 let mac = TargetInfo::new("aarch64-apple-darwin".parse().unwrap());
1017 assert_eq!(mac.long_double_width, 64);
1018 assert_eq!(mac.long_double_format, Format::Double);
1019 let linux = TargetInfo::new("x86_64-unknown-linux-gnu".parse().unwrap());
1020 assert_eq!(linux.long_double_width, 128);
1021 }
1022
1023 #[test]
1024 fn apples_x86_64_is_not_one_of_the_targets_that_narrowed_long_double() {
1025 // The bug the layout facts moving to `rucc-abi` fixed. This crate used to decide the
1026 // width from the operating system, which took both Apple targets, and Apple made the
1027 // change on AArch64 only. `facts/x86_64-macos.facts` in tamnd/rucc-cross records
1028 // `long_double_format=x87_extended` with `sizeof_long_double=16`, from a reference
1029 // compiler, and this used to answer a sixty four bit `double`.
1030 //
1031 // It is the quiet kind of wrong. `sizeof(long double)` came out at eight where the
1032 // headers say sixteen, so `printf("%Lf")` read the wrong bytes and every structure with
1033 // a `long double` in it laid out differently from the system's own.
1034 let mac = TargetInfo::new("x86_64-apple-darwin".parse().unwrap());
1035 assert_eq!(mac.long_double_width, 128);
1036 assert_eq!(mac.long_double_format, Format::X87Extended);
1037
1038 let linux = TargetInfo::new("x86_64-unknown-linux-gnu".parse().unwrap());
1039 assert_eq!(
1040 (mac.long_double_width, mac.long_double_format),
1041 (linux.long_double_width, linux.long_double_format)
1042 );
1043 }
1044
1045 #[test]
1046 fn every_triple_describes_a_machine() {
1047 // `Triple::tuple` panics on a pair that is not a machine and this is what says there is
1048 // no such pair. All forty eight combinations, including the ones the parser will produce
1049 // from a string somebody can type and no machine has, such as a Darwin target claiming
1050 // glibc.
1051 let mut built = 0;
1052 for arch in [Arch::X86_64, Arch::Aarch64, Arch::Riscv64] {
1053 for os in [Os::Linux, Os::Darwin, Os::Windows, Os::None] {
1054 for env in [Env::None, Env::Gnu, Env::Musl, Env::Msvc] {
1055 let triple = Triple::new(arch, os, env);
1056 let tuple = triple.tuple();
1057 assert_eq!(tuple.pointer_width(), 64, "{triple}");
1058 // The one field the narrowing has to preserve, because mingw and MSVC are the
1059 // same operating system with two different `long double`s.
1060 if os == Os::Windows {
1061 let expected = match env {
1062 Env::Gnu => rucc_tuple::Env::Gnu,
1063 _ => rucc_tuple::Env::Msvc,
1064 };
1065 assert_eq!(tuple.env(), expected, "{triple}");
1066 }
1067 built += 1;
1068 }
1069 }
1070 }
1071 assert_eq!(built, 48);
1072 }
1073
1074 #[test]
1075 fn from_tuple_undoes_the_narrowing() {
1076 // Every triple's tuple comes back as a triple describing the same machine. It is not
1077 // always the triple it started as, because the narrowing is many to one: a Darwin target
1078 // claiming glibc and the same one claiming nothing are one machine, and the answer is the
1079 // spelling that names no libc.
1080 for arch in [Arch::X86_64, Arch::Aarch64, Arch::Riscv64] {
1081 for os in [Os::Linux, Os::Darwin, Os::Windows, Os::None] {
1082 for env in [Env::None, Env::Gnu, Env::Musl, Env::Msvc] {
1083 let triple = Triple::new(arch, os, env);
1084 let back = Triple::from_tuple(triple.tuple())
1085 .unwrap_or_else(|| panic!("{triple} has a tuple and no way back"));
1086 assert_eq!(back.tuple(), triple.tuple(), "{triple}");
1087 assert_eq!(back.arch, arch, "{triple}");
1088 assert_eq!(back.os, os, "{triple}");
1089 }
1090 }
1091 }
1092 }
1093
1094 #[test]
1095 fn from_tuple_gives_the_canonical_environment() {
1096 let musl = Triple::from_tuple("aarch64-linux-musl".parse().unwrap()).unwrap();
1097 assert_eq!(musl, Triple::new(Arch::Aarch64, Os::Linux, Env::Musl));
1098 let gnu = Triple::from_tuple("x86_64-linux-gnu".parse().unwrap()).unwrap();
1099 assert_eq!(gnu, Triple::new(Arch::X86_64, Os::Linux, Env::Gnu));
1100 // Darwin and freestanding name no libc, so the answer does too, even though the parser
1101 // will hand this type a Darwin triple with `gnu` on the end.
1102 let macos = Triple::from_tuple("aarch64-macos".parse().unwrap()).unwrap();
1103 assert_eq!(macos, Triple::new(Arch::Aarch64, Os::Darwin, Env::None));
1104 let bare = Triple::from_tuple("riscv64-none".parse().unwrap()).unwrap();
1105 assert_eq!(bare, Triple::new(Arch::Riscv64, Os::None, Env::None));
1106 // The two Windows environments stay apart, which is the whole reason the narrowing keeps
1107 // the environment there and nowhere else.
1108 let mingw = Triple::from_tuple("x86_64-windows-gnu".parse().unwrap()).unwrap();
1109 assert_eq!(mingw.env, Env::Gnu);
1110 let msvc = Triple::from_tuple("x86_64-windows-msvc".parse().unwrap()).unwrap();
1111 assert_eq!(msvc.env, Env::Msvc);
1112 }
1113
1114 #[test]
1115 fn from_tuple_says_no_rather_than_saying_something_near() {
1116 // Twenty five of the forty two rows have no triple, and the answer is `None` rather than
1117 // a neighbour. `rucc-abi` knows the scalar layout of every one of these and this type
1118 // cannot hold any of them, which is the gap the record layout engine inherits.
1119 for tuple in [
1120 "i686-linux-gnu",
1121 "armv7-linux-gnueabihf",
1122 "s390x-linux-gnu",
1123 "powerpc64le-linux-gnu",
1124 "loongarch64-linux-gnu",
1125 "x86_64-linux-gnux32",
1126 "aarch64-linux-android",
1127 "aarch64-ios",
1128 "wasm32-wasip1",
1129 "x86_64-freebsd",
1130 ] {
1131 let target = tuple.parse().unwrap();
1132 assert_eq!(Triple::from_tuple(target), None, "{tuple}");
1133 }
1134 }
1135
1136 #[test]
1137 fn mingw_and_msvc_are_one_operating_system_with_two_long_doubles() {
1138 // The narrowing in `Triple::tuple` keeps the environment on Windows for this reason and
1139 // throws it away everywhere else. GCC's Windows targets keep the eighty bit `long double`
1140 // and Microsoft's make it a `double`, on the same processor and the same OS.
1141 let mingw = TargetInfo::new("x86_64-pc-windows-gnu".parse().unwrap());
1142 assert_eq!(mingw.long_double_width, 128);
1143 assert_eq!(mingw.long_double_format, Format::X87Extended);
1144
1145 let msvc = TargetInfo::new("x86_64-pc-windows-msvc".parse().unwrap());
1146 assert_eq!(msvc.long_double_width, 64);
1147 assert_eq!(msvc.long_double_format, Format::Double);
1148
1149 // And they agree about everything the operating system does decide.
1150 assert_eq!(mingw.long_width, msvc.long_width);
1151 assert_eq!(mingw.wchar_width, msvc.wchar_width);
1152 assert_eq!(mingw.object_format, msvc.object_format);
1153 }
1154
1155 #[test]
1156 fn wchar_t_divides_the_targets_in_two_directions_at_once() {
1157 // Windows narrows it to sixteen bits, which makes a wide string UTF-16 there and
1158 // UTF-32 everywhere else, and AArch64 Linux makes it unsigned without narrowing it.
1159 let windows = TargetInfo::new("x86_64-pc-windows-msvc".parse().unwrap());
1160 assert_eq!((windows.wchar_width, windows.wchar_is_signed), (16, false));
1161 let arm = TargetInfo::new("aarch64-unknown-linux-gnu".parse().unwrap());
1162 assert_eq!((arm.wchar_width, arm.wchar_is_signed), (32, false));
1163 let linux = TargetInfo::new("x86_64-unknown-linux-gnu".parse().unwrap());
1164 assert_eq!((linux.wchar_width, linux.wchar_is_signed), (32, true));
1165 // Apple keeps it signed on the same processor where Linux does not, in the same way it
1166 // keeps plain `char` signed there.
1167 let mac = TargetInfo::new("aarch64-apple-darwin".parse().unwrap());
1168 assert_eq!((mac.wchar_width, mac.wchar_is_signed), (32, true));
1169 }
1170
1171 #[test]
1172 fn va_list_is_the_psabis_type_and_not_one_type_with_four_spellings() {
1173 let linux = TargetInfo::new("x86_64-unknown-linux-gnu".parse().unwrap());
1174 assert_eq!(linux.va_list, Some(VaList::SysV));
1175 // x86-64 Darwin follows SysV here, and AArch64 Darwin does not follow AAPCS64.
1176 let mac = TargetInfo::new("x86_64-apple-darwin".parse().unwrap());
1177 assert_eq!(mac.va_list, Some(VaList::SysV));
1178 let arm_mac = TargetInfo::new("aarch64-apple-darwin".parse().unwrap());
1179 assert_eq!(arm_mac.va_list, Some(VaList::CharPointer));
1180 let arm = TargetInfo::new("aarch64-unknown-linux-gnu".parse().unwrap());
1181 assert_eq!(arm.va_list, Some(VaList::Aapcs));
1182 // Windows passes everything one way on both processors, so both get the simple one.
1183 let win = TargetInfo::new("x86_64-pc-windows-msvc".parse().unwrap());
1184 assert_eq!(win.va_list, Some(VaList::CharPointer));
1185 let arm_win = TargetInfo::new("aarch64-pc-windows-msvc".parse().unwrap());
1186 assert_eq!(arm_win.va_list, Some(VaList::CharPointer));
1187 let riscv = TargetInfo::new("riscv64-unknown-linux-gnu".parse().unwrap());
1188 assert_eq!(riscv.va_list, Some(VaList::VoidPointer));
1189 }
1190
1191 #[test]
1192 fn two_targets_agree_on_the_width_of_long_double_and_not_on_the_type() {
1193 // Sixteen bytes on both, and a different number in them: the x87 format has sixty four
1194 // bits of significand and quad precision has a hundred and thirteen, so a constant
1195 // converted for one is the wrong bits for the other.
1196 let x86 = TargetInfo::new("x86_64-unknown-linux-gnu".parse().unwrap());
1197 let arm = TargetInfo::new("aarch64-unknown-linux-gnu".parse().unwrap());
1198 assert_eq!(x86.long_double_width, arm.long_double_width);
1199 assert_eq!(x86.long_double_format, Format::X87Extended);
1200 assert_eq!(arm.long_double_format, Format::Quad);
1201 assert_eq!(x86.long_double_format.precision(), 64);
1202 assert_eq!(arm.long_double_format.precision(), 113);
1203 // Windows keeps the name and drops the type, the way Apple does.
1204 let windows = TargetInfo::new("x86_64-pc-windows-msvc".parse().unwrap());
1205 assert_eq!(windows.long_double_format, Format::Double);
1206 }
1207
1208 #[test]
1209 fn float64x_follows_the_processor_where_long_double_follows_the_operating_system() {
1210 // `_Float64x` is the widest format the hardware has, and no ABI takes it away the way
1211 // Apple and Windows take `long double` away. So the two fields say the same thing on
1212 // Linux and disagree everywhere else, which is the whole reason there are two of them.
1213 let x86 = TargetInfo::new("x86_64-unknown-linux-gnu".parse().unwrap());
1214 assert_eq!(x86.float64x_format, Some(Format::X87Extended));
1215 let arm = TargetInfo::new("aarch64-unknown-linux-gnu".parse().unwrap());
1216 assert_eq!(arm.float64x_format, Some(Format::Quad));
1217 let riscv = TargetInfo::new("riscv64-unknown-linux-gnu".parse().unwrap());
1218 assert_eq!(riscv.float64x_format, Some(Format::Quad));
1219
1220 let mac = TargetInfo::new("aarch64-apple-darwin".parse().unwrap());
1221 assert_eq!(mac.long_double_format, Format::Double);
1222 assert_eq!(mac.float64x_format, Some(Format::Quad));
1223 let windows = TargetInfo::new("x86_64-pc-windows-msvc".parse().unwrap());
1224 assert_eq!(windows.long_double_format, Format::Double);
1225 assert_eq!(windows.float64x_format, Some(Format::X87Extended));
1226 }
1227
1228 #[test]
1229 fn the_named_floating_types_are_not_on_every_machine() {
1230 // gcc 13, measured with the cross compilers rather than reasoned about. `_Float16` is on
1231 // three of these seven and `_Float128` is on six, and the two lists are not the same
1232 // list, which is why there are two fields.
1233 // The three field triple spells three architectures, and four of these rows are not
1234 // among them, so this asks the tuple the way the layout tests do.
1235 let of = |tuple: &str| TargetInfo::for_tuple(tuple.parse().expect("a row in the table"));
1236 let rows = [
1237 ("x86_64-linux-gnu", true, true),
1238 ("i686-linux-gnu", false, true),
1239 ("aarch64-linux-gnu", true, true),
1240 ("armv7-linux-gnueabihf", false, false),
1241 ("powerpc64le-linux-gnu", false, true),
1242 ("riscv64-linux-gnu", true, true),
1243 ("s390x-linux-gnu", false, true),
1244 ];
1245 for (tuple, float16, float128) in rows {
1246 let target = of(tuple);
1247 assert_eq!(target.has_float16, float16, "{tuple} `_Float16`");
1248 assert_eq!(target.has_float128, float128, "{tuple} `_Float128`");
1249 }
1250 // The operating system has nothing to do with it, the way it has nothing to do with
1251 // `_Float64x`, so Apple and Windows keep both types.
1252 assert!(of("aarch64-apple-darwin").has_float16);
1253 assert!(of("x86_64-pc-windows-msvc").has_float128);
1254 }
1255
1256 #[test]
1257 fn the_object_format_follows_the_operating_system() {
1258 assert_eq!(Os::Linux.object_format(), ObjectFormat::Elf);
1259 assert_eq!(Os::Darwin.object_format(), ObjectFormat::MachO);
1260 assert_eq!(Os::Windows.object_format(), ObjectFormat::Coff);
1261 }
1262
1263 #[test]
1264 fn a_target_carries_its_registers_and_says_so_when_it_has_none() {
1265 let of = |triple: &str| TargetInfo::new(triple.parse().unwrap());
1266 let linux = of("x86_64-unknown-linux-gnu");
1267 assert_eq!(linux.regs.reg_named("rdi"), Some((x86_64::GPR, x86_64::RDI)));
1268 assert_eq!(linux.call_regs.map(|regs| regs.int_args[0]), Some(x86_64::RDI));
1269 // Apple's x86-64 is SysV and Windows is the one that is not.
1270 let apple = of("x86_64-apple-darwin");
1271 assert_eq!(apple.call_regs.map(|regs| regs.int_args[0]), Some(x86_64::RDI));
1272 let windows = of("x86_64-pc-windows-msvc");
1273 assert_eq!(windows.regs.len(x86_64::GPR), 16);
1274 assert_eq!(windows.call_regs.map(|regs| regs.int_args[0]), Some(x86_64::RCX));
1275 // Not described yet, and saying nothing is the answer rather than saying x86-64's.
1276 let arm = of("aarch64-unknown-linux-gnu");
1277 assert!(arm.regs.is_empty());
1278 assert!(arm.call_regs.is_none());
1279 }
1280
1281 #[test]
1282 fn the_host_triple_is_one_we_support() {
1283 // Every host in spec/15-testing.md section 15.7 must be recognised, and CI runs on
1284 // all three, so a failure here means a host we claim support for stopped resolving.
1285 let host = Triple::host().expect("the host must be a supported target");
1286 assert_eq!(host.to_string().parse::<Triple>().unwrap(), host);
1287 }
1288}