Skip to main content

rucc_sysroot/
layout.rs

1//! The directory layout of one target's sysroot, and the cache key that names it.
2//!
3//! Design: `spec/cross-compile/08-sysroots.md` sections 8.2 and 8.3.
4//!
5//! # Why the directories are split the way they are
6//!
7//! Section 8.3 is about a multiplication. Headers are naively `arch x os x libc x libc-version`
8//! trees, which for glibc alone is eight architectures times six versions and several hundred
9//! megabytes, and `spec/cross-compile/13-distribution.md` has a size budget that number destroys.
10//!
11//! The fix is two splits that turn the product into a sum. The per version differences go inside
12//! the files as `#if __GLIBC_MINOR__ >= n`, so one tree serves every version. The per architecture
13//! differences stay in directories, because they are whole files rather than lines, but only for
14//! the small part of a libc that has any: `bits/` and a handful of others. Everything else is one
15//! copy.
16//!
17//! That is why a sysroot here has two include directories rather than one. [`Sysroot::arch_include`]
18//! holds the files that differ by architecture and is searched first, and
19//! [`Sysroot::generic_include`] holds the copy that every architecture shares.
20//!
21//! A Linux target searches four directories and not two, because the kernel's headers are a second
22//! pair with the same split and a different owner. They are [`Kernel`], their root is the cache
23//! rather than a sysroot, and the order is the libc's two and then the kernel's two, which is the
24//! order `zig cc -E -v` prints for a glibc target. `linux/` and `asm/` are nine megabytes of files
25//! that are the same for every target, so one tree is shared and only `asm/` is copied per
26//! architecture.
27//!
28//! # Why the root is a function of the tuple
29//!
30//! `spec/cross-compile/02-the-goal.md` claim 5 asks for byte identical output from different hosts.
31//! A sysroot that lands in a directory named after the host, or after the day it was built, or
32//! after a hash of an absolute path, breaks that before anything is compiled. So the root is the
33//! cache directory the caller chose plus the canonical spelling of the tuple, and nothing else.
34//!
35//! The canonical spelling is the key rather than a hash of it because it is already unique, it is
36//! already a legal directory name, and a cache a person can read is a cache a person can debug. It
37//! carries the whole ten field model, so `x86_64-linux-gnu` and `x86_64-linux-gnu.2.28` are
38//! different directories, which is the point of `env_version` being in the tuple at all.
39//!
40//! It is not the hash of the contents either, which is what
41//! `spec/cross-compile/13-distribution.md` section 13.2 asked for until tamnd/rucc#1021 settled it.
42//! A name cannot carry one: this function is what the producer calls to find out where to write
43//! files it has not written yet, so there are no contents to hash when the question is asked. The
44//! hash lives in the record instead, which is [`crate::Manifest::digest`], and the one thing the
45//! name has to be is the same on two hosts.
46
47use std::fmt;
48use std::path::{Path, PathBuf};
49
50use rucc_tuple::{Arch, DataModel, Endian, Env, Os, TargetTuple, Version};
51
52/// One target's sysroot: where its headers are, where its link inputs are, and where the record
53/// of what they are is.
54///
55/// Constructed rather than discovered. Nothing here checks that any of these directories exists,
56/// because the caller that is about to produce a sysroot needs the same answer as the caller that
57/// is about to read one, and a constructor that failed for an absent directory would give the
58/// first one nothing to create.
59#[derive(Debug, Clone, PartialEq, Eq)]
60pub struct Sysroot {
61    target: TargetTuple,
62    root: PathBuf,
63}
64
65impl Sysroot {
66    /// The sysroot for this target inside this cache directory.
67    ///
68    /// The path is `<cache>/sysroots/<canonical tuple>`. Two hosts running this with the same
69    /// cache directory and the same target get the same path, which is what makes the tuple a
70    /// cache key and is the reason `spec/cross-compile/03-target-model.md` section 3.2 admits a
71    /// field only when it changes how a call is made or a struct is laid out.
72    #[must_use]
73    pub fn in_cache(cache: &Path, target: TargetTuple) -> Self {
74        let root = cache.join("sysroots").join(target.to_canonical_string());
75        Sysroot { target, root }
76    }
77
78    /// A sysroot rooted at a directory the user named, with `--sysroot` or `-isysroot`.
79    ///
80    /// The layout below the root is the same, so a user who assembled a tree the way we lay one
81    /// out is served by every other method here. A user who did not is served by
82    /// [`Options::sysroot`](crate::Options::sysroot), which replaces step 3 of section 8.5
83    /// wholesale rather than assuming a shape.
84    #[must_use]
85    pub fn at(root: PathBuf, target: TargetTuple) -> Self {
86        Sysroot { target, root }
87    }
88
89    /// The target this sysroot is for.
90    #[must_use]
91    pub const fn target(&self) -> TargetTuple {
92        self.target
93    }
94
95    /// The directory everything else here is under.
96    #[must_use]
97    pub fn root(&self) -> &Path {
98        &self.root
99    }
100
101    /// The cache key, which is the canonical spelling of the tuple.
102    ///
103    /// The whole tuple and not a summary of it. A key that dropped `env_version` would serve a
104    /// sysroot built against glibc 2.28 to a target that pinned 2.34, and the failure would be a
105    /// missing symbol at link time on one machine and not on another.
106    #[must_use]
107    pub fn cache_key(&self) -> String {
108        self.target.to_canonical_string()
109    }
110
111    /// The headers that differ by architecture, which are searched before the generic ones.
112    ///
113    /// Section 8.3's second split. For musl this is `bits/`, which is a few dozen small files
114    /// against a few hundred shared ones, so the copy per architecture is cheap and the
115    /// alternative of one whole tree per architecture is not.
116    #[must_use]
117    pub fn arch_include(&self) -> PathBuf {
118        self.root.join("include").join(self.header_arch())
119    }
120
121    /// The headers every architecture shares, which is almost all of them.
122    #[must_use]
123    pub fn generic_include(&self) -> PathBuf {
124        self.root.join("include").join("generic")
125    }
126
127    /// The libc's include directories, in search order, most specific first.
128    ///
129    /// Two rather than four. A Linux target also needs the kernel's own headers, which are
130    /// [`Kernel`] and are not under this root, because they are the same files for every target
131    /// that shares an architecture and carrying a copy of them per tuple is nine megabytes times
132    /// the size of the table.
133    #[must_use]
134    pub fn includes(&self) -> Vec<PathBuf> {
135        vec![self.arch_include(), self.generic_include()]
136    }
137
138    /// The link inputs: the start files, the libc archive or its generated stubs, and the
139    /// compiler's own runtime for this target.
140    #[must_use]
141    pub fn lib(&self) -> PathBuf {
142        self.root.join("lib")
143    }
144
145    /// The manifest naming every input with its source, its hash and its licence.
146    ///
147    /// A file rather than a directory, and at the top rather than beside the libraries, because
148    /// the thing a person does with it is read it first.
149    #[must_use]
150    pub fn manifest_path(&self) -> PathBuf {
151        self.root.join("manifest")
152    }
153
154    /// The name the target's libc gives to its per architecture header directory.
155    ///
156    /// Not the architecture component of the canonical tuple, which carries a baseline the headers
157    /// do not care about: `armv7a-linux-musleabihf` and `armv5te-linux-musleabi` read the same
158    /// `arm` directory, because a header does not know which instructions the chip has. 32-bit x86
159    /// is `i386` in musl's source tree whatever the tuple spells it.
160    ///
161    /// # Why the libc is part of the answer
162    ///
163    /// The two libcs do not split their headers at the same place, and the name has to follow the
164    /// libc rather than a scheme of ours, because the producer installs what the libc's own build
165    /// system installs and the compiler has to look where that put it.
166    ///
167    /// musl splits per architecture and per ABI, which is what `arch/` in its source tree is, so
168    /// `x86_64`, `i386` and `x32` are three directories. glibc splits per architecture family and
169    /// handles the rest inside the files: one `x86` directory serves i386, x86-64 and x32, and 22
170    /// of the 31 files in its `bits/` branch on `__x86_64__`, `__ILP32__` or `__WORDSIZE` to do it,
171    /// starting with `bits/wordsize.h`. Checked against Zig 0.16, which ships twelve glibc
172    /// directories named after families and seventeen musl directories named after architectures.
173    ///
174    /// # The rule this is here to enforce
175    ///
176    /// An ILP32 ABI on a 64-bit architecture cannot read the LP64 headers. Every type that carries
177    /// a pointer or a `long` is a different size, and `x86_64-linux-gnux32` is the row that proves
178    /// it. For musl that is a separate directory, which is what the suffix below is. For glibc it
179    /// is a branch inside glibc's own files, so the directory is shared and the thing that checks
180    /// it is section 8.4's structural equivalence corpus rather than a path.
181    #[must_use]
182    pub fn header_arch(&self) -> &'static str {
183        if self.target.env() == Env::Gnu {
184            return self.header_family();
185        }
186        let narrow = self.target.data_model() == DataModel::Ilp32On64;
187        match (self.target.arch(), narrow) {
188            // x32 is what everyone calls it, including musl and glibc, so it does not get the
189            // suffix the rule below would give it.
190            (Arch::X86_64, true) => "x32",
191            (Arch::X86_64, false) => "x86_64",
192            (Arch::X86, _) => "i386",
193            (Arch::Aarch64 | Arch::Arm64Ec, true) => "aarch64_ilp32",
194            (Arch::Aarch64 | Arch::Arm64Ec, false) => "aarch64",
195            (Arch::Arm, _) => "arm",
196            (Arch::Riscv64, true) => "riscv64_ilp32",
197            (Arch::Riscv64, false) => "riscv64",
198            (Arch::Riscv32, _) => "riscv32",
199            (Arch::S390x, true) => "s390x_ilp32",
200            (Arch::S390x, false) => "s390x",
201            (Arch::PowerPc64, true) => "powerpc64_ilp32",
202            (Arch::PowerPc64, false) => "powerpc64",
203            (Arch::LoongArch64, true) => "loongarch64_ilp32",
204            (Arch::LoongArch64, false) => "loongarch64",
205            (Arch::Wasm32, _) => "wasm32",
206        }
207    }
208
209    /// The architecture family, which is how glibc names its per architecture header directory.
210    ///
211    /// The data model is not in it, on purpose, for the reason [`Sysroot::header_arch`] gives: the
212    /// family's files carry the branch themselves. `s390x` and `loongarch` are spelled the way
213    /// glibc's own `sysdeps` tree spells them, which is not the same shortening for both.
214    ///
215    /// # Why powerpc is the only family whose byte order is in the name
216    ///
217    /// The byte order is in the name exactly where the installed text depends on it, and that is one
218    /// family. Measured on glibc 2.44, by installing a family's headers twice with
219    /// `make install-headers` and diffing the two installs. `aarch64_be-linux-gnu` against
220    /// `aarch64-linux-gnu` is 474 files each and an empty diff, so one directory serves both orders.
221    /// `powerpc64-linux-gnu` against `powerpc64le-linux-gnu` is 474 files each and one file that
222    /// differs, `bits/long-double.h`, because little endian powerpc can redirect `long double` to
223    /// the float128 ABI and big endian powerpc cannot, so one install defines
224    /// `__LDOUBLE_REDIRECTS_TO_FLOAT128_ABI` as `(__LDBL_MANT_DIG__ == 113)` and the other defines
225    /// it as `0`. One directory for both orders would hand half of the powerpc rows a macro that is
226    /// wrong about their own ABI.
227    ///
228    /// The word size is not in the name, for powerpc either. The third run of the same experiment,
229    /// `powerpc-linux-gnu` against `powerpc64-linux-gnu` with the order held fixed, is 474 files
230    /// each and an empty diff, which is the x86 answer again: `bits/wordsize.h` is two files in
231    /// glibc's `sysdeps` tree for powerpc and they are byte identical, and both of them branch on
232    /// `__powerpc64__`. So the name is the family and the order and nothing else, which is why it is
233    /// `powerpc` and `powerpcle` rather than a spelling per width.
234    ///
235    /// `bits/endianness.h` is not the reason, which is worth saying because it reads like the
236    /// obvious one and tamnd/rucc#940 was written around it. glibc's copies of that file for arm,
237    /// aarch64 and powerpc branch on `__BIG_ENDIAN__` and `_BIG_ENDIAN` inside the file, the same
238    /// way `bits/wordsize.h` branches on `__x86_64__`, so both orders install the same text into it.
239    /// musl 1.2.5 does the same in `bits/alltypes.h` and `bits/signal.h` and ships no per order
240    /// directory under `arch/` at all, which is why [`Sysroot::header_arch`]'s musl names carry no
241    /// order either.
242    fn header_family(&self) -> &'static str {
243        match (self.target.arch(), self.target.endian()) {
244            (Arch::X86_64 | Arch::X86, _) => "x86",
245            // Arm64EC is a Windows ABI and never has glibc headers. It answers with the family it
246            // belongs to rather than with a word that is not a directory anywhere.
247            (Arch::Aarch64 | Arch::Arm64Ec, _) => "aarch64",
248            (Arch::Arm, _) => "arm",
249            (Arch::Riscv64 | Arch::Riscv32, _) => "riscv",
250            (Arch::S390x, _) => "s390x",
251            // `powerpc` is the big endian directory because that is the name glibc's own `sysdeps`
252            // tree uses and big endian is what the bare spelling means everywhere in this tuple
253            // model. `powerpcle` is the GNU spelling of the other one.
254            (Arch::PowerPc64, Endian::Big) => "powerpc",
255            (Arch::PowerPc64, Endian::Little) => "powerpcle",
256            (Arch::LoongArch64, _) => "loongarch",
257            // There is no glibc for wasm. The arm of the match exists because the type is closed
258            // and a wildcard here would quietly name a directory for a future architecture.
259            (Arch::Wasm32, _) => "wasm32",
260        }
261    }
262}
263
264/// The kernel's own headers, which are not the libc's and are shared by every target that can read
265/// them.
266///
267/// `linux/` and `asm/` are the system call interface rather than the C library, and a sysroot
268/// without them does not compile 31 of glibc's installed headers or 3 of musl's, `sys/quota.h` and
269/// `net/ethernet.h` among them. So they are part of what section 8.2 calls a sysroot even though no
270/// libc produced them.
271///
272/// # Why they are not under [`Sysroot`]
273///
274/// One tree serves every libc and every architecture except `asm/`, which is per architecture and
275/// small. Copying the shared part into each tuple's sysroot would be nine megabytes times the
276/// number of Linux rows in the table, for files that are identical in every copy. So the root is
277/// the cache directory rather than a sysroot, and a sysroot that was produced with it records the
278/// version in its manifest, which is [`crate::Manifest::kernel`] and the `kernel` line of the
279/// format. The tree carries its own record as well, headed `rucc kernel headers manifest 1`, because
280/// one tree serves every target and a sysroot manifest names one. Nothing checks the two against
281/// each other: a sysroot can be produced beside one tree and read beside another, and the manifest
282/// makes that visible rather than preventing it.
283///
284/// # Why the version is not in the path
285///
286/// The driver has to be able to compute this path before it reads anything, and a version in the
287/// path would mean asking the cache what it has before being able to ask where it is. It is the
288/// same decision [`Sysroot::in_cache`] makes about the libc version, where the tuple carries the
289/// version only because `env_version` is part of the target's identity, and the same gap: a cache
290/// populated by one release and read by the next gets whatever is there.
291///
292/// That gap does not close by putting a hash in a path, which is what
293/// `spec/cross-compile/13-distribution.md` section 13.2 used to say and what tamnd/rucc#1021
294/// settled: a path has to be known before anything has been read. What closes it is comparing a
295/// record against one somebody published, and the record says which release it was, here as the
296/// tree's own `manifest` and in a sysroot as [`crate::Manifest::kernel`].
297#[derive(Debug, Clone, PartialEq, Eq)]
298pub struct Kernel {
299    arch: &'static str,
300    root: PathBuf,
301}
302
303impl Kernel {
304    /// The kernel headers in this cache directory for this target, when the target has any.
305    ///
306    /// [`None`] for everything that is not Linux with a libc we produce a tree for. Windows, the
307    /// BSDs and Darwin have their own system headers and no `linux/` at all, freestanding has no
308    /// system call interface by definition, and Android is Linux but bionic carries its own
309    /// scrubbed copy of the uapi headers, which is a different tree from this one and not a subset
310    /// of it.
311    #[must_use]
312    pub fn for_target(cache: &Path, target: TargetTuple) -> Option<Kernel> {
313        if target.os() != Os::Linux || !matches!(target.env(), Env::Gnu | Env::Musl) {
314            return None;
315        }
316        let arch = kernel_arch(target.arch())?;
317        Some(Kernel { arch, root: cache.join("kernel-headers") })
318    }
319
320    /// The directory both of these are under.
321    #[must_use]
322    pub fn root(&self) -> &Path {
323        &self.root
324    }
325
326    /// `asm/`, which is the part of the interface that is per architecture.
327    ///
328    /// Named after the kernel's own architecture directory and not after ours or the libc's, which
329    /// is a third spelling of the same machine and the reason this is a method rather than a format
330    /// string at the call site.
331    #[must_use]
332    pub fn arch_include(&self) -> PathBuf {
333        self.root.join(self.arch)
334    }
335
336    /// `linux/`, `asm-generic/` and the rest, which are the same files for every architecture.
337    #[must_use]
338    pub fn generic_include(&self) -> PathBuf {
339        self.root.join("generic")
340    }
341
342    /// Both directories, in search order, most specific first.
343    #[must_use]
344    pub fn includes(&self) -> Vec<PathBuf> {
345        vec![self.arch_include(), self.generic_include()]
346    }
347
348    /// The kernel's name for this architecture.
349    #[must_use]
350    pub const fn arch(&self) -> &'static str {
351        self.arch
352    }
353}
354
355/// What `make headers_install ARCH=` takes, which is a third naming of the machine.
356///
357/// `arm64` rather than `aarch64` and `s390` rather than `s390x`, because those are the directories
358/// under `arch/` in the kernel's source tree, and the 31-bit s390 port leaving did not rename the
359/// one that stayed. One directory serves both widths of x86, of riscv and of powerpc, the same way
360/// glibc's family does and for the same reason: the uapi headers branch on the compiler's macros.
361///
362/// [`None`] for an architecture the kernel does not have, which is wasm.
363const fn kernel_arch(arch: Arch) -> Option<&'static str> {
364    match arch {
365        Arch::X86_64 | Arch::X86 => Some("x86"),
366        Arch::Aarch64 | Arch::Arm64Ec => Some("arm64"),
367        Arch::Arm => Some("arm"),
368        Arch::Riscv64 | Arch::Riscv32 => Some("riscv"),
369        Arch::S390x => Some("s390"),
370        Arch::PowerPc64 => Some("powerpc"),
371        Arch::LoongArch64 => Some("loongarch"),
372        Arch::Wasm32 => None,
373    }
374}
375
376/// Whether we can produce a sysroot for this target without the user fetching anything.
377///
378/// Section 8.2's table has seven rows and two of them are legal walls rather than engineering.
379/// The macOS SDK is restricted by the Xcode licence to Apple-branded hardware and the Windows SDK
380/// is not redistributable, so for those two the answer is a path the user supplies under their own
381/// licence, and `spec/cross-compile/13-distribution.md` owns the mechanism.
382///
383/// This returns false for those two and true for everything else, including freestanding, which
384/// needs nine compiler headers and no link inputs at all.
385///
386/// Which of the two walls a target is behind is [`crate::Wall`], and this is that question asked
387/// without caring about the answer. One of them is a predicate a producer filters a table with and
388/// the other is what a message has to say, and they are the same rule either way round.
389#[must_use]
390pub fn can_be_bundled(target: TargetTuple) -> bool {
391    crate::Wall::of(target).is_none()
392}
393
394/// The glibc our bundled header tree is derived from.
395///
396/// A fact about the tree and not a choice. `sysroots/manifest` in `tamnd/rucc-cross` pins the glibc
397/// source by version and hash, the tree is produced from that source, and this is that version. It
398/// moves when the pin moves and the two are checked against each other by the producer.
399pub const BUNDLED_GLIBC: Version = Version::new(2, 44);
400
401/// The `__GLIBC_MINOR__` a target gets when it is compiled against the bundled glibc tree.
402///
403/// Design: `spec/cross-compile/08-sysroots.md` section 8.3.
404///
405/// One tree serves every glibc release, with the differences written inside the files as
406/// `#if __GLIBC_MINOR__ >= n`, so the release is the part of it the target supplies. That is Zig's
407/// patch to the same tree and the same macro, which is where the spelling comes from: `features.h`
408/// keeps `__GLIBC__` at 2 and leaves the minor to the compiler, and `__GLIBC_PREREQ` reads both.
409///
410/// [`None`] for anything that is not glibc, because there is no such macro on musl or mingw and
411/// defining one would have every probe for it answer yes on a libc that does not have it.
412///
413/// The version is the one the tuple asked for, which is the point of `env_version` being in the
414/// tuple, and [`BUNDLED_GLIBC`] when it asked for nothing. Asking for an older release is how a
415/// program is kept off symbols and declarations the target's libc does not have, and it is honest
416/// only as far as the text goes: the declarations are guarded by the macro and the structure
417/// layouts in the same files are one release's. Issue #926's last box is where that is finished and
418/// it is the same direction as the compat symbol gap of #920, too permissive rather than wrong
419/// about what it does say.
420///
421/// # Errors
422///
423/// A release newer than the tree, which is the one direction that cannot be approximated. Every
424/// `__GLIBC_PREREQ` in the program would answer yes and the declarations behind them would not be
425/// there, so the failure would be a missing declaration at best and a missing symbol at link time
426/// at worst. Both versions are in the error, because the two things a person can do about it are
427/// pin a release the tree has and name a sysroot that has the one they asked for, and neither is a
428/// choice they can make without being told which release the tree is.
429pub fn bundled_glibc_minor(target: TargetTuple) -> Result<Option<u32>, GlibcSkew> {
430    if target.os() != Os::Linux || target.env() != Env::Gnu {
431        return Ok(None);
432    }
433    let Some(asked) = target.env_version() else {
434        return Ok(Some(BUNDLED_GLIBC.minor_part().unwrap_or(0)));
435    };
436    // A glibc version is two components and a tuple will hold one or three, so a request this
437    // cannot read as a glibc release is a request for the tree's own version rather than an error:
438    // `gnu.2` is somebody naming the libc and not pinning it.
439    let Some(minor) = asked.minor_part() else {
440        return Ok(Some(BUNDLED_GLIBC.minor_part().unwrap_or(0)));
441    };
442    if asked.major_part() != BUNDLED_GLIBC.major_part()
443        || minor > BUNDLED_GLIBC.minor_part().unwrap_or(0)
444    {
445        return Err(GlibcSkew { asked, tree: BUNDLED_GLIBC });
446    }
447    Ok(Some(minor))
448}
449
450/// A glibc release the bundled tree cannot serve, and the release the tree is.
451///
452/// A type rather than a pair, because the two versions read the same way round in the message as
453/// they do here and a caller that swapped them would produce a diagnostic exactly as wrong as it is
454/// convincing.
455#[derive(Debug, Clone, Copy, PartialEq, Eq)]
456pub struct GlibcSkew {
457    /// What the target asked for.
458    pub asked: Version,
459    /// What the bundled tree is, which is [`BUNDLED_GLIBC`].
460    pub tree: Version,
461}
462
463impl fmt::Display for GlibcSkew {
464    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
465        write!(
466            f,
467            "the target asked for glibc {}, and the bundled headers are glibc {}",
468            self.asked, self.tree
469        )
470    }
471}
472
473impl std::error::Error for GlibcSkew {}