1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
//! Which instruction-set rung this CPU can run, resolved once.
//!
//! trex ships every rung of its SIMD ladders in one binary and lets the
//! running CPU choose, so the build stays portable and a machine with
//! AVX-512 uses it without a separate artifact. That choice has to be made
//! somewhere, and making it at each call site costs a load and a bit test
//! per call - the entropy pass would run it once per chunk, the literal
//! search once per search.
//!
//! [`tier`] resolves the whole question once per process and hands back a
//! value every ladder can match on. `std::is_x86_feature_detected!` already
//! avoids re-issuing CPUID after its first use, so what this removes is the
//! repeated probe and the scattering of the policy, not the instruction.
//!
//! # The rung is a floor, not a menu
//!
//! [`Tier`] is ordered, and a rung implies every rung below it: a host that
//! reports [`Tier::Avx512`] can also run the AVX2 and SSE2 bodies. A ladder
//! matches from the top and falls through, so adding a rung above does not
//! disturb the ones under it.
//!
//! # What a rung promises
//!
//! Each names the exact features its bodies may use, because
//! `#[target_feature]` on a body the CPU cannot run is undefined behavior
//! and the promise here is what discharges it:
//!
//! - [`Tier::Avx512`] - `avx512f` and `avx512bw`, and everything below.
//! - [`Tier::Avx2`] - `avx2` and `bmi2`, and everything below.
//! - [`Tier::Sse2`] - `sse2`, which is architectural on x86-64.
//! - [`Tier::Scalar`] - nothing beyond the target's baseline. Every
//! non-x86-64 architecture reports this, so the scalar body is the
//! portable floor rather than a fallback that never runs.
use std::sync::OnceLock;
/// The highest ladder rung this CPU can run.
///
/// Ordered from least to most capable, so a ladder can compare rather than
/// enumerate.
#[derive(Clone, Copy, Debug, Default, PartialEq, Eq, PartialOrd, Ord)]
pub enum Tier {
/// The target's baseline instructions and nothing more.
#[default]
Scalar,
/// `sse2`. Architectural on x86-64, so every x86-64 host reaches at
/// least this.
Sse2,
/// `avx2` and `bmi2`.
Avx2,
/// `avx512f` and `avx512bw`.
Avx512,
}
impl Tier {
/// The rung's name, for a bench or a diagnostic that reports which body
/// actually ran.
#[must_use]
pub fn name(self) -> &'static str {
match self {
Tier::Scalar => "scalar",
Tier::Sse2 => "sse2",
Tier::Avx2 => "avx2",
Tier::Avx512 => "avx512",
}
}
}
/// The rung this CPU can run, probed on first call and reused after.
///
/// A ladder calls this once and matches on the result, rather than probing
/// each feature at each call site.
#[must_use]
#[inline]
pub fn tier() -> Tier {
static TIER: OnceLock<Tier> = OnceLock::new();
*TIER.get_or_init(probe)
}
#[cfg(target_arch = "x86_64")]
fn probe() -> Tier {
// Highest first: the rungs are cumulative, so the first match is the
// answer and the tests below it would all pass too.
if std::is_x86_feature_detected!("avx512f") && std::is_x86_feature_detected!("avx512bw") {
return Tier::Avx512;
}
if std::is_x86_feature_detected!("avx2") && std::is_x86_feature_detected!("bmi2") {
return Tier::Avx2;
}
if std::is_x86_feature_detected!("sse2") {
return Tier::Sse2;
}
// Unreachable on a conforming x86-64, which guarantees sse2; kept
// because the probe promises a total answer rather than assuming one.
Tier::Scalar
}
#[cfg(not(target_arch = "x86_64"))]
fn probe() -> Tier {
Tier::Scalar
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn the_tier_is_stable_across_calls() {
// A ladder reads this on a hot path and must not see it move.
let first = tier();
for _ in 0..1000 {
assert_eq!(tier(), first, "the resolved tier must not change within a process");
}
}
#[test]
fn the_rungs_are_ordered_by_capability() {
assert!(Tier::Avx512 > Tier::Avx2);
assert!(Tier::Avx2 > Tier::Sse2);
assert!(Tier::Sse2 > Tier::Scalar);
}
#[cfg(target_arch = "x86_64")]
#[test]
fn every_x86_64_host_reaches_at_least_sse2() {
// sse2 is architectural on x86-64, so a Scalar answer there means
// the probe failed rather than that the CPU is old.
assert!(tier() >= Tier::Sse2, "x86-64 guarantees sse2; probe returned {:?}", tier());
}
#[cfg(target_arch = "x86_64")]
#[test]
fn the_tier_agrees_with_a_direct_probe() {
// The cached answer must be the one the feature tests give, or the
// ladder dispatches on something other than this CPU.
let direct = if std::is_x86_feature_detected!("avx512f")
&& std::is_x86_feature_detected!("avx512bw")
{
Tier::Avx512
} else if std::is_x86_feature_detected!("avx2") && std::is_x86_feature_detected!("bmi2") {
Tier::Avx2
} else {
Tier::Sse2
};
assert_eq!(tier(), direct, "the cached tier disagrees with a direct feature probe");
eprintln!("this host resolves to the {} rung", tier().name());
}
}