rucc_codegen/select/x86_64.rs
1//! The x86-64 lowering table.
2//!
3//! Everything below the module comment is generated from `rules/x86-64.rules` by `rucc-rules`
4//! when this crate is built, and none of it is in the repository. The rule file is the only
5//! place the rules are written, which is what makes the table that is matched with and the
6//! table `rucc-verify` proves things about the same table.
7//!
8//! To read the rules, read the rule file. To read the automaton they compile into, build the
9//! crate and read `x86-64.rs` under the build directory, which is a file worth looking at once
10//! for the shape of it and never again.
11
12// A guard is emitted as the comparison the rule writes, so a rule saying a shift count is at
13// least zero and less than the width comes out as two comparisons rather than as a range. That
14// is deliberate: the generated line and the rule it came from should read the same, and the
15// suggestion to write it another way is advice for somebody editing code, which nobody here is.
16#![allow(clippy::manual_range_contains)]
17
18include!(concat!(env!("OUT_DIR"), "/x86-64.rs"));
19
20#[cfg(test)]
21mod tests {
22 use rucc_target::x86_64;
23
24 use super::TABLE;
25 use crate::select::Piece;
26
27 /// The prefix a rule file puts in front of a machine term, which is how it says which target
28 /// the term belongs to. It is not part of the opcode.
29 const PREFIX: &str = "x64.";
30
31 /// The two address constructors, which are not instructions. An addressing mode is an
32 /// argument to `lea` and to every memory operand after it, so it is written as a term in the
33 /// rule file and built by the selector into the instruction that takes it.
34 const AMODES: &[&str] =
35 &["amode_base_index_scale", "amode_index_scale", "amode_base", "amode_base_offset"];
36
37 /// Every head this table can write, in and under the replacements.
38 fn heads() -> Vec<&'static str> {
39 let mut found: Vec<&'static str> = TABLE
40 .rules
41 .iter()
42 .flat_map(|rule| rule.replacement.iter())
43 .filter_map(|piece| match piece {
44 Piece::App { head, .. } => Some(*head),
45 _ => None,
46 })
47 .collect();
48 found.sort_unstable();
49 found.dedup();
50 found
51 }
52
53 #[test]
54 fn every_instruction_the_table_writes_is_described() {
55 for head in heads() {
56 if AMODES.contains(&head) {
57 continue;
58 }
59 let opcode = head.strip_prefix(PREFIX).unwrap_or_else(|| {
60 panic!("{head} is neither an x86-64 term nor an addressing mode")
61 });
62 assert!(
63 x86_64::form(opcode).is_some(),
64 "{head} is selected by a rule and `rucc_target::x86_64` does not say what it \
65 does with its operands"
66 );
67 }
68 }
69
70 /// The order the operands of a store are written in, which is the IR's and not a choice this
71 /// file makes.
72 ///
73 /// A pattern is matched against an instruction's operand list by position, so a rule that
74 /// names the address where the IR holds the value is a rule that stores to the value and
75 /// writes the address into memory. Nothing in a proof would catch it, because a proof is
76 /// about the rule file agreeing with itself, and both halves would be wrong in the same way.
77 /// `rucc_ir::Builder::store` takes the value first and the machine instruction takes it last,
78 /// which is why the two halves of one of these rules read in opposite orders.
79 #[test]
80 fn a_store_is_written_with_the_value_first_because_that_is_where_the_ir_keeps_it() {
81 let mut seen = 0;
82 for rule in TABLE.rules {
83 let Some(rest) = rule.pattern.strip_prefix("(store.") else { continue };
84 let (width, operands) = rest.split_once(' ').expect("a store takes operands");
85 assert!(
86 operands.starts_with(&format!("(value.{width} ")),
87 "line {}: {} binds something other than the value it is storing first",
88 rule.line,
89 rule.pattern
90 );
91 assert!(
92 operands.contains("(value.i64 "),
93 "line {}: {} reaches no address",
94 rule.line,
95 rule.pattern
96 );
97 seen += 1;
98 }
99 assert_eq!(seen, 16, "the store rules moved and this test did not follow them");
100 }
101
102 /// Every comparison can be made against a constant as well as against a register.
103 ///
104 /// Four comparisons in five in the corpus are against a constant, and without a rule for one
105 /// the constant is loaded into a register first, which is an instruction and a register the
106 /// machine never needed. A missing width or a missing condition would not fail anything else:
107 /// the register rule still matches, the output is still correct, and the only sign is code
108 /// that is one instruction longer in a place nobody is looking. So the two lists are counted
109 /// against each other here.
110 ///
111 /// What this cannot check is that the condition on the immediate rule is the right one, since
112 /// both halves of a wrong pair would be a consistent pair. That is what the `spec` clause is
113 /// for, and `rucc-verify` is what reads it.
114 #[test]
115 fn a_comparison_against_a_constant_is_written_for_every_one_against_a_register() {
116 let mut against_register = Vec::new();
117 let mut against_constant = Vec::new();
118 for rule in TABLE.rules {
119 let Some(rest) = rule.pattern.strip_prefix("(icmp_") else { continue };
120 let (condition, operands) = rest.split_once(".i1 ").expect("a comparison takes two");
121 let width = operands
122 .strip_prefix("(value.")
123 .and_then(|rest| rest.split_once(' '))
124 .map(|(width, _)| width)
125 .expect("a comparison reads a value first");
126 let named = format!("{condition}.{width}");
127 if operands.contains("(iconst.") {
128 // The constant is the second operand and never the first, because a comparison is
129 // not symmetric and the same condition on the other side means the opposite.
130 assert!(
131 !operands.starts_with("(iconst."),
132 "line {}: {} compares a constant against a value",
133 rule.line,
134 rule.pattern
135 );
136 against_constant.push(named);
137 } else {
138 against_register.push(named);
139 }
140 }
141 against_register.sort_unstable();
142 against_constant.sort_unstable();
143 assert_eq!(against_register, against_constant);
144 assert_eq!(against_register.len(), 40, "ten conditions at four widths");
145 }
146
147 /// The instructions the calling convention writes rather than a rule.
148 ///
149 /// Three kinds of them. Naming the register an argument arrived in, where an argument is
150 /// depends on its position in the signature and on the classification of every argument before
151 /// it, and a rule pattern sees one term and has no way to say any of that, so `crate::abi`
152 /// builds these from the convention instead. Calling a name is the same the other way round:
153 /// what its operands are is whatever the signature made them, and a call through an address is
154 /// the same instruction with one operand more.
155 ///
156 /// The second half of a value that comes back in two registers is the third. A return of one
157 /// value is a rule, because where that value goes depends on nothing but the value, which is
158 /// exactly what a rule can say. A return of two is not, because which register the second half
159 /// is in depends on the first half: the two register files are counted separately, so a
160 /// `double` and a `long` both come back at place zero and two `long`s do not.
161 const CONVENTION: &[&str] = &[
162 "arg_val_8",
163 "arg_val_16",
164 "arg_val_32",
165 "arg_val_64",
166 "arg_val_f32",
167 "arg_val_f64",
168 "arg_val_f128",
169 "ret_val2_8",
170 "ret_val2_16",
171 "ret_val2_32",
172 "ret_val2_64",
173 "ret_val2_f32",
174 "ret_val2_f64",
175 "ret_val2_f128",
176 "call",
177 "call_reg",
178 ];
179
180 /// The instructions the block layout writes rather than a rule.
181 ///
182 /// A rule sees one branch and the layout is about the order of every block in the function, so
183 /// which arm falls through is not something any pattern could say. That answer is what decides
184 /// whether the jump goes to the arm the condition is true for or the other one, and whether
185 /// there is a second jump after it, so all of these are written where the answer is.
186 ///
187 /// The comparisons are here for a second reason on top of that one. A branch on a comparison
188 /// is a comparison and a jump on the flags it set, and the flags are not a value: no pattern
189 /// could bind one and no `spec` clause could say anything about one. So the pair is put
190 /// together by the layout, out of a comparison a rule did select and the branch behind it,
191 /// which is the same argument `rucc_target::x86_64::Form::CmpSet` is one form rather than two
192 /// under.
193 const LAYOUT: &[&str] = &[
194 "test_rr_8",
195 "cmp_rr_8",
196 "cmp_rr_16",
197 "cmp_rr_32",
198 "cmp_rr_64",
199 "cmp_ri_8",
200 "cmp_ri_16",
201 "cmp_ri_32",
202 "cmp_ri_64",
203 "cmp_rm_8",
204 "cmp_rm_16",
205 "cmp_rm_32",
206 "cmp_rm_64",
207 "jcc_e",
208 "jcc_ne",
209 "jcc_l",
210 "jcc_le",
211 "jcc_g",
212 "jcc_ge",
213 "jcc_b",
214 "jcc_be",
215 "jcc_a",
216 "jcc_ae",
217 "jmp",
218 ];
219
220 /// The instructions the compare pass writes rather than a rule.
221 ///
222 /// The other half of the argument the comparisons above are here under. A rule selects a
223 /// comparison that keeps its answer in a byte, because that is the shape a value has. What is
224 /// left of one when the machine has already made the comparison is the byte with no comparison
225 /// in front of it, and there is no pattern for that: the term it would compute is the same term
226 /// the full comparison computes, and what makes the short one right is the instruction three
227 /// places back rather than anything about the value. So `crate::compare` writes them by name,
228 /// in place of a comparison it found was already made.
229 const COMPARE: &[&str] = &[
230 "set_e", "set_ne", "set_l", "set_le", "set_g", "set_ge", "set_b", "set_be", "set_a",
231 "set_ae",
232 ];
233
234 /// The instruction a computed `goto` is written as rather than a rule.
235 ///
236 /// The one branch `crate::lower` writes by name, and the one the block layout does not write
237 /// either. What it reads is the address, which a pattern could have bound, so it is not
238 /// exempt for the reason the branches above are. What no pattern can say is the rest of it:
239 /// how many arms the block has, which is every label of the function the program took the
240 /// address of, and a rule says what an instruction reads rather than where a block goes.
241 const LABELS: &[&str] = &["jmp_reg"];
242
243 /// The instruction the memory model writes rather than a rule.
244 ///
245 /// A barrier computes nothing, so there is no equality for the solver to discharge and no
246 /// pattern for a rule to be written as. What makes it the right answer is what the machine
247 /// promises about the order two other instructions become visible in, which is a claim about
248 /// the program around it rather than about any value. `crate::lower` writes it by name, at the
249 /// strongest ordering and nowhere else, and `crate::expand` says why the strongest is the only
250 /// one that costs anything here.
251 const BARRIER: &[&str] = &["mfence"];
252
253 /// The instruction a program stops on, which `crate::lower` writes rather than a rule.
254 ///
255 /// The first half of the barrier's reason and not the second. It computes nothing, so there is
256 /// no equality for the solver and no pattern for a rule. What makes it right is not a claim
257 /// about the order anything becomes visible in either: it is what the operating system does
258 /// with the fault, which is a fact about neither the values nor the program around it.
259 const STOP: &[&str] = &["ud2"];
260
261 /// The instructions that are a hint rather than a computation.
262 ///
263 /// The same shape of exemption the barrier gets and for a reason one step further out. A
264 /// barrier computes nothing and still has to be where it is, so there is at least a claim about
265 /// the program around it. A prefetch does not even have that: a machine that drops the whole
266 /// instruction runs the program correctly, because the only thing it can change is how long the
267 /// program takes.
268 ///
269 /// So there is no equality for the solver and no pattern for a rule, and which of the four a
270 /// program gets is decided by a number in the builtin's own arguments rather than by anything
271 /// about the value being prefetched. `crate::lower` writes them by name, out of the hint the IR
272 /// carries beside the instruction.
273 const HINT: &[&str] = &["prefetch_nta", "prefetch_t0", "prefetch_t1", "prefetch_t2"];
274
275 /// The instructions nothing but an `asm` statement asks for.
276 ///
277 /// One step further out again. A prefetch is a hint and is still something the compiler decides
278 /// to write, out of a builtin the program called. These are instructions the program wrote down
279 /// itself, by name, in a template, and nothing else in the language reaches them: there is no
280 /// builtin for either, no rule could match a term that produces one, and `crate::lower` writes
281 /// them only because [`rucc_target::x86_64::read`] found the name in a template and said which
282 /// opcode that is.
283 ///
284 /// `pause` is the hint a spin lock writes between two tries at the lock. `cpuid` is how a
285 /// program asks the processor what it can do, which there is no other way to ask, so every
286 /// program that takes a faster path on some machines than on others has one of these in it.
287 ///
288 /// The alignment is the third, and it is on this list rather than one of its own because it
289 /// meets the claim below outright: an instruction is exempt for this reason exactly when there
290 /// is nothing about it for a rule to name, and an opcode with no operands and no addressing mode
291 /// has nothing. It is not an instruction at all, which is more than the test asks and is the
292 /// reason no rule could have been written for it however the rule language grew.
293 const TEMPLATE: &[&str] = &["cpuid", "pause", "align"];
294
295 /// The instructions a template asks for that are right because of the line above them.
296 ///
297 /// These are exempt for the reason the ten bytes in [`COMPARE`] are, one step further out. A
298 /// rule selects a conditional move with its comparison in front of it, because that pair is the
299 /// shape a select has. The move on its own computes the same term and what makes it right is the
300 /// comparison somewhere behind it rather than anything about its own operands, so no pattern
301 /// could say what it means. The compare pass does not write one either, because it replaces a
302 /// comparison it found was already made and there is no earlier move here to replace: what
303 /// writes one is a program that put the comparison on one line of a template and the move on the
304 /// next, which is what zstd does to keep a bounds check from becoming a branch.
305 ///
306 /// So these have operands a rule could have named, unlike everything in [`TEMPLATE`], and they
307 /// are still not instructions a rule could have been written for.
308 const CONDITIONAL: &[&str] = &[
309 "cmov_e_16",
310 "cmov_e_32",
311 "cmov_e_64",
312 "cmov_ne_16",
313 "cmov_ne_32",
314 "cmov_ne_64",
315 "cmov_l_16",
316 "cmov_l_32",
317 "cmov_l_64",
318 "cmov_le_16",
319 "cmov_le_32",
320 "cmov_le_64",
321 "cmov_g_16",
322 "cmov_g_32",
323 "cmov_g_64",
324 "cmov_ge_16",
325 "cmov_ge_32",
326 "cmov_ge_64",
327 "cmov_b_16",
328 "cmov_b_32",
329 "cmov_b_64",
330 "cmov_be_16",
331 "cmov_be_32",
332 "cmov_be_64",
333 "cmov_a_16",
334 "cmov_a_32",
335 "cmov_a_64",
336 "cmov_ae_16",
337 "cmov_ae_32",
338 "cmov_ae_64",
339 ];
340
341 /// The instructions that look for a set bit, which a template asks for and nothing else does.
342 ///
343 /// These have a source and a destination a rule could have named, the way the conditional moves
344 /// above do, and the reason no rule names them is a different one again. It is not that their
345 /// meaning comes from the line in front of them: each of these says on its own exactly what it
346 /// computes. It is that [`crate::expand`] already answers the question they answer, out of
347 /// arithmetic every machine has, and it does that because what these do when the source is zero
348 /// is four different things on four families of processor. A rule that selected one would be a
349 /// rule whose answer depends on which machine ran it.
350 ///
351 /// So the only thing that reaches one is a program that wrote the name in a template, which is
352 /// what the libraries that were counting bits before there was a builtin for it all do.
353 /// `crate::lower` writes them for the reason it writes the three in [`TEMPLATE`], and they are
354 /// not on that list because they are not bare: a rule could have named these operands and the
355 /// claim that list makes would be false of them.
356 const SEARCH: &[&str] = &[
357 "bsf_16", "bsf_32", "bsf_64", "bsr_16", "bsr_32", "bsr_64", "lzcnt_32", "lzcnt_64",
358 "tzcnt_32", "tzcnt_64",
359 ];
360
361 /// The instruction that turns a register round, which a template asks for and nothing else does.
362 ///
363 /// The list above, one step simpler. A search is unselected because what it does with a source
364 /// of zero is not the same on every processor, so a rule that chose one would depend on what ran
365 /// it. A byte reversal has no such case: it means exactly one thing everywhere. What keeps it
366 /// off the rule set is a choice made once, in [`crate::expand`], which builds a reversal out of
367 /// shifts and masks so that the answer is the same on every target this compiler has rather than
368 /// good on the one that happens to have the instruction. tamnd/rucc#310 is where that trade is
369 /// written down, and the day a target grows its own reversal is the day to reopen it.
370 ///
371 /// So the only thing that reaches one is a program that wrote the name in a template, which is
372 /// what libgmp does in `gmp-impl.h` to put a limb the other way round.
373 const SWAP: &[&str] = &["bswap_32", "bswap_64"];
374
375 /// The multiply that keeps both halves of its product and the division that reads both halves
376 /// of its dividend, which a template asks for and nothing else does.
377 ///
378 /// A third reason again, and the plainest of the three. A search is unselected because its
379 /// answer depends on the processor and a reversal because a choice was made to build one out of
380 /// arithmetic. This one is unselected because there is nothing in the IR to select it from: a
381 /// multiply in C takes two values of a type and produces a value of that type, so the term a
382 /// rule would match on is the narrow product, and the wide product is not a term at all. A rule
383 /// that fired on the narrow one and wrote this would be writing an instruction that computes
384 /// twice as much as was asked for and leaves the rest in a register nobody asked about.
385 ///
386 /// So the only thing that reaches one is a program that wrote the name in a template, which is
387 /// what `umul_ppmm` in libgmp's `longlong.h` does, and what every library that is building
388 /// arithmetic out of limbs does somewhere.
389 ///
390 /// The division is the same claim upside down and is on this list because the reason is the same
391 /// one. A division in C divides a number by a number of its own width, so the term a rule would
392 /// match is the narrow one, and this compiler already has two opcodes for that: each of them
393 /// fills the high half of the dividend itself and then throws one of the two answers away. A
394 /// dividend the program filled both halves of is not a term the IR has, and `udiv_qrnnd` beside
395 /// the multiply in the same header is how long division a limb at a time is written.
396 const WIDE: &[&str] = &[
397 "mul_wide_16",
398 "mul_wide_32",
399 "mul_wide_64",
400 "imul_wide_16",
401 "imul_wide_32",
402 "imul_wide_64",
403 "div_wide_16",
404 "div_wide_32",
405 "div_wide_64",
406 "idiv_wide_16",
407 "idiv_wide_32",
408 "idiv_wide_64",
409 ];
410
411 /// The instructions that produce two values, which is one more than a rule can name.
412 ///
413 /// A rule replaces a term with a term, and a term is the value one instruction computes. A
414 /// compare and exchange computes two: what it found at the address, and whether what it found
415 /// was what the program expected. There is no way to write the second one down in the rule
416 /// language, and inventing one would be inventing a language for a single instruction.
417 ///
418 /// So `crate::lower` writes it by name, the way it writes the barrier by name, and for a reason
419 /// that is about the rule language rather than about the machine. What the solver would have
420 /// been asked to prove about it is the easy half in any case: the arithmetic is a comparison
421 /// and a select, and what is hard is that the whole of it happens at once, which is the same
422 /// claim about the program around it that a barrier makes.
423 const ATOMIC: &[&str] = &["cmpxchg_8", "cmpxchg_16", "cmpxchg_32", "cmpxchg_64"];
424
425 /// The instructions whose operation is in the payload rather than in the head.
426 ///
427 /// A different exemption from the one above, on instructions that produce one value each and so
428 /// could be named by a rule if the rule had anything to match on. The head a pattern matches is
429 /// an opcode and a type, and every read modify write in the IR is the one opcode `atomic_rmw`.
430 /// Which of the thirteen operations it performs is carried beside the instruction rather than in
431 /// its name, so a pattern written for the exchange would match the add and the nand as well, and
432 /// the rule language has no way to look at what a rule matched to tell them apart.
433 ///
434 /// Giving each operation its own opcode is the other way out and is a worse trade: it is
435 /// thirteen opcodes at four widths where the IR wants one, and every pass that treats a read
436 /// modify write as one thing would then have a list of fifty two.
437 ///
438 /// So `crate::lower` writes these by name too. Three operations here, out of the thirteen: the
439 /// bitwise ones need a loop around a compare and exchange, which is control flow and so is built
440 /// before selection rather than during it, and they are the rest of `tamnd/rucc#311`.
441 const PAYLOAD: &[&str] =
442 &["xchg_8", "xchg_16", "xchg_32", "xchg_64", "xadd_8", "xadd_16", "xadd_32", "xadd_64"];
443
444 /// The instructions a frame writes rather than a rule.
445 ///
446 /// A prologue, an epilogue, a copy, a spill and a reload are not in the program. They are what
447 /// the allocator's answer costs, so they are written after it, by `crate::finish` reading
448 /// `x86_64::FRAME`. Six of the names that describes are already reachable from a rule, since a
449 /// prologue taking its frame is a subtraction and a spill is a store, and those are not here:
450 /// this is only the ones nothing else can reach.
451 const FRAME: &[&str] = &[
452 "push_64",
453 "pop_64",
454 "ret",
455 "mov_rr_64",
456 "movaps_rr",
457 // The touch a probing prologue puts on each page as it reaches it, the landing pad a
458 // prologue opens with, and the byte that does nothing which one reserves room with. All
459 // three are written by a frame and none on a command line that did not ask for it.
460 "or_mi_8",
461 "endbr64",
462 "nop",
463 ];
464
465 /// The instructions that reach the x87 stack, which are selected but not from here.
466 ///
467 /// A third kind of exemption, and the same reason all the way down the list.
468 ///
469 /// Every one of these is written by `crate::lower`, as part of a group rather than on its own.
470 /// What one of them leaves behind and the next picks up is the top of the x87 stack, which is
471 /// not a register anything allocates from and not a value a pattern could bind, so a rule
472 /// could neither match the middle of a group nor name what its replacement produced. And an
473 /// add here reads two addresses and writes a third, where one machine IR instruction carries
474 /// one addressing mode, so the group cannot be folded into a single opcode the way
475 /// `ucomisd_set_e` folds a comparison and a `setcc` either.
476 ///
477 /// So these are exempt for the reason `FRAME` is exempt rather than for the reason the list
478 /// below is, and they will stay exempt. Two of them are not reached by anything yet all the
479 /// same: `fsub_p` and `fdiv_p` are the other direction of the subtraction and the division,
480 /// which a code generator that pushed its operands the other way round would need and this one
481 /// does not. `fabs` is a third, since C spells that as a call to a library function.
482 const X87: &[&str] = &[
483 "fld_t",
484 "fstp_t",
485 "fld_s",
486 "fld_l",
487 "fild_l",
488 "fild_ll",
489 "fstp_s",
490 "fstp_l",
491 "fistp_l",
492 "fistp_ll",
493 "fnstcw",
494 "fldcw",
495 "fadd_p",
496 "fsub_p",
497 "fsubr_p",
498 "fmul_p",
499 "fdiv_p",
500 "fdivr_p",
501 "fchs",
502 "fabs",
503 "fucomip_set_a",
504 "fucomip_set_ae",
505 "fucomip_set_b",
506 "fucomip_set_be",
507 "fucomip_set_e",
508 "fucomip_set_ne",
509 "fucomip_set_p",
510 "fucomip_set_np",
511 "fucomip_set_e_and_np",
512 "fucomip_set_ne_or_p",
513 ];
514
515 /// The instructions no rule selects yet, because the rules that selected them were taken out.
516 ///
517 /// A different kind of exemption from the three above. Those say an instruction is written
518 /// somewhere a rule cannot reach and always will be. These say nobody reaches one at all right
519 /// now, and name the work that puts the rules back.
520 ///
521 /// The rules went out under `tamnd/rucc#368`. C promotes the operands of an arithmetic
522 /// operator to `int`, so a byte add and a two byte compare are things no C program asks the
523 /// back end for, and the rules at those widths sat proved and never selected over the whole
524 /// torture corpus at every optimization level. The width narrowing pass in `tamnd/rucc#375` is
525 /// what asks for them, and the rules come back with it.
526 ///
527 /// The descriptions stayed. A description says what an x86-64 instruction is, how long it is
528 /// and how it encodes, and that is true whether or not anything selects it. Taking them out
529 /// would be deleting a correct account of the machine to make a list shorter, and putting them
530 /// back is then a second thing to get right rather than a line of a rule file.
531 const NARROW: &[&str] = &[
532 // Three of the two address forms against an immediate. The `narrow` pass does write the
533 // shape, since `char c = a | 1;` narrows to a byte `or` against a byte constant, and no
534 // rule selects these yet: the constant goes into a register and the register with
535 // register rule takes it. Their `add`, `sub` and `and` siblings do have rules and are
536 // reached by the bitfield lowering, so this is six rules missing rather than a shape
537 // nothing writes.
538 "or_ri_8",
539 "or_ri_16",
540 "xor_ri_8",
541 "xor_ri_16",
542 "imul_ri_8",
543 "imul_ri_16",
544 // The divides, which are four instructions per width because the quotient and the
545 // remainder come out of one division in two different registers. `narrow` refuses these
546 // on purpose: the most negative byte over minus one is a defined hundred and twenty eight
547 // at four bytes and is the overflow that raises at one, so narrowing a division wants a
548 // range that rules the pair out and there is no range analysis yet.
549 "idiv_quo_8",
550 "idiv_quo_16",
551 "idiv_rem_8",
552 "idiv_rem_16",
553 "div_quo_8",
554 "div_quo_16",
555 "div_rem_8",
556 "div_rem_16",
557 // The shifts by a value, whose count is in `cl` whatever the width being shifted is. The
558 // same refusal for the same kind of reason: a count of twenty is a defined shift to zero
559 // at four bytes and is poison at one, so only a count that is a constant below the narrow
560 // width narrows, and that one selects the immediate forms which do have rules.
561 "shl_rcl_8",
562 "shl_rcl_16",
563 "shr_rcl_8",
564 "shr_rcl_16",
565 "sar_rcl_8",
566 "sar_rcl_16",
567 ];
568
569 /// The arithmetic that reaches memory, which [`crate::combine`] writes: the forms that read a
570 /// source out of it and the forms that leave the answer in it.
571 ///
572 /// A function rather than a list, for the reason the compare pass's exemption is taken from the
573 /// flag description rather than typed out: the pass already writes down which instructions it
574 /// can produce, and a second copy of that here would be a second opinion about one pass.
575 ///
576 /// No rule selects one of these because a rule matches a term and one of these is two terms, a
577 /// load and an arithmetic operation, put together, or three where the answer goes back to
578 /// memory. Whether they may be put together depends on what is written between them and on
579 /// whether anything else wants what the load read, and neither is a fact about any of the
580 /// terms. That is the whole reason the pass exists and the module documentation there says it
581 /// at length.
582 fn combine() -> Vec<&'static str> {
583 let loads = crate::combine::FOLDS.iter().map(|fold| fold.into);
584 // And the instruction a load on the other side comes to, which for most rows is the one
585 // above and for a comparison is the condition the other way round.
586 let swapped = crate::combine::FOLDS.iter().filter_map(|fold| fold.swapped);
587 let stores = crate::combine::UPDATES.iter().map(|update| update.into);
588 let constants = crate::combine::BUMPS.iter().map(|bump| bump.into);
589 loads.chain(swapped).chain(stores).chain(constants).collect()
590 }
591
592 #[test]
593 fn every_instruction_exempt_from_a_rule_is_one_a_frame_really_writes() {
594 // The same claim as the one about the convention, so that this list cannot grow an opcode
595 // that no frame asks for. In the order `x86_64::FRAME` names them, the copies after the
596 // return because there is one set of them per class the allocator may spill.
597 let frame = &x86_64::FRAME;
598 let mut written = vec![frame.push, frame.pop, frame.ret];
599 for class in frame.classes {
600 written.extend([class.mov, class.load, class.store]);
601 }
602 // And the touch a probing prologue puts on a page, which the target names as an option
603 // because a target with no instruction that writes an address without changing it takes
604 // every frame in one subtraction and has nothing to exempt.
605 written.extend(frame.probe.map(|probe| probe.inst));
606 // And the landing pad and the byte that does nothing, which are options for the same
607 // reason.
608 written.extend(frame.landing);
609 written.extend(frame.pad);
610 // What is left after the ones a rule already reaches, which are the loads and the stores of
611 // both register files, since those are the same instructions a program's own reads and
612 // writes of memory are. The vector pair joined them with the rules for a quad float, and a
613 // spill of one is now the same instruction as a program reading a `_Float128` variable.
614 written.retain(|opcode| !heads().contains(&format!("{PREFIX}{opcode}").as_str()));
615 assert_eq!(written, FRAME);
616 }
617
618 #[test]
619 fn every_instruction_exempt_from_a_rule_is_one_the_convention_really_writes() {
620 // An exemption list that nothing checks is a hole, since an opcode dropped into it stops
621 // being covered by either direction of the pinning. These are the ones `crate::abi` can
622 // name, at the four integer widths and the two float formats it has names for an
623 // argument in, and no others.
624 let strip = |head: &'static str| head.strip_prefix(PREFIX).expect("an x86-64 term");
625 let named = |ty| strip(crate::abi::head_of(ty).expect("every width the pseudos cover"));
626 // The second half of a pair at place one, which is the place a rule cannot name. The first
627 // half at place zero is `ret_val_*` and is reached by a rule, so it is not on this list.
628 let second = |ty| strip(crate::abi::ret_of(ty, 1).expect("every width the pseudos cover"));
629 let widths = || {
630 [8, 16, 32, 64].into_iter().map(rucc_ir::Type::int).chain(
631 [rucc_ir::Float::F32, rucc_ir::Float::F64, rucc_ir::Float::F128]
632 .map(rucc_ir::Type::float),
633 )
634 };
635 let written: Vec<&str> = widths()
636 .map(named)
637 .chain(widths().map(second))
638 .chain([strip(crate::abi::CALL), strip(crate::abi::CALL_REG)])
639 .collect();
640 assert_eq!(written, CONVENTION);
641 }
642
643 /// The same claim about the block layout's list, which is longer than it looks.
644 ///
645 /// A name here that the layout does not write is an opcode exempted from needing a rule and
646 /// reached by nothing, and a name the layout writes that is not here is a failing test in
647 /// `every_described_instruction_is_reachable_from_a_rule` with a misleading message. Both are
648 /// avoided by taking the list from `rucc_target::x86_64::BRANCH` rather than believing it.
649 #[test]
650 fn every_instruction_exempt_from_a_rule_is_one_the_block_layout_really_writes() {
651 let branch = &x86_64::BRANCH;
652 // Eighty entries name sixteen instructions between them, so this is a set rather than a
653 // list and both sides are sorted before they are held against each other. What the order
654 // of the list itself is for is reading it.
655 let mut written: Vec<&str> = vec![branch.test, branch.jump];
656 written.extend(branch.fused.iter().map(|fusion| fusion.cmp));
657 written.extend(branch.fused.iter().flat_map(|fusion| [fusion.if_true, fusion.if_false]));
658 written.sort_unstable();
659 written.dedup();
660 let mut exempt = LAYOUT.to_vec();
661 exempt.sort_unstable();
662 assert_eq!(written, exempt);
663 }
664
665 /// The same claim about the compare pass. What it writes is what the flag description says is
666 /// left of a comparison, so the exemption is taken from that rather than typed out twice, and
667 /// an entry added there without a rule to go with it shows up here rather than in a build that
668 /// fails somewhere else.
669 #[test]
670 fn every_instruction_exempt_from_a_rule_is_one_the_compare_pass_really_writes() {
671 let mut written: Vec<&str> =
672 x86_64::FLAGS.compares.iter().filter_map(|entry| entry.kept).collect();
673 written.sort_unstable();
674 written.dedup();
675 let mut exempt = COMPARE.to_vec();
676 exempt.sort_unstable();
677 assert_eq!(written, exempt);
678 }
679
680 /// And the same claim about the one the lowering writes, held against the name the target gave
681 /// it rather than against the spelling written above.
682 #[test]
683 fn the_instruction_a_computed_goto_is_exempt_for_is_the_one_the_target_names() {
684 assert_eq!(LABELS, [x86_64::BRANCH.indirect]);
685 }
686
687 /// The rows of the constant table that take nothing yet are exactly the narrow ones waiting on
688 /// the width narrowing, so the day `NARROW` shrinks is the day this says so.
689 ///
690 /// `crate::combine::BUMPS` has a row per instruction this machine has, which is the whole five
691 /// operations at the whole four widths. Four of those instructions arrive out of a rule that is
692 /// not written yet, so four of the rows sit there taking nothing. That is a fact worth holding
693 /// rather than a thing to notice again later.
694 #[test]
695 fn the_constant_runs_that_take_nothing_are_the_ones_no_rule_selects_yet() {
696 let written = heads();
697 let mut waiting = Vec::new();
698 for bump in crate::combine::BUMPS {
699 if !written.contains(&format!("{PREFIX}{}", bump.from).as_str()) {
700 waiting.push(bump.from);
701 }
702 }
703 assert_eq!(waiting, ["or_ri_8", "or_ri_16", "xor_ri_8", "xor_ri_16"]);
704 for from in waiting {
705 assert!(NARROW.contains(&from), "{from} is unselected and is not on the list");
706 }
707 }
708
709 #[test]
710 fn every_described_instruction_is_reachable_from_a_rule() {
711 let written = heads();
712 let combine = combine();
713 for &(opcode, _) in x86_64::INSTS {
714 if combine.contains(&opcode) {
715 continue;
716 }
717 if CONVENTION.contains(&opcode) || LAYOUT.contains(&opcode) || FRAME.contains(&opcode) {
718 continue;
719 }
720 if NARROW.contains(&opcode) || BARRIER.contains(&opcode) || X87.contains(&opcode) {
721 continue;
722 }
723 if ATOMIC.contains(&opcode) || PAYLOAD.contains(&opcode) || HINT.contains(&opcode) {
724 continue;
725 }
726 if CONDITIONAL.contains(&opcode) {
727 continue;
728 }
729 if COMPARE.contains(&opcode) || TEMPLATE.contains(&opcode) {
730 continue;
731 }
732 if SEARCH.contains(&opcode) || SWAP.contains(&opcode) || WIDE.contains(&opcode) {
733 continue;
734 }
735 if LABELS.contains(&opcode) || STOP.contains(&opcode) {
736 continue;
737 }
738 let head = format!("{PREFIX}{opcode}");
739 assert!(
740 written.contains(&head.as_str()),
741 "{opcode} is described and no rule in {} selects it",
742 TABLE.source
743 );
744 }
745 }
746
747 /// The same claim about the barrier as the ones above make about the convention and the frame:
748 /// the list holds instructions this target really describes, and holds only the ones that have
749 /// no operands, since an instruction with an operand is one a rule could have been written for.
750 #[test]
751 fn every_instruction_exempt_from_a_rule_is_one_the_memory_model_really_writes() {
752 for &opcode in BARRIER {
753 let form = x86_64::form(opcode).expect("an instruction this target describes");
754 assert!(form.operands().is_empty(), "{opcode} has operands, so a rule could name it");
755 }
756 }
757
758 /// The same claim about the instruction a program stops on, which is the barrier's shape
759 /// exactly: no operands, because an instruction with one is an instruction a rule could have
760 /// been written for, and no addressing mode either, because it is given nothing at all.
761 #[test]
762 fn the_instruction_exempt_from_a_rule_because_it_stops_the_program_is_bare() {
763 for &opcode in STOP {
764 let form = x86_64::form(opcode).expect("an instruction this target describes");
765 assert!(form.operands().is_empty(), "{opcode} has operands, so a rule could name it");
766 assert!(!form.takes_mem(), "{opcode} is given an address and stopping needs none");
767 }
768 }
769
770 /// The same claim about the hints, with the one difference between them written down. A hint is
771 /// given an address and nothing else, so it has no operands for the reason a barrier has none
772 /// and it does carry an addressing mode, which is what a rule would have had to match on.
773 #[test]
774 fn every_instruction_exempt_from_a_rule_because_it_is_a_hint_is_given_only_an_address() {
775 for &opcode in HINT {
776 let form = x86_64::form(opcode).expect("an instruction this target describes");
777 assert!(form.operands().is_empty(), "{opcode} has operands, so a rule could name it");
778 assert!(form.takes_mem(), "{opcode} is a hint about an address and is given none");
779 }
780 }
781
782 /// The same claim about the template list. An instruction is exempt for this reason exactly
783 /// when there is nothing about it for a rule to name, and there are two ways to have nothing.
784 /// No operands and no address, which is the hint. Or every operand fixed to one register by the
785 /// description, which is the question put to the processor: a rule names the operands of a term
786 /// and binds them to the values underneath it, and an operand that can be nothing but `rax` is
787 /// not a place a value goes. Either way the whole of the claim holds, which is that there was
788 /// nowhere else for the instruction to come from.
789 #[test]
790 fn every_instruction_exempt_from_a_rule_because_only_a_template_asks_for_it_is_bare() {
791 for &opcode in TEMPLATE {
792 let form = x86_64::form(opcode).expect("an instruction this target describes");
793 let fixed = form
794 .operands()
795 .iter()
796 .all(|desc| matches!(desc.constraint, rucc_target::Constraint::Fixed(_)));
797 assert!(fixed, "{opcode} has an operand a rule could name");
798 assert!(!form.takes_mem(), "{opcode} is given an address, so a rule could name it");
799 }
800 }
801
802 /// The same claim about the bit searches, read off the description that put them there and read
803 /// both ways round. An instruction is exempt for this reason exactly when the machine describes
804 /// it as a search, so the list cannot grow an opcode that is something else, and a search this
805 /// target grows later cannot be left off the list and quietly go unselected with nobody saying
806 /// why. Nothing in the rule set selects one, which is the other half of the reason and is what
807 /// the check above would have caught in any case.
808 #[test]
809 fn every_instruction_exempt_from_a_rule_because_only_a_template_searches_for_a_bit_is_one() {
810 let written = heads();
811 for &opcode in SEARCH {
812 let form = x86_64::form(opcode).expect("an instruction this target describes");
813 assert_eq!(form, x86_64::Form::Search, "{opcode} is not a search");
814 assert!(
815 !written.contains(&format!("{PREFIX}{opcode}").as_str()),
816 "a rule in {} selects {opcode}, which only a template asks for",
817 TABLE.source
818 );
819 }
820 for &(opcode, form) in x86_64::INSTS {
821 if form == x86_64::Form::Search {
822 assert!(SEARCH.contains(&opcode), "{opcode} is a search and is not on the list");
823 }
824 }
825 }
826
827 /// The same claim about the byte reversal, read both ways round the way the searches are, and
828 /// with the one thing that is different about it checked as well: this is the instruction of its
829 /// shape that leaves the condition state alone, which is the whole reason it has a form rather
830 /// than being a unary operation, so a description that stopped saying that would stop being the
831 /// reason this list exists.
832 #[test]
833 fn every_instruction_exempt_from_a_rule_because_only_a_template_turns_a_register_round_is_one()
834 {
835 let written = heads();
836 for &opcode in SWAP {
837 let form = x86_64::form(opcode).expect("an instruction this target describes");
838 assert_eq!(form, x86_64::Form::Swap, "{opcode} is not a byte reversal");
839 assert!(
840 !(x86_64::FLAGS.writes)(opcode),
841 "{opcode} writes the condition state, so it is a unary operation after all"
842 );
843 assert!(
844 !written.contains(&format!("{PREFIX}{opcode}").as_str()),
845 "a rule in {} selects {opcode}, which only a template asks for",
846 TABLE.source
847 );
848 }
849 for &(opcode, form) in x86_64::INSTS {
850 if form == x86_64::Form::Swap {
851 assert!(SWAP.contains(&opcode), "{opcode} is a reversal and is not on the list");
852 }
853 }
854 }
855
856 /// The same claim about the two that work on a pair of registers, read both ways round and with
857 /// the thing that puts them out of reach of a rule checked rather than asserted in prose: each
858 /// writes two registers, and a rule replaces a term with a term, so there is no way to say the
859 /// second answer in the rule language at all. That is the same bar the compare and exchange is
860 /// exempt at, and this list is separate from that one because the reason it is nobody's to select
861 /// is different: an atomic is written by name where it is needed, and nothing in this compiler
862 /// needs one of these.
863 #[test]
864 fn every_instruction_exempt_from_a_rule_because_only_a_template_wants_both_halves_writes_two() {
865 let written = heads();
866 let both = [x86_64::Form::MulWide, x86_64::Form::DivWide];
867 for &opcode in WIDE {
868 let form = x86_64::form(opcode).expect("an instruction this target describes");
869 assert!(both.contains(&form), "{opcode} works on one register rather than on a pair");
870 let defs = form.operands().iter().filter(|desc| desc.role.is_def()).count();
871 assert_eq!(defs, 2, "{opcode} writes {defs} registers and a pair takes two");
872 assert!(
873 !written.contains(&format!("{PREFIX}{opcode}").as_str()),
874 "a rule in {} selects {opcode}, which only a template asks for",
875 TABLE.source
876 );
877 }
878 for &(opcode, form) in x86_64::INSTS {
879 if both.contains(&form) {
880 assert!(WIDE.contains(&opcode), "{opcode} works on a pair and is not on the list");
881 }
882 }
883 }
884
885 /// The same claim about the conditional moves, read off the flag description the way the compare
886 /// pass's list is taken from it rather than typed out twice. An instruction is exempt for this
887 /// reason exactly when it reads the condition state and leaves it as it found it, which is what
888 /// says the instruction in front of it is where its meaning comes from. One that wrote the state
889 /// as well would be one a pattern could match on its own.
890 #[test]
891 fn every_instruction_exempt_from_a_rule_because_a_comparison_gives_it_its_meaning_reads_one() {
892 for &opcode in CONDITIONAL {
893 x86_64::form(opcode).expect("an instruction this target describes");
894 assert!(
895 x86_64::FLAGS.reads(opcode).is_some(),
896 "{opcode} reads no comparison, so a rule could name it"
897 );
898 assert!(
899 !(x86_64::FLAGS.writes)(opcode),
900 "{opcode} writes the condition state, so a rule could name it"
901 );
902 }
903 }
904
905 /// The same claim about the atomic list, read off the thing that put the entry there: an
906 /// instruction is exempt for this reason exactly when it writes more than one value, and an
907 /// instruction that writes one is one a rule could have been written for.
908 #[test]
909 fn every_instruction_exempt_from_a_rule_is_one_that_writes_more_than_one_value() {
910 let written = heads();
911 for &opcode in ATOMIC {
912 let form = x86_64::form(opcode).expect("an instruction this target describes");
913 let writes = form.operands().iter().filter(|desc| desc.role.is_def()).count();
914 assert!(writes > 1, "{opcode} writes one value, so a rule could name it");
915 assert!(
916 !written.contains(&format!("{PREFIX}{opcode}").as_str()),
917 "a rule in {} selects {opcode}, which `crate::lower` also writes by hand",
918 TABLE.source
919 );
920 }
921 }
922
923 /// The same claim about the payload list, read off the thing that puts an entry there.
924 ///
925 /// Two halves. Each of these writes one value, which is what says the reason above is not the
926 /// reason here, so a list that grew to cover an instruction the atomic list should have had
927 /// fails. And there really is more than one operation behind the one IR opcode, which is the
928 /// whole of why a pattern cannot name any of them, and is a fact about the IR that would stop
929 /// being true if the operations were ever given opcodes of their own.
930 #[test]
931 fn every_instruction_exempt_because_its_operation_is_beside_it_writes_one_value() {
932 assert!(
933 rucc_ir::RmwOp::all().count() > 1,
934 "one operation per opcode would be a head a rule could match"
935 );
936 let written = heads();
937 for &opcode in PAYLOAD {
938 let form = x86_64::form(opcode).expect("an instruction this target describes");
939 let writes = form.operands().iter().filter(|desc| desc.role.is_def()).count();
940 assert_eq!(writes, 1, "{opcode} writes more than one value, so it is the other list's");
941 assert!(
942 !written.contains(&format!("{PREFIX}{opcode}").as_str()),
943 "a rule in {} selects {opcode}, which `crate::lower` also writes by hand",
944 TABLE.source
945 );
946 }
947 }
948
949 /// The staleness rule every list in this project is kept under, on the one list here whose
950 /// entries are meant to leave. A rule that starts selecting one of these is `tamnd/rucc#375`
951 /// arriving, and the entry goes with it. An entry naming an instruction nothing describes is a
952 /// misspelling, and it would sit here exempting nothing.
953 #[test]
954 fn an_instruction_a_rule_now_selects_is_off_the_list_of_the_ones_left_for_later() {
955 let written = heads();
956 for &opcode in NARROW {
957 let head = format!("{PREFIX}{opcode}");
958 assert!(
959 !written.contains(&head.as_str()),
960 "a rule in {} selects {opcode} now, so it is not waiting on tamnd/rucc#375",
961 TABLE.source
962 );
963 assert!(
964 x86_64::INSTS.iter().any(|&(described, _)| described == opcode),
965 "{opcode} is not an instruction anything describes"
966 );
967 }
968 }
969
970 /// The same staleness rule on the x87 pair, and one thing more that is particular to them.
971 ///
972 /// They are a pair. An instruction that pushes onto the x87 stack and nothing that pops off it
973 /// again would leave the stack one deeper than the function found it, which is not a mistake
974 /// the allocator or the block layout could catch, since neither of them knows the stack is
975 /// there. So the two arrive together and leave together, and that is what this says.
976 #[test]
977 fn the_x87_stack_is_reached_by_a_pair_and_by_nothing_else() {
978 let written = heads();
979 for &opcode in X87 {
980 assert!(
981 x86_64::INSTS.iter().any(|&(described, _)| described == opcode),
982 "{opcode} is not an instruction anything describes"
983 );
984 assert!(
985 !written.contains(&format!("{PREFIX}{opcode}").as_str()),
986 "a rule in {} selects {opcode}, which `crate::lower` also writes by hand",
987 TABLE.source
988 );
989 }
990 // One way onto the stack per format a value can be read from, one way off it per format a
991 // value can be written to, the control word pair that is neither, and the arithmetic. The
992 // count is here as well as in the target description because this list is what says none
993 // of them is reachable, and a name that arrived here without its partner would be a format
994 // this target can convert in one direction and not the other.
995 assert_eq!(X87.len(), 30, "twelve that move a value and eighteen that work on one");
996 }
997}