rucc_codegen/select/x86_64.rs
1//! The x86-64 lowering table.
2//!
3//! Everything below the module comment is generated from `rules/x86-64.rules` by `rucc-rules`
4//! when this crate is built, and none of it is in the repository. The rule file is the only
5//! place the rules are written, which is what makes the table that is matched with and the
6//! table `rucc-verify` proves things about the same table.
7//!
8//! To read the rules, read the rule file. To read the automaton they compile into, build the
9//! crate and read `x86-64.rs` under the build directory, which is a file worth looking at once
10//! for the shape of it and never again.
11
12// A guard is emitted as the comparison the rule writes, so a rule saying a shift count is at
13// least zero and less than the width comes out as two comparisons rather than as a range. That
14// is deliberate: the generated line and the rule it came from should read the same, and the
15// suggestion to write it another way is advice for somebody editing code, which nobody here is.
16#![allow(clippy::manual_range_contains)]
17
18include!(concat!(env!("OUT_DIR"), "/x86-64.rs"));
19
20#[cfg(test)]
21mod tests {
22 use rucc_target::x86_64;
23
24 use super::TABLE;
25 use crate::select::Piece;
26
27 /// The prefix a rule file puts in front of a machine term, which is how it says which target
28 /// the term belongs to. It is not part of the opcode.
29 const PREFIX: &str = "x64.";
30
31 /// The two address constructors, which are not instructions. An addressing mode is an
32 /// argument to `lea` and to every memory operand after it, so it is written as a term in the
33 /// rule file and built by the selector into the instruction that takes it.
34 const AMODES: &[&str] =
35 &["amode_base_index_scale", "amode_index_scale", "amode_base", "amode_base_offset"];
36
37 /// Every head this table can write, in and under the replacements.
38 fn heads() -> Vec<&'static str> {
39 let mut found: Vec<&'static str> = TABLE
40 .rules
41 .iter()
42 .flat_map(|rule| rule.replacement.iter())
43 .filter_map(|piece| match piece {
44 Piece::App { head, .. } => Some(*head),
45 _ => None,
46 })
47 .collect();
48 found.sort_unstable();
49 found.dedup();
50 found
51 }
52
53 #[test]
54 fn every_instruction_the_table_writes_is_described() {
55 for head in heads() {
56 if AMODES.contains(&head) {
57 continue;
58 }
59 let opcode = head.strip_prefix(PREFIX).unwrap_or_else(|| {
60 panic!("{head} is neither an x86-64 term nor an addressing mode")
61 });
62 assert!(
63 x86_64::form(opcode).is_some(),
64 "{head} is selected by a rule and `rucc_target::x86_64` does not say what it \
65 does with its operands"
66 );
67 }
68 }
69
70 /// The order the operands of a store are written in, which is the IR's and not a choice this
71 /// file makes.
72 ///
73 /// A pattern is matched against an instruction's operand list by position, so a rule that
74 /// names the address where the IR holds the value is a rule that stores to the value and
75 /// writes the address into memory. Nothing in a proof would catch it, because a proof is
76 /// about the rule file agreeing with itself, and both halves would be wrong in the same way.
77 /// `rucc_ir::Builder::store` takes the value first and the machine instruction takes it last,
78 /// which is why the two halves of one of these rules read in opposite orders.
79 #[test]
80 fn a_store_is_written_with_the_value_first_because_that_is_where_the_ir_keeps_it() {
81 let mut seen = 0;
82 for rule in TABLE.rules {
83 let Some(rest) = rule.pattern.strip_prefix("(store.") else { continue };
84 let (width, operands) = rest.split_once(' ').expect("a store takes operands");
85 assert!(
86 operands.starts_with(&format!("(value.{width} ")),
87 "line {}: {} binds something other than the value it is storing first",
88 rule.line,
89 rule.pattern
90 );
91 assert!(
92 operands.contains("(value.i64 "),
93 "line {}: {} reaches no address",
94 rule.line,
95 rule.pattern
96 );
97 seen += 1;
98 }
99 assert_eq!(seen, 16, "the store rules moved and this test did not follow them");
100 }
101
102 /// Every comparison can be made against a constant as well as against a register.
103 ///
104 /// Four comparisons in five in the corpus are against a constant, and without a rule for one
105 /// the constant is loaded into a register first, which is an instruction and a register the
106 /// machine never needed. A missing width or a missing condition would not fail anything else:
107 /// the register rule still matches, the output is still correct, and the only sign is code
108 /// that is one instruction longer in a place nobody is looking. So the two lists are counted
109 /// against each other here.
110 ///
111 /// What this cannot check is that the condition on the immediate rule is the right one, since
112 /// both halves of a wrong pair would be a consistent pair. That is what the `spec` clause is
113 /// for, and `rucc-verify` is what reads it.
114 #[test]
115 fn a_comparison_against_a_constant_is_written_for_every_one_against_a_register() {
116 let mut against_register = Vec::new();
117 let mut against_constant = Vec::new();
118 for rule in TABLE.rules {
119 let Some(rest) = rule.pattern.strip_prefix("(icmp_") else { continue };
120 let (condition, operands) = rest.split_once(".i1 ").expect("a comparison takes two");
121 let width = operands
122 .strip_prefix("(value.")
123 .and_then(|rest| rest.split_once(' '))
124 .map(|(width, _)| width)
125 .expect("a comparison reads a value first");
126 let named = format!("{condition}.{width}");
127 if operands.contains("(iconst.") {
128 // The constant is the second operand and never the first, because a comparison is
129 // not symmetric and the same condition on the other side means the opposite.
130 assert!(
131 !operands.starts_with("(iconst."),
132 "line {}: {} compares a constant against a value",
133 rule.line,
134 rule.pattern
135 );
136 against_constant.push(named);
137 } else {
138 against_register.push(named);
139 }
140 }
141 against_register.sort_unstable();
142 against_constant.sort_unstable();
143 assert_eq!(against_register, against_constant);
144 assert_eq!(against_register.len(), 40, "ten conditions at four widths");
145 }
146
147 /// The instructions the calling convention writes rather than a rule.
148 ///
149 /// Three kinds of them. Naming the register an argument arrived in, where an argument is
150 /// depends on its position in the signature and on the classification of every argument before
151 /// it, and a rule pattern sees one term and has no way to say any of that, so `crate::abi`
152 /// builds these from the convention instead. Calling a name is the same the other way round:
153 /// what its operands are is whatever the signature made them, and a call through an address is
154 /// the same instruction with one operand more.
155 ///
156 /// The second half of a value that comes back in two registers is the third. A return of one
157 /// value is a rule, because where that value goes depends on nothing but the value, which is
158 /// exactly what a rule can say. A return of two is not, because which register the second half
159 /// is in depends on the first half: the two register files are counted separately, so a
160 /// `double` and a `long` both come back at place zero and two `long`s do not.
161 const CONVENTION: &[&str] = &[
162 "arg_val_8",
163 "arg_val_16",
164 "arg_val_32",
165 "arg_val_64",
166 "arg_val_f32",
167 "arg_val_f64",
168 "arg_val_f128",
169 "ret_val2_8",
170 "ret_val2_16",
171 "ret_val2_32",
172 "ret_val2_64",
173 "ret_val2_f32",
174 "ret_val2_f64",
175 "ret_val2_f128",
176 "call",
177 "call_reg",
178 ];
179
180 /// The instructions the block layout writes rather than a rule.
181 ///
182 /// A rule sees one branch and the layout is about the order of every block in the function, so
183 /// which arm falls through is not something any pattern could say. That answer is what decides
184 /// whether the jump goes to the arm the condition is true for or the other one, and whether
185 /// there is a second jump after it, so all of these are written where the answer is.
186 ///
187 /// The comparisons are here for a second reason on top of that one. A branch on a comparison
188 /// is a comparison and a jump on the flags it set, and the flags are not a value: no pattern
189 /// could bind one and no `spec` clause could say anything about one. So the pair is put
190 /// together by the layout, out of a comparison a rule did select and the branch behind it,
191 /// which is the same argument `rucc_target::x86_64::Form::CmpSet` is one form rather than two
192 /// under.
193 const LAYOUT: &[&str] = &[
194 "test_rr_8",
195 "cmp_rr_8",
196 "cmp_rr_16",
197 "cmp_rr_32",
198 "cmp_rr_64",
199 "cmp_ri_8",
200 "cmp_ri_16",
201 "cmp_ri_32",
202 "cmp_ri_64",
203 "cmp_rm_8",
204 "cmp_rm_16",
205 "cmp_rm_32",
206 "cmp_rm_64",
207 "cmp_mi_8",
208 "cmp_mi_16",
209 "cmp_mi_32",
210 "cmp_mi_64",
211 "jcc_e",
212 "jcc_ne",
213 "jcc_l",
214 "jcc_le",
215 "jcc_g",
216 "jcc_ge",
217 "jcc_b",
218 "jcc_be",
219 "jcc_a",
220 "jcc_ae",
221 "jmp",
222 ];
223
224 /// The instructions the compare pass writes rather than a rule.
225 ///
226 /// The other half of the argument the comparisons above are here under. A rule selects a
227 /// comparison that keeps its answer in a byte, because that is the shape a value has. What is
228 /// left of one when the machine has already made the comparison is the byte with no comparison
229 /// in front of it, and there is no pattern for that: the term it would compute is the same term
230 /// the full comparison computes, and what makes the short one right is the instruction three
231 /// places back rather than anything about the value. So `crate::compare` writes them by name,
232 /// in place of a comparison it found was already made.
233 const COMPARE: &[&str] = &[
234 "set_e", "set_ne", "set_l", "set_le", "set_g", "set_ge", "set_b", "set_be", "set_a",
235 "set_ae",
236 ];
237
238 /// The instruction a computed `goto` is written as rather than a rule.
239 ///
240 /// The one branch `crate::lower` writes by name, and the one the block layout does not write
241 /// either. What it reads is the address, which a pattern could have bound, so it is not
242 /// exempt for the reason the branches above are. What no pattern can say is the rest of it:
243 /// how many arms the block has, which is every label of the function the program took the
244 /// address of, and a rule says what an instruction reads rather than where a block goes.
245 const LABELS: &[&str] = &["jmp_reg"];
246
247 /// The instruction the memory model writes rather than a rule.
248 ///
249 /// A barrier computes nothing, so there is no equality for the solver to discharge and no
250 /// pattern for a rule to be written as. What makes it the right answer is what the machine
251 /// promises about the order two other instructions become visible in, which is a claim about
252 /// the program around it rather than about any value. `crate::lower` writes it by name, at the
253 /// strongest ordering and nowhere else, and `crate::expand` says why the strongest is the only
254 /// one that costs anything here.
255 const BARRIER: &[&str] = &["mfence"];
256
257 /// The instruction a program stops on, which `crate::lower` writes rather than a rule.
258 ///
259 /// The first half of the barrier's reason and not the second. It computes nothing, so there is
260 /// no equality for the solver and no pattern for a rule. What makes it right is not a claim
261 /// about the order anything becomes visible in either: it is what the operating system does
262 /// with the fault, which is a fact about neither the values nor the program around it.
263 const STOP: &[&str] = &["ud2"];
264
265 /// The instructions that are a hint rather than a computation.
266 ///
267 /// The same shape of exemption the barrier gets and for a reason one step further out. A
268 /// barrier computes nothing and still has to be where it is, so there is at least a claim about
269 /// the program around it. A prefetch does not even have that: a machine that drops the whole
270 /// instruction runs the program correctly, because the only thing it can change is how long the
271 /// program takes.
272 ///
273 /// So there is no equality for the solver and no pattern for a rule, and which of the four a
274 /// program gets is decided by a number in the builtin's own arguments rather than by anything
275 /// about the value being prefetched. `crate::lower` writes them by name, out of the hint the IR
276 /// carries beside the instruction.
277 const HINT: &[&str] = &["prefetch_nta", "prefetch_t0", "prefetch_t1", "prefetch_t2"];
278
279 /// The instructions nothing but an `asm` statement asks for.
280 ///
281 /// One step further out again. A prefetch is a hint and is still something the compiler decides
282 /// to write, out of a builtin the program called. These are instructions the program wrote down
283 /// itself, by name, in a template, and nothing else in the language reaches them: there is no
284 /// builtin for either, no rule could match a term that produces one, and `crate::lower` writes
285 /// them only because [`rucc_target::x86_64::read`] found the name in a template and said which
286 /// opcode that is.
287 ///
288 /// `pause` is the hint a spin lock writes between two tries at the lock. `cpuid` is how a
289 /// program asks the processor what it can do, which there is no other way to ask, so every
290 /// program that takes a faster path on some machines than on others has one of these in it.
291 ///
292 /// The alignment is the third, and it is on this list rather than one of its own because it
293 /// meets the claim below outright: an instruction is exempt for this reason exactly when there
294 /// is nothing about it for a rule to name, and an opcode with no operands and no addressing mode
295 /// has nothing. It is not an instruction at all, which is more than the test asks and is the
296 /// reason no rule could have been written for it however the rule language grew.
297 const TEMPLATE: &[&str] = &["cpuid", "pause", "align"];
298
299 /// The instructions a template asks for that are right because of the line above them.
300 ///
301 /// These are exempt for the reason the ten bytes in [`COMPARE`] are, one step further out. A
302 /// rule selects a conditional move with its comparison in front of it, because that pair is the
303 /// shape a select has. The move on its own computes the same term and what makes it right is the
304 /// comparison somewhere behind it rather than anything about its own operands, so no pattern
305 /// could say what it means. The compare pass does not write one either, because it replaces a
306 /// comparison it found was already made and there is no earlier move here to replace: what
307 /// writes one is a program that put the comparison on one line of a template and the move on the
308 /// next, which is what zstd does to keep a bounds check from becoming a branch.
309 ///
310 /// So these have operands a rule could have named, unlike everything in [`TEMPLATE`], and they
311 /// are still not instructions a rule could have been written for.
312 /// The add with carry and the subtract with borrow, which read a bit off the instruction in
313 /// front of them.
314 ///
315 /// Exempt one step further out again than [`CONDITIONAL`]. A conditional move reads the
316 /// condition state and leaves it alone, so what is missing from a rule that named one is the
317 /// comparison. These read it and write it both, and what is missing is worse than a comparison:
318 /// the bit they read is the carry out of an addition, and an addition in the IR is an addition
319 /// of a width with no carry out at all, so there is no term a rule could match that the bit is
320 /// a part of. A program gets one by writing both halves itself in a template, which is what
321 /// `add_ssaaaa` and `sub_ddmmss` in libgmp's `longlong.h` are.
322 ///
323 /// What keeps the two halves together once they are two instructions in a block is not here. It
324 /// is `rucc_target::FlagInsts`, which the scheduler reads for exactly this, and the test below
325 /// checks the entry is there rather than trusting that somebody remembered.
326 const CARRY: &[&str] = &[
327 "adc_rr_8",
328 "adc_rr_16",
329 "adc_rr_32",
330 "adc_rr_64",
331 "sbb_rr_8",
332 "sbb_rr_16",
333 "sbb_rr_32",
334 "sbb_rr_64",
335 ];
336
337 const CONDITIONAL: &[&str] = &[
338 "cmov_e_16",
339 "cmov_e_32",
340 "cmov_e_64",
341 "cmov_ne_16",
342 "cmov_ne_32",
343 "cmov_ne_64",
344 "cmov_l_16",
345 "cmov_l_32",
346 "cmov_l_64",
347 "cmov_le_16",
348 "cmov_le_32",
349 "cmov_le_64",
350 "cmov_g_16",
351 "cmov_g_32",
352 "cmov_g_64",
353 "cmov_ge_16",
354 "cmov_ge_32",
355 "cmov_ge_64",
356 "cmov_b_16",
357 "cmov_b_32",
358 "cmov_b_64",
359 "cmov_be_16",
360 "cmov_be_32",
361 "cmov_be_64",
362 "cmov_a_16",
363 "cmov_a_32",
364 "cmov_a_64",
365 "cmov_ae_16",
366 "cmov_ae_32",
367 "cmov_ae_64",
368 ];
369
370 /// The instructions that look for a set bit, which a template asks for and nothing else does.
371 ///
372 /// These have a source and a destination a rule could have named, the way the conditional moves
373 /// above do, and the reason no rule names them is a different one again. It is not that their
374 /// meaning comes from the line in front of them: each of these says on its own exactly what it
375 /// computes. It is that [`crate::expand`] already answers the question they answer, out of
376 /// arithmetic every machine has, and it does that because what these do when the source is zero
377 /// is four different things on four families of processor. A rule that selected one would be a
378 /// rule whose answer depends on which machine ran it.
379 ///
380 /// So the only thing that reaches one is a program that wrote the name in a template, which is
381 /// what the libraries that were counting bits before there was a builtin for it all do.
382 /// `crate::lower` writes them for the reason it writes the three in [`TEMPLATE`], and they are
383 /// not on that list because they are not bare: a rule could have named these operands and the
384 /// claim that list makes would be false of them.
385 const SEARCH: &[&str] = &[
386 "bsf_16", "bsf_32", "bsf_64", "bsr_16", "bsr_32", "bsr_64", "lzcnt_32", "lzcnt_64",
387 "tzcnt_32", "tzcnt_64",
388 ];
389
390 /// The instruction that turns a register round, which a template asks for and nothing else does.
391 ///
392 /// The list above, one step simpler. A search is unselected because what it does with a source
393 /// of zero is not the same on every processor, so a rule that chose one would depend on what ran
394 /// it. A byte reversal has no such case: it means exactly one thing everywhere. What keeps it
395 /// off the rule set is a choice made once, in [`crate::expand`], which builds a reversal out of
396 /// shifts and masks so that the answer is the same on every target this compiler has rather than
397 /// good on the one that happens to have the instruction. tamnd/rucc#310 is where that trade is
398 /// written down, and the day a target grows its own reversal is the day to reopen it.
399 ///
400 /// So the only thing that reaches one is a program that wrote the name in a template, which is
401 /// what libgmp does in `gmp-impl.h` to put a limb the other way round.
402 const SWAP: &[&str] = &["bswap_32", "bswap_64"];
403
404 /// The multiply that keeps both halves of its product and the division that reads both halves
405 /// of its dividend, which a template asks for and nothing else does.
406 ///
407 /// A third reason again, and the plainest of the three. A search is unselected because its
408 /// answer depends on the processor and a reversal because a choice was made to build one out of
409 /// arithmetic. This one is unselected because there is nothing in the IR to select it from: a
410 /// multiply in C takes two values of a type and produces a value of that type, so the term a
411 /// rule would match on is the narrow product, and the wide product is not a term at all. A rule
412 /// that fired on the narrow one and wrote this would be writing an instruction that computes
413 /// twice as much as was asked for and leaves the rest in a register nobody asked about.
414 ///
415 /// So the only thing that reaches one is a program that wrote the name in a template, which is
416 /// what `umul_ppmm` in libgmp's `longlong.h` does, and what every library that is building
417 /// arithmetic out of limbs does somewhere.
418 ///
419 /// The division is the same claim upside down and is on this list because the reason is the same
420 /// one. A division in C divides a number by a number of its own width, so the term a rule would
421 /// match is the narrow one, and this compiler already has two opcodes for that: each of them
422 /// fills the high half of the dividend itself and then throws one of the two answers away. A
423 /// dividend the program filled both halves of is not a term the IR has, and `udiv_qrnnd` beside
424 /// the multiply in the same header is how long division a limb at a time is written.
425 const WIDE: &[&str] = &[
426 "mul_wide_16",
427 "mul_wide_32",
428 "mul_wide_64",
429 "imul_wide_16",
430 "imul_wide_32",
431 "imul_wide_64",
432 "div_wide_16",
433 "div_wide_32",
434 "div_wide_64",
435 "idiv_wide_16",
436 "idiv_wide_32",
437 "idiv_wide_64",
438 ];
439
440 /// The instructions that produce two values, which is one more than a rule can name.
441 ///
442 /// A rule replaces a term with a term, and a term is the value one instruction computes. A
443 /// compare and exchange computes two: what it found at the address, and whether what it found
444 /// was what the program expected. There is no way to write the second one down in the rule
445 /// language, and inventing one would be inventing a language for a single instruction.
446 ///
447 /// So `crate::lower` writes it by name, the way it writes the barrier by name, and for a reason
448 /// that is about the rule language rather than about the machine. What the solver would have
449 /// been asked to prove about it is the easy half in any case: the arithmetic is a comparison
450 /// and a select, and what is hard is that the whole of it happens at once, which is the same
451 /// claim about the program around it that a barrier makes.
452 const ATOMIC: &[&str] = &["cmpxchg_8", "cmpxchg_16", "cmpxchg_32", "cmpxchg_64"];
453
454 /// The instructions whose operation is in the payload rather than in the head.
455 ///
456 /// A different exemption from the one above, on instructions that produce one value each and so
457 /// could be named by a rule if the rule had anything to match on. The head a pattern matches is
458 /// an opcode and a type, and every read modify write in the IR is the one opcode `atomic_rmw`.
459 /// Which of the thirteen operations it performs is carried beside the instruction rather than in
460 /// its name, so a pattern written for the exchange would match the add and the nand as well, and
461 /// the rule language has no way to look at what a rule matched to tell them apart.
462 ///
463 /// Giving each operation its own opcode is the other way out and is a worse trade: it is
464 /// thirteen opcodes at four widths where the IR wants one, and every pass that treats a read
465 /// modify write as one thing would then have a list of fifty two.
466 ///
467 /// So `crate::lower` writes these by name too. Three operations here, out of the thirteen: the
468 /// bitwise ones need a loop around a compare and exchange, which is control flow and so is built
469 /// before selection rather than during it, and they are the rest of `tamnd/rucc#311`.
470 const PAYLOAD: &[&str] =
471 &["xchg_8", "xchg_16", "xchg_32", "xchg_64", "xadd_8", "xadd_16", "xadd_32", "xadd_64"];
472
473 /// The instructions a frame writes rather than a rule.
474 ///
475 /// A prologue, an epilogue, a copy, a spill and a reload are not in the program. They are what
476 /// the allocator's answer costs, so they are written after it, by `crate::finish` reading
477 /// `x86_64::FRAME`. Six of the names that describes are already reachable from a rule, since a
478 /// prologue taking its frame is a subtraction and a spill is a store, and those are not here:
479 /// this is only the ones nothing else can reach.
480 const FRAME: &[&str] = &[
481 "push_64",
482 "pop_64",
483 "ret",
484 "mov_rr_64",
485 "movaps_rr",
486 // The touch a probing prologue puts on each page as it reaches it, the landing pad a
487 // prologue opens with, and the byte that does nothing which one reserves room with. All
488 // three are written by a frame and none on a command line that did not ask for it.
489 "or_mi_8",
490 "endbr64",
491 "nop",
492 ];
493
494 /// The instructions that reach the x87 stack, which are selected but not from here.
495 ///
496 /// A third kind of exemption, and the same reason all the way down the list.
497 ///
498 /// Every one of these is written by `crate::lower`, as part of a group rather than on its own.
499 /// What one of them leaves behind and the next picks up is the top of the x87 stack, which is
500 /// not a register anything allocates from and not a value a pattern could bind, so a rule
501 /// could neither match the middle of a group nor name what its replacement produced. And an
502 /// add here reads two addresses and writes a third, where one machine IR instruction carries
503 /// one addressing mode, so the group cannot be folded into a single opcode the way
504 /// `ucomisd_set_e` folds a comparison and a `setcc` either.
505 ///
506 /// So these are exempt for the reason `FRAME` is exempt rather than for the reason the list
507 /// below is, and they will stay exempt. Two of them are not reached by anything yet all the
508 /// same: `fsub_p` and `fdiv_p` are the other direction of the subtraction and the division,
509 /// which a code generator that pushed its operands the other way round would need and this one
510 /// does not. `fabs` is a third, since C spells that as a call to a library function.
511 const X87: &[&str] = &[
512 "fld_t",
513 "fstp_t",
514 "fld_s",
515 "fld_l",
516 "fild_l",
517 "fild_ll",
518 "fstp_s",
519 "fstp_l",
520 "fistp_l",
521 "fistp_ll",
522 "fnstcw",
523 "fldcw",
524 "fadd_p",
525 "fsub_p",
526 "fsubr_p",
527 "fmul_p",
528 "fdiv_p",
529 "fdivr_p",
530 "fchs",
531 "fabs",
532 "fucomip_set_a",
533 "fucomip_set_ae",
534 "fucomip_set_b",
535 "fucomip_set_be",
536 "fucomip_set_e",
537 "fucomip_set_ne",
538 "fucomip_set_p",
539 "fucomip_set_np",
540 "fucomip_set_e_and_np",
541 "fucomip_set_ne_or_p",
542 ];
543
544 /// The instructions no rule selects yet, because the rules that selected them were taken out.
545 ///
546 /// A different kind of exemption from the three above. Those say an instruction is written
547 /// somewhere a rule cannot reach and always will be. These say nobody reaches one at all right
548 /// now, and name the work that puts the rules back.
549 ///
550 /// The rules went out under `tamnd/rucc#368`. C promotes the operands of an arithmetic
551 /// operator to `int`, so a byte add and a two byte compare are things no C program asks the
552 /// back end for, and the rules at those widths sat proved and never selected over the whole
553 /// torture corpus at every optimization level. The width narrowing pass in `tamnd/rucc#375` is
554 /// what asks for them, and the rules come back with it.
555 ///
556 /// The descriptions stayed. A description says what an x86-64 instruction is, how long it is
557 /// and how it encodes, and that is true whether or not anything selects it. Taking them out
558 /// would be deleting a correct account of the machine to make a list shorter, and putting them
559 /// back is then a second thing to get right rather than a line of a rule file.
560 const NARROW: &[&str] = &[
561 // Three of the two address forms against an immediate. The `narrow` pass does write the
562 // shape, since `char c = a | 1;` narrows to a byte `or` against a byte constant, and no
563 // rule selects these yet: the constant goes into a register and the register with
564 // register rule takes it. Their `add`, `sub` and `and` siblings do have rules and are
565 // reached by the bitfield lowering, so this is six rules missing rather than a shape
566 // nothing writes.
567 "or_ri_8",
568 "or_ri_16",
569 "xor_ri_8",
570 "xor_ri_16",
571 "imul_ri_8",
572 "imul_ri_16",
573 // The divides, which are four instructions per width because the quotient and the
574 // remainder come out of one division in two different registers. `narrow` refuses these
575 // on purpose: the most negative byte over minus one is a defined hundred and twenty eight
576 // at four bytes and is the overflow that raises at one, so narrowing a division wants a
577 // range that rules the pair out and there is no range analysis yet.
578 "idiv_quo_8",
579 "idiv_quo_16",
580 "idiv_rem_8",
581 "idiv_rem_16",
582 "div_quo_8",
583 "div_quo_16",
584 "div_rem_8",
585 "div_rem_16",
586 // The shifts by a value, whose count is in `cl` whatever the width being shifted is. The
587 // same refusal for the same kind of reason: a count of twenty is a defined shift to zero
588 // at four bytes and is poison at one, so only a count that is a constant below the narrow
589 // width narrows, and that one selects the immediate forms which do have rules.
590 "shl_rcl_8",
591 "shl_rcl_16",
592 "shr_rcl_8",
593 "shr_rcl_16",
594 "sar_rcl_8",
595 "sar_rcl_16",
596 ];
597
598 /// The arithmetic that reaches memory, which [`crate::combine`] writes: the forms that read a
599 /// source out of it and the forms that leave the answer in it.
600 ///
601 /// A function rather than a list, for the reason the compare pass's exemption is taken from the
602 /// flag description rather than typed out: the pass already writes down which instructions it
603 /// can produce, and a second copy of that here would be a second opinion about one pass.
604 ///
605 /// No rule selects one of these because a rule matches a term and one of these is two terms, a
606 /// load and an arithmetic operation, put together, or three where the answer goes back to
607 /// memory. Whether they may be put together depends on what is written between them and on
608 /// whether anything else wants what the load read, and neither is a fact about any of the
609 /// terms. That is the whole reason the pass exists and the module documentation there says it
610 /// at length.
611 fn combine() -> Vec<&'static str> {
612 let loads = crate::combine::FOLDS.iter().map(|fold| fold.into);
613 // And the instruction a load on the other side comes to, which for most rows is the one
614 // above and for a comparison is the condition the other way round.
615 let swapped = crate::combine::FOLDS.iter().filter_map(|fold| fold.swapped);
616 let stores = crate::combine::UPDATES.iter().map(|update| update.into);
617 let constants = crate::combine::BUMPS.iter().map(|bump| bump.into);
618 loads.chain(swapped).chain(stores).chain(constants).collect()
619 }
620
621 #[test]
622 fn every_instruction_exempt_from_a_rule_is_one_a_frame_really_writes() {
623 // The same claim as the one about the convention, so that this list cannot grow an opcode
624 // that no frame asks for. In the order `x86_64::FRAME` names them, the copies after the
625 // return because there is one set of them per class the allocator may spill.
626 let frame = &x86_64::FRAME;
627 let mut written = vec![frame.push, frame.pop, frame.ret];
628 for class in frame.classes {
629 written.extend([class.mov, class.load, class.store]);
630 }
631 // And the touch a probing prologue puts on a page, which the target names as an option
632 // because a target with no instruction that writes an address without changing it takes
633 // every frame in one subtraction and has nothing to exempt.
634 written.extend(frame.probe.map(|probe| probe.inst));
635 // And the landing pad and the byte that does nothing, which are options for the same
636 // reason.
637 written.extend(frame.landing);
638 written.extend(frame.pad);
639 // What is left after the ones a rule already reaches, which are the loads and the stores of
640 // both register files, since those are the same instructions a program's own reads and
641 // writes of memory are. The vector pair joined them with the rules for a quad float, and a
642 // spill of one is now the same instruction as a program reading a `_Float128` variable.
643 written.retain(|opcode| !heads().contains(&format!("{PREFIX}{opcode}").as_str()));
644 assert_eq!(written, FRAME);
645 }
646
647 #[test]
648 fn every_instruction_exempt_from_a_rule_is_one_the_convention_really_writes() {
649 // An exemption list that nothing checks is a hole, since an opcode dropped into it stops
650 // being covered by either direction of the pinning. These are the ones `crate::abi` can
651 // name, at the four integer widths and the two float formats it has names for an
652 // argument in, and no others.
653 let strip = |head: &'static str| head.strip_prefix(PREFIX).expect("an x86-64 term");
654 let named = |ty| strip(crate::abi::head_of(ty).expect("every width the pseudos cover"));
655 // The second half of a pair at place one, which is the place a rule cannot name. The first
656 // half at place zero is `ret_val_*` and is reached by a rule, so it is not on this list.
657 let second = |ty| strip(crate::abi::ret_of(ty, 1).expect("every width the pseudos cover"));
658 let widths = || {
659 [8, 16, 32, 64].into_iter().map(rucc_ir::Type::int).chain(
660 [rucc_ir::Float::F32, rucc_ir::Float::F64, rucc_ir::Float::F128]
661 .map(rucc_ir::Type::float),
662 )
663 };
664 let written: Vec<&str> = widths()
665 .map(named)
666 .chain(widths().map(second))
667 .chain([strip(crate::abi::CALL), strip(crate::abi::CALL_REG)])
668 .collect();
669 assert_eq!(written, CONVENTION);
670 }
671
672 /// The same claim about the block layout's list, which is longer than it looks.
673 ///
674 /// A name here that the layout does not write is an opcode exempted from needing a rule and
675 /// reached by nothing, and a name the layout writes that is not here is a failing test in
676 /// `every_described_instruction_is_reachable_from_a_rule` with a misleading message. Both are
677 /// avoided by taking the list from `rucc_target::x86_64::BRANCH` rather than believing it.
678 #[test]
679 fn every_instruction_exempt_from_a_rule_is_one_the_block_layout_really_writes() {
680 let branch = &x86_64::BRANCH;
681 // Eighty entries name sixteen instructions between them, so this is a set rather than a
682 // list and both sides are sorted before they are held against each other. What the order
683 // of the list itself is for is reading it.
684 let mut written: Vec<&str> = vec![branch.test, branch.jump];
685 written.extend(branch.fused.iter().map(|fusion| fusion.cmp));
686 written.extend(branch.fused.iter().flat_map(|fusion| [fusion.if_true, fusion.if_false]));
687 written.sort_unstable();
688 written.dedup();
689 let mut exempt = LAYOUT.to_vec();
690 exempt.sort_unstable();
691 assert_eq!(written, exempt);
692 }
693
694 /// The same claim about the compare pass. What it writes is what the flag description says is
695 /// left of a comparison, so the exemption is taken from that rather than typed out twice, and
696 /// an entry added there without a rule to go with it shows up here rather than in a build that
697 /// fails somewhere else.
698 #[test]
699 fn every_instruction_exempt_from_a_rule_is_one_the_compare_pass_really_writes() {
700 let mut written: Vec<&str> =
701 x86_64::FLAGS.compares.iter().filter_map(|entry| entry.kept).collect();
702 written.sort_unstable();
703 written.dedup();
704 let mut exempt = COMPARE.to_vec();
705 exempt.sort_unstable();
706 assert_eq!(written, exempt);
707 }
708
709 /// And the same claim about the one the lowering writes, held against the name the target gave
710 /// it rather than against the spelling written above.
711 #[test]
712 fn the_instruction_a_computed_goto_is_exempt_for_is_the_one_the_target_names() {
713 assert_eq!(LABELS, [x86_64::BRANCH.indirect]);
714 }
715
716 /// The rows of the constant table that take nothing yet are exactly the narrow ones waiting on
717 /// the width narrowing, so the day `NARROW` shrinks is the day this says so.
718 ///
719 /// `crate::combine::BUMPS` has a row per instruction this machine has, which is the whole five
720 /// operations at the whole four widths. Four of those instructions arrive out of a rule that is
721 /// not written yet, so four of the rows sit there taking nothing. That is a fact worth holding
722 /// rather than a thing to notice again later.
723 #[test]
724 fn the_constant_runs_that_take_nothing_are_the_ones_no_rule_selects_yet() {
725 let written = heads();
726 let mut waiting = Vec::new();
727 for bump in crate::combine::BUMPS {
728 if !written.contains(&format!("{PREFIX}{}", bump.from).as_str()) {
729 waiting.push(bump.from);
730 }
731 }
732 assert_eq!(waiting, ["or_ri_8", "or_ri_16", "xor_ri_8", "xor_ri_16"]);
733 for from in waiting {
734 assert!(NARROW.contains(&from), "{from} is unselected and is not on the list");
735 }
736 }
737
738 #[test]
739 fn every_described_instruction_is_reachable_from_a_rule() {
740 let written = heads();
741 let combine = combine();
742 for &(opcode, _) in x86_64::INSTS {
743 if combine.contains(&opcode) {
744 continue;
745 }
746 if CONVENTION.contains(&opcode) || LAYOUT.contains(&opcode) || FRAME.contains(&opcode) {
747 continue;
748 }
749 if NARROW.contains(&opcode) || BARRIER.contains(&opcode) || X87.contains(&opcode) {
750 continue;
751 }
752 if ATOMIC.contains(&opcode) || PAYLOAD.contains(&opcode) || HINT.contains(&opcode) {
753 continue;
754 }
755 if CONDITIONAL.contains(&opcode) {
756 continue;
757 }
758 if CARRY.contains(&opcode) {
759 continue;
760 }
761 if COMPARE.contains(&opcode) || TEMPLATE.contains(&opcode) {
762 continue;
763 }
764 if SEARCH.contains(&opcode) || SWAP.contains(&opcode) || WIDE.contains(&opcode) {
765 continue;
766 }
767 if LABELS.contains(&opcode) || STOP.contains(&opcode) {
768 continue;
769 }
770 let head = format!("{PREFIX}{opcode}");
771 assert!(
772 written.contains(&head.as_str()),
773 "{opcode} is described and no rule in {} selects it",
774 TABLE.source
775 );
776 }
777 }
778
779 /// The same claim about the barrier as the ones above make about the convention and the frame:
780 /// the list holds instructions this target really describes, and holds only the ones that have
781 /// no operands, since an instruction with an operand is one a rule could have been written for.
782 #[test]
783 fn every_instruction_exempt_from_a_rule_is_one_the_memory_model_really_writes() {
784 for &opcode in BARRIER {
785 let form = x86_64::form(opcode).expect("an instruction this target describes");
786 assert!(form.operands().is_empty(), "{opcode} has operands, so a rule could name it");
787 }
788 }
789
790 /// The same claim about the instruction a program stops on, which is the barrier's shape
791 /// exactly: no operands, because an instruction with one is an instruction a rule could have
792 /// been written for, and no addressing mode either, because it is given nothing at all.
793 #[test]
794 fn the_instruction_exempt_from_a_rule_because_it_stops_the_program_is_bare() {
795 for &opcode in STOP {
796 let form = x86_64::form(opcode).expect("an instruction this target describes");
797 assert!(form.operands().is_empty(), "{opcode} has operands, so a rule could name it");
798 assert!(!form.takes_mem(), "{opcode} is given an address and stopping needs none");
799 }
800 }
801
802 /// The same claim about the hints, with the one difference between them written down. A hint is
803 /// given an address and nothing else, so it has no operands for the reason a barrier has none
804 /// and it does carry an addressing mode, which is what a rule would have had to match on.
805 #[test]
806 fn every_instruction_exempt_from_a_rule_because_it_is_a_hint_is_given_only_an_address() {
807 for &opcode in HINT {
808 let form = x86_64::form(opcode).expect("an instruction this target describes");
809 assert!(form.operands().is_empty(), "{opcode} has operands, so a rule could name it");
810 assert!(form.takes_mem(), "{opcode} is a hint about an address and is given none");
811 }
812 }
813
814 /// The same claim about the template list. An instruction is exempt for this reason exactly
815 /// when there is nothing about it for a rule to name, and there are two ways to have nothing.
816 /// No operands and no address, which is the hint. Or every operand fixed to one register by the
817 /// description, which is the question put to the processor: a rule names the operands of a term
818 /// and binds them to the values underneath it, and an operand that can be nothing but `rax` is
819 /// not a place a value goes. Either way the whole of the claim holds, which is that there was
820 /// nowhere else for the instruction to come from.
821 #[test]
822 fn every_instruction_exempt_from_a_rule_because_only_a_template_asks_for_it_is_bare() {
823 for &opcode in TEMPLATE {
824 let form = x86_64::form(opcode).expect("an instruction this target describes");
825 let fixed = form
826 .operands()
827 .iter()
828 .all(|desc| matches!(desc.constraint, rucc_target::Constraint::Fixed(_)));
829 assert!(fixed, "{opcode} has an operand a rule could name");
830 assert!(!form.takes_mem(), "{opcode} is given an address, so a rule could name it");
831 }
832 }
833
834 /// The same claim about the bit searches, read off the description that put them there and read
835 /// both ways round. An instruction is exempt for this reason exactly when the machine describes
836 /// it as a search, so the list cannot grow an opcode that is something else, and a search this
837 /// target grows later cannot be left off the list and quietly go unselected with nobody saying
838 /// why. Nothing in the rule set selects one, which is the other half of the reason and is what
839 /// the check above would have caught in any case.
840 #[test]
841 fn every_instruction_exempt_from_a_rule_because_only_a_template_searches_for_a_bit_is_one() {
842 let written = heads();
843 for &opcode in SEARCH {
844 let form = x86_64::form(opcode).expect("an instruction this target describes");
845 assert_eq!(form, x86_64::Form::Search, "{opcode} is not a search");
846 assert!(
847 !written.contains(&format!("{PREFIX}{opcode}").as_str()),
848 "a rule in {} selects {opcode}, which only a template asks for",
849 TABLE.source
850 );
851 }
852 for &(opcode, form) in x86_64::INSTS {
853 if form == x86_64::Form::Search {
854 assert!(SEARCH.contains(&opcode), "{opcode} is a search and is not on the list");
855 }
856 }
857 }
858
859 /// The same claim about the byte reversal, read both ways round the way the searches are, and
860 /// with the one thing that is different about it checked as well: this is the instruction of its
861 /// shape that leaves the condition state alone, which is the whole reason it has a form rather
862 /// than being a unary operation, so a description that stopped saying that would stop being the
863 /// reason this list exists.
864 #[test]
865 fn every_instruction_exempt_from_a_rule_because_only_a_template_turns_a_register_round_is_one()
866 {
867 let written = heads();
868 for &opcode in SWAP {
869 let form = x86_64::form(opcode).expect("an instruction this target describes");
870 assert_eq!(form, x86_64::Form::Swap, "{opcode} is not a byte reversal");
871 assert!(
872 !(x86_64::FLAGS.writes)(opcode),
873 "{opcode} writes the condition state, so it is a unary operation after all"
874 );
875 assert!(
876 !written.contains(&format!("{PREFIX}{opcode}").as_str()),
877 "a rule in {} selects {opcode}, which only a template asks for",
878 TABLE.source
879 );
880 }
881 for &(opcode, form) in x86_64::INSTS {
882 if form == x86_64::Form::Swap {
883 assert!(SWAP.contains(&opcode), "{opcode} is a reversal and is not on the list");
884 }
885 }
886 }
887
888 /// The same claim about the two that work on a pair of registers, read both ways round and with
889 /// the thing that puts them out of reach of a rule checked rather than asserted in prose: each
890 /// writes two registers, and a rule replaces a term with a term, so there is no way to say the
891 /// second answer in the rule language at all. That is the same bar the compare and exchange is
892 /// exempt at, and this list is separate from that one because the reason it is nobody's to select
893 /// is different: an atomic is written by name where it is needed, and nothing in this compiler
894 /// needs one of these.
895 #[test]
896 fn every_instruction_exempt_from_a_rule_because_only_a_template_wants_both_halves_writes_two() {
897 let written = heads();
898 let both = [x86_64::Form::MulWide, x86_64::Form::DivWide];
899 for &opcode in WIDE {
900 let form = x86_64::form(opcode).expect("an instruction this target describes");
901 assert!(both.contains(&form), "{opcode} works on one register rather than on a pair");
902 let defs = form.operands().iter().filter(|desc| desc.role.is_def()).count();
903 assert_eq!(defs, 2, "{opcode} writes {defs} registers and a pair takes two");
904 assert!(
905 !written.contains(&format!("{PREFIX}{opcode}").as_str()),
906 "a rule in {} selects {opcode}, which only a template asks for",
907 TABLE.source
908 );
909 }
910 for &(opcode, form) in x86_64::INSTS {
911 if both.contains(&form) {
912 assert!(WIDE.contains(&opcode), "{opcode} works on a pair and is not on the list");
913 }
914 }
915 }
916
917 /// The same claim about the carry pair, and the one thing that has to be true of them that is
918 /// not true of anything else on any of these lists. An instruction here reads the condition
919 /// state and writes it, which is what makes it half of a pair and not a rewrite of its own, and
920 /// the scheduler will only keep it behind the instruction that set the bit if the target says
921 /// it reads one.
922 #[test]
923 fn every_instruction_exempt_from_a_rule_because_it_reads_a_carry_says_it_reads_the_state() {
924 let written = heads();
925 for &opcode in CARRY {
926 let form = x86_64::form(opcode).expect("an instruction this target describes");
927 assert_eq!(form, x86_64::Form::AluCarry, "{opcode} is not one of the pair");
928 assert_eq!(
929 x86_64::FLAGS.reads(opcode),
930 Some(rucc_target::Reads::Carry),
931 "{opcode} does not say it reads the carry, so the scheduler may move it"
932 );
933 assert!(
934 (x86_64::FLAGS.writes)(opcode),
935 "{opcode} is said to leave the condition state alone"
936 );
937 assert!(
938 !written.contains(&format!("{PREFIX}{opcode}").as_str()),
939 "a rule in {} selects {opcode}, which only a template asks for",
940 TABLE.source
941 );
942 }
943 for &(opcode, form) in x86_64::INSTS {
944 if form == x86_64::Form::AluCarry {
945 assert!(CARRY.contains(&opcode), "{opcode} reads a carry and is not on the list");
946 }
947 }
948 }
949
950 /// The same claim about the conditional moves, read off the flag description the way the compare
951 /// pass's list is taken from it rather than typed out twice. An instruction is exempt for this
952 /// reason exactly when it reads the condition state and leaves it as it found it, which is what
953 /// says the instruction in front of it is where its meaning comes from. One that wrote the state
954 /// as well would be one a pattern could match on its own.
955 #[test]
956 fn every_instruction_exempt_from_a_rule_because_a_comparison_gives_it_its_meaning_reads_one() {
957 for &opcode in CONDITIONAL {
958 x86_64::form(opcode).expect("an instruction this target describes");
959 assert!(
960 x86_64::FLAGS.reads(opcode).is_some(),
961 "{opcode} reads no comparison, so a rule could name it"
962 );
963 assert!(
964 !(x86_64::FLAGS.writes)(opcode),
965 "{opcode} writes the condition state, so a rule could name it"
966 );
967 }
968 }
969
970 /// The same claim about the atomic list, read off the thing that put the entry there: an
971 /// instruction is exempt for this reason exactly when it writes more than one value, and an
972 /// instruction that writes one is one a rule could have been written for.
973 #[test]
974 fn every_instruction_exempt_from_a_rule_is_one_that_writes_more_than_one_value() {
975 let written = heads();
976 for &opcode in ATOMIC {
977 let form = x86_64::form(opcode).expect("an instruction this target describes");
978 let writes = form.operands().iter().filter(|desc| desc.role.is_def()).count();
979 assert!(writes > 1, "{opcode} writes one value, so a rule could name it");
980 assert!(
981 !written.contains(&format!("{PREFIX}{opcode}").as_str()),
982 "a rule in {} selects {opcode}, which `crate::lower` also writes by hand",
983 TABLE.source
984 );
985 }
986 }
987
988 /// The same claim about the payload list, read off the thing that puts an entry there.
989 ///
990 /// Two halves. Each of these writes one value, which is what says the reason above is not the
991 /// reason here, so a list that grew to cover an instruction the atomic list should have had
992 /// fails. And there really is more than one operation behind the one IR opcode, which is the
993 /// whole of why a pattern cannot name any of them, and is a fact about the IR that would stop
994 /// being true if the operations were ever given opcodes of their own.
995 #[test]
996 fn every_instruction_exempt_because_its_operation_is_beside_it_writes_one_value() {
997 assert!(
998 rucc_ir::RmwOp::all().count() > 1,
999 "one operation per opcode would be a head a rule could match"
1000 );
1001 let written = heads();
1002 for &opcode in PAYLOAD {
1003 let form = x86_64::form(opcode).expect("an instruction this target describes");
1004 let writes = form.operands().iter().filter(|desc| desc.role.is_def()).count();
1005 assert_eq!(writes, 1, "{opcode} writes more than one value, so it is the other list's");
1006 assert!(
1007 !written.contains(&format!("{PREFIX}{opcode}").as_str()),
1008 "a rule in {} selects {opcode}, which `crate::lower` also writes by hand",
1009 TABLE.source
1010 );
1011 }
1012 }
1013
1014 /// The staleness rule every list in this project is kept under, on the one list here whose
1015 /// entries are meant to leave. A rule that starts selecting one of these is `tamnd/rucc#375`
1016 /// arriving, and the entry goes with it. An entry naming an instruction nothing describes is a
1017 /// misspelling, and it would sit here exempting nothing.
1018 #[test]
1019 fn an_instruction_a_rule_now_selects_is_off_the_list_of_the_ones_left_for_later() {
1020 let written = heads();
1021 for &opcode in NARROW {
1022 let head = format!("{PREFIX}{opcode}");
1023 assert!(
1024 !written.contains(&head.as_str()),
1025 "a rule in {} selects {opcode} now, so it is not waiting on tamnd/rucc#375",
1026 TABLE.source
1027 );
1028 assert!(
1029 x86_64::INSTS.iter().any(|&(described, _)| described == opcode),
1030 "{opcode} is not an instruction anything describes"
1031 );
1032 }
1033 }
1034
1035 /// The same staleness rule on the x87 pair, and one thing more that is particular to them.
1036 ///
1037 /// They are a pair. An instruction that pushes onto the x87 stack and nothing that pops off it
1038 /// again would leave the stack one deeper than the function found it, which is not a mistake
1039 /// the allocator or the block layout could catch, since neither of them knows the stack is
1040 /// there. So the two arrive together and leave together, and that is what this says.
1041 #[test]
1042 fn the_x87_stack_is_reached_by_a_pair_and_by_nothing_else() {
1043 let written = heads();
1044 for &opcode in X87 {
1045 assert!(
1046 x86_64::INSTS.iter().any(|&(described, _)| described == opcode),
1047 "{opcode} is not an instruction anything describes"
1048 );
1049 assert!(
1050 !written.contains(&format!("{PREFIX}{opcode}").as_str()),
1051 "a rule in {} selects {opcode}, which `crate::lower` also writes by hand",
1052 TABLE.source
1053 );
1054 }
1055 // One way onto the stack per format a value can be read from, one way off it per format a
1056 // value can be written to, the control word pair that is neither, and the arithmetic. The
1057 // count is here as well as in the target description because this list is what says none
1058 // of them is reachable, and a name that arrived here without its partner would be a format
1059 // this target can convert in one direction and not the other.
1060 assert_eq!(X87.len(), 30, "twelve that move a value and eighteen that work on one");
1061 }
1062}