1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
//! Whole-unit tasks over three equally-long mutable buffers, split in lockstep.
use drive_unit_tasks;
use cratechunk_shards;
use crateExecutionPolicy;
chunk_shards!
/// Apply `f(state, first_unit, a_units, b_units, c_units)` to aligned runs of
/// whole units of three mutable buffers, each run sized to about
/// [`crate::UNIT_TASK_BYTES`] of work.
///
/// This is [`crate::for_each_unit_task_pair_mut_with`] for a pass that writes three
/// fields per element: `a`, `b` and `c` hold the same number of
/// `unit_len`-element units, and each task receives the same units of all
/// three. `unit_bytes` counts one unit of each buffer and any input read beside
/// them.
///
/// # Panics
///
/// Panics if `unit_len` is zero, `a.len()` is not a multiple of it, or `b` or
/// `c` is not the length of `a`, and propagates a panic raised by `init` or `f`.
///
/// # Examples
///
/// ```
/// use moirai_parallel::{for_each_unit_task_triple_mut_with, WorkBytes};
///
/// let source: Vec<u64> = (0..12).collect();
/// let (mut xs, mut ys, mut zs) = (vec![0_u64; 12], vec![0_u64; 12], vec![0_u64; 12]);
/// // Rows of 4, each moving 32 bytes of each output and 32 of the source.
/// for_each_unit_task_triple_mut_with::<WorkBytes<{ 1 << 20 }>, _, _, _, _, _, _>(
/// &mut xs,
/// &mut ys,
/// &mut zs,
/// 4,
/// 128,
/// || (),
/// |(), first_unit, xs, ys, zs| {
/// let start = first_unit * 4;
/// let outputs = xs.iter_mut().zip(ys.iter_mut()).zip(zs.iter_mut());
/// for (((x, y), z), &value) in outputs.zip(&source[start..]) {
/// *x += value;
/// *y += value;
/// *z += value;
/// }
/// },
/// );
/// assert!(xs == source && ys == source && zs == source);
/// ```