1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
#![allow(clippy::too_many_arguments)]
#![allow(clippy::not_unsafe_ptr_arg_deref)]
#[inline(always)]
pub fn get_rgb(pixmap: &[u8], i: usize) -> (i32, i32, i32) {
let r = pixmap[4 * i] as i32;
let g = pixmap[4 * i + 1] as i32;
let b = pixmap[4 * i + 2] as i32;
(r, g, b)
}
/// ! Publicly exported only for benchmarking.
/// We support only yuv420 format as for now so we can pretty efficiently convert the bitmap buffer.
/// yuv420 represented by y per each pixel and uv (cb and cr) per each 2x2 pixel block.
#[allow(clippy::precedence)]
pub fn fill_yuv420_from_rgba_pixmap_base(
width: i32,
height: i32,
y_linesize: i32,
cb_linesize: i32,
cr_linesize: i32,
rgba_pixels: &[u8],
y_pixels_destination: *mut u8,
cb_pixels_destination: *mut u8,
cr_pixels_destination: *mut u8,
) {
unsafe {
// an important note that linesize here can be different from the width of an image so it is required to fill the buffer correctly.
let width = width as usize;
let height = height as usize;
let frame_size = height * (y_linesize as usize) + width;
let y_pixels = std::slice::from_raw_parts_mut(y_pixels_destination, frame_size);
let cb_pixels = std::slice::from_raw_parts_mut(cb_pixels_destination, frame_size / 2);
let cr_pixels = std::slice::from_raw_parts_mut(cr_pixels_destination, frame_size / 2);
for y in 0..height {
for x in 0..width {
let (r, g, b) = get_rgb(rgba_pixels, y * width + x);
// use a linesize to get the correct index for the pixel as it can differ for different dimensions.
// BT.601 limited range: black is 16, white is 235 (+128 rounds the >> 8).
y_pixels[y * y_linesize as usize + x] =
(16 + ((66 * r + 129 * g + 25 * b + 128) >> 8)) as u8;
if y % 2 == 0 && x % 2 == 0 {
// the bounds are 1/4 of the image size
let x = x / 2;
let y = y / 2;
cb_pixels[y * cb_linesize as usize + x] =
(128 + ((-38 * r) - (74 * g) + (112 * b) >> 8)) as u8;
cr_pixels[y * cr_linesize as usize + x] =
(128 + ((112 * r) - (94 * g) - (18 * b) >> 8)) as u8;
}
}
}
}
}
/// ! Publicly exported only for benchmarking.
/// Accelerated version of the yuv420 for neon using SIMD instructions.
#[cfg(target_feature = "neon")]
pub unsafe fn fill_yuv420_from_rgba_pixmap_accelerated(
width: i32,
height: i32,
y_linesize: i32,
cb_linesize: i32,
cr_linesize: i32,
rgba_pixels: &[u8],
y_pixels_destination: *mut u8,
cb_pixels_destination: *mut u8,
cr_pixels_destination: *mut u8,
) {
// the asm implementation rely on the fact that the resolution is dividable by v8
// which is true for the most common resolution
if width % 8 != 0 {
return fill_yuv420_from_rgba_pixmap_base(
width,
height,
y_linesize,
cb_linesize,
cr_linesize,
rgba_pixels,
y_pixels_destination,
cb_pixels_destination,
cr_pixels_destination,
);
}
unsafe {
std::arch::asm!(
// setup conversion coefficients
"movi v20.8h, #66", // r coef for y
"movi v21.8h, #129", // g coef for y
"movi v22.8h, #25", // b coef for y
"movi v23.8h, #38", // setup cb coeffs
"neg v23.8h, v23.8h", // -38 r (this is correct as negative)
"movi v24.8h, #74", // 74 g (positive)
"movi v25.8h, #112", // 112 b (positive)
"movi v26.8h, #112", // setup cr coeffs (r is positive)
"movi v27.8h, #94", // 94 g (positive)
"movi v28.8h, #18", // 18 b (positive)
// constants
// y offset: (16 << 8) + 128, so the high-half narrow below yields
// 16 + round(sum / 256) (black = 16, white = 235)
"movi v29.8h, #16, lsl #8",
"orr v29.8h, #128",
"movi v30.8h, #128", // cb/cr offset
"mov w9, wzr", // y = 0
"2:", // row loop
"add x1, {src}, {width:x}, lsl #2", // Next row start (width * 4 bytes per pixel)
"prfm pldl1keep, [x1]", // Prefetch next row data
"mov w10, wzr", // x = 0
"3:", // col loop (8 pixels)
// load 8 rgba pixels
"ld4 {{v0.8b, v1.8b, v2.8b, v3.8b}}, [{src}], #32",
// convert rgb to 16-bit
"uxtl v4.8h, v0.8b", // r
"uxtl v6.8h, v1.8b", // g
"uxtl v8.8h, v2.8b", // b
// calc y
"mul.8h v10, v4, v20", // r * 66
"mla.8h v10, v6, v21", // + g * 129
"mla.8h v10, v8, v22", // + b * 25
"addhn.8b v12, v10, v29", // (sum + offset) >> 8 and pack
// store y
"st1 {{v12.8b}}, [{dst_y}], #8",
// only process cb/cr on even rows
"tbnz w9, #0, 5f",
// Extract even-indexed pixels (0, 2, 4, 6)
"uzp1 v13.8h, v4.8h, v4.8h",
"uzp1 v14.8h, v6.8h, v6.8h",
"uzp1 v15.8h, v8.8h, v8.8h",
// calc cb for even pixels: 128 + ((-38*R - 74*G + 112*B) >> 8)
"mul.8h v16, v13, v23", // r * -38
"mls.8h v16, v14, v24", // - g * 74
"mla.8h v16, v15, v25", // + b * 112
"sshr.8h v16, v16, #8", // >> 8
"add.8h v16, v16, v30", // add 128 offset
"sqxtun.8b v17, v16", // convert to unsigned
// calc cr for even pixels: 128 + ((112*R - 94*G - 18*B) >> 8)
"mul.8h v18, v13, v26", // r * 112
"mls.8h v18, v14, v27", // - g * 94 (subtract using mls)
"mls.8h v18, v15, v28", // - b * 18 (subtract using mls)
"sshr.8h v18, v18, #8", // >> 8
"add.8h v18, v18, v30", // add 128 offset
"sqxtun.8b v19, v18", // convert to unsigned
// Store 4 bytes
"str s17, [{dst_cb}], #4",
"str s19, [{dst_cr}], #4",
"5:",
"add w10, w10, #8", // go to next 8 pixels
"cmp w10, {width:w}",
"b.lt 3b",
// end of row
"add {dst_y}, {dst_y}, {y_pad:x}",
// only update padding on even rows
"tbnz w9, #0, 7f",
"add {dst_cb}, {dst_cb}, {cb_pad:x}",
"add {dst_cr}, {dst_cr}, {cr_pad:x}",
"7:",
"add w9, w9, #1", // next row
"cmp w9, {height:w}",
"b.lt 2b",
// the pointers are advanced by the loop, and every value used as a 64 bit
// register has to be passed as one (the upper half of an i32 is undefined)
src = inout(reg) rgba_pixels.as_ptr() => _,
dst_y = inout(reg) y_pixels_destination => _,
dst_cb = inout(reg) cb_pixels_destination => _,
dst_cr = inout(reg) cr_pixels_destination => _,
width = in(reg) width as i64,
height = in(reg) height as i64,
y_pad = in(reg) (y_linesize - width) as i64,
cb_pad = in(reg) (cb_linesize - (width / 2)) as i64,
cr_pad = in(reg) (cr_linesize - (width / 2)) as i64,
out("x1") _, out("w4") _, out("x5") _, out("w6") _, out("w7") _, out("x8") _,
out("w9") _, out("w10") _, out("w11") _, out("w12") _,
out("v0") _, out("v1") _, out("v2") _, out("v3") _, out("v4") _,
out("v6") _, out("v8") _, out("v10") _, out("v12") _, out("v13") _,
out("v14") _, out("v15") _, out("v16") _, out("v17") _, out("v18") _,
out("v19") _, out("v20") _, out("v21") _, out("v22") _, out("v23") _,
out("v24") _, out("v25") _, out("v26") _, out("v27") _, out("v28") _,
out("v29") _, out("v30") _,
options(nostack)
);
}
}
#[cfg(not(target_feature = "neon"))]
pub unsafe fn fill_yuv420_from_rgba_pixmap_accelerated(
width: i32,
height: i32,
y_linesize: i32,
cb_linesize: i32,
cr_linesize: i32,
rgba_pixels: &[u8],
y_pixels_destination: *mut u8,
cb_pixels_destination: *mut u8,
cr_pixels_destination: *mut u8,
) {
fill_yuv420_from_rgba_pixmap_base(
width,
height,
y_linesize,
cb_linesize,
cr_linesize,
rgba_pixels,
y_pixels_destination,
cb_pixels_destination,
cr_pixels_destination,
);
}