Small optimisation of divide_alpha_row_native()

This commit is contained in:
Kirill Kuzminykh
2021-07-06 21:46:22 +03:00
parent 6214456ddd
commit c3be3bc77a
2 changed files with 18 additions and 15 deletions
+15 -12
View File
@@ -157,16 +157,19 @@ fn div_and_clip(v: u8, rev_alpha: f32) -> u8 {
#[inline(always)]
fn divide_alpha_row_native(src_row: &[u32], dst_row: &mut [u32]) {
for (src_pixel, dst_pixel) in src_row.iter().zip(dst_row) {
let components: [u8; 4] = src_pixel.to_le_bytes();
let alpha = components[3];
let recip_alpha = if alpha == 0 { 0. } else { 255. / alpha as f32 };
let res = [
div_and_clip(components[0], recip_alpha),
div_and_clip(components[1], recip_alpha),
div_and_clip(components[2], recip_alpha),
alpha,
];
*dst_pixel = u32::from_le_bytes(res);
}
src_row
.iter()
.zip(dst_row)
.for_each(|(src_pixel, dst_pixel)| {
let components: [u8; 4] = src_pixel.to_le_bytes();
let alpha = components[3];
let recip_alpha = if alpha == 0 { 0. } else { 255. / alpha as f32 };
let res = [
div_and_clip(components[0], recip_alpha),
div_and_clip(components[1], recip_alpha),
div_and_clip(components[2], recip_alpha),
alpha,
];
*dst_pixel = u32::from_le_bytes(res);
});
}
+3 -3
View File
@@ -64,8 +64,8 @@ pub fn precompute_coefficients(
// Maximum number of coeffs per out pixel
let window_size = filter_radius.ceil() as usize * 2 + 1;
// Optimization: replace division by filter_scale
// with multiplication by inv_filter_scale.
let inv_filter_scale = 1.0 / filter_scale;
// with multiplication by recip_filter_scale
let recip_filter_scale = 1.0 / filter_scale;
let count_of_coeffs = window_size * out_size as usize;
let mut coeffs: Vec<f64> = Vec::with_capacity(count_of_coeffs);
@@ -91,7 +91,7 @@ pub fn precompute_coefficients(
let center = in_center - 0.5;
for x in x_min..x_max {
let w: f64 = filter((x as f64 - center) * inv_filter_scale);
let w: f64 = filter((x as f64 - center) * recip_filter_scale);
coeffs.push(w);
ww += w;
}