Slightly improved speed of Convolution implementation for U8x2 images and Wasm32 SIMD128 instructions.

This commit is contained in:
Kirill Kuzminykh
2023-02-21 00:09:21 +03:00
parent 94070bbdba
commit 999c684fce
9 changed files with 92 additions and 180 deletions
+5
View File
@@ -1,3 +1,8 @@
## [Unreleased] - ReleaseDate
- Slightly improved speed of `Convolution` implementation for `U8x2` images
and `Wasm32 SIMD128` instructions.
## [2.5.0] - 2023-01-29
### Crate
+5 -1
View File
@@ -106,10 +106,14 @@ opt-level = 3
opt-level = 3
#incremental = true
lto = true
codegen-units = 1
#codegen-units = 1
strip = true
[profile.release.package.fast_image_resize]
codegen-units = 1
[profile.test]
opt-level = 3
+1 -1
View File
@@ -27,7 +27,7 @@ Supported pixel formats and available optimisations:
Resizer from this crate does not convert image into linear colorspace
during resize process. If it is important for you to resize images with a
non-linear color space (e.g. sRGB) correctly, then you hove to convert
non-linear color space (e.g. sRGB) correctly, then you have to convert
it to a linear color space before resizing and convert back to the color space of
result image. [Read more](https://legacy.imagemagick.org/Usage/resize/#resize_colorspace)
about resizing with respect to color space.
+2 -2
View File
@@ -5,8 +5,8 @@ Environment:
- CPU: AMD Ryzen 9 5950X
- RAM: DDR4 3800 MHz
- Ubuntu 22.04 (linux 5.15.0)
- Rust 1.67
- wasmtime = "5.0.0"
- Rust 1.67.1
- wasmtime = "6.0.0"
- criterion = "0.4"
- fast_image_resize = "2.5.0"
+53 -139
View File
@@ -64,18 +64,12 @@ unsafe fn horiz_convolution_four_rows(
A: |-1 07| |-1 05| |-1 03| |-1 01|
L: |-1 06| |-1 04| |-1 02| |-1 00|
*/
#[rustfmt::skip]
const SH1: v128 = i8x16(
0, -1, 2, -1, 4, -1, 6, -1, 1, -1, 3, -1, 5, -1, 7, -1
);
const SH1: v128 = i8x16(0, -1, 2, -1, 4, -1, 6, -1, 1, -1, 3, -1, 5, -1, 7, -1);
/*
A: |-1 15| |-1 13| |-1 11| |-1 09|
L: |-1 14| |-1 12| |-1 10| |-1 08|
*/
#[rustfmt::skip]
const SH2: v128 = i8x16(
8, -1, 10, -1, 12, -1, 14, -1, 9, -1, 11, -1, 13, -1, 15, -1
);
const SH2: v128 = i8x16(8, -1, 10, -1, 12, -1, 14, -1, 9, -1, 11, -1, 13, -1, 15, -1);
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x = coeffs_chunk.start as usize;
@@ -102,7 +96,6 @@ unsafe fn horiz_convolution_four_rows(
let coeffs_by_4 = reminder.chunks_exact(4);
let reminder = coeffs_by_4.remainder();
for k in coeffs_by_4 {
let mmk = wasm32_utils::ptr_i16_to_set1_i64(k, 0);
@@ -116,7 +109,6 @@ unsafe fn horiz_convolution_four_rows(
let coeffs_by_2 = reminder.chunks_exact(2);
let reminder = coeffs_by_2.remainder();
for k in coeffs_by_2 {
let mmk = wasm32_utils::ptr_i16_to_set1_i32(k, 0);
@@ -144,23 +136,6 @@ unsafe fn horiz_convolution_four_rows(
}
}
#[inline]
#[target_feature(enable = "simd128")]
unsafe fn set_dst_pixel(
raw: v128,
d_row: &mut &mut [U8x2],
dst_x: usize,
normalizer: &optimisations::Normalizer16,
) {
let l32x2 = i64x2_extract_lane::<0>(raw);
let a32x2 = i64x2_extract_lane::<1>(raw);
let l32 = ((l32x2 >> 32) as i32).saturating_add((l32x2 & 0xffffffff) as i32);
let a32 = ((a32x2 >> 32) as i32).saturating_add((a32x2 & 0xffffffff) as i32);
let l8 = normalizer.clip(l32);
let a8 = normalizer.clip(a32);
d_row.get_unchecked_mut(dst_x).0 = u16::from_le_bytes([l8, a8]);
}
/// For safety, it is necessary to ensure the following conditions:
/// - bounds.len() == dst_row.len()
/// - coeffs.len() == dst_rows.0.len() * window_size
@@ -174,103 +149,35 @@ unsafe fn horiz_convolution_one_row(
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
normalizer: &optimisations::Normalizer16,
) {
const SH1: v128 = i8x16(0, -1, 2, -1, 4, -1, 6, -1, 1, -1, 3, -1, 5, -1, 7, -1);
/*
A: |-1 15| |-1 13| |-1 11| |-1 09|
L: |-1 14| |-1 12| |-1 10| |-1 08|
*/
const SH2: v128 = i8x16(8, -1, 10, -1, 12, -1, 14, -1, 9, -1, 11, -1, 13, -1, 15, -1);
// Lower part will be added to higher, use only half of the error
let precision = normalizer.precision();
/*
|L A | |L A | |L A | |L A | |L A | |L A | |L A | |L A |
|00 01| |02 03| |04 05| |06 07| |08 09| |10 11| |12 13| |14 15|
Scale first four pixels into i16:
A: |-1 07| |-1 05|
L: |-1 06| |-1 04|
A: |-1 03| |-1 01|
L: |-1 02| |-1 00|
*/
#[rustfmt::skip]
const PIX_SH1: v128 = i8x16(
0, -1, 2, -1, 1, -1, 3, -1, 4, -1, 6, -1, 5, -1, 7, -1
);
/*
|C0 | |C1 | |C2 | |C3 | |C4 | |C5 | |C6 | |C7 |
|00 01| |02 03| |04 05| |06 07| |08 09| |10 11| |12 13| |14 15|
Duplicate first four coefficients for A and L components of pixels:
CA: |07 06| |05 04|
CL: |07 06| |05 04|
CA: |03 02| |01 00|
CL: |03 02| |01 00|
*/
#[rustfmt::skip]
const COEFF_SH1: v128 = i8x16(
0, 1, 2, 3, 0, 1, 2, 3, 4, 5, 6, 7, 4, 5, 6, 7
);
/*
|L A | |L A | |L A | |L A | |L A | |L A | |L A | |L A |
|00 01| |02 03| |04 05| |06 07| |08 09| |10 11| |12 13| |14 15|
Scale second four pixels into i16:
A: |-1 15| |-1 13|
L: |-1 14| |-1 12|
A: |-1 11| |-1 09|
L: |-1 10| |-1 08|
*/
#[rustfmt::skip]
const PIX_SH2: v128 = i8x16(
8, -1, 10, -1, 9, -1, 11, -1, 12, -1, 14, -1, 13, -1, 15, -1
);
/*
|C0 | |C1 | |C2 | |C3 | |C4 | |C5 | |C6 | |C7 |
|00 01| |02 03| |04 05| |06 07| |08 09| |10 11| |12 13| |14 15|
Duplicate second four coefficients for A and L components of pixels:
CA: |15 14| |13 12|
CL: |15 14| |13 12|
CA: |11 10| |09 08|
CL: |11 10| |09 08|
*/
#[rustfmt::skip]
const COEFF_SH2: v128 = i8x16(
8, 9, 10, 11, 8, 9, 10, 11, 12, 13, 14, 15, 12, 13, 14, 15
);
/*
|L A | |L A | |L A | |L A |
|00 01| |02 03| |04 05| |06 07| |08 09| |10 11| |12 13| |14 15|
Scale four pixels into i16:
A: |-1 07| |-1 05|
L: |-1 06| |-1 04|
A: |-1 03| |-1 01|
L: |-1 02| |-1 00|
*/
const PIX_SH3: v128 = i8x16(0, -1, 2, -1, 1, -1, 3, -1, 4, -1, 6, -1, 5, -1, 7, -1);
let initial = i32x4_splat(1 << (precision - 2));
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x = coeffs_chunk.start as usize;
let mut coeffs = coeffs_chunk.values;
// Lower part will be added to higher, use only half of the error
let mut sss = i32x4_splat(1 << (precision - 2));
let mut sss = initial;
let coeffs_by_8 = coeffs.chunks_exact(8);
coeffs = coeffs_by_8.remainder();
for k in coeffs_by_8 {
let ksource = wasm32_utils::load_v128(k, 0);
let mmk0 = wasm32_utils::ptr_i16_to_set1_i64(k, 0);
let mmk1 = wasm32_utils::ptr_i16_to_set1_i64(k, 4);
let source = wasm32_utils::load_v128(src_row, x);
let pix = i8x16_swizzle(source, PIX_SH1);
let mmk = i8x16_swizzle(ksource, COEFF_SH1);
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
let pix = i8x16_swizzle(source, PIX_SH2);
let mmk = i8x16_swizzle(ksource, COEFF_SH2);
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
let pix = i8x16_swizzle(source, SH1);
let tmp_sum = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk0));
let pix = i8x16_swizzle(source, SH2);
sss = i32x4_add(tmp_sum, i32x4_dot_i16x8(pix, mmk1));
x += 8;
}
@@ -279,42 +186,49 @@ unsafe fn horiz_convolution_one_row(
let reminder1 = coeffs_by_4.remainder();
for k in coeffs_by_4 {
let mmk = i16x8(k[0], k[1], k[0], k[1], k[2], k[3], k[2], k[3]);
let mmk = wasm32_utils::ptr_i16_to_set1_i64(k, 0);
let source = wasm32_utils::loadl_i64(src_row, x);
let pix = i8x16_swizzle(source, PIX_SH3);
let pix = i8x16_swizzle(source, SH1);
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
x += 4
}
if !reminder1.is_empty() {
let mut pixels: [i16; 6] = [0; 6];
let mut coeffs: [i16; 3] = [0; 3];
for (i, &coeff) in reminder1.iter().enumerate() {
coeffs[i] = coeff;
let pixel: [u8; 2] = (*src_row.get_unchecked(x)).0.to_le_bytes();
pixels[i * 2] = pixel[0] as i16;
pixels[i * 2 + 1] = pixel[1] as i16;
x += 1;
}
let coeffs_by_2 = reminder1.chunks_exact(2);
let reminder = coeffs_by_2.remainder();
for k in coeffs_by_2 {
let mmk = wasm32_utils::ptr_i16_to_set1_i32(k, 0);
let pix = i16x8(
pixels[0], pixels[2], pixels[1], pixels[3], pixels[4], 0, pixels[5], 0,
);
let mmk = i16x8(
coeffs[0], coeffs[1], coeffs[0], coeffs[1], coeffs[2], 0, coeffs[2], 0,
);
let source = wasm32_utils::loadl_i32(src_row, x);
let pix = i8x16_swizzle(source, SH1);
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
x += 2;
}
if let Some(&k) = reminder.first() {
let mmk = i32x4_splat(k as i32);
let source = wasm32_utils::loadl_i16(src_row, x);
let pix = i8x16_swizzle(source, SH1);
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
}
let lo = i64x2_extract_lane::<0>(sss);
let hi = i64x2_extract_lane::<1>(sss);
let a32 = ((lo >> 32) as i32).saturating_add((hi >> 32) as i32);
let l32 = ((lo & 0xffffffff) as i32).saturating_add((hi & 0xffffffff) as i32);
let a8 = normalizer.clip(a32);
let l8 = normalizer.clip(l32);
dst_row.get_unchecked_mut(dst_x).0 = u16::from_le_bytes([l8, a8]);
set_dst_pixel(sss, dst_row, dst_x, normalizer);
}
}
#[inline]
#[target_feature(enable = "simd128")]
unsafe fn set_dst_pixel(
raw: v128,
d_row: &mut [U8x2],
dst_x: usize,
normalizer: &optimisations::Normalizer16,
) {
let mut buf = [0i32; 4];
v128_store(buf.as_mut_ptr() as _, raw);
let l32 = buf[0].saturating_add(buf[1]);
let a32 = buf[2].saturating_add(buf[3]);
let l8 = normalizer.clip(l32);
let a8 = normalizer.clip(a32);
d_row.get_unchecked_mut(dst_x).0 = u16::from_le_bytes([l8, a8]);
}
+1 -1
View File
@@ -26,7 +26,7 @@ pub(crate) fn vert_convolution<T: PixelExt<Component = u8>>(
}
#[target_feature(enable = "sse4.1")]
pub(crate) unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
src_img: &ImageView<T>,
dst_row: &mut [T],
mut src_x: usize,
+14 -29
View File
@@ -25,8 +25,9 @@ pub(crate) fn vert_convolution<T: PixelExt<Component = u8>>(
}
}
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
src_img: &ImageView<T>,
dst_row: &mut [T],
mut src_x: usize,
@@ -37,7 +38,7 @@ pub(crate) unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8
let y_start = coeffs_chunk.start;
let coeffs = coeffs_chunk.values;
let max_y = y_start + coeffs.len() as u32;
let precision = normalizer.precision();
let precision = normalizer.precision() as u32;
let mut dst_u8 = T::components_mut(dst_row);
let initial = i32x4_splat(1 << (precision - 1));
@@ -143,19 +144,14 @@ pub(crate) unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8
sss7 = i32x4_add(sss7, i32x4_dot_i16x8(pix, mmk));
}
macro_rules! call {
($imm8:expr) => {{
sss0 = i32x4_shr(sss0, $imm8);
sss1 = i32x4_shr(sss1, $imm8);
sss2 = i32x4_shr(sss2, $imm8);
sss3 = i32x4_shr(sss3, $imm8);
sss4 = i32x4_shr(sss4, $imm8);
sss5 = i32x4_shr(sss5, $imm8);
sss6 = i32x4_shr(sss6, $imm8);
sss7 = i32x4_shr(sss7, $imm8);
}};
}
constify_imm8!(precision, call);
sss0 = i32x4_shr(sss0, precision);
sss1 = i32x4_shr(sss1, precision);
sss2 = i32x4_shr(sss2, precision);
sss3 = i32x4_shr(sss3, precision);
sss4 = i32x4_shr(sss4, precision);
sss5 = i32x4_shr(sss5, precision);
sss6 = i32x4_shr(sss6, precision);
sss7 = i32x4_shr(sss7, precision);
sss0 = i16x8_narrow_i32x4(sss0, sss1);
sss2 = i16x8_narrow_i32x4(sss2, sss3);
@@ -214,13 +210,8 @@ pub(crate) unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8
sss1 = i32x4_add(sss1, i32x4_dot_i16x8(pix, mmk));
}
macro_rules! call {
($imm8:expr) => {{
sss0 = i32x4_shr(sss0, $imm8);
sss1 = i32x4_shr(sss1, $imm8);
}};
}
constify_imm8!(precision, call);
sss0 = i32x4_shr(sss0, precision);
sss1 = i32x4_shr(sss1, precision);
sss0 = i16x8_narrow_i32x4(sss0, sss1);
sss0 = u8x16_narrow_i16x8(sss0, sss0);
@@ -262,13 +253,7 @@ pub(crate) unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
}
macro_rules! call {
($imm8:expr) => {{
sss = i32x4_shr(sss, $imm8);
}};
}
constify_imm8!(precision, call);
sss = i32x4_shr(sss, precision);
sss = i16x8_narrow_i32x4(sss, sss);
let dst_ptr = dst_chunk.as_mut_ptr() as *mut i32;
*dst_ptr = i32x4_extract_lane::<0>(u8x16_narrow_i16x8(sss, sss));
+10 -5
View File
@@ -1,4 +1,5 @@
use std::arch::wasm32::*;
use std::ptr;
use crate::pixels::{U8x3, U8x4};
@@ -12,33 +13,37 @@ pub(crate) unsafe fn load_v128<T>(buf: &[T], index: usize) -> v128 {
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn loadl_i64<T>(buf: &[T], index: usize) -> v128 {
let i = buf.get_unchecked(index..).as_ptr() as *const i64;
i64x2(*i, 0)
i64x2(ptr::read_unaligned(i), 0)
}
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn loadl_i32<T>(buf: &[T], index: usize) -> v128 {
let i = buf.get_unchecked(index..).as_ptr() as *const i32;
i32x4(*i, 0, 0, 0)
i32x4(ptr::read_unaligned(i), 0, 0, 0)
}
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn loadl_i16<T>(buf: &[T], index: usize) -> v128 {
let i = buf.get_unchecked(index..).as_ptr() as *const i16;
i16x8(*i, 0, 0, 0, 0, 0, 0, 0)
i16x8(ptr::read_unaligned(i), 0, 0, 0, 0, 0, 0, 0)
}
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn ptr_i16_to_set1_i64(buf: &[i16], index: usize) -> v128 {
i64x2_splat(*(buf.get_unchecked(index..).as_ptr() as *const i64))
i64x2_splat(ptr::read_unaligned(
buf.get_unchecked(index..).as_ptr() as *const i64
))
}
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn ptr_i16_to_set1_i32(buf: &[i16], index: usize) -> v128 {
i32x4_splat(*(buf.get_unchecked(index..).as_ptr() as *const i32))
i32x4_splat(ptr::read_unaligned(
buf.get_unchecked(index..).as_ptr() as *const i32
))
}
#[inline]
+1 -2
View File
@@ -303,8 +303,7 @@ where
assert_eq!(
testing::image_checksum::<Self, CC>(&result),
checksum,
"Error in checksum for {:?}",
cpu_extensions
"Error in checksum for {cpu_extensions:?}",
);
}