Wasm simd for u16x1.

This commit is contained in:
Colin
2023-01-23 16:46:38 -05:00
committed by Colin Murphy
parent b3d03d03ef
commit 11fbe03a8f
13 changed files with 743 additions and 792 deletions
+6
View File
@@ -12,6 +12,8 @@ mod native;
mod neon;
#[cfg(target_arch = "x86_64")]
mod sse4;
#[cfg(target_arch = "wasm32")]
mod wasm32;
impl Convolution for U16 {
fn horiz_convolution(
@@ -28,6 +30,10 @@ impl Convolution for U16 {
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => neon::horiz_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => {
wasm32::horiz_convolution(src_image, dst_image, offset, coeffs)
}
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
}
}
+297
View File
@@ -0,0 +1,297 @@
use std::arch::wasm32::*;
use crate::convolution::{optimisations, Coefficients};
use crate::pixels::U16;
use crate::wasm32_utils;
use crate::{ImageView, ImageViewMut};
/*
use std::fs;
use std::path::Path;
*/
#[inline]
pub(crate) fn horiz_convolution(
src_image: &ImageView<U16>,
dst_image: &mut ImageViewMut<U16>,
offset: u32,
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer32::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_image.height().get();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_image.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
}
}
let mut yy = dst_height - dst_height % 4;
while yy < dst_height {
unsafe {
horiz_convolution_one_row(
src_image.get_row(yy + offset).unwrap(),
dst_image.get_row_mut(yy).unwrap(),
&coefficients_chunks,
&normalizer,
);
}
yy += 1;
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - length of all rows in src_rows must be equal
/// - length of all rows in dst_rows must be equal
/// - coefficients_chunks.len() == dst_rows.0.len()
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
#[target_feature(enable = "simd128")]
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U16]; 4],
dst_rows: [&mut &mut [U16]; 4],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
normalizer: &optimisations::Normalizer32,
) {
let precision = normalizer.precision();
let half_error = 1i64 << (precision - 1);
let mut ll_buf = [0i64; 2];
/*
|L0 | |L1 | |L2 | |L3 | |L4 | |L5 | |L6 | |L7 |
|0001| |0203| |0405| |0607| |0809| |1011| |1213| |1415|
Shuffle to extract L0 and L1 as i64:
-1, -1, -1, -1, -1, -1, 3, 2, -1, -1, -1, -1, -1, -1, 1, 0
Shuffle to extract L2 and L3 as i64:
-1, -1, -1, -1, -1, -1, 7, 6, -1, -1, -1, -1, -1, -1, 5, 4
Shuffle to extract L4 and L5 as i64:
-1, -1, -1, -1, -1, -1, 11, 10, -1, -1, -1, -1, -1, -1, 9, 8
Shuffle to extract L6 and L7 as i64:
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12
*/
let l0l1_shuffle = i8x16(-1, -1, -1, -1, -1, -1, 3, 2, -1, -1, -1, -1, -1, -1, 1, 0);
let l2l3_shuffle = i8x16(-1, -1, -1, -1, -1, -1, 7, 6, -1, -1, -1, -1, -1, -1, 5, 4);
let l4l5_shuffle = i8x16(-1, -1, -1, -1, -1, -1, 11, 10, -1, -1, -1, -1, -1, -1, 9, 8);
let l6l7_shuffle = i8x16(
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12,
);
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
let mut ll_sum: [v128; 4] = [i64x2(0i64, 0i64); 4];
let mut coeffs = coeffs_chunk.values;
let coeffs_by_8 = coeffs.chunks_exact(8);
coeffs = coeffs_by_8.remainder();
for k in coeffs_by_8 {
let coeff01_i64x2 = i64x2(k[0] as i64, k[1] as i64);
let coeff23_i64x2 = i64x2(k[2] as i64, k[3] as i64);
let coeff45_i64x2 = i64x2(k[4] as i64, k[5] as i64);
let coeff67_i64x2 = i64x2(k[6] as i64, k[7] as i64);
for i in 0..4 {
let mut sum = ll_sum[i];
let source = wasm32_utils::load_v128(src_rows[i], x);
let l0l1_i64x2 = i8x16_swizzle(source, l0l1_shuffle);
sum = i64x2_add(sum, i64x2_mul(l0l1_i64x2, coeff01_i64x2));
let l2l3_i64x2 = i8x16_swizzle(source, l2l3_shuffle);
sum = i64x2_add(sum, i64x2_mul(l2l3_i64x2, coeff23_i64x2));
let l4l5_i64x2 = i8x16_swizzle(source, l4l5_shuffle);
sum = i64x2_add(sum, i64x2_mul(l4l5_i64x2, coeff45_i64x2));
let l6l7_i64x2 = i8x16_swizzle(source, l6l7_shuffle);
sum = i64x2_add(sum, i64x2_mul(l6l7_i64x2, coeff67_i64x2));
ll_sum[i] = sum;
}
x += 8;
}
let coeffs_by_4 = coeffs.chunks_exact(4);
coeffs = coeffs_by_4.remainder();
for k in coeffs_by_4 {
let coeff01_i64x2 = i64x2(k[0] as i64, k[1] as i64);
let coeff23_i64x2 = i64x2(k[2] as i64, k[3] as i64);
for i in 0..4 {
let mut sum = ll_sum[i];
let source = wasm32_utils::load_v128(src_rows[i], x);
let l0l1_i64x2 = i8x16_swizzle(source, l0l1_shuffle);
sum = i64x2_add(sum, i64x2_mul(l0l1_i64x2, coeff01_i64x2));
let l2l3_i64x2 = i8x16_swizzle(source, l2l3_shuffle);
sum = i64x2_add(sum, i64x2_mul(l2l3_i64x2, coeff23_i64x2));
ll_sum[i] = sum;
}
x += 4;
}
let coeffs_by_2 = coeffs.chunks_exact(2);
coeffs = coeffs_by_2.remainder();
for k in coeffs_by_2 {
let coeff01_i64x2 = i64x2(k[0] as i64, k[1] as i64);
for i in 0..4 {
let source = wasm32_utils::load_v128(src_rows[i], x);
let l_i64x2 = i8x16_swizzle(source, l0l1_shuffle);
ll_sum[i] = i64x2_add(ll_sum[i], i64x2_mul(l_i64x2, coeff01_i64x2));
}
x += 2;
}
if let Some(&k) = coeffs.first() {
let coeff01_i64x2 = i64x2(k as i64, 0);
for i in 0..4 {
let pixel = (*src_rows[i].get_unchecked(x)).0 as i64;
let source = i64x2(pixel, 0);
ll_sum[i] = i64x2_add(ll_sum[i], i64x2_mul(source, coeff01_i64x2));
}
}
for i in 0..4 {
v128_store((&mut ll_buf).as_mut_ptr() as *mut v128, ll_sum[i]);
let dst_pixel = dst_rows[i].get_unchecked_mut(dst_x);
dst_pixel.0 = normalizer.clip(ll_buf.iter().sum::<i64>() + half_error);
}
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - bounds.len() == dst_row.len()
/// - coefficients_chunks.len() == dst_row.len()
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
#[target_feature(enable = "simd128")]
unsafe fn horiz_convolution_one_row(
src_row: &[U16],
dst_row: &mut [U16],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
normalizer: &optimisations::Normalizer32,
) {
/*
let file = "wasm32";
let file_exists = Path::new(file).exists();
if file_exists {
fs::write(file, format!("src_row: {:?}", src_row)).unwrap();
}
*/
let precision = normalizer.precision();
let half_error = 1i64 << (precision - 1);
let mut ll_buf = [0i64; 2];
/*
|L0 | |L1 | |L2 | |L3 | |L4 | |L5 | |L6 | |L7 |
|0001| |0203| |0405| |0607| |0809| |1011| |1213| |1415|
Shuffle to extract L0 and L1 as i64:
-1, -1, -1, -1, -1, -1, 0, 1, -1, -1, -1, -1, -1, -1, 2, 3
Shuffle to extract L2 and L3 as i64:
-1, -1, -1, -1, -1, -1, 4, 5, -1, -1, -1, -1, -1, -1, 6, 7
Shuffle to extract L4 and L5 as i64:
-1, -1, -1, -1, -1, -1, 8, 9, -1, -1, -1, -1, -1, -1, 10, 11
Shuffle to extract L6 and L7 as i64:
-1, -1, -1, -1, -1, -1, 12, 13, -1, -1, -1, -1, -1, -1, 14, 15
*/
let l01_shuffle = i8x16(-1, -1, -1, -1, -1, -1, 0, 1, -1, -1, -1, -1, -1, -1, 0, 1);
let l23_shuffle = i8x16(-1, -1, -1, -1, -1, -1, 4, 5, -1, -1, -1, -1, -1, -1, 6, 7);
let l45_shuffle = i8x16(-1, -1, -1, -1, -1, -1, 8, 9, -1, -1, -1, -1, -1, -1, 10, 11);
let l67_shuffle = i8x16(
-1, -1, -1, -1, -1, -1, 12, 13, -1, -1, -1, -1, -1, -1, 14, 15,
);
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
let mut ll_sum = i64x2(0, 0);
let mut coeffs = coeffs_chunk.values;
let coeffs_by_8 = coeffs.chunks_exact(8);
coeffs = coeffs_by_8.remainder();
for k in coeffs_by_8 {
let coeff01_i64x2 = i64x2(k[0] as i64, k[1] as i64);
let coeff23_i64x2 = i64x2(k[2] as i64, k[3] as i64);
let coeff45_i64x2 = i64x2(k[4] as i64, k[5] as i64);
let coeff67_i64x2 = i64x2(k[6] as i64, k[7] as i64);
let source = wasm32_utils::load_v128(src_row, x);
let l_i64x2 = i8x16_swizzle(source, l01_shuffle);
ll_sum = i64x2_add(ll_sum, i64x2_mul(l_i64x2, coeff01_i64x2));
let l_i64x2 = i8x16_swizzle(source, l23_shuffle);
ll_sum = i64x2_add(ll_sum, i64x2_mul(l_i64x2, coeff23_i64x2));
let l_i64x2 = i8x16_swizzle(source, l45_shuffle);
ll_sum = i64x2_add(ll_sum, i64x2_mul(l_i64x2, coeff45_i64x2));
let l_i64x2 = i8x16_swizzle(source, l67_shuffle);
ll_sum = i64x2_add(ll_sum, i64x2_mul(l_i64x2, coeff67_i64x2));
x += 8;
}
let coeffs_by_4 = coeffs.chunks_exact(4);
coeffs = coeffs_by_4.remainder();
for k in coeffs_by_4 {
let coeff01_i64x2 = i64x2(k[0] as i64, k[1] as i64);
let coeff23_i64x2 = i64x2(k[2] as i64, k[3] as i64);
let source = wasm32_utils::load_v128(src_row, x);
let l_i64x2 = i8x16_swizzle(source, l01_shuffle);
ll_sum = i64x2_add(ll_sum, i64x2_mul(l_i64x2, coeff01_i64x2));
let l_i64x2 = i8x16_swizzle(source, l23_shuffle);
ll_sum = i64x2_add(ll_sum, i64x2_mul(l_i64x2, coeff23_i64x2));
x += 4;
}
let coeffs_by_2 = coeffs.chunks_exact(2);
coeffs = coeffs_by_2.remainder();
for k in coeffs_by_2 {
let coeff01_i64x2 = i64x2(k[0] as i64, k[1] as i64);
let source = wasm32_utils::load_v128(src_row, x);
let l_i64x2 = i8x16_swizzle(source, l01_shuffle);
ll_sum = i64x2_add(ll_sum, i64x2_mul(l_i64x2, coeff01_i64x2));
x += 2;
}
if let Some(&k) = coeffs.first() {
let coeff01_i64x2 = i64x2(k as i64, 0);
let pixel = (*src_row.get_unchecked(x)).0 as i64;
let source = i64x2(0, pixel);
ll_sum = i64x2_add(ll_sum, i64x2_mul(source, coeff01_i64x2));
}
v128_store((&mut ll_buf).as_mut_ptr() as *mut v128, ll_sum);
let dst_pixel = dst_row.get_unchecked_mut(dst_x);
dst_pixel.0 = normalizer.clip(ll_buf[0] + ll_buf[1] + half_error);
}
/*
if file_exists {
fs::write(file, format!("dst_row: {:?}", dst_row)).unwrap();
}
*/
}
+4
View File
@@ -28,6 +28,10 @@ impl Convolution for U16x2 {
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => neon::horiz_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => {
native::horiz_convolution(src_image, dst_image, offset, coeffs)
}
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
}
}
+4
View File
@@ -28,6 +28,10 @@ impl Convolution for U16x3 {
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => neon::horiz_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => {
native::horiz_convolution(src_image, dst_image, offset, coeffs)
}
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
}
}
+4
View File
@@ -28,6 +28,10 @@ impl Convolution for U16x4 {
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => neon::horiz_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => {
native::horiz_convolution(src_image, dst_image, offset, coeffs)
}
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
}
}
+6
View File
@@ -12,6 +12,8 @@ mod native;
mod neon;
#[cfg(target_arch = "x86_64")]
mod sse4;
#[cfg(target_arch = "wasm32")]
mod wasm32;
impl Convolution for U8x2 {
fn horiz_convolution(
@@ -28,6 +30,10 @@ impl Convolution for U8x2 {
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => neon::horiz_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => {
wasm32::horiz_convolution(src_image, dst_image, offset, coeffs)
}
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
}
}
+320
View File
@@ -0,0 +1,320 @@
use std::arch::wasm32::*;
use crate::convolution::{optimisations, Coefficients};
use crate::pixels::U8x2;
use crate::wasm32_utils;
use crate::{ImageView, ImageViewMut};
#[inline]
pub(crate) fn horiz_convolution(
src_image: &ImageView<U8x2>,
dst_image: &mut ImageViewMut<U8x2>,
offset: u32,
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer16::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_image.height().get();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_image.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
}
}
let mut yy = dst_height - dst_height % 4;
while yy < dst_height {
unsafe {
horiz_convolution_one_row(
src_image.get_row(yy + offset).unwrap(),
dst_image.get_row_mut(yy).unwrap(),
&coefficients_chunks,
&normalizer,
);
}
yy += 1;
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - length of all rows in src_rows must be equal
/// - length of all rows in dst_rows must be equal
/// - coefficients_chunks.len() == dst_rows.0.len()
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
#[inline]
#[target_feature(enable = "simd128")]
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U8x2]; 4],
dst_rows: [&mut &mut [U8x2]; 4],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
normalizer: &optimisations::Normalizer16,
) {
let precision = normalizer.precision();
let initial = i32x4_splat(1 << (precision - 2));
/*
|L A | |L A | |L A | |L A | |L A | |L A | |L A | |L A |
|00 01| |02 03| |04 05| |06 07| |08 09| |10 11| |12 13| |14 15|
Shuffle components with converting from u8 into i16:
A: |-1 07| |-1 05| |-1 03| |-1 01|
L: |-1 06| |-1 04| |-1 02| |-1 00|
*/
#[rustfmt::skip]
let sh1 = i8x16(
-1, 7, -1, 5, -1, 3, -1, 1, -1, 6, -1, 4, -1, 2, -1, 0,
);
/*
A: |-1 15| |-1 13| |-1 11| |-1 09|
L: |-1 14| |-1 12| |-1 10| |-1 08|
*/
#[rustfmt::skip]
let sh2 = i8x16(
-1, 15, -1, 13, -1, 11, -1, 9, -1, 14, -1, 12, -1, 10, -1, 8,
);
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x = coeffs_chunk.start as usize;
let mut sss: [v128; 4] = [initial; 4];
let coeffs = coeffs_chunk.values;
let coeffs_by_8 = coeffs.chunks_exact(8);
let reminder = coeffs_by_8.remainder();
for k in coeffs_by_8 {
let mmk0 = wasm32_utils::ptr_i16_to_set1_i64(k, 0);
let mmk1 = wasm32_utils::ptr_i16_to_set1_i64(k, 4);
for i in 0..4 {
let source = wasm32_utils::load_v128(src_rows[i], x);
let pix = wasm32_utils::i8x16_shuffle(source, sh1);
let tmp_sum = i32x4_add(sss[i], i32x4_dot_i16x8(pix, mmk0));
let pix = wasm32_utils::i8x16_shuffle(source, sh2);
sss[i] = i32x4_add(tmp_sum, i32x4_dot_i16x8(pix, mmk1));
}
x += 8;
}
let coeffs_by_4 = reminder.chunks_exact(4);
let reminder = coeffs_by_4.remainder();
for k in coeffs_by_4 {
let mmk = wasm32_utils::ptr_i16_to_set1_i64(k, 0);
for i in 0..4 {
let source = wasm32_utils::loadl_i64(src_rows[i], x);
let pix = wasm32_utils::i8x16_shuffle(source, sh1);
sss[i] = i32x4_add(sss[i], i32x4_dot_i16x8(pix, mmk));
}
x += 4;
}
let coeffs_by_2 = reminder.chunks_exact(2);
let reminder = coeffs_by_2.remainder();
for k in coeffs_by_2 {
let mmk = wasm32_utils::ptr_i16_to_set1_i32(k, 0);
for i in 0..4 {
let source = wasm32_utils::loadl_i32(src_rows[i], x);
let pix = wasm32_utils::i8x16_shuffle(source, sh1);
sss[i] = i32x4_add(sss[i], i32x4_dot_i16x8(pix, mmk));
}
x += 2;
}
if let Some(&k) = reminder.first() {
let mmk = i32x4(k as i32, 0, 0, 0);
for i in 0..4 {
let source = wasm32_utils::loadl_i16(src_rows[i], x);
let pix = wasm32_utils::i8x16_shuffle(source, sh1);
sss[i] = i32x4_add(sss[i], i32x4_dot_i16x8(pix, mmk));
}
}
for i in 0..4 {
set_dst_pixel(sss[i], dst_rows[i], dst_x, normalizer);
}
}
}
#[inline]
#[target_feature(enable = "simd128")]
unsafe fn set_dst_pixel(
raw: v128,
d_row: &mut &mut [U8x2],
dst_x: usize,
normalizer: &optimisations::Normalizer16,
) {
let l32x2 = i64x2_extract_lane::<0>(raw);
let a32x2 = i64x2_extract_lane::<1>(raw);
let l32 = ((l32x2 >> 32) as i32).saturating_add((l32x2 & 0xffffffff) as i32);
let a32 = ((a32x2 >> 32) as i32).saturating_add((a32x2 & 0xffffffff) as i32);
let l8 = normalizer.clip(l32);
let a8 = normalizer.clip(a32);
d_row.get_unchecked_mut(dst_x).0 = u16::from_le_bytes([l8, a8]);
}
/// For safety, it is necessary to ensure the following conditions:
/// - bounds.len() == dst_row.len()
/// - coeffs.len() == dst_rows.0.len() * window_size
/// - max(bound.start + bound.size for bound in bounds) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
#[inline]
#[target_feature(enable = "simd128")]
unsafe fn horiz_convolution_one_row(
src_row: &[U8x2],
dst_row: &mut [U8x2],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
normalizer: &optimisations::Normalizer16,
) {
let precision = normalizer.precision();
/*
|L A | |L A | |L A | |L A | |L A | |L A | |L A | |L A |
|00 01| |02 03| |04 05| |06 07| |08 09| |10 11| |12 13| |14 15|
Scale first four pixels into i16:
A: |-1 07| |-1 05|
L: |-1 06| |-1 04|
A: |-1 03| |-1 01|
L: |-1 02| |-1 00|
*/
#[rustfmt::skip]
let pix_sh1 = i8x16(
-1, 7, -1, 5, -1, 6, -1, 4, -1, 3, -1, 1, -1, 2, -1, 0,
);
/*
|C0 | |C1 | |C2 | |C3 | |C4 | |C5 | |C6 | |C7 |
|00 01| |02 03| |04 05| |06 07| |08 09| |10 11| |12 13| |14 15|
Duplicate first four coefficients for A and L components of pixels:
CA: |07 06| |05 04|
CL: |07 06| |05 04|
CA: |03 02| |01 00|
CL: |03 02| |01 00|
*/
#[rustfmt::skip]
let coeff_sh1 = i8x16(
7, 6, 5, 4, 7, 6, 5, 4, 3, 2, 1, 0, 3,2, 1, 0,
);
/*
|L A | |L A | |L A | |L A | |L A | |L A | |L A | |L A |
|00 01| |02 03| |04 05| |06 07| |08 09| |10 11| |12 13| |14 15|
Scale second four pixels into i16:
A: |-1 15| |-1 13|
L: |-1 14| |-1 12|
A: |-1 11| |-1 09|
L: |-1 10| |-1 08|
*/
#[rustfmt::skip]
let pix_sh2 = i8x16(
-1, 15, -1, 13, -1, 14, -1, 12, -1, 11, -1, 9, -1, 10, -1, 8,
);
/*
|C0 | |C1 | |C2 | |C3 | |C4 | |C5 | |C6 | |C7 |
|00 01| |02 03| |04 05| |06 07| |08 09| |10 11| |12 13| |14 15|
Duplicate second four coefficients for A and L components of pixels:
CA: |15 14| |13 12|
CL: |15 14| |13 12|
CA: |11 10| |09 08|
CL: |11 10| |09 08|
*/
#[rustfmt::skip]
let coeff_sh2 = i8x16(
15, 14, 13, 12, 15, 14, 13, 12, 11, 10, 9, 8, 11, 10, 9, 8,
);
/*
|L A | |L A | |L A | |L A |
|00 01| |02 03| |04 05| |06 07| |08 09| |10 11| |12 13| |14 15|
Scale four pixels into i16:
A: |-1 07| |-1 05|
L: |-1 06| |-1 04|
A: |-1 03| |-1 01|
L: |-1 02| |-1 00|
*/
let pix_sh3 = i8x16(-1, 7, -1, 5, -1, 6, -1, 4, -1, 3, -1, 1, -1, 2, -1, 0);
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x = coeffs_chunk.start as usize;
let mut coeffs = coeffs_chunk.values;
// Lower part will be added to higher, use only half of the error
let mut sss = i32x4(1 << (precision - 2), 0, 0, 0);
let coeffs_by_8 = coeffs.chunks_exact(8);
coeffs = coeffs_by_8.remainder();
for k in coeffs_by_8 {
let ksource = wasm32_utils::load_v128(k, 0);
let source = wasm32_utils::load_v128(src_row, x);
let pix = wasm32_utils::i8x16_shuffle(source, pix_sh1);
let mmk = wasm32_utils::i8x16_shuffle(ksource, coeff_sh1);
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
let pix = wasm32_utils::i8x16_shuffle(source, pix_sh2);
let mmk = wasm32_utils::i8x16_shuffle(ksource, coeff_sh2);
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
x += 8;
}
let coeffs_by_4 = coeffs.chunks_exact(4);
let reminder1 = coeffs_by_4.remainder();
for k in coeffs_by_4 {
let mmk = i16x8(k[3], k[2], k[3], k[2], k[1], k[0], k[1], k[0]);
let source = wasm32_utils::loadl_i64(src_row, x);
let pix = wasm32_utils::i8x16_shuffle(source, pix_sh3);
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
x += 4
}
if !reminder1.is_empty() {
let mut pixels: [i16; 6] = [0; 6];
let mut coeffs: [i16; 3] = [0; 3];
for (i, &coeff) in reminder1.iter().enumerate() {
coeffs[i] = coeff;
let pixel: [u8; 2] = (*src_row.get_unchecked(x)).0.to_le_bytes();
pixels[i * 2] = pixel[0] as i16;
pixels[i * 2 + 1] = pixel[1] as i16;
x += 1;
}
let pix = i16x8(
0, pixels[5], 0, pixels[4], pixels[3], pixels[1], pixels[2], pixels[0],
);
let mmk = i16x8(
0, coeffs[2], 0, coeffs[2], coeffs[1], coeffs[0], coeffs[1], coeffs[0],
);
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
}
let lo = i64x2_extract_lane::<0>(sss);
let hi = i64x2_extract_lane::<1>(sss);
let a32 = ((lo >> 32) as i32).saturating_add((hi >> 32) as i32);
let l32 = ((lo & 0xffffffff) as i32).saturating_add((hi & 0xffffffff) as i32);
let a8 = normalizer.clip(a32);
let l8 = normalizer.clip(l32);
dst_row.get_unchecked_mut(dst_x).0 = u16::from_le_bytes([l8, a8]);
}
}
+2
View File
@@ -29,6 +29,8 @@ pub(crate) fn vert_convolution_u16<T: PixelExt<Component = u16>>(
CpuExtensions::Sse4_1 => sse4::vert_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => neon::vert_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => native::vert_convolution(src_image, dst_image, offset, coeffs),
_ => native::vert_convolution(src_image, dst_image, offset, coeffs),
}
}
+2
View File
@@ -34,3 +34,5 @@ mod resizer;
mod simd_utils;
#[cfg(feature = "for_test")]
pub mod testing;
#[cfg(target_arch = "wasm32")]
mod wasm32_utils;
+13 -1
View File
@@ -16,6 +16,8 @@ pub enum CpuExtensions {
Avx2,
#[cfg(target_arch = "aarch64")]
Neon,
#[cfg(target_arch = "wasm32")]
Wasm32,
}
impl CpuExtensions {
@@ -28,6 +30,8 @@ impl CpuExtensions {
Self::Sse4_1 => is_x86_feature_detected!("sse4.1"),
#[cfg(target_arch = "aarch64")]
Self::Neon => true,
#[cfg(target_arch = "wasm32")]
Self::Wasm32 => true,
Self::None => true,
}
}
@@ -54,8 +58,16 @@ impl Default for CpuExtensions {
Self::None
}
}
#[cfg(target_arch = "wasm32")]
fn default() -> Self {
Self::Wasm32
}
#[cfg(not(any(target_arch = "x86_64", target_arch = "aarch64")))]
#[cfg(not(any(
target_arch = "x86_64",
target_arch = "aarch64",
target_arch = "wasm32"
)))]
fn default() -> Self {
Self::None
}
+50
View File
@@ -0,0 +1,50 @@
use std::arch::wasm32::*;
#[inline(always)]
pub unsafe fn load_v128<T>(buf: &[T], index: usize) -> v128 {
v128_load(buf.get_unchecked(index..).as_ptr() as *const v128)
}
#[inline(always)]
pub unsafe fn loadl_i64<T>(buf: &[T], index: usize) -> v128 {
i64x2(buf.get_unchecked(index..).as_ptr() as i64, 0)
}
#[inline(always)]
pub unsafe fn loadl_i32<T>(buf: &[T], index: usize) -> v128 {
i32x4(buf.get_unchecked(index..).as_ptr() as i32, 0, 0, 0)
}
#[inline(always)]
pub unsafe fn loadl_i16<T>(buf: &[T], index: usize) -> v128 {
i16x8(
buf.get_unchecked(index..).as_ptr() as i16,
0,
0,
0,
0,
0,
0,
0,
)
}
#[inline(always)]
pub fn i8x16_shuffle(a: v128, b: v128) -> v128 {
u8x16_swizzle(a, v128_and(b, i8x16_splat(-113)))
}
#[inline(always)]
pub unsafe fn ptr_i16_to_set1_i64(buf: &[i16], index: usize) -> v128 {
i64x2(*(buf.get_unchecked(index..).as_ptr() as *const i64), 0)
}
#[inline(always)]
pub unsafe fn ptr_i16_to_set1_i32(buf: &[i16], index: usize) -> v128 {
i32x4(
*(buf.get_unchecked(index..).as_ptr() as *const i32),
0,
0,
0,
)
}
+2
View File
@@ -272,5 +272,7 @@ pub fn cpu_ext_into_str(cpu_extensions: CpuExtensions) -> &'static str {
CpuExtensions::Avx2 => "avx2",
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => "neon",
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => "wasm32",
}
}
+33 -791
View File
@@ -1,796 +1,38 @@
use std::cmp::Ordering;
use std::fmt::Debug;
use fast_image_resize::pixels::*;
use fast_image_resize::{
testing as fr_testing, CpuExtensions, CropBox, DifferentTypesOfPixelsError, DynamicImageView,
FilterType, Image, PixelType, ResizeAlg, Resizer,
};
use testing::{cpu_ext_into_str, nonzero, PixelTestingExt};
fn get_new_height(src_image: &DynamicImageView, new_width: u32) -> u32 {
let scale = new_width as f32 / src_image.width().get() as f32;
(src_image.height().get() as f32 * scale).round() as u32
}
const NEW_WIDTH: u32 = 255;
const NEW_BIG_WIDTH: u32 = 5016;
use std::arch::wasm32::*;
#[test]
fn try_resize_to_other_pixel_type() {
let src_image = U8x4::load_big_src_image();
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
let mut dst_image = Image::new(nonzero(1024), nonzero(256), PixelType::U8);
assert!(matches!(
resizer.resize(&src_image.view(), &mut dst_image.view_mut()),
Err(DifferentTypesOfPixelsError)
));
}
fn convolution_one_row() {
let mut ll_sum = i64x2(0, 0);
let k: [i64; 8] = [3, 1, 4, 1, 5, 9, 2, 6];
let source = u16x8(129, 380, 867, 408, 235, 809, 7824, 4752);
#[test]
fn resize_to_same_size() {
let width = nonzero(100);
let height = nonzero(80);
let buffer: Vec<u8> = (0..8000).map(|v| (v & 0xff) as u8).collect();
let src_image = Image::from_vec_u8(width, height, buffer, PixelType::U8).unwrap();
let mut dst_image = Image::new(width, height, PixelType::U8);
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
resizer
.resize(&src_image.view(), &mut dst_image.view_mut())
.unwrap();
assert!(matches!(
src_image.buffer().cmp(dst_image.buffer()),
Ordering::Equal
));
}
#[test]
fn resize_to_same_size_after_cropping() {
let width = nonzero(100);
let height = nonzero(80);
let src_width = nonzero(120);
let src_height = nonzero(100);
let buffer: Vec<u8> = (0..12000).map(|v| (v & 0xff) as u8).collect();
let src_image = Image::from_vec_u8(src_width, src_height, buffer, PixelType::U8).unwrap();
let mut src_view = src_image.view();
src_view
.set_crop_box(CropBox {
top: 10,
left: 10,
width,
height,
})
.unwrap();
let mut dst_image = Image::new(width, height, PixelType::U8);
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
resizer
.resize(&src_view, &mut dst_image.view_mut())
.unwrap();
let cropped_buffer: Vec<u8> = (0..12000u32)
.filter_map(|v| {
let row = v / 120;
let col = v % 120;
if (10..90u32).contains(&row) && (10..110u32).contains(&col) {
Some((v & 0xff) as u8)
} else {
None
}
})
.collect();
let dst_buffer = dst_image.into_vec();
assert!(matches!(cropped_buffer.cmp(&dst_buffer), Ordering::Equal));
}
/// In this test, we check that resizer won't use horizontal convolution
/// if width of destination image is equal to width of cropped source image.
fn resize_to_same_width<const C: usize>(
pixel_type: PixelType,
cpu_extensions: CpuExtensions,
create_pixel: fn(v: u8) -> [u8; C],
) {
fr_testing::clear_log();
let width = nonzero(100);
let height = nonzero(80);
let src_width = nonzero(120);
let src_height = nonzero(100);
// Image columns are made up of pixels of the same color.
let buffer: Vec<u8> = (0..12000)
.flat_map(|v| create_pixel((v % 120) as u8))
.collect();
let src_image = Image::from_vec_u8(src_width, src_height, buffer, pixel_type).unwrap();
let mut src_view = src_image.view();
src_view
.set_crop_box(CropBox {
left: 10,
top: 0,
width,
height: src_height,
})
.unwrap();
let mut dst_image = Image::new(width, height, pixel_type);
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
unsafe {
resizer.set_cpu_extensions(cpu_extensions);
}
resizer
.resize(&src_view, &mut dst_image.view_mut())
.unwrap();
let expected_result: Vec<u8> = (0..8000u32)
.flat_map(|v| create_pixel((10 + v % 100) as u8))
.collect();
let dst_buffer = dst_image.into_vec();
assert!(
matches!(expected_result.cmp(&dst_buffer), Ordering::Equal),
"Resizing result is not equal to expected ones ({:?}, {:?})",
pixel_type,
cpu_extensions
let l01_shuffle = i8x16(-1, -1, -1, -1, -1, -1, 0, 1, -1, -1, -1, -1, -1, -1, 0, 1);
let l23_shuffle = i8x16(-1, -1, -1, -1, -1, -1, 4, 5, -1, -1, -1, -1, -1, -1, 6, 7);
let l45_shuffle = i8x16(-1, -1, -1, -1, -1, -1, 8, 9, -1, -1, -1, -1, -1, -1, 10, 11);
let l67_shuffle = i8x16(
-1, -1, -1, -1, -1, -1, 12, 13, -1, -1, -1, -1, -1, -1, 14, 15,
);
assert!(fr_testing::logs_contain(
"compute vertical convolution coefficients"
));
assert!(!fr_testing::logs_contain(
"compute horizontal convolution coefficients"
));
}
/// In this test, we check that resizer won't use vertical convolution
/// if height of destination image is equal to height of cropped source image.
fn resize_to_same_height<const C: usize>(
pixel_type: PixelType,
cpu_extensions: CpuExtensions,
create_pixel: fn(v: u8) -> [u8; C],
) {
fr_testing::clear_log();
let width = nonzero(100);
let height = nonzero(80);
let src_width = nonzero(120);
let src_height = nonzero(100);
// Image rows are made up of pixels of the same color.
let buffer: Vec<u8> = (0..12000)
.flat_map(|v| create_pixel((v / 120) as u8))
.collect();
let src_image = Image::from_vec_u8(src_width, src_height, buffer, pixel_type).unwrap();
let mut src_view = src_image.view();
src_view
.set_crop_box(CropBox {
left: 0,
top: 10,
width: src_width,
height,
})
.unwrap();
let mut dst_image = Image::new(width, height, pixel_type);
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
unsafe {
resizer.set_cpu_extensions(cpu_extensions);
}
resizer
.resize(&src_view, &mut dst_image.view_mut())
.unwrap();
let expected_result: Vec<u8> = (0..8000u32)
.flat_map(|v| create_pixel((10 + v / 100) as u8))
.collect();
let dst_buffer = dst_image.into_vec();
assert!(
matches!(expected_result.cmp(&dst_buffer), Ordering::Equal),
"Resizing result is not equal to expected ones ({:?}, {:?})",
pixel_type,
cpu_extensions
);
assert!(!fr_testing::logs_contain(
"compute vertical convolution coefficients"
));
assert!(fr_testing::logs_contain(
"compute horizontal convolution coefficients"
));
}
#[test]
fn resize_to_same_width_after_cropping() {
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
if !cpu_extensions.is_supported() {
continue;
}
resize_to_same_width(PixelType::U8, cpu_extensions, |v| [v]);
resize_to_same_width(PixelType::U8x2, cpu_extensions, |v| [v; 2]);
resize_to_same_width(PixelType::U8x3, cpu_extensions, |v| [v; 3]);
resize_to_same_width(PixelType::U8x4, cpu_extensions, |v| [v; 4]);
resize_to_same_width(PixelType::U16, cpu_extensions, |v| [v, 0]);
resize_to_same_width(PixelType::U16x2, cpu_extensions, |v| [v, 0, v, 0]);
resize_to_same_width(PixelType::U16x3, cpu_extensions, |v| [v, 0, v, 0, v, 0]);
resize_to_same_width(PixelType::U16x4, cpu_extensions, |v| {
[v, 0, v, 0, v, 0, v, 0]
});
resize_to_same_height(PixelType::U8, cpu_extensions, |v| [v]);
resize_to_same_height(PixelType::U8x2, cpu_extensions, |v| [v; 2]);
resize_to_same_height(PixelType::U8x3, cpu_extensions, |v| [v; 3]);
resize_to_same_height(PixelType::U8x4, cpu_extensions, |v| [v; 4]);
resize_to_same_height(PixelType::U16, cpu_extensions, |v| [v, 0]);
resize_to_same_height(PixelType::U16x2, cpu_extensions, |v| [v, 0, v, 0]);
resize_to_same_height(PixelType::U16x3, cpu_extensions, |v| [v, 0, v, 0, v, 0]);
resize_to_same_height(PixelType::U16x4, cpu_extensions, |v| {
[v, 0, v, 0, v, 0, v, 0]
});
}
resize_to_same_width(PixelType::I32, CpuExtensions::None, |v| {
(v as i32).to_le_bytes()
});
resize_to_same_width(PixelType::F32, CpuExtensions::None, |v| {
(v as f32).to_le_bytes()
});
resize_to_same_height(PixelType::I32, CpuExtensions::None, |v| {
(v as i32).to_le_bytes()
});
resize_to_same_height(PixelType::F32, CpuExtensions::None, |v| {
(v as f32).to_le_bytes()
});
}
trait ResizeTest<const CC: usize> {
fn downscale_test(resize_alg: ResizeAlg, cpu_extensions: CpuExtensions, checksum: [u64; CC]);
fn upscale_test(resize_alg: ResizeAlg, cpu_extensions: CpuExtensions, checksum: [u64; CC]);
}
impl<T, C, const CC: usize> ResizeTest<CC> for Pixel<T, C, CC>
where
Self: PixelTestingExt,
T: Sized + Copy + Clone + Debug + PartialEq + 'static,
C: PixelComponent,
{
fn downscale_test(resize_alg: ResizeAlg, cpu_extensions: CpuExtensions, checksum: [u64; CC]) {
if !cpu_extensions.is_supported() {
println!(
"Cpu Extensions '{}' not supported by your CPU",
cpu_ext_into_str(cpu_extensions)
);
return;
}
let image = Self::load_big_src_image();
assert_eq!(image.pixel_type(), Self::pixel_type());
let mut resizer = Resizer::new(resize_alg);
unsafe {
resizer.set_cpu_extensions(cpu_extensions);
}
let image_view = image.view();
let new_height = get_new_height(&image_view, NEW_WIDTH);
let mut result = Image::new(nonzero(NEW_WIDTH), nonzero(new_height), image.pixel_type());
assert!(resizer.resize(&image_view, &mut result.view_mut()).is_ok());
let alg_name = match resize_alg {
ResizeAlg::Nearest => "nearest",
ResizeAlg::Convolution(filter) => match filter {
FilterType::Box => "box",
FilterType::Bilinear => "bilinear",
FilterType::Hamming => "hamming",
FilterType::Mitchell => "mitchell",
FilterType::CatmullRom => "catmullrom",
FilterType::Lanczos3 => "lanczos3",
_ => "unknown",
},
ResizeAlg::SuperSampling(_, _) => "supersampling",
_ => "unknown",
};
let name = format!(
"downscale-{}-{}-{}",
Self::pixel_type_str(),
alg_name,
cpu_ext_into_str(cpu_extensions),
);
testing::save_result(&result, &name);
assert_eq!(
testing::image_checksum::<Self, CC>(&result),
checksum,
"Error in checksum for {:?}",
cpu_extensions
);
}
fn upscale_test(resize_alg: ResizeAlg, cpu_extensions: CpuExtensions, checksum: [u64; CC]) {
if !cpu_extensions.is_supported() {
println!(
"Cpu Extensions '{}' not supported by your CPU",
cpu_ext_into_str(cpu_extensions)
);
return;
}
let image = Self::load_small_src_image();
assert_eq!(image.pixel_type(), Self::pixel_type());
let mut resizer = Resizer::new(resize_alg);
unsafe {
resizer.set_cpu_extensions(cpu_extensions);
}
let new_height = get_new_height(&image.view(), NEW_BIG_WIDTH);
let mut result = Image::new(
nonzero(NEW_BIG_WIDTH),
nonzero(new_height),
image.pixel_type(),
);
assert!(resizer
.resize(&image.view(), &mut result.view_mut())
.is_ok());
let alg_name = match resize_alg {
ResizeAlg::Nearest => "nearest",
ResizeAlg::Convolution(filter) => match filter {
FilterType::Box => "box",
FilterType::Bilinear => "bilinear",
FilterType::Hamming => "hamming",
FilterType::Mitchell => "mitchell",
FilterType::CatmullRom => "catmullrom",
FilterType::Lanczos3 => "lanczos3",
_ => "unknown",
},
ResizeAlg::SuperSampling(_, _) => "supersampling",
_ => "unknown",
};
let name = format!(
"upscale-{}-{}-{}",
Self::pixel_type_str(),
alg_name,
cpu_ext_into_str(cpu_extensions),
);
testing::save_result(&result, &name);
assert_eq!(
testing::image_checksum::<Self, CC>(&result),
checksum,
"Error in checksum for {:?}",
cpu_extensions
);
}
}
#[test]
fn downscale_u8() {
type P = U8;
P::downscale_test(ResizeAlg::Nearest, CpuExtensions::None, [2920348]);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[2923557],
);
}
}
#[test]
fn upscale_u8() {
type P = U8;
P::upscale_test(ResizeAlg::Nearest, CpuExtensions::None, [1148754010]);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[1148811406],
);
}
}
#[test]
fn downscale_u8x2() {
type P = U8x2;
P::downscale_test(ResizeAlg::Nearest, CpuExtensions::None, [2920348, 6121802]);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[2923557, 6122818],
);
}
}
#[test]
fn upscale_u8x2() {
type P = U8x2;
P::upscale_test(
ResizeAlg::Nearest,
CpuExtensions::None,
[1146218632, 2364895380],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[1146283728, 2364890194],
);
}
}
#[test]
fn downscale_u8x3() {
type P = U8x3;
P::downscale_test(
ResizeAlg::Nearest,
CpuExtensions::None,
[2937940, 2945380, 2882679],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[2942479, 2947850, 2885072],
);
}
}
#[test]
fn upscale_u8x3() {
type P = U8x3;
P::upscale_test(
ResizeAlg::Nearest,
CpuExtensions::None,
[1156008260, 1158417906, 1135087540],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[1156107005, 1158443335, 1135101759],
);
}
}
#[test]
fn downscale_u8x4() {
type P = U8x4;
P::downscale_test(
ResizeAlg::Nearest,
CpuExtensions::None,
[2937940, 2945380, 2882679, 6121802],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[2942479, 2947850, 2885072, 6122818],
);
P::downscale_test(
ResizeAlg::SuperSampling(FilterType::Lanczos3, 2),
cpu_extensions,
[2942546, 2947627, 2884866, 6123158],
);
}
}
#[test]
fn upscale_u8x4() {
type P = U8x4;
P::upscale_test(
ResizeAlg::Nearest,
CpuExtensions::None,
[1155096957, 1152644783, 1123285879, 2364895380],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[1155201788, 1152688479, 1123328716, 2364890194],
);
}
}
#[test]
fn downscale_u16() {
type P = U16;
P::downscale_test(ResizeAlg::Nearest, CpuExtensions::None, [750529436]);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[751401243],
);
}
}
#[test]
fn upscale_u16() {
type P = U16;
P::upscale_test(ResizeAlg::Nearest, CpuExtensions::None, [295229780570]);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[295246940755],
);
}
}
#[test]
fn downscale_u16x2() {
type P = U16x2;
P::downscale_test(
ResizeAlg::Nearest,
CpuExtensions::None,
[750529436, 1573303114],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[751401243, 1573563971],
);
}
}
#[test]
fn upscale_u16x2() {
type P = U16x2;
P::upscale_test(
ResizeAlg::Nearest,
CpuExtensions::None,
[294578188424, 607778112660],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[294597368766, 607776760273],
);
}
}
#[test]
fn downscale_u16x3() {
type P = U16x3;
P::downscale_test(
ResizeAlg::Nearest,
CpuExtensions::None,
[755050580, 756962660, 740848503],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[756269847, 757632467, 741478612],
);
}
}
#[test]
fn upscale_u16x3() {
type P = U16x3;
P::upscale_test(
ResizeAlg::Nearest,
CpuExtensions::None,
[297094122820, 297713401842, 291717497780],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[297122154090, 297723994984, 291725294637],
);
}
}
#[test]
fn downscale_u16x4() {
type P = U16x4;
P::downscale_test(
ResizeAlg::Nearest,
CpuExtensions::None,
[755050580, 756962660, 740848503, 1573303114],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[756269847, 757632467, 741478612, 1573563971],
);
}
}
#[test]
fn upscale_u16x4() {
type P = U16x4;
P::upscale_test(
ResizeAlg::Nearest,
CpuExtensions::None,
[296859917949, 296229709231, 288684470903, 607778112660],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[296888688348, 296243667797, 288698172180, 607776760273],
);
}
let coeff01_i64x2 = i64x2(k[0] as i64, k[1] as i64);
let coeff23_i64x2 = i64x2(k[2] as i64, k[3] as i64);
let coeff45_i64x2 = i64x2(k[4] as i64, k[5] as i64);
let coeff67_i64x2 = i64x2(k[6] as i64, k[7] as i64);
let l_i64x2 = i8x16_swizzle(source, l01_shuffle);
ll_sum = i64x2_add(ll_sum, i64x2_mul(l_i64x2, coeff01_i64x2));
let l_i64x2 = i8x16_swizzle(source, l23_shuffle);
ll_sum = i64x2_add(ll_sum, i64x2_mul(l_i64x2, coeff23_i64x2));
let l_i64x2 = i8x16_swizzle(source, l45_shuffle);
ll_sum = i64x2_add(ll_sum, i64x2_mul(l_i64x2, coeff45_i64x2));
let l_i64x2 = i8x16_swizzle(source, l67_shuffle);
ll_sum = i64x2_add(ll_sum, i64x2_mul(l_i64x2, coeff67_i64x2));
let simd_answer = i64x2_extract_lane::<0>(ll_sum) + i64x2_extract_lane::<1>(ll_sum);
// Non simd calculation
let source: [u16; 8] = [129, 380, 867, 408, 235, 809, 7824, 4752];
let native_ans = source
.into_iter()
.map(|x| k.into_iter().map(|ki| x as i64 * ki as i64).sum::<i64>())
.sum::<i64>();
assert_eq!(native_ans, simd_answer);
}