Working vertical u16

This commit is contained in:
Colin
2023-01-23 16:46:42 -05:00
committed by Colin Murphy
parent 11fbe03a8f
commit f0979ec3b9
7 changed files with 1148 additions and 109 deletions
+19 -37
View File
@@ -4,10 +4,6 @@ use crate::convolution::{optimisations, Coefficients};
use crate::pixels::U16;
use crate::wasm32_utils;
use crate::{ImageView, ImageViewMut};
/*
use std::fs;
use std::path::Path;
*/
#[inline]
pub(crate) fn horiz_convolution(
@@ -47,7 +43,6 @@ pub(crate) fn horiz_convolution(
/// - length of all rows in dst_rows must be equal
/// - coefficients_chunks.len() == dst_rows.0.len()
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
#[target_feature(enable = "simd128")]
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U16]; 4],
dst_rows: [&mut &mut [U16]; 4],
@@ -63,28 +58,28 @@ unsafe fn horiz_convolution_four_rows(
|0001| |0203| |0405| |0607| |0809| |1011| |1213| |1415|
Shuffle to extract L0 and L1 as i64:
-1, -1, -1, -1, -1, -1, 3, 2, -1, -1, -1, -1, -1, -1, 1, 0
0, 1, -1, -1, -1, -1, -1, -1, 2, 3, -1, -1, -1, -1, -1, -1
Shuffle to extract L2 and L3 as i64:
-1, -1, -1, -1, -1, -1, 7, 6, -1, -1, -1, -1, -1, -1, 5, 4
4, 5, -1, -1, -1, -1, -1, -1, 6, 7, -1, -1, -1, -1, -1, -1
Shuffle to extract L4 and L5 as i64:
-1, -1, -1, -1, -1, -1, 11, 10, -1, -1, -1, -1, -1, -1, 9, 8
8, 9, -1, -1, -1, -1, -1, -1, 10, 11, -1, -1, -1, -1, -1, -1
Shuffle to extract L6 and L7 as i64:
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12
12, 13, -1, -1, -1, -1, -1, -1, 14, 15, -1, -1, -1, -1, -1, -1
*/
let l0l1_shuffle = i8x16(-1, -1, -1, -1, -1, -1, 3, 2, -1, -1, -1, -1, -1, -1, 1, 0);
let l2l3_shuffle = i8x16(-1, -1, -1, -1, -1, -1, 7, 6, -1, -1, -1, -1, -1, -1, 5, 4);
let l4l5_shuffle = i8x16(-1, -1, -1, -1, -1, -1, 11, 10, -1, -1, -1, -1, -1, -1, 9, 8);
let l0l1_shuffle = i8x16(0, 1, -1, -1, -1, -1, -1, -1, 2, 3, -1, -1, -1, -1, -1, -1);
let l2l3_shuffle = i8x16(4, 5, -1, -1, -1, -1, -1, -1, 6, 7, -1, -1, -1, -1, -1, -1);
let l4l5_shuffle = i8x16(8, 9, -1, -1, -1, -1, -1, -1, 10, 11, -1, -1, -1, -1, -1, -1);
let l6l7_shuffle = i8x16(
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12,
12, 13, -1, -1, -1, -1, -1, -1, 14, 15, -1, -1, -1, -1, -1, -1,
);
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
let mut ll_sum: [v128; 4] = [i64x2(0i64, 0i64); 4];
let mut ll_sum: [v128; 4] = [i64x2_splat(0i64); 4];
let mut coeffs = coeffs_chunk.values;
@@ -174,20 +169,12 @@ unsafe fn horiz_convolution_four_rows(
/// - bounds.len() == dst_row.len()
/// - coefficients_chunks.len() == dst_row.len()
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
#[target_feature(enable = "simd128")]
unsafe fn horiz_convolution_one_row(
src_row: &[U16],
dst_row: &mut [U16],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
normalizer: &optimisations::Normalizer32,
) {
/*
let file = "wasm32";
let file_exists = Path::new(file).exists();
if file_exists {
fs::write(file, format!("src_row: {:?}", src_row)).unwrap();
}
*/
let precision = normalizer.precision();
let half_error = 1i64 << (precision - 1);
let mut ll_buf = [0i64; 2];
@@ -197,28 +184,28 @@ unsafe fn horiz_convolution_one_row(
|0001| |0203| |0405| |0607| |0809| |1011| |1213| |1415|
Shuffle to extract L0 and L1 as i64:
-1, -1, -1, -1, -1, -1, 0, 1, -1, -1, -1, -1, -1, -1, 2, 3
0, 1, -1, -1, -1, -1, -1, -1, 2, 3, -1, -1, -1, -1, -1, -1
Shuffle to extract L2 and L3 as i64:
-1, -1, -1, -1, -1, -1, 4, 5, -1, -1, -1, -1, -1, -1, 6, 7
4, 5, -1, -1, -1, -1, -1, -1, 6, 7, -1, -1, -1, -1, -1, -1
Shuffle to extract L4 and L5 as i64:
-1, -1, -1, -1, -1, -1, 8, 9, -1, -1, -1, -1, -1, -1, 10, 11
8, 9, -1, -1, -1, -1, -1, -1, 10, 11, -1, -1, -1, -1, -1, -1
Shuffle to extract L6 and L7 as i64:
-1, -1, -1, -1, -1, -1, 12, 13, -1, -1, -1, -1, -1, -1, 14, 15
12, 13, -1, -1, -1, -1, -1, -1, 14, 15, -1, -1, -1, -1, -1, -1
*/
let l01_shuffle = i8x16(-1, -1, -1, -1, -1, -1, 0, 1, -1, -1, -1, -1, -1, -1, 0, 1);
let l23_shuffle = i8x16(-1, -1, -1, -1, -1, -1, 4, 5, -1, -1, -1, -1, -1, -1, 6, 7);
let l45_shuffle = i8x16(-1, -1, -1, -1, -1, -1, 8, 9, -1, -1, -1, -1, -1, -1, 10, 11);
let l01_shuffle = i8x16(0, 1, -1, -1, -1, -1, -1, -1, 2, 3, -1, -1, -1, -1, -1, -1);
let l23_shuffle = i8x16(4, 5, -1, -1, -1, -1, -1, -1, 6, 7, -1, -1, -1, -1, -1, -1);
let l45_shuffle = i8x16(8, 9, -1, -1, -1, -1, -1, -1, 10, 11, -1, -1, -1, -1, -1, -1);
let l67_shuffle = i8x16(
-1, -1, -1, -1, -1, -1, 12, 13, -1, -1, -1, -1, -1, -1, 14, 15,
12, 13, -1, -1, -1, -1, -1, -1, 14, 15, -1, -1, -1, -1, -1, -1,
);
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
let mut ll_sum = i64x2(0, 0);
let mut ll_sum = i64x2_splat(0);
let mut coeffs = coeffs_chunk.values;
let coeffs_by_8 = coeffs.chunks_exact(8);
@@ -281,7 +268,7 @@ unsafe fn horiz_convolution_one_row(
if let Some(&k) = coeffs.first() {
let coeff01_i64x2 = i64x2(k as i64, 0);
let pixel = (*src_row.get_unchecked(x)).0 as i64;
let source = i64x2(0, pixel);
let source = i64x2(pixel, 0);
ll_sum = i64x2_add(ll_sum, i64x2_mul(source, coeff01_i64x2));
}
@@ -289,9 +276,4 @@ unsafe fn horiz_convolution_one_row(
let dst_pixel = dst_row.get_unchecked_mut(dst_x);
dst_pixel.0 = normalizer.clip(ll_buf[0] + ll_buf[1] + half_error);
}
/*
if file_exists {
fs::write(file, format!("dst_row: {:?}", dst_row)).unwrap();
}
*/
}
+22 -25
View File
@@ -45,7 +45,6 @@ pub(crate) fn horiz_convolution(
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
#[inline]
#[target_feature(enable = "simd128")]
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U8x2]; 4],
dst_rows: [&mut &mut [U8x2]; 4],
@@ -66,7 +65,7 @@ unsafe fn horiz_convolution_four_rows(
*/
#[rustfmt::skip]
let sh1 = i8x16(
-1, 7, -1, 5, -1, 3, -1, 1, -1, 6, -1, 4, -1, 2, -1, 0,
0, -1, 2, -1, 4, -1, 6, -1, 1, -1, 3, -1, 5, -1, 7, -1
);
/*
A: |-1 15| |-1 13| |-1 11| |-1 09|
@@ -74,7 +73,7 @@ unsafe fn horiz_convolution_four_rows(
*/
#[rustfmt::skip]
let sh2 = i8x16(
-1, 15, -1, 13, -1, 11, -1, 9, -1, 14, -1, 12, -1, 10, -1, 8,
8, -1, 10, -1, 12, -1, 14, -1, 9, -1, 11, -1, 13, -1, 15, -1
);
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
@@ -92,9 +91,9 @@ unsafe fn horiz_convolution_four_rows(
for i in 0..4 {
let source = wasm32_utils::load_v128(src_rows[i], x);
let pix = wasm32_utils::i8x16_shuffle(source, sh1);
let pix = i8x16_swizzle(source, sh1);
let tmp_sum = i32x4_add(sss[i], i32x4_dot_i16x8(pix, mmk0));
let pix = wasm32_utils::i8x16_shuffle(source, sh2);
let pix = i8x16_swizzle(source, sh2);
sss[i] = i32x4_add(tmp_sum, i32x4_dot_i16x8(pix, mmk1));
}
x += 8;
@@ -108,7 +107,7 @@ unsafe fn horiz_convolution_four_rows(
for i in 0..4 {
let source = wasm32_utils::loadl_i64(src_rows[i], x);
let pix = wasm32_utils::i8x16_shuffle(source, sh1);
let pix = i8x16_swizzle(source, sh1);
sss[i] = i32x4_add(sss[i], i32x4_dot_i16x8(pix, mmk));
}
x += 4;
@@ -122,18 +121,18 @@ unsafe fn horiz_convolution_four_rows(
for i in 0..4 {
let source = wasm32_utils::loadl_i32(src_rows[i], x);
let pix = wasm32_utils::i8x16_shuffle(source, sh1);
let pix = i8x16_swizzle(source, sh1);
sss[i] = i32x4_add(sss[i], i32x4_dot_i16x8(pix, mmk));
}
x += 2;
}
if let Some(&k) = reminder.first() {
let mmk = i32x4(k as i32, 0, 0, 0);
let mmk = i32x4_splat(k as i32);
for i in 0..4 {
let source = wasm32_utils::loadl_i16(src_rows[i], x);
let pix = wasm32_utils::i8x16_shuffle(source, sh1);
let pix = i8x16_swizzle(source, sh1);
sss[i] = i32x4_add(sss[i], i32x4_dot_i16x8(pix, mmk));
}
}
@@ -145,7 +144,6 @@ unsafe fn horiz_convolution_four_rows(
}
#[inline]
#[target_feature(enable = "simd128")]
unsafe fn set_dst_pixel(
raw: v128,
d_row: &mut &mut [U8x2],
@@ -167,7 +165,6 @@ unsafe fn set_dst_pixel(
/// - max(bound.start + bound.size for bound in bounds) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
#[inline]
#[target_feature(enable = "simd128")]
unsafe fn horiz_convolution_one_row(
src_row: &[U8x2],
dst_row: &mut [U8x2],
@@ -188,7 +185,7 @@ unsafe fn horiz_convolution_one_row(
*/
#[rustfmt::skip]
let pix_sh1 = i8x16(
-1, 7, -1, 5, -1, 6, -1, 4, -1, 3, -1, 1, -1, 2, -1, 0,
0, -1, 2, -1, 1, -1, 3, -1, 4, -1, 6, -1, 5, -1, 7, -1
);
/*
|C0 | |C1 | |C2 | |C3 | |C4 | |C5 | |C6 | |C7 |
@@ -203,7 +200,7 @@ unsafe fn horiz_convolution_one_row(
*/
#[rustfmt::skip]
let coeff_sh1 = i8x16(
7, 6, 5, 4, 7, 6, 5, 4, 3, 2, 1, 0, 3,2, 1, 0,
0, 1, 2, 3, 0, 1, 2, 3, 4, 5, 6, 7, 4, 5, 6, 7
);
/*
@@ -219,7 +216,7 @@ unsafe fn horiz_convolution_one_row(
*/
#[rustfmt::skip]
let pix_sh2 = i8x16(
-1, 15, -1, 13, -1, 14, -1, 12, -1, 11, -1, 9, -1, 10, -1, 8,
8, -1, 10, -1, 9, -1, 11, -1, 12, -1, 14, -1, 13, -1, 15, -1
);
/*
|C0 | |C1 | |C2 | |C3 | |C4 | |C5 | |C6 | |C7 |
@@ -234,7 +231,7 @@ unsafe fn horiz_convolution_one_row(
*/
#[rustfmt::skip]
let coeff_sh2 = i8x16(
15, 14, 13, 12, 15, 14, 13, 12, 11, 10, 9, 8, 11, 10, 9, 8,
8, 9, 10, 11, 8, 9, 10, 11, 12, 13, 14, 15, 12, 13, 14, 15
);
/*
@@ -248,14 +245,14 @@ unsafe fn horiz_convolution_one_row(
A: |-1 03| |-1 01|
L: |-1 02| |-1 00|
*/
let pix_sh3 = i8x16(-1, 7, -1, 5, -1, 6, -1, 4, -1, 3, -1, 1, -1, 2, -1, 0);
let pix_sh3 = i8x16(0, -1, 2, -1, 1, -1, 3, -1, 4, -1, 6, -1, 5, -1, 7, -1);
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x = coeffs_chunk.start as usize;
let mut coeffs = coeffs_chunk.values;
// Lower part will be added to higher, use only half of the error
let mut sss = i32x4(1 << (precision - 2), 0, 0, 0);
let mut sss = i32x4_splat(1 << (precision - 2));
let coeffs_by_8 = coeffs.chunks_exact(8);
coeffs = coeffs_by_8.remainder();
@@ -264,12 +261,12 @@ unsafe fn horiz_convolution_one_row(
let ksource = wasm32_utils::load_v128(k, 0);
let source = wasm32_utils::load_v128(src_row, x);
let pix = wasm32_utils::i8x16_shuffle(source, pix_sh1);
let mmk = wasm32_utils::i8x16_shuffle(ksource, coeff_sh1);
let pix = i8x16_swizzle(source, pix_sh1);
let mmk = i8x16_swizzle(ksource, coeff_sh1);
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
let pix = wasm32_utils::i8x16_shuffle(source, pix_sh2);
let mmk = wasm32_utils::i8x16_shuffle(ksource, coeff_sh2);
let pix = i8x16_swizzle(source, pix_sh2);
let mmk = i8x16_swizzle(ksource, coeff_sh2);
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
x += 8;
@@ -279,9 +276,9 @@ unsafe fn horiz_convolution_one_row(
let reminder1 = coeffs_by_4.remainder();
for k in coeffs_by_4 {
let mmk = i16x8(k[3], k[2], k[3], k[2], k[1], k[0], k[1], k[0]);
let mmk = i16x8(k[0], k[1], k[0], k[1], k[2], k[3], k[2], k[3]);
let source = wasm32_utils::loadl_i64(src_row, x);
let pix = wasm32_utils::i8x16_shuffle(source, pix_sh3);
let pix = i8x16_swizzle(source, pix_sh3);
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
x += 4
@@ -299,10 +296,10 @@ unsafe fn horiz_convolution_one_row(
}
let pix = i16x8(
0, pixels[5], 0, pixels[4], pixels[3], pixels[1], pixels[2], pixels[0],
pixels[0], pixels[2], pixels[1], pixels[3], pixels[4], 0, pixels[5], 0,
);
let mmk = i16x8(
0, coeffs[2], 0, coeffs[2], coeffs[1], coeffs[0], coeffs[1], coeffs[0],
coeffs[0], coeffs[1], coeffs[0], coeffs[1], coeffs[2], 0, coeffs[2], 0,
);
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
+3 -1
View File
@@ -10,6 +10,8 @@ pub(crate) mod native;
mod neon;
#[cfg(target_arch = "x86_64")]
pub(crate) mod sse4;
#[cfg(target_arch = "wasm32")]
pub(crate) mod wasm32;
pub(crate) fn vert_convolution_u16<T: PixelExt<Component = u16>>(
src_image: &ImageView<T>,
@@ -30,7 +32,7 @@ pub(crate) fn vert_convolution_u16<T: PixelExt<Component = u16>>(
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => neon::vert_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => native::vert_convolution(src_image, dst_image, offset, coeffs),
CpuExtensions::Wasm32 => wasm32::vert_convolution(src_image, dst_image, offset, coeffs),
_ => native::vert_convolution(src_image, dst_image, offset, coeffs),
}
}
+36
View File
@@ -6,6 +6,8 @@ use crate::convolution::{optimisations, Coefficients};
use crate::pixels::PixelExt;
use crate::simd_utils;
use crate::{ImageView, ImageViewMut};
use std::fs;
use std::path::Path;
pub(crate) fn vert_convolution<T: PixelExt<Component = u16>>(
src_image: &ImageView<T>,
@@ -33,6 +35,9 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
coeffs_chunk: CoefficientsI32Chunk,
normalizer: &optimisations::Normalizer32,
) {
let file = "vsse4";
let mut debugout = String::new();
let file_exists = Path::new(file).exists();
let y_start = coeffs_chunk.start;
let coeffs = coeffs_chunk.values;
let max_y = y_start + coeffs.len() as u32;
@@ -111,6 +116,13 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
for x in 0..2 {
for sum in sums {
_mm_storeu_si128((&mut c_buf).as_mut_ptr() as *mut __m128i, sum[x]);
if !file_exists {
debugout += &format!(
"119: {:?} {:?}\n",
_mm_extract_epi64(sum[x], 0),
_mm_extract_epi64(sum[x], 1)
);
}
*dst_ptr = normalizer.clip(c_buf[0]);
dst_ptr = dst_ptr.add(1);
*dst_ptr = normalizer.clip(c_buf[1]);
@@ -165,6 +177,13 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
// sums[i] = _mm_srl_epi64(sums[i] , precision_i64);
// _mm_packus_epi32(sums[i] , sums[i] );
_mm_storeu_si128((&mut c_buf).as_mut_ptr() as *mut __m128i, sum);
if !file_exists {
debugout += &format!(
"176: {:?} {:?}\n",
_mm_extract_epi64(sum, 0),
_mm_extract_epi64(sum, 1)
);
}
*dst_ptr = normalizer.clip(c_buf[0]);
dst_ptr = dst_ptr.add(1);
*dst_ptr = normalizer.clip(c_buf[1]);
@@ -213,17 +232,34 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
let mut dst_ptr = dst_chunk.as_mut_ptr();
_mm_storeu_si128((&mut c_buf).as_mut_ptr() as *mut __m128i, c01);
if !file_exists {
debugout += &format!(
"227: {:?} {:?}\n",
_mm_extract_epi64(c01, 0),
_mm_extract_epi64(c01, 1)
);
}
*dst_ptr = normalizer.clip(c_buf[0]);
dst_ptr = dst_ptr.add(1);
*dst_ptr = normalizer.clip(c_buf[1]);
dst_ptr = dst_ptr.add(1);
_mm_storeu_si128((&mut c_buf).as_mut_ptr() as *mut __m128i, c23);
if !file_exists {
debugout += &format!(
"236: {:?} {:?}\n",
_mm_extract_epi64(c23, 0),
_mm_extract_epi64(c23, 1)
);
}
*dst_ptr = normalizer.clip(c_buf[0]);
dst_ptr = dst_ptr.add(1);
*dst_ptr = normalizer.clip(c_buf[1]);
src_x += 4;
}
if !file_exists {
fs::write(file, debugout).unwrap();
}
dst_u16 = dst_chunks_4.into_remainder();
if !dst_u16.is_empty() {
+270
View File
@@ -0,0 +1,270 @@
use std::arch::wasm32::*;
use crate::convolution::optimisations::CoefficientsI32Chunk;
use crate::convolution::vertical_u16::native::convolution_by_u16;
use crate::convolution::{optimisations, Coefficients};
use crate::pixels::PixelExt;
use crate::wasm32_utils;
use crate::{ImageView, ImageViewMut};
use std::fs;
use std::path::Path;
pub(crate) fn vert_convolution<T: PixelExt<Component = u16>>(
src_image: &ImageView<T>,
dst_image: &mut ImageViewMut<T>,
offset: u32,
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer32::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let src_x = offset as usize * T::count_of_components();
let dst_rows = dst_image.iter_rows_mut();
for (dst_row, coeffs_chunk) in dst_rows.zip(coefficients_chunks) {
unsafe {
vert_convolution_into_one_row_u16(src_image, dst_row, src_x, coeffs_chunk, &normalizer);
}
}
}
unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
src_img: &ImageView<T>,
dst_row: &mut [T],
mut src_x: usize,
coeffs_chunk: CoefficientsI32Chunk,
normalizer: &optimisations::Normalizer32,
) {
let file = "vwasm32";
let mut debugout = String::new();
let file_exists = Path::new(file).exists();
let y_start = coeffs_chunk.start;
let coeffs = coeffs_chunk.values;
let max_y = y_start + coeffs.len() as u32;
let mut dst_u16 = T::components_mut(dst_row);
/*
|0 1 2 3 4 5 6 7 |
|0001 0203 0405 0607 0809 1011 1213 1415|
Shuffle to extract 0-1 components as i64:
0, 1, -1, -1, -1, -1, -1, -1, 2, 3, -1, -1, -1, -1, -1, -1
Shuffle to extract 2-3 components as i64:
4, 5, -1, -1, -1, -1, -1, -1, 6, 7, -1, -1, -1, -1, -1, -1
Shuffle to extract 4-5 components as i64:
8, 9, -1, -1, -1, -1, -1, -1, 10, 11, -1, -1, -1, -1, -1, -1
Shuffle to extract 6-7 components as i64:
12, 13, -1, -1, -1, -1, -1, -1, 14, 15, -1, -1, -1, -1, -1, -1
*/
let c_shuffles = [
i8x16(0, 1, -1, -1, -1, -1, -1, -1, 2, 3, -1, -1, -1, -1, -1, -1),
i8x16(4, 5, -1, -1, -1, -1, -1, -1, 6, 7, -1, -1, -1, -1, -1, -1),
i8x16(8, 9, -1, -1, -1, -1, -1, -1, 10, 11, -1, -1, -1, -1, -1, -1),
i8x16(
12, 13, -1, -1, -1, -1, -1, -1, 14, 15, -1, -1, -1, -1, -1, -1,
),
];
let precision = normalizer.precision();
let initial = i64x2_splat(1 << (precision - 1));
let mut c_buf = [0i64; 2];
let mut dst_chunks_16 = dst_u16.chunks_exact_mut(16);
for dst_chunk in &mut dst_chunks_16 {
let mut sums = [[initial; 2], [initial; 2], [initial; 2], [initial; 2]];
let mut y: u32 = 0;
let coeffs_2 = coeffs.chunks_exact(2);
let coeffs_reminder = coeffs_2.remainder();
for (src_rows, two_coeffs) in src_img.iter_2_rows(y_start, max_y).zip(coeffs_2) {
let src_rows = src_rows.map(|row| T::components(row));
for r in 0..2 {
let coeff_i64x2 = i64x2_splat(two_coeffs[r] as i64);
for x in 0..2 {
let source = wasm32_utils::load_v128(src_rows[r], src_x + x * 8);
for i in 0..4 {
let c_i64x2 = i8x16_swizzle(source, c_shuffles[i]);
sums[i][x] = i64x2_add(sums[i][x], i64x2_mul(c_i64x2, coeff_i64x2));
}
}
}
y += 2;
}
if let Some(&k) = coeffs_reminder.first() {
let s_row = src_img.get_row(y_start + y).unwrap();
let components = T::components(s_row);
let coeff_i64x2 = i64x2_splat(k as i64);
for x in 0..2 {
let source = wasm32_utils::load_v128(components, src_x + x * 8);
for i in 0..4 {
let c_i64x2 = i8x16_swizzle(source, c_shuffles[i]);
sums[i][x] = i64x2_add(sums[i][x], i64x2_mul(c_i64x2, coeff_i64x2));
}
}
}
let mut dst_ptr = dst_chunk.as_mut_ptr();
for x in 0..2 {
for sum in sums {
v128_store((&mut c_buf).as_mut_ptr() as *mut v128, sum[x]);
if !file_exists {
debugout += &format!(
"119: {:?} {:?}\n",
i64x2_extract_lane::<0>(sum[x]),
i64x2_extract_lane::<1>(sum[x])
);
}
*dst_ptr = normalizer.clip(c_buf[0]);
dst_ptr = dst_ptr.add(1);
*dst_ptr = normalizer.clip(c_buf[1]);
dst_ptr = dst_ptr.add(1);
}
}
src_x += 16;
}
dst_u16 = dst_chunks_16.into_remainder();
let mut dst_chunks_8 = dst_u16.chunks_exact_mut(8);
if let Some(dst_chunk) = dst_chunks_8.next() {
let mut sums = [initial, initial, initial, initial];
let mut y: u32 = 0;
let coeffs_2 = coeffs.chunks_exact(2);
let coeffs_reminder = coeffs_2.remainder();
for (src_rows, two_coeffs) in src_img.iter_2_rows(y_start, max_y).zip(coeffs_2) {
let src_rows = src_rows.map(|row| T::components(row));
let coeffs_i64 = [
i64x2_splat(two_coeffs[0] as i64),
i64x2_splat(two_coeffs[1] as i64),
];
for r in 0..2 {
let source = wasm32_utils::load_v128(src_rows[r], src_x);
for i in 0..4 {
let c_i64x2 = i8x16_swizzle(source, c_shuffles[i]);
sums[i] = i64x2_add(sums[i], i64x2_mul(c_i64x2, coeffs_i64[r]));
}
}
y += 2;
}
if let Some(&k) = coeffs_reminder.first() {
let s_row = src_img.get_row(y_start + y).unwrap();
let components = T::components(s_row);
let coeff_i64x2 = i64x2_splat(k as i64);
let source = wasm32_utils::load_v128(components, src_x);
for i in 0..4 {
let c_i64x2 = i8x16_swizzle(source, c_shuffles[i]);
sums[i] = i64x2_add(sums[i], i64x2_mul(c_i64x2, coeff_i64x2));
}
}
let mut dst_ptr = dst_chunk.as_mut_ptr();
for sum in sums {
// let mask = _mm_cmpgt_epi64(sums[i], zero);
// sums[i] = _mm_and_si128(sums[i] , mask);
// sums[i] = _mm_srl_epi64(sums[i] , precision_i64);
// _mm_packus_epi32(sums[i] , sums[i] );
v128_store((&mut c_buf).as_mut_ptr() as *mut v128, sum);
if !file_exists {
debugout += &format!(
"176: {:?} {:?}\n",
i64x2_extract_lane::<0>(sum),
i64x2_extract_lane::<1>(sum)
);
}
*dst_ptr = normalizer.clip(c_buf[0]);
dst_ptr = dst_ptr.add(1);
*dst_ptr = normalizer.clip(c_buf[1]);
dst_ptr = dst_ptr.add(1);
}
src_x += 8;
}
dst_u16 = dst_chunks_8.into_remainder();
let mut dst_chunks_4 = dst_u16.chunks_exact_mut(4);
if let Some(dst_chunk) = dst_chunks_4.next() {
let mut c01 = initial;
let mut c23 = initial;
let mut y: u32 = 0;
let coeffs_2 = coeffs.chunks_exact(2);
let coeffs_reminder = coeffs_2.remainder();
for (src_rows, two_coeffs) in src_img.iter_2_rows(y_start, max_y).zip(coeffs_2) {
let src_rows = src_rows.map(|row| T::components(row));
let coeffs_i64 = [
i64x2_splat(two_coeffs[0] as i64),
i64x2_splat(two_coeffs[1] as i64),
];
for r in 0..2 {
let comp_x4 = src_rows[r].get_unchecked(src_x..src_x + 4);
let c_i64x2 = i64x2(comp_x4[0] as i64, comp_x4[1] as i64);
c01 = i64x2_add(c01, i64x2_mul(c_i64x2, coeffs_i64[r]));
let c_i64x2 = i64x2(comp_x4[2] as i64, comp_x4[3] as i64);
c23 = i64x2_add(c23, i64x2_mul(c_i64x2, coeffs_i64[r]));
}
y += 2;
}
if let Some(&k) = coeffs_reminder.first() {
let s_row = src_img.get_row(y_start + y).unwrap();
let components = T::components(s_row);
let coeff_i64x2 = i64x2_splat(k as i64);
let comp_x4 = components.get_unchecked(src_x..src_x + 4);
let c_i64x2 = i64x2(comp_x4[0] as i64, comp_x4[1] as i64);
c01 = i64x2_add(c01, i64x2_mul(c_i64x2, coeff_i64x2));
let c_i64x2 = i64x2(comp_x4[2] as i64, comp_x4[3] as i64);
c23 = i64x2_add(c23, i64x2_mul(c_i64x2, coeff_i64x2));
}
let mut dst_ptr = dst_chunk.as_mut_ptr();
v128_store((&mut c_buf).as_mut_ptr() as *mut v128, c01);
if !file_exists {
debugout += &format!(
"227: {:?} {:?}\n",
i64x2_extract_lane::<0>(c01),
i64x2_extract_lane::<1>(c01)
);
}
*dst_ptr = normalizer.clip(c_buf[0]);
dst_ptr = dst_ptr.add(1);
*dst_ptr = normalizer.clip(c_buf[1]);
dst_ptr = dst_ptr.add(1);
v128_store((&mut c_buf).as_mut_ptr() as *mut v128, c23);
if !file_exists {
debugout += &format!(
"236: {:?} {:?}\n",
i64x2_extract_lane::<0>(c23),
i64x2_extract_lane::<1>(c23)
);
}
*dst_ptr = normalizer.clip(c_buf[0]);
dst_ptr = dst_ptr.add(1);
*dst_ptr = normalizer.clip(c_buf[1]);
src_x += 4;
}
if !file_exists {
fs::write(file, debugout).unwrap();
}
dst_u16 = dst_chunks_4.into_remainder();
if !dst_u16.is_empty() {
let initial = 1 << (precision - 1);
convolution_by_u16(
src_img, normalizer, initial, dst_u16, src_x, y_start, coeffs,
);
}
}
+2 -12
View File
@@ -29,22 +29,12 @@ pub unsafe fn loadl_i16<T>(buf: &[T], index: usize) -> v128 {
)
}
#[inline(always)]
pub fn i8x16_shuffle(a: v128, b: v128) -> v128 {
u8x16_swizzle(a, v128_and(b, i8x16_splat(-113)))
}
#[inline(always)]
pub unsafe fn ptr_i16_to_set1_i64(buf: &[i16], index: usize) -> v128 {
i64x2(*(buf.get_unchecked(index..).as_ptr() as *const i64), 0)
i64x2_splat(*(buf.get_unchecked(index..).as_ptr() as *const i64))
}
#[inline(always)]
pub unsafe fn ptr_i16_to_set1_i32(buf: &[i16], index: usize) -> v128 {
i32x4(
*(buf.get_unchecked(index..).as_ptr() as *const i32),
0,
0,
0,
)
i32x4_splat(*(buf.get_unchecked(index..).as_ptr() as *const i32))
}
+796 -34
View File
@@ -1,38 +1,800 @@
use std::arch::wasm32::*;
#[test]
fn convolution_one_row() {
let mut ll_sum = i64x2(0, 0);
let k: [i64; 8] = [3, 1, 4, 1, 5, 9, 2, 6];
let source = u16x8(129, 380, 867, 408, 235, 809, 7824, 4752);
use std::cmp::Ordering;
use std::fmt::Debug;
let l01_shuffle = i8x16(-1, -1, -1, -1, -1, -1, 0, 1, -1, -1, -1, -1, -1, -1, 0, 1);
let l23_shuffle = i8x16(-1, -1, -1, -1, -1, -1, 4, 5, -1, -1, -1, -1, -1, -1, 6, 7);
let l45_shuffle = i8x16(-1, -1, -1, -1, -1, -1, 8, 9, -1, -1, -1, -1, -1, -1, 10, 11);
let l67_shuffle = i8x16(
-1, -1, -1, -1, -1, -1, 12, 13, -1, -1, -1, -1, -1, -1, 14, 15,
use fast_image_resize::pixels::*;
use fast_image_resize::{
testing as fr_testing, CpuExtensions, CropBox, DifferentTypesOfPixelsError, DynamicImageView,
FilterType, Image, PixelType, ResizeAlg, Resizer,
};
use testing::{cpu_ext_into_str, nonzero, PixelTestingExt};
fn get_new_height(src_image: &DynamicImageView, new_width: u32) -> u32 {
let scale = new_width as f32 / src_image.width().get() as f32;
(src_image.height().get() as f32 * scale).round() as u32
}
const NEW_WIDTH: u32 = 255;
const NEW_BIG_WIDTH: u32 = 5016;
#[test]
fn try_resize_to_other_pixel_type() {
let src_image = U8x4::load_big_src_image();
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
let mut dst_image = Image::new(nonzero(1024), nonzero(256), PixelType::U8);
assert!(matches!(
resizer.resize(&src_image.view(), &mut dst_image.view_mut()),
Err(DifferentTypesOfPixelsError)
));
}
#[test]
fn resize_to_same_size() {
let width = nonzero(100);
let height = nonzero(80);
let buffer: Vec<u8> = (0..8000).map(|v| (v & 0xff) as u8).collect();
let src_image = Image::from_vec_u8(width, height, buffer, PixelType::U8).unwrap();
let mut dst_image = Image::new(width, height, PixelType::U8);
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
resizer
.resize(&src_image.view(), &mut dst_image.view_mut())
.unwrap();
assert!(matches!(
src_image.buffer().cmp(dst_image.buffer()),
Ordering::Equal
));
}
#[test]
fn resize_to_same_size_after_cropping() {
let width = nonzero(100);
let height = nonzero(80);
let src_width = nonzero(120);
let src_height = nonzero(100);
let buffer: Vec<u8> = (0..12000).map(|v| (v & 0xff) as u8).collect();
let src_image = Image::from_vec_u8(src_width, src_height, buffer, PixelType::U8).unwrap();
let mut src_view = src_image.view();
src_view
.set_crop_box(CropBox {
top: 10,
left: 10,
width,
height,
})
.unwrap();
let mut dst_image = Image::new(width, height, PixelType::U8);
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
resizer
.resize(&src_view, &mut dst_image.view_mut())
.unwrap();
let cropped_buffer: Vec<u8> = (0..12000u32)
.filter_map(|v| {
let row = v / 120;
let col = v % 120;
if (10..90u32).contains(&row) && (10..110u32).contains(&col) {
Some((v & 0xff) as u8)
} else {
None
}
})
.collect();
let dst_buffer = dst_image.into_vec();
assert!(matches!(cropped_buffer.cmp(&dst_buffer), Ordering::Equal));
}
/// In this test, we check that resizer won't use horizontal convolution
/// if width of destination image is equal to width of cropped source image.
fn resize_to_same_width<const C: usize>(
pixel_type: PixelType,
cpu_extensions: CpuExtensions,
create_pixel: fn(v: u8) -> [u8; C],
) {
fr_testing::clear_log();
let width = nonzero(100);
let height = nonzero(80);
let src_width = nonzero(120);
let src_height = nonzero(100);
// Image columns are made up of pixels of the same color.
let buffer: Vec<u8> = (0..12000)
.flat_map(|v| create_pixel((v % 120) as u8))
.collect();
let src_image = Image::from_vec_u8(src_width, src_height, buffer, pixel_type).unwrap();
let mut src_view = src_image.view();
src_view
.set_crop_box(CropBox {
left: 10,
top: 0,
width,
height: src_height,
})
.unwrap();
let mut dst_image = Image::new(width, height, pixel_type);
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
unsafe {
resizer.set_cpu_extensions(cpu_extensions);
}
resizer
.resize(&src_view, &mut dst_image.view_mut())
.unwrap();
let expected_result: Vec<u8> = (0..8000u32)
.flat_map(|v| create_pixel((10 + v % 100) as u8))
.collect();
let dst_buffer = dst_image.into_vec();
assert!(
matches!(expected_result.cmp(&dst_buffer), Ordering::Equal),
"Resizing result is not equal to expected ones ({:?}, {:?})",
pixel_type,
cpu_extensions
);
let coeff01_i64x2 = i64x2(k[0] as i64, k[1] as i64);
let coeff23_i64x2 = i64x2(k[2] as i64, k[3] as i64);
let coeff45_i64x2 = i64x2(k[4] as i64, k[5] as i64);
let coeff67_i64x2 = i64x2(k[6] as i64, k[7] as i64);
let l_i64x2 = i8x16_swizzle(source, l01_shuffle);
ll_sum = i64x2_add(ll_sum, i64x2_mul(l_i64x2, coeff01_i64x2));
let l_i64x2 = i8x16_swizzle(source, l23_shuffle);
ll_sum = i64x2_add(ll_sum, i64x2_mul(l_i64x2, coeff23_i64x2));
let l_i64x2 = i8x16_swizzle(source, l45_shuffle);
ll_sum = i64x2_add(ll_sum, i64x2_mul(l_i64x2, coeff45_i64x2));
let l_i64x2 = i8x16_swizzle(source, l67_shuffle);
ll_sum = i64x2_add(ll_sum, i64x2_mul(l_i64x2, coeff67_i64x2));
let simd_answer = i64x2_extract_lane::<0>(ll_sum) + i64x2_extract_lane::<1>(ll_sum);
// Non simd calculation
let source: [u16; 8] = [129, 380, 867, 408, 235, 809, 7824, 4752];
let native_ans = source
.into_iter()
.map(|x| k.into_iter().map(|ki| x as i64 * ki as i64).sum::<i64>())
.sum::<i64>();
assert_eq!(native_ans, simd_answer);
assert!(fr_testing::logs_contain(
"compute vertical convolution coefficients"
));
assert!(!fr_testing::logs_contain(
"compute horizontal convolution coefficients"
));
}
/// In this test, we check that resizer won't use vertical convolution
/// if height of destination image is equal to height of cropped source image.
fn resize_to_same_height<const C: usize>(
pixel_type: PixelType,
cpu_extensions: CpuExtensions,
create_pixel: fn(v: u8) -> [u8; C],
) {
fr_testing::clear_log();
let width = nonzero(100);
let height = nonzero(80);
let src_width = nonzero(120);
let src_height = nonzero(100);
// Image rows are made up of pixels of the same color.
let buffer: Vec<u8> = (0..12000)
.flat_map(|v| create_pixel((v / 120) as u8))
.collect();
let src_image = Image::from_vec_u8(src_width, src_height, buffer, pixel_type).unwrap();
let mut src_view = src_image.view();
src_view
.set_crop_box(CropBox {
left: 0,
top: 10,
width: src_width,
height,
})
.unwrap();
let mut dst_image = Image::new(width, height, pixel_type);
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
unsafe {
resizer.set_cpu_extensions(cpu_extensions);
}
resizer
.resize(&src_view, &mut dst_image.view_mut())
.unwrap();
let expected_result: Vec<u8> = (0..8000u32)
.flat_map(|v| create_pixel((10 + v / 100) as u8))
.collect();
let dst_buffer = dst_image.into_vec();
assert!(
matches!(expected_result.cmp(&dst_buffer), Ordering::Equal),
"Resizing result is not equal to expected ones ({:?}, {:?})",
pixel_type,
cpu_extensions
);
assert!(!fr_testing::logs_contain(
"compute vertical convolution coefficients"
));
assert!(fr_testing::logs_contain(
"compute horizontal convolution coefficients"
));
}
#[test]
fn resize_to_same_width_after_cropping() {
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
if !cpu_extensions.is_supported() {
continue;
}
resize_to_same_width(PixelType::U8, cpu_extensions, |v| [v]);
resize_to_same_width(PixelType::U8x2, cpu_extensions, |v| [v; 2]);
resize_to_same_width(PixelType::U8x3, cpu_extensions, |v| [v; 3]);
resize_to_same_width(PixelType::U8x4, cpu_extensions, |v| [v; 4]);
resize_to_same_width(PixelType::U16, cpu_extensions, |v| [v, 0]);
resize_to_same_width(PixelType::U16x2, cpu_extensions, |v| [v, 0, v, 0]);
resize_to_same_width(PixelType::U16x3, cpu_extensions, |v| [v, 0, v, 0, v, 0]);
resize_to_same_width(PixelType::U16x4, cpu_extensions, |v| {
[v, 0, v, 0, v, 0, v, 0]
});
resize_to_same_height(PixelType::U8, cpu_extensions, |v| [v]);
resize_to_same_height(PixelType::U8x2, cpu_extensions, |v| [v; 2]);
resize_to_same_height(PixelType::U8x3, cpu_extensions, |v| [v; 3]);
resize_to_same_height(PixelType::U8x4, cpu_extensions, |v| [v; 4]);
resize_to_same_height(PixelType::U16, cpu_extensions, |v| [v, 0]);
resize_to_same_height(PixelType::U16x2, cpu_extensions, |v| [v, 0, v, 0]);
resize_to_same_height(PixelType::U16x3, cpu_extensions, |v| [v, 0, v, 0, v, 0]);
resize_to_same_height(PixelType::U16x4, cpu_extensions, |v| {
[v, 0, v, 0, v, 0, v, 0]
});
}
resize_to_same_width(PixelType::I32, CpuExtensions::None, |v| {
(v as i32).to_le_bytes()
});
resize_to_same_width(PixelType::F32, CpuExtensions::None, |v| {
(v as f32).to_le_bytes()
});
resize_to_same_height(PixelType::I32, CpuExtensions::None, |v| {
(v as i32).to_le_bytes()
});
resize_to_same_height(PixelType::F32, CpuExtensions::None, |v| {
(v as f32).to_le_bytes()
});
}
trait ResizeTest<const CC: usize> {
fn downscale_test(resize_alg: ResizeAlg, cpu_extensions: CpuExtensions, checksum: [u64; CC]);
fn upscale_test(resize_alg: ResizeAlg, cpu_extensions: CpuExtensions, checksum: [u64; CC]);
}
impl<T, C, const CC: usize> ResizeTest<CC> for Pixel<T, C, CC>
where
Self: PixelTestingExt,
T: Sized + Copy + Clone + Debug + PartialEq + 'static,
C: PixelComponent,
{
fn downscale_test(resize_alg: ResizeAlg, cpu_extensions: CpuExtensions, checksum: [u64; CC]) {
if !cpu_extensions.is_supported() {
println!(
"Cpu Extensions '{}' not supported by your CPU",
cpu_ext_into_str(cpu_extensions)
);
return;
}
let image = Self::load_big_src_image();
assert_eq!(image.pixel_type(), Self::pixel_type());
let mut resizer = Resizer::new(resize_alg);
unsafe {
resizer.set_cpu_extensions(cpu_extensions);
}
let image_view = image.view();
let new_height = get_new_height(&image_view, NEW_WIDTH);
let mut result = Image::new(nonzero(NEW_WIDTH), nonzero(new_height), image.pixel_type());
assert!(resizer.resize(&image_view, &mut result.view_mut()).is_ok());
let alg_name = match resize_alg {
ResizeAlg::Nearest => "nearest",
ResizeAlg::Convolution(filter) => match filter {
FilterType::Box => "box",
FilterType::Bilinear => "bilinear",
FilterType::Hamming => "hamming",
FilterType::Mitchell => "mitchell",
FilterType::CatmullRom => "catmullrom",
FilterType::Lanczos3 => "lanczos3",
_ => "unknown",
},
ResizeAlg::SuperSampling(_, _) => "supersampling",
_ => "unknown",
};
let name = format!(
"downscale-{}-{}-{}",
Self::pixel_type_str(),
alg_name,
cpu_ext_into_str(cpu_extensions),
);
testing::save_result(&result, &name);
assert_eq!(
testing::image_checksum::<Self, CC>(&result),
checksum,
"Error in checksum for {:?}",
cpu_extensions
);
}
fn upscale_test(resize_alg: ResizeAlg, cpu_extensions: CpuExtensions, checksum: [u64; CC]) {
if !cpu_extensions.is_supported() {
println!(
"Cpu Extensions '{}' not supported by your CPU",
cpu_ext_into_str(cpu_extensions)
);
return;
}
let image = Self::load_small_src_image();
assert_eq!(image.pixel_type(), Self::pixel_type());
let mut resizer = Resizer::new(resize_alg);
unsafe {
resizer.set_cpu_extensions(cpu_extensions);
}
let new_height = get_new_height(&image.view(), NEW_BIG_WIDTH);
let mut result = Image::new(
nonzero(NEW_BIG_WIDTH),
nonzero(new_height),
image.pixel_type(),
);
assert!(resizer
.resize(&image.view(), &mut result.view_mut())
.is_ok());
let alg_name = match resize_alg {
ResizeAlg::Nearest => "nearest",
ResizeAlg::Convolution(filter) => match filter {
FilterType::Box => "box",
FilterType::Bilinear => "bilinear",
FilterType::Hamming => "hamming",
FilterType::Mitchell => "mitchell",
FilterType::CatmullRom => "catmullrom",
FilterType::Lanczos3 => "lanczos3",
_ => "unknown",
},
ResizeAlg::SuperSampling(_, _) => "supersampling",
_ => "unknown",
};
let name = format!(
"upscale-{}-{}-{}",
Self::pixel_type_str(),
alg_name,
cpu_ext_into_str(cpu_extensions),
);
testing::save_result(&result, &name);
assert_eq!(
testing::image_checksum::<Self, CC>(&result),
checksum,
"Error in checksum for {:?}",
cpu_extensions
);
}
}
#[test]
fn downscale_u8() {
type P = U8;
P::downscale_test(ResizeAlg::Nearest, CpuExtensions::None, [2920348]);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[2923557],
);
}
}
#[test]
fn upscale_u8() {
type P = U8;
P::upscale_test(ResizeAlg::Nearest, CpuExtensions::None, [1148754010]);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[1148811406],
);
}
}
#[test]
fn downscale_u8x2() {
type P = U8x2;
P::downscale_test(ResizeAlg::Nearest, CpuExtensions::None, [2920348, 6121802]);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[2923557, 6122818],
);
}
}
#[test]
fn upscale_u8x2() {
type P = U8x2;
P::upscale_test(
ResizeAlg::Nearest,
CpuExtensions::None,
[1146218632, 2364895380],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[1146283728, 2364890194],
);
}
}
#[test]
fn downscale_u8x3() {
type P = U8x3;
P::downscale_test(
ResizeAlg::Nearest,
CpuExtensions::None,
[2937940, 2945380, 2882679],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[2942479, 2947850, 2885072],
);
}
}
#[test]
fn upscale_u8x3() {
type P = U8x3;
P::upscale_test(
ResizeAlg::Nearest,
CpuExtensions::None,
[1156008260, 1158417906, 1135087540],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[1156107005, 1158443335, 1135101759],
);
}
}
#[test]
fn downscale_u8x4() {
type P = U8x4;
P::downscale_test(
ResizeAlg::Nearest,
CpuExtensions::None,
[2937940, 2945380, 2882679, 6121802],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[2942479, 2947850, 2885072, 6122818],
);
P::downscale_test(
ResizeAlg::SuperSampling(FilterType::Lanczos3, 2),
cpu_extensions,
[2942546, 2947627, 2884866, 6123158],
);
}
}
#[test]
fn upscale_u8x4() {
type P = U8x4;
P::upscale_test(
ResizeAlg::Nearest,
CpuExtensions::None,
[1155096957, 1152644783, 1123285879, 2364895380],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[1155201788, 1152688479, 1123328716, 2364890194],
);
}
}
#[test]
fn downscale_u16() {
type P = U16;
P::downscale_test(ResizeAlg::Nearest, CpuExtensions::None, [750529436]);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
}
for cpu_extensions in cpu_extensions_vec {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[751401243],
);
}
}
#[test]
fn upscale_u16() {
type P = U16;
P::upscale_test(ResizeAlg::Nearest, CpuExtensions::None, [295229780570]);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[295246940755],
);
}
}
#[test]
fn downscale_u16x2() {
type P = U16x2;
P::downscale_test(
ResizeAlg::Nearest,
CpuExtensions::None,
[750529436, 1573303114],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[751401243, 1573563971],
);
}
}
#[test]
fn upscale_u16x2() {
type P = U16x2;
P::upscale_test(
ResizeAlg::Nearest,
CpuExtensions::None,
[294578188424, 607778112660],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[294597368766, 607776760273],
);
}
}
#[test]
fn downscale_u16x3() {
type P = U16x3;
P::downscale_test(
ResizeAlg::Nearest,
CpuExtensions::None,
[755050580, 756962660, 740848503],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[756269847, 757632467, 741478612],
);
}
}
#[test]
fn upscale_u16x3() {
type P = U16x3;
P::upscale_test(
ResizeAlg::Nearest,
CpuExtensions::None,
[297094122820, 297713401842, 291717497780],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[297122154090, 297723994984, 291725294637],
);
}
}
#[test]
fn downscale_u16x4() {
type P = U16x4;
P::downscale_test(
ResizeAlg::Nearest,
CpuExtensions::None,
[755050580, 756962660, 740848503, 1573303114],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[756269847, 757632467, 741478612, 1573563971],
);
}
}
#[test]
fn upscale_u16x4() {
type P = U16x4;
P::upscale_test(
ResizeAlg::Nearest,
CpuExtensions::None,
[296859917949, 296229709231, 288684470903, 607778112660],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[296888688348, 296243667797, 288698172180, 607776760273],
);
}
}