mirror of
https://github.com/Cykooz/fast_image_resize.git
synced 2026-10-08 01:11:09 +00:00
U8 complete.
This commit is contained in:
@@ -12,6 +12,8 @@ mod native;
|
||||
mod neon;
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod sse4;
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
mod wasm32;
|
||||
|
||||
impl Convolution for U16x2 {
|
||||
fn horiz_convolution(
|
||||
@@ -30,7 +32,7 @@ impl Convolution for U16x2 {
|
||||
CpuExtensions::Neon => neon::horiz_convolution(src_image, dst_image, offset, coeffs),
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Wasm32 => {
|
||||
native::horiz_convolution(src_image, dst_image, offset, coeffs)
|
||||
wasm32::horiz_convolution(src_image, dst_image, offset, coeffs)
|
||||
}
|
||||
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
|
||||
}
|
||||
|
||||
@@ -12,6 +12,8 @@ mod native;
|
||||
mod neon;
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod sse4;
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
mod wasm32;
|
||||
|
||||
impl Convolution for U8 {
|
||||
fn horiz_convolution(
|
||||
@@ -28,6 +30,10 @@ impl Convolution for U8 {
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_image, dst_image, offset, coeffs),
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => neon::horiz_convolution(src_image, dst_image, offset, coeffs),
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Wasm32 => {
|
||||
wasm32::horiz_convolution(src_image, dst_image, offset, coeffs)
|
||||
}
|
||||
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,161 @@
|
||||
use std::arch::wasm32::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::pixels::U8;
|
||||
use crate::wasm32_utils;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn horiz_convolution(
|
||||
src_image: &ImageView<U8>,
|
||||
dst_image: &mut ImageViewMut<U8>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
let mut yy = dst_height - dst_height % 4;
|
||||
while yy < dst_height {
|
||||
unsafe {
|
||||
horiz_convolution_row(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
&coefficients_chunks,
|
||||
&normalizer,
|
||||
);
|
||||
}
|
||||
yy += 1;
|
||||
}
|
||||
}
|
||||
|
||||
/// For safety, it is necessary to ensure the following conditions:
|
||||
/// - length of all rows in src_rows must be equal
|
||||
/// - length of all rows in dst_rows must be equal
|
||||
/// - coefficients_chunks.len() == dst_rows.0.len()
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[inline]
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U8]; 4],
|
||||
dst_rows: [&mut &mut [U8]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
let zero = i64x2_splat(0);
|
||||
let initial = 1 << (normalizer.precision() - 1);
|
||||
let mut buf = [0, 0, 0, 0, initial];
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let coeffs = coeffs_chunk.values;
|
||||
let mut x = coeffs_chunk.start as usize;
|
||||
let mut result_i32x4 = [zero, zero, zero, zero];
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
let reminder8 = coeffs_by_8.remainder();
|
||||
for k in coeffs_by_8 {
|
||||
let coeffs_i16x8 = v128_load(k.as_ptr() as *const v128);
|
||||
for i in 0..4 {
|
||||
let pixels_u8x8 = wasm32_utils::loadl_i64(src_rows[i], x);
|
||||
let pixels_i16x8 = u16x8_extend_low_u8x16(pixels_u8x8);
|
||||
result_i32x4[i] =
|
||||
i32x4_add(result_i32x4[i], i32x4_dot_i16x8(pixels_i16x8, coeffs_i16x8));
|
||||
}
|
||||
x += 8;
|
||||
}
|
||||
|
||||
let mut coeffs_by_4 = reminder8.chunks_exact(4);
|
||||
let reminder4 = coeffs_by_4.remainder();
|
||||
if let Some(k) = coeffs_by_4.next() {
|
||||
let coeffs_i16x4 = wasm32_utils::loadl_i64(k, 0);
|
||||
for i in 0..4 {
|
||||
let pixels_u8x4 = wasm32_utils::loadl_i32(src_rows[i], x);
|
||||
let pixels_i16x4 = u16x8_extend_low_u8x16(pixels_u8x4);
|
||||
result_i32x4[i] =
|
||||
i32x4_add(result_i32x4[i], i32x4_dot_i16x8(pixels_i16x4, coeffs_i16x4));
|
||||
}
|
||||
x += 4;
|
||||
}
|
||||
|
||||
let mut result_i32x4 = result_i32x4.map(|v| {
|
||||
v128_store(buf.as_mut_ptr() as *mut v128, v);
|
||||
buf.iter().sum()
|
||||
});
|
||||
|
||||
for &coeff in reminder4 {
|
||||
let coeff_i32 = coeff as i32;
|
||||
for i in 0..4 {
|
||||
result_i32x4[i] += src_rows[i].get_unchecked(x).0.to_owned() as i32 * coeff_i32;
|
||||
}
|
||||
x += 1;
|
||||
}
|
||||
|
||||
let result_u8x4 = result_i32x4.map(|v| normalizer.clip(v));
|
||||
for i in 0..4 {
|
||||
dst_rows[i].get_unchecked_mut(dst_x).0 = result_u8x4[i];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// For safety, it is necessary to ensure the following conditions:
|
||||
/// - bounds.len() == dst_row.len()
|
||||
/// - coeffs.len() == dst_rows.0.len() * window_size
|
||||
/// - max(bound.start + bound.size for bound in bounds) <= src_row.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[inline]
|
||||
unsafe fn horiz_convolution_row(
|
||||
src_row: &[U8],
|
||||
dst_row: &mut [U8],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
let zero = i64x2_splat(0);
|
||||
let initial = 1 << (normalizer.precision() - 1);
|
||||
let mut buf = [0, 0, 0, 0, initial];
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let coeffs = coeffs_chunk.values;
|
||||
let mut x = coeffs_chunk.start as usize;
|
||||
let mut result_i32x4 = zero;
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
let reminder8 = coeffs_by_8.remainder();
|
||||
for k in coeffs_by_8 {
|
||||
let coeffs_i16x8 = v128_load(k.as_ptr() as *const v128);
|
||||
let pixels_u8x8 = wasm32_utils::loadl_i64(src_row, x);
|
||||
let pixels_i16x8 = u16x8_extend_low_u8x16(pixels_u8x8);
|
||||
result_i32x4 = i32x4_add(result_i32x4, i32x4_dot_i16x8(pixels_i16x8, coeffs_i16x8));
|
||||
x += 8;
|
||||
}
|
||||
|
||||
let mut coeffs_by_4 = reminder8.chunks_exact(4);
|
||||
let reminder4 = coeffs_by_4.remainder();
|
||||
if let Some(k) = coeffs_by_4.next() {
|
||||
let coeffs_i16x4 = wasm32_utils::loadl_i64(k, 0);
|
||||
let pixels_u8x4 = wasm32_utils::loadl_i32(src_row, x);
|
||||
let pixels_i16x4 = u16x8_extend_low_u8x16(pixels_u8x4);
|
||||
result_i32x4 = i32x4_add(result_i32x4, i32x4_dot_i16x8(pixels_i16x4, coeffs_i16x4));
|
||||
x += 4;
|
||||
}
|
||||
|
||||
v128_store(buf.as_mut_ptr() as *mut v128, result_i32x4);
|
||||
let mut result_i32 = buf.iter().sum();
|
||||
|
||||
for &coeff in reminder4 {
|
||||
let coeff_i32 = coeff as i32;
|
||||
result_i32 += src_row.get_unchecked(x).0 as i32 * coeff_i32;
|
||||
x += 1;
|
||||
}
|
||||
|
||||
dst_row.get_unchecked_mut(dst_x).0 = normalizer.clip(result_i32);
|
||||
}
|
||||
}
|
||||
@@ -12,6 +12,8 @@ mod native;
|
||||
mod neon;
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod sse4;
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
mod wasm32;
|
||||
|
||||
impl Convolution for U8x3 {
|
||||
fn horiz_convolution(
|
||||
@@ -28,6 +30,10 @@ impl Convolution for U8x3 {
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_image, dst_image, offset, coeffs),
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => neon::horiz_convolution(src_image, dst_image, offset, coeffs),
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Wasm32 => {
|
||||
wasm32::horiz_convolution(src_image, dst_image, offset, coeffs)
|
||||
}
|
||||
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,290 @@
|
||||
use std::arch::wasm32::*;
|
||||
use std::intrinsics::transmute;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::pixels::U8x3;
|
||||
use crate::wasm32_utils;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn horiz_convolution(
|
||||
src_image: &ImageView<U8x3>,
|
||||
dst_image: &mut ImageViewMut<U8x3>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_8u4x(src_rows, dst_rows, &coefficients_chunks, precision);
|
||||
}
|
||||
}
|
||||
|
||||
let mut yy = dst_height - dst_height % 4;
|
||||
while yy < dst_height {
|
||||
unsafe {
|
||||
horiz_convolution_8u(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
&coefficients_chunks,
|
||||
precision,
|
||||
);
|
||||
}
|
||||
yy += 1;
|
||||
}
|
||||
}
|
||||
|
||||
/// For safety, it is necessary to ensure the following conditions:
|
||||
/// - length of all rows in src_rows must be equal
|
||||
/// - length of all rows in dst_rows must be equal
|
||||
/// - coefficients_chunks.len() == dst_rows.0.len()
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[inline]
|
||||
unsafe fn horiz_convolution_8u4x(
|
||||
src_rows: [&[U8x3]; 4],
|
||||
dst_rows: [&mut &mut [U8x3]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
precision: u8,
|
||||
) {
|
||||
let zero = i64x2_splat(0);
|
||||
let initial = i32x4_splat(1 << (precision - 1));
|
||||
let src_width = src_rows[0].len();
|
||||
|
||||
/*
|
||||
|R G B | |R G B | |R G B | |R G B | |R G B | |R |
|
||||
|00 01 02| |03 04 05| |06 07 08| |09 10 11| |12 13 14| |15|
|
||||
|
||||
Ignore 12-15 bytes in register and
|
||||
shuffle other components with converting from u8 into i16:
|
||||
|
||||
x: |-1 -1| |-1 -1|
|
||||
B: |-1 05| |-1 02|
|
||||
G: |-1 04| |-1 01|
|
||||
R: |-1 03| |-1 00|
|
||||
*/
|
||||
#[rustfmt::skip]
|
||||
let sh_lo = i8x16(
|
||||
0, -1, 3, -1, 1, -1, 4, -1, 2, -1, 5, -1, -1, -1, -1, -1
|
||||
);
|
||||
/*
|
||||
x: |-1 -1| |-1 -1|
|
||||
B: |-1 11| |-1 08|
|
||||
G: |-1 10| |-1 07|
|
||||
R: |-1 09| |-1 06|
|
||||
*/
|
||||
#[rustfmt::skip]
|
||||
let sh_hi = i8x16(
|
||||
6, -1, 9, -1, 7, -1, 10, -1, 8, -1, 11, -1, -1, -1, -1, -1
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let x_start = coeffs_chunk.start as usize;
|
||||
let mut x = x_start;
|
||||
|
||||
let mut sss_a = [initial; 4];
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
|
||||
// Next block of code will be load source pixels by 16 bytes per time.
|
||||
// We must guarantee what this process will not go beyond
|
||||
// the one row of image.
|
||||
// (16 bytes) / (3 bytes per pixel) = 5 whole pixels + 1 byte
|
||||
let max_x = src_width.saturating_sub(5);
|
||||
if x < max_x {
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
|
||||
for k in coeffs_by_4 {
|
||||
let mmk0 = wasm32_utils::ptr_i16_to_set1_i32(k, 0);
|
||||
let mmk1 = wasm32_utils::ptr_i16_to_set1_i32(k, 2);
|
||||
for i in 0..4 {
|
||||
let source = wasm32_utils::load_v128(src_rows[i], x);
|
||||
let pix = i8x16_swizzle(source, sh_lo);
|
||||
let mut sss = sss_a[i];
|
||||
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk0));
|
||||
let pix = i8x16_swizzle(source, sh_hi);
|
||||
sss_a[i] = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk1));
|
||||
}
|
||||
|
||||
x += 4;
|
||||
if x >= max_x {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Next block of code will be load source pixels by 8 bytes per time.
|
||||
// We must guarantee what this process will not go beyond
|
||||
// the one row of image.
|
||||
// (8 bytes) / (3 bytes per pixel) = 2 whole pixels + 2 bytes
|
||||
let max_x = src_width.saturating_sub(2);
|
||||
if x < max_x {
|
||||
let coeffs_by_2 = coeffs[x - x_start..].chunks_exact(2);
|
||||
|
||||
for k in coeffs_by_2 {
|
||||
let mmk = wasm32_utils::ptr_i16_to_set1_i32(k, 0);
|
||||
|
||||
for i in 0..4 {
|
||||
let source = wasm32_utils::loadl_i64(src_rows[i], x);
|
||||
let pix = i8x16_swizzle(source, sh_lo);
|
||||
sss_a[i] = i32x4_add(sss_a[i], i32x4_dot_i16x8(pix, mmk));
|
||||
}
|
||||
|
||||
x += 2;
|
||||
if x >= max_x {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
coeffs = coeffs.split_at(x - x_start).1;
|
||||
for &k in coeffs {
|
||||
let mmk = i32x4_splat(k as i32);
|
||||
for i in 0..4 {
|
||||
let pix = wasm32_utils::i32x4_extend_low_ptr_u8x3(src_rows[i], x);
|
||||
sss_a[i] = i32x4_add(sss_a[i], i32x4_dot_i16x8(pix, mmk));
|
||||
}
|
||||
|
||||
x += 1;
|
||||
}
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss_a[0] = i32x4_shr(sss_a[0], $imm8);
|
||||
sss_a[1] = i32x4_shr(sss_a[1], $imm8);
|
||||
sss_a[2] = i32x4_shr(sss_a[2], $imm8);
|
||||
sss_a[3] = i32x4_shr(sss_a[3], $imm8);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
|
||||
for i in 0..4 {
|
||||
let sss = i16x8_narrow_i32x4(sss_a[i], zero);
|
||||
let pixel: u32 = transmute(i32x4_extract_lane::<0>(u8x16_narrow_i16x8(sss, zero)));
|
||||
let bytes = pixel.to_le_bytes();
|
||||
dst_rows[i].get_unchecked_mut(dst_x).0 = [bytes[0], bytes[1], bytes[2]];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// For safety, it is necessary to ensure the following conditions:
|
||||
/// - bounds.len() == dst_row.len()
|
||||
/// - coeffs.len() == dst_rows.0.len() * window_size
|
||||
/// - max(bound.start + bound.size for bound in bounds) <= src_row.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[inline]
|
||||
unsafe fn horiz_convolution_8u(
|
||||
src_row: &[U8x3],
|
||||
dst_row: &mut [U8x3],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
precision: u8,
|
||||
) {
|
||||
#[rustfmt::skip]
|
||||
let pix_sh1 = i8x16(
|
||||
0, -1, 3, -1, 1, -1, 4, -1, 2, -1, 5, -1, -1, -1, -1, -1
|
||||
);
|
||||
#[rustfmt::skip]
|
||||
let coef_sh1 = i8x16(
|
||||
0, 1, 2, 3, 0, 1, 2, 3, 0, 1, 2, 3, 0, 1, 2, 3
|
||||
);
|
||||
#[rustfmt::skip]
|
||||
let pix_sh2 = i8x16(
|
||||
6, -1, 9, -1, 7, -1, 10, -1, 8, -1, 11, -1, -1, -1, -1, -1
|
||||
);
|
||||
#[rustfmt::skip]
|
||||
let coef_sh2 = i8x16(
|
||||
4, 5, 6, 7, 4, 5, 6, 7, 4, 5, 6, 7, 4, 5, 6, 7
|
||||
);
|
||||
/*
|
||||
Load 8 bytes from memory into low half of 16-bytes register:
|
||||
|R G B | |R G B | |R G |
|
||||
|00 01 02| |03 04 05| |06 07| 08 09 10 11 12 13 14 15
|
||||
|
||||
Ignore 06-16 bytes in 16-bytes register and
|
||||
shuffle other components with converting from u8 into i16:
|
||||
|
||||
x: |-1 -1| |-1 -1|
|
||||
B: |-1 05| |-1 02|
|
||||
G: |-1 04| |-1 01|
|
||||
R: |-1 03| |-1 00|
|
||||
*/
|
||||
let src_width = src_row.len();
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let x_start = coeffs_chunk.start as usize;
|
||||
let mut x = x_start;
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut sss = i32x4_splat(1 << (precision - 1));
|
||||
|
||||
// Next block of code will be load source pixels by 16 bytes per time.
|
||||
// We must guarantee what this process will not go beyond
|
||||
// the one row of image.
|
||||
// (16 bytes) / (3 bytes per pixel) = 5 whole pixels + 1 bytes
|
||||
let max_x = src_width.saturating_sub(5);
|
||||
if x < max_x {
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
for k in coeffs_by_4 {
|
||||
let ksource = wasm32_utils::loadl_i64(k, 0);
|
||||
let source = wasm32_utils::load_v128(src_row, x);
|
||||
|
||||
let pix = i8x16_swizzle(source, pix_sh1);
|
||||
let mmk = i8x16_swizzle(ksource, coef_sh1);
|
||||
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
let pix = i8x16_swizzle(source, pix_sh2);
|
||||
let mmk = i8x16_swizzle(ksource, coef_sh2);
|
||||
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
x += 4;
|
||||
if x >= max_x {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Next block of code will be load source pixels by 8 bytes per time.
|
||||
// We must guarantee what this process will not go beyond
|
||||
// the one row of image.
|
||||
// (8 bytes) / (3 bytes per pixel) = 2 whole pixels + 2 bytes
|
||||
let max_x = src_width.saturating_sub(2);
|
||||
if x < max_x {
|
||||
let coeffs_by_2 = coeffs[x - x_start..].chunks_exact(2);
|
||||
|
||||
for k in coeffs_by_2 {
|
||||
let mmk = wasm32_utils::ptr_i16_to_set1_i32(k, 0);
|
||||
let source = wasm32_utils::loadl_i64(src_row, x);
|
||||
let pix = i8x16_swizzle(source, pix_sh1);
|
||||
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
x += 2;
|
||||
if x >= max_x {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
coeffs = coeffs.split_at(x - x_start).1;
|
||||
for &k in coeffs {
|
||||
let pix = wasm32_utils::i32x4_extend_low_ptr_u8x3(src_row, x);
|
||||
let mmk = i32x4_splat(k as i32);
|
||||
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
|
||||
x += 1;
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss = i32x4_shr(sss, $imm8);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
|
||||
sss = i16x8_narrow_i32x4(sss, sss);
|
||||
let pixel: u32 = transmute(i32x4_extract_lane::<0>(u8x16_narrow_i16x8(sss, sss)));
|
||||
let bytes = pixel.to_le_bytes();
|
||||
dst_row.get_unchecked_mut(dst_x).0 = [bytes[0], bytes[1], bytes[2]];
|
||||
}
|
||||
}
|
||||
@@ -12,6 +12,8 @@ mod native;
|
||||
mod neon;
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod sse4;
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
mod wasm32;
|
||||
|
||||
impl Convolution for U8x4 {
|
||||
fn horiz_convolution(
|
||||
@@ -28,6 +30,10 @@ impl Convolution for U8x4 {
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_image, dst_image, offset, coeffs),
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => neon::horiz_convolution(src_image, dst_image, offset, coeffs),
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Wasm32 => {
|
||||
wasm32::horiz_convolution(src_image, dst_image, offset, coeffs)
|
||||
}
|
||||
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,280 @@
|
||||
use std::arch::wasm32::*;
|
||||
use std::intrinsics::transmute;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::pixels::U8x4;
|
||||
use crate::wasm32_utils;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
// This code is based on C-implementation from Pillow-SIMD package for Python
|
||||
// https://github.com/uploadcare/pillow-simd
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn horiz_convolution(
|
||||
src_image: &ImageView<U8x4>,
|
||||
dst_image: &mut ImageViewMut<U8x4>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_8u4x(src_rows, dst_rows, &coefficients_chunks, precision);
|
||||
}
|
||||
}
|
||||
|
||||
let mut yy = dst_height - dst_height % 4;
|
||||
while yy < dst_height {
|
||||
unsafe {
|
||||
horiz_convolution_8u(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
&coefficients_chunks,
|
||||
precision,
|
||||
);
|
||||
}
|
||||
yy += 1;
|
||||
}
|
||||
}
|
||||
|
||||
/// For safety, it is necessary to ensure the following conditions:
|
||||
/// - length of all rows in src_rows must be equal
|
||||
/// - length of all rows in dst_rows must be equal
|
||||
/// - coefficients_chunks.len() == dst_rows.0.len()
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
unsafe fn horiz_convolution_8u4x(
|
||||
src_rows: [&[U8x4]; 4],
|
||||
dst_rows: [&mut &mut [U8x4]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
precision: u8,
|
||||
) {
|
||||
let initial = i32x4_splat(1 << (precision - 1));
|
||||
let mask_lo = i8x16(0, -1, 4, -1, 1, -1, 5, -1, 2, -1, 6, -1, 3, -1, 7, -1);
|
||||
let mask_hi = i8x16(8, -1, 12, -1, 9, -1, 13, -1, 10, -1, 14, -1, 11, -1, 15, -1);
|
||||
let mask = i8x16(0, -1, 4, -1, 1, -1, 5, -1, 2, -1, 6, -1, 3, -1, 7, -1);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
|
||||
let mut sss0 = initial;
|
||||
let mut sss1 = initial;
|
||||
let mut sss2 = initial;
|
||||
let mut sss3 = initial;
|
||||
|
||||
let coeffs = coeffs_chunk.values;
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
let reminder1 = coeffs_by_4.remainder();
|
||||
|
||||
for k in coeffs_by_4 {
|
||||
let mmk_lo = wasm32_utils::ptr_i16_to_set1_i32(k, 0);
|
||||
let mmk_hi = wasm32_utils::ptr_i16_to_set1_i32(k, 2);
|
||||
|
||||
// [8] a3 b3 g3 r3 a2 b2 g2 r2 a1 b1 g1 r1 a0 b0 g0 r0
|
||||
let mut source = wasm32_utils::load_v128(src_rows[0], x);
|
||||
// [16] a1 a0 b1 b0 g1 g0 r1 r0
|
||||
let mut pix = i8x16_swizzle(source, mask_lo);
|
||||
sss0 = i32x4_add(sss0, i32x4_dot_i16x8(pix, mmk_lo));
|
||||
// [16] a3 a2 b3 b2 g3 g2 r3 r2
|
||||
pix = i8x16_swizzle(source, mask_hi);
|
||||
sss0 = i32x4_add(sss0, i32x4_dot_i16x8(pix, mmk_hi));
|
||||
|
||||
source = wasm32_utils::load_v128(src_rows[1], x);
|
||||
pix = i8x16_swizzle(source, mask_lo);
|
||||
sss1 = i32x4_add(sss1, i32x4_dot_i16x8(pix, mmk_lo));
|
||||
pix = i8x16_swizzle(source, mask_hi);
|
||||
sss1 = i32x4_add(sss1, i32x4_dot_i16x8(pix, mmk_hi));
|
||||
|
||||
source = wasm32_utils::load_v128(src_rows[2], x);
|
||||
pix = i8x16_swizzle(source, mask_lo);
|
||||
sss2 = i32x4_add(sss2, i32x4_dot_i16x8(pix, mmk_lo));
|
||||
pix = i8x16_swizzle(source, mask_hi);
|
||||
sss2 = i32x4_add(sss2, i32x4_dot_i16x8(pix, mmk_hi));
|
||||
|
||||
source = wasm32_utils::load_v128(src_rows[3], x);
|
||||
pix = i8x16_swizzle(source, mask_lo);
|
||||
sss3 = i32x4_add(sss3, i32x4_dot_i16x8(pix, mmk_lo));
|
||||
pix = i8x16_swizzle(source, mask_hi);
|
||||
sss3 = i32x4_add(sss3, i32x4_dot_i16x8(pix, mmk_hi));
|
||||
x += 4;
|
||||
}
|
||||
|
||||
let coeffs_by_2 = reminder1.chunks_exact(2);
|
||||
let reminder2 = coeffs_by_2.remainder();
|
||||
|
||||
for k in coeffs_by_2 {
|
||||
// [16] k1 k0 k1 k0 k1 k0 k1 k0
|
||||
let mmk = wasm32_utils::ptr_i16_to_set1_i32(k, 0);
|
||||
|
||||
// [8] x x x x x x x x a1 b1 g1 r1 a0 b0 g0 r0
|
||||
let mut pix = wasm32_utils::loadl_i64(src_rows[0], x);
|
||||
// [16] a1 a0 b1 b0 g1 g0 r1 r0
|
||||
pix = i8x16_swizzle(pix, mask);
|
||||
sss0 = i32x4_add(sss0, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
pix = wasm32_utils::loadl_i64(src_rows[1], x);
|
||||
pix = i8x16_swizzle(pix, mask);
|
||||
sss1 = i32x4_add(sss1, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
pix = wasm32_utils::loadl_i64(src_rows[2], x);
|
||||
pix = i8x16_swizzle(pix, mask);
|
||||
sss2 = i32x4_add(sss2, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
pix = wasm32_utils::loadl_i64(src_rows[3], x);
|
||||
pix = i8x16_swizzle(pix, mask);
|
||||
sss3 = i32x4_add(sss3, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
x += 2;
|
||||
}
|
||||
|
||||
if let Some(&k) = reminder2.first() {
|
||||
// [16] xx k0 xx k0 xx k0 xx k0
|
||||
let mmk = i32x4_splat(k as i32);
|
||||
// [16] xx a0 xx b0 xx g0 xx r0
|
||||
let mut pix = wasm32_utils::i32x4_extend_low_ptr_u8x4(src_rows[0], x);
|
||||
sss0 = i32x4_add(sss0, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
pix = wasm32_utils::i32x4_extend_low_ptr_u8x4(src_rows[1], x);
|
||||
sss1 = i32x4_add(sss1, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
pix = wasm32_utils::i32x4_extend_low_ptr_u8x4(src_rows[2], x);
|
||||
sss2 = i32x4_add(sss2, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
pix = wasm32_utils::i32x4_extend_low_ptr_u8x4(src_rows[3], x);
|
||||
sss3 = i32x4_add(sss3, i32x4_dot_i16x8(pix, mmk));
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss0 = i32x4_shr(sss0, $imm8);
|
||||
sss1 = i32x4_shr(sss1, $imm8);
|
||||
sss2 = i32x4_shr(sss2, $imm8);
|
||||
sss3 = i32x4_shr(sss3, $imm8);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
|
||||
sss0 = i16x8_narrow_i32x4(sss0, sss0);
|
||||
sss1 = i16x8_narrow_i32x4(sss1, sss1);
|
||||
sss2 = i16x8_narrow_i32x4(sss2, sss2);
|
||||
sss3 = i16x8_narrow_i32x4(sss3, sss3);
|
||||
*dst_rows[0].get_unchecked_mut(dst_x) =
|
||||
transmute(i32x4_extract_lane::<0>(u8x16_narrow_i16x8(sss0, sss0)));
|
||||
*dst_rows[1].get_unchecked_mut(dst_x) =
|
||||
transmute(i32x4_extract_lane::<0>(u8x16_narrow_i16x8(sss1, sss1)));
|
||||
*dst_rows[2].get_unchecked_mut(dst_x) =
|
||||
transmute(i32x4_extract_lane::<0>(u8x16_narrow_i16x8(sss2, sss2)));
|
||||
*dst_rows[3].get_unchecked_mut(dst_x) =
|
||||
transmute(i32x4_extract_lane::<0>(u8x16_narrow_i16x8(sss3, sss3)));
|
||||
}
|
||||
}
|
||||
|
||||
/// For safety, it is necessary to ensure the following conditions:
|
||||
/// - bounds.len() == dst_row.len()
|
||||
/// - coefficients_chunks.len() == dst_row.len()
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
unsafe fn horiz_convolution_8u(
|
||||
src_row: &[U8x4],
|
||||
dst_row: &mut [U8x4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
precision: u8,
|
||||
) {
|
||||
let initial = i32x4_splat(1 << (precision - 1));
|
||||
let sh1 = i8x16(0, -1, 8, -1, 1, -1, 9, -1, 2, -1, 10, -1, 3, -1, 11, -1);
|
||||
let sh2 = i8x16(0, 1, 4, 5, 0, 1, 4, 5, 0, 1, 4, 5, 0, 1, 4, 5);
|
||||
let sh3 = i8x16(4, -1, 12, -1, 5, -1, 13, -1, 6, -1, 14, -1, 7, -1, 15, -1);
|
||||
let sh4 = i8x16(2, 3, 6, 7, 2, 3, 6, 7, 2, 3, 6, 7, 2, 3, 6, 7);
|
||||
let sh5 = i8x16(8, 9, 12, 13, 8, 9, 12, 13, 8, 9, 12, 13, 8, 9, 12, 13);
|
||||
let sh6 = i8x16(
|
||||
10, 11, 14, 15, 10, 11, 14, 15, 10, 11, 14, 15, 10, 11, 14, 15,
|
||||
);
|
||||
let sh7 = i8x16(0, -1, 4, -1, 1, -1, 5, -1, 2, -1, 6, -1, 3, -1, 7, -1);
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut sss = initial;
|
||||
|
||||
let coeffs_by_8 = coeffs_chunk.values.chunks_exact(8);
|
||||
let reminder8 = coeffs_by_8.remainder();
|
||||
|
||||
for k in coeffs_by_8 {
|
||||
let ksource = wasm32_utils::load_v128(k, 0);
|
||||
|
||||
let mut source = wasm32_utils::load_v128(src_row, x);
|
||||
|
||||
let mut pix = i8x16_swizzle(source, sh1);
|
||||
let mut mmk = i8x16_swizzle(ksource, sh2);
|
||||
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
pix = i8x16_swizzle(source, sh3);
|
||||
mmk = i8x16_swizzle(ksource, sh4);
|
||||
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
source = wasm32_utils::load_v128(src_row, x + 4);
|
||||
|
||||
pix = i8x16_swizzle(source, sh1);
|
||||
mmk = i8x16_swizzle(ksource, sh5);
|
||||
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
pix = i8x16_swizzle(source, sh3);
|
||||
mmk = i8x16_swizzle(ksource, sh6);
|
||||
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
x += 8;
|
||||
}
|
||||
|
||||
let coeffs_by_4 = reminder8.chunks_exact(4);
|
||||
let reminder4 = coeffs_by_4.remainder();
|
||||
|
||||
for k in coeffs_by_4 {
|
||||
let source = wasm32_utils::load_v128(src_row, x);
|
||||
let ksource = wasm32_utils::loadl_i64(k, 0);
|
||||
|
||||
let mut pix = i8x16_swizzle(source, sh1);
|
||||
let mut mmk = i8x16_swizzle(ksource, sh2);
|
||||
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
pix = i8x16_swizzle(source, sh3);
|
||||
mmk = i8x16_swizzle(ksource, sh4);
|
||||
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
x += 4;
|
||||
}
|
||||
|
||||
let coeffs_by_2 = reminder4.chunks_exact(2);
|
||||
let reminder2 = coeffs_by_2.remainder();
|
||||
|
||||
for k in coeffs_by_2 {
|
||||
let mmk = wasm32_utils::ptr_i16_to_set1_i32(k, 0);
|
||||
let source = wasm32_utils::loadl_i64(src_row, x);
|
||||
let pix = i8x16_swizzle(source, sh7);
|
||||
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
x += 2
|
||||
}
|
||||
|
||||
if let Some(&k) = reminder2.first() {
|
||||
let pix = wasm32_utils::i32x4_extend_low_ptr_u8x4(src_row, x);
|
||||
let mmk = i32x4_splat(k as i32);
|
||||
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss = i32x4_shr(sss, $imm8);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
|
||||
sss = i16x8_narrow_i32x4(sss, sss);
|
||||
*dst_row.get_unchecked_mut(dst_x) =
|
||||
transmute(i32x4_extract_lane::<0>(u8x16_narrow_i16x8(sss, sss)));
|
||||
}
|
||||
}
|
||||
@@ -6,8 +6,6 @@ use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::pixels::PixelExt;
|
||||
use crate::simd_utils;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
use std::fs;
|
||||
use std::path::Path;
|
||||
|
||||
pub(crate) fn vert_convolution<T: PixelExt<Component = u16>>(
|
||||
src_image: &ImageView<T>,
|
||||
@@ -35,9 +33,6 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
|
||||
coeffs_chunk: CoefficientsI32Chunk,
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let file = "vsse4";
|
||||
let mut debugout = String::new();
|
||||
let file_exists = Path::new(file).exists();
|
||||
let y_start = coeffs_chunk.start;
|
||||
let coeffs = coeffs_chunk.values;
|
||||
let max_y = y_start + coeffs.len() as u32;
|
||||
@@ -116,13 +111,6 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
|
||||
for x in 0..2 {
|
||||
for sum in sums {
|
||||
_mm_storeu_si128((&mut c_buf).as_mut_ptr() as *mut __m128i, sum[x]);
|
||||
if !file_exists {
|
||||
debugout += &format!(
|
||||
"119: {:?} {:?}\n",
|
||||
_mm_extract_epi64(sum[x], 0),
|
||||
_mm_extract_epi64(sum[x], 1)
|
||||
);
|
||||
}
|
||||
*dst_ptr = normalizer.clip(c_buf[0]);
|
||||
dst_ptr = dst_ptr.add(1);
|
||||
*dst_ptr = normalizer.clip(c_buf[1]);
|
||||
@@ -177,13 +165,6 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
|
||||
// sums[i] = _mm_srl_epi64(sums[i] , precision_i64);
|
||||
// _mm_packus_epi32(sums[i] , sums[i] );
|
||||
_mm_storeu_si128((&mut c_buf).as_mut_ptr() as *mut __m128i, sum);
|
||||
if !file_exists {
|
||||
debugout += &format!(
|
||||
"176: {:?} {:?}\n",
|
||||
_mm_extract_epi64(sum, 0),
|
||||
_mm_extract_epi64(sum, 1)
|
||||
);
|
||||
}
|
||||
*dst_ptr = normalizer.clip(c_buf[0]);
|
||||
dst_ptr = dst_ptr.add(1);
|
||||
*dst_ptr = normalizer.clip(c_buf[1]);
|
||||
@@ -232,34 +213,17 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
|
||||
|
||||
let mut dst_ptr = dst_chunk.as_mut_ptr();
|
||||
_mm_storeu_si128((&mut c_buf).as_mut_ptr() as *mut __m128i, c01);
|
||||
if !file_exists {
|
||||
debugout += &format!(
|
||||
"227: {:?} {:?}\n",
|
||||
_mm_extract_epi64(c01, 0),
|
||||
_mm_extract_epi64(c01, 1)
|
||||
);
|
||||
}
|
||||
*dst_ptr = normalizer.clip(c_buf[0]);
|
||||
dst_ptr = dst_ptr.add(1);
|
||||
*dst_ptr = normalizer.clip(c_buf[1]);
|
||||
dst_ptr = dst_ptr.add(1);
|
||||
_mm_storeu_si128((&mut c_buf).as_mut_ptr() as *mut __m128i, c23);
|
||||
if !file_exists {
|
||||
debugout += &format!(
|
||||
"236: {:?} {:?}\n",
|
||||
_mm_extract_epi64(c23, 0),
|
||||
_mm_extract_epi64(c23, 1)
|
||||
);
|
||||
}
|
||||
*dst_ptr = normalizer.clip(c_buf[0]);
|
||||
dst_ptr = dst_ptr.add(1);
|
||||
*dst_ptr = normalizer.clip(c_buf[1]);
|
||||
|
||||
src_x += 4;
|
||||
}
|
||||
if !file_exists {
|
||||
fs::write(file, debugout).unwrap();
|
||||
}
|
||||
|
||||
dst_u16 = dst_chunks_4.into_remainder();
|
||||
if !dst_u16.is_empty() {
|
||||
|
||||
@@ -6,8 +6,6 @@ use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::pixels::PixelExt;
|
||||
use crate::wasm32_utils;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
use std::fs;
|
||||
use std::path::Path;
|
||||
|
||||
pub(crate) fn vert_convolution<T: PixelExt<Component = u16>>(
|
||||
src_image: &ImageView<T>,
|
||||
@@ -34,9 +32,6 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
|
||||
coeffs_chunk: CoefficientsI32Chunk,
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let file = "vwasm32";
|
||||
let mut debugout = String::new();
|
||||
let file_exists = Path::new(file).exists();
|
||||
let y_start = coeffs_chunk.start;
|
||||
let coeffs = coeffs_chunk.values;
|
||||
let max_y = y_start + coeffs.len() as u32;
|
||||
@@ -115,13 +110,6 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
|
||||
for x in 0..2 {
|
||||
for sum in sums {
|
||||
v128_store((&mut c_buf).as_mut_ptr() as *mut v128, sum[x]);
|
||||
if !file_exists {
|
||||
debugout += &format!(
|
||||
"119: {:?} {:?}\n",
|
||||
i64x2_extract_lane::<0>(sum[x]),
|
||||
i64x2_extract_lane::<1>(sum[x])
|
||||
);
|
||||
}
|
||||
*dst_ptr = normalizer.clip(c_buf[0]);
|
||||
dst_ptr = dst_ptr.add(1);
|
||||
*dst_ptr = normalizer.clip(c_buf[1]);
|
||||
@@ -176,13 +164,6 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
|
||||
// sums[i] = _mm_srl_epi64(sums[i] , precision_i64);
|
||||
// _mm_packus_epi32(sums[i] , sums[i] );
|
||||
v128_store((&mut c_buf).as_mut_ptr() as *mut v128, sum);
|
||||
if !file_exists {
|
||||
debugout += &format!(
|
||||
"176: {:?} {:?}\n",
|
||||
i64x2_extract_lane::<0>(sum),
|
||||
i64x2_extract_lane::<1>(sum)
|
||||
);
|
||||
}
|
||||
*dst_ptr = normalizer.clip(c_buf[0]);
|
||||
dst_ptr = dst_ptr.add(1);
|
||||
*dst_ptr = normalizer.clip(c_buf[1]);
|
||||
@@ -231,34 +212,17 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
|
||||
|
||||
let mut dst_ptr = dst_chunk.as_mut_ptr();
|
||||
v128_store((&mut c_buf).as_mut_ptr() as *mut v128, c01);
|
||||
if !file_exists {
|
||||
debugout += &format!(
|
||||
"227: {:?} {:?}\n",
|
||||
i64x2_extract_lane::<0>(c01),
|
||||
i64x2_extract_lane::<1>(c01)
|
||||
);
|
||||
}
|
||||
*dst_ptr = normalizer.clip(c_buf[0]);
|
||||
dst_ptr = dst_ptr.add(1);
|
||||
*dst_ptr = normalizer.clip(c_buf[1]);
|
||||
dst_ptr = dst_ptr.add(1);
|
||||
v128_store((&mut c_buf).as_mut_ptr() as *mut v128, c23);
|
||||
if !file_exists {
|
||||
debugout += &format!(
|
||||
"236: {:?} {:?}\n",
|
||||
i64x2_extract_lane::<0>(c23),
|
||||
i64x2_extract_lane::<1>(c23)
|
||||
);
|
||||
}
|
||||
*dst_ptr = normalizer.clip(c_buf[0]);
|
||||
dst_ptr = dst_ptr.add(1);
|
||||
*dst_ptr = normalizer.clip(c_buf[1]);
|
||||
|
||||
src_x += 4;
|
||||
}
|
||||
if !file_exists {
|
||||
fs::write(file, debugout).unwrap();
|
||||
}
|
||||
|
||||
dst_u16 = dst_chunks_4.into_remainder();
|
||||
if !dst_u16.is_empty() {
|
||||
|
||||
@@ -10,6 +10,8 @@ pub(crate) mod native;
|
||||
mod neon;
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
pub(crate) mod sse4;
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
pub(crate) mod wasm32;
|
||||
|
||||
pub(crate) fn vert_convolution_u8<T: PixelExt<Component = u8>>(
|
||||
src_image: &ImageView<T>,
|
||||
@@ -29,6 +31,8 @@ pub(crate) fn vert_convolution_u8<T: PixelExt<Component = u8>>(
|
||||
CpuExtensions::Sse4_1 => sse4::vert_convolution(src_image, dst_image, offset, coeffs),
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => neon::vert_convolution(src_image, dst_image, offset, coeffs),
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Wasm32 => wasm32::vert_convolution(src_image, dst_image, offset, coeffs),
|
||||
_ => native::vert_convolution(src_image, dst_image, offset, coeffs),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,292 @@
|
||||
use std::arch::wasm32::*;
|
||||
|
||||
use crate::convolution::vertical_u8::native;
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::pixels::PixelExt;
|
||||
use crate::wasm32_utils;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn vert_convolution<T: PixelExt<Component = u8>>(
|
||||
src_image: &ImageView<T>,
|
||||
dst_image: &mut ImageViewMut<T>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let src_x = offset as usize * T::count_of_components();
|
||||
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for (dst_row, coeffs_chunk) in dst_rows.zip(coefficients_chunks) {
|
||||
unsafe {
|
||||
vert_convolution_into_one_row_u8(src_image, dst_row, src_x, coeffs_chunk, &normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
|
||||
src_img: &ImageView<T>,
|
||||
dst_row: &mut [T],
|
||||
mut src_x: usize,
|
||||
coeffs_chunk: optimisations::CoefficientsI16Chunk,
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
let y_start = coeffs_chunk.start;
|
||||
let coeffs = coeffs_chunk.values;
|
||||
let max_y = y_start + coeffs.len() as u32;
|
||||
let precision = normalizer.precision();
|
||||
let mut dst_u8 = T::components_mut(dst_row);
|
||||
|
||||
let initial = i32x4_splat(1 << (precision - 1));
|
||||
|
||||
let mut dst_chunks_32 = dst_u8.chunks_exact_mut(32);
|
||||
for dst_chunk in &mut dst_chunks_32 {
|
||||
let mut sss0 = initial;
|
||||
let mut sss1 = initial;
|
||||
let mut sss2 = initial;
|
||||
let mut sss3 = initial;
|
||||
let mut sss4 = initial;
|
||||
let mut sss5 = initial;
|
||||
let mut sss6 = initial;
|
||||
let mut sss7 = initial;
|
||||
|
||||
let mut y: u32 = 0;
|
||||
|
||||
for src_rows in src_img.iter_2_rows(y_start, max_y) {
|
||||
let components1 = T::components(src_rows[0]);
|
||||
let components2 = T::components(src_rows[1]);
|
||||
|
||||
// Load two coefficients at once
|
||||
let mmk = wasm32_utils::ptr_i16_to_set1_i32(coeffs, y as usize);
|
||||
|
||||
let source1 = wasm32_utils::load_v128(components1, src_x); // top line
|
||||
let source2 = wasm32_utils::load_v128(components2, src_x); // bottom line
|
||||
|
||||
let source = i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(
|
||||
source1, source2,
|
||||
);
|
||||
let pix = i16x8_extend_low_u8x16(source);
|
||||
sss0 = i32x4_add(sss0, i32x4_dot_i16x8(pix, mmk));
|
||||
let pix = i16x8_extend_high_u8x16(source);
|
||||
sss1 = i32x4_add(sss1, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
let source =
|
||||
i8x16_shuffle::<8, 24, 9, 25, 10, 26, 11, 27, 12, 28, 13, 29, 14, 30, 15, 31>(
|
||||
source1, source2,
|
||||
);
|
||||
let pix = i16x8_extend_low_u8x16(source);
|
||||
sss2 = i32x4_add(sss2, i32x4_dot_i16x8(pix, mmk));
|
||||
let pix = i16x8_extend_high_u8x16(source);
|
||||
sss3 = i32x4_add(sss3, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
let source1 = wasm32_utils::load_v128(components1, src_x + 16); // top line
|
||||
let source2 = wasm32_utils::load_v128(components2, src_x + 16); // bottom line
|
||||
|
||||
let source = i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(
|
||||
source1, source2,
|
||||
);
|
||||
let pix = i16x8_extend_low_u8x16(source);
|
||||
sss4 = i32x4_add(sss4, i32x4_dot_i16x8(pix, mmk));
|
||||
let pix = i16x8_extend_high_u8x16(source);
|
||||
sss5 = i32x4_add(sss5, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
let source =
|
||||
i8x16_shuffle::<8, 24, 9, 25, 10, 26, 11, 27, 12, 28, 13, 29, 14, 30, 15, 31>(
|
||||
source1, source2,
|
||||
);
|
||||
let pix = i16x8_extend_low_u8x16(source);
|
||||
sss6 = i32x4_add(sss6, i32x4_dot_i16x8(pix, mmk));
|
||||
let pix = i16x8_extend_high_u8x16(source);
|
||||
sss7 = i32x4_add(sss7, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
y += 2;
|
||||
}
|
||||
|
||||
if let Some(&k) = coeffs.get(y as usize) {
|
||||
let s_row = src_img.get_row(y_start + y).unwrap();
|
||||
let components = T::components(s_row);
|
||||
let mmk = i32x4_splat(k as i32);
|
||||
|
||||
let source1 = wasm32_utils::load_v128(components, src_x); // top line
|
||||
|
||||
let source = i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(
|
||||
source1,
|
||||
i64x2_splat(0),
|
||||
);
|
||||
let pix = i16x8_extend_low_u8x16(source);
|
||||
sss0 = i32x4_add(sss0, i32x4_dot_i16x8(pix, mmk));
|
||||
let pix = i16x8_extend_high_u8x16(source);
|
||||
sss1 = i32x4_add(sss1, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
let source = i16x8_extend_high_u8x16(source1);
|
||||
let pix = i16x8_extend_low_u8x16(source);
|
||||
sss2 = i32x4_add(sss2, i32x4_dot_i16x8(pix, mmk));
|
||||
let pix = i16x8_extend_high_u8x16(source);
|
||||
sss3 = i32x4_add(sss3, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
let source1 = wasm32_utils::load_v128(components, src_x + 16); // top line
|
||||
|
||||
let source = i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(
|
||||
source1,
|
||||
i64x2_splat(0),
|
||||
);
|
||||
let pix = i16x8_extend_low_u8x16(source);
|
||||
sss4 = i32x4_add(sss4, i32x4_dot_i16x8(pix, mmk));
|
||||
let pix = i16x8_extend_high_u8x16(source);
|
||||
sss5 = i32x4_add(sss5, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
let source = i16x8_extend_high_u8x16(source1);
|
||||
let pix = i16x8_extend_low_u8x16(source);
|
||||
sss6 = i32x4_add(sss6, i32x4_dot_i16x8(pix, mmk));
|
||||
let pix = i16x8_extend_high_u8x16(source);
|
||||
sss7 = i32x4_add(sss7, i32x4_dot_i16x8(pix, mmk));
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss0 = i32x4_shr(sss0, $imm8);
|
||||
sss1 = i32x4_shr(sss1, $imm8);
|
||||
sss2 = i32x4_shr(sss2, $imm8);
|
||||
sss3 = i32x4_shr(sss3, $imm8);
|
||||
sss4 = i32x4_shr(sss4, $imm8);
|
||||
sss5 = i32x4_shr(sss5, $imm8);
|
||||
sss6 = i32x4_shr(sss6, $imm8);
|
||||
sss7 = i32x4_shr(sss7, $imm8);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
|
||||
sss0 = i16x8_narrow_i32x4(sss0, sss1);
|
||||
sss2 = i16x8_narrow_i32x4(sss2, sss3);
|
||||
sss0 = u8x16_narrow_i16x8(sss0, sss2);
|
||||
let dst_ptr = dst_chunk.as_mut_ptr() as *mut v128;
|
||||
v128_store(dst_ptr, sss0);
|
||||
sss4 = i16x8_narrow_i32x4(sss4, sss5);
|
||||
sss6 = i16x8_narrow_i32x4(sss6, sss7);
|
||||
sss4 = u8x16_narrow_i16x8(sss4, sss6);
|
||||
let dst_ptr = dst_ptr.add(1);
|
||||
v128_store(dst_ptr, sss4);
|
||||
|
||||
src_x += 32;
|
||||
}
|
||||
|
||||
dst_u8 = dst_chunks_32.into_remainder();
|
||||
let mut dst_chunks_8 = dst_u8.chunks_exact_mut(8);
|
||||
for dst_chunk in &mut dst_chunks_8 {
|
||||
let mut sss0 = initial; // left row
|
||||
let mut sss1 = initial; // right row
|
||||
let mut y: u32 = 0;
|
||||
|
||||
for src_rows in src_img.iter_2_rows(y_start, max_y) {
|
||||
let components1 = T::components(src_rows[0]);
|
||||
let components2 = T::components(src_rows[1]);
|
||||
// Load two coefficients at once
|
||||
let mmk = wasm32_utils::ptr_i16_to_set1_i32(coeffs, y as usize);
|
||||
|
||||
let source1 = wasm32_utils::loadl_i64(components1, src_x); // top line
|
||||
let source2 = wasm32_utils::loadl_i64(components2, src_x); // bottom line
|
||||
|
||||
let source = i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(
|
||||
source1, source2,
|
||||
);
|
||||
let pix = i16x8_extend_low_u8x16(source);
|
||||
sss0 = i32x4_add(sss0, i32x4_dot_i16x8(pix, mmk));
|
||||
let pix = i16x8_extend_high_u8x16(source);
|
||||
sss1 = i32x4_add(sss1, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
y += 2;
|
||||
}
|
||||
|
||||
if let Some(&k) = coeffs.get(y as usize) {
|
||||
let s_row = src_img.get_row(y_start + y).unwrap();
|
||||
let components = T::components(s_row);
|
||||
let mmk = i32x4_splat(k as i32);
|
||||
|
||||
let source1 = wasm32_utils::loadl_i64(components, src_x); // top line
|
||||
|
||||
let source = i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(
|
||||
source1,
|
||||
i64x2_splat(0),
|
||||
);
|
||||
let pix = i16x8_extend_low_u8x16(source);
|
||||
sss0 = i32x4_add(sss0, i32x4_dot_i16x8(pix, mmk));
|
||||
let pix = i16x8_extend_high_u8x16(source);
|
||||
sss1 = i32x4_add(sss1, i32x4_dot_i16x8(pix, mmk));
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss0 = i32x4_shr(sss0, $imm8);
|
||||
sss1 = i32x4_shr(sss1, $imm8);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
|
||||
sss0 = i16x8_narrow_i32x4(sss0, sss1);
|
||||
sss0 = u8x16_narrow_i16x8(sss0, sss0);
|
||||
let dst_ptr = dst_chunk.as_mut_ptr() as *mut [i64; 2];
|
||||
(*dst_ptr)[0] = i64x2_extract_lane::<0>(sss0);
|
||||
|
||||
src_x += 8;
|
||||
}
|
||||
|
||||
dst_u8 = dst_chunks_8.into_remainder();
|
||||
let mut dst_chunks_4 = dst_u8.chunks_exact_mut(4);
|
||||
if let Some(dst_chunk) = dst_chunks_4.next() {
|
||||
let mut sss = initial;
|
||||
let mut y: u32 = 0;
|
||||
|
||||
for src_rows in src_img.iter_2_rows(y_start, max_y) {
|
||||
let components1 = T::components(src_rows[0]);
|
||||
let components2 = T::components(src_rows[1]);
|
||||
// Load two coefficients at once
|
||||
let mmk = wasm32_utils::ptr_i16_to_set1_i32(coeffs, y as usize);
|
||||
|
||||
let source1 = wasm32_utils::i32_v128_from_u8(components1, src_x); // top line
|
||||
let source2 = wasm32_utils::i32_v128_from_u8(components2, src_x); // bottom line
|
||||
|
||||
let source = i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(
|
||||
source1, source2,
|
||||
);
|
||||
let pix = i16x8_extend_low_u8x16(source);
|
||||
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
y += 2;
|
||||
}
|
||||
|
||||
if let Some(&k) = coeffs.get(y as usize) {
|
||||
let s_row = src_img.get_row(y_start + y).unwrap();
|
||||
let components = T::components(s_row);
|
||||
let pix = wasm32_utils::i32x4_extend_low_ptr_u8(components, src_x);
|
||||
let mmk = i32x4_splat(k as i32);
|
||||
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss = i32x4_shr(sss, $imm8);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
|
||||
sss = i16x8_narrow_i32x4(sss, sss);
|
||||
let dst_ptr = dst_chunk.as_mut_ptr() as *mut i32;
|
||||
*dst_ptr = i32x4_extract_lane::<0>(u8x16_narrow_i16x8(sss, sss));
|
||||
|
||||
src_x += 4;
|
||||
}
|
||||
|
||||
dst_u8 = dst_chunks_4.into_remainder();
|
||||
if !dst_u8.is_empty() {
|
||||
native::convolution_by_u8(
|
||||
src_img,
|
||||
normalizer,
|
||||
1 << (precision - 1),
|
||||
dst_u8,
|
||||
src_x,
|
||||
y_start,
|
||||
coeffs,
|
||||
);
|
||||
}
|
||||
}
|
||||
+35
-12
@@ -1,4 +1,6 @@
|
||||
use crate::pixels::{U8x3, U8x4};
|
||||
use std::arch::wasm32::*;
|
||||
use std::intrinsics::transmute;
|
||||
|
||||
#[inline(always)]
|
||||
pub unsafe fn load_v128<T>(buf: &[T], index: usize) -> v128 {
|
||||
@@ -7,26 +9,23 @@ pub unsafe fn load_v128<T>(buf: &[T], index: usize) -> v128 {
|
||||
|
||||
#[inline(always)]
|
||||
pub unsafe fn loadl_i64<T>(buf: &[T], index: usize) -> v128 {
|
||||
i64x2(buf.get_unchecked(index..).as_ptr() as i64, 0)
|
||||
let v = v128_load(buf.get_unchecked(index..).as_ptr() as *const v128);
|
||||
let k = i8x16(0, 1, 2, 3, 4, 5, 6, 7, -1, -1, -1, -1, -1, -1, -1, -1);
|
||||
i8x16_swizzle(v, k)
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub unsafe fn loadl_i32<T>(buf: &[T], index: usize) -> v128 {
|
||||
i32x4(buf.get_unchecked(index..).as_ptr() as i32, 0, 0, 0)
|
||||
let v = v128_load(buf.get_unchecked(index..).as_ptr() as *const v128);
|
||||
let k = i8x16(0, 1, 2, 3, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1);
|
||||
i8x16_swizzle(v, k)
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub unsafe fn loadl_i16<T>(buf: &[T], index: usize) -> v128 {
|
||||
i16x8(
|
||||
buf.get_unchecked(index..).as_ptr() as i16,
|
||||
0,
|
||||
0,
|
||||
0,
|
||||
0,
|
||||
0,
|
||||
0,
|
||||
0,
|
||||
)
|
||||
let v = v128_load(buf.get_unchecked(index..).as_ptr() as *const v128);
|
||||
let k = i8x16(0, 1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1);
|
||||
i8x16_swizzle(v, k)
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
@@ -38,3 +37,27 @@ pub unsafe fn ptr_i16_to_set1_i64(buf: &[i16], index: usize) -> v128 {
|
||||
pub unsafe fn ptr_i16_to_set1_i32(buf: &[i16], index: usize) -> v128 {
|
||||
i32x4_splat(*(buf.get_unchecked(index..).as_ptr() as *const i32))
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub unsafe fn i32x4_extend_low_ptr_u8(buf: &[u8], index: usize) -> v128 {
|
||||
let ptr = buf.get_unchecked(index..).as_ptr() as *const v128;
|
||||
u32x4_extend_low_u16x8(i16x8_extend_low_u8x16(v128_load(ptr)))
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub unsafe fn i32x4_extend_low_ptr_u8x4(buf: &[U8x4], index: usize) -> v128 {
|
||||
let v: i32 = transmute(buf.get_unchecked(index).0);
|
||||
u32x4_extend_low_u16x8(i16x8_extend_low_u8x16(i32x4(v, 0, 0, 0)))
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub unsafe fn i32x4_extend_low_ptr_u8x3(buf: &[U8x3], index: usize) -> v128 {
|
||||
let pixel = buf.get_unchecked(index).0;
|
||||
i32x4(pixel[0] as i32, pixel[1] as i32, pixel[2] as i32, 0)
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub unsafe fn i32_v128_from_u8(buf: &[u8], index: usize) -> v128 {
|
||||
let ptr = buf.get_unchecked(index..).as_ptr() as *const i32;
|
||||
i32x4(*ptr, 0, 0, 0)
|
||||
}
|
||||
|
||||
@@ -376,6 +376,10 @@ fn downscale_u8() {
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Neon);
|
||||
}
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Wasm32);
|
||||
}
|
||||
for cpu_extensions in cpu_extensions_vec {
|
||||
P::downscale_test(
|
||||
ResizeAlg::Convolution(FilterType::Lanczos3),
|
||||
@@ -400,6 +404,10 @@ fn upscale_u8() {
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Neon);
|
||||
}
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Wasm32);
|
||||
}
|
||||
for cpu_extensions in cpu_extensions_vec {
|
||||
P::upscale_test(
|
||||
ResizeAlg::Convolution(FilterType::Lanczos3),
|
||||
@@ -424,6 +432,10 @@ fn downscale_u8x2() {
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Neon);
|
||||
}
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Wasm32);
|
||||
}
|
||||
for cpu_extensions in cpu_extensions_vec {
|
||||
P::downscale_test(
|
||||
ResizeAlg::Convolution(FilterType::Lanczos3),
|
||||
@@ -452,6 +464,10 @@ fn upscale_u8x2() {
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Neon);
|
||||
}
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Wasm32);
|
||||
}
|
||||
for cpu_extensions in cpu_extensions_vec {
|
||||
P::upscale_test(
|
||||
ResizeAlg::Convolution(FilterType::Lanczos3),
|
||||
@@ -480,6 +496,10 @@ fn downscale_u8x3() {
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Neon);
|
||||
}
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Wasm32);
|
||||
}
|
||||
for cpu_extensions in cpu_extensions_vec {
|
||||
P::downscale_test(
|
||||
ResizeAlg::Convolution(FilterType::Lanczos3),
|
||||
@@ -508,6 +528,10 @@ fn upscale_u8x3() {
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Neon);
|
||||
}
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Wasm32);
|
||||
}
|
||||
for cpu_extensions in cpu_extensions_vec {
|
||||
P::upscale_test(
|
||||
ResizeAlg::Convolution(FilterType::Lanczos3),
|
||||
@@ -536,6 +560,10 @@ fn downscale_u8x4() {
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Neon);
|
||||
}
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Wasm32);
|
||||
}
|
||||
for cpu_extensions in cpu_extensions_vec {
|
||||
P::downscale_test(
|
||||
ResizeAlg::Convolution(FilterType::Lanczos3),
|
||||
@@ -570,6 +598,10 @@ fn upscale_u8x4() {
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Neon);
|
||||
}
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Wasm32);
|
||||
}
|
||||
for cpu_extensions in cpu_extensions_vec {
|
||||
P::upscale_test(
|
||||
ResizeAlg::Convolution(FilterType::Lanczos3),
|
||||
|
||||
Reference in New Issue
Block a user