Fixed implementation of vertical convolution for Wasm32

This commit is contained in:
Kirill Kuzminykh
2024-05-04 21:10:15 +03:00
parent 2ae2702634
commit 833b25c282
13 changed files with 56 additions and 75 deletions
Generated
+4 -4
View File
@@ -124,9 +124,9 @@ checksum = "96d30a06541fbafbc7f82ed10c06164cfbd2c401138f6addd8404629c4b16711"
[[package]]
name = "autocfg"
version = "1.2.0"
version = "1.3.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f1fdabc7756949593fe60f30ec81974b613357de856987752631dea1e3394c80"
checksum = "0c4b4d0bd25bd0b74681c0ad21497610ce1b7c91b1022cd21c80c6fbdd9476b0"
[[package]]
name = "av1-grain"
@@ -1074,9 +1074,9 @@ dependencies = [
[[package]]
name = "num-traits"
version = "0.2.18"
version = "0.2.19"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "da0df0e5185db44f69b44f26786fe401b6c293d1907744beaa7fa62b2e5a517a"
checksum = "071dfc062690e90b734c0b2273ce72ad0ffa95f0c74596bc250dcfd960262841"
dependencies = [
"autocfg",
]
+1 -1
View File
@@ -21,7 +21,7 @@ exclude = ["/data"]
[dependencies]
cfg-if = "1.0"
num-traits = "0.2.18"
num-traits = "0.2.19"
thiserror = "1.0"
bytemuck = "1.15"
document-features = "0.2.8"
+1 -1
View File
@@ -1,7 +1,7 @@
use std::arch::wasm32::*;
use crate::pixels::U8x2;
use crate::wasm32_utils::{u16x8_mul_add_shr16, u16x8_mul_shr16};
use crate::wasm32_utils::u16x8_mul_add_shr16;
use crate::{ImageView, ImageViewMut};
use super::native;
+2 -5
View File
@@ -68,12 +68,9 @@ where
}
#[inline]
fn next_chunk<I: Iterator, const N: usize>(
fn next_chunk<I: Iterator + Sized, const N: usize>(
iter: &mut I,
) -> Result<[I::Item; N], Take<IntoIter<I::Item, N>>>
where
I: Sized,
{
) -> Result<[I::Item; N], Take<IntoIter<I::Item, N>>> {
iter_next_chunk(iter)
}
+4 -4
View File
@@ -151,14 +151,14 @@ unsafe fn horiz_convolution_four_rows(
if let Some(&k) = coeffs.first() {
let coeff01_i64x2 = i64x2(k as i64, 0);
for i in 0..4 {
let pixel = (*src_rows[i].get_unchecked(x)).0 as i64;
let pixel = src_rows[i].get_unchecked(x).0 as i64;
let source = i64x2(pixel, 0);
ll_sum[i] = i64x2_add(ll_sum[i], wasm32_utils::i64x2_mul_lo(source, coeff01_i64x2));
}
}
for i in 0..4 {
v128_store((&mut ll_buf).as_mut_ptr() as *mut v128, ll_sum[i]);
v128_store(ll_buf.as_mut_ptr() as *mut v128, ll_sum[i]);
let dst_pixel = dst_rows[i].get_unchecked_mut(dst_x);
dst_pixel.0 = normalizer.clip(ll_buf.iter().sum::<i64>() + half_error);
}
@@ -268,12 +268,12 @@ unsafe fn horiz_convolution_one_row(
if let Some(&k) = coeffs.first() {
let coeff01_i64x2 = i64x2(k as i64, 0);
let pixel = (*src_row.get_unchecked(x)).0 as i64;
let pixel = src_row.get_unchecked(x).0 as i64;
let source = i64x2(pixel, 0);
ll_sum = i64x2_add(ll_sum, wasm32_utils::i64x2_mul_lo(source, coeff01_i64x2));
}
v128_store((&mut ll_buf).as_mut_ptr() as *mut v128, ll_sum);
v128_store(ll_buf.as_mut_ptr() as *mut v128, ll_sum);
let dst_pixel = dst_row.get_unchecked_mut(dst_x);
dst_pixel.0 = normalizer.clip(ll_buf[0] + ll_buf[1] + half_error);
}
+2 -2
View File
@@ -143,7 +143,7 @@ unsafe fn horiz_convolution_four_rows(
}
for i in 0..4 {
v128_store((&mut ll_buf).as_mut_ptr() as *mut v128, ll_sum[i]);
v128_store(ll_buf.as_mut_ptr() as *mut v128, ll_sum[i]);
let dst_pixel = dst_rows[i].get_unchecked_mut(dst_x);
dst_pixel.0 = [normalizer.clip(ll_buf[0]), normalizer.clip(ll_buf[1])];
}
@@ -248,7 +248,7 @@ unsafe fn horiz_convolution_one_row(
ll_sum = i64x2_add(ll_sum, wasm32_utils::i64x2_mul_lo(p_i64x2, coeff0_i64x2));
}
v128_store((&mut ll_buf).as_mut_ptr() as *mut v128, ll_sum);
v128_store(ll_buf.as_mut_ptr() as *mut v128, ll_sum);
let dst_pixel = dst_row.get_unchecked_mut(dst_x);
dst_pixel.0 = [normalizer.clip(ll_buf[0]), normalizer.clip(ll_buf[1])];
}
+4 -4
View File
@@ -128,8 +128,8 @@ unsafe fn horiz_convolution_four_rows(
}
for i in 0..4 {
v128_store((&mut rg_buf).as_mut_ptr() as *mut v128, rg_sum[i]);
v128_store((&mut bb_buf).as_mut_ptr() as *mut v128, bb_sum[i]);
v128_store(rg_buf.as_mut_ptr() as *mut v128, rg_sum[i]);
v128_store(bb_buf.as_mut_ptr() as *mut v128, bb_sum[i]);
let dst_pixel = dst_rows[i].get_unchecked_mut(dst_x);
dst_pixel.0[0] = normalizer.clip(rg_buf[0] + half_error);
dst_pixel.0[1] = normalizer.clip(rg_buf[1] + half_error);
@@ -221,8 +221,8 @@ unsafe fn horiz_convolution_one_row(
x += 1;
}
v128_store((&mut rg_buf).as_mut_ptr() as *mut v128, rg_sum);
v128_store((&mut bb_buf).as_mut_ptr() as *mut v128, bb_sum);
v128_store(rg_buf.as_mut_ptr() as *mut v128, rg_sum);
v128_store(bb_buf.as_mut_ptr() as *mut v128, bb_sum);
let dst_pixel = dst_row.get_unchecked_mut(dst_x);
dst_pixel.0[0] = normalizer.clip(rg_buf[0]);
dst_pixel.0[1] = normalizer.clip(rg_buf[1]);
+4 -4
View File
@@ -127,8 +127,8 @@ unsafe fn horiz_convolution_four_rows(
}
for i in 0..4 {
v128_store((&mut rg_buf).as_mut_ptr() as *mut v128, rg_sum[i]);
v128_store((&mut ba_buf).as_mut_ptr() as *mut v128, ba_sum[i]);
v128_store(rg_buf.as_mut_ptr() as *mut v128, rg_sum[i]);
v128_store(ba_buf.as_mut_ptr() as *mut v128, ba_sum[i]);
let dst_pixel = dst_rows[i].get_unchecked_mut(dst_x);
dst_pixel.0 = [
normalizer.clip(rg_buf[0]),
@@ -219,8 +219,8 @@ unsafe fn horiz_convolution_one_row(
ba_sum = i64x2_add(ba_sum, wasm32_utils::i64x2_mul_lo(ba_i64x2, coeff0_i64x2));
}
v128_store((&mut rg_buf).as_mut_ptr() as *mut v128, rg_sum);
v128_store((&mut ba_buf).as_mut_ptr() as *mut v128, ba_sum);
v128_store(rg_buf.as_mut_ptr() as *mut v128, rg_sum);
v128_store(ba_buf.as_mut_ptr() as *mut v128, ba_sum);
let dst_pixel = dst_row.get_unchecked_mut(dst_x);
dst_pixel.0 = [
normalizer.clip(rg_buf[0]),
+8 -8
View File
@@ -35,7 +35,7 @@ unsafe fn vert_convolution_into_one_row_u16<T: InnerPixel<Component = u16>>(
) {
let y_start = coeffs_chunk.start;
let coeffs = coeffs_chunk.values;
let max_y = y_start + coeffs.len() as u32;
let max_rows = coeffs.len() as u32;
let mut dst_u16 = T::components_mut(dst_row);
/*
@@ -77,7 +77,7 @@ unsafe fn vert_convolution_into_one_row_u16<T: InnerPixel<Component = u16>>(
let coeffs_2 = coeffs.chunks_exact(2);
let coeffs_reminder = coeffs_2.remainder();
for (src_rows, two_coeffs) in src_view.iter_2_rows(y_start, max_y).zip(coeffs_2) {
for (src_rows, two_coeffs) in src_view.iter_2_rows(y_start, max_rows).zip(coeffs_2) {
let src_rows = src_rows.map(|row| T::components(row));
for r in 0..2 {
@@ -113,7 +113,7 @@ unsafe fn vert_convolution_into_one_row_u16<T: InnerPixel<Component = u16>>(
let mut dst_ptr = dst_chunk.as_mut_ptr();
for x in 0..2 {
for sum in sums {
v128_store((&mut c_buf).as_mut_ptr() as *mut v128, sum[x]);
v128_store(c_buf.as_mut_ptr() as *mut v128, sum[x]);
*dst_ptr = normalizer.clip(c_buf[0]);
dst_ptr = dst_ptr.add(1);
*dst_ptr = normalizer.clip(c_buf[1]);
@@ -133,7 +133,7 @@ unsafe fn vert_convolution_into_one_row_u16<T: InnerPixel<Component = u16>>(
let coeffs_2 = coeffs.chunks_exact(2);
let coeffs_reminder = coeffs_2.remainder();
for (src_rows, two_coeffs) in src_view.iter_2_rows(y_start, max_y).zip(coeffs_2) {
for (src_rows, two_coeffs) in src_view.iter_2_rows(y_start, max_rows).zip(coeffs_2) {
let src_rows = src_rows.map(|row| T::components(row));
let coeffs_i64 = [
i64x2_splat(two_coeffs[0] as i64),
@@ -169,7 +169,7 @@ unsafe fn vert_convolution_into_one_row_u16<T: InnerPixel<Component = u16>>(
// sums[i] = _mm_and_si128(sums[i] , mask);
// sums[i] = _mm_srl_epi64(sums[i] , precision_i64);
// _mm_packus_epi32(sums[i] , sums[i] );
v128_store((&mut c_buf).as_mut_ptr() as *mut v128, sum);
v128_store(c_buf.as_mut_ptr() as *mut v128, sum);
*dst_ptr = normalizer.clip(c_buf[0]);
dst_ptr = dst_ptr.add(1);
*dst_ptr = normalizer.clip(c_buf[1]);
@@ -188,7 +188,7 @@ unsafe fn vert_convolution_into_one_row_u16<T: InnerPixel<Component = u16>>(
let coeffs_2 = coeffs.chunks_exact(2);
let coeffs_reminder = coeffs_2.remainder();
for (src_rows, two_coeffs) in src_view.iter_2_rows(y_start, max_y).zip(coeffs_2) {
for (src_rows, two_coeffs) in src_view.iter_2_rows(y_start, max_rows).zip(coeffs_2) {
let src_rows = src_rows.map(|row| T::components(row));
let coeffs_i64 = [
i64x2_splat(two_coeffs[0] as i64),
@@ -218,12 +218,12 @@ unsafe fn vert_convolution_into_one_row_u16<T: InnerPixel<Component = u16>>(
}
let mut dst_ptr = dst_chunk.as_mut_ptr();
v128_store((&mut c_buf).as_mut_ptr() as *mut v128, c01);
v128_store(c_buf.as_mut_ptr() as *mut v128, c01);
*dst_ptr = normalizer.clip(c_buf[0]);
dst_ptr = dst_ptr.add(1);
*dst_ptr = normalizer.clip(c_buf[1]);
dst_ptr = dst_ptr.add(1);
v128_store((&mut c_buf).as_mut_ptr() as *mut v128, c23);
v128_store(c_buf.as_mut_ptr() as *mut v128, c23);
*dst_ptr = normalizer.clip(c_buf[0]);
dst_ptr = dst_ptr.add(1);
*dst_ptr = normalizer.clip(c_buf[1]);
+16 -32
View File
@@ -37,8 +37,8 @@ unsafe fn vert_convolution_into_one_row_u8<T: InnerPixel<Component = u8>>(
const ZERO: v128 = i64x2(0, 0);
let y_start = coeffs_chunk.start;
let coeffs = coeffs_chunk.values;
let max_y = y_start + coeffs.len() as u32;
let precision = normalizer.precision();
let max_rows = coeffs.len() as u32;
let precision = normalizer.precision() as u32;
let mut dst_u8 = T::components_mut(dst_row);
let initial = i32x4_splat(1 << (precision - 1));
@@ -56,7 +56,7 @@ unsafe fn vert_convolution_into_one_row_u8<T: InnerPixel<Component = u8>>(
let mut y: u32 = 0;
for src_rows in src_view.iter_2_rows(y_start, max_y) {
for src_rows in src_view.iter_2_rows(y_start, max_rows) {
let components1 = T::components(src_rows[0]);
let components2 = T::components(src_rows[1]);
@@ -145,20 +145,14 @@ unsafe fn vert_convolution_into_one_row_u8<T: InnerPixel<Component = u8>>(
}
}
// This version of code works faster.
macro_rules! call {
($imm8:expr) => {{
sss0 = i32x4_shr(sss0, $imm8);
sss1 = i32x4_shr(sss1, $imm8);
sss2 = i32x4_shr(sss2, $imm8);
sss3 = i32x4_shr(sss3, $imm8);
sss4 = i32x4_shr(sss4, $imm8);
sss5 = i32x4_shr(sss5, $imm8);
sss6 = i32x4_shr(sss6, $imm8);
sss7 = i32x4_shr(sss7, $imm8);
}};
}
constify_imm8!(precision, call);
sss0 = i32x4_shr(sss0, precision);
sss1 = i32x4_shr(sss1, precision);
sss2 = i32x4_shr(sss2, precision);
sss3 = i32x4_shr(sss3, precision);
sss4 = i32x4_shr(sss4, precision);
sss5 = i32x4_shr(sss5, precision);
sss6 = i32x4_shr(sss6, precision);
sss7 = i32x4_shr(sss7, precision);
sss0 = i16x8_narrow_i32x4(sss0, sss1);
sss2 = i16x8_narrow_i32x4(sss2, sss3);
@@ -181,7 +175,7 @@ unsafe fn vert_convolution_into_one_row_u8<T: InnerPixel<Component = u8>>(
let mut sss1 = initial; // right row
let mut y: u32 = 0;
for src_rows in src_view.iter_2_rows(y_start, max_y) {
for src_rows in src_view.iter_2_rows(y_start, max_rows) {
let components1 = T::components(src_rows[0]);
let components2 = T::components(src_rows[1]);
// Load two coefficients at once
@@ -218,13 +212,8 @@ unsafe fn vert_convolution_into_one_row_u8<T: InnerPixel<Component = u8>>(
}
}
macro_rules! call {
($imm8:expr) => {{
sss0 = i32x4_shr(sss0, $imm8);
sss1 = i32x4_shr(sss1, $imm8);
}};
}
constify_imm8!(precision, call);
sss0 = i32x4_shr(sss0, precision);
sss1 = i32x4_shr(sss1, precision);
sss0 = i16x8_narrow_i32x4(sss0, sss1);
sss0 = u8x16_narrow_i16x8(sss0, sss0);
@@ -240,7 +229,7 @@ unsafe fn vert_convolution_into_one_row_u8<T: InnerPixel<Component = u8>>(
let mut sss = initial;
let mut y: u32 = 0;
for src_rows in src_view.iter_2_rows(y_start, max_y) {
for src_rows in src_view.iter_2_rows(y_start, max_rows) {
let components1 = T::components(src_rows[0]);
let components2 = T::components(src_rows[1]);
// Load two coefficients at once
@@ -267,12 +256,7 @@ unsafe fn vert_convolution_into_one_row_u8<T: InnerPixel<Component = u8>>(
}
}
macro_rules! call {
($imm8:expr) => {{
sss = i32x4_shr(sss, $imm8);
}};
}
constify_imm8!(precision, call);
sss = i32x4_shr(sss, precision);
sss = i16x8_narrow_i32x4(sss, sss);
let dst_ptr = dst_chunk.as_mut_ptr() as *mut i32;
+1 -1
View File
@@ -44,7 +44,7 @@ pub mod testing;
#[cfg(target_arch = "wasm32")]
mod wasm32_utils;
/// This trait must be used in your code instead of [InnerPixel](crate::pixels::InnerPixel).
/// This trait must be used in your code instead of [InnerPixel](pixels::InnerPixel).
#[allow(private_bounds)]
pub trait PixelTrait: Convolution + AlphaMulDiv {}
+7 -7
View File
@@ -71,13 +71,13 @@ pub(crate) unsafe fn i32x4_v128_from_u8(buf: &[u8], index: usize) -> v128 {
i32x4(p.read_unaligned(), 0, 0, 0)
}
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn u16x8_mul_shr16(a_u16x8: v128, b_u16x8: v128) -> v128 {
let lo_u32x4 = u32x4_extmul_low_u16x8(a_u16x8, b_u16x8);
let hi_u32x4 = u32x4_extmul_high_u16x8(a_u16x8, b_u16x8);
i16x8_shuffle::<1, 3, 5, 7, 9, 11, 13, 15>(lo_u32x4, hi_u32x4)
}
// #[inline]
// #[target_feature(enable = "simd128")]
// pub(crate) unsafe fn u16x8_mul_shr16(a_u16x8: v128, b_u16x8: v128) -> v128 {
// let lo_u32x4 = u32x4_extmul_low_u16x8(a_u16x8, b_u16x8);
// let hi_u32x4 = u32x4_extmul_high_u16x8(a_u16x8, b_u16x8);
// i16x8_shuffle::<1, 3, 5, 7, 9, 11, 13, 15>(lo_u32x4, hi_u32x4)
// }
pub(crate) unsafe fn u16x8_mul_add_shr16(a_u16x8: v128, b_u16x8: v128, c: v128) -> v128 {
let lo_u32x4 = u32x4_extmul_low_u16x8(a_u16x8, b_u16x8);
+2 -2
View File
@@ -5,7 +5,7 @@ use image::{ColorType, DynamicImage};
use fast_image_resize::images::Image;
use fast_image_resize::pixels::*;
use fast_image_resize::{CpuExtensions, PixelType};
use fast_image_resize::{CpuExtensions, PixelTrait, PixelType};
pub fn nonzero(v: u32) -> NonZeroU32 {
NonZeroU32::new(v).unwrap()
@@ -32,7 +32,7 @@ pub fn image_checksum<P: InnerPixel, const N: usize>(image: &Image) -> [u64; N]
res
}
pub trait PixelTestingExt: InnerPixel {
pub trait PixelTestingExt: PixelTrait {
fn pixel_type_str() -> &'static str {
match Self::pixel_type() {
PixelType::U8 => "u8",