mirror of
https://github.com/Cykooz/fast_image_resize.git
synced 2026-10-08 01:11:09 +00:00
Fixed implementation of vertical convolution for Wasm32
This commit is contained in:
Generated
+4
-4
@@ -124,9 +124,9 @@ checksum = "96d30a06541fbafbc7f82ed10c06164cfbd2c401138f6addd8404629c4b16711"
|
||||
|
||||
[[package]]
|
||||
name = "autocfg"
|
||||
version = "1.2.0"
|
||||
version = "1.3.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f1fdabc7756949593fe60f30ec81974b613357de856987752631dea1e3394c80"
|
||||
checksum = "0c4b4d0bd25bd0b74681c0ad21497610ce1b7c91b1022cd21c80c6fbdd9476b0"
|
||||
|
||||
[[package]]
|
||||
name = "av1-grain"
|
||||
@@ -1074,9 +1074,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "num-traits"
|
||||
version = "0.2.18"
|
||||
version = "0.2.19"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "da0df0e5185db44f69b44f26786fe401b6c293d1907744beaa7fa62b2e5a517a"
|
||||
checksum = "071dfc062690e90b734c0b2273ce72ad0ffa95f0c74596bc250dcfd960262841"
|
||||
dependencies = [
|
||||
"autocfg",
|
||||
]
|
||||
|
||||
+1
-1
@@ -21,7 +21,7 @@ exclude = ["/data"]
|
||||
|
||||
[dependencies]
|
||||
cfg-if = "1.0"
|
||||
num-traits = "0.2.18"
|
||||
num-traits = "0.2.19"
|
||||
thiserror = "1.0"
|
||||
bytemuck = "1.15"
|
||||
document-features = "0.2.8"
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
use std::arch::wasm32::*;
|
||||
|
||||
use crate::pixels::U8x2;
|
||||
use crate::wasm32_utils::{u16x8_mul_add_shr16, u16x8_mul_shr16};
|
||||
use crate::wasm32_utils::u16x8_mul_add_shr16;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
use super::native;
|
||||
|
||||
+2
-5
@@ -68,12 +68,9 @@ where
|
||||
}
|
||||
|
||||
#[inline]
|
||||
fn next_chunk<I: Iterator, const N: usize>(
|
||||
fn next_chunk<I: Iterator + Sized, const N: usize>(
|
||||
iter: &mut I,
|
||||
) -> Result<[I::Item; N], Take<IntoIter<I::Item, N>>>
|
||||
where
|
||||
I: Sized,
|
||||
{
|
||||
) -> Result<[I::Item; N], Take<IntoIter<I::Item, N>>> {
|
||||
iter_next_chunk(iter)
|
||||
}
|
||||
|
||||
|
||||
@@ -151,14 +151,14 @@ unsafe fn horiz_convolution_four_rows(
|
||||
if let Some(&k) = coeffs.first() {
|
||||
let coeff01_i64x2 = i64x2(k as i64, 0);
|
||||
for i in 0..4 {
|
||||
let pixel = (*src_rows[i].get_unchecked(x)).0 as i64;
|
||||
let pixel = src_rows[i].get_unchecked(x).0 as i64;
|
||||
let source = i64x2(pixel, 0);
|
||||
ll_sum[i] = i64x2_add(ll_sum[i], wasm32_utils::i64x2_mul_lo(source, coeff01_i64x2));
|
||||
}
|
||||
}
|
||||
|
||||
for i in 0..4 {
|
||||
v128_store((&mut ll_buf).as_mut_ptr() as *mut v128, ll_sum[i]);
|
||||
v128_store(ll_buf.as_mut_ptr() as *mut v128, ll_sum[i]);
|
||||
let dst_pixel = dst_rows[i].get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = normalizer.clip(ll_buf.iter().sum::<i64>() + half_error);
|
||||
}
|
||||
@@ -268,12 +268,12 @@ unsafe fn horiz_convolution_one_row(
|
||||
|
||||
if let Some(&k) = coeffs.first() {
|
||||
let coeff01_i64x2 = i64x2(k as i64, 0);
|
||||
let pixel = (*src_row.get_unchecked(x)).0 as i64;
|
||||
let pixel = src_row.get_unchecked(x).0 as i64;
|
||||
let source = i64x2(pixel, 0);
|
||||
ll_sum = i64x2_add(ll_sum, wasm32_utils::i64x2_mul_lo(source, coeff01_i64x2));
|
||||
}
|
||||
|
||||
v128_store((&mut ll_buf).as_mut_ptr() as *mut v128, ll_sum);
|
||||
v128_store(ll_buf.as_mut_ptr() as *mut v128, ll_sum);
|
||||
let dst_pixel = dst_row.get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = normalizer.clip(ll_buf[0] + ll_buf[1] + half_error);
|
||||
}
|
||||
|
||||
@@ -143,7 +143,7 @@ unsafe fn horiz_convolution_four_rows(
|
||||
}
|
||||
|
||||
for i in 0..4 {
|
||||
v128_store((&mut ll_buf).as_mut_ptr() as *mut v128, ll_sum[i]);
|
||||
v128_store(ll_buf.as_mut_ptr() as *mut v128, ll_sum[i]);
|
||||
let dst_pixel = dst_rows[i].get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = [normalizer.clip(ll_buf[0]), normalizer.clip(ll_buf[1])];
|
||||
}
|
||||
@@ -248,7 +248,7 @@ unsafe fn horiz_convolution_one_row(
|
||||
ll_sum = i64x2_add(ll_sum, wasm32_utils::i64x2_mul_lo(p_i64x2, coeff0_i64x2));
|
||||
}
|
||||
|
||||
v128_store((&mut ll_buf).as_mut_ptr() as *mut v128, ll_sum);
|
||||
v128_store(ll_buf.as_mut_ptr() as *mut v128, ll_sum);
|
||||
let dst_pixel = dst_row.get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = [normalizer.clip(ll_buf[0]), normalizer.clip(ll_buf[1])];
|
||||
}
|
||||
|
||||
@@ -128,8 +128,8 @@ unsafe fn horiz_convolution_four_rows(
|
||||
}
|
||||
|
||||
for i in 0..4 {
|
||||
v128_store((&mut rg_buf).as_mut_ptr() as *mut v128, rg_sum[i]);
|
||||
v128_store((&mut bb_buf).as_mut_ptr() as *mut v128, bb_sum[i]);
|
||||
v128_store(rg_buf.as_mut_ptr() as *mut v128, rg_sum[i]);
|
||||
v128_store(bb_buf.as_mut_ptr() as *mut v128, bb_sum[i]);
|
||||
let dst_pixel = dst_rows[i].get_unchecked_mut(dst_x);
|
||||
dst_pixel.0[0] = normalizer.clip(rg_buf[0] + half_error);
|
||||
dst_pixel.0[1] = normalizer.clip(rg_buf[1] + half_error);
|
||||
@@ -221,8 +221,8 @@ unsafe fn horiz_convolution_one_row(
|
||||
x += 1;
|
||||
}
|
||||
|
||||
v128_store((&mut rg_buf).as_mut_ptr() as *mut v128, rg_sum);
|
||||
v128_store((&mut bb_buf).as_mut_ptr() as *mut v128, bb_sum);
|
||||
v128_store(rg_buf.as_mut_ptr() as *mut v128, rg_sum);
|
||||
v128_store(bb_buf.as_mut_ptr() as *mut v128, bb_sum);
|
||||
let dst_pixel = dst_row.get_unchecked_mut(dst_x);
|
||||
dst_pixel.0[0] = normalizer.clip(rg_buf[0]);
|
||||
dst_pixel.0[1] = normalizer.clip(rg_buf[1]);
|
||||
|
||||
@@ -127,8 +127,8 @@ unsafe fn horiz_convolution_four_rows(
|
||||
}
|
||||
|
||||
for i in 0..4 {
|
||||
v128_store((&mut rg_buf).as_mut_ptr() as *mut v128, rg_sum[i]);
|
||||
v128_store((&mut ba_buf).as_mut_ptr() as *mut v128, ba_sum[i]);
|
||||
v128_store(rg_buf.as_mut_ptr() as *mut v128, rg_sum[i]);
|
||||
v128_store(ba_buf.as_mut_ptr() as *mut v128, ba_sum[i]);
|
||||
let dst_pixel = dst_rows[i].get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = [
|
||||
normalizer.clip(rg_buf[0]),
|
||||
@@ -219,8 +219,8 @@ unsafe fn horiz_convolution_one_row(
|
||||
ba_sum = i64x2_add(ba_sum, wasm32_utils::i64x2_mul_lo(ba_i64x2, coeff0_i64x2));
|
||||
}
|
||||
|
||||
v128_store((&mut rg_buf).as_mut_ptr() as *mut v128, rg_sum);
|
||||
v128_store((&mut ba_buf).as_mut_ptr() as *mut v128, ba_sum);
|
||||
v128_store(rg_buf.as_mut_ptr() as *mut v128, rg_sum);
|
||||
v128_store(ba_buf.as_mut_ptr() as *mut v128, ba_sum);
|
||||
let dst_pixel = dst_row.get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = [
|
||||
normalizer.clip(rg_buf[0]),
|
||||
|
||||
@@ -35,7 +35,7 @@ unsafe fn vert_convolution_into_one_row_u16<T: InnerPixel<Component = u16>>(
|
||||
) {
|
||||
let y_start = coeffs_chunk.start;
|
||||
let coeffs = coeffs_chunk.values;
|
||||
let max_y = y_start + coeffs.len() as u32;
|
||||
let max_rows = coeffs.len() as u32;
|
||||
let mut dst_u16 = T::components_mut(dst_row);
|
||||
|
||||
/*
|
||||
@@ -77,7 +77,7 @@ unsafe fn vert_convolution_into_one_row_u16<T: InnerPixel<Component = u16>>(
|
||||
let coeffs_2 = coeffs.chunks_exact(2);
|
||||
let coeffs_reminder = coeffs_2.remainder();
|
||||
|
||||
for (src_rows, two_coeffs) in src_view.iter_2_rows(y_start, max_y).zip(coeffs_2) {
|
||||
for (src_rows, two_coeffs) in src_view.iter_2_rows(y_start, max_rows).zip(coeffs_2) {
|
||||
let src_rows = src_rows.map(|row| T::components(row));
|
||||
|
||||
for r in 0..2 {
|
||||
@@ -113,7 +113,7 @@ unsafe fn vert_convolution_into_one_row_u16<T: InnerPixel<Component = u16>>(
|
||||
let mut dst_ptr = dst_chunk.as_mut_ptr();
|
||||
for x in 0..2 {
|
||||
for sum in sums {
|
||||
v128_store((&mut c_buf).as_mut_ptr() as *mut v128, sum[x]);
|
||||
v128_store(c_buf.as_mut_ptr() as *mut v128, sum[x]);
|
||||
*dst_ptr = normalizer.clip(c_buf[0]);
|
||||
dst_ptr = dst_ptr.add(1);
|
||||
*dst_ptr = normalizer.clip(c_buf[1]);
|
||||
@@ -133,7 +133,7 @@ unsafe fn vert_convolution_into_one_row_u16<T: InnerPixel<Component = u16>>(
|
||||
let coeffs_2 = coeffs.chunks_exact(2);
|
||||
let coeffs_reminder = coeffs_2.remainder();
|
||||
|
||||
for (src_rows, two_coeffs) in src_view.iter_2_rows(y_start, max_y).zip(coeffs_2) {
|
||||
for (src_rows, two_coeffs) in src_view.iter_2_rows(y_start, max_rows).zip(coeffs_2) {
|
||||
let src_rows = src_rows.map(|row| T::components(row));
|
||||
let coeffs_i64 = [
|
||||
i64x2_splat(two_coeffs[0] as i64),
|
||||
@@ -169,7 +169,7 @@ unsafe fn vert_convolution_into_one_row_u16<T: InnerPixel<Component = u16>>(
|
||||
// sums[i] = _mm_and_si128(sums[i] , mask);
|
||||
// sums[i] = _mm_srl_epi64(sums[i] , precision_i64);
|
||||
// _mm_packus_epi32(sums[i] , sums[i] );
|
||||
v128_store((&mut c_buf).as_mut_ptr() as *mut v128, sum);
|
||||
v128_store(c_buf.as_mut_ptr() as *mut v128, sum);
|
||||
*dst_ptr = normalizer.clip(c_buf[0]);
|
||||
dst_ptr = dst_ptr.add(1);
|
||||
*dst_ptr = normalizer.clip(c_buf[1]);
|
||||
@@ -188,7 +188,7 @@ unsafe fn vert_convolution_into_one_row_u16<T: InnerPixel<Component = u16>>(
|
||||
let coeffs_2 = coeffs.chunks_exact(2);
|
||||
let coeffs_reminder = coeffs_2.remainder();
|
||||
|
||||
for (src_rows, two_coeffs) in src_view.iter_2_rows(y_start, max_y).zip(coeffs_2) {
|
||||
for (src_rows, two_coeffs) in src_view.iter_2_rows(y_start, max_rows).zip(coeffs_2) {
|
||||
let src_rows = src_rows.map(|row| T::components(row));
|
||||
let coeffs_i64 = [
|
||||
i64x2_splat(two_coeffs[0] as i64),
|
||||
@@ -218,12 +218,12 @@ unsafe fn vert_convolution_into_one_row_u16<T: InnerPixel<Component = u16>>(
|
||||
}
|
||||
|
||||
let mut dst_ptr = dst_chunk.as_mut_ptr();
|
||||
v128_store((&mut c_buf).as_mut_ptr() as *mut v128, c01);
|
||||
v128_store(c_buf.as_mut_ptr() as *mut v128, c01);
|
||||
*dst_ptr = normalizer.clip(c_buf[0]);
|
||||
dst_ptr = dst_ptr.add(1);
|
||||
*dst_ptr = normalizer.clip(c_buf[1]);
|
||||
dst_ptr = dst_ptr.add(1);
|
||||
v128_store((&mut c_buf).as_mut_ptr() as *mut v128, c23);
|
||||
v128_store(c_buf.as_mut_ptr() as *mut v128, c23);
|
||||
*dst_ptr = normalizer.clip(c_buf[0]);
|
||||
dst_ptr = dst_ptr.add(1);
|
||||
*dst_ptr = normalizer.clip(c_buf[1]);
|
||||
|
||||
@@ -37,8 +37,8 @@ unsafe fn vert_convolution_into_one_row_u8<T: InnerPixel<Component = u8>>(
|
||||
const ZERO: v128 = i64x2(0, 0);
|
||||
let y_start = coeffs_chunk.start;
|
||||
let coeffs = coeffs_chunk.values;
|
||||
let max_y = y_start + coeffs.len() as u32;
|
||||
let precision = normalizer.precision();
|
||||
let max_rows = coeffs.len() as u32;
|
||||
let precision = normalizer.precision() as u32;
|
||||
let mut dst_u8 = T::components_mut(dst_row);
|
||||
|
||||
let initial = i32x4_splat(1 << (precision - 1));
|
||||
@@ -56,7 +56,7 @@ unsafe fn vert_convolution_into_one_row_u8<T: InnerPixel<Component = u8>>(
|
||||
|
||||
let mut y: u32 = 0;
|
||||
|
||||
for src_rows in src_view.iter_2_rows(y_start, max_y) {
|
||||
for src_rows in src_view.iter_2_rows(y_start, max_rows) {
|
||||
let components1 = T::components(src_rows[0]);
|
||||
let components2 = T::components(src_rows[1]);
|
||||
|
||||
@@ -145,20 +145,14 @@ unsafe fn vert_convolution_into_one_row_u8<T: InnerPixel<Component = u8>>(
|
||||
}
|
||||
}
|
||||
|
||||
// This version of code works faster.
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss0 = i32x4_shr(sss0, $imm8);
|
||||
sss1 = i32x4_shr(sss1, $imm8);
|
||||
sss2 = i32x4_shr(sss2, $imm8);
|
||||
sss3 = i32x4_shr(sss3, $imm8);
|
||||
sss4 = i32x4_shr(sss4, $imm8);
|
||||
sss5 = i32x4_shr(sss5, $imm8);
|
||||
sss6 = i32x4_shr(sss6, $imm8);
|
||||
sss7 = i32x4_shr(sss7, $imm8);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
sss0 = i32x4_shr(sss0, precision);
|
||||
sss1 = i32x4_shr(sss1, precision);
|
||||
sss2 = i32x4_shr(sss2, precision);
|
||||
sss3 = i32x4_shr(sss3, precision);
|
||||
sss4 = i32x4_shr(sss4, precision);
|
||||
sss5 = i32x4_shr(sss5, precision);
|
||||
sss6 = i32x4_shr(sss6, precision);
|
||||
sss7 = i32x4_shr(sss7, precision);
|
||||
|
||||
sss0 = i16x8_narrow_i32x4(sss0, sss1);
|
||||
sss2 = i16x8_narrow_i32x4(sss2, sss3);
|
||||
@@ -181,7 +175,7 @@ unsafe fn vert_convolution_into_one_row_u8<T: InnerPixel<Component = u8>>(
|
||||
let mut sss1 = initial; // right row
|
||||
let mut y: u32 = 0;
|
||||
|
||||
for src_rows in src_view.iter_2_rows(y_start, max_y) {
|
||||
for src_rows in src_view.iter_2_rows(y_start, max_rows) {
|
||||
let components1 = T::components(src_rows[0]);
|
||||
let components2 = T::components(src_rows[1]);
|
||||
// Load two coefficients at once
|
||||
@@ -218,13 +212,8 @@ unsafe fn vert_convolution_into_one_row_u8<T: InnerPixel<Component = u8>>(
|
||||
}
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss0 = i32x4_shr(sss0, $imm8);
|
||||
sss1 = i32x4_shr(sss1, $imm8);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
sss0 = i32x4_shr(sss0, precision);
|
||||
sss1 = i32x4_shr(sss1, precision);
|
||||
|
||||
sss0 = i16x8_narrow_i32x4(sss0, sss1);
|
||||
sss0 = u8x16_narrow_i16x8(sss0, sss0);
|
||||
@@ -240,7 +229,7 @@ unsafe fn vert_convolution_into_one_row_u8<T: InnerPixel<Component = u8>>(
|
||||
let mut sss = initial;
|
||||
let mut y: u32 = 0;
|
||||
|
||||
for src_rows in src_view.iter_2_rows(y_start, max_y) {
|
||||
for src_rows in src_view.iter_2_rows(y_start, max_rows) {
|
||||
let components1 = T::components(src_rows[0]);
|
||||
let components2 = T::components(src_rows[1]);
|
||||
// Load two coefficients at once
|
||||
@@ -267,12 +256,7 @@ unsafe fn vert_convolution_into_one_row_u8<T: InnerPixel<Component = u8>>(
|
||||
}
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss = i32x4_shr(sss, $imm8);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
sss = i32x4_shr(sss, precision);
|
||||
|
||||
sss = i16x8_narrow_i32x4(sss, sss);
|
||||
let dst_ptr = dst_chunk.as_mut_ptr() as *mut i32;
|
||||
|
||||
+1
-1
@@ -44,7 +44,7 @@ pub mod testing;
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
mod wasm32_utils;
|
||||
|
||||
/// This trait must be used in your code instead of [InnerPixel](crate::pixels::InnerPixel).
|
||||
/// This trait must be used in your code instead of [InnerPixel](pixels::InnerPixel).
|
||||
#[allow(private_bounds)]
|
||||
pub trait PixelTrait: Convolution + AlphaMulDiv {}
|
||||
|
||||
|
||||
+7
-7
@@ -71,13 +71,13 @@ pub(crate) unsafe fn i32x4_v128_from_u8(buf: &[u8], index: usize) -> v128 {
|
||||
i32x4(p.read_unaligned(), 0, 0, 0)
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "simd128")]
|
||||
pub(crate) unsafe fn u16x8_mul_shr16(a_u16x8: v128, b_u16x8: v128) -> v128 {
|
||||
let lo_u32x4 = u32x4_extmul_low_u16x8(a_u16x8, b_u16x8);
|
||||
let hi_u32x4 = u32x4_extmul_high_u16x8(a_u16x8, b_u16x8);
|
||||
i16x8_shuffle::<1, 3, 5, 7, 9, 11, 13, 15>(lo_u32x4, hi_u32x4)
|
||||
}
|
||||
// #[inline]
|
||||
// #[target_feature(enable = "simd128")]
|
||||
// pub(crate) unsafe fn u16x8_mul_shr16(a_u16x8: v128, b_u16x8: v128) -> v128 {
|
||||
// let lo_u32x4 = u32x4_extmul_low_u16x8(a_u16x8, b_u16x8);
|
||||
// let hi_u32x4 = u32x4_extmul_high_u16x8(a_u16x8, b_u16x8);
|
||||
// i16x8_shuffle::<1, 3, 5, 7, 9, 11, 13, 15>(lo_u32x4, hi_u32x4)
|
||||
// }
|
||||
|
||||
pub(crate) unsafe fn u16x8_mul_add_shr16(a_u16x8: v128, b_u16x8: v128, c: v128) -> v128 {
|
||||
let lo_u32x4 = u32x4_extmul_low_u16x8(a_u16x8, b_u16x8);
|
||||
|
||||
+2
-2
@@ -5,7 +5,7 @@ use image::{ColorType, DynamicImage};
|
||||
|
||||
use fast_image_resize::images::Image;
|
||||
use fast_image_resize::pixels::*;
|
||||
use fast_image_resize::{CpuExtensions, PixelType};
|
||||
use fast_image_resize::{CpuExtensions, PixelTrait, PixelType};
|
||||
|
||||
pub fn nonzero(v: u32) -> NonZeroU32 {
|
||||
NonZeroU32::new(v).unwrap()
|
||||
@@ -32,7 +32,7 @@ pub fn image_checksum<P: InnerPixel, const N: usize>(image: &Image) -> [u64; N]
|
||||
res
|
||||
}
|
||||
|
||||
pub trait PixelTestingExt: InnerPixel {
|
||||
pub trait PixelTestingExt: PixelTrait {
|
||||
fn pixel_type_str() -> &'static str {
|
||||
match Self::pixel_type() {
|
||||
PixelType::U8 => "u8",
|
||||
|
||||
Reference in New Issue
Block a user