mirror of
https://github.com/Cykooz/fast_image_resize.git
synced 2026-10-07 17:01:09 +00:00
Added support of optimisation with the help of Wasm32 SIMD128 for all F32 based pixel types (#61)
This commit is contained in:
+10
-8
@@ -8,6 +8,8 @@ mod avx2;
|
||||
mod native;
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod sse4;
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
mod wasm32;
|
||||
|
||||
type P = F32x2;
|
||||
|
||||
@@ -67,8 +69,8 @@ fn multiple(
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha(src_view, dst_view) },
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => unsafe { neon::multiply_alpha(src_view, dst_view) },
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha(src_view, dst_view) },
|
||||
_ => native::multiply_alpha(src_view, dst_view),
|
||||
}
|
||||
}
|
||||
@@ -81,8 +83,8 @@ fn multiply_inplace(image_view: &mut impl ImageViewMut<Pixel = P>, cpu_extension
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha_inplace(image_view) },
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => unsafe { neon::multiply_alpha_inplace(image_view) },
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha_inplace(image_view) },
|
||||
_ => native::multiply_alpha_inplace(image_view),
|
||||
}
|
||||
}
|
||||
@@ -99,8 +101,8 @@ fn divide(
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha(src_view, dst_view) },
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => unsafe { neon::divide_alpha(src_view, dst_view) },
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha(src_view, dst_view) },
|
||||
_ => native::divide_alpha(src_view, dst_view),
|
||||
}
|
||||
}
|
||||
@@ -113,8 +115,8 @@ fn divide_inplace(image_view: &mut impl ImageViewMut<Pixel = P>, cpu_extensions:
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha_inplace(image_view) },
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => unsafe { neon::divide_alpha_inplace(image_view) },
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha_inplace(image_view) },
|
||||
_ => native::divide_alpha_inplace(image_view),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,151 @@
|
||||
use core::arch::wasm32::*;
|
||||
|
||||
use super::native;
|
||||
use crate::pixels::F32x2;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
#[target_feature(enable = "simd128")]
|
||||
pub(crate) unsafe fn multiply_alpha(
|
||||
src_view: &impl ImageView<Pixel = F32x2>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = F32x2>,
|
||||
) {
|
||||
let src_rows = src_view.iter_rows(0);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
multiply_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "simd128")]
|
||||
pub(crate) unsafe fn multiply_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = F32x2>) {
|
||||
for row in image_view.iter_rows_mut(0) {
|
||||
multiply_alpha_row_inplace(row);
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "simd128")]
|
||||
pub(crate) unsafe fn multiply_alpha_row(src_row: &[F32x2], dst_row: &mut [F32x2]) {
|
||||
let src_chunks = src_row.chunks_exact(4);
|
||||
let src_remainder = src_chunks.remainder();
|
||||
let mut dst_chunks = dst_row.chunks_exact_mut(4);
|
||||
for (src_chunk, dst_chunk) in src_chunks.zip(&mut dst_chunks) {
|
||||
let src_ptr = src_chunk.as_ptr() as *const v128;
|
||||
let src_pixels01 = v128_load(src_ptr);
|
||||
let src_pixels23 = v128_load(src_ptr.add(1));
|
||||
multiply_alpha_4_pixels(src_pixels01, src_pixels23, dst_chunk);
|
||||
}
|
||||
|
||||
if !src_remainder.is_empty() {
|
||||
let dst_reminder = dst_chunks.into_remainder();
|
||||
native::multiply_alpha_row(src_remainder, dst_reminder);
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "simd128")]
|
||||
pub(crate) unsafe fn multiply_alpha_row_inplace(row: &mut [F32x2]) {
|
||||
let mut chunks = row.chunks_exact_mut(4);
|
||||
for chunk in &mut chunks {
|
||||
let src_ptr = chunk.as_ptr() as *const v128;
|
||||
let src_pixels01 = v128_load(src_ptr);
|
||||
let src_pixels23 = v128_load(src_ptr.add(1));
|
||||
multiply_alpha_4_pixels(src_pixels01, src_pixels23, chunk);
|
||||
}
|
||||
|
||||
let reminder = chunks.into_remainder();
|
||||
if !reminder.is_empty() {
|
||||
native::multiply_alpha_row_inplace(reminder);
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn multiply_alpha_4_pixels(pixels01: v128, pixels23: v128, dst_chunk: &mut [F32x2]) {
|
||||
let luma03 = i32x4_shuffle::<0, 2, 4, 6>(pixels01, pixels23);
|
||||
let alpha03 = i32x4_shuffle::<1, 3, 5, 7>(pixels01, pixels23);
|
||||
let multiplied_luma03 = f32x4_mul(luma03, alpha03);
|
||||
|
||||
let dst_pixel01 = i32x4_shuffle::<0, 4, 1, 5>(multiplied_luma03, alpha03);
|
||||
let dst_pixel23 = i32x4_shuffle::<2, 6, 3, 7>(multiplied_luma03, alpha03);
|
||||
let dst_ptr = dst_chunk.as_mut_ptr() as *mut v128;
|
||||
v128_store(dst_ptr, dst_pixel01);
|
||||
v128_store(dst_ptr.add(1), dst_pixel23);
|
||||
}
|
||||
|
||||
// Divide
|
||||
|
||||
#[target_feature(enable = "simd128")]
|
||||
pub(crate) unsafe fn divide_alpha(
|
||||
src_view: &impl ImageView<Pixel = F32x2>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = F32x2>,
|
||||
) {
|
||||
let src_rows = src_view.iter_rows(0);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
divide_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "simd128")]
|
||||
pub(crate) unsafe fn divide_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = F32x2>) {
|
||||
for row in image_view.iter_rows_mut(0) {
|
||||
divide_alpha_row_inplace(row);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "simd128")]
|
||||
pub(crate) unsafe fn divide_alpha_row(src_row: &[F32x2], dst_row: &mut [F32x2]) {
|
||||
let src_chunks = src_row.chunks_exact(4);
|
||||
let src_remainder = src_chunks.remainder();
|
||||
let mut dst_chunks = dst_row.chunks_exact_mut(4);
|
||||
|
||||
for (src_chunk, dst_chunk) in src_chunks.zip(&mut dst_chunks) {
|
||||
let src_ptr = src_chunk.as_ptr() as *const v128;
|
||||
let src_pixels01 = v128_load(src_ptr);
|
||||
let src_pixels23 = v128_load(src_ptr.add(1));
|
||||
divide_alpha_4_pixels(src_pixels01, src_pixels23, dst_chunk);
|
||||
}
|
||||
|
||||
if !src_remainder.is_empty() {
|
||||
let dst_reminder = dst_chunks.into_remainder();
|
||||
native::divide_alpha_row(src_remainder, dst_reminder);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "simd128")]
|
||||
pub(crate) unsafe fn divide_alpha_row_inplace(row: &mut [F32x2]) {
|
||||
let mut chunks = row.chunks_exact_mut(4);
|
||||
for chunk in &mut chunks {
|
||||
let src_ptr = chunk.as_ptr() as *const v128;
|
||||
let src_pixels01 = v128_load(src_ptr);
|
||||
let src_pixels23 = v128_load(src_ptr.add(1));
|
||||
divide_alpha_4_pixels(src_pixels01, src_pixels23, chunk);
|
||||
}
|
||||
|
||||
let reminder = chunks.into_remainder();
|
||||
if !reminder.is_empty() {
|
||||
native::divide_alpha_row_inplace(reminder);
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn divide_alpha_4_pixels(pixels01: v128, pixels23: v128, dst_chunk: &mut [F32x2]) {
|
||||
let zero = f32x4_splat(0.);
|
||||
|
||||
let luma03 = i32x4_shuffle::<0, 2, 4, 6>(pixels01, pixels23);
|
||||
let alpha03 = i32x4_shuffle::<1, 3, 5, 7>(pixels01, pixels23);
|
||||
let mut multiplied_luma03 = f32x4_div(luma03, alpha03);
|
||||
|
||||
let mask_zero = f32x4_ne(alpha03, zero);
|
||||
multiplied_luma03 = v128_and(mask_zero, multiplied_luma03);
|
||||
|
||||
let dst_pixel01 = i32x4_shuffle::<0, 4, 1, 5>(multiplied_luma03, alpha03);
|
||||
let dst_pixel23 = i32x4_shuffle::<2, 6, 3, 7>(multiplied_luma03, alpha03);
|
||||
let dst_ptr = dst_chunk.as_mut_ptr() as *mut v128;
|
||||
v128_store(dst_ptr, dst_pixel01);
|
||||
v128_store(dst_ptr.add(1), dst_pixel23);
|
||||
}
|
||||
+10
-8
@@ -8,6 +8,8 @@ mod avx2;
|
||||
mod native;
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod sse4;
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
mod wasm32;
|
||||
|
||||
type P = F32x4;
|
||||
|
||||
@@ -67,8 +69,8 @@ fn multiple(
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha(src_view, dst_view) },
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => unsafe { neon::multiply_alpha(src_view, dst_view) },
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha(src_view, dst_view) },
|
||||
_ => native::multiply_alpha(src_view, dst_view),
|
||||
}
|
||||
}
|
||||
@@ -81,8 +83,8 @@ fn multiply_inplace(image_view: &mut impl ImageViewMut<Pixel = P>, cpu_extension
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha_inplace(image_view) },
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => unsafe { neon::multiply_alpha_inplace(image_view) },
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha_inplace(image_view) },
|
||||
_ => native::multiply_alpha_inplace(image_view),
|
||||
}
|
||||
}
|
||||
@@ -99,8 +101,8 @@ fn divide(
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha(src_view, dst_view) },
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => unsafe { neon::divide_alpha(src_view, dst_view) },
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha(src_view, dst_view) },
|
||||
_ => native::divide_alpha(src_view, dst_view),
|
||||
}
|
||||
}
|
||||
@@ -113,8 +115,8 @@ fn divide_inplace(image_view: &mut impl ImageViewMut<Pixel = P>, cpu_extensions:
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha_inplace(image_view) },
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => unsafe { neon::divide_alpha_inplace(image_view) },
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha_inplace(image_view) },
|
||||
_ => native::divide_alpha_inplace(image_view),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,172 @@
|
||||
use core::arch::wasm32::*;
|
||||
|
||||
use super::native;
|
||||
use crate::pixels::F32x4;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
#[target_feature(enable = "simd128")]
|
||||
pub(crate) unsafe fn multiply_alpha(
|
||||
src_view: &impl ImageView<Pixel = F32x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = F32x4>,
|
||||
) {
|
||||
let src_rows = src_view.iter_rows(0);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
multiply_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "simd128")]
|
||||
pub(crate) unsafe fn multiply_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = F32x4>) {
|
||||
for row in image_view.iter_rows_mut(0) {
|
||||
multiply_alpha_row_inplace(row);
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "simd128")]
|
||||
pub(crate) unsafe fn multiply_alpha_row(src_row: &[F32x4], dst_row: &mut [F32x4]) {
|
||||
let src_chunks = src_row.chunks_exact(4);
|
||||
let src_remainder = src_chunks.remainder();
|
||||
let mut dst_chunks = dst_row.chunks_exact_mut(4);
|
||||
for (src_chunk, dst_chunk) in src_chunks.zip(&mut dst_chunks) {
|
||||
let src_pixels = load_4_pixels(src_chunk);
|
||||
multiply_alpha_4_pixels(src_pixels, dst_chunk);
|
||||
}
|
||||
|
||||
if !src_remainder.is_empty() {
|
||||
let dst_reminder = dst_chunks.into_remainder();
|
||||
native::multiply_alpha_row(src_remainder, dst_reminder);
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "simd128")]
|
||||
pub(crate) unsafe fn multiply_alpha_row_inplace(row: &mut [F32x4]) {
|
||||
let mut chunks = row.chunks_exact_mut(4);
|
||||
for chunk in &mut chunks {
|
||||
let src_pixels = load_4_pixels(chunk);
|
||||
multiply_alpha_4_pixels(src_pixels, chunk);
|
||||
}
|
||||
|
||||
let reminder = chunks.into_remainder();
|
||||
if !reminder.is_empty() {
|
||||
native::multiply_alpha_row_inplace(reminder);
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn multiply_alpha_4_pixels(pixels: [v128; 4], dst_chunk: &mut [F32x4]) {
|
||||
let r_f32x4 = f32x4_mul(pixels[0], pixels[3]);
|
||||
let g_f32x4 = f32x4_mul(pixels[1], pixels[3]);
|
||||
let b_f32x4 = f32x4_mul(pixels[2], pixels[3]);
|
||||
store_4_pixels([r_f32x4, g_f32x4, b_f32x4, pixels[3]], dst_chunk);
|
||||
}
|
||||
|
||||
// Divide
|
||||
|
||||
#[target_feature(enable = "simd128")]
|
||||
pub(crate) unsafe fn divide_alpha(
|
||||
src_view: &impl ImageView<Pixel = F32x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = F32x4>,
|
||||
) {
|
||||
let src_rows = src_view.iter_rows(0);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
divide_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "simd128")]
|
||||
pub(crate) unsafe fn divide_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = F32x4>) {
|
||||
for row in image_view.iter_rows_mut(0) {
|
||||
divide_alpha_row_inplace(row);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "simd128")]
|
||||
pub(crate) unsafe fn divide_alpha_row(src_row: &[F32x4], dst_row: &mut [F32x4]) {
|
||||
let src_chunks = src_row.chunks_exact(4);
|
||||
let src_remainder = src_chunks.remainder();
|
||||
let mut dst_chunks = dst_row.chunks_exact_mut(4);
|
||||
|
||||
for (src_chunk, dst_chunk) in src_chunks.zip(&mut dst_chunks) {
|
||||
let src_pixels = load_4_pixels(src_chunk);
|
||||
divide_alpha_4_pixels(src_pixels, dst_chunk);
|
||||
}
|
||||
|
||||
if !src_remainder.is_empty() {
|
||||
let dst_reminder = dst_chunks.into_remainder();
|
||||
native::divide_alpha_row(src_remainder, dst_reminder);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "simd128")]
|
||||
pub(crate) unsafe fn divide_alpha_row_inplace(row: &mut [F32x4]) {
|
||||
let mut chunks = row.chunks_exact_mut(4);
|
||||
for chunk in &mut chunks {
|
||||
let src_pixels = load_4_pixels(chunk);
|
||||
divide_alpha_4_pixels(src_pixels, chunk);
|
||||
}
|
||||
|
||||
let reminder = chunks.into_remainder();
|
||||
if !reminder.is_empty() {
|
||||
native::divide_alpha_row_inplace(reminder);
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn divide_alpha_4_pixels(pixels: [v128; 4], dst_chunk: &mut [F32x4]) {
|
||||
let mut r_f32x4 = f32x4_div(pixels[0], pixels[3]);
|
||||
let mut g_f32x4 = f32x4_div(pixels[1], pixels[3]);
|
||||
let mut b_f32x4 = f32x4_div(pixels[2], pixels[3]);
|
||||
let zero = f32x4_splat(0.);
|
||||
let mask_zero = f32x4_ne(pixels[3], zero);
|
||||
r_f32x4 = v128_and(mask_zero, r_f32x4);
|
||||
g_f32x4 = v128_and(mask_zero, g_f32x4);
|
||||
b_f32x4 = v128_and(mask_zero, b_f32x4);
|
||||
|
||||
store_4_pixels([r_f32x4, g_f32x4, b_f32x4, pixels[3]], dst_chunk);
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn load_4_pixels(pixels: &[F32x4]) -> [v128; 4] {
|
||||
let ptr = pixels.as_ptr() as *const v128;
|
||||
cols_into_rows([
|
||||
v128_load(ptr),
|
||||
v128_load(ptr.add(1)),
|
||||
v128_load(ptr.add(2)),
|
||||
v128_load(ptr.add(3)),
|
||||
])
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn store_4_pixels(pixels: [v128; 4], dst_chunk: &mut [F32x4]) {
|
||||
let pixels = cols_into_rows(pixels);
|
||||
let mut dst_ptr = dst_chunk.as_mut_ptr() as *mut v128;
|
||||
for rgba in pixels {
|
||||
v128_store(dst_ptr, rgba);
|
||||
dst_ptr = dst_ptr.add(1);
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn cols_into_rows(pixels: [v128; 4]) -> [v128; 4] {
|
||||
let rrgg01 = i32x4_shuffle::<0, 4, 1, 5>(pixels[0], pixels[1]);
|
||||
let rrgg23 = i32x4_shuffle::<0, 4, 1, 5>(pixels[2], pixels[3]);
|
||||
let r0123 = i64x2_shuffle::<0, 2>(rrgg01, rrgg23);
|
||||
let g0123 = i64x2_shuffle::<1, 3>(rrgg01, rrgg23);
|
||||
|
||||
let bbaa01 = i32x4_shuffle::<2, 6, 3, 7>(pixels[0], pixels[1]);
|
||||
let bbaa23 = i32x4_shuffle::<2, 6, 3, 7>(pixels[2], pixels[3]);
|
||||
let b0123 = i64x2_shuffle::<0, 2>(bbaa01, bbaa23);
|
||||
let a0123 = i64x2_shuffle::<1, 3>(bbaa01, bbaa23);
|
||||
[r0123, g0123, b0123, a0123]
|
||||
}
|
||||
@@ -11,8 +11,8 @@ mod native;
|
||||
// mod neon;
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod sse4;
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// mod wasm32;
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
mod wasm32;
|
||||
|
||||
type P = F32;
|
||||
|
||||
@@ -75,8 +75,8 @@ fn horiz_convolution(
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
_ => native::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,108 @@
|
||||
use core::arch::wasm32::*;
|
||||
|
||||
use crate::convolution::{Coefficients, CoefficientsChunk};
|
||||
use crate::pixels::F32;
|
||||
use crate::wasm32_utils;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = F32>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = F32>,
|
||||
offset: u32,
|
||||
coeffs: &Coefficients,
|
||||
) {
|
||||
let coefficients_chunks = coeffs.get_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_rows(src_rows, dst_rows, &coefficients_chunks);
|
||||
}
|
||||
}
|
||||
|
||||
let yy = dst_height - dst_height % 4;
|
||||
let src_rows = src_view.iter_rows(yy + offset);
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_rows([src_row], [dst_row], &coefficients_chunks);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// For safety, it is necessary to ensure the following conditions:
|
||||
/// - length of all rows in src_rows must be equal
|
||||
/// - length of all rows in dst_rows must be equal
|
||||
/// - coefficients_chunks.len() == dst_rows.0.len()
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn horiz_convolution_rows<const ROWS_COUNT: usize>(
|
||||
src_rows: [&[F32]; ROWS_COUNT],
|
||||
dst_rows: [&mut [F32]; ROWS_COUNT],
|
||||
coefficients_chunks: &[CoefficientsChunk],
|
||||
) {
|
||||
let mut ll_buf = [0f64; 2];
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut sums = [f64x2_splat(0.); ROWS_COUNT];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
coeffs = coeffs_by_4.remainder();
|
||||
|
||||
for k in coeffs_by_4 {
|
||||
let coeff01_f64x2 = wasm32_utils::load_v128(k, 0);
|
||||
let coeff23_f64x2 = wasm32_utils::load_v128(k, 2);
|
||||
|
||||
for i in 0..ROWS_COUNT {
|
||||
let mut sum = sums[i];
|
||||
let source = wasm32_utils::load_v128(src_rows[i], x);
|
||||
|
||||
let pixel01_f64 = f64x2_promote_low_f32x4(source);
|
||||
sum = wasm32_utils::f64x2_mul_add(sum, pixel01_f64, coeff01_f64x2);
|
||||
|
||||
let pixel23_f64 = wasm32_utils::f64x2_promote_high_f32x4(source);
|
||||
sum = wasm32_utils::f64x2_mul_add(sum, pixel23_f64, coeff23_f64x2);
|
||||
|
||||
sums[i] = sum;
|
||||
}
|
||||
x += 4;
|
||||
}
|
||||
|
||||
let coeffs_by_2 = coeffs.chunks_exact(2);
|
||||
coeffs = coeffs_by_2.remainder();
|
||||
for k in coeffs_by_2 {
|
||||
let coeff01_f64x2 = wasm32_utils::load_v128(k, 0);
|
||||
|
||||
for i in 0..ROWS_COUNT {
|
||||
let pixel0 = src_rows[i].get_unchecked(x).0;
|
||||
let pixel1 = src_rows[i].get_unchecked(x + 1).0;
|
||||
let pixel01_f64 = f64x2(pixel0 as f64, pixel1 as f64);
|
||||
sums[i] = wasm32_utils::f64x2_mul_add(sums[i], pixel01_f64, coeff01_f64x2);
|
||||
}
|
||||
x += 2;
|
||||
}
|
||||
|
||||
if let Some(&k) = coeffs.first() {
|
||||
let coeff0_f64x2 = f64x2_splat(k);
|
||||
|
||||
for i in 0..ROWS_COUNT {
|
||||
let pixel0 = src_rows[i].get_unchecked(x).0;
|
||||
let pixel0_f64 = f64x2(pixel0 as f64, 0.);
|
||||
sums[i] = wasm32_utils::f64x2_mul_add(sums[i], pixel0_f64, coeff0_f64x2);
|
||||
}
|
||||
}
|
||||
|
||||
for i in 0..ROWS_COUNT {
|
||||
v128_store(ll_buf.as_mut_ptr() as *mut v128, sums[i]);
|
||||
let dst_pixel = dst_rows[i].get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = (ll_buf[0] + ll_buf[1]) as f32;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -11,8 +11,8 @@ mod native;
|
||||
// mod neon;
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod sse4;
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// mod wasm32;
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
mod wasm32;
|
||||
|
||||
type P = F32x2;
|
||||
|
||||
@@ -75,8 +75,8 @@ fn horiz_convolution(
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
_ => native::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,96 @@
|
||||
use core::arch::wasm32::*;
|
||||
|
||||
use crate::convolution::{Coefficients, CoefficientsChunk};
|
||||
use crate::pixels::F32x2;
|
||||
use crate::wasm32_utils;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = F32x2>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = F32x2>,
|
||||
offset: u32,
|
||||
coeffs: &Coefficients,
|
||||
) {
|
||||
let coefficients_chunks = coeffs.get_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_rows(src_rows, dst_rows, &coefficients_chunks);
|
||||
}
|
||||
}
|
||||
|
||||
let yy = dst_height - dst_height % 4;
|
||||
let src_rows = src_view.iter_rows(yy + offset);
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_rows([src_row], [dst_row], &coefficients_chunks);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// For safety, it is necessary to ensure the following conditions:
|
||||
/// - length of all rows in src_rows must be equal
|
||||
/// - length of all rows in dst_rows must be equal
|
||||
/// - coefficients_chunks.len() == dst_rows.0.len()
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn horiz_convolution_rows<const ROWS_COUNT: usize>(
|
||||
src_rows: [&[F32x2]; ROWS_COUNT],
|
||||
dst_rows: [&mut [F32x2]; ROWS_COUNT],
|
||||
coefficients_chunks: &[CoefficientsChunk],
|
||||
) {
|
||||
let mut ll_buf = [0f64; 2];
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut ll_sum = [f64x2_splat(0.); ROWS_COUNT];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
|
||||
let coeffs_by_2 = coeffs.chunks_exact(2);
|
||||
coeffs = coeffs_by_2.remainder();
|
||||
|
||||
for k in coeffs_by_2 {
|
||||
let coeff0_f64x2 = f64x2_splat(k[0]);
|
||||
let coeff1_f64x2 = f64x2_splat(k[1]);
|
||||
|
||||
for i in 0..ROWS_COUNT {
|
||||
let mut sum = ll_sum[i];
|
||||
let source = wasm32_utils::load_v128(src_rows[i], x);
|
||||
|
||||
let pixel0_f64 = f64x2_promote_low_f32x4(source);
|
||||
sum = wasm32_utils::f64x2_mul_add(sum, pixel0_f64, coeff0_f64x2);
|
||||
|
||||
let pixel1_f64 = wasm32_utils::f64x2_promote_high_f32x4(source);
|
||||
sum = wasm32_utils::f64x2_mul_add(sum, pixel1_f64, coeff1_f64x2);
|
||||
|
||||
ll_sum[i] = sum;
|
||||
}
|
||||
x += 2;
|
||||
}
|
||||
|
||||
if let Some(&k) = coeffs.first() {
|
||||
let coeff0_f64x2 = f64x2_splat(k);
|
||||
|
||||
for i in 0..ROWS_COUNT {
|
||||
let pixel = src_rows[i].get_unchecked(x);
|
||||
let source = f32x4(pixel.0[0], pixel.0[1], 0., 0.);
|
||||
|
||||
let pixel0_f64 = f64x2_promote_low_f32x4(source);
|
||||
ll_sum[i] = wasm32_utils::f64x2_mul_add(ll_sum[i], pixel0_f64, coeff0_f64x2);
|
||||
}
|
||||
}
|
||||
|
||||
for i in 0..ROWS_COUNT {
|
||||
v128_store(ll_buf.as_mut_ptr() as *mut v128, ll_sum[i]);
|
||||
let dst_pixel = dst_rows[i].get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = ll_buf.map(|v| v as f32);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -11,8 +11,8 @@ mod native;
|
||||
// mod neon;
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod sse4;
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// mod wasm32;
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
mod wasm32;
|
||||
|
||||
type P = F32x3;
|
||||
|
||||
@@ -75,8 +75,8 @@ fn horiz_convolution(
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
_ => native::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,161 @@
|
||||
use core::arch::wasm32::*;
|
||||
|
||||
use crate::convolution::{Coefficients, CoefficientsChunk};
|
||||
use crate::pixels::{F32x3, InnerPixel};
|
||||
use crate::wasm32_utils;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = F32x3>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = F32x3>,
|
||||
offset: u32,
|
||||
coeffs: &Coefficients,
|
||||
) {
|
||||
let coefficients_chunks = coeffs.get_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_2_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_2_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_rows(src_rows, dst_rows, &coefficients_chunks);
|
||||
}
|
||||
}
|
||||
|
||||
let yy = dst_height - dst_height % 2;
|
||||
let src_rows = src_view.iter_rows(yy + offset);
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_rows([src_row], [dst_row], &coefficients_chunks);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// For safety, it is necessary to ensure the following conditions:
|
||||
/// - length of all rows in src_rows must be equal
|
||||
/// - length of all rows in dst_rows must be equal
|
||||
/// - coefficients_chunks.len() == dst_rows.0.len()
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn horiz_convolution_rows<const ROWS_COUNT: usize>(
|
||||
src_rows: [&[F32x3]; ROWS_COUNT],
|
||||
dst_rows: [&mut [F32x3]; ROWS_COUNT],
|
||||
coefficients_chunks: &[CoefficientsChunk],
|
||||
) {
|
||||
/*
|
||||
|R0 G0 B0| |R1|
|
||||
|00 01 02| |03|
|
||||
|
||||
|G1 B1| |R2 G2|
|
||||
|00 01| |02 03|
|
||||
|
||||
|B2| |R3 G3 B3|
|
||||
|00| |01 02 03|
|
||||
*/
|
||||
|
||||
let mut rg_buf = [0f64; 2];
|
||||
let mut br_buf = [0f64; 2];
|
||||
let mut gb_buf = [0f64; 2];
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut rg_sums = [f64x2_splat(0.); ROWS_COUNT];
|
||||
let mut br_sums = [f64x2_splat(0.); ROWS_COUNT];
|
||||
let mut gb_sums = [f64x2_splat(0.); ROWS_COUNT];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
coeffs = coeffs_by_4.remainder();
|
||||
|
||||
for k in coeffs_by_4 {
|
||||
let coeff00_f64x2 = f64x2_splat(k[0]);
|
||||
let coeff01_f64x2 = f64x2(k[0], k[1]);
|
||||
let coeff11_f64x2 = f64x2_splat(k[1]);
|
||||
let coeff22_f64x2 = f64x2_splat(k[2]);
|
||||
let coeff23_f64x2 = f64x2(k[2], k[3]);
|
||||
let coeff33_f64x2 = f64x2_splat(k[3]);
|
||||
|
||||
for i in 0..ROWS_COUNT {
|
||||
let c = x * 3;
|
||||
let components = F32x3::components(src_rows[i]);
|
||||
|
||||
let rgb0r1 = wasm32_utils::load_v128(components, c);
|
||||
let rg0_f64x2 = f64x2_promote_low_f32x4(rgb0r1);
|
||||
rg_sums[i] = wasm32_utils::f64x2_mul_add(rg_sums[i], rg0_f64x2, coeff00_f64x2);
|
||||
let b0r1_f64x2 = wasm32_utils::f64x2_promote_high_f32x4(rgb0r1);
|
||||
br_sums[i] = wasm32_utils::f64x2_mul_add(br_sums[i], b0r1_f64x2, coeff01_f64x2);
|
||||
|
||||
let gb1rg2 = wasm32_utils::load_v128(components, c + 4);
|
||||
let gb1_f64x2 = f64x2_promote_low_f32x4(gb1rg2);
|
||||
gb_sums[i] = wasm32_utils::f64x2_mul_add(gb_sums[i], gb1_f64x2, coeff11_f64x2);
|
||||
let rg2_f64x2 = wasm32_utils::f64x2_promote_high_f32x4(gb1rg2);
|
||||
rg_sums[i] = wasm32_utils::f64x2_mul_add(rg_sums[i], rg2_f64x2, coeff22_f64x2);
|
||||
|
||||
let b2rgb3 = wasm32_utils::load_v128(components, c + 8);
|
||||
let b2r3_f64x2 = f64x2_promote_low_f32x4(b2rgb3);
|
||||
br_sums[i] = wasm32_utils::f64x2_mul_add(br_sums[i], b2r3_f64x2, coeff23_f64x2);
|
||||
let gb3_f64x2 = wasm32_utils::f64x2_promote_high_f32x4(b2rgb3);
|
||||
gb_sums[i] = wasm32_utils::f64x2_mul_add(gb_sums[i], gb3_f64x2, coeff33_f64x2);
|
||||
}
|
||||
x += 4;
|
||||
}
|
||||
|
||||
let coeffs_by_2 = coeffs.chunks_exact(2);
|
||||
coeffs = coeffs_by_2.remainder();
|
||||
for k in coeffs_by_2 {
|
||||
let coeff00_f64x2 = f64x2_splat(k[0]);
|
||||
let coeff01_f64x2 = f64x2(k[0], k[1]);
|
||||
let coeff11_f64x2 = f64x2_splat(k[1]);
|
||||
|
||||
for i in 0..ROWS_COUNT {
|
||||
let c = x * 3;
|
||||
let components = F32x3::components(src_rows[i]);
|
||||
|
||||
let rgb0r1 = wasm32_utils::load_v128(components, c);
|
||||
let rg0_f64x2 = f64x2_promote_low_f32x4(rgb0r1);
|
||||
rg_sums[i] = wasm32_utils::f64x2_mul_add(rg_sums[i], rg0_f64x2, coeff00_f64x2);
|
||||
let b0r1_f64x2 = wasm32_utils::f64x2_promote_high_f32x4(rgb0r1);
|
||||
br_sums[i] = wasm32_utils::f64x2_mul_add(br_sums[i], b0r1_f64x2, coeff01_f64x2);
|
||||
|
||||
let g1 = *components.get_unchecked(c + 4);
|
||||
let b1 = *components.get_unchecked(c + 5);
|
||||
let gb1_f64x2 = f64x2(g1 as f64, b1 as f64);
|
||||
gb_sums[i] = wasm32_utils::f64x2_mul_add(gb_sums[i], gb1_f64x2, coeff11_f64x2);
|
||||
}
|
||||
x += 2;
|
||||
}
|
||||
|
||||
for &k in coeffs {
|
||||
let coeff00_f64x2 = f64x2_splat(k);
|
||||
let coeff0x_f64x2 = f64x2(k, 0.);
|
||||
|
||||
for i in 0..ROWS_COUNT {
|
||||
let pixel = src_rows[i].get_unchecked(x);
|
||||
let rgb0x = f32x4(pixel.0[0], pixel.0[1], pixel.0[2], 0.);
|
||||
|
||||
let rg0_f64x2 = f64x2_promote_low_f32x4(rgb0x);
|
||||
rg_sums[i] = wasm32_utils::f64x2_mul_add(rg_sums[i], rg0_f64x2, coeff00_f64x2);
|
||||
|
||||
let b0x_f64x2 = wasm32_utils::f64x2_promote_high_f32x4(rgb0x);
|
||||
br_sums[i] = wasm32_utils::f64x2_mul_add(br_sums[i], b0x_f64x2, coeff0x_f64x2);
|
||||
}
|
||||
x += 1;
|
||||
}
|
||||
|
||||
for i in 0..ROWS_COUNT {
|
||||
v128_store(rg_buf.as_mut_ptr() as *mut v128, rg_sums[i]);
|
||||
v128_store(br_buf.as_mut_ptr() as *mut v128, br_sums[i]);
|
||||
v128_store(gb_buf.as_mut_ptr() as *mut v128, gb_sums[i]);
|
||||
let dst_pixel = dst_rows[i].get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = [
|
||||
(rg_buf[0] + br_buf[1]) as f32,
|
||||
(rg_buf[1] + gb_buf[0]) as f32,
|
||||
(br_buf[0] + gb_buf[1]) as f32,
|
||||
];
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -11,8 +11,8 @@ mod native;
|
||||
// mod neon;
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod sse4;
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// mod wasm32;
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
mod wasm32;
|
||||
|
||||
type P = F32x4;
|
||||
|
||||
@@ -75,8 +75,8 @@ fn horiz_convolution(
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
_ => native::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,81 @@
|
||||
use core::arch::wasm32::*;
|
||||
|
||||
use crate::convolution::{Coefficients, CoefficientsChunk};
|
||||
use crate::pixels::F32x4;
|
||||
use crate::wasm32_utils;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = F32x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = F32x4>,
|
||||
offset: u32,
|
||||
coeffs: &Coefficients,
|
||||
) {
|
||||
let coefficients_chunks = coeffs.get_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_rows(src_rows, dst_rows, &coefficients_chunks);
|
||||
}
|
||||
}
|
||||
|
||||
let yy = dst_height - dst_height % 4;
|
||||
let src_rows = src_view.iter_rows(yy + offset);
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_rows([src_row], [dst_row], &coefficients_chunks);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// For safety, it is necessary to ensure the following conditions:
|
||||
/// - length of all rows in src_rows must be equal
|
||||
/// - length of all rows in dst_rows must be equal
|
||||
/// - coefficients_chunks.len() == dst_rows.0.len()
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn horiz_convolution_rows<const ROWS_COUNT: usize>(
|
||||
src_rows: [&[F32x4]; ROWS_COUNT],
|
||||
dst_rows: [&mut [F32x4]; ROWS_COUNT],
|
||||
coefficients_chunks: &[CoefficientsChunk],
|
||||
) {
|
||||
let mut rg_buf = [0f64; 2];
|
||||
let mut ba_buf = [0f64; 2];
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut rg_sums = [f64x2_splat(0.); ROWS_COUNT];
|
||||
let mut ba_sums = [f64x2_splat(0.); ROWS_COUNT];
|
||||
|
||||
for &k in coeffs_chunk.values {
|
||||
let coeffs_f64x2 = f64x2_splat(k);
|
||||
|
||||
for r in 0..ROWS_COUNT {
|
||||
let pixel = wasm32_utils::load_v128(src_rows[r], x);
|
||||
let rg_f64x2 = f64x2_promote_low_f32x4(pixel);
|
||||
rg_sums[r] = wasm32_utils::f64x2_mul_add(rg_sums[r], rg_f64x2, coeffs_f64x2);
|
||||
let ba_f64x2 = wasm32_utils::f64x2_promote_high_f32x4(pixel);
|
||||
ba_sums[r] = wasm32_utils::f64x2_mul_add(ba_sums[r], ba_f64x2, coeffs_f64x2);
|
||||
}
|
||||
x += 1;
|
||||
}
|
||||
|
||||
for i in 0..ROWS_COUNT {
|
||||
v128_store(rg_buf.as_mut_ptr() as *mut v128, rg_sums[i]);
|
||||
v128_store(ba_buf.as_mut_ptr() as *mut v128, ba_sums[i]);
|
||||
let dst_pixel = dst_rows[i].get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = [
|
||||
rg_buf[0] as f32,
|
||||
rg_buf[1] as f32,
|
||||
ba_buf[0] as f32,
|
||||
ba_buf[1] as f32,
|
||||
];
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -9,8 +9,8 @@ pub(crate) mod native;
|
||||
// mod neon;
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
pub(crate) mod sse4;
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// pub mod wasm32;
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
pub mod wasm32;
|
||||
|
||||
pub(crate) fn vert_convolution_f32<T: InnerPixel<Component = f32>>(
|
||||
src_view: &impl ImageView<Pixel = T>,
|
||||
@@ -30,8 +30,8 @@ pub(crate) fn vert_convolution_f32<T: InnerPixel<Component = f32>>(
|
||||
CpuExtensions::Sse4_1 => sse4::vert_convolution(src_view, dst_view, offset, coeffs),
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => neon::vert_convolution(src_view, dst_view, offset, coeffs),
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => wasm32::vert_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => wasm32::vert_convolution(src_view, dst_view, offset, coeffs),
|
||||
_ => native::vert_convolution(src_view, dst_view, offset, coeffs),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,131 @@
|
||||
use core::arch::wasm32::*;
|
||||
|
||||
use super::native;
|
||||
use crate::convolution::{Coefficients, CoefficientsChunk};
|
||||
use crate::pixels::InnerPixel;
|
||||
use crate::wasm32_utils;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
pub(crate) fn vert_convolution<T>(
|
||||
src_view: &impl ImageView<Pixel = T>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = T>,
|
||||
offset: u32,
|
||||
coeffs: &Coefficients,
|
||||
) where
|
||||
T: InnerPixel<Component = f32>,
|
||||
{
|
||||
let coefficients_chunks = coeffs.get_chunks();
|
||||
let src_x = offset as usize * T::count_of_components();
|
||||
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
for (dst_row, coeffs_chunk) in dst_rows.zip(coefficients_chunks) {
|
||||
unsafe {
|
||||
vert_convolution_into_one_row_f32(src_view, dst_row, src_x, coeffs_chunk);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn vert_convolution_into_one_row_f32<T: InnerPixel<Component = f32>>(
|
||||
src_view: &impl ImageView<Pixel = T>,
|
||||
dst_row: &mut [T],
|
||||
mut src_x: usize,
|
||||
coeffs_chunk: CoefficientsChunk,
|
||||
) {
|
||||
let mut c_buf = [0f64; 2];
|
||||
let mut dst_f32 = T::components_mut(dst_row);
|
||||
|
||||
let mut dst_chunks = dst_f32.chunks_exact_mut(16);
|
||||
for dst_chunk in &mut dst_chunks {
|
||||
multiply_components_of_rows::<_, 8>(src_view, src_x, coeffs_chunk, dst_chunk, &mut c_buf);
|
||||
src_x += 16;
|
||||
}
|
||||
|
||||
dst_f32 = dst_chunks.into_remainder();
|
||||
dst_chunks = dst_f32.chunks_exact_mut(8);
|
||||
for dst_chunk in &mut dst_chunks {
|
||||
multiply_components_of_rows::<_, 4>(src_view, src_x, coeffs_chunk, dst_chunk, &mut c_buf);
|
||||
src_x += 8;
|
||||
}
|
||||
|
||||
dst_f32 = dst_chunks.into_remainder();
|
||||
dst_chunks = dst_f32.chunks_exact_mut(4);
|
||||
if let Some(dst_chunk) = dst_chunks.next() {
|
||||
multiply_components_of_rows::<_, 2>(src_view, src_x, coeffs_chunk, dst_chunk, &mut c_buf);
|
||||
src_x += 4;
|
||||
}
|
||||
|
||||
dst_f32 = dst_chunks.into_remainder();
|
||||
if !dst_f32.is_empty() {
|
||||
let y_start = coeffs_chunk.start;
|
||||
let coeffs = coeffs_chunk.values;
|
||||
native::convolution_by_f32(src_view, dst_f32, src_x, y_start, coeffs);
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "simd128")]
|
||||
pub(crate) unsafe fn multiply_components_of_rows<
|
||||
T: InnerPixel<Component = f32>,
|
||||
const SUMS_COUNT: usize,
|
||||
>(
|
||||
src_view: &impl ImageView<Pixel = T>,
|
||||
src_x: usize,
|
||||
coeffs_chunk: CoefficientsChunk,
|
||||
dst_chunk: &mut [f32],
|
||||
c_buf: &mut [f64; 2],
|
||||
) {
|
||||
let mut sums = [f64x2_splat(0.); SUMS_COUNT];
|
||||
let y_start = coeffs_chunk.start;
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut y: u32 = 0;
|
||||
let max_rows = coeffs.len() as u32;
|
||||
|
||||
let coeffs_2 = coeffs.chunks_exact(2);
|
||||
coeffs = coeffs_2.remainder();
|
||||
for (src_rows, two_coeffs) in src_view.iter_2_rows(y_start, max_rows).zip(coeffs_2) {
|
||||
let src_rows = src_rows.map(|row| T::components(row).get_unchecked(src_x..));
|
||||
for (&coeff, src_row) in two_coeffs.iter().zip(src_rows) {
|
||||
multiply_components_of_row(&mut sums, coeff, src_row);
|
||||
}
|
||||
y += 2;
|
||||
}
|
||||
|
||||
if let Some(&coeff) = coeffs.first() {
|
||||
if let Some(s_row) = src_view.iter_rows(y_start + y).next() {
|
||||
let src_row = T::components(s_row).get_unchecked(src_x..);
|
||||
multiply_components_of_row(&mut sums, coeff, src_row);
|
||||
}
|
||||
}
|
||||
|
||||
let mut dst_ptr = dst_chunk.as_mut_ptr();
|
||||
for sum in sums {
|
||||
v128_store(c_buf.as_mut_ptr() as *mut v128, sum);
|
||||
for &v in c_buf.iter() {
|
||||
*dst_ptr = v as f32;
|
||||
dst_ptr = dst_ptr.add(1);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn multiply_components_of_row<const SUMS_COUNT: usize>(
|
||||
sums: &mut [v128; SUMS_COUNT],
|
||||
coeff: f64,
|
||||
src_row: &[f32],
|
||||
) {
|
||||
let coeff_f64x2 = f64x2_splat(coeff);
|
||||
let mut i = 0;
|
||||
while i < SUMS_COUNT {
|
||||
let comp03_f32x4 = wasm32_utils::load_v128(src_row, i * 2);
|
||||
|
||||
let comp01_f64x2 = f64x2_promote_low_f32x4(comp03_f32x4);
|
||||
sums[i] = wasm32_utils::f64x2_mul_add(sums[i], comp01_f64x2, coeff_f64x2);
|
||||
i += 1;
|
||||
|
||||
let comp23_f64x2 = wasm32_utils::f64x2_promote_high_f32x4(comp03_f32x4);
|
||||
sums[i] = wasm32_utils::f64x2_mul_add(sums[i], comp23_f64x2, coeff_f64x2);
|
||||
i += 1;
|
||||
}
|
||||
}
|
||||
@@ -98,6 +98,33 @@ pub unsafe fn u8x16_narrow_u16x8(mut a_u8x16: v128, mut b_u8x16: v128) -> v128 {
|
||||
u8x16_shuffle::<0, 2, 4, 6, 8, 10, 12, 14, 16, 18, 20, 22, 24, 26, 28, 30>(a_u8x16, b_u8x16)
|
||||
}
|
||||
|
||||
/// Promotes the two high `f32` lanes of `v` into a `f64x2`.
|
||||
///
|
||||
/// This is the WASM counterpart of `_mm_cvtps_pd(_mm_movehl_ps(v, v))`.
|
||||
#[inline]
|
||||
#[target_feature(enable = "simd128")]
|
||||
pub(crate) unsafe fn f64x2_promote_high_f32x4(v: v128) -> v128 {
|
||||
f64x2_promote_low_f32x4(i32x4_shuffle::<2, 3, 2, 3>(v, v))
|
||||
}
|
||||
|
||||
/// Computes `acc + a * b` for `f64x2` vectors.
|
||||
///
|
||||
/// When the crate is built with the `relaxed-simd` target feature, this uses
|
||||
/// the possibly fused `f64x2_relaxed_madd` instruction. Otherwise it falls
|
||||
/// back to a separate multiply and add.
|
||||
#[inline]
|
||||
#[target_feature(enable = "simd128")]
|
||||
pub(crate) unsafe fn f64x2_mul_add(acc: v128, a: v128, b: v128) -> v128 {
|
||||
#[cfg(target_feature = "relaxed-simd")]
|
||||
{
|
||||
f64x2_relaxed_madd(a, b, acc)
|
||||
}
|
||||
#[cfg(not(target_feature = "relaxed-simd"))]
|
||||
{
|
||||
f64x2_add(acc, f64x2_mul(a, b))
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "simd128")]
|
||||
pub(crate) unsafe fn i64x2_mul_lo(a: v128, b: v128) -> v128 {
|
||||
|
||||
Reference in New Issue
Block a user