Added support of optimisation with the help of Wasm32 SIMD128 for all F32 based pixel types (#61)

This commit is contained in:
Colin D Murphy
2026-08-21 01:21:59 +03:00
committed by GitHub
parent 6dc1fe2750
commit f22335a9b4
15 changed files with 967 additions and 36 deletions
+10 -8
View File
@@ -8,6 +8,8 @@ mod avx2;
mod native;
#[cfg(target_arch = "x86_64")]
mod sse4;
#[cfg(target_arch = "wasm32")]
mod wasm32;
type P = F32x2;
@@ -67,8 +69,8 @@ fn multiple(
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha(src_view, dst_view) },
// #[cfg(target_arch = "aarch64")]
// CpuExtensions::Neon => unsafe { neon::multiply_alpha(src_view, dst_view) },
// #[cfg(target_arch = "wasm32")]
// CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha(src_view, dst_view) },
#[cfg(target_arch = "wasm32")]
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha(src_view, dst_view) },
_ => native::multiply_alpha(src_view, dst_view),
}
}
@@ -81,8 +83,8 @@ fn multiply_inplace(image_view: &mut impl ImageViewMut<Pixel = P>, cpu_extension
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha_inplace(image_view) },
// #[cfg(target_arch = "aarch64")]
// CpuExtensions::Neon => unsafe { neon::multiply_alpha_inplace(image_view) },
// #[cfg(target_arch = "wasm32")]
// CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha_inplace(image_view) },
#[cfg(target_arch = "wasm32")]
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha_inplace(image_view) },
_ => native::multiply_alpha_inplace(image_view),
}
}
@@ -99,8 +101,8 @@ fn divide(
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha(src_view, dst_view) },
// #[cfg(target_arch = "aarch64")]
// CpuExtensions::Neon => unsafe { neon::divide_alpha(src_view, dst_view) },
// #[cfg(target_arch = "wasm32")]
// CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha(src_view, dst_view) },
#[cfg(target_arch = "wasm32")]
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha(src_view, dst_view) },
_ => native::divide_alpha(src_view, dst_view),
}
}
@@ -113,8 +115,8 @@ fn divide_inplace(image_view: &mut impl ImageViewMut<Pixel = P>, cpu_extensions:
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha_inplace(image_view) },
// #[cfg(target_arch = "aarch64")]
// CpuExtensions::Neon => unsafe { neon::divide_alpha_inplace(image_view) },
// #[cfg(target_arch = "wasm32")]
// CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha_inplace(image_view) },
#[cfg(target_arch = "wasm32")]
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha_inplace(image_view) },
_ => native::divide_alpha_inplace(image_view),
}
}
+151
View File
@@ -0,0 +1,151 @@
use core::arch::wasm32::*;
use super::native;
use crate::pixels::F32x2;
use crate::{ImageView, ImageViewMut};
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn multiply_alpha(
src_view: &impl ImageView<Pixel = F32x2>,
dst_view: &mut impl ImageViewMut<Pixel = F32x2>,
) {
let src_rows = src_view.iter_rows(0);
let dst_rows = dst_view.iter_rows_mut(0);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
multiply_alpha_row(src_row, dst_row);
}
}
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn multiply_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = F32x2>) {
for row in image_view.iter_rows_mut(0) {
multiply_alpha_row_inplace(row);
}
}
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn multiply_alpha_row(src_row: &[F32x2], dst_row: &mut [F32x2]) {
let src_chunks = src_row.chunks_exact(4);
let src_remainder = src_chunks.remainder();
let mut dst_chunks = dst_row.chunks_exact_mut(4);
for (src_chunk, dst_chunk) in src_chunks.zip(&mut dst_chunks) {
let src_ptr = src_chunk.as_ptr() as *const v128;
let src_pixels01 = v128_load(src_ptr);
let src_pixels23 = v128_load(src_ptr.add(1));
multiply_alpha_4_pixels(src_pixels01, src_pixels23, dst_chunk);
}
if !src_remainder.is_empty() {
let dst_reminder = dst_chunks.into_remainder();
native::multiply_alpha_row(src_remainder, dst_reminder);
}
}
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn multiply_alpha_row_inplace(row: &mut [F32x2]) {
let mut chunks = row.chunks_exact_mut(4);
for chunk in &mut chunks {
let src_ptr = chunk.as_ptr() as *const v128;
let src_pixels01 = v128_load(src_ptr);
let src_pixels23 = v128_load(src_ptr.add(1));
multiply_alpha_4_pixels(src_pixels01, src_pixels23, chunk);
}
let reminder = chunks.into_remainder();
if !reminder.is_empty() {
native::multiply_alpha_row_inplace(reminder);
}
}
#[inline]
#[target_feature(enable = "simd128")]
unsafe fn multiply_alpha_4_pixels(pixels01: v128, pixels23: v128, dst_chunk: &mut [F32x2]) {
let luma03 = i32x4_shuffle::<0, 2, 4, 6>(pixels01, pixels23);
let alpha03 = i32x4_shuffle::<1, 3, 5, 7>(pixels01, pixels23);
let multiplied_luma03 = f32x4_mul(luma03, alpha03);
let dst_pixel01 = i32x4_shuffle::<0, 4, 1, 5>(multiplied_luma03, alpha03);
let dst_pixel23 = i32x4_shuffle::<2, 6, 3, 7>(multiplied_luma03, alpha03);
let dst_ptr = dst_chunk.as_mut_ptr() as *mut v128;
v128_store(dst_ptr, dst_pixel01);
v128_store(dst_ptr.add(1), dst_pixel23);
}
// Divide
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn divide_alpha(
src_view: &impl ImageView<Pixel = F32x2>,
dst_view: &mut impl ImageViewMut<Pixel = F32x2>,
) {
let src_rows = src_view.iter_rows(0);
let dst_rows = dst_view.iter_rows_mut(0);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
divide_alpha_row(src_row, dst_row);
}
}
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn divide_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = F32x2>) {
for row in image_view.iter_rows_mut(0) {
divide_alpha_row_inplace(row);
}
}
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn divide_alpha_row(src_row: &[F32x2], dst_row: &mut [F32x2]) {
let src_chunks = src_row.chunks_exact(4);
let src_remainder = src_chunks.remainder();
let mut dst_chunks = dst_row.chunks_exact_mut(4);
for (src_chunk, dst_chunk) in src_chunks.zip(&mut dst_chunks) {
let src_ptr = src_chunk.as_ptr() as *const v128;
let src_pixels01 = v128_load(src_ptr);
let src_pixels23 = v128_load(src_ptr.add(1));
divide_alpha_4_pixels(src_pixels01, src_pixels23, dst_chunk);
}
if !src_remainder.is_empty() {
let dst_reminder = dst_chunks.into_remainder();
native::divide_alpha_row(src_remainder, dst_reminder);
}
}
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn divide_alpha_row_inplace(row: &mut [F32x2]) {
let mut chunks = row.chunks_exact_mut(4);
for chunk in &mut chunks {
let src_ptr = chunk.as_ptr() as *const v128;
let src_pixels01 = v128_load(src_ptr);
let src_pixels23 = v128_load(src_ptr.add(1));
divide_alpha_4_pixels(src_pixels01, src_pixels23, chunk);
}
let reminder = chunks.into_remainder();
if !reminder.is_empty() {
native::divide_alpha_row_inplace(reminder);
}
}
#[inline]
#[target_feature(enable = "simd128")]
unsafe fn divide_alpha_4_pixels(pixels01: v128, pixels23: v128, dst_chunk: &mut [F32x2]) {
let zero = f32x4_splat(0.);
let luma03 = i32x4_shuffle::<0, 2, 4, 6>(pixels01, pixels23);
let alpha03 = i32x4_shuffle::<1, 3, 5, 7>(pixels01, pixels23);
let mut multiplied_luma03 = f32x4_div(luma03, alpha03);
let mask_zero = f32x4_ne(alpha03, zero);
multiplied_luma03 = v128_and(mask_zero, multiplied_luma03);
let dst_pixel01 = i32x4_shuffle::<0, 4, 1, 5>(multiplied_luma03, alpha03);
let dst_pixel23 = i32x4_shuffle::<2, 6, 3, 7>(multiplied_luma03, alpha03);
let dst_ptr = dst_chunk.as_mut_ptr() as *mut v128;
v128_store(dst_ptr, dst_pixel01);
v128_store(dst_ptr.add(1), dst_pixel23);
}
+10 -8
View File
@@ -8,6 +8,8 @@ mod avx2;
mod native;
#[cfg(target_arch = "x86_64")]
mod sse4;
#[cfg(target_arch = "wasm32")]
mod wasm32;
type P = F32x4;
@@ -67,8 +69,8 @@ fn multiple(
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha(src_view, dst_view) },
// #[cfg(target_arch = "aarch64")]
// CpuExtensions::Neon => unsafe { neon::multiply_alpha(src_view, dst_view) },
// #[cfg(target_arch = "wasm32")]
// CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha(src_view, dst_view) },
#[cfg(target_arch = "wasm32")]
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha(src_view, dst_view) },
_ => native::multiply_alpha(src_view, dst_view),
}
}
@@ -81,8 +83,8 @@ fn multiply_inplace(image_view: &mut impl ImageViewMut<Pixel = P>, cpu_extension
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha_inplace(image_view) },
// #[cfg(target_arch = "aarch64")]
// CpuExtensions::Neon => unsafe { neon::multiply_alpha_inplace(image_view) },
// #[cfg(target_arch = "wasm32")]
// CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha_inplace(image_view) },
#[cfg(target_arch = "wasm32")]
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha_inplace(image_view) },
_ => native::multiply_alpha_inplace(image_view),
}
}
@@ -99,8 +101,8 @@ fn divide(
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha(src_view, dst_view) },
// #[cfg(target_arch = "aarch64")]
// CpuExtensions::Neon => unsafe { neon::divide_alpha(src_view, dst_view) },
// #[cfg(target_arch = "wasm32")]
// CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha(src_view, dst_view) },
#[cfg(target_arch = "wasm32")]
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha(src_view, dst_view) },
_ => native::divide_alpha(src_view, dst_view),
}
}
@@ -113,8 +115,8 @@ fn divide_inplace(image_view: &mut impl ImageViewMut<Pixel = P>, cpu_extensions:
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha_inplace(image_view) },
// #[cfg(target_arch = "aarch64")]
// CpuExtensions::Neon => unsafe { neon::divide_alpha_inplace(image_view) },
// #[cfg(target_arch = "wasm32")]
// CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha_inplace(image_view) },
#[cfg(target_arch = "wasm32")]
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha_inplace(image_view) },
_ => native::divide_alpha_inplace(image_view),
}
}
+172
View File
@@ -0,0 +1,172 @@
use core::arch::wasm32::*;
use super::native;
use crate::pixels::F32x4;
use crate::{ImageView, ImageViewMut};
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn multiply_alpha(
src_view: &impl ImageView<Pixel = F32x4>,
dst_view: &mut impl ImageViewMut<Pixel = F32x4>,
) {
let src_rows = src_view.iter_rows(0);
let dst_rows = dst_view.iter_rows_mut(0);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
multiply_alpha_row(src_row, dst_row);
}
}
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn multiply_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = F32x4>) {
for row in image_view.iter_rows_mut(0) {
multiply_alpha_row_inplace(row);
}
}
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn multiply_alpha_row(src_row: &[F32x4], dst_row: &mut [F32x4]) {
let src_chunks = src_row.chunks_exact(4);
let src_remainder = src_chunks.remainder();
let mut dst_chunks = dst_row.chunks_exact_mut(4);
for (src_chunk, dst_chunk) in src_chunks.zip(&mut dst_chunks) {
let src_pixels = load_4_pixels(src_chunk);
multiply_alpha_4_pixels(src_pixels, dst_chunk);
}
if !src_remainder.is_empty() {
let dst_reminder = dst_chunks.into_remainder();
native::multiply_alpha_row(src_remainder, dst_reminder);
}
}
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn multiply_alpha_row_inplace(row: &mut [F32x4]) {
let mut chunks = row.chunks_exact_mut(4);
for chunk in &mut chunks {
let src_pixels = load_4_pixels(chunk);
multiply_alpha_4_pixels(src_pixels, chunk);
}
let reminder = chunks.into_remainder();
if !reminder.is_empty() {
native::multiply_alpha_row_inplace(reminder);
}
}
#[inline]
#[target_feature(enable = "simd128")]
unsafe fn multiply_alpha_4_pixels(pixels: [v128; 4], dst_chunk: &mut [F32x4]) {
let r_f32x4 = f32x4_mul(pixels[0], pixels[3]);
let g_f32x4 = f32x4_mul(pixels[1], pixels[3]);
let b_f32x4 = f32x4_mul(pixels[2], pixels[3]);
store_4_pixels([r_f32x4, g_f32x4, b_f32x4, pixels[3]], dst_chunk);
}
// Divide
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn divide_alpha(
src_view: &impl ImageView<Pixel = F32x4>,
dst_view: &mut impl ImageViewMut<Pixel = F32x4>,
) {
let src_rows = src_view.iter_rows(0);
let dst_rows = dst_view.iter_rows_mut(0);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
divide_alpha_row(src_row, dst_row);
}
}
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn divide_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = F32x4>) {
for row in image_view.iter_rows_mut(0) {
divide_alpha_row_inplace(row);
}
}
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn divide_alpha_row(src_row: &[F32x4], dst_row: &mut [F32x4]) {
let src_chunks = src_row.chunks_exact(4);
let src_remainder = src_chunks.remainder();
let mut dst_chunks = dst_row.chunks_exact_mut(4);
for (src_chunk, dst_chunk) in src_chunks.zip(&mut dst_chunks) {
let src_pixels = load_4_pixels(src_chunk);
divide_alpha_4_pixels(src_pixels, dst_chunk);
}
if !src_remainder.is_empty() {
let dst_reminder = dst_chunks.into_remainder();
native::divide_alpha_row(src_remainder, dst_reminder);
}
}
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn divide_alpha_row_inplace(row: &mut [F32x4]) {
let mut chunks = row.chunks_exact_mut(4);
for chunk in &mut chunks {
let src_pixels = load_4_pixels(chunk);
divide_alpha_4_pixels(src_pixels, chunk);
}
let reminder = chunks.into_remainder();
if !reminder.is_empty() {
native::divide_alpha_row_inplace(reminder);
}
}
#[inline]
#[target_feature(enable = "simd128")]
unsafe fn divide_alpha_4_pixels(pixels: [v128; 4], dst_chunk: &mut [F32x4]) {
let mut r_f32x4 = f32x4_div(pixels[0], pixels[3]);
let mut g_f32x4 = f32x4_div(pixels[1], pixels[3]);
let mut b_f32x4 = f32x4_div(pixels[2], pixels[3]);
let zero = f32x4_splat(0.);
let mask_zero = f32x4_ne(pixels[3], zero);
r_f32x4 = v128_and(mask_zero, r_f32x4);
g_f32x4 = v128_and(mask_zero, g_f32x4);
b_f32x4 = v128_and(mask_zero, b_f32x4);
store_4_pixels([r_f32x4, g_f32x4, b_f32x4, pixels[3]], dst_chunk);
}
#[inline]
#[target_feature(enable = "simd128")]
unsafe fn load_4_pixels(pixels: &[F32x4]) -> [v128; 4] {
let ptr = pixels.as_ptr() as *const v128;
cols_into_rows([
v128_load(ptr),
v128_load(ptr.add(1)),
v128_load(ptr.add(2)),
v128_load(ptr.add(3)),
])
}
#[inline]
#[target_feature(enable = "simd128")]
unsafe fn store_4_pixels(pixels: [v128; 4], dst_chunk: &mut [F32x4]) {
let pixels = cols_into_rows(pixels);
let mut dst_ptr = dst_chunk.as_mut_ptr() as *mut v128;
for rgba in pixels {
v128_store(dst_ptr, rgba);
dst_ptr = dst_ptr.add(1);
}
}
#[inline]
#[target_feature(enable = "simd128")]
unsafe fn cols_into_rows(pixels: [v128; 4]) -> [v128; 4] {
let rrgg01 = i32x4_shuffle::<0, 4, 1, 5>(pixels[0], pixels[1]);
let rrgg23 = i32x4_shuffle::<0, 4, 1, 5>(pixels[2], pixels[3]);
let r0123 = i64x2_shuffle::<0, 2>(rrgg01, rrgg23);
let g0123 = i64x2_shuffle::<1, 3>(rrgg01, rrgg23);
let bbaa01 = i32x4_shuffle::<2, 6, 3, 7>(pixels[0], pixels[1]);
let bbaa23 = i32x4_shuffle::<2, 6, 3, 7>(pixels[2], pixels[3]);
let b0123 = i64x2_shuffle::<0, 2>(bbaa01, bbaa23);
let a0123 = i64x2_shuffle::<1, 3>(bbaa01, bbaa23);
[r0123, g0123, b0123, a0123]
}
+4 -4
View File
@@ -11,8 +11,8 @@ mod native;
// mod neon;
#[cfg(target_arch = "x86_64")]
mod sse4;
// #[cfg(target_arch = "wasm32")]
// mod wasm32;
#[cfg(target_arch = "wasm32")]
mod wasm32;
type P = F32;
@@ -75,8 +75,8 @@ fn horiz_convolution(
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, coeffs),
// #[cfg(target_arch = "aarch64")]
// CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, coeffs),
// #[cfg(target_arch = "wasm32")]
// CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
_ => native::horiz_convolution(src_view, dst_view, offset, coeffs),
}
}
+108
View File
@@ -0,0 +1,108 @@
use core::arch::wasm32::*;
use crate::convolution::{Coefficients, CoefficientsChunk};
use crate::pixels::F32;
use crate::wasm32_utils;
use crate::{ImageView, ImageViewMut};
#[inline]
pub(crate) fn horiz_convolution(
src_view: &impl ImageView<Pixel = F32>,
dst_view: &mut impl ImageViewMut<Pixel = F32>,
offset: u32,
coeffs: &Coefficients,
) {
let coefficients_chunks = coeffs.get_chunks();
let dst_height = dst_view.height();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_rows(src_rows, dst_rows, &coefficients_chunks);
}
}
let yy = dst_height - dst_height % 4;
let src_rows = src_view.iter_rows(yy + offset);
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_rows([src_row], [dst_row], &coefficients_chunks);
}
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - length of all rows in src_rows must be equal
/// - length of all rows in dst_rows must be equal
/// - coefficients_chunks.len() == dst_rows.0.len()
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "simd128")]
unsafe fn horiz_convolution_rows<const ROWS_COUNT: usize>(
src_rows: [&[F32]; ROWS_COUNT],
dst_rows: [&mut [F32]; ROWS_COUNT],
coefficients_chunks: &[CoefficientsChunk],
) {
let mut ll_buf = [0f64; 2];
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
let mut sums = [f64x2_splat(0.); ROWS_COUNT];
let mut coeffs = coeffs_chunk.values;
let coeffs_by_4 = coeffs.chunks_exact(4);
coeffs = coeffs_by_4.remainder();
for k in coeffs_by_4 {
let coeff01_f64x2 = wasm32_utils::load_v128(k, 0);
let coeff23_f64x2 = wasm32_utils::load_v128(k, 2);
for i in 0..ROWS_COUNT {
let mut sum = sums[i];
let source = wasm32_utils::load_v128(src_rows[i], x);
let pixel01_f64 = f64x2_promote_low_f32x4(source);
sum = wasm32_utils::f64x2_mul_add(sum, pixel01_f64, coeff01_f64x2);
let pixel23_f64 = wasm32_utils::f64x2_promote_high_f32x4(source);
sum = wasm32_utils::f64x2_mul_add(sum, pixel23_f64, coeff23_f64x2);
sums[i] = sum;
}
x += 4;
}
let coeffs_by_2 = coeffs.chunks_exact(2);
coeffs = coeffs_by_2.remainder();
for k in coeffs_by_2 {
let coeff01_f64x2 = wasm32_utils::load_v128(k, 0);
for i in 0..ROWS_COUNT {
let pixel0 = src_rows[i].get_unchecked(x).0;
let pixel1 = src_rows[i].get_unchecked(x + 1).0;
let pixel01_f64 = f64x2(pixel0 as f64, pixel1 as f64);
sums[i] = wasm32_utils::f64x2_mul_add(sums[i], pixel01_f64, coeff01_f64x2);
}
x += 2;
}
if let Some(&k) = coeffs.first() {
let coeff0_f64x2 = f64x2_splat(k);
for i in 0..ROWS_COUNT {
let pixel0 = src_rows[i].get_unchecked(x).0;
let pixel0_f64 = f64x2(pixel0 as f64, 0.);
sums[i] = wasm32_utils::f64x2_mul_add(sums[i], pixel0_f64, coeff0_f64x2);
}
}
for i in 0..ROWS_COUNT {
v128_store(ll_buf.as_mut_ptr() as *mut v128, sums[i]);
let dst_pixel = dst_rows[i].get_unchecked_mut(dst_x);
dst_pixel.0 = (ll_buf[0] + ll_buf[1]) as f32;
}
}
}
+4 -4
View File
@@ -11,8 +11,8 @@ mod native;
// mod neon;
#[cfg(target_arch = "x86_64")]
mod sse4;
// #[cfg(target_arch = "wasm32")]
// mod wasm32;
#[cfg(target_arch = "wasm32")]
mod wasm32;
type P = F32x2;
@@ -75,8 +75,8 @@ fn horiz_convolution(
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, coeffs),
// #[cfg(target_arch = "aarch64")]
// CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, coeffs),
// #[cfg(target_arch = "wasm32")]
// CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
_ => native::horiz_convolution(src_view, dst_view, offset, coeffs),
}
}
+96
View File
@@ -0,0 +1,96 @@
use core::arch::wasm32::*;
use crate::convolution::{Coefficients, CoefficientsChunk};
use crate::pixels::F32x2;
use crate::wasm32_utils;
use crate::{ImageView, ImageViewMut};
#[inline]
pub(crate) fn horiz_convolution(
src_view: &impl ImageView<Pixel = F32x2>,
dst_view: &mut impl ImageViewMut<Pixel = F32x2>,
offset: u32,
coeffs: &Coefficients,
) {
let coefficients_chunks = coeffs.get_chunks();
let dst_height = dst_view.height();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_rows(src_rows, dst_rows, &coefficients_chunks);
}
}
let yy = dst_height - dst_height % 4;
let src_rows = src_view.iter_rows(yy + offset);
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_rows([src_row], [dst_row], &coefficients_chunks);
}
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - length of all rows in src_rows must be equal
/// - length of all rows in dst_rows must be equal
/// - coefficients_chunks.len() == dst_rows.0.len()
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "simd128")]
unsafe fn horiz_convolution_rows<const ROWS_COUNT: usize>(
src_rows: [&[F32x2]; ROWS_COUNT],
dst_rows: [&mut [F32x2]; ROWS_COUNT],
coefficients_chunks: &[CoefficientsChunk],
) {
let mut ll_buf = [0f64; 2];
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
let mut ll_sum = [f64x2_splat(0.); ROWS_COUNT];
let mut coeffs = coeffs_chunk.values;
let coeffs_by_2 = coeffs.chunks_exact(2);
coeffs = coeffs_by_2.remainder();
for k in coeffs_by_2 {
let coeff0_f64x2 = f64x2_splat(k[0]);
let coeff1_f64x2 = f64x2_splat(k[1]);
for i in 0..ROWS_COUNT {
let mut sum = ll_sum[i];
let source = wasm32_utils::load_v128(src_rows[i], x);
let pixel0_f64 = f64x2_promote_low_f32x4(source);
sum = wasm32_utils::f64x2_mul_add(sum, pixel0_f64, coeff0_f64x2);
let pixel1_f64 = wasm32_utils::f64x2_promote_high_f32x4(source);
sum = wasm32_utils::f64x2_mul_add(sum, pixel1_f64, coeff1_f64x2);
ll_sum[i] = sum;
}
x += 2;
}
if let Some(&k) = coeffs.first() {
let coeff0_f64x2 = f64x2_splat(k);
for i in 0..ROWS_COUNT {
let pixel = src_rows[i].get_unchecked(x);
let source = f32x4(pixel.0[0], pixel.0[1], 0., 0.);
let pixel0_f64 = f64x2_promote_low_f32x4(source);
ll_sum[i] = wasm32_utils::f64x2_mul_add(ll_sum[i], pixel0_f64, coeff0_f64x2);
}
}
for i in 0..ROWS_COUNT {
v128_store(ll_buf.as_mut_ptr() as *mut v128, ll_sum[i]);
let dst_pixel = dst_rows[i].get_unchecked_mut(dst_x);
dst_pixel.0 = ll_buf.map(|v| v as f32);
}
}
}
+4 -4
View File
@@ -11,8 +11,8 @@ mod native;
// mod neon;
#[cfg(target_arch = "x86_64")]
mod sse4;
// #[cfg(target_arch = "wasm32")]
// mod wasm32;
#[cfg(target_arch = "wasm32")]
mod wasm32;
type P = F32x3;
@@ -75,8 +75,8 @@ fn horiz_convolution(
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, coeffs),
// #[cfg(target_arch = "aarch64")]
// CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, coeffs),
// #[cfg(target_arch = "wasm32")]
// CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
_ => native::horiz_convolution(src_view, dst_view, offset, coeffs),
}
}
+161
View File
@@ -0,0 +1,161 @@
use core::arch::wasm32::*;
use crate::convolution::{Coefficients, CoefficientsChunk};
use crate::pixels::{F32x3, InnerPixel};
use crate::wasm32_utils;
use crate::{ImageView, ImageViewMut};
#[inline]
pub(crate) fn horiz_convolution(
src_view: &impl ImageView<Pixel = F32x3>,
dst_view: &mut impl ImageViewMut<Pixel = F32x3>,
offset: u32,
coeffs: &Coefficients,
) {
let coefficients_chunks = coeffs.get_chunks();
let dst_height = dst_view.height();
let src_iter = src_view.iter_2_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_2_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_rows(src_rows, dst_rows, &coefficients_chunks);
}
}
let yy = dst_height - dst_height % 2;
let src_rows = src_view.iter_rows(yy + offset);
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_rows([src_row], [dst_row], &coefficients_chunks);
}
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - length of all rows in src_rows must be equal
/// - length of all rows in dst_rows must be equal
/// - coefficients_chunks.len() == dst_rows.0.len()
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "simd128")]
unsafe fn horiz_convolution_rows<const ROWS_COUNT: usize>(
src_rows: [&[F32x3]; ROWS_COUNT],
dst_rows: [&mut [F32x3]; ROWS_COUNT],
coefficients_chunks: &[CoefficientsChunk],
) {
/*
|R0 G0 B0| |R1|
|00 01 02| |03|
|G1 B1| |R2 G2|
|00 01| |02 03|
|B2| |R3 G3 B3|
|00| |01 02 03|
*/
let mut rg_buf = [0f64; 2];
let mut br_buf = [0f64; 2];
let mut gb_buf = [0f64; 2];
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
let mut rg_sums = [f64x2_splat(0.); ROWS_COUNT];
let mut br_sums = [f64x2_splat(0.); ROWS_COUNT];
let mut gb_sums = [f64x2_splat(0.); ROWS_COUNT];
let mut coeffs = coeffs_chunk.values;
let coeffs_by_4 = coeffs.chunks_exact(4);
coeffs = coeffs_by_4.remainder();
for k in coeffs_by_4 {
let coeff00_f64x2 = f64x2_splat(k[0]);
let coeff01_f64x2 = f64x2(k[0], k[1]);
let coeff11_f64x2 = f64x2_splat(k[1]);
let coeff22_f64x2 = f64x2_splat(k[2]);
let coeff23_f64x2 = f64x2(k[2], k[3]);
let coeff33_f64x2 = f64x2_splat(k[3]);
for i in 0..ROWS_COUNT {
let c = x * 3;
let components = F32x3::components(src_rows[i]);
let rgb0r1 = wasm32_utils::load_v128(components, c);
let rg0_f64x2 = f64x2_promote_low_f32x4(rgb0r1);
rg_sums[i] = wasm32_utils::f64x2_mul_add(rg_sums[i], rg0_f64x2, coeff00_f64x2);
let b0r1_f64x2 = wasm32_utils::f64x2_promote_high_f32x4(rgb0r1);
br_sums[i] = wasm32_utils::f64x2_mul_add(br_sums[i], b0r1_f64x2, coeff01_f64x2);
let gb1rg2 = wasm32_utils::load_v128(components, c + 4);
let gb1_f64x2 = f64x2_promote_low_f32x4(gb1rg2);
gb_sums[i] = wasm32_utils::f64x2_mul_add(gb_sums[i], gb1_f64x2, coeff11_f64x2);
let rg2_f64x2 = wasm32_utils::f64x2_promote_high_f32x4(gb1rg2);
rg_sums[i] = wasm32_utils::f64x2_mul_add(rg_sums[i], rg2_f64x2, coeff22_f64x2);
let b2rgb3 = wasm32_utils::load_v128(components, c + 8);
let b2r3_f64x2 = f64x2_promote_low_f32x4(b2rgb3);
br_sums[i] = wasm32_utils::f64x2_mul_add(br_sums[i], b2r3_f64x2, coeff23_f64x2);
let gb3_f64x2 = wasm32_utils::f64x2_promote_high_f32x4(b2rgb3);
gb_sums[i] = wasm32_utils::f64x2_mul_add(gb_sums[i], gb3_f64x2, coeff33_f64x2);
}
x += 4;
}
let coeffs_by_2 = coeffs.chunks_exact(2);
coeffs = coeffs_by_2.remainder();
for k in coeffs_by_2 {
let coeff00_f64x2 = f64x2_splat(k[0]);
let coeff01_f64x2 = f64x2(k[0], k[1]);
let coeff11_f64x2 = f64x2_splat(k[1]);
for i in 0..ROWS_COUNT {
let c = x * 3;
let components = F32x3::components(src_rows[i]);
let rgb0r1 = wasm32_utils::load_v128(components, c);
let rg0_f64x2 = f64x2_promote_low_f32x4(rgb0r1);
rg_sums[i] = wasm32_utils::f64x2_mul_add(rg_sums[i], rg0_f64x2, coeff00_f64x2);
let b0r1_f64x2 = wasm32_utils::f64x2_promote_high_f32x4(rgb0r1);
br_sums[i] = wasm32_utils::f64x2_mul_add(br_sums[i], b0r1_f64x2, coeff01_f64x2);
let g1 = *components.get_unchecked(c + 4);
let b1 = *components.get_unchecked(c + 5);
let gb1_f64x2 = f64x2(g1 as f64, b1 as f64);
gb_sums[i] = wasm32_utils::f64x2_mul_add(gb_sums[i], gb1_f64x2, coeff11_f64x2);
}
x += 2;
}
for &k in coeffs {
let coeff00_f64x2 = f64x2_splat(k);
let coeff0x_f64x2 = f64x2(k, 0.);
for i in 0..ROWS_COUNT {
let pixel = src_rows[i].get_unchecked(x);
let rgb0x = f32x4(pixel.0[0], pixel.0[1], pixel.0[2], 0.);
let rg0_f64x2 = f64x2_promote_low_f32x4(rgb0x);
rg_sums[i] = wasm32_utils::f64x2_mul_add(rg_sums[i], rg0_f64x2, coeff00_f64x2);
let b0x_f64x2 = wasm32_utils::f64x2_promote_high_f32x4(rgb0x);
br_sums[i] = wasm32_utils::f64x2_mul_add(br_sums[i], b0x_f64x2, coeff0x_f64x2);
}
x += 1;
}
for i in 0..ROWS_COUNT {
v128_store(rg_buf.as_mut_ptr() as *mut v128, rg_sums[i]);
v128_store(br_buf.as_mut_ptr() as *mut v128, br_sums[i]);
v128_store(gb_buf.as_mut_ptr() as *mut v128, gb_sums[i]);
let dst_pixel = dst_rows[i].get_unchecked_mut(dst_x);
dst_pixel.0 = [
(rg_buf[0] + br_buf[1]) as f32,
(rg_buf[1] + gb_buf[0]) as f32,
(br_buf[0] + gb_buf[1]) as f32,
];
}
}
}
+4 -4
View File
@@ -11,8 +11,8 @@ mod native;
// mod neon;
#[cfg(target_arch = "x86_64")]
mod sse4;
// #[cfg(target_arch = "wasm32")]
// mod wasm32;
#[cfg(target_arch = "wasm32")]
mod wasm32;
type P = F32x4;
@@ -75,8 +75,8 @@ fn horiz_convolution(
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, coeffs),
// #[cfg(target_arch = "aarch64")]
// CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, coeffs),
// #[cfg(target_arch = "wasm32")]
// CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
_ => native::horiz_convolution(src_view, dst_view, offset, coeffs),
}
}
+81
View File
@@ -0,0 +1,81 @@
use core::arch::wasm32::*;
use crate::convolution::{Coefficients, CoefficientsChunk};
use crate::pixels::F32x4;
use crate::wasm32_utils;
use crate::{ImageView, ImageViewMut};
#[inline]
pub(crate) fn horiz_convolution(
src_view: &impl ImageView<Pixel = F32x4>,
dst_view: &mut impl ImageViewMut<Pixel = F32x4>,
offset: u32,
coeffs: &Coefficients,
) {
let coefficients_chunks = coeffs.get_chunks();
let dst_height = dst_view.height();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_rows(src_rows, dst_rows, &coefficients_chunks);
}
}
let yy = dst_height - dst_height % 4;
let src_rows = src_view.iter_rows(yy + offset);
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_rows([src_row], [dst_row], &coefficients_chunks);
}
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - length of all rows in src_rows must be equal
/// - length of all rows in dst_rows must be equal
/// - coefficients_chunks.len() == dst_rows.0.len()
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "simd128")]
unsafe fn horiz_convolution_rows<const ROWS_COUNT: usize>(
src_rows: [&[F32x4]; ROWS_COUNT],
dst_rows: [&mut [F32x4]; ROWS_COUNT],
coefficients_chunks: &[CoefficientsChunk],
) {
let mut rg_buf = [0f64; 2];
let mut ba_buf = [0f64; 2];
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
let mut rg_sums = [f64x2_splat(0.); ROWS_COUNT];
let mut ba_sums = [f64x2_splat(0.); ROWS_COUNT];
for &k in coeffs_chunk.values {
let coeffs_f64x2 = f64x2_splat(k);
for r in 0..ROWS_COUNT {
let pixel = wasm32_utils::load_v128(src_rows[r], x);
let rg_f64x2 = f64x2_promote_low_f32x4(pixel);
rg_sums[r] = wasm32_utils::f64x2_mul_add(rg_sums[r], rg_f64x2, coeffs_f64x2);
let ba_f64x2 = wasm32_utils::f64x2_promote_high_f32x4(pixel);
ba_sums[r] = wasm32_utils::f64x2_mul_add(ba_sums[r], ba_f64x2, coeffs_f64x2);
}
x += 1;
}
for i in 0..ROWS_COUNT {
v128_store(rg_buf.as_mut_ptr() as *mut v128, rg_sums[i]);
v128_store(ba_buf.as_mut_ptr() as *mut v128, ba_sums[i]);
let dst_pixel = dst_rows[i].get_unchecked_mut(dst_x);
dst_pixel.0 = [
rg_buf[0] as f32,
rg_buf[1] as f32,
ba_buf[0] as f32,
ba_buf[1] as f32,
];
}
}
}
+4 -4
View File
@@ -9,8 +9,8 @@ pub(crate) mod native;
// mod neon;
#[cfg(target_arch = "x86_64")]
pub(crate) mod sse4;
// #[cfg(target_arch = "wasm32")]
// pub mod wasm32;
#[cfg(target_arch = "wasm32")]
pub mod wasm32;
pub(crate) fn vert_convolution_f32<T: InnerPixel<Component = f32>>(
src_view: &impl ImageView<Pixel = T>,
@@ -30,8 +30,8 @@ pub(crate) fn vert_convolution_f32<T: InnerPixel<Component = f32>>(
CpuExtensions::Sse4_1 => sse4::vert_convolution(src_view, dst_view, offset, coeffs),
// #[cfg(target_arch = "aarch64")]
// CpuExtensions::Neon => neon::vert_convolution(src_view, dst_view, offset, coeffs),
// #[cfg(target_arch = "wasm32")]
// CpuExtensions::Simd128 => wasm32::vert_convolution(src_view, dst_view, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Simd128 => wasm32::vert_convolution(src_view, dst_view, offset, coeffs),
_ => native::vert_convolution(src_view, dst_view, offset, coeffs),
}
}
+131
View File
@@ -0,0 +1,131 @@
use core::arch::wasm32::*;
use super::native;
use crate::convolution::{Coefficients, CoefficientsChunk};
use crate::pixels::InnerPixel;
use crate::wasm32_utils;
use crate::{ImageView, ImageViewMut};
pub(crate) fn vert_convolution<T>(
src_view: &impl ImageView<Pixel = T>,
dst_view: &mut impl ImageViewMut<Pixel = T>,
offset: u32,
coeffs: &Coefficients,
) where
T: InnerPixel<Component = f32>,
{
let coefficients_chunks = coeffs.get_chunks();
let src_x = offset as usize * T::count_of_components();
let dst_rows = dst_view.iter_rows_mut(0);
for (dst_row, coeffs_chunk) in dst_rows.zip(coefficients_chunks) {
unsafe {
vert_convolution_into_one_row_f32(src_view, dst_row, src_x, coeffs_chunk);
}
}
}
#[target_feature(enable = "simd128")]
unsafe fn vert_convolution_into_one_row_f32<T: InnerPixel<Component = f32>>(
src_view: &impl ImageView<Pixel = T>,
dst_row: &mut [T],
mut src_x: usize,
coeffs_chunk: CoefficientsChunk,
) {
let mut c_buf = [0f64; 2];
let mut dst_f32 = T::components_mut(dst_row);
let mut dst_chunks = dst_f32.chunks_exact_mut(16);
for dst_chunk in &mut dst_chunks {
multiply_components_of_rows::<_, 8>(src_view, src_x, coeffs_chunk, dst_chunk, &mut c_buf);
src_x += 16;
}
dst_f32 = dst_chunks.into_remainder();
dst_chunks = dst_f32.chunks_exact_mut(8);
for dst_chunk in &mut dst_chunks {
multiply_components_of_rows::<_, 4>(src_view, src_x, coeffs_chunk, dst_chunk, &mut c_buf);
src_x += 8;
}
dst_f32 = dst_chunks.into_remainder();
dst_chunks = dst_f32.chunks_exact_mut(4);
if let Some(dst_chunk) = dst_chunks.next() {
multiply_components_of_rows::<_, 2>(src_view, src_x, coeffs_chunk, dst_chunk, &mut c_buf);
src_x += 4;
}
dst_f32 = dst_chunks.into_remainder();
if !dst_f32.is_empty() {
let y_start = coeffs_chunk.start;
let coeffs = coeffs_chunk.values;
native::convolution_by_f32(src_view, dst_f32, src_x, y_start, coeffs);
}
}
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn multiply_components_of_rows<
T: InnerPixel<Component = f32>,
const SUMS_COUNT: usize,
>(
src_view: &impl ImageView<Pixel = T>,
src_x: usize,
coeffs_chunk: CoefficientsChunk,
dst_chunk: &mut [f32],
c_buf: &mut [f64; 2],
) {
let mut sums = [f64x2_splat(0.); SUMS_COUNT];
let y_start = coeffs_chunk.start;
let mut coeffs = coeffs_chunk.values;
let mut y: u32 = 0;
let max_rows = coeffs.len() as u32;
let coeffs_2 = coeffs.chunks_exact(2);
coeffs = coeffs_2.remainder();
for (src_rows, two_coeffs) in src_view.iter_2_rows(y_start, max_rows).zip(coeffs_2) {
let src_rows = src_rows.map(|row| T::components(row).get_unchecked(src_x..));
for (&coeff, src_row) in two_coeffs.iter().zip(src_rows) {
multiply_components_of_row(&mut sums, coeff, src_row);
}
y += 2;
}
if let Some(&coeff) = coeffs.first() {
if let Some(s_row) = src_view.iter_rows(y_start + y).next() {
let src_row = T::components(s_row).get_unchecked(src_x..);
multiply_components_of_row(&mut sums, coeff, src_row);
}
}
let mut dst_ptr = dst_chunk.as_mut_ptr();
for sum in sums {
v128_store(c_buf.as_mut_ptr() as *mut v128, sum);
for &v in c_buf.iter() {
*dst_ptr = v as f32;
dst_ptr = dst_ptr.add(1);
}
}
}
#[inline]
#[target_feature(enable = "simd128")]
unsafe fn multiply_components_of_row<const SUMS_COUNT: usize>(
sums: &mut [v128; SUMS_COUNT],
coeff: f64,
src_row: &[f32],
) {
let coeff_f64x2 = f64x2_splat(coeff);
let mut i = 0;
while i < SUMS_COUNT {
let comp03_f32x4 = wasm32_utils::load_v128(src_row, i * 2);
let comp01_f64x2 = f64x2_promote_low_f32x4(comp03_f32x4);
sums[i] = wasm32_utils::f64x2_mul_add(sums[i], comp01_f64x2, coeff_f64x2);
i += 1;
let comp23_f64x2 = wasm32_utils::f64x2_promote_high_f32x4(comp03_f32x4);
sums[i] = wasm32_utils::f64x2_mul_add(sums[i], comp23_f64x2, coeff_f64x2);
i += 1;
}
}
+27
View File
@@ -98,6 +98,33 @@ pub unsafe fn u8x16_narrow_u16x8(mut a_u8x16: v128, mut b_u8x16: v128) -> v128 {
u8x16_shuffle::<0, 2, 4, 6, 8, 10, 12, 14, 16, 18, 20, 22, 24, 26, 28, 30>(a_u8x16, b_u8x16)
}
/// Promotes the two high `f32` lanes of `v` into a `f64x2`.
///
/// This is the WASM counterpart of `_mm_cvtps_pd(_mm_movehl_ps(v, v))`.
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn f64x2_promote_high_f32x4(v: v128) -> v128 {
f64x2_promote_low_f32x4(i32x4_shuffle::<2, 3, 2, 3>(v, v))
}
/// Computes `acc + a * b` for `f64x2` vectors.
///
/// When the crate is built with the `relaxed-simd` target feature, this uses
/// the possibly fused `f64x2_relaxed_madd` instruction. Otherwise it falls
/// back to a separate multiply and add.
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn f64x2_mul_add(acc: v128, a: v128, b: v128) -> v128 {
#[cfg(target_feature = "relaxed-simd")]
{
f64x2_relaxed_madd(a, b, acc)
}
#[cfg(not(target_feature = "relaxed-simd"))]
{
f64x2_add(acc, f64x2_mul(a, b))
}
}
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn i64x2_mul_lo(a: v128, b: v128) -> v128 {