U8 complete.

This commit is contained in:
Colin
2023-01-23 16:46:42 -05:00
committed by Colin Murphy
parent f0979ec3b9
commit 9b15dcbc71
13 changed files with 1115 additions and 85 deletions
+3 -1
View File
@@ -12,6 +12,8 @@ mod native;
mod neon;
#[cfg(target_arch = "x86_64")]
mod sse4;
#[cfg(target_arch = "wasm32")]
mod wasm32;
impl Convolution for U16x2 {
fn horiz_convolution(
@@ -30,7 +32,7 @@ impl Convolution for U16x2 {
CpuExtensions::Neon => neon::horiz_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => {
native::horiz_convolution(src_image, dst_image, offset, coeffs)
wasm32::horiz_convolution(src_image, dst_image, offset, coeffs)
}
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
}
+6
View File
@@ -12,6 +12,8 @@ mod native;
mod neon;
#[cfg(target_arch = "x86_64")]
mod sse4;
#[cfg(target_arch = "wasm32")]
mod wasm32;
impl Convolution for U8 {
fn horiz_convolution(
@@ -28,6 +30,10 @@ impl Convolution for U8 {
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => neon::horiz_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => {
wasm32::horiz_convolution(src_image, dst_image, offset, coeffs)
}
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
}
}
+161
View File
@@ -0,0 +1,161 @@
use std::arch::wasm32::*;
use crate::convolution::{optimisations, Coefficients};
use crate::pixels::U8;
use crate::wasm32_utils;
use crate::{ImageView, ImageViewMut};
#[inline]
pub(crate) fn horiz_convolution(
src_image: &ImageView<U8>,
dst_image: &mut ImageViewMut<U8>,
offset: u32,
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer16::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_image.height().get();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_image.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
}
}
let mut yy = dst_height - dst_height % 4;
while yy < dst_height {
unsafe {
horiz_convolution_row(
src_image.get_row(yy + offset).unwrap(),
dst_image.get_row_mut(yy).unwrap(),
&coefficients_chunks,
&normalizer,
);
}
yy += 1;
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - length of all rows in src_rows must be equal
/// - length of all rows in dst_rows must be equal
/// - coefficients_chunks.len() == dst_rows.0.len()
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
#[inline]
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U8]; 4],
dst_rows: [&mut &mut [U8]; 4],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
normalizer: &optimisations::Normalizer16,
) {
let zero = i64x2_splat(0);
let initial = 1 << (normalizer.precision() - 1);
let mut buf = [0, 0, 0, 0, initial];
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let coeffs = coeffs_chunk.values;
let mut x = coeffs_chunk.start as usize;
let mut result_i32x4 = [zero, zero, zero, zero];
let coeffs_by_8 = coeffs.chunks_exact(8);
let reminder8 = coeffs_by_8.remainder();
for k in coeffs_by_8 {
let coeffs_i16x8 = v128_load(k.as_ptr() as *const v128);
for i in 0..4 {
let pixels_u8x8 = wasm32_utils::loadl_i64(src_rows[i], x);
let pixels_i16x8 = u16x8_extend_low_u8x16(pixels_u8x8);
result_i32x4[i] =
i32x4_add(result_i32x4[i], i32x4_dot_i16x8(pixels_i16x8, coeffs_i16x8));
}
x += 8;
}
let mut coeffs_by_4 = reminder8.chunks_exact(4);
let reminder4 = coeffs_by_4.remainder();
if let Some(k) = coeffs_by_4.next() {
let coeffs_i16x4 = wasm32_utils::loadl_i64(k, 0);
for i in 0..4 {
let pixels_u8x4 = wasm32_utils::loadl_i32(src_rows[i], x);
let pixels_i16x4 = u16x8_extend_low_u8x16(pixels_u8x4);
result_i32x4[i] =
i32x4_add(result_i32x4[i], i32x4_dot_i16x8(pixels_i16x4, coeffs_i16x4));
}
x += 4;
}
let mut result_i32x4 = result_i32x4.map(|v| {
v128_store(buf.as_mut_ptr() as *mut v128, v);
buf.iter().sum()
});
for &coeff in reminder4 {
let coeff_i32 = coeff as i32;
for i in 0..4 {
result_i32x4[i] += src_rows[i].get_unchecked(x).0.to_owned() as i32 * coeff_i32;
}
x += 1;
}
let result_u8x4 = result_i32x4.map(|v| normalizer.clip(v));
for i in 0..4 {
dst_rows[i].get_unchecked_mut(dst_x).0 = result_u8x4[i];
}
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - bounds.len() == dst_row.len()
/// - coeffs.len() == dst_rows.0.len() * window_size
/// - max(bound.start + bound.size for bound in bounds) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
#[inline]
unsafe fn horiz_convolution_row(
src_row: &[U8],
dst_row: &mut [U8],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
normalizer: &optimisations::Normalizer16,
) {
let zero = i64x2_splat(0);
let initial = 1 << (normalizer.precision() - 1);
let mut buf = [0, 0, 0, 0, initial];
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let coeffs = coeffs_chunk.values;
let mut x = coeffs_chunk.start as usize;
let mut result_i32x4 = zero;
let coeffs_by_8 = coeffs.chunks_exact(8);
let reminder8 = coeffs_by_8.remainder();
for k in coeffs_by_8 {
let coeffs_i16x8 = v128_load(k.as_ptr() as *const v128);
let pixels_u8x8 = wasm32_utils::loadl_i64(src_row, x);
let pixels_i16x8 = u16x8_extend_low_u8x16(pixels_u8x8);
result_i32x4 = i32x4_add(result_i32x4, i32x4_dot_i16x8(pixels_i16x8, coeffs_i16x8));
x += 8;
}
let mut coeffs_by_4 = reminder8.chunks_exact(4);
let reminder4 = coeffs_by_4.remainder();
if let Some(k) = coeffs_by_4.next() {
let coeffs_i16x4 = wasm32_utils::loadl_i64(k, 0);
let pixels_u8x4 = wasm32_utils::loadl_i32(src_row, x);
let pixels_i16x4 = u16x8_extend_low_u8x16(pixels_u8x4);
result_i32x4 = i32x4_add(result_i32x4, i32x4_dot_i16x8(pixels_i16x4, coeffs_i16x4));
x += 4;
}
v128_store(buf.as_mut_ptr() as *mut v128, result_i32x4);
let mut result_i32 = buf.iter().sum();
for &coeff in reminder4 {
let coeff_i32 = coeff as i32;
result_i32 += src_row.get_unchecked(x).0 as i32 * coeff_i32;
x += 1;
}
dst_row.get_unchecked_mut(dst_x).0 = normalizer.clip(result_i32);
}
}
+6
View File
@@ -12,6 +12,8 @@ mod native;
mod neon;
#[cfg(target_arch = "x86_64")]
mod sse4;
#[cfg(target_arch = "wasm32")]
mod wasm32;
impl Convolution for U8x3 {
fn horiz_convolution(
@@ -28,6 +30,10 @@ impl Convolution for U8x3 {
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => neon::horiz_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => {
wasm32::horiz_convolution(src_image, dst_image, offset, coeffs)
}
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
}
}
+290
View File
@@ -0,0 +1,290 @@
use std::arch::wasm32::*;
use std::intrinsics::transmute;
use crate::convolution::{optimisations, Coefficients};
use crate::pixels::U8x3;
use crate::wasm32_utils;
use crate::{ImageView, ImageViewMut};
#[inline]
pub(crate) fn horiz_convolution(
src_image: &ImageView<U8x3>,
dst_image: &mut ImageViewMut<U8x3>,
offset: u32,
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer16::new(coeffs);
let precision = normalizer.precision();
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_image.height().get();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_image.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_8u4x(src_rows, dst_rows, &coefficients_chunks, precision);
}
}
let mut yy = dst_height - dst_height % 4;
while yy < dst_height {
unsafe {
horiz_convolution_8u(
src_image.get_row(yy + offset).unwrap(),
dst_image.get_row_mut(yy).unwrap(),
&coefficients_chunks,
precision,
);
}
yy += 1;
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - length of all rows in src_rows must be equal
/// - length of all rows in dst_rows must be equal
/// - coefficients_chunks.len() == dst_rows.0.len()
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
#[inline]
unsafe fn horiz_convolution_8u4x(
src_rows: [&[U8x3]; 4],
dst_rows: [&mut &mut [U8x3]; 4],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
precision: u8,
) {
let zero = i64x2_splat(0);
let initial = i32x4_splat(1 << (precision - 1));
let src_width = src_rows[0].len();
/*
|R G B | |R G B | |R G B | |R G B | |R G B | |R |
|00 01 02| |03 04 05| |06 07 08| |09 10 11| |12 13 14| |15|
Ignore 12-15 bytes in register and
shuffle other components with converting from u8 into i16:
x: |-1 -1| |-1 -1|
B: |-1 05| |-1 02|
G: |-1 04| |-1 01|
R: |-1 03| |-1 00|
*/
#[rustfmt::skip]
let sh_lo = i8x16(
0, -1, 3, -1, 1, -1, 4, -1, 2, -1, 5, -1, -1, -1, -1, -1
);
/*
x: |-1 -1| |-1 -1|
B: |-1 11| |-1 08|
G: |-1 10| |-1 07|
R: |-1 09| |-1 06|
*/
#[rustfmt::skip]
let sh_hi = i8x16(
6, -1, 9, -1, 7, -1, 10, -1, 8, -1, 11, -1, -1, -1, -1, -1
);
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let x_start = coeffs_chunk.start as usize;
let mut x = x_start;
let mut sss_a = [initial; 4];
let mut coeffs = coeffs_chunk.values;
// Next block of code will be load source pixels by 16 bytes per time.
// We must guarantee what this process will not go beyond
// the one row of image.
// (16 bytes) / (3 bytes per pixel) = 5 whole pixels + 1 byte
let max_x = src_width.saturating_sub(5);
if x < max_x {
let coeffs_by_4 = coeffs.chunks_exact(4);
for k in coeffs_by_4 {
let mmk0 = wasm32_utils::ptr_i16_to_set1_i32(k, 0);
let mmk1 = wasm32_utils::ptr_i16_to_set1_i32(k, 2);
for i in 0..4 {
let source = wasm32_utils::load_v128(src_rows[i], x);
let pix = i8x16_swizzle(source, sh_lo);
let mut sss = sss_a[i];
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk0));
let pix = i8x16_swizzle(source, sh_hi);
sss_a[i] = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk1));
}
x += 4;
if x >= max_x {
break;
}
}
}
// Next block of code will be load source pixels by 8 bytes per time.
// We must guarantee what this process will not go beyond
// the one row of image.
// (8 bytes) / (3 bytes per pixel) = 2 whole pixels + 2 bytes
let max_x = src_width.saturating_sub(2);
if x < max_x {
let coeffs_by_2 = coeffs[x - x_start..].chunks_exact(2);
for k in coeffs_by_2 {
let mmk = wasm32_utils::ptr_i16_to_set1_i32(k, 0);
for i in 0..4 {
let source = wasm32_utils::loadl_i64(src_rows[i], x);
let pix = i8x16_swizzle(source, sh_lo);
sss_a[i] = i32x4_add(sss_a[i], i32x4_dot_i16x8(pix, mmk));
}
x += 2;
if x >= max_x {
break;
}
}
}
coeffs = coeffs.split_at(x - x_start).1;
for &k in coeffs {
let mmk = i32x4_splat(k as i32);
for i in 0..4 {
let pix = wasm32_utils::i32x4_extend_low_ptr_u8x3(src_rows[i], x);
sss_a[i] = i32x4_add(sss_a[i], i32x4_dot_i16x8(pix, mmk));
}
x += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss_a[0] = i32x4_shr(sss_a[0], $imm8);
sss_a[1] = i32x4_shr(sss_a[1], $imm8);
sss_a[2] = i32x4_shr(sss_a[2], $imm8);
sss_a[3] = i32x4_shr(sss_a[3], $imm8);
}};
}
constify_imm8!(precision, call);
for i in 0..4 {
let sss = i16x8_narrow_i32x4(sss_a[i], zero);
let pixel: u32 = transmute(i32x4_extract_lane::<0>(u8x16_narrow_i16x8(sss, zero)));
let bytes = pixel.to_le_bytes();
dst_rows[i].get_unchecked_mut(dst_x).0 = [bytes[0], bytes[1], bytes[2]];
}
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - bounds.len() == dst_row.len()
/// - coeffs.len() == dst_rows.0.len() * window_size
/// - max(bound.start + bound.size for bound in bounds) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
#[inline]
unsafe fn horiz_convolution_8u(
src_row: &[U8x3],
dst_row: &mut [U8x3],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
precision: u8,
) {
#[rustfmt::skip]
let pix_sh1 = i8x16(
0, -1, 3, -1, 1, -1, 4, -1, 2, -1, 5, -1, -1, -1, -1, -1
);
#[rustfmt::skip]
let coef_sh1 = i8x16(
0, 1, 2, 3, 0, 1, 2, 3, 0, 1, 2, 3, 0, 1, 2, 3
);
#[rustfmt::skip]
let pix_sh2 = i8x16(
6, -1, 9, -1, 7, -1, 10, -1, 8, -1, 11, -1, -1, -1, -1, -1
);
#[rustfmt::skip]
let coef_sh2 = i8x16(
4, 5, 6, 7, 4, 5, 6, 7, 4, 5, 6, 7, 4, 5, 6, 7
);
/*
Load 8 bytes from memory into low half of 16-bytes register:
|R G B | |R G B | |R G |
|00 01 02| |03 04 05| |06 07| 08 09 10 11 12 13 14 15
Ignore 06-16 bytes in 16-bytes register and
shuffle other components with converting from u8 into i16:
x: |-1 -1| |-1 -1|
B: |-1 05| |-1 02|
G: |-1 04| |-1 01|
R: |-1 03| |-1 00|
*/
let src_width = src_row.len();
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let x_start = coeffs_chunk.start as usize;
let mut x = x_start;
let mut coeffs = coeffs_chunk.values;
let mut sss = i32x4_splat(1 << (precision - 1));
// Next block of code will be load source pixels by 16 bytes per time.
// We must guarantee what this process will not go beyond
// the one row of image.
// (16 bytes) / (3 bytes per pixel) = 5 whole pixels + 1 bytes
let max_x = src_width.saturating_sub(5);
if x < max_x {
let coeffs_by_4 = coeffs.chunks_exact(4);
for k in coeffs_by_4 {
let ksource = wasm32_utils::loadl_i64(k, 0);
let source = wasm32_utils::load_v128(src_row, x);
let pix = i8x16_swizzle(source, pix_sh1);
let mmk = i8x16_swizzle(ksource, coef_sh1);
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
let pix = i8x16_swizzle(source, pix_sh2);
let mmk = i8x16_swizzle(ksource, coef_sh2);
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
x += 4;
if x >= max_x {
break;
}
}
}
// Next block of code will be load source pixels by 8 bytes per time.
// We must guarantee what this process will not go beyond
// the one row of image.
// (8 bytes) / (3 bytes per pixel) = 2 whole pixels + 2 bytes
let max_x = src_width.saturating_sub(2);
if x < max_x {
let coeffs_by_2 = coeffs[x - x_start..].chunks_exact(2);
for k in coeffs_by_2 {
let mmk = wasm32_utils::ptr_i16_to_set1_i32(k, 0);
let source = wasm32_utils::loadl_i64(src_row, x);
let pix = i8x16_swizzle(source, pix_sh1);
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
x += 2;
if x >= max_x {
break;
}
}
}
coeffs = coeffs.split_at(x - x_start).1;
for &k in coeffs {
let pix = wasm32_utils::i32x4_extend_low_ptr_u8x3(src_row, x);
let mmk = i32x4_splat(k as i32);
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
x += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss = i32x4_shr(sss, $imm8);
}};
}
constify_imm8!(precision, call);
sss = i16x8_narrow_i32x4(sss, sss);
let pixel: u32 = transmute(i32x4_extract_lane::<0>(u8x16_narrow_i16x8(sss, sss)));
let bytes = pixel.to_le_bytes();
dst_row.get_unchecked_mut(dst_x).0 = [bytes[0], bytes[1], bytes[2]];
}
}
+6
View File
@@ -12,6 +12,8 @@ mod native;
mod neon;
#[cfg(target_arch = "x86_64")]
mod sse4;
#[cfg(target_arch = "wasm32")]
mod wasm32;
impl Convolution for U8x4 {
fn horiz_convolution(
@@ -28,6 +30,10 @@ impl Convolution for U8x4 {
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => neon::horiz_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => {
wasm32::horiz_convolution(src_image, dst_image, offset, coeffs)
}
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
}
}
+280
View File
@@ -0,0 +1,280 @@
use std::arch::wasm32::*;
use std::intrinsics::transmute;
use crate::convolution::{optimisations, Coefficients};
use crate::pixels::U8x4;
use crate::wasm32_utils;
use crate::{ImageView, ImageViewMut};
// This code is based on C-implementation from Pillow-SIMD package for Python
// https://github.com/uploadcare/pillow-simd
#[inline]
pub(crate) fn horiz_convolution(
src_image: &ImageView<U8x4>,
dst_image: &mut ImageViewMut<U8x4>,
offset: u32,
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer16::new(coeffs);
let precision = normalizer.precision();
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_image.height().get();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_image.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_8u4x(src_rows, dst_rows, &coefficients_chunks, precision);
}
}
let mut yy = dst_height - dst_height % 4;
while yy < dst_height {
unsafe {
horiz_convolution_8u(
src_image.get_row(yy + offset).unwrap(),
dst_image.get_row_mut(yy).unwrap(),
&coefficients_chunks,
precision,
);
}
yy += 1;
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - length of all rows in src_rows must be equal
/// - length of all rows in dst_rows must be equal
/// - coefficients_chunks.len() == dst_rows.0.len()
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
unsafe fn horiz_convolution_8u4x(
src_rows: [&[U8x4]; 4],
dst_rows: [&mut &mut [U8x4]; 4],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
precision: u8,
) {
let initial = i32x4_splat(1 << (precision - 1));
let mask_lo = i8x16(0, -1, 4, -1, 1, -1, 5, -1, 2, -1, 6, -1, 3, -1, 7, -1);
let mask_hi = i8x16(8, -1, 12, -1, 9, -1, 13, -1, 10, -1, 14, -1, 11, -1, 15, -1);
let mask = i8x16(0, -1, 4, -1, 1, -1, 5, -1, 2, -1, 6, -1, 3, -1, 7, -1);
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
let mut sss0 = initial;
let mut sss1 = initial;
let mut sss2 = initial;
let mut sss3 = initial;
let coeffs = coeffs_chunk.values;
let coeffs_by_4 = coeffs.chunks_exact(4);
let reminder1 = coeffs_by_4.remainder();
for k in coeffs_by_4 {
let mmk_lo = wasm32_utils::ptr_i16_to_set1_i32(k, 0);
let mmk_hi = wasm32_utils::ptr_i16_to_set1_i32(k, 2);
// [8] a3 b3 g3 r3 a2 b2 g2 r2 a1 b1 g1 r1 a0 b0 g0 r0
let mut source = wasm32_utils::load_v128(src_rows[0], x);
// [16] a1 a0 b1 b0 g1 g0 r1 r0
let mut pix = i8x16_swizzle(source, mask_lo);
sss0 = i32x4_add(sss0, i32x4_dot_i16x8(pix, mmk_lo));
// [16] a3 a2 b3 b2 g3 g2 r3 r2
pix = i8x16_swizzle(source, mask_hi);
sss0 = i32x4_add(sss0, i32x4_dot_i16x8(pix, mmk_hi));
source = wasm32_utils::load_v128(src_rows[1], x);
pix = i8x16_swizzle(source, mask_lo);
sss1 = i32x4_add(sss1, i32x4_dot_i16x8(pix, mmk_lo));
pix = i8x16_swizzle(source, mask_hi);
sss1 = i32x4_add(sss1, i32x4_dot_i16x8(pix, mmk_hi));
source = wasm32_utils::load_v128(src_rows[2], x);
pix = i8x16_swizzle(source, mask_lo);
sss2 = i32x4_add(sss2, i32x4_dot_i16x8(pix, mmk_lo));
pix = i8x16_swizzle(source, mask_hi);
sss2 = i32x4_add(sss2, i32x4_dot_i16x8(pix, mmk_hi));
source = wasm32_utils::load_v128(src_rows[3], x);
pix = i8x16_swizzle(source, mask_lo);
sss3 = i32x4_add(sss3, i32x4_dot_i16x8(pix, mmk_lo));
pix = i8x16_swizzle(source, mask_hi);
sss3 = i32x4_add(sss3, i32x4_dot_i16x8(pix, mmk_hi));
x += 4;
}
let coeffs_by_2 = reminder1.chunks_exact(2);
let reminder2 = coeffs_by_2.remainder();
for k in coeffs_by_2 {
// [16] k1 k0 k1 k0 k1 k0 k1 k0
let mmk = wasm32_utils::ptr_i16_to_set1_i32(k, 0);
// [8] x x x x x x x x a1 b1 g1 r1 a0 b0 g0 r0
let mut pix = wasm32_utils::loadl_i64(src_rows[0], x);
// [16] a1 a0 b1 b0 g1 g0 r1 r0
pix = i8x16_swizzle(pix, mask);
sss0 = i32x4_add(sss0, i32x4_dot_i16x8(pix, mmk));
pix = wasm32_utils::loadl_i64(src_rows[1], x);
pix = i8x16_swizzle(pix, mask);
sss1 = i32x4_add(sss1, i32x4_dot_i16x8(pix, mmk));
pix = wasm32_utils::loadl_i64(src_rows[2], x);
pix = i8x16_swizzle(pix, mask);
sss2 = i32x4_add(sss2, i32x4_dot_i16x8(pix, mmk));
pix = wasm32_utils::loadl_i64(src_rows[3], x);
pix = i8x16_swizzle(pix, mask);
sss3 = i32x4_add(sss3, i32x4_dot_i16x8(pix, mmk));
x += 2;
}
if let Some(&k) = reminder2.first() {
// [16] xx k0 xx k0 xx k0 xx k0
let mmk = i32x4_splat(k as i32);
// [16] xx a0 xx b0 xx g0 xx r0
let mut pix = wasm32_utils::i32x4_extend_low_ptr_u8x4(src_rows[0], x);
sss0 = i32x4_add(sss0, i32x4_dot_i16x8(pix, mmk));
pix = wasm32_utils::i32x4_extend_low_ptr_u8x4(src_rows[1], x);
sss1 = i32x4_add(sss1, i32x4_dot_i16x8(pix, mmk));
pix = wasm32_utils::i32x4_extend_low_ptr_u8x4(src_rows[2], x);
sss2 = i32x4_add(sss2, i32x4_dot_i16x8(pix, mmk));
pix = wasm32_utils::i32x4_extend_low_ptr_u8x4(src_rows[3], x);
sss3 = i32x4_add(sss3, i32x4_dot_i16x8(pix, mmk));
}
macro_rules! call {
($imm8:expr) => {{
sss0 = i32x4_shr(sss0, $imm8);
sss1 = i32x4_shr(sss1, $imm8);
sss2 = i32x4_shr(sss2, $imm8);
sss3 = i32x4_shr(sss3, $imm8);
}};
}
constify_imm8!(precision, call);
sss0 = i16x8_narrow_i32x4(sss0, sss0);
sss1 = i16x8_narrow_i32x4(sss1, sss1);
sss2 = i16x8_narrow_i32x4(sss2, sss2);
sss3 = i16x8_narrow_i32x4(sss3, sss3);
*dst_rows[0].get_unchecked_mut(dst_x) =
transmute(i32x4_extract_lane::<0>(u8x16_narrow_i16x8(sss0, sss0)));
*dst_rows[1].get_unchecked_mut(dst_x) =
transmute(i32x4_extract_lane::<0>(u8x16_narrow_i16x8(sss1, sss1)));
*dst_rows[2].get_unchecked_mut(dst_x) =
transmute(i32x4_extract_lane::<0>(u8x16_narrow_i16x8(sss2, sss2)));
*dst_rows[3].get_unchecked_mut(dst_x) =
transmute(i32x4_extract_lane::<0>(u8x16_narrow_i16x8(sss3, sss3)));
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - bounds.len() == dst_row.len()
/// - coefficients_chunks.len() == dst_row.len()
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
unsafe fn horiz_convolution_8u(
src_row: &[U8x4],
dst_row: &mut [U8x4],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
precision: u8,
) {
let initial = i32x4_splat(1 << (precision - 1));
let sh1 = i8x16(0, -1, 8, -1, 1, -1, 9, -1, 2, -1, 10, -1, 3, -1, 11, -1);
let sh2 = i8x16(0, 1, 4, 5, 0, 1, 4, 5, 0, 1, 4, 5, 0, 1, 4, 5);
let sh3 = i8x16(4, -1, 12, -1, 5, -1, 13, -1, 6, -1, 14, -1, 7, -1, 15, -1);
let sh4 = i8x16(2, 3, 6, 7, 2, 3, 6, 7, 2, 3, 6, 7, 2, 3, 6, 7);
let sh5 = i8x16(8, 9, 12, 13, 8, 9, 12, 13, 8, 9, 12, 13, 8, 9, 12, 13);
let sh6 = i8x16(
10, 11, 14, 15, 10, 11, 14, 15, 10, 11, 14, 15, 10, 11, 14, 15,
);
let sh7 = i8x16(0, -1, 4, -1, 1, -1, 5, -1, 2, -1, 6, -1, 3, -1, 7, -1);
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
let mut sss = initial;
let coeffs_by_8 = coeffs_chunk.values.chunks_exact(8);
let reminder8 = coeffs_by_8.remainder();
for k in coeffs_by_8 {
let ksource = wasm32_utils::load_v128(k, 0);
let mut source = wasm32_utils::load_v128(src_row, x);
let mut pix = i8x16_swizzle(source, sh1);
let mut mmk = i8x16_swizzle(ksource, sh2);
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
pix = i8x16_swizzle(source, sh3);
mmk = i8x16_swizzle(ksource, sh4);
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
source = wasm32_utils::load_v128(src_row, x + 4);
pix = i8x16_swizzle(source, sh1);
mmk = i8x16_swizzle(ksource, sh5);
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
pix = i8x16_swizzle(source, sh3);
mmk = i8x16_swizzle(ksource, sh6);
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
x += 8;
}
let coeffs_by_4 = reminder8.chunks_exact(4);
let reminder4 = coeffs_by_4.remainder();
for k in coeffs_by_4 {
let source = wasm32_utils::load_v128(src_row, x);
let ksource = wasm32_utils::loadl_i64(k, 0);
let mut pix = i8x16_swizzle(source, sh1);
let mut mmk = i8x16_swizzle(ksource, sh2);
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
pix = i8x16_swizzle(source, sh3);
mmk = i8x16_swizzle(ksource, sh4);
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
x += 4;
}
let coeffs_by_2 = reminder4.chunks_exact(2);
let reminder2 = coeffs_by_2.remainder();
for k in coeffs_by_2 {
let mmk = wasm32_utils::ptr_i16_to_set1_i32(k, 0);
let source = wasm32_utils::loadl_i64(src_row, x);
let pix = i8x16_swizzle(source, sh7);
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
x += 2
}
if let Some(&k) = reminder2.first() {
let pix = wasm32_utils::i32x4_extend_low_ptr_u8x4(src_row, x);
let mmk = i32x4_splat(k as i32);
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
}
macro_rules! call {
($imm8:expr) => {{
sss = i32x4_shr(sss, $imm8);
}};
}
constify_imm8!(precision, call);
sss = i16x8_narrow_i32x4(sss, sss);
*dst_row.get_unchecked_mut(dst_x) =
transmute(i32x4_extract_lane::<0>(u8x16_narrow_i16x8(sss, sss)));
}
}
-36
View File
@@ -6,8 +6,6 @@ use crate::convolution::{optimisations, Coefficients};
use crate::pixels::PixelExt;
use crate::simd_utils;
use crate::{ImageView, ImageViewMut};
use std::fs;
use std::path::Path;
pub(crate) fn vert_convolution<T: PixelExt<Component = u16>>(
src_image: &ImageView<T>,
@@ -35,9 +33,6 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
coeffs_chunk: CoefficientsI32Chunk,
normalizer: &optimisations::Normalizer32,
) {
let file = "vsse4";
let mut debugout = String::new();
let file_exists = Path::new(file).exists();
let y_start = coeffs_chunk.start;
let coeffs = coeffs_chunk.values;
let max_y = y_start + coeffs.len() as u32;
@@ -116,13 +111,6 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
for x in 0..2 {
for sum in sums {
_mm_storeu_si128((&mut c_buf).as_mut_ptr() as *mut __m128i, sum[x]);
if !file_exists {
debugout += &format!(
"119: {:?} {:?}\n",
_mm_extract_epi64(sum[x], 0),
_mm_extract_epi64(sum[x], 1)
);
}
*dst_ptr = normalizer.clip(c_buf[0]);
dst_ptr = dst_ptr.add(1);
*dst_ptr = normalizer.clip(c_buf[1]);
@@ -177,13 +165,6 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
// sums[i] = _mm_srl_epi64(sums[i] , precision_i64);
// _mm_packus_epi32(sums[i] , sums[i] );
_mm_storeu_si128((&mut c_buf).as_mut_ptr() as *mut __m128i, sum);
if !file_exists {
debugout += &format!(
"176: {:?} {:?}\n",
_mm_extract_epi64(sum, 0),
_mm_extract_epi64(sum, 1)
);
}
*dst_ptr = normalizer.clip(c_buf[0]);
dst_ptr = dst_ptr.add(1);
*dst_ptr = normalizer.clip(c_buf[1]);
@@ -232,34 +213,17 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
let mut dst_ptr = dst_chunk.as_mut_ptr();
_mm_storeu_si128((&mut c_buf).as_mut_ptr() as *mut __m128i, c01);
if !file_exists {
debugout += &format!(
"227: {:?} {:?}\n",
_mm_extract_epi64(c01, 0),
_mm_extract_epi64(c01, 1)
);
}
*dst_ptr = normalizer.clip(c_buf[0]);
dst_ptr = dst_ptr.add(1);
*dst_ptr = normalizer.clip(c_buf[1]);
dst_ptr = dst_ptr.add(1);
_mm_storeu_si128((&mut c_buf).as_mut_ptr() as *mut __m128i, c23);
if !file_exists {
debugout += &format!(
"236: {:?} {:?}\n",
_mm_extract_epi64(c23, 0),
_mm_extract_epi64(c23, 1)
);
}
*dst_ptr = normalizer.clip(c_buf[0]);
dst_ptr = dst_ptr.add(1);
*dst_ptr = normalizer.clip(c_buf[1]);
src_x += 4;
}
if !file_exists {
fs::write(file, debugout).unwrap();
}
dst_u16 = dst_chunks_4.into_remainder();
if !dst_u16.is_empty() {
-36
View File
@@ -6,8 +6,6 @@ use crate::convolution::{optimisations, Coefficients};
use crate::pixels::PixelExt;
use crate::wasm32_utils;
use crate::{ImageView, ImageViewMut};
use std::fs;
use std::path::Path;
pub(crate) fn vert_convolution<T: PixelExt<Component = u16>>(
src_image: &ImageView<T>,
@@ -34,9 +32,6 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
coeffs_chunk: CoefficientsI32Chunk,
normalizer: &optimisations::Normalizer32,
) {
let file = "vwasm32";
let mut debugout = String::new();
let file_exists = Path::new(file).exists();
let y_start = coeffs_chunk.start;
let coeffs = coeffs_chunk.values;
let max_y = y_start + coeffs.len() as u32;
@@ -115,13 +110,6 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
for x in 0..2 {
for sum in sums {
v128_store((&mut c_buf).as_mut_ptr() as *mut v128, sum[x]);
if !file_exists {
debugout += &format!(
"119: {:?} {:?}\n",
i64x2_extract_lane::<0>(sum[x]),
i64x2_extract_lane::<1>(sum[x])
);
}
*dst_ptr = normalizer.clip(c_buf[0]);
dst_ptr = dst_ptr.add(1);
*dst_ptr = normalizer.clip(c_buf[1]);
@@ -176,13 +164,6 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
// sums[i] = _mm_srl_epi64(sums[i] , precision_i64);
// _mm_packus_epi32(sums[i] , sums[i] );
v128_store((&mut c_buf).as_mut_ptr() as *mut v128, sum);
if !file_exists {
debugout += &format!(
"176: {:?} {:?}\n",
i64x2_extract_lane::<0>(sum),
i64x2_extract_lane::<1>(sum)
);
}
*dst_ptr = normalizer.clip(c_buf[0]);
dst_ptr = dst_ptr.add(1);
*dst_ptr = normalizer.clip(c_buf[1]);
@@ -231,34 +212,17 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
let mut dst_ptr = dst_chunk.as_mut_ptr();
v128_store((&mut c_buf).as_mut_ptr() as *mut v128, c01);
if !file_exists {
debugout += &format!(
"227: {:?} {:?}\n",
i64x2_extract_lane::<0>(c01),
i64x2_extract_lane::<1>(c01)
);
}
*dst_ptr = normalizer.clip(c_buf[0]);
dst_ptr = dst_ptr.add(1);
*dst_ptr = normalizer.clip(c_buf[1]);
dst_ptr = dst_ptr.add(1);
v128_store((&mut c_buf).as_mut_ptr() as *mut v128, c23);
if !file_exists {
debugout += &format!(
"236: {:?} {:?}\n",
i64x2_extract_lane::<0>(c23),
i64x2_extract_lane::<1>(c23)
);
}
*dst_ptr = normalizer.clip(c_buf[0]);
dst_ptr = dst_ptr.add(1);
*dst_ptr = normalizer.clip(c_buf[1]);
src_x += 4;
}
if !file_exists {
fs::write(file, debugout).unwrap();
}
dst_u16 = dst_chunks_4.into_remainder();
if !dst_u16.is_empty() {
+4
View File
@@ -10,6 +10,8 @@ pub(crate) mod native;
mod neon;
#[cfg(target_arch = "x86_64")]
pub(crate) mod sse4;
#[cfg(target_arch = "wasm32")]
pub(crate) mod wasm32;
pub(crate) fn vert_convolution_u8<T: PixelExt<Component = u8>>(
src_image: &ImageView<T>,
@@ -29,6 +31,8 @@ pub(crate) fn vert_convolution_u8<T: PixelExt<Component = u8>>(
CpuExtensions::Sse4_1 => sse4::vert_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => neon::vert_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => wasm32::vert_convolution(src_image, dst_image, offset, coeffs),
_ => native::vert_convolution(src_image, dst_image, offset, coeffs),
}
}
+292
View File
@@ -0,0 +1,292 @@
use std::arch::wasm32::*;
use crate::convolution::vertical_u8::native;
use crate::convolution::{optimisations, Coefficients};
use crate::pixels::PixelExt;
use crate::wasm32_utils;
use crate::{ImageView, ImageViewMut};
#[inline]
pub(crate) fn vert_convolution<T: PixelExt<Component = u8>>(
src_image: &ImageView<T>,
dst_image: &mut ImageViewMut<T>,
offset: u32,
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer16::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let src_x = offset as usize * T::count_of_components();
let dst_rows = dst_image.iter_rows_mut();
for (dst_row, coeffs_chunk) in dst_rows.zip(coefficients_chunks) {
unsafe {
vert_convolution_into_one_row_u8(src_image, dst_row, src_x, coeffs_chunk, &normalizer);
}
}
}
pub(crate) unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
src_img: &ImageView<T>,
dst_row: &mut [T],
mut src_x: usize,
coeffs_chunk: optimisations::CoefficientsI16Chunk,
normalizer: &optimisations::Normalizer16,
) {
let y_start = coeffs_chunk.start;
let coeffs = coeffs_chunk.values;
let max_y = y_start + coeffs.len() as u32;
let precision = normalizer.precision();
let mut dst_u8 = T::components_mut(dst_row);
let initial = i32x4_splat(1 << (precision - 1));
let mut dst_chunks_32 = dst_u8.chunks_exact_mut(32);
for dst_chunk in &mut dst_chunks_32 {
let mut sss0 = initial;
let mut sss1 = initial;
let mut sss2 = initial;
let mut sss3 = initial;
let mut sss4 = initial;
let mut sss5 = initial;
let mut sss6 = initial;
let mut sss7 = initial;
let mut y: u32 = 0;
for src_rows in src_img.iter_2_rows(y_start, max_y) {
let components1 = T::components(src_rows[0]);
let components2 = T::components(src_rows[1]);
// Load two coefficients at once
let mmk = wasm32_utils::ptr_i16_to_set1_i32(coeffs, y as usize);
let source1 = wasm32_utils::load_v128(components1, src_x); // top line
let source2 = wasm32_utils::load_v128(components2, src_x); // bottom line
let source = i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(
source1, source2,
);
let pix = i16x8_extend_low_u8x16(source);
sss0 = i32x4_add(sss0, i32x4_dot_i16x8(pix, mmk));
let pix = i16x8_extend_high_u8x16(source);
sss1 = i32x4_add(sss1, i32x4_dot_i16x8(pix, mmk));
let source =
i8x16_shuffle::<8, 24, 9, 25, 10, 26, 11, 27, 12, 28, 13, 29, 14, 30, 15, 31>(
source1, source2,
);
let pix = i16x8_extend_low_u8x16(source);
sss2 = i32x4_add(sss2, i32x4_dot_i16x8(pix, mmk));
let pix = i16x8_extend_high_u8x16(source);
sss3 = i32x4_add(sss3, i32x4_dot_i16x8(pix, mmk));
let source1 = wasm32_utils::load_v128(components1, src_x + 16); // top line
let source2 = wasm32_utils::load_v128(components2, src_x + 16); // bottom line
let source = i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(
source1, source2,
);
let pix = i16x8_extend_low_u8x16(source);
sss4 = i32x4_add(sss4, i32x4_dot_i16x8(pix, mmk));
let pix = i16x8_extend_high_u8x16(source);
sss5 = i32x4_add(sss5, i32x4_dot_i16x8(pix, mmk));
let source =
i8x16_shuffle::<8, 24, 9, 25, 10, 26, 11, 27, 12, 28, 13, 29, 14, 30, 15, 31>(
source1, source2,
);
let pix = i16x8_extend_low_u8x16(source);
sss6 = i32x4_add(sss6, i32x4_dot_i16x8(pix, mmk));
let pix = i16x8_extend_high_u8x16(source);
sss7 = i32x4_add(sss7, i32x4_dot_i16x8(pix, mmk));
y += 2;
}
if let Some(&k) = coeffs.get(y as usize) {
let s_row = src_img.get_row(y_start + y).unwrap();
let components = T::components(s_row);
let mmk = i32x4_splat(k as i32);
let source1 = wasm32_utils::load_v128(components, src_x); // top line
let source = i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(
source1,
i64x2_splat(0),
);
let pix = i16x8_extend_low_u8x16(source);
sss0 = i32x4_add(sss0, i32x4_dot_i16x8(pix, mmk));
let pix = i16x8_extend_high_u8x16(source);
sss1 = i32x4_add(sss1, i32x4_dot_i16x8(pix, mmk));
let source = i16x8_extend_high_u8x16(source1);
let pix = i16x8_extend_low_u8x16(source);
sss2 = i32x4_add(sss2, i32x4_dot_i16x8(pix, mmk));
let pix = i16x8_extend_high_u8x16(source);
sss3 = i32x4_add(sss3, i32x4_dot_i16x8(pix, mmk));
let source1 = wasm32_utils::load_v128(components, src_x + 16); // top line
let source = i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(
source1,
i64x2_splat(0),
);
let pix = i16x8_extend_low_u8x16(source);
sss4 = i32x4_add(sss4, i32x4_dot_i16x8(pix, mmk));
let pix = i16x8_extend_high_u8x16(source);
sss5 = i32x4_add(sss5, i32x4_dot_i16x8(pix, mmk));
let source = i16x8_extend_high_u8x16(source1);
let pix = i16x8_extend_low_u8x16(source);
sss6 = i32x4_add(sss6, i32x4_dot_i16x8(pix, mmk));
let pix = i16x8_extend_high_u8x16(source);
sss7 = i32x4_add(sss7, i32x4_dot_i16x8(pix, mmk));
}
macro_rules! call {
($imm8:expr) => {{
sss0 = i32x4_shr(sss0, $imm8);
sss1 = i32x4_shr(sss1, $imm8);
sss2 = i32x4_shr(sss2, $imm8);
sss3 = i32x4_shr(sss3, $imm8);
sss4 = i32x4_shr(sss4, $imm8);
sss5 = i32x4_shr(sss5, $imm8);
sss6 = i32x4_shr(sss6, $imm8);
sss7 = i32x4_shr(sss7, $imm8);
}};
}
constify_imm8!(precision, call);
sss0 = i16x8_narrow_i32x4(sss0, sss1);
sss2 = i16x8_narrow_i32x4(sss2, sss3);
sss0 = u8x16_narrow_i16x8(sss0, sss2);
let dst_ptr = dst_chunk.as_mut_ptr() as *mut v128;
v128_store(dst_ptr, sss0);
sss4 = i16x8_narrow_i32x4(sss4, sss5);
sss6 = i16x8_narrow_i32x4(sss6, sss7);
sss4 = u8x16_narrow_i16x8(sss4, sss6);
let dst_ptr = dst_ptr.add(1);
v128_store(dst_ptr, sss4);
src_x += 32;
}
dst_u8 = dst_chunks_32.into_remainder();
let mut dst_chunks_8 = dst_u8.chunks_exact_mut(8);
for dst_chunk in &mut dst_chunks_8 {
let mut sss0 = initial; // left row
let mut sss1 = initial; // right row
let mut y: u32 = 0;
for src_rows in src_img.iter_2_rows(y_start, max_y) {
let components1 = T::components(src_rows[0]);
let components2 = T::components(src_rows[1]);
// Load two coefficients at once
let mmk = wasm32_utils::ptr_i16_to_set1_i32(coeffs, y as usize);
let source1 = wasm32_utils::loadl_i64(components1, src_x); // top line
let source2 = wasm32_utils::loadl_i64(components2, src_x); // bottom line
let source = i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(
source1, source2,
);
let pix = i16x8_extend_low_u8x16(source);
sss0 = i32x4_add(sss0, i32x4_dot_i16x8(pix, mmk));
let pix = i16x8_extend_high_u8x16(source);
sss1 = i32x4_add(sss1, i32x4_dot_i16x8(pix, mmk));
y += 2;
}
if let Some(&k) = coeffs.get(y as usize) {
let s_row = src_img.get_row(y_start + y).unwrap();
let components = T::components(s_row);
let mmk = i32x4_splat(k as i32);
let source1 = wasm32_utils::loadl_i64(components, src_x); // top line
let source = i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(
source1,
i64x2_splat(0),
);
let pix = i16x8_extend_low_u8x16(source);
sss0 = i32x4_add(sss0, i32x4_dot_i16x8(pix, mmk));
let pix = i16x8_extend_high_u8x16(source);
sss1 = i32x4_add(sss1, i32x4_dot_i16x8(pix, mmk));
}
macro_rules! call {
($imm8:expr) => {{
sss0 = i32x4_shr(sss0, $imm8);
sss1 = i32x4_shr(sss1, $imm8);
}};
}
constify_imm8!(precision, call);
sss0 = i16x8_narrow_i32x4(sss0, sss1);
sss0 = u8x16_narrow_i16x8(sss0, sss0);
let dst_ptr = dst_chunk.as_mut_ptr() as *mut [i64; 2];
(*dst_ptr)[0] = i64x2_extract_lane::<0>(sss0);
src_x += 8;
}
dst_u8 = dst_chunks_8.into_remainder();
let mut dst_chunks_4 = dst_u8.chunks_exact_mut(4);
if let Some(dst_chunk) = dst_chunks_4.next() {
let mut sss = initial;
let mut y: u32 = 0;
for src_rows in src_img.iter_2_rows(y_start, max_y) {
let components1 = T::components(src_rows[0]);
let components2 = T::components(src_rows[1]);
// Load two coefficients at once
let mmk = wasm32_utils::ptr_i16_to_set1_i32(coeffs, y as usize);
let source1 = wasm32_utils::i32_v128_from_u8(components1, src_x); // top line
let source2 = wasm32_utils::i32_v128_from_u8(components2, src_x); // bottom line
let source = i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(
source1, source2,
);
let pix = i16x8_extend_low_u8x16(source);
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
y += 2;
}
if let Some(&k) = coeffs.get(y as usize) {
let s_row = src_img.get_row(y_start + y).unwrap();
let components = T::components(s_row);
let pix = wasm32_utils::i32x4_extend_low_ptr_u8(components, src_x);
let mmk = i32x4_splat(k as i32);
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
}
macro_rules! call {
($imm8:expr) => {{
sss = i32x4_shr(sss, $imm8);
}};
}
constify_imm8!(precision, call);
sss = i16x8_narrow_i32x4(sss, sss);
let dst_ptr = dst_chunk.as_mut_ptr() as *mut i32;
*dst_ptr = i32x4_extract_lane::<0>(u8x16_narrow_i16x8(sss, sss));
src_x += 4;
}
dst_u8 = dst_chunks_4.into_remainder();
if !dst_u8.is_empty() {
native::convolution_by_u8(
src_img,
normalizer,
1 << (precision - 1),
dst_u8,
src_x,
y_start,
coeffs,
);
}
}
+35 -12
View File
@@ -1,4 +1,6 @@
use crate::pixels::{U8x3, U8x4};
use std::arch::wasm32::*;
use std::intrinsics::transmute;
#[inline(always)]
pub unsafe fn load_v128<T>(buf: &[T], index: usize) -> v128 {
@@ -7,26 +9,23 @@ pub unsafe fn load_v128<T>(buf: &[T], index: usize) -> v128 {
#[inline(always)]
pub unsafe fn loadl_i64<T>(buf: &[T], index: usize) -> v128 {
i64x2(buf.get_unchecked(index..).as_ptr() as i64, 0)
let v = v128_load(buf.get_unchecked(index..).as_ptr() as *const v128);
let k = i8x16(0, 1, 2, 3, 4, 5, 6, 7, -1, -1, -1, -1, -1, -1, -1, -1);
i8x16_swizzle(v, k)
}
#[inline(always)]
pub unsafe fn loadl_i32<T>(buf: &[T], index: usize) -> v128 {
i32x4(buf.get_unchecked(index..).as_ptr() as i32, 0, 0, 0)
let v = v128_load(buf.get_unchecked(index..).as_ptr() as *const v128);
let k = i8x16(0, 1, 2, 3, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1);
i8x16_swizzle(v, k)
}
#[inline(always)]
pub unsafe fn loadl_i16<T>(buf: &[T], index: usize) -> v128 {
i16x8(
buf.get_unchecked(index..).as_ptr() as i16,
0,
0,
0,
0,
0,
0,
0,
)
let v = v128_load(buf.get_unchecked(index..).as_ptr() as *const v128);
let k = i8x16(0, 1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1);
i8x16_swizzle(v, k)
}
#[inline(always)]
@@ -38,3 +37,27 @@ pub unsafe fn ptr_i16_to_set1_i64(buf: &[i16], index: usize) -> v128 {
pub unsafe fn ptr_i16_to_set1_i32(buf: &[i16], index: usize) -> v128 {
i32x4_splat(*(buf.get_unchecked(index..).as_ptr() as *const i32))
}
#[inline(always)]
pub unsafe fn i32x4_extend_low_ptr_u8(buf: &[u8], index: usize) -> v128 {
let ptr = buf.get_unchecked(index..).as_ptr() as *const v128;
u32x4_extend_low_u16x8(i16x8_extend_low_u8x16(v128_load(ptr)))
}
#[inline(always)]
pub unsafe fn i32x4_extend_low_ptr_u8x4(buf: &[U8x4], index: usize) -> v128 {
let v: i32 = transmute(buf.get_unchecked(index).0);
u32x4_extend_low_u16x8(i16x8_extend_low_u8x16(i32x4(v, 0, 0, 0)))
}
#[inline(always)]
pub unsafe fn i32x4_extend_low_ptr_u8x3(buf: &[U8x3], index: usize) -> v128 {
let pixel = buf.get_unchecked(index).0;
i32x4(pixel[0] as i32, pixel[1] as i32, pixel[2] as i32, 0)
}
#[inline(always)]
pub unsafe fn i32_v128_from_u8(buf: &[u8], index: usize) -> v128 {
let ptr = buf.get_unchecked(index..).as_ptr() as *const i32;
i32x4(*ptr, 0, 0, 0)
}
+32
View File
@@ -376,6 +376,10 @@ fn downscale_u8() {
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
}
for cpu_extensions in cpu_extensions_vec {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
@@ -400,6 +404,10 @@ fn upscale_u8() {
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
@@ -424,6 +432,10 @@ fn downscale_u8x2() {
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
}
for cpu_extensions in cpu_extensions_vec {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
@@ -452,6 +464,10 @@ fn upscale_u8x2() {
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
@@ -480,6 +496,10 @@ fn downscale_u8x3() {
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
}
for cpu_extensions in cpu_extensions_vec {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
@@ -508,6 +528,10 @@ fn upscale_u8x3() {
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
@@ -536,6 +560,10 @@ fn downscale_u8x4() {
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
}
for cpu_extensions in cpu_extensions_vec {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
@@ -570,6 +598,10 @@ fn upscale_u8x4() {
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),