Fixed code for Wasm32

This commit is contained in:
Kirill Kuzminykh
2024-04-18 00:41:39 +03:00
parent d90d235fb2
commit 4086b46d0b
15 changed files with 290 additions and 323 deletions
+14 -6
View File
@@ -1,14 +1,16 @@
# Preparation
Install system libraries:
- libvips-dev (used in benchmarks)
Install additional toolchains:
- Arm64:
```shell
rustup target add aarch64-unknown-linux-gnu
```
- Wasm32:
- Wasm32:
```shell
rustup target add wasm32-wasi
```
@@ -16,45 +18,51 @@ Install additional toolchains:
# Tests
Run tests without saving result images as files in `./data` directory:
Run tests with saving result images as files in `./data` directory:
```shell
DONT_SAVE_RESULT=1 cargo test
SAVE_RESULT=1 cargo test
```
# Benchmarks
Run benchmarks to compare with other crates for image resizing and write results into
report files, such as `./benchmarks-x86_64.md`:
```shell
WRITE_COMPARE_RESULT=1 cargo bench -- Compare
```
# Wasm32
Specify build target in `.cargo/config.toml` file.
```toml
[build]
target = "wasm32-wasi"
```
Run tests:
```shell
CARGO_TARGET_WASM32_WASI_RUNNER="wasmtime --dir=. --" cargo test
```
Run tests without saving result images as files in `./data` directory:
Run tests with saving result images as files in `./data` directory:
```shell
CARGO_TARGET_WASM32_WASI_RUNNER="wasmtime --dir=. --env DONT_SAVE_RESULT=1 --" cargo test
CARGO_TARGET_WASM32_WASI_RUNNER="wasmtime --dir=. --env SAVE_RESULT=1 --" cargo test
```
Run a specific benchmark in `quick` mode:
```shell
CARGO_TARGET_WASM32_WASI_RUNNER="wasmtime --dir=. --" cargo bench --bench bench_resize -- --color=always --quick
```
Run benchmarks to compare with other crates for image resizing and write results into
report files, such as `./benchmarks-x86_64.md`:
```shell
CARGO_TARGET_WASM32_WASI_RUNNER="wasmtime --dir=. --env WRITE_COMPARE_RESULT=1 --" cargo bench -- --color=always Compare
```
+12 -12
View File
@@ -7,19 +7,19 @@ use crate::{ImageView, ImageViewMut};
use super::native;
pub(crate) unsafe fn multiply_alpha(
src_image: &ImageView<U16x2>,
dst_image: &mut ImageViewMut<U16x2>,
src_view: &impl ImageView<Pixel = U16x2>,
dst_view: &mut impl ImageViewMut<Pixel = U16x2>,
) {
let src_rows = src_image.iter_rows(0);
let dst_rows = dst_image.iter_rows_mut();
let src_rows = src_view.iter_rows(0);
let dst_rows = dst_view.iter_rows_mut(0);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
multiply_alpha_row(src_row, dst_row);
}
}
pub(crate) unsafe fn multiply_alpha_inplace(image: &mut ImageViewMut<U16x2>) {
for row in image.iter_rows_mut() {
pub(crate) unsafe fn multiply_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = U16x2>) {
for row in image_view.iter_rows_mut(0) {
multiply_alpha_row_inplace(row);
}
}
@@ -93,19 +93,19 @@ unsafe fn multiply_alpha_4_pixels(pixels: v128) -> v128 {
// Divide
pub(crate) unsafe fn divide_alpha(
src_image: &ImageView<U16x2>,
dst_image: &mut ImageViewMut<U16x2>,
src_view: &impl ImageView<Pixel = U16x2>,
dst_view: &mut impl ImageViewMut<Pixel = U16x2>,
) {
let src_rows = src_image.iter_rows(0);
let dst_rows = dst_image.iter_rows_mut();
let src_rows = src_view.iter_rows(0);
let dst_rows = dst_view.iter_rows_mut(0);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
divide_alpha_row(src_row, dst_row);
}
}
pub(crate) unsafe fn divide_alpha_inplace(image: &mut ImageViewMut<U16x2>) {
for row in image.iter_rows_mut() {
pub(crate) unsafe fn divide_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = U16x2>) {
for row in image_view.iter_rows_mut(0) {
divide_alpha_row_inplace(row);
}
}
+12 -12
View File
@@ -7,19 +7,19 @@ use crate::{ImageView, ImageViewMut};
use super::native;
pub(crate) unsafe fn multiply_alpha(
src_image: &ImageView<U16x4>,
dst_image: &mut ImageViewMut<U16x4>,
src_view: &impl ImageView<Pixel = U16x4>,
dst_view: &mut impl ImageViewMut<Pixel = U16x4>,
) {
let src_rows = src_image.iter_rows(0);
let dst_rows = dst_image.iter_rows_mut();
let src_rows = src_view.iter_rows(0);
let dst_rows = dst_view.iter_rows_mut(0);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
multiply_alpha_row(src_row, dst_row);
}
}
pub(crate) unsafe fn multiply_alpha_inplace(image: &mut ImageViewMut<U16x4>) {
for row in image.iter_rows_mut() {
pub(crate) unsafe fn multiply_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = U16x4>) {
for row in image_view.iter_rows_mut(0) {
multiply_alpha_row_inplace(row);
}
}
@@ -105,19 +105,19 @@ unsafe fn multiply_alpha_2_pixels(pixels: v128) -> v128 {
// Divide
pub(crate) unsafe fn divide_alpha(
src_image: &ImageView<U16x4>,
dst_image: &mut ImageViewMut<U16x4>,
src_view: &impl ImageView<Pixel = U16x4>,
dst_view: &mut impl ImageViewMut<Pixel = U16x4>,
) {
let src_rows = src_image.iter_rows(0);
let dst_rows = dst_image.iter_rows_mut();
let src_rows = src_view.iter_rows(0);
let dst_rows = dst_view.iter_rows_mut(0);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
divide_alpha_row(src_row, dst_row);
}
}
pub(crate) unsafe fn divide_alpha_inplace(image: &mut ImageViewMut<U16x4>) {
for row in image.iter_rows_mut() {
pub(crate) unsafe fn divide_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = U16x4>) {
for row in image_view.iter_rows_mut(0) {
divide_alpha_row_inplace(row);
}
}
+18 -15
View File
@@ -6,19 +6,19 @@ use crate::{ImageView, ImageViewMut};
use super::native;
pub(crate) unsafe fn multiply_alpha(
src_image: &ImageView<U8x2>,
dst_image: &mut ImageViewMut<U8x2>,
src_view: &impl ImageView<Pixel = U8x2>,
dst_view: &mut impl ImageViewMut<Pixel = U8x2>,
) {
let src_rows = src_image.iter_rows(0);
let dst_rows = dst_image.iter_rows_mut();
let src_rows = src_view.iter_rows(0);
let dst_rows = dst_view.iter_rows_mut(0);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
multiply_alpha_row(src_row, dst_row);
}
}
pub(crate) unsafe fn multiply_alpha_inplace(image: &mut ImageViewMut<U8x2>) {
for row in image.iter_rows_mut() {
pub(crate) unsafe fn multiply_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = U8x2>) {
for row in image_view.iter_rows_mut(0) {
multiply_alpha_row_inplace(row);
}
}
@@ -91,17 +91,20 @@ unsafe fn multiplies_alpha_8_pixels(pixels: v128) -> v128 {
// Divide
pub(crate) unsafe fn divide_alpha(src_image: &ImageView<U8x2>, dst_image: &mut ImageViewMut<U8x2>) {
let src_rows = src_image.iter_rows(0);
let dst_rows = dst_image.iter_rows_mut();
pub(crate) unsafe fn divide_alpha(
src_view: &impl ImageView<Pixel = U8x2>,
dst_view: &mut impl ImageViewMut<Pixel = U8x2>,
) {
let src_rows = src_view.iter_rows(0);
let dst_rows = dst_view.iter_rows_mut(0);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
divide_alpha_row(src_row, dst_row);
}
}
pub(crate) unsafe fn divide_alpha_inplace(image: &mut ImageViewMut<U8x2>) {
for row in image.iter_rows_mut() {
pub(crate) unsafe fn divide_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = U8x2>) {
for row in image_view.iter_rows_mut(0) {
divide_alpha_row_inplace(row);
}
}
@@ -122,13 +125,13 @@ unsafe fn divide_alpha_row(src_row: &[U8x2], dst_row: &mut [U8x2]) {
if !src_remainder.is_empty() {
let dst_reminder = dst_chunks.into_remainder();
let mut src_pixels = [U8x2::new(0); 8];
let mut src_pixels = [U8x2::new([0, 0]); 8];
src_pixels
.iter_mut()
.zip(src_remainder)
.for_each(|(d, s)| *d = *s);
let mut dst_pixels = [U8x2::new(0); 8];
let mut dst_pixels = [U8x2::new([0; 2]); 8];
let mut pixels = v128_load(src_pixels.as_ptr() as *const v128);
pixels = divide_alpha_8_pixels(pixels);
v128_store(dst_pixels.as_mut_ptr() as *mut v128, pixels);
@@ -153,13 +156,13 @@ unsafe fn divide_alpha_row_inplace(row: &mut [U8x2]) {
let reminder = chunks.into_remainder();
if !reminder.is_empty() {
let mut src_pixels = [U8x2::new(0); 8];
let mut src_pixels = [U8x2::new([0; 2]); 8];
src_pixels
.iter_mut()
.zip(reminder.iter())
.for_each(|(d, s)| *d = *s);
let mut dst_pixels = [U8x2::new(0); 8];
let mut dst_pixels = [U8x2::new([0; 2]); 8];
let mut pixels = v128_load(src_pixels.as_ptr() as *const v128);
pixels = divide_alpha_8_pixels(pixels);
v128_store(dst_pixels.as_mut_ptr() as *mut v128, pixels);
+17 -15
View File
@@ -7,19 +7,19 @@ use crate::{ImageView, ImageViewMut};
use super::native;
pub(crate) unsafe fn multiply_alpha(
src_image: &ImageView<U8x4>,
dst_image: &mut ImageViewMut<U8x4>,
src_view: &impl ImageView<Pixel = U8x4>,
dst_view: &mut impl ImageViewMut<Pixel = U8x4>,
) {
let src_rows = src_image.iter_rows(0);
let dst_rows = dst_image.iter_rows_mut();
let src_rows = src_view.iter_rows(0);
let dst_rows = dst_view.iter_rows_mut(0);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
multiply_alpha_row(src_row, dst_row);
}
}
pub(crate) unsafe fn multiply_alpha_inplace(image: &mut ImageViewMut<U8x4>) {
for row in image.iter_rows_mut() {
pub(crate) unsafe fn multiply_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = U8x4>) {
for row in image_view.iter_rows_mut(0) {
multiply_alpha_row_inplace(row);
}
}
@@ -64,13 +64,12 @@ unsafe fn multiply_alpha_row_inplace(row: &mut [U8x4]) {
#[inline]
#[target_feature(enable = "simd128")]
unsafe fn multiply_alpha_4_pixels(pixels: v128) -> v128 {
let half = u16x8_splat(128);
let max_alpha = u32x4_splat(0xff000000);
const FACTOR_MASK: v128 = i8x16(3, 3, 3, 3, 7, 7, 7, 7, 11, 11, 11, 11, 15, 15, 15, 15);
let factor_pixels = u8x16_swizzle(pixels, FACTOR_MASK);
let factor_pixels = v128_or(factor_pixels, max_alpha);
let max_alpha = u32x4_splat(0xff000000);
let factor_pixels = v128_or(u8x16_swizzle(pixels, FACTOR_MASK), max_alpha);
let half = u16x8_splat(128);
let src_u16_lo = u16x8_extend_low_u8x16(pixels);
let factors = u16x8_extend_low_u8x16(factor_pixels);
let mut dst_u16_lo = u16x8_add(u16x8_mul(src_u16_lo, factors), half);
@@ -88,16 +87,19 @@ unsafe fn multiply_alpha_4_pixels(pixels: v128) -> v128 {
// Divide
pub(crate) unsafe fn divide_alpha(src_image: &ImageView<U8x4>, dst_image: &mut ImageViewMut<U8x4>) {
let src_rows = src_image.iter_rows(0);
let dst_rows = dst_image.iter_rows_mut();
pub(crate) unsafe fn divide_alpha(
src_view: &impl ImageView<Pixel = U8x4>,
dst_view: &mut impl ImageViewMut<Pixel = U8x4>,
) {
let src_rows = src_view.iter_rows(0);
let dst_rows = dst_view.iter_rows_mut(0);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
divide_alpha_row(src_row, dst_row);
}
}
pub(crate) unsafe fn divide_alpha_inplace(image: &mut ImageViewMut<U8x4>) {
for row in image.iter_rows_mut() {
pub(crate) unsafe fn divide_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = U8x4>) {
for row in image_view.iter_rows_mut(0) {
divide_alpha_row_inplace(row);
}
}
+11 -15
View File
@@ -7,34 +7,30 @@ use crate::{ImageView, ImageViewMut};
#[inline]
pub(crate) fn horiz_convolution(
src_image: &ImageView<U16>,
dst_image: &mut ImageViewMut<U16>,
src_view: &impl ImageView<Pixel = U16>,
dst_view: &mut impl ImageViewMut<Pixel = U16>,
offset: u32,
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer32::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_image.height().get();
let dst_height = dst_view.height();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_image.iter_4_rows_mut();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
}
}
let mut yy = dst_height - dst_height % 4;
while yy < dst_height {
let yy = dst_height - dst_height % 4;
let src_rows = src_view.iter_rows(yy + offset);
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_one_row(
src_image.get_row(yy + offset).unwrap(),
dst_image.get_row_mut(yy).unwrap(),
&coefficients_chunks,
&normalizer,
);
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
}
yy += 1;
}
}
@@ -46,7 +42,7 @@ pub(crate) fn horiz_convolution(
#[target_feature(enable = "simd128")]
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U16]; 4],
dst_rows: [&mut &mut [U16]; 4],
dst_rows: [&mut [U16]; 4],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
normalizer: &optimisations::Normalizer32,
) {
+11 -15
View File
@@ -7,34 +7,30 @@ use crate::{ImageView, ImageViewMut};
#[inline]
pub(crate) fn horiz_convolution(
src_image: &ImageView<U16x2>,
dst_image: &mut ImageViewMut<U16x2>,
src_view: &impl ImageView<Pixel = U16x2>,
dst_view: &mut impl ImageViewMut<Pixel = U16x2>,
offset: u32,
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer32::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_image.height().get();
let dst_height = dst_view.height();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_image.iter_4_rows_mut();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
}
}
let mut yy = dst_height - dst_height % 4;
while yy < dst_height {
let yy = dst_height - dst_height % 4;
let src_rows = src_view.iter_rows(yy + offset);
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_one_row(
src_image.get_row(yy + offset).unwrap(),
dst_image.get_row_mut(yy).unwrap(),
&coefficients_chunks,
&normalizer,
);
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
}
yy += 1;
}
}
@@ -47,7 +43,7 @@ pub(crate) fn horiz_convolution(
#[target_feature(enable = "simd128")]
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U16x2]; 4],
dst_rows: [&mut &mut [U16x2]; 4],
dst_rows: [&mut [U16x2]; 4],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
normalizer: &optimisations::Normalizer32,
) {
+16 -21
View File
@@ -1,6 +1,5 @@
use std::arch::wasm32::*;
use crate::convolution::optimisations::CoefficientsI32Chunk;
use crate::convolution::{optimisations, Coefficients};
use crate::pixels::U16x3;
use crate::wasm32_utils;
@@ -8,34 +7,30 @@ use crate::{ImageView, ImageViewMut};
#[inline]
pub(crate) fn horiz_convolution(
src_image: &ImageView<U16x3>,
dst_image: &mut ImageViewMut<U16x3>,
src_view: &impl ImageView<Pixel = U16x3>,
dst_view: &mut impl ImageViewMut<Pixel = U16x3>,
offset: u32,
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer32::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_image.height().get();
let dst_height = dst_view.height();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_image.iter_4_rows_mut();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_8u4x(src_rows, dst_rows, &coefficients_chunks, &normalizer);
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
}
}
let mut yy = dst_height - dst_height % 4;
while yy < dst_height {
let yy = dst_height - dst_height % 4;
let src_rows = src_view.iter_rows(yy + offset);
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_8u(
src_image.get_row(yy + offset).unwrap(),
dst_image.get_row_mut(yy).unwrap(),
&coefficients_chunks,
&normalizer,
);
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
}
yy += 1;
}
}
@@ -46,10 +41,10 @@ pub(crate) fn horiz_convolution(
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "simd128")]
unsafe fn horiz_convolution_8u4x(
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U16x3]; 4],
dst_rows: [&mut &mut [U16x3]; 4],
coefficients_chunks: &[CoefficientsI32Chunk],
dst_rows: [&mut [U16x3]; 4],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
normalizer: &optimisations::Normalizer32,
) {
const ZERO: v128 = i64x2(0, 0);
@@ -149,10 +144,10 @@ unsafe fn horiz_convolution_8u4x(
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "simd128")]
unsafe fn horiz_convolution_8u(
unsafe fn horiz_convolution_one_row(
src_row: &[U16x3],
dst_row: &mut [U16x3],
coefficients_chunks: &[CoefficientsI32Chunk],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
normalizer: &optimisations::Normalizer32,
) {
let precision = normalizer.precision();
+11 -15
View File
@@ -7,34 +7,30 @@ use crate::{ImageView, ImageViewMut};
#[inline]
pub(crate) fn horiz_convolution(
src_image: &ImageView<U16x4>,
dst_image: &mut ImageViewMut<U16x4>,
src_view: &impl ImageView<Pixel = U16x4>,
dst_view: &mut impl ImageViewMut<Pixel = U16x4>,
offset: u32,
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer32::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_image.height().get();
let dst_height = dst_view.height();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_image.iter_4_rows_mut();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
}
}
let mut yy = dst_height - dst_height % 4;
while yy < dst_height {
let yy = dst_height - dst_height % 4;
let src_rows = src_view.iter_rows(yy + offset);
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_one_row(
src_image.get_row(yy + offset).unwrap(),
dst_image.get_row_mut(yy).unwrap(),
&coefficients_chunks,
&normalizer,
);
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
}
yy += 1;
}
}
@@ -47,7 +43,7 @@ pub(crate) fn horiz_convolution(
#[target_feature(enable = "simd128")]
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U16x4]; 4],
dst_rows: [&mut &mut [U16x4]; 4],
dst_rows: [&mut [U16x4]; 4],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
normalizer: &optimisations::Normalizer32,
) {
+12 -16
View File
@@ -7,34 +7,30 @@ use crate::{ImageView, ImageViewMut};
#[inline]
pub(crate) fn horiz_convolution(
src_image: &ImageView<U8>,
dst_image: &mut ImageViewMut<U8>,
src_view: &impl ImageView<Pixel = U8>,
dst_view: &mut impl ImageViewMut<Pixel = U8>,
offset: u32,
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer16::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_image.height().get();
let dst_height = dst_view.height();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_image.iter_4_rows_mut();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
}
}
let mut yy = dst_height - dst_height % 4;
while yy < dst_height {
let yy = dst_height - dst_height % 4;
let src_rows = src_view.iter_rows(yy + offset);
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_row(
src_image.get_row(yy + offset).unwrap(),
dst_image.get_row_mut(yy).unwrap(),
&coefficients_chunks,
&normalizer,
);
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
}
yy += 1;
}
}
@@ -48,7 +44,7 @@ pub(crate) fn horiz_convolution(
#[target_feature(enable = "simd128")]
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U8]; 4],
dst_rows: [&mut &mut [U8]; 4],
dst_rows: [&mut [U8]; 4],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
normalizer: &optimisations::Normalizer16,
) {
@@ -114,7 +110,7 @@ unsafe fn horiz_convolution_four_rows(
/// - precision <= MAX_COEFS_PRECISION
#[inline]
#[target_feature(enable = "simd128")]
unsafe fn horiz_convolution_row(
unsafe fn horiz_convolution_one_row(
src_row: &[U8],
dst_row: &mut [U8],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
+12 -16
View File
@@ -7,34 +7,30 @@ use crate::{ImageView, ImageViewMut};
#[inline]
pub(crate) fn horiz_convolution(
src_image: &ImageView<U8x2>,
dst_image: &mut ImageViewMut<U8x2>,
src_view: &impl ImageView<Pixel = U8x2>,
dst_view: &mut impl ImageViewMut<Pixel = U8x2>,
offset: u32,
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer16::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_image.height().get();
let dst_height = dst_view.height();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_image.iter_4_rows_mut();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
}
}
let mut yy = dst_height - dst_height % 4;
while yy < dst_height {
let yy = dst_height - dst_height % 4;
let src_rows = src_view.iter_rows(yy + offset);
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_one_row(
src_image.get_row(yy + offset).unwrap(),
dst_image.get_row_mut(yy).unwrap(),
&coefficients_chunks,
&normalizer,
);
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
}
yy += 1;
}
}
@@ -48,7 +44,7 @@ pub(crate) fn horiz_convolution(
#[target_feature(enable = "simd128")]
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U8x2]; 4],
dst_rows: [&mut &mut [U8x2]; 4],
dst_rows: [&mut [U8x2]; 4],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
normalizer: &optimisations::Normalizer16,
) {
@@ -230,5 +226,5 @@ unsafe fn set_dst_pixel(
let a32 = buf[2].saturating_add(buf[3]);
let l8 = normalizer.clip(l32);
let a8 = normalizer.clip(a32);
d_row.get_unchecked_mut(dst_x).0 = u16::from_le_bytes([l8, a8]);
d_row.get_unchecked_mut(dst_x).0 = [l8, a8];
}
+23 -36
View File
@@ -8,35 +8,31 @@ use crate::{ImageView, ImageViewMut};
#[inline]
pub(crate) fn horiz_convolution(
src_image: &ImageView<U8x3>,
dst_image: &mut ImageViewMut<U8x3>,
src_view: &impl ImageView<Pixel = U8x3>,
dst_view: &mut impl ImageViewMut<Pixel = U8x3>,
offset: u32,
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer16::new(coeffs);
let precision = normalizer.precision();
let precision = normalizer.precision() as u32;
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_image.height().get();
let dst_height = dst_view.height();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_image.iter_4_rows_mut();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_8u4x(src_rows, dst_rows, &coefficients_chunks, precision);
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, precision);
}
}
let mut yy = dst_height - dst_height % 4;
while yy < dst_height {
let yy = dst_height - dst_height % 4;
let src_rows = src_view.iter_rows(yy + offset);
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_8u(
src_image.get_row(yy + offset).unwrap(),
dst_image.get_row_mut(yy).unwrap(),
&coefficients_chunks,
precision,
);
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, precision);
}
yy += 1;
}
}
@@ -48,11 +44,11 @@ pub(crate) fn horiz_convolution(
/// - precision <= MAX_COEFS_PRECISION
#[inline]
#[target_feature(enable = "simd128")]
unsafe fn horiz_convolution_8u4x(
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U8x3]; 4],
dst_rows: [&mut &mut [U8x3]; 4],
dst_rows: [&mut [U8x3]; 4],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
precision: u8,
precision: u32,
) {
const ZERO: v128 = i64x2(0, 0);
let initial = i32x4_splat(1 << (precision - 1));
@@ -153,15 +149,11 @@ unsafe fn horiz_convolution_8u4x(
x += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss_a[0] = i32x4_shr(sss_a[0], $imm8);
sss_a[1] = i32x4_shr(sss_a[1], $imm8);
sss_a[2] = i32x4_shr(sss_a[2], $imm8);
sss_a[3] = i32x4_shr(sss_a[3], $imm8);
}};
}
constify_imm8!(precision, call);
sss_a[0] = i32x4_shr(sss_a[0], precision);
sss_a[1] = i32x4_shr(sss_a[1], precision);
sss_a[2] = i32x4_shr(sss_a[2], precision);
sss_a[3] = i32x4_shr(sss_a[3], precision);
for i in 0..4 {
let sss = i16x8_narrow_i32x4(sss_a[i], ZERO);
@@ -179,11 +171,11 @@ unsafe fn horiz_convolution_8u4x(
/// - precision <= MAX_COEFS_PRECISION
#[inline]
#[target_feature(enable = "simd128")]
unsafe fn horiz_convolution_8u(
unsafe fn horiz_convolution_one_row(
src_row: &[U8x3],
dst_row: &mut [U8x3],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
precision: u8,
precision: u32,
) {
#[rustfmt::skip]
const PIX_SH1: v128 = i8x16(
@@ -278,12 +270,7 @@ unsafe fn horiz_convolution_8u(
x += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss = i32x4_shr(sss, $imm8);
}};
}
constify_imm8!(precision, call);
sss = i32x4_shr(sss, precision);
sss = i16x8_narrow_i32x4(sss, sss);
let pixel: u32 = transmute(i32x4_extract_lane::<0>(u8x16_narrow_i16x8(sss, sss)));
+22 -36
View File
@@ -11,35 +11,31 @@ use crate::{ImageView, ImageViewMut};
#[inline]
pub(crate) fn horiz_convolution(
src_image: &ImageView<U8x4>,
dst_image: &mut ImageViewMut<U8x4>,
src_view: &impl ImageView<Pixel = U8x4>,
dst_view: &mut impl ImageViewMut<Pixel = U8x4>,
offset: u32,
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer16::new(coeffs);
let precision = normalizer.precision();
let precision = normalizer.precision() as u32;
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_image.height().get();
let dst_height = dst_view.height();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_image.iter_4_rows_mut();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_8u4x(src_rows, dst_rows, &coefficients_chunks, precision);
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, precision);
}
}
let mut yy = dst_height - dst_height % 4;
while yy < dst_height {
let yy = dst_height - dst_height % 4;
let src_rows = src_view.iter_rows(yy + offset);
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_8u(
src_image.get_row(yy + offset).unwrap(),
dst_image.get_row_mut(yy).unwrap(),
&coefficients_chunks,
precision,
);
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, precision);
}
yy += 1;
}
}
@@ -50,11 +46,11 @@ pub(crate) fn horiz_convolution(
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "simd128")]
unsafe fn horiz_convolution_8u4x(
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U8x4]; 4],
dst_rows: [&mut &mut [U8x4]; 4],
dst_rows: [&mut [U8x4]; 4],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
precision: u8,
precision: u32,
) {
let initial = i32x4_splat(1 << (precision - 1));
const MASK_LO: v128 = i8x16(0, -1, 4, -1, 1, -1, 5, -1, 2, -1, 6, -1, 3, -1, 7, -1);
@@ -151,15 +147,10 @@ unsafe fn horiz_convolution_8u4x(
sss3 = i32x4_add(sss3, i32x4_dot_i16x8(pix, mmk));
}
macro_rules! call {
($imm8:expr) => {{
sss0 = i32x4_shr(sss0, $imm8);
sss1 = i32x4_shr(sss1, $imm8);
sss2 = i32x4_shr(sss2, $imm8);
sss3 = i32x4_shr(sss3, $imm8);
}};
}
constify_imm8!(precision, call);
sss0 = i32x4_shr(sss0, precision);
sss1 = i32x4_shr(sss1, precision);
sss2 = i32x4_shr(sss2, precision);
sss3 = i32x4_shr(sss3, precision);
sss0 = i16x8_narrow_i32x4(sss0, sss0);
sss1 = i16x8_narrow_i32x4(sss1, sss1);
@@ -182,11 +173,11 @@ unsafe fn horiz_convolution_8u4x(
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "simd128")]
unsafe fn horiz_convolution_8u(
unsafe fn horiz_convolution_one_row(
src_row: &[U8x4],
dst_row: &mut [U8x4],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
precision: u8,
precision: u32,
) {
let initial = i32x4_splat(1 << (precision - 1));
const SH1: v128 = i8x16(0, -1, 8, -1, 1, -1, 9, -1, 2, -1, 10, -1, 3, -1, 11, -1);
@@ -268,12 +259,7 @@ unsafe fn horiz_convolution_8u(
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
}
macro_rules! call {
($imm8:expr) => {{
sss = i32x4_shr(sss, $imm8);
}};
}
constify_imm8!(precision, call);
sss = i32x4_shr(sss, precision);
sss = i16x8_narrow_i32x4(sss, sss);
*dst_row.get_unchecked_mut(dst_x) =
+39 -36
View File
@@ -3,13 +3,13 @@ use std::arch::wasm32::*;
use crate::convolution::optimisations::CoefficientsI32Chunk;
use crate::convolution::vertical_u16::native::convolution_by_u16;
use crate::convolution::{optimisations, Coefficients};
use crate::pixels::PixelExt;
use crate::pixels::InnerPixel;
use crate::wasm32_utils;
use crate::{ImageView, ImageViewMut};
pub(crate) fn vert_convolution<T: PixelExt<Component = u16>>(
src_image: &ImageView<T>,
dst_image: &mut ImageViewMut<T>,
pub(crate) fn vert_convolution<T: InnerPixel<Component = u16>>(
src_view: &impl ImageView<Pixel = T>,
dst_view: &mut impl ImageViewMut<Pixel = T>,
offset: u32,
coeffs: Coefficients,
) {
@@ -17,17 +17,17 @@ pub(crate) fn vert_convolution<T: PixelExt<Component = u16>>(
let coefficients_chunks = normalizer.normalized_chunks();
let src_x = offset as usize * T::count_of_components();
let dst_rows = dst_image.iter_rows_mut();
let dst_rows = dst_view.iter_rows_mut(0);
for (dst_row, coeffs_chunk) in dst_rows.zip(coefficients_chunks) {
unsafe {
vert_convolution_into_one_row_u16(src_image, dst_row, src_x, coeffs_chunk, &normalizer);
vert_convolution_into_one_row_u16(src_view, dst_row, src_x, coeffs_chunk, &normalizer);
}
}
}
#[target_feature(enable = "simd128")]
unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
src_img: &ImageView<T>,
unsafe fn vert_convolution_into_one_row_u16<T: InnerPixel<Component = u16>>(
src_view: &impl ImageView<Pixel = T>,
dst_row: &mut [T],
mut src_x: usize,
coeffs_chunk: CoefficientsI32Chunk,
@@ -77,7 +77,7 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
let coeffs_2 = coeffs.chunks_exact(2);
let coeffs_reminder = coeffs_2.remainder();
for (src_rows, two_coeffs) in src_img.iter_2_rows(y_start, max_y).zip(coeffs_2) {
for (src_rows, two_coeffs) in src_view.iter_2_rows(y_start, max_y).zip(coeffs_2) {
let src_rows = src_rows.map(|row| T::components(row));
for r in 0..2 {
@@ -95,16 +95,17 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
}
if let Some(&k) = coeffs_reminder.first() {
let s_row = src_img.get_row(y_start + y).unwrap();
let components = T::components(s_row);
let coeff_i64x2 = i64x2_splat(k as i64);
if let Some(s_row) = src_view.iter_rows(y_start + y).next() {
let components = T::components(s_row);
let coeff_i64x2 = i64x2_splat(k as i64);
for x in 0..2 {
let source = wasm32_utils::load_v128(components, src_x + x * 8);
for i in 0..4 {
let c_i64x2 = i8x16_swizzle(source, c_shuffles[i]);
sums[i][x] =
i64x2_add(sums[i][x], wasm32_utils::i64x2_mul_lo(c_i64x2, coeff_i64x2));
for x in 0..2 {
let source = wasm32_utils::load_v128(components, src_x + x * 8);
for i in 0..4 {
let c_i64x2 = i8x16_swizzle(source, c_shuffles[i]);
sums[i][x] =
i64x2_add(sums[i][x], wasm32_utils::i64x2_mul_lo(c_i64x2, coeff_i64x2));
}
}
}
}
@@ -132,7 +133,7 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
let coeffs_2 = coeffs.chunks_exact(2);
let coeffs_reminder = coeffs_2.remainder();
for (src_rows, two_coeffs) in src_img.iter_2_rows(y_start, max_y).zip(coeffs_2) {
for (src_rows, two_coeffs) in src_view.iter_2_rows(y_start, max_y).zip(coeffs_2) {
let src_rows = src_rows.map(|row| T::components(row));
let coeffs_i64 = [
i64x2_splat(two_coeffs[0] as i64),
@@ -151,13 +152,14 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
}
if let Some(&k) = coeffs_reminder.first() {
let s_row = src_img.get_row(y_start + y).unwrap();
let components = T::components(s_row);
let coeff_i64x2 = i64x2_splat(k as i64);
let source = wasm32_utils::load_v128(components, src_x);
for i in 0..4 {
let c_i64x2 = i8x16_swizzle(source, c_shuffles[i]);
sums[i] = i64x2_add(sums[i], wasm32_utils::i64x2_mul_lo(c_i64x2, coeff_i64x2));
if let Some(s_row) = src_view.iter_rows(y_start + y).next() {
let components = T::components(s_row);
let coeff_i64x2 = i64x2_splat(k as i64);
let source = wasm32_utils::load_v128(components, src_x);
for i in 0..4 {
let c_i64x2 = i8x16_swizzle(source, c_shuffles[i]);
sums[i] = i64x2_add(sums[i], wasm32_utils::i64x2_mul_lo(c_i64x2, coeff_i64x2));
}
}
}
@@ -186,7 +188,7 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
let coeffs_2 = coeffs.chunks_exact(2);
let coeffs_reminder = coeffs_2.remainder();
for (src_rows, two_coeffs) in src_img.iter_2_rows(y_start, max_y).zip(coeffs_2) {
for (src_rows, two_coeffs) in src_view.iter_2_rows(y_start, max_y).zip(coeffs_2) {
let src_rows = src_rows.map(|row| T::components(row));
let coeffs_i64 = [
i64x2_splat(two_coeffs[0] as i64),
@@ -203,15 +205,16 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
}
if let Some(&k) = coeffs_reminder.first() {
let s_row = src_img.get_row(y_start + y).unwrap();
let components = T::components(s_row);
let coeff_i64x2 = i64x2_splat(k as i64);
if let Some(s_row) = src_view.iter_rows(y_start + y).next() {
let components = T::components(s_row);
let coeff_i64x2 = i64x2_splat(k as i64);
let comp_x4 = components.get_unchecked(src_x..src_x + 4);
let c_i64x2 = i64x2(comp_x4[0] as i64, comp_x4[1] as i64);
c01 = i64x2_add(c01, wasm32_utils::i64x2_mul_lo(c_i64x2, coeff_i64x2));
let c_i64x2 = i64x2(comp_x4[2] as i64, comp_x4[3] as i64);
c23 = i64x2_add(c23, wasm32_utils::i64x2_mul_lo(c_i64x2, coeff_i64x2));
let comp_x4 = components.get_unchecked(src_x..src_x + 4);
let c_i64x2 = i64x2(comp_x4[0] as i64, comp_x4[1] as i64);
c01 = i64x2_add(c01, wasm32_utils::i64x2_mul_lo(c_i64x2, coeff_i64x2));
let c_i64x2 = i64x2(comp_x4[2] as i64, comp_x4[3] as i64);
c23 = i64x2_add(c23, wasm32_utils::i64x2_mul_lo(c_i64x2, coeff_i64x2));
}
}
let mut dst_ptr = dst_chunk.as_mut_ptr();
@@ -232,7 +235,7 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
if !dst_u16.is_empty() {
let initial = 1 << (precision - 1);
convolution_by_u16(
src_img, normalizer, initial, dst_u16, src_x, y_start, coeffs,
src_view, normalizer, initial, dst_u16, src_x, y_start, coeffs,
);
}
}
+60 -57
View File
@@ -2,14 +2,14 @@ use std::arch::wasm32::*;
use crate::convolution::vertical_u8::native;
use crate::convolution::{optimisations, Coefficients};
use crate::pixels::PixelExt;
use crate::pixels::InnerPixel;
use crate::wasm32_utils;
use crate::{ImageView, ImageViewMut};
#[inline]
pub(crate) fn vert_convolution<T: PixelExt<Component = u8>>(
src_image: &ImageView<T>,
dst_image: &mut ImageViewMut<T>,
pub(crate) fn vert_convolution<T: InnerPixel<Component = u8>>(
src_view: &impl ImageView<Pixel = T>,
dst_view: &mut impl ImageViewMut<Pixel = T>,
offset: u32,
coeffs: Coefficients,
) {
@@ -17,18 +17,18 @@ pub(crate) fn vert_convolution<T: PixelExt<Component = u8>>(
let coefficients_chunks = normalizer.normalized_chunks();
let src_x = offset as usize * T::count_of_components();
let dst_rows = dst_image.iter_rows_mut();
let dst_rows = dst_view.iter_rows_mut(0);
for (dst_row, coeffs_chunk) in dst_rows.zip(coefficients_chunks) {
unsafe {
vert_convolution_into_one_row_u8(src_image, dst_row, src_x, coeffs_chunk, &normalizer);
vert_convolution_into_one_row_u8(src_view, dst_row, src_x, coeffs_chunk, &normalizer);
}
}
}
#[inline]
#[target_feature(enable = "simd128")]
unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
src_img: &ImageView<T>,
unsafe fn vert_convolution_into_one_row_u8<T: InnerPixel<Component = u8>>(
src_view: &impl ImageView<Pixel = T>,
dst_row: &mut [T],
mut src_x: usize,
coeffs_chunk: optimisations::CoefficientsI16Chunk,
@@ -56,7 +56,7 @@ unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
let mut y: u32 = 0;
for src_rows in src_img.iter_2_rows(y_start, max_y) {
for src_rows in src_view.iter_2_rows(y_start, max_y) {
let components1 = T::components(src_rows[0]);
let components2 = T::components(src_rows[1]);
@@ -107,41 +107,42 @@ unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
}
if let Some(&k) = coeffs.get(y as usize) {
let s_row = src_img.get_row(y_start + y).unwrap();
let components = T::components(s_row);
let mmk = i32x4_splat(k as i32);
if let Some(s_row) = src_view.iter_rows(y_start + y).next() {
let components = T::components(s_row);
let mmk = i32x4_splat(k as i32);
let source1 = wasm32_utils::load_v128(components, src_x); // top line
let source1 = wasm32_utils::load_v128(components, src_x); // top line
let source = i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(
source1, ZERO,
);
let pix = i16x8_extend_low_u8x16(source);
sss0 = i32x4_add(sss0, i32x4_dot_i16x8(pix, mmk));
let pix = i16x8_extend_high_u8x16(source);
sss1 = i32x4_add(sss1, i32x4_dot_i16x8(pix, mmk));
let source = i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(
source1, ZERO,
);
let pix = i16x8_extend_low_u8x16(source);
sss0 = i32x4_add(sss0, i32x4_dot_i16x8(pix, mmk));
let pix = i16x8_extend_high_u8x16(source);
sss1 = i32x4_add(sss1, i32x4_dot_i16x8(pix, mmk));
let source = i16x8_extend_high_u8x16(source1);
let pix = i16x8_extend_low_u8x16(source);
sss2 = i32x4_add(sss2, i32x4_dot_i16x8(pix, mmk));
let pix = i16x8_extend_high_u8x16(source);
sss3 = i32x4_add(sss3, i32x4_dot_i16x8(pix, mmk));
let source = i16x8_extend_high_u8x16(source1);
let pix = i16x8_extend_low_u8x16(source);
sss2 = i32x4_add(sss2, i32x4_dot_i16x8(pix, mmk));
let pix = i16x8_extend_high_u8x16(source);
sss3 = i32x4_add(sss3, i32x4_dot_i16x8(pix, mmk));
let source1 = wasm32_utils::load_v128(components, src_x + 16); // top line
let source1 = wasm32_utils::load_v128(components, src_x + 16); // top line
let source = i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(
source1, ZERO,
);
let pix = i16x8_extend_low_u8x16(source);
sss4 = i32x4_add(sss4, i32x4_dot_i16x8(pix, mmk));
let pix = i16x8_extend_high_u8x16(source);
sss5 = i32x4_add(sss5, i32x4_dot_i16x8(pix, mmk));
let source = i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(
source1, ZERO,
);
let pix = i16x8_extend_low_u8x16(source);
sss4 = i32x4_add(sss4, i32x4_dot_i16x8(pix, mmk));
let pix = i16x8_extend_high_u8x16(source);
sss5 = i32x4_add(sss5, i32x4_dot_i16x8(pix, mmk));
let source = i16x8_extend_high_u8x16(source1);
let pix = i16x8_extend_low_u8x16(source);
sss6 = i32x4_add(sss6, i32x4_dot_i16x8(pix, mmk));
let pix = i16x8_extend_high_u8x16(source);
sss7 = i32x4_add(sss7, i32x4_dot_i16x8(pix, mmk));
let source = i16x8_extend_high_u8x16(source1);
let pix = i16x8_extend_low_u8x16(source);
sss6 = i32x4_add(sss6, i32x4_dot_i16x8(pix, mmk));
let pix = i16x8_extend_high_u8x16(source);
sss7 = i32x4_add(sss7, i32x4_dot_i16x8(pix, mmk));
}
}
// This version of code works faster.
@@ -180,7 +181,7 @@ unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
let mut sss1 = initial; // right row
let mut y: u32 = 0;
for src_rows in src_img.iter_2_rows(y_start, max_y) {
for src_rows in src_view.iter_2_rows(y_start, max_y) {
let components1 = T::components(src_rows[0]);
let components2 = T::components(src_rows[1]);
// Load two coefficients at once
@@ -201,19 +202,20 @@ unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
}
if let Some(&k) = coeffs.get(y as usize) {
let s_row = src_img.get_row(y_start + y).unwrap();
let components = T::components(s_row);
let mmk = i32x4_splat(k as i32);
if let Some(s_row) = src_view.iter_rows(y_start + y).next() {
let components = T::components(s_row);
let mmk = i32x4_splat(k as i32);
let source1 = wasm32_utils::loadl_i64(components, src_x); // top line
let source1 = wasm32_utils::loadl_i64(components, src_x); // top line
let source = i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(
source1, ZERO,
);
let pix = i16x8_extend_low_u8x16(source);
sss0 = i32x4_add(sss0, i32x4_dot_i16x8(pix, mmk));
let pix = i16x8_extend_high_u8x16(source);
sss1 = i32x4_add(sss1, i32x4_dot_i16x8(pix, mmk));
let source = i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(
source1, ZERO,
);
let pix = i16x8_extend_low_u8x16(source);
sss0 = i32x4_add(sss0, i32x4_dot_i16x8(pix, mmk));
let pix = i16x8_extend_high_u8x16(source);
sss1 = i32x4_add(sss1, i32x4_dot_i16x8(pix, mmk));
}
}
macro_rules! call {
@@ -238,7 +240,7 @@ unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
let mut sss = initial;
let mut y: u32 = 0;
for src_rows in src_img.iter_2_rows(y_start, max_y) {
for src_rows in src_view.iter_2_rows(y_start, max_y) {
let components1 = T::components(src_rows[0]);
let components2 = T::components(src_rows[1]);
// Load two coefficients at once
@@ -257,11 +259,12 @@ unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
}
if let Some(&k) = coeffs.get(y as usize) {
let s_row = src_img.get_row(y_start + y).unwrap();
let components = T::components(s_row);
let pix = wasm32_utils::i32x4_extend_low_ptr_u8(components, src_x);
let mmk = i32x4_splat(k as i32);
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
if let Some(s_row) = src_view.iter_rows(y_start + y).next() {
let components = T::components(s_row);
let pix = wasm32_utils::i32x4_extend_low_ptr_u8(components, src_x);
let mmk = i32x4_splat(k as i32);
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
}
}
macro_rules! call {
@@ -281,7 +284,7 @@ unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
dst_u8 = dst_chunks_4.into_remainder();
if !dst_u8.is_empty() {
native::convolution_by_u8(
src_img,
src_view,
normalizer,
1 << (precision - 1),
dst_u8,