mirror of
https://github.com/Cykooz/fast_image_resize.git
synced 2026-10-08 01:11:09 +00:00
Fixed code for Wasm32
This commit is contained in:
@@ -1,14 +1,16 @@
|
||||
# Preparation
|
||||
|
||||
Install system libraries:
|
||||
|
||||
- libvips-dev (used in benchmarks)
|
||||
|
||||
Install additional toolchains:
|
||||
|
||||
- Arm64:
|
||||
```shell
|
||||
rustup target add aarch64-unknown-linux-gnu
|
||||
```
|
||||
- Wasm32:
|
||||
- Wasm32:
|
||||
```shell
|
||||
rustup target add wasm32-wasi
|
||||
```
|
||||
@@ -16,45 +18,51 @@ Install additional toolchains:
|
||||
|
||||
# Tests
|
||||
|
||||
Run tests without saving result images as files in `./data` directory:
|
||||
Run tests with saving result images as files in `./data` directory:
|
||||
|
||||
```shell
|
||||
DONT_SAVE_RESULT=1 cargo test
|
||||
SAVE_RESULT=1 cargo test
|
||||
```
|
||||
|
||||
# Benchmarks
|
||||
|
||||
Run benchmarks to compare with other crates for image resizing and write results into
|
||||
report files, such as `./benchmarks-x86_64.md`:
|
||||
|
||||
```shell
|
||||
WRITE_COMPARE_RESULT=1 cargo bench -- Compare
|
||||
```
|
||||
|
||||
|
||||
# Wasm32
|
||||
|
||||
Specify build target in `.cargo/config.toml` file.
|
||||
|
||||
```toml
|
||||
[build]
|
||||
target = "wasm32-wasi"
|
||||
```
|
||||
|
||||
Run tests:
|
||||
|
||||
```shell
|
||||
CARGO_TARGET_WASM32_WASI_RUNNER="wasmtime --dir=. --" cargo test
|
||||
```
|
||||
|
||||
Run tests without saving result images as files in `./data` directory:
|
||||
Run tests with saving result images as files in `./data` directory:
|
||||
|
||||
```shell
|
||||
CARGO_TARGET_WASM32_WASI_RUNNER="wasmtime --dir=. --env DONT_SAVE_RESULT=1 --" cargo test
|
||||
CARGO_TARGET_WASM32_WASI_RUNNER="wasmtime --dir=. --env SAVE_RESULT=1 --" cargo test
|
||||
```
|
||||
|
||||
Run a specific benchmark in `quick` mode:
|
||||
|
||||
```shell
|
||||
CARGO_TARGET_WASM32_WASI_RUNNER="wasmtime --dir=. --" cargo bench --bench bench_resize -- --color=always --quick
|
||||
```
|
||||
|
||||
Run benchmarks to compare with other crates for image resizing and write results into
|
||||
report files, such as `./benchmarks-x86_64.md`:
|
||||
|
||||
```shell
|
||||
CARGO_TARGET_WASM32_WASI_RUNNER="wasmtime --dir=. --env WRITE_COMPARE_RESULT=1 --" cargo bench -- --color=always Compare
|
||||
```
|
||||
|
||||
+12
-12
@@ -7,19 +7,19 @@ use crate::{ImageView, ImageViewMut};
|
||||
use super::native;
|
||||
|
||||
pub(crate) unsafe fn multiply_alpha(
|
||||
src_image: &ImageView<U16x2>,
|
||||
dst_image: &mut ImageViewMut<U16x2>,
|
||||
src_view: &impl ImageView<Pixel = U16x2>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16x2>,
|
||||
) {
|
||||
let src_rows = src_image.iter_rows(0);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
let src_rows = src_view.iter_rows(0);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
multiply_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) unsafe fn multiply_alpha_inplace(image: &mut ImageViewMut<U16x2>) {
|
||||
for row in image.iter_rows_mut() {
|
||||
pub(crate) unsafe fn multiply_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = U16x2>) {
|
||||
for row in image_view.iter_rows_mut(0) {
|
||||
multiply_alpha_row_inplace(row);
|
||||
}
|
||||
}
|
||||
@@ -93,19 +93,19 @@ unsafe fn multiply_alpha_4_pixels(pixels: v128) -> v128 {
|
||||
// Divide
|
||||
|
||||
pub(crate) unsafe fn divide_alpha(
|
||||
src_image: &ImageView<U16x2>,
|
||||
dst_image: &mut ImageViewMut<U16x2>,
|
||||
src_view: &impl ImageView<Pixel = U16x2>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16x2>,
|
||||
) {
|
||||
let src_rows = src_image.iter_rows(0);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
let src_rows = src_view.iter_rows(0);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
divide_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) unsafe fn divide_alpha_inplace(image: &mut ImageViewMut<U16x2>) {
|
||||
for row in image.iter_rows_mut() {
|
||||
pub(crate) unsafe fn divide_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = U16x2>) {
|
||||
for row in image_view.iter_rows_mut(0) {
|
||||
divide_alpha_row_inplace(row);
|
||||
}
|
||||
}
|
||||
|
||||
+12
-12
@@ -7,19 +7,19 @@ use crate::{ImageView, ImageViewMut};
|
||||
use super::native;
|
||||
|
||||
pub(crate) unsafe fn multiply_alpha(
|
||||
src_image: &ImageView<U16x4>,
|
||||
dst_image: &mut ImageViewMut<U16x4>,
|
||||
src_view: &impl ImageView<Pixel = U16x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16x4>,
|
||||
) {
|
||||
let src_rows = src_image.iter_rows(0);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
let src_rows = src_view.iter_rows(0);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
multiply_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) unsafe fn multiply_alpha_inplace(image: &mut ImageViewMut<U16x4>) {
|
||||
for row in image.iter_rows_mut() {
|
||||
pub(crate) unsafe fn multiply_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = U16x4>) {
|
||||
for row in image_view.iter_rows_mut(0) {
|
||||
multiply_alpha_row_inplace(row);
|
||||
}
|
||||
}
|
||||
@@ -105,19 +105,19 @@ unsafe fn multiply_alpha_2_pixels(pixels: v128) -> v128 {
|
||||
// Divide
|
||||
|
||||
pub(crate) unsafe fn divide_alpha(
|
||||
src_image: &ImageView<U16x4>,
|
||||
dst_image: &mut ImageViewMut<U16x4>,
|
||||
src_view: &impl ImageView<Pixel = U16x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16x4>,
|
||||
) {
|
||||
let src_rows = src_image.iter_rows(0);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
let src_rows = src_view.iter_rows(0);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
divide_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) unsafe fn divide_alpha_inplace(image: &mut ImageViewMut<U16x4>) {
|
||||
for row in image.iter_rows_mut() {
|
||||
pub(crate) unsafe fn divide_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = U16x4>) {
|
||||
for row in image_view.iter_rows_mut(0) {
|
||||
divide_alpha_row_inplace(row);
|
||||
}
|
||||
}
|
||||
|
||||
+18
-15
@@ -6,19 +6,19 @@ use crate::{ImageView, ImageViewMut};
|
||||
use super::native;
|
||||
|
||||
pub(crate) unsafe fn multiply_alpha(
|
||||
src_image: &ImageView<U8x2>,
|
||||
dst_image: &mut ImageViewMut<U8x2>,
|
||||
src_view: &impl ImageView<Pixel = U8x2>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x2>,
|
||||
) {
|
||||
let src_rows = src_image.iter_rows(0);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
let src_rows = src_view.iter_rows(0);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
multiply_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) unsafe fn multiply_alpha_inplace(image: &mut ImageViewMut<U8x2>) {
|
||||
for row in image.iter_rows_mut() {
|
||||
pub(crate) unsafe fn multiply_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = U8x2>) {
|
||||
for row in image_view.iter_rows_mut(0) {
|
||||
multiply_alpha_row_inplace(row);
|
||||
}
|
||||
}
|
||||
@@ -91,17 +91,20 @@ unsafe fn multiplies_alpha_8_pixels(pixels: v128) -> v128 {
|
||||
|
||||
// Divide
|
||||
|
||||
pub(crate) unsafe fn divide_alpha(src_image: &ImageView<U8x2>, dst_image: &mut ImageViewMut<U8x2>) {
|
||||
let src_rows = src_image.iter_rows(0);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
pub(crate) unsafe fn divide_alpha(
|
||||
src_view: &impl ImageView<Pixel = U8x2>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x2>,
|
||||
) {
|
||||
let src_rows = src_view.iter_rows(0);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
divide_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) unsafe fn divide_alpha_inplace(image: &mut ImageViewMut<U8x2>) {
|
||||
for row in image.iter_rows_mut() {
|
||||
pub(crate) unsafe fn divide_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = U8x2>) {
|
||||
for row in image_view.iter_rows_mut(0) {
|
||||
divide_alpha_row_inplace(row);
|
||||
}
|
||||
}
|
||||
@@ -122,13 +125,13 @@ unsafe fn divide_alpha_row(src_row: &[U8x2], dst_row: &mut [U8x2]) {
|
||||
|
||||
if !src_remainder.is_empty() {
|
||||
let dst_reminder = dst_chunks.into_remainder();
|
||||
let mut src_pixels = [U8x2::new(0); 8];
|
||||
let mut src_pixels = [U8x2::new([0, 0]); 8];
|
||||
src_pixels
|
||||
.iter_mut()
|
||||
.zip(src_remainder)
|
||||
.for_each(|(d, s)| *d = *s);
|
||||
|
||||
let mut dst_pixels = [U8x2::new(0); 8];
|
||||
let mut dst_pixels = [U8x2::new([0; 2]); 8];
|
||||
let mut pixels = v128_load(src_pixels.as_ptr() as *const v128);
|
||||
pixels = divide_alpha_8_pixels(pixels);
|
||||
v128_store(dst_pixels.as_mut_ptr() as *mut v128, pixels);
|
||||
@@ -153,13 +156,13 @@ unsafe fn divide_alpha_row_inplace(row: &mut [U8x2]) {
|
||||
|
||||
let reminder = chunks.into_remainder();
|
||||
if !reminder.is_empty() {
|
||||
let mut src_pixels = [U8x2::new(0); 8];
|
||||
let mut src_pixels = [U8x2::new([0; 2]); 8];
|
||||
src_pixels
|
||||
.iter_mut()
|
||||
.zip(reminder.iter())
|
||||
.for_each(|(d, s)| *d = *s);
|
||||
|
||||
let mut dst_pixels = [U8x2::new(0); 8];
|
||||
let mut dst_pixels = [U8x2::new([0; 2]); 8];
|
||||
let mut pixels = v128_load(src_pixels.as_ptr() as *const v128);
|
||||
pixels = divide_alpha_8_pixels(pixels);
|
||||
v128_store(dst_pixels.as_mut_ptr() as *mut v128, pixels);
|
||||
|
||||
+17
-15
@@ -7,19 +7,19 @@ use crate::{ImageView, ImageViewMut};
|
||||
use super::native;
|
||||
|
||||
pub(crate) unsafe fn multiply_alpha(
|
||||
src_image: &ImageView<U8x4>,
|
||||
dst_image: &mut ImageViewMut<U8x4>,
|
||||
src_view: &impl ImageView<Pixel = U8x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x4>,
|
||||
) {
|
||||
let src_rows = src_image.iter_rows(0);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
let src_rows = src_view.iter_rows(0);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
multiply_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) unsafe fn multiply_alpha_inplace(image: &mut ImageViewMut<U8x4>) {
|
||||
for row in image.iter_rows_mut() {
|
||||
pub(crate) unsafe fn multiply_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = U8x4>) {
|
||||
for row in image_view.iter_rows_mut(0) {
|
||||
multiply_alpha_row_inplace(row);
|
||||
}
|
||||
}
|
||||
@@ -64,13 +64,12 @@ unsafe fn multiply_alpha_row_inplace(row: &mut [U8x4]) {
|
||||
#[inline]
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn multiply_alpha_4_pixels(pixels: v128) -> v128 {
|
||||
let half = u16x8_splat(128);
|
||||
let max_alpha = u32x4_splat(0xff000000);
|
||||
const FACTOR_MASK: v128 = i8x16(3, 3, 3, 3, 7, 7, 7, 7, 11, 11, 11, 11, 15, 15, 15, 15);
|
||||
|
||||
let factor_pixels = u8x16_swizzle(pixels, FACTOR_MASK);
|
||||
let factor_pixels = v128_or(factor_pixels, max_alpha);
|
||||
let max_alpha = u32x4_splat(0xff000000);
|
||||
let factor_pixels = v128_or(u8x16_swizzle(pixels, FACTOR_MASK), max_alpha);
|
||||
|
||||
let half = u16x8_splat(128);
|
||||
let src_u16_lo = u16x8_extend_low_u8x16(pixels);
|
||||
let factors = u16x8_extend_low_u8x16(factor_pixels);
|
||||
let mut dst_u16_lo = u16x8_add(u16x8_mul(src_u16_lo, factors), half);
|
||||
@@ -88,16 +87,19 @@ unsafe fn multiply_alpha_4_pixels(pixels: v128) -> v128 {
|
||||
|
||||
// Divide
|
||||
|
||||
pub(crate) unsafe fn divide_alpha(src_image: &ImageView<U8x4>, dst_image: &mut ImageViewMut<U8x4>) {
|
||||
let src_rows = src_image.iter_rows(0);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
pub(crate) unsafe fn divide_alpha(
|
||||
src_view: &impl ImageView<Pixel = U8x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x4>,
|
||||
) {
|
||||
let src_rows = src_view.iter_rows(0);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
divide_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) unsafe fn divide_alpha_inplace(image: &mut ImageViewMut<U8x4>) {
|
||||
for row in image.iter_rows_mut() {
|
||||
pub(crate) unsafe fn divide_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = U8x4>) {
|
||||
for row in image_view.iter_rows_mut(0) {
|
||||
divide_alpha_row_inplace(row);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -7,34 +7,30 @@ use crate::{ImageView, ImageViewMut};
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn horiz_convolution(
|
||||
src_image: &ImageView<U16>,
|
||||
dst_image: &mut ImageViewMut<U16>,
|
||||
src_view: &impl ImageView<Pixel = U16>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_image.height().get();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
let mut yy = dst_height - dst_height % 4;
|
||||
while yy < dst_height {
|
||||
let yy = dst_height - dst_height % 4;
|
||||
let src_rows = src_view.iter_rows(yy + offset);
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
&coefficients_chunks,
|
||||
&normalizer,
|
||||
);
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
|
||||
}
|
||||
yy += 1;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -46,7 +42,7 @@ pub(crate) fn horiz_convolution(
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U16]; 4],
|
||||
dst_rows: [&mut &mut [U16]; 4],
|
||||
dst_rows: [&mut [U16]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
|
||||
@@ -7,34 +7,30 @@ use crate::{ImageView, ImageViewMut};
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn horiz_convolution(
|
||||
src_image: &ImageView<U16x2>,
|
||||
dst_image: &mut ImageViewMut<U16x2>,
|
||||
src_view: &impl ImageView<Pixel = U16x2>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16x2>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_image.height().get();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
let mut yy = dst_height - dst_height % 4;
|
||||
while yy < dst_height {
|
||||
let yy = dst_height - dst_height % 4;
|
||||
let src_rows = src_view.iter_rows(yy + offset);
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
&coefficients_chunks,
|
||||
&normalizer,
|
||||
);
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
|
||||
}
|
||||
yy += 1;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -47,7 +43,7 @@ pub(crate) fn horiz_convolution(
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U16x2]; 4],
|
||||
dst_rows: [&mut &mut [U16x2]; 4],
|
||||
dst_rows: [&mut [U16x2]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
|
||||
@@ -1,6 +1,5 @@
|
||||
use std::arch::wasm32::*;
|
||||
|
||||
use crate::convolution::optimisations::CoefficientsI32Chunk;
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::pixels::U16x3;
|
||||
use crate::wasm32_utils;
|
||||
@@ -8,34 +7,30 @@ use crate::{ImageView, ImageViewMut};
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn horiz_convolution(
|
||||
src_image: &ImageView<U16x3>,
|
||||
dst_image: &mut ImageViewMut<U16x3>,
|
||||
src_view: &impl ImageView<Pixel = U16x3>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16x3>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_image.height().get();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_8u4x(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
let mut yy = dst_height - dst_height % 4;
|
||||
while yy < dst_height {
|
||||
let yy = dst_height - dst_height % 4;
|
||||
let src_rows = src_view.iter_rows(yy + offset);
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_8u(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
&coefficients_chunks,
|
||||
&normalizer,
|
||||
);
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
|
||||
}
|
||||
yy += 1;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -46,10 +41,10 @@ pub(crate) fn horiz_convolution(
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn horiz_convolution_8u4x(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U16x3]; 4],
|
||||
dst_rows: [&mut &mut [U16x3]; 4],
|
||||
coefficients_chunks: &[CoefficientsI32Chunk],
|
||||
dst_rows: [&mut [U16x3]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
const ZERO: v128 = i64x2(0, 0);
|
||||
@@ -149,10 +144,10 @@ unsafe fn horiz_convolution_8u4x(
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn horiz_convolution_8u(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16x3],
|
||||
dst_row: &mut [U16x3],
|
||||
coefficients_chunks: &[CoefficientsI32Chunk],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
|
||||
@@ -7,34 +7,30 @@ use crate::{ImageView, ImageViewMut};
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn horiz_convolution(
|
||||
src_image: &ImageView<U16x4>,
|
||||
dst_image: &mut ImageViewMut<U16x4>,
|
||||
src_view: &impl ImageView<Pixel = U16x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16x4>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_image.height().get();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
let mut yy = dst_height - dst_height % 4;
|
||||
while yy < dst_height {
|
||||
let yy = dst_height - dst_height % 4;
|
||||
let src_rows = src_view.iter_rows(yy + offset);
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
&coefficients_chunks,
|
||||
&normalizer,
|
||||
);
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
|
||||
}
|
||||
yy += 1;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -47,7 +43,7 @@ pub(crate) fn horiz_convolution(
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U16x4]; 4],
|
||||
dst_rows: [&mut &mut [U16x4]; 4],
|
||||
dst_rows: [&mut [U16x4]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
|
||||
@@ -7,34 +7,30 @@ use crate::{ImageView, ImageViewMut};
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn horiz_convolution(
|
||||
src_image: &ImageView<U8>,
|
||||
dst_image: &mut ImageViewMut<U8>,
|
||||
src_view: &impl ImageView<Pixel = U8>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_image.height().get();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
let mut yy = dst_height - dst_height % 4;
|
||||
while yy < dst_height {
|
||||
let yy = dst_height - dst_height % 4;
|
||||
let src_rows = src_view.iter_rows(yy + offset);
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_row(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
&coefficients_chunks,
|
||||
&normalizer,
|
||||
);
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
|
||||
}
|
||||
yy += 1;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -48,7 +44,7 @@ pub(crate) fn horiz_convolution(
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U8]; 4],
|
||||
dst_rows: [&mut &mut [U8]; 4],
|
||||
dst_rows: [&mut [U8]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
@@ -114,7 +110,7 @@ unsafe fn horiz_convolution_four_rows(
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[inline]
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn horiz_convolution_row(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U8],
|
||||
dst_row: &mut [U8],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
|
||||
@@ -7,34 +7,30 @@ use crate::{ImageView, ImageViewMut};
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn horiz_convolution(
|
||||
src_image: &ImageView<U8x2>,
|
||||
dst_image: &mut ImageViewMut<U8x2>,
|
||||
src_view: &impl ImageView<Pixel = U8x2>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x2>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_image.height().get();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
let mut yy = dst_height - dst_height % 4;
|
||||
while yy < dst_height {
|
||||
let yy = dst_height - dst_height % 4;
|
||||
let src_rows = src_view.iter_rows(yy + offset);
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
&coefficients_chunks,
|
||||
&normalizer,
|
||||
);
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
|
||||
}
|
||||
yy += 1;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -48,7 +44,7 @@ pub(crate) fn horiz_convolution(
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U8x2]; 4],
|
||||
dst_rows: [&mut &mut [U8x2]; 4],
|
||||
dst_rows: [&mut [U8x2]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
@@ -230,5 +226,5 @@ unsafe fn set_dst_pixel(
|
||||
let a32 = buf[2].saturating_add(buf[3]);
|
||||
let l8 = normalizer.clip(l32);
|
||||
let a8 = normalizer.clip(a32);
|
||||
d_row.get_unchecked_mut(dst_x).0 = u16::from_le_bytes([l8, a8]);
|
||||
d_row.get_unchecked_mut(dst_x).0 = [l8, a8];
|
||||
}
|
||||
|
||||
@@ -8,35 +8,31 @@ use crate::{ImageView, ImageViewMut};
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn horiz_convolution(
|
||||
src_image: &ImageView<U8x3>,
|
||||
dst_image: &mut ImageViewMut<U8x3>,
|
||||
src_view: &impl ImageView<Pixel = U8x3>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x3>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let precision = normalizer.precision() as u32;
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_image.height().get();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_8u4x(src_rows, dst_rows, &coefficients_chunks, precision);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, precision);
|
||||
}
|
||||
}
|
||||
|
||||
let mut yy = dst_height - dst_height % 4;
|
||||
while yy < dst_height {
|
||||
let yy = dst_height - dst_height % 4;
|
||||
let src_rows = src_view.iter_rows(yy + offset);
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_8u(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
&coefficients_chunks,
|
||||
precision,
|
||||
);
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, precision);
|
||||
}
|
||||
yy += 1;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -48,11 +44,11 @@ pub(crate) fn horiz_convolution(
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[inline]
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn horiz_convolution_8u4x(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U8x3]; 4],
|
||||
dst_rows: [&mut &mut [U8x3]; 4],
|
||||
dst_rows: [&mut [U8x3]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
precision: u8,
|
||||
precision: u32,
|
||||
) {
|
||||
const ZERO: v128 = i64x2(0, 0);
|
||||
let initial = i32x4_splat(1 << (precision - 1));
|
||||
@@ -153,15 +149,11 @@ unsafe fn horiz_convolution_8u4x(
|
||||
|
||||
x += 1;
|
||||
}
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss_a[0] = i32x4_shr(sss_a[0], $imm8);
|
||||
sss_a[1] = i32x4_shr(sss_a[1], $imm8);
|
||||
sss_a[2] = i32x4_shr(sss_a[2], $imm8);
|
||||
sss_a[3] = i32x4_shr(sss_a[3], $imm8);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
|
||||
sss_a[0] = i32x4_shr(sss_a[0], precision);
|
||||
sss_a[1] = i32x4_shr(sss_a[1], precision);
|
||||
sss_a[2] = i32x4_shr(sss_a[2], precision);
|
||||
sss_a[3] = i32x4_shr(sss_a[3], precision);
|
||||
|
||||
for i in 0..4 {
|
||||
let sss = i16x8_narrow_i32x4(sss_a[i], ZERO);
|
||||
@@ -179,11 +171,11 @@ unsafe fn horiz_convolution_8u4x(
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[inline]
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn horiz_convolution_8u(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U8x3],
|
||||
dst_row: &mut [U8x3],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
precision: u8,
|
||||
precision: u32,
|
||||
) {
|
||||
#[rustfmt::skip]
|
||||
const PIX_SH1: v128 = i8x16(
|
||||
@@ -278,12 +270,7 @@ unsafe fn horiz_convolution_8u(
|
||||
x += 1;
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss = i32x4_shr(sss, $imm8);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
sss = i32x4_shr(sss, precision);
|
||||
|
||||
sss = i16x8_narrow_i32x4(sss, sss);
|
||||
let pixel: u32 = transmute(i32x4_extract_lane::<0>(u8x16_narrow_i16x8(sss, sss)));
|
||||
|
||||
@@ -11,35 +11,31 @@ use crate::{ImageView, ImageViewMut};
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn horiz_convolution(
|
||||
src_image: &ImageView<U8x4>,
|
||||
dst_image: &mut ImageViewMut<U8x4>,
|
||||
src_view: &impl ImageView<Pixel = U8x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x4>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let precision = normalizer.precision() as u32;
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_image.height().get();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_8u4x(src_rows, dst_rows, &coefficients_chunks, precision);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, precision);
|
||||
}
|
||||
}
|
||||
|
||||
let mut yy = dst_height - dst_height % 4;
|
||||
while yy < dst_height {
|
||||
let yy = dst_height - dst_height % 4;
|
||||
let src_rows = src_view.iter_rows(yy + offset);
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_8u(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
&coefficients_chunks,
|
||||
precision,
|
||||
);
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, precision);
|
||||
}
|
||||
yy += 1;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -50,11 +46,11 @@ pub(crate) fn horiz_convolution(
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn horiz_convolution_8u4x(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U8x4]; 4],
|
||||
dst_rows: [&mut &mut [U8x4]; 4],
|
||||
dst_rows: [&mut [U8x4]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
precision: u8,
|
||||
precision: u32,
|
||||
) {
|
||||
let initial = i32x4_splat(1 << (precision - 1));
|
||||
const MASK_LO: v128 = i8x16(0, -1, 4, -1, 1, -1, 5, -1, 2, -1, 6, -1, 3, -1, 7, -1);
|
||||
@@ -151,15 +147,10 @@ unsafe fn horiz_convolution_8u4x(
|
||||
sss3 = i32x4_add(sss3, i32x4_dot_i16x8(pix, mmk));
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss0 = i32x4_shr(sss0, $imm8);
|
||||
sss1 = i32x4_shr(sss1, $imm8);
|
||||
sss2 = i32x4_shr(sss2, $imm8);
|
||||
sss3 = i32x4_shr(sss3, $imm8);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
sss0 = i32x4_shr(sss0, precision);
|
||||
sss1 = i32x4_shr(sss1, precision);
|
||||
sss2 = i32x4_shr(sss2, precision);
|
||||
sss3 = i32x4_shr(sss3, precision);
|
||||
|
||||
sss0 = i16x8_narrow_i32x4(sss0, sss0);
|
||||
sss1 = i16x8_narrow_i32x4(sss1, sss1);
|
||||
@@ -182,11 +173,11 @@ unsafe fn horiz_convolution_8u4x(
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn horiz_convolution_8u(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U8x4],
|
||||
dst_row: &mut [U8x4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
precision: u8,
|
||||
precision: u32,
|
||||
) {
|
||||
let initial = i32x4_splat(1 << (precision - 1));
|
||||
const SH1: v128 = i8x16(0, -1, 8, -1, 1, -1, 9, -1, 2, -1, 10, -1, 3, -1, 11, -1);
|
||||
@@ -268,12 +259,7 @@ unsafe fn horiz_convolution_8u(
|
||||
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss = i32x4_shr(sss, $imm8);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
sss = i32x4_shr(sss, precision);
|
||||
|
||||
sss = i16x8_narrow_i32x4(sss, sss);
|
||||
*dst_row.get_unchecked_mut(dst_x) =
|
||||
|
||||
@@ -3,13 +3,13 @@ use std::arch::wasm32::*;
|
||||
use crate::convolution::optimisations::CoefficientsI32Chunk;
|
||||
use crate::convolution::vertical_u16::native::convolution_by_u16;
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::pixels::PixelExt;
|
||||
use crate::pixels::InnerPixel;
|
||||
use crate::wasm32_utils;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
pub(crate) fn vert_convolution<T: PixelExt<Component = u16>>(
|
||||
src_image: &ImageView<T>,
|
||||
dst_image: &mut ImageViewMut<T>,
|
||||
pub(crate) fn vert_convolution<T: InnerPixel<Component = u16>>(
|
||||
src_view: &impl ImageView<Pixel = T>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = T>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
@@ -17,17 +17,17 @@ pub(crate) fn vert_convolution<T: PixelExt<Component = u16>>(
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let src_x = offset as usize * T::count_of_components();
|
||||
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
for (dst_row, coeffs_chunk) in dst_rows.zip(coefficients_chunks) {
|
||||
unsafe {
|
||||
vert_convolution_into_one_row_u16(src_image, dst_row, src_x, coeffs_chunk, &normalizer);
|
||||
vert_convolution_into_one_row_u16(src_view, dst_row, src_x, coeffs_chunk, &normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
|
||||
src_img: &ImageView<T>,
|
||||
unsafe fn vert_convolution_into_one_row_u16<T: InnerPixel<Component = u16>>(
|
||||
src_view: &impl ImageView<Pixel = T>,
|
||||
dst_row: &mut [T],
|
||||
mut src_x: usize,
|
||||
coeffs_chunk: CoefficientsI32Chunk,
|
||||
@@ -77,7 +77,7 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
|
||||
let coeffs_2 = coeffs.chunks_exact(2);
|
||||
let coeffs_reminder = coeffs_2.remainder();
|
||||
|
||||
for (src_rows, two_coeffs) in src_img.iter_2_rows(y_start, max_y).zip(coeffs_2) {
|
||||
for (src_rows, two_coeffs) in src_view.iter_2_rows(y_start, max_y).zip(coeffs_2) {
|
||||
let src_rows = src_rows.map(|row| T::components(row));
|
||||
|
||||
for r in 0..2 {
|
||||
@@ -95,16 +95,17 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
|
||||
}
|
||||
|
||||
if let Some(&k) = coeffs_reminder.first() {
|
||||
let s_row = src_img.get_row(y_start + y).unwrap();
|
||||
let components = T::components(s_row);
|
||||
let coeff_i64x2 = i64x2_splat(k as i64);
|
||||
if let Some(s_row) = src_view.iter_rows(y_start + y).next() {
|
||||
let components = T::components(s_row);
|
||||
let coeff_i64x2 = i64x2_splat(k as i64);
|
||||
|
||||
for x in 0..2 {
|
||||
let source = wasm32_utils::load_v128(components, src_x + x * 8);
|
||||
for i in 0..4 {
|
||||
let c_i64x2 = i8x16_swizzle(source, c_shuffles[i]);
|
||||
sums[i][x] =
|
||||
i64x2_add(sums[i][x], wasm32_utils::i64x2_mul_lo(c_i64x2, coeff_i64x2));
|
||||
for x in 0..2 {
|
||||
let source = wasm32_utils::load_v128(components, src_x + x * 8);
|
||||
for i in 0..4 {
|
||||
let c_i64x2 = i8x16_swizzle(source, c_shuffles[i]);
|
||||
sums[i][x] =
|
||||
i64x2_add(sums[i][x], wasm32_utils::i64x2_mul_lo(c_i64x2, coeff_i64x2));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -132,7 +133,7 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
|
||||
let coeffs_2 = coeffs.chunks_exact(2);
|
||||
let coeffs_reminder = coeffs_2.remainder();
|
||||
|
||||
for (src_rows, two_coeffs) in src_img.iter_2_rows(y_start, max_y).zip(coeffs_2) {
|
||||
for (src_rows, two_coeffs) in src_view.iter_2_rows(y_start, max_y).zip(coeffs_2) {
|
||||
let src_rows = src_rows.map(|row| T::components(row));
|
||||
let coeffs_i64 = [
|
||||
i64x2_splat(two_coeffs[0] as i64),
|
||||
@@ -151,13 +152,14 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
|
||||
}
|
||||
|
||||
if let Some(&k) = coeffs_reminder.first() {
|
||||
let s_row = src_img.get_row(y_start + y).unwrap();
|
||||
let components = T::components(s_row);
|
||||
let coeff_i64x2 = i64x2_splat(k as i64);
|
||||
let source = wasm32_utils::load_v128(components, src_x);
|
||||
for i in 0..4 {
|
||||
let c_i64x2 = i8x16_swizzle(source, c_shuffles[i]);
|
||||
sums[i] = i64x2_add(sums[i], wasm32_utils::i64x2_mul_lo(c_i64x2, coeff_i64x2));
|
||||
if let Some(s_row) = src_view.iter_rows(y_start + y).next() {
|
||||
let components = T::components(s_row);
|
||||
let coeff_i64x2 = i64x2_splat(k as i64);
|
||||
let source = wasm32_utils::load_v128(components, src_x);
|
||||
for i in 0..4 {
|
||||
let c_i64x2 = i8x16_swizzle(source, c_shuffles[i]);
|
||||
sums[i] = i64x2_add(sums[i], wasm32_utils::i64x2_mul_lo(c_i64x2, coeff_i64x2));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -186,7 +188,7 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
|
||||
let coeffs_2 = coeffs.chunks_exact(2);
|
||||
let coeffs_reminder = coeffs_2.remainder();
|
||||
|
||||
for (src_rows, two_coeffs) in src_img.iter_2_rows(y_start, max_y).zip(coeffs_2) {
|
||||
for (src_rows, two_coeffs) in src_view.iter_2_rows(y_start, max_y).zip(coeffs_2) {
|
||||
let src_rows = src_rows.map(|row| T::components(row));
|
||||
let coeffs_i64 = [
|
||||
i64x2_splat(two_coeffs[0] as i64),
|
||||
@@ -203,15 +205,16 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
|
||||
}
|
||||
|
||||
if let Some(&k) = coeffs_reminder.first() {
|
||||
let s_row = src_img.get_row(y_start + y).unwrap();
|
||||
let components = T::components(s_row);
|
||||
let coeff_i64x2 = i64x2_splat(k as i64);
|
||||
if let Some(s_row) = src_view.iter_rows(y_start + y).next() {
|
||||
let components = T::components(s_row);
|
||||
let coeff_i64x2 = i64x2_splat(k as i64);
|
||||
|
||||
let comp_x4 = components.get_unchecked(src_x..src_x + 4);
|
||||
let c_i64x2 = i64x2(comp_x4[0] as i64, comp_x4[1] as i64);
|
||||
c01 = i64x2_add(c01, wasm32_utils::i64x2_mul_lo(c_i64x2, coeff_i64x2));
|
||||
let c_i64x2 = i64x2(comp_x4[2] as i64, comp_x4[3] as i64);
|
||||
c23 = i64x2_add(c23, wasm32_utils::i64x2_mul_lo(c_i64x2, coeff_i64x2));
|
||||
let comp_x4 = components.get_unchecked(src_x..src_x + 4);
|
||||
let c_i64x2 = i64x2(comp_x4[0] as i64, comp_x4[1] as i64);
|
||||
c01 = i64x2_add(c01, wasm32_utils::i64x2_mul_lo(c_i64x2, coeff_i64x2));
|
||||
let c_i64x2 = i64x2(comp_x4[2] as i64, comp_x4[3] as i64);
|
||||
c23 = i64x2_add(c23, wasm32_utils::i64x2_mul_lo(c_i64x2, coeff_i64x2));
|
||||
}
|
||||
}
|
||||
|
||||
let mut dst_ptr = dst_chunk.as_mut_ptr();
|
||||
@@ -232,7 +235,7 @@ unsafe fn vert_convolution_into_one_row_u16<T: PixelExt<Component = u16>>(
|
||||
if !dst_u16.is_empty() {
|
||||
let initial = 1 << (precision - 1);
|
||||
convolution_by_u16(
|
||||
src_img, normalizer, initial, dst_u16, src_x, y_start, coeffs,
|
||||
src_view, normalizer, initial, dst_u16, src_x, y_start, coeffs,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2,14 +2,14 @@ use std::arch::wasm32::*;
|
||||
|
||||
use crate::convolution::vertical_u8::native;
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::pixels::PixelExt;
|
||||
use crate::pixels::InnerPixel;
|
||||
use crate::wasm32_utils;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn vert_convolution<T: PixelExt<Component = u8>>(
|
||||
src_image: &ImageView<T>,
|
||||
dst_image: &mut ImageViewMut<T>,
|
||||
pub(crate) fn vert_convolution<T: InnerPixel<Component = u8>>(
|
||||
src_view: &impl ImageView<Pixel = T>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = T>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
@@ -17,18 +17,18 @@ pub(crate) fn vert_convolution<T: PixelExt<Component = u8>>(
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let src_x = offset as usize * T::count_of_components();
|
||||
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
for (dst_row, coeffs_chunk) in dst_rows.zip(coefficients_chunks) {
|
||||
unsafe {
|
||||
vert_convolution_into_one_row_u8(src_image, dst_row, src_x, coeffs_chunk, &normalizer);
|
||||
vert_convolution_into_one_row_u8(src_view, dst_row, src_x, coeffs_chunk, &normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
|
||||
src_img: &ImageView<T>,
|
||||
unsafe fn vert_convolution_into_one_row_u8<T: InnerPixel<Component = u8>>(
|
||||
src_view: &impl ImageView<Pixel = T>,
|
||||
dst_row: &mut [T],
|
||||
mut src_x: usize,
|
||||
coeffs_chunk: optimisations::CoefficientsI16Chunk,
|
||||
@@ -56,7 +56,7 @@ unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
|
||||
|
||||
let mut y: u32 = 0;
|
||||
|
||||
for src_rows in src_img.iter_2_rows(y_start, max_y) {
|
||||
for src_rows in src_view.iter_2_rows(y_start, max_y) {
|
||||
let components1 = T::components(src_rows[0]);
|
||||
let components2 = T::components(src_rows[1]);
|
||||
|
||||
@@ -107,41 +107,42 @@ unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
|
||||
}
|
||||
|
||||
if let Some(&k) = coeffs.get(y as usize) {
|
||||
let s_row = src_img.get_row(y_start + y).unwrap();
|
||||
let components = T::components(s_row);
|
||||
let mmk = i32x4_splat(k as i32);
|
||||
if let Some(s_row) = src_view.iter_rows(y_start + y).next() {
|
||||
let components = T::components(s_row);
|
||||
let mmk = i32x4_splat(k as i32);
|
||||
|
||||
let source1 = wasm32_utils::load_v128(components, src_x); // top line
|
||||
let source1 = wasm32_utils::load_v128(components, src_x); // top line
|
||||
|
||||
let source = i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(
|
||||
source1, ZERO,
|
||||
);
|
||||
let pix = i16x8_extend_low_u8x16(source);
|
||||
sss0 = i32x4_add(sss0, i32x4_dot_i16x8(pix, mmk));
|
||||
let pix = i16x8_extend_high_u8x16(source);
|
||||
sss1 = i32x4_add(sss1, i32x4_dot_i16x8(pix, mmk));
|
||||
let source = i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(
|
||||
source1, ZERO,
|
||||
);
|
||||
let pix = i16x8_extend_low_u8x16(source);
|
||||
sss0 = i32x4_add(sss0, i32x4_dot_i16x8(pix, mmk));
|
||||
let pix = i16x8_extend_high_u8x16(source);
|
||||
sss1 = i32x4_add(sss1, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
let source = i16x8_extend_high_u8x16(source1);
|
||||
let pix = i16x8_extend_low_u8x16(source);
|
||||
sss2 = i32x4_add(sss2, i32x4_dot_i16x8(pix, mmk));
|
||||
let pix = i16x8_extend_high_u8x16(source);
|
||||
sss3 = i32x4_add(sss3, i32x4_dot_i16x8(pix, mmk));
|
||||
let source = i16x8_extend_high_u8x16(source1);
|
||||
let pix = i16x8_extend_low_u8x16(source);
|
||||
sss2 = i32x4_add(sss2, i32x4_dot_i16x8(pix, mmk));
|
||||
let pix = i16x8_extend_high_u8x16(source);
|
||||
sss3 = i32x4_add(sss3, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
let source1 = wasm32_utils::load_v128(components, src_x + 16); // top line
|
||||
let source1 = wasm32_utils::load_v128(components, src_x + 16); // top line
|
||||
|
||||
let source = i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(
|
||||
source1, ZERO,
|
||||
);
|
||||
let pix = i16x8_extend_low_u8x16(source);
|
||||
sss4 = i32x4_add(sss4, i32x4_dot_i16x8(pix, mmk));
|
||||
let pix = i16x8_extend_high_u8x16(source);
|
||||
sss5 = i32x4_add(sss5, i32x4_dot_i16x8(pix, mmk));
|
||||
let source = i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(
|
||||
source1, ZERO,
|
||||
);
|
||||
let pix = i16x8_extend_low_u8x16(source);
|
||||
sss4 = i32x4_add(sss4, i32x4_dot_i16x8(pix, mmk));
|
||||
let pix = i16x8_extend_high_u8x16(source);
|
||||
sss5 = i32x4_add(sss5, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
let source = i16x8_extend_high_u8x16(source1);
|
||||
let pix = i16x8_extend_low_u8x16(source);
|
||||
sss6 = i32x4_add(sss6, i32x4_dot_i16x8(pix, mmk));
|
||||
let pix = i16x8_extend_high_u8x16(source);
|
||||
sss7 = i32x4_add(sss7, i32x4_dot_i16x8(pix, mmk));
|
||||
let source = i16x8_extend_high_u8x16(source1);
|
||||
let pix = i16x8_extend_low_u8x16(source);
|
||||
sss6 = i32x4_add(sss6, i32x4_dot_i16x8(pix, mmk));
|
||||
let pix = i16x8_extend_high_u8x16(source);
|
||||
sss7 = i32x4_add(sss7, i32x4_dot_i16x8(pix, mmk));
|
||||
}
|
||||
}
|
||||
|
||||
// This version of code works faster.
|
||||
@@ -180,7 +181,7 @@ unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
|
||||
let mut sss1 = initial; // right row
|
||||
let mut y: u32 = 0;
|
||||
|
||||
for src_rows in src_img.iter_2_rows(y_start, max_y) {
|
||||
for src_rows in src_view.iter_2_rows(y_start, max_y) {
|
||||
let components1 = T::components(src_rows[0]);
|
||||
let components2 = T::components(src_rows[1]);
|
||||
// Load two coefficients at once
|
||||
@@ -201,19 +202,20 @@ unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
|
||||
}
|
||||
|
||||
if let Some(&k) = coeffs.get(y as usize) {
|
||||
let s_row = src_img.get_row(y_start + y).unwrap();
|
||||
let components = T::components(s_row);
|
||||
let mmk = i32x4_splat(k as i32);
|
||||
if let Some(s_row) = src_view.iter_rows(y_start + y).next() {
|
||||
let components = T::components(s_row);
|
||||
let mmk = i32x4_splat(k as i32);
|
||||
|
||||
let source1 = wasm32_utils::loadl_i64(components, src_x); // top line
|
||||
let source1 = wasm32_utils::loadl_i64(components, src_x); // top line
|
||||
|
||||
let source = i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(
|
||||
source1, ZERO,
|
||||
);
|
||||
let pix = i16x8_extend_low_u8x16(source);
|
||||
sss0 = i32x4_add(sss0, i32x4_dot_i16x8(pix, mmk));
|
||||
let pix = i16x8_extend_high_u8x16(source);
|
||||
sss1 = i32x4_add(sss1, i32x4_dot_i16x8(pix, mmk));
|
||||
let source = i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(
|
||||
source1, ZERO,
|
||||
);
|
||||
let pix = i16x8_extend_low_u8x16(source);
|
||||
sss0 = i32x4_add(sss0, i32x4_dot_i16x8(pix, mmk));
|
||||
let pix = i16x8_extend_high_u8x16(source);
|
||||
sss1 = i32x4_add(sss1, i32x4_dot_i16x8(pix, mmk));
|
||||
}
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
@@ -238,7 +240,7 @@ unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
|
||||
let mut sss = initial;
|
||||
let mut y: u32 = 0;
|
||||
|
||||
for src_rows in src_img.iter_2_rows(y_start, max_y) {
|
||||
for src_rows in src_view.iter_2_rows(y_start, max_y) {
|
||||
let components1 = T::components(src_rows[0]);
|
||||
let components2 = T::components(src_rows[1]);
|
||||
// Load two coefficients at once
|
||||
@@ -257,11 +259,12 @@ unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
|
||||
}
|
||||
|
||||
if let Some(&k) = coeffs.get(y as usize) {
|
||||
let s_row = src_img.get_row(y_start + y).unwrap();
|
||||
let components = T::components(s_row);
|
||||
let pix = wasm32_utils::i32x4_extend_low_ptr_u8(components, src_x);
|
||||
let mmk = i32x4_splat(k as i32);
|
||||
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
|
||||
if let Some(s_row) = src_view.iter_rows(y_start + y).next() {
|
||||
let components = T::components(s_row);
|
||||
let pix = wasm32_utils::i32x4_extend_low_ptr_u8(components, src_x);
|
||||
let mmk = i32x4_splat(k as i32);
|
||||
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
|
||||
}
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
@@ -281,7 +284,7 @@ unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
|
||||
dst_u8 = dst_chunks_4.into_remainder();
|
||||
if !dst_u8.is_empty() {
|
||||
native::convolution_by_u8(
|
||||
src_img,
|
||||
src_view,
|
||||
normalizer,
|
||||
1 << (precision - 1),
|
||||
dst_u8,
|
||||
|
||||
Reference in New Issue
Block a user