- CpuExtension::Wasm32 renamed into CpuExtension::Simd128.

- Optimized MulDiv implementation for Wasm32 SIMD128.
This commit is contained in:
Kirill Kuzminykh
2023-01-29 00:43:25 +04:00
parent b765af521c
commit 76a87bc305
35 changed files with 436 additions and 465 deletions
+3
View File
@@ -2,6 +2,9 @@
## Benchmarks
- Added support of optimisation with helps of `Wasm32 SIMD128` for
all type of images exclude `I32` and `F32`
(thanks to @cdmurph32, [#11](https://github.com/Cykooz/fast_image_resize/pull/11)).
- Benchmark framework `glassbench` replaced by `criterion`.
- Added report with results of benchmarks for `wasm32-wasi` target.
Generated
+10 -10
View File
@@ -95,9 +95,9 @@ checksum = "37b2a672a2cb129a2e41c10b1224bb368f9f37a2b16b612598138befd7b37eb5"
[[package]]
name = "cc"
version = "1.0.78"
version = "1.0.79"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "a20104e2335ce8a659d6dd92a51a767a0c062599c73b343fd152cb401e828c3d"
checksum = "50d30906286121d95be3d479533b458f87493b30a4b5f79a607db8f5d11aa91f"
[[package]]
name = "cfg-if"
@@ -146,9 +146,9 @@ dependencies = [
[[package]]
name = "clap"
version = "4.1.1"
version = "4.1.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "4ec7a4128863c188deefe750ac1d1dfe66c236909f845af04beed823638dc1b2"
checksum = "f13b9c79b5d1dd500d20ef541215a6423c75829ef43117e1b4d17fd8af0b5d76"
dependencies = [
"bitflags",
"clap_derive",
@@ -165,7 +165,7 @@ version = "2.0.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "23e2b6c3dcdb73299f48ae05b294da14e2f560b3ed2c09e742269eb1b22af231"
dependencies = [
"clap 4.1.1",
"clap 4.1.4",
"log",
]
@@ -300,9 +300,9 @@ checksum = "7a81dae078cea95a014a339291cec439d2f232ebe854a9d672b796c6afafa9b7"
[[package]]
name = "either"
version = "1.8.0"
version = "1.8.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "90e5c1c8368803113bf0c9584fc495a58b86dc8a29edbf8fe877d21d9507e797"
checksum = "7fcaabb2fef8c910e7f4c7ce9f67a1283a1715879a7c230ca9d6d1ae31f16d91"
[[package]]
name = "env_logger"
@@ -810,9 +810,9 @@ dependencies = [
[[package]]
name = "rayon-core"
version = "1.10.1"
version = "1.10.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "cac410af5d00ab6884528b4ab69d1e8e146e8d471201800fa1b4524126de6ad3"
checksum = "356a0625f1954f730c0201cdab48611198dc6ce21f4acff55089b5a78e6e835b"
dependencies = [
"crossbeam-channel",
"crossbeam-deque",
@@ -852,7 +852,7 @@ name = "resizer"
version = "0.1.0"
dependencies = [
"anyhow",
"clap 4.1.1",
"clap 4.1.4",
"clap-verbosity-flag",
"env_logger",
"fast_image_resize",
+9 -9
View File
@@ -20,8 +20,8 @@ exclude = ["/data"]
[dependencies]
num-traits = "0.2.15"
thiserror = "1.0.37"
num-traits = "0.2"
thiserror = "1.0"
[features]
for_test = []
@@ -30,17 +30,17 @@ for_test = []
fast_image_resize = { path = ".", features = ["for_test"] }
image = "0.24.5"
resize = "0.7.4"
rgb = "0.8.34"
png = "0.17.7"
testing = { path = "testing" }
rgb = "0.8"
png = "0.17"
serde = { version = "1.0", features = ["serde_derive"] }
serde_json = "1.0"
serde_json = "1"
walkdir = "2"
itertools = "0.10.5"
criterion = { version = "0.4.0", default-features = false, features = ["cargo_bench_support"] }
itertools = "0.10"
criterion = { version = "0.4", default-features = false, features = ["cargo_bench_support"] }
testing = { path = "testing" }
[target.'cfg(not(target_arch = "wasm32"))'.dev-dependencies]
nix = { version = "0.26.1", default-features = false, features = ["sched"] }
nix = { version = "0.26", default-features = false, features = ["sched"] }
[[bench]]
+16 -15
View File
@@ -10,25 +10,25 @@ Rust library for fast image resizing with using of SIMD instructions.
Supported pixel formats and available optimisations:
| Format | Description | Native Rust | SSE4.1 | AVX2 | Neon |
|:------:|:--------------------------------------------------------------|:-----------:|:------:|:----:|:----:|
| U8 | One `u8` component per pixel (e.g. L) | + | + | + | + |
| U8x2 | Two `u8` components per pixel (e.g. LA) | + | + | + | + |
| U8x3 | Three `u8` components per pixel (e.g. RGB) | + | + | + | + |
| U8x4 | Four `u8` components per pixel (e.g. RGBA, RGBx, CMYK) | + | + | + | + |
| U16 | One `u16` components per pixel (e.g. L16) | + | + | + | + |
| U16x2 | Two `u16` components per pixel (e.g. LA16) | + | + | + | + |
| U16x3 | Three `u16` components per pixel (e.g. RGB16) | + | + | + | + |
| U16x4 | Four `u16` components per pixel (e.g. RGBA16, RGBx16, CMYK16) | + | + | + | + |
| I32 | One `i32` component per pixel | + | - | - | - |
| F32 | One `f32` component per pixel | + | - | - | - |
| Format | Description | SSE4.1 | AVX2 | Neon | Wasm32 SIMD128 |
|:------:|:--------------------------------------------------------------|:------:|:----:|:----:|:--------------:|
| U8 | One `u8` component per pixel (e.g. L) | + | + | + | + |
| U8x2 | Two `u8` components per pixel (e.g. LA) | + | + | + | + |
| U8x3 | Three `u8` components per pixel (e.g. RGB) | + | + | + | + |
| U8x4 | Four `u8` components per pixel (e.g. RGBA, RGBx, CMYK) | + | + | + | + |
| U16 | One `u16` components per pixel (e.g. L16) | + | + | + | + |
| U16x2 | Two `u16` components per pixel (e.g. LA16) | + | + | + | + |
| U16x3 | Three `u16` components per pixel (e.g. RGB16) | + | + | + | + |
| U16x4 | Four `u16` components per pixel (e.g. RGBA16, RGBx16, CMYK16) | + | + | + | + |
| I32 | One `i32` component per pixel | - | - | - | - |
| F32 | One `f32` component per pixel | - | - | - | - |
## Colorspace
Resizer from this crate does not convert image into linear colorspace
during resize process. If it is important for you to resize images with a
non-linear color space (e.g. sRGB) correctly, then you need to convert
it to a linear color space before resizing and convert back a color space of
non-linear color space (e.g. sRGB) correctly, then you hove to convert
it to a linear color space before resizing and convert back to the color space of
result image. [Read more](https://legacy.imagemagick.org/Usage/resize/#resize_colorspace)
about resizing with respect to color space.
@@ -45,7 +45,8 @@ that converts images from sRGB or gamma 2.2 into linear colorspace and back.
- [All x86_64 benchmarks.](https://github.com/Cykooz/fast_image_resize/blob/main/benchmarks-x86_64.md)
- [All arm64 benchmarks.](https://github.com/Cykooz/fast_image_resize/blob/main/benchmarks-arm64.md)
- [All wasm32 benchmarks.](https://github.com/Cykooz/fast_image_resize/blob/main/benchmarks-wasm32.md)
-
Rust libraries used to compare of resizing speed:
- image (<https://crates.io/crates/image>)
+21 -18
View File
@@ -3,6 +3,7 @@ use std::num::NonZeroU32;
use fast_image_resize::MulDiv;
use fast_image_resize::PixelType;
use fast_image_resize::{CpuExtensions, Image};
use testing::cpu_ext_into_str;
mod utils;
@@ -25,7 +26,6 @@ fn multiplies_alpha(
bench_group: &mut utils::BenchGroup,
pixel_type: PixelType,
cpu_extensions: CpuExtensions,
ext_name: &str,
) {
let sample_size = 100;
let width = NonZeroU32::new(4096).unwrap();
@@ -49,8 +49,8 @@ fn multiplies_alpha(
utils::bench(
bench_group,
sample_size,
format!("Multiplies alpha {:?}", pixel_type),
ext_name,
format!("Multiplies alpha {pixel_type:?}"),
cpu_ext_into_str(cpu_extensions),
|bencher| {
bencher.iter(|| {
alpha_mul_div
@@ -64,8 +64,8 @@ fn multiplies_alpha(
utils::bench(
bench_group,
sample_size,
format!("Multiplies alpha inplace {:?}", pixel_type),
ext_name,
format!("Multiplies alpha inplace {pixel_type:?}"),
cpu_ext_into_str(cpu_extensions),
|bencher| {
let mut image = src_image.copy();
let mut view = image.view_mut();
@@ -80,7 +80,6 @@ fn divides_alpha(
bench_group: &mut utils::BenchGroup,
pixel_type: PixelType,
cpu_extensions: CpuExtensions,
ext_name: &str,
) {
let sample_size = 100;
let width = NonZeroU32::new(4095).unwrap();
@@ -104,8 +103,8 @@ fn divides_alpha(
utils::bench(
bench_group,
sample_size,
format!("Divides alpha {:?}", pixel_type),
ext_name,
format!("Divides alpha {pixel_type:?}"),
cpu_ext_into_str(cpu_extensions),
|bencher| {
bencher.iter(|| {
alpha_mul_div
@@ -119,8 +118,8 @@ fn divides_alpha(
utils::bench(
bench_group,
sample_size,
format!("Divides alpha inplace {:?}", pixel_type),
ext_name,
format!("Divides alpha inplace {pixel_type:?}"),
cpu_ext_into_str(cpu_extensions),
|bencher| {
let mut image = src_image.copy();
let mut view = image.view_mut();
@@ -138,24 +137,28 @@ fn bench_alpha(bench_group: &mut utils::BenchGroup) {
PixelType::U16x2,
PixelType::U16x4,
];
let mut cpu_ext_and_name = vec![(CpuExtensions::None, "rust")];
let mut cpu_extensions = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_ext_and_name.push((CpuExtensions::Sse4_1, "sse4.1"));
cpu_ext_and_name.push((CpuExtensions::Avx2, "avx2"));
cpu_extensions.push(CpuExtensions::Sse4_1);
cpu_extensions.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_ext_and_name.push((CpuExtensions::Neon, "neon"));
cpu_extensions.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions.push(CpuExtensions::Simd128);
}
for pixel_type in pixel_types {
for &(cpu_ext, ext_name) in cpu_ext_and_name.iter() {
multiplies_alpha(bench_group, pixel_type, cpu_ext, ext_name);
for &cpu_ext in cpu_extensions.iter() {
multiplies_alpha(bench_group, pixel_type, cpu_ext);
}
}
for pixel_type in pixel_types {
for &(cpu_ext, ext_name) in cpu_ext_and_name.iter() {
divides_alpha(bench_group, pixel_type, cpu_ext, ext_name);
for &cpu_ext in cpu_extensions.iter() {
divides_alpha(bench_group, pixel_type, cpu_ext);
}
}
}
+36 -31
View File
@@ -23,7 +23,27 @@ fn native_nearest_u8x4_bench(bench_group: &mut utils::BenchGroup) {
unsafe {
resizer.set_cpu_extensions(CpuExtensions::None);
}
bench_group.bench_function("nearest wo SIMD", |bencher| {
utils::bench(bench_group, 100, "U8x4 Nearest", "rust", |bencher| {
bencher.iter(|| {
resizer.resize(&src_image, &mut dst_image).unwrap();
})
});
}
fn native_nearest_u8_bench(bench_group: &mut utils::BenchGroup) {
let image = U8::load_big_src_image();
let mut res_image = Image::new(
NonZeroU32::new(NEW_WIDTH).unwrap(),
NonZeroU32::new(NEW_HEIGHT).unwrap(),
image.pixel_type(),
);
let src_image = image.view();
let mut dst_image = res_image.view_mut();
let mut resizer = Resizer::new(ResizeAlg::Nearest);
unsafe {
resizer.set_cpu_extensions(CpuExtensions::None);
}
utils::bench(bench_group, 100, "U8 Nearest", "rust", |bencher| {
bencher.iter(|| {
resizer.resize(&src_image, &mut dst_image).unwrap();
})
@@ -47,37 +67,17 @@ fn downscale_bench(
unsafe {
resizer.set_cpu_extensions(cpu_extensions);
}
let bench_name = &format!(
"{:?}-{:?}-{}",
image.pixel_type(),
filter_type,
utils::bench(
bench_group,
100,
&format!("{:?} {:?}", image.pixel_type(), filter_type),
cpu_ext_into_str(cpu_extensions),
|bencher| {
bencher.iter(|| {
resizer.resize(&src_image, &mut dst_image).unwrap();
})
},
);
bench_group.bench_function(bench_name, |bencher| {
bencher.iter(|| {
resizer.resize(&src_image, &mut dst_image).unwrap();
})
});
}
fn native_nearest_u8_bench(bench_group: &mut utils::BenchGroup) {
let image = U8::load_big_src_image();
let mut res_image = Image::new(
NonZeroU32::new(NEW_WIDTH).unwrap(),
NonZeroU32::new(NEW_HEIGHT).unwrap(),
image.pixel_type(),
);
let src_image = image.view();
let mut dst_image = res_image.view_mut();
let mut resizer = Resizer::new(ResizeAlg::Nearest);
unsafe {
resizer.set_cpu_extensions(CpuExtensions::None);
}
bench_group.bench_function("u8 nearest wo SIMD", |bencher| {
bencher.iter(|| {
resizer.resize(&src_image, &mut dst_image).unwrap();
})
});
}
pub fn resize_bench(bench_group: &mut utils::BenchGroup) {
@@ -102,6 +102,10 @@ pub fn resize_bench(bench_group: &mut utils::BenchGroup) {
{
cpu_extensions.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions.push(CpuExtensions::Simd128);
}
for pixel_type in pixel_types {
for &cpu_extension in cpu_extensions.iter() {
let image = match pixel_type {
@@ -125,5 +129,6 @@ pub fn resize_bench(bench_group: &mut utils::BenchGroup) {
}
fn main() {
utils::run_bench(resize_bench, "Resize");
let results = utils::run_bench(resize_bench, "Resize");
println!("{}", utils::build_md_table(&results));
}
+11 -7
View File
@@ -4,7 +4,7 @@ use image::{imageops, ImageBuffer};
use crate::utils::bencher::{bench, BenchGroup};
use fast_image_resize::{CpuExtensions, FilterType, Image, MulDiv, ResizeAlg, Resizer};
use testing::{nonzero, PixelTestingExt};
use testing::{cpu_ext_into_str, nonzero, PixelTestingExt};
const ALG_NAMES: [&str; 4] = ["Nearest", "Bilinear", "CatmullRom", "Lanczos3"];
const NEW_WIDTH: u32 = 852;
@@ -86,17 +86,21 @@ pub fn fir_resize<P: PixelTestingExt>(bench_group: &mut BenchGroup) {
);
let mut dst_view = dst_image.view_mut();
let mut cpu_ext_and_name = vec![(CpuExtensions::None, "rust")];
let mut cpu_extensions = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_ext_and_name.push((CpuExtensions::Sse4_1, "sse4.1"));
cpu_ext_and_name.push((CpuExtensions::Avx2, "avx2"));
cpu_extensions.push(CpuExtensions::Sse4_1);
cpu_extensions.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_ext_and_name.push((CpuExtensions::Neon, "neon"));
cpu_extensions.push(CpuExtensions::Neon);
}
for (cpu_ext, ext_name) in cpu_ext_and_name {
#[cfg(target_arch = "wasm32")]
{
cpu_extensions.push(CpuExtensions::Simd128);
}
for cpu_ext in cpu_extensions {
for alg_name in ALG_NAMES {
let resize_alg = match alg_name {
"Nearest" => {
@@ -120,7 +124,7 @@ pub fn fir_resize<P: PixelTestingExt>(bench_group: &mut BenchGroup) {
bench(
bench_group,
sample_size,
format!("fir {}", ext_name),
format!("fir {}", cpu_ext_into_str(cpu_ext)),
alg_name,
|bencher| {
fast_resizer.reset_internal_buffers();
+1 -1
View File
@@ -4,7 +4,7 @@ Environment:
- CPU: Neoverse-N1 2GHz (Oracle Cloud Compute, VM.Standard.A1.Flex)
- Ubuntu 22.04 (linux 5.15.0)
- Rust 1.66.1
- Rust 1.67
- criterion = "0.4"
- fast_image_resize = "2.4.0"
+4 -4
View File
@@ -5,7 +5,7 @@ Environment:
- CPU: AMD Ryzen 9 5950X
- RAM: DDR4 3800 MHz
- Ubuntu 22.04 (linux 5.15.0)
- Rust 1.66.1
- Rust 1.67
- wasmtime = "5.0.0"
- criterion = "0.4"
- fast_image_resize = "2.4.0"
@@ -70,9 +70,9 @@ Pipeline:
<!-- bench_compare_l start -->
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|----------|:-------:|:--------:|:----------:|:--------:|
| image | 20.67 | 113.49 | 204.72 | 296.08 |
| resize | - | 30.10 | 60.87 | 90.92 |
| fir rust | 0.21 | 24.54 | 40.10 | 57.36 |
| image | 55.66 | 203.36 | 379.55 | 525.72 |
| resize | - | 45.42 | 84.39 | 122.73 |
| fir rust | 0.29 | 35.84 | 58.16 | 81.65 |
<!-- bench_compare_l end -->
### Resize LA8 image (U8x2) 4928x3279 => 852x567
+1 -1
View File
@@ -5,7 +5,7 @@ Environment:
- CPU: AMD Ryzen 9 5950X
- RAM: DDR4 3800 MHz
- Ubuntu 22.04 (linux 5.15.0)
- Rust 1.66.1
- Rust 1.67
- criterion = "0.4"
- fast_image_resize = "2.4.0"
+4 -10
View File
@@ -8,7 +8,6 @@ Install additional toolchains.
- Wasm32:
```shell
rustup target add wasm32-wasi
cargo install cargo-wasi
```
Install [Wasmtime](https://wasmtime.dev/).
@@ -36,28 +35,23 @@ Specify build target in `.cargo/config.toml` file.
target = "wasm32-wasi"
```
Template of command to run `cargo` commands with using `Wasmtime`:
```
CARGO_TARGET_WASM32_WASI_RUNNER="wasmtime --dir=." cargo wasi <any cargo command>
```
Run tests:
```shell
CARGO_TARGET_WASM32_WASI_RUNNER="wasmtime --dir=." cargo wasi test
CARGO_TARGET_WASM32_WASI_RUNNER="wasmtime --dir=. --" cargo test
```
Run tests without saving result images as files in `./data` directory:
```shell
CARGO_TARGET_WASM32_WASI_RUNNER="wasmtime --dir=. --env DONT_SAVE_RESULT=1" cargo wasi test
CARGO_TARGET_WASM32_WASI_RUNNER="wasmtime --dir=. --env DONT_SAVE_RESULT=1 --" cargo test
```
Run a specific benchmark in `quick` mode:
```shell
CARGO_TARGET_WASM32_WASI_RUNNER="wasmtime --dir=." cargo wasi bench --bench bench_resize -- --quick
CARGO_TARGET_WASM32_WASI_RUNNER="wasmtime --dir=. --" cargo bench --bench bench_resize -- --color=always --quick
```
Run benchmarks to compare with other crates for image resizing and write results into
report files, such as `./benchmarks-x86_64.md`:
```shell
CARGO_TARGET_WASM32_WASI_RUNNER="wasmtime --dir=. --env WRITE_COMPARE_RESULT=1" cargo wasi bench -- Compare
CARGO_TARGET_WASM32_WASI_RUNNER="wasmtime --dir=. --env WRITE_COMPARE_RESULT=1 --" cargo bench -- --color=always Compare
```
+4 -4
View File
@@ -28,7 +28,7 @@ impl AlphaMulDiv for U16x2 {
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => unsafe { neon::multiply_alpha(src_image, dst_image) },
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => unsafe { wasm32::multiply_alpha(src_image, dst_image) },
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha(src_image, dst_image) },
_ => native::multiply_alpha(src_image, dst_image),
}
}
@@ -42,7 +42,7 @@ impl AlphaMulDiv for U16x2 {
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => unsafe { neon::multiply_alpha_inplace(image) },
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => unsafe { wasm32::multiply_alpha_inplace(image) },
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha_inplace(image) },
_ => native::multiply_alpha_inplace(image),
}
}
@@ -60,7 +60,7 @@ impl AlphaMulDiv for U16x2 {
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => unsafe { neon::divide_alpha(src_image, dst_image) },
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => unsafe { wasm32::divide_alpha(src_image, dst_image) },
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha(src_image, dst_image) },
_ => native::divide_alpha(src_image, dst_image),
}
}
@@ -74,7 +74,7 @@ impl AlphaMulDiv for U16x2 {
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => unsafe { neon::divide_alpha_inplace(image) },
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => unsafe { wasm32::divide_alpha_inplace(image) },
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha_inplace(image) },
_ => native::divide_alpha_inplace(image),
}
}
+41 -49
View File
@@ -25,23 +25,18 @@ pub(crate) unsafe fn multiply_alpha_inplace(image: &mut ImageViewMut<U16x2>) {
}
#[inline]
pub(crate) unsafe fn multiply_alpha_row(src_row: &[U16x2], dst_row: &mut [U16x2]) {
#[target_feature(enable = "simd128")]
unsafe fn multiply_alpha_row(src_row: &[U16x2], dst_row: &mut [U16x2]) {
let src_chunks = src_row.chunks_exact(4);
let src_remainder = src_chunks.remainder();
let mut dst_chunks = dst_row.chunks_exact_mut(4);
let src_dst = src_chunks.zip(&mut dst_chunks);
foreach_with_pre_reading(
src_dst,
|(src, dst)| {
let pixels = v128_load(src.as_ptr() as *const v128);
let dst_ptr = dst.as_mut_ptr() as *mut v128;
(pixels, dst_ptr)
},
|(mut pixels, dst_ptr)| {
pixels = multiplies_alpha_4_pixels(pixels);
v128_store(dst_ptr, pixels);
},
);
// A simple for-loop in this case is faster than implementation with pre-reading
for (src, dst) in src_dst {
let mut pixels = v128_load(src.as_ptr() as *const v128);
pixels = multiply_alpha_4_pixels(pixels);
v128_store(dst.as_mut_ptr() as *mut v128, pixels);
}
if !src_remainder.is_empty() {
let dst_reminder = dst_chunks.into_remainder();
@@ -50,20 +45,15 @@ pub(crate) unsafe fn multiply_alpha_row(src_row: &[U16x2], dst_row: &mut [U16x2]
}
#[inline]
pub(crate) unsafe fn multiply_alpha_row_inplace(row: &mut [U16x2]) {
#[target_feature(enable = "simd128")]
unsafe fn multiply_alpha_row_inplace(row: &mut [U16x2]) {
let mut chunks = row.chunks_exact_mut(4);
foreach_with_pre_reading(
&mut chunks,
|chunk| {
let pixels = v128_load(chunk.as_ptr() as *const v128);
let dst_ptr = chunk.as_mut_ptr() as *mut v128;
(pixels, dst_ptr)
},
|(mut pixels, dst_ptr)| {
pixels = multiplies_alpha_4_pixels(pixels);
v128_store(dst_ptr, pixels);
},
);
// A simple for-loop in this case is faster than implementation with pre-reading
for chunk in &mut chunks {
let mut pixels = v128_load(chunk.as_ptr() as *const v128);
pixels = multiply_alpha_4_pixels(pixels);
v128_store(chunk.as_mut_ptr() as *mut v128, pixels);
}
let reminder = chunks.into_remainder();
if !reminder.is_empty() {
@@ -72,9 +62,9 @@ pub(crate) unsafe fn multiply_alpha_row_inplace(row: &mut [U16x2]) {
}
#[inline]
unsafe fn multiplies_alpha_4_pixels(pixels: v128) -> v128 {
const HALF: v128 = i32x4(0x8000, 0x8000, 0x8000, 0x8000);
#[target_feature(enable = "simd128")]
unsafe fn multiply_alpha_4_pixels(pixels: v128) -> v128 {
const HALF: v128 = u32x4(0x8000, 0x8000, 0x8000, 0x8000);
const MAX_ALPHA: v128 = u32x4(0xffff0000u32, 0xffff0000u32, 0xffff0000u32, 0xffff0000u32);
/*
|L0 A0 | |L1 A1 | |L2 A2 | |L3 A3 |
@@ -87,15 +77,15 @@ unsafe fn multiplies_alpha_4_pixels(pixels: v128) -> v128 {
let src_u32_lo = u32x4_extend_low_u16x8(pixels);
let factors = u32x4_extend_low_u16x8(factor_pixels);
let src_i32_lo = i32x4_add(i32x4_mul(src_u32_lo, factors), HALF);
let dst_i32_lo = i32x4_add(src_i32_lo, u32x4_shr(src_i32_lo, 16));
let dst_i32_lo = u32x4_shr(dst_i32_lo, 16);
let mut dst_i32_lo = u32x4_add(u32x4_mul(src_u32_lo, factors), HALF);
dst_i32_lo = u32x4_add(dst_i32_lo, u32x4_shr(dst_i32_lo, 16));
dst_i32_lo = u32x4_shr(dst_i32_lo, 16);
let src_u32_hi = u32x4_extend_high_u16x8(pixels);
let factors = u32x4_extend_high_u16x8(factor_pixels);
let src_i32_hi = i32x4_add(i32x4_mul(src_u32_hi, factors), HALF);
let dst_i32_hi = i32x4_add(src_i32_hi, u32x4_shr(src_i32_hi, 16));
let dst_i32_hi = u32x4_shr(dst_i32_hi, 16);
let mut dst_i32_hi = u32x4_add(u32x4_mul(src_u32_hi, factors), HALF);
dst_i32_hi = u32x4_add(dst_i32_hi, u32x4_shr(dst_i32_hi, 16));
dst_i32_hi = u32x4_shr(dst_i32_hi, 16);
u16x8_narrow_i32x4(dst_i32_lo, dst_i32_hi)
}
@@ -120,7 +110,9 @@ pub(crate) unsafe fn divide_alpha_inplace(image: &mut ImageViewMut<U16x2>) {
}
}
pub(crate) unsafe fn divide_alpha_row(src_row: &[U16x2], dst_row: &mut [U16x2]) {
#[inline]
#[target_feature(enable = "simd128")]
unsafe fn divide_alpha_row(src_row: &[U16x2], dst_row: &mut [U16x2]) {
let src_chunks = src_row.chunks_exact(4);
let src_remainder = src_chunks.remainder();
let mut dst_chunks = dst_row.chunks_exact_mut(4);
@@ -158,9 +150,11 @@ pub(crate) unsafe fn divide_alpha_row(src_row: &[U16x2], dst_row: &mut [U16x2])
}
}
pub(crate) unsafe fn divide_alpha_row_inplace(row: &mut [U16x2]) {
#[inline]
#[target_feature(enable = "simd128")]
unsafe fn divide_alpha_row_inplace(row: &mut [U16x2]) {
let mut chunks = row.chunks_exact_mut(4);
// Using a simple for-loop in this case is faster than implementation with pre-reading
// A simple for-loop in this case is as fast as implementation with pre-reading
for chunk in &mut chunks {
let mut pixels = v128_load(chunk.as_ptr() as *const v128);
pixels = divide_alpha_4_pixels(pixels);
@@ -185,11 +179,11 @@ pub(crate) unsafe fn divide_alpha_row_inplace(row: &mut [U16x2]) {
}
#[inline]
#[target_feature(enable = "simd128")]
unsafe fn divide_alpha_4_pixels(pixels: v128) -> v128 {
const ALPHA_MASK: v128 = u32x4(0xffff0000u32, 0xffff0000u32, 0xffff0000u32, 0xffff0000u32);
const LUMA_MASK: v128 = i32x4(0xffff, 0xffff, 0xffff, 0xffff);
const ALPHA_MASK: v128 = u32x4(0xffff0000, 0xffff0000, 0xffff0000, 0xffff0000);
const LUMA_MASK: v128 = u32x4(0xffff, 0xffff, 0xffff, 0xffff);
const ALPHA_MAX: v128 = f32x4(65535.0, 65535.0, 65535.0, 65535.0);
const ALPHA_SCALE_MAX: v128 = f32x4(2147483648f32, 2147483648f32, 2147483648f32, 2147483648f32);
/*
|L0 A0 | |L1 A1 | |L2 A2 | |L3 A3 |
|0001 0203| |0405 0607| |0809 1011| |1213 1415|
@@ -199,14 +193,12 @@ unsafe fn divide_alpha_4_pixels(pixels: v128) -> v128 {
let alpha_f32x4 = f32x4_convert_i32x4(u8x16_swizzle(pixels, ALPHA32_SH));
let luma_f32x4 = f32x4_convert_i32x4(v128_and(pixels, LUMA_MASK));
let scaled_luma_f32x4 = f32x4_mul(luma_f32x4, ALPHA_MAX);
let divided_luma_u32x4 = u32x4_trunc_sat_f32x4(f32x4_pmin(
f32x4_div(scaled_luma_f32x4, alpha_f32x4),
ALPHA_SCALE_MAX,
));
// In case of zero division the result will be u32::MAX or 0.
let divided_luma_u32x4 = u32x4_trunc_sat_f32x4(f32x4_div(scaled_luma_f32x4, alpha_f32x4));
// All u32::MAX values in arguments will interpreted as -1i32.
// u16x8_narrow_i32x4() converts all negative values into 0.
let divided_luma_u16 = u16x8_narrow_i32x4(divided_luma_u32x4, divided_luma_u32x4);
let alpha = v128_and(pixels, ALPHA_MASK);
u8x16_shuffle::<0, 1, 18, 19, 4, 5, 22, 23, 8, 9, 26, 27, 12, 13, 30, 31>(
divided_luma_u32x4,
alpha,
)
v128_or(u32x4_extend_low_u16x8(divided_luma_u16), alpha)
}
+4 -4
View File
@@ -28,7 +28,7 @@ impl AlphaMulDiv for U16x4 {
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => unsafe { neon::multiply_alpha(src_image, dst_image) },
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => unsafe { wasm32::multiply_alpha(src_image, dst_image) },
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha(src_image, dst_image) },
_ => native::multiply_alpha(src_image, dst_image),
}
}
@@ -42,7 +42,7 @@ impl AlphaMulDiv for U16x4 {
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => unsafe { neon::multiply_alpha_inplace(image) },
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => unsafe { wasm32::multiply_alpha_inplace(image) },
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha_inplace(image) },
_ => native::multiply_alpha_inplace(image),
}
}
@@ -60,7 +60,7 @@ impl AlphaMulDiv for U16x4 {
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => unsafe { neon::divide_alpha(src_image, dst_image) },
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => unsafe { wasm32::divide_alpha(src_image, dst_image) },
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha(src_image, dst_image) },
_ => native::divide_alpha(src_image, dst_image),
}
}
@@ -74,7 +74,7 @@ impl AlphaMulDiv for U16x4 {
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => unsafe { neon::divide_alpha_inplace(image) },
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => unsafe { wasm32::divide_alpha_inplace(image) },
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha_inplace(image) },
_ => native::divide_alpha_inplace(image),
}
}
+46 -41
View File
@@ -25,7 +25,8 @@ pub(crate) unsafe fn multiply_alpha_inplace(image: &mut ImageViewMut<U16x4>) {
}
#[inline]
pub(crate) unsafe fn multiply_alpha_row(src_row: &[U16x4], dst_row: &mut [U16x4]) {
#[target_feature(enable = "simd128")]
unsafe fn multiply_alpha_row(src_row: &[U16x4], dst_row: &mut [U16x4]) {
let src_chunks = src_row.chunks_exact(2);
let src_remainder = src_chunks.remainder();
let mut dst_chunks = dst_row.chunks_exact_mut(2);
@@ -50,7 +51,8 @@ pub(crate) unsafe fn multiply_alpha_row(src_row: &[U16x4], dst_row: &mut [U16x4]
}
#[inline]
pub(crate) unsafe fn multiply_alpha_row_inplace(row: &mut [U16x4]) {
#[target_feature(enable = "simd128")]
unsafe fn multiply_alpha_row_inplace(row: &mut [U16x4]) {
let mut chunks = row.chunks_exact_mut(2);
foreach_with_pre_reading(
&mut chunks,
@@ -72,11 +74,10 @@ pub(crate) unsafe fn multiply_alpha_row_inplace(row: &mut [U16x4]) {
}
#[inline]
#[target_feature(enable = "simd128")]
unsafe fn multiply_alpha_2_pixels(pixels: v128) -> v128 {
let zero = i64x2_splat(0);
let half = i32x4_splat(0x8000);
const MAX_A: i64 = 0xffff000000000000u64 as i64;
let max_alpha = i64x2_splat(MAX_A);
let half = u32x4_splat(0x8000);
let max_alpha = u64x2_splat(0xffff000000000000);
/*
|R0 G0 B0 A0 | |R1 G1 B1 A1 |
|0001 0203 0405 0607| |0809 1011 1213 1415|
@@ -86,19 +87,19 @@ unsafe fn multiply_alpha_2_pixels(pixels: v128) -> v128 {
let factor_pixels = u8x16_swizzle(pixels, FACTOR_MASK);
let factor_pixels = v128_or(factor_pixels, max_alpha);
let src_i32_lo = i16x8_shuffle::<0, 8, 1, 9, 2, 10, 3, 11>(pixels, zero);
let factors = i16x8_shuffle::<0, 8, 1, 9, 2, 10, 3, 11>(factor_pixels, zero);
let src_i32_lo = i32x4_add(i32x4_mul(src_i32_lo, factors), half);
let dst_i32_lo = i32x4_add(src_i32_lo, u32x4_shr(src_i32_lo, 16));
let dst_i32_lo = u32x4_shr(dst_i32_lo, 16);
let src_u32_lo = u32x4_extend_low_u16x8(pixels);
let factors = u32x4_extend_low_u16x8(factor_pixels);
let mut dst_u32_lo = u32x4_add(u32x4_mul(src_u32_lo, factors), half);
dst_u32_lo = u32x4_add(dst_u32_lo, u32x4_shr(dst_u32_lo, 16));
dst_u32_lo = u32x4_shr(dst_u32_lo, 16);
let src_i32_hi = i16x8_shuffle::<4, 12, 5, 13, 6, 14, 7, 15>(pixels, zero);
let factors = i16x8_shuffle::<4, 12, 5, 13, 6, 14, 7, 15>(factor_pixels, zero);
let src_i32_hi = i32x4_add(i32x4_mul(src_i32_hi, factors), half);
let dst_i32_hi = i32x4_add(src_i32_hi, u32x4_shr(src_i32_hi, 16));
let dst_i32_hi = u32x4_shr(dst_i32_hi, 16);
let src_u32_hi = u32x4_extend_high_u16x8(pixels);
let factors = u32x4_extend_high_u16x8(factor_pixels);
let mut dst_u32_hi = u32x4_add(u32x4_mul(src_u32_hi, factors), half);
dst_u32_hi = u32x4_add(dst_u32_hi, u32x4_shr(dst_u32_hi, 16));
dst_u32_hi = u32x4_shr(dst_u32_hi, 16);
u16x8_narrow_i32x4(dst_i32_lo, dst_i32_hi)
u16x8_narrow_i32x4(dst_u32_lo, dst_u32_hi)
}
// Divide
@@ -121,7 +122,9 @@ pub(crate) unsafe fn divide_alpha_inplace(image: &mut ImageViewMut<U16x4>) {
}
}
pub(crate) unsafe fn divide_alpha_row(src_row: &[U16x4], dst_row: &mut [U16x4]) {
#[inline]
#[target_feature(enable = "simd128")]
unsafe fn divide_alpha_row(src_row: &[U16x4], dst_row: &mut [U16x4]) {
let src_chunks = src_row.chunks_exact(2);
let src_remainder = src_chunks.remainder();
let mut dst_chunks = dst_row.chunks_exact_mut(2);
@@ -154,7 +157,9 @@ pub(crate) unsafe fn divide_alpha_row(src_row: &[U16x4], dst_row: &mut [U16x4])
}
}
pub(crate) unsafe fn divide_alpha_row_inplace(row: &mut [U16x4]) {
#[inline]
#[target_feature(enable = "simd128")]
unsafe fn divide_alpha_row_inplace(row: &mut [U16x4]) {
let mut chunks = row.chunks_exact_mut(2);
foreach_with_pre_reading(
&mut chunks,
@@ -182,39 +187,39 @@ pub(crate) unsafe fn divide_alpha_row_inplace(row: &mut [U16x4]) {
}
#[inline]
#[target_feature(enable = "simd128")]
unsafe fn divide_alpha_2_pixels(pixels: v128) -> v128 {
let zero = i64x2_splat(0);
let alpha_mask = i64x2_splat(0xffff000000000000u64 as i64);
let zero = u64x2_splat(0);
let alpha_mask = u64x2_splat(0xffff000000000000);
let alpha_max = f32x4_splat(65535.0);
let alpha_scale_max = f32x4_splat(2147483648f32);
/*
|R0 G0 B0 A0 | |R1 G1 B1 A1 |
|0001 0203 0405 0607| |0809 1011 1213 1415|
*/
const ALPHA32_SH0: v128 = i8x16(6, 7, -1, -1, 6, 7, -1, -1, 6, 7, -1, -1, 6, 7, -1, -1);
const ALPHA32_SH1: v128 = i8x16(
14, 15, -1, -1, 14, 15, -1, -1, 14, 15, -1, -1, 14, 15, -1, -1,
const ALPHA32_LO_SH: v128 = i8x16(6, 7, -1, -1, 6, 7, -1, -1, 6, 7, -1, -1, -1, -1, -1, -1);
const ALPHA32_HI_SH: v128 = i8x16(
14, 15, -1, -1, 14, 15, -1, -1, 14, 15, -1, -1, -1, -1, -1, -1,
);
let alpha0_f32x4 = f32x4_convert_i32x4(u8x16_swizzle(pixels, ALPHA32_SH0));
let alpha1_f32x4 = f32x4_convert_i32x4(u8x16_swizzle(pixels, ALPHA32_SH1));
let alpha_lo_f32x4 = f32x4_convert_i32x4(u8x16_swizzle(pixels, ALPHA32_LO_SH));
let alpha_hi_f32x4 = f32x4_convert_i32x4(u8x16_swizzle(pixels, ALPHA32_HI_SH));
let pix0_f32x4 = f32x4_convert_i32x4(i16x8_shuffle::<0, 8, 1, 9, 2, 10, 3, 11>(pixels, zero));
let pix1_f32x4 = f32x4_convert_i32x4(i16x8_shuffle::<4, 12, 5, 13, 6, 14, 7, 15>(pixels, zero));
let pix_lo_f32x4 = f32x4_convert_i32x4(i16x8_shuffle::<0, 8, 1, 9, 2, 10, 3, 11>(pixels, zero));
let pix_hi_f32x4 =
f32x4_convert_i32x4(i16x8_shuffle::<4, 12, 5, 13, 6, 14, 7, 15>(pixels, zero));
let scaled_pix0_f32x4 = f32x4_mul(pix0_f32x4, alpha_max);
let scaled_pix1_f32x4 = f32x4_mul(pix1_f32x4, alpha_max);
let scaled_pix_lo_f32x4 = f32x4_mul(pix_lo_f32x4, alpha_max);
let scaled_pix_hi_f32x4 = f32x4_mul(pix_hi_f32x4, alpha_max);
let divided_pix0_i32x4 = u32x4_trunc_sat_f32x4(f32x4_pmin(
f32x4_div(scaled_pix0_f32x4, alpha0_f32x4),
alpha_scale_max,
));
let divided_pix1_i32x4 = u32x4_trunc_sat_f32x4(f32x4_pmin(
f32x4_div(scaled_pix1_f32x4, alpha1_f32x4),
alpha_scale_max,
));
// In case of zero division the result will be u32::MAX or 0.
let divided_pix_lo_u32x4 =
u32x4_trunc_sat_f32x4(f32x4_div(scaled_pix_lo_f32x4, alpha_lo_f32x4));
let divided_pix_hi_u32x4 =
u32x4_trunc_sat_f32x4(f32x4_div(scaled_pix_hi_f32x4, alpha_hi_f32x4));
let two_pixels_i16x8 = u16x8_narrow_i32x4(divided_pix0_i32x4, divided_pix1_i32x4);
// All u32::MAX values in arguments will interpreted as -1i32.
// u16x8_narrow_i32x4() converts all negative values into 0.
let two_pixels_i16x8 = u16x8_narrow_i32x4(divided_pix_lo_u32x4, divided_pix_hi_u32x4);
let alpha = v128_and(pixels, alpha_mask);
u8x16_shuffle::<0, 1, 2, 3, 4, 5, 22, 23, 8, 9, 10, 11, 12, 13, 30, 31>(two_pixels_i16x8, alpha)
v128_or(two_pixels_i16x8, alpha)
}
+4 -4
View File
@@ -28,7 +28,7 @@ impl AlphaMulDiv for U8x2 {
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => unsafe { neon::multiply_alpha(src_image, dst_image) },
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => unsafe { wasm32::multiply_alpha(src_image, dst_image) },
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha(src_image, dst_image) },
_ => native::multiply_alpha(src_image, dst_image),
}
}
@@ -42,7 +42,7 @@ impl AlphaMulDiv for U8x2 {
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => unsafe { neon::multiply_alpha_inplace(image) },
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => unsafe { wasm32::multiply_alpha_inplace(image) },
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha_inplace(image) },
_ => native::multiply_alpha_inplace(image),
}
}
@@ -60,7 +60,7 @@ impl AlphaMulDiv for U8x2 {
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => unsafe { neon::divide_alpha(src_image, dst_image) },
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => unsafe { wasm32::divide_alpha(src_image, dst_image) },
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha(src_image, dst_image) },
_ => native::divide_alpha(src_image, dst_image),
}
}
@@ -74,7 +74,7 @@ impl AlphaMulDiv for U8x2 {
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => unsafe { neon::divide_alpha_inplace(image) },
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => unsafe { wasm32::divide_alpha_inplace(image) },
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha_inplace(image) },
_ => native::divide_alpha_inplace(image),
}
}
+2
View File
@@ -205,9 +205,11 @@ unsafe fn divide_alpha_8_pixels(pixels: __m128i) -> __m128i {
let alpha_scale = _mm_set1_ps(255.0 * 256.0);
let alpha_lo_f32 = _mm_cvtepi32_ps(_mm_shuffle_epi8(pixels, alpha32_sh_lo));
// In case of zero division the `scaled_alpha_lo_i32` will contain negative value (-2147483648).
let scaled_alpha_lo_i32 = _mm_cvtps_epi32(_mm_div_ps(alpha_scale, alpha_lo_f32));
let alpha_hi_f32 = _mm_cvtepi32_ps(_mm_shuffle_epi8(pixels, alpha32_sh_hi));
let scaled_alpha_hi_i32 = _mm_cvtps_epi32(_mm_div_ps(alpha_scale, alpha_hi_f32));
// All negative values will stored as 0.
let scaled_alpha_i16 = _mm_packus_epi32(scaled_alpha_lo_i32, scaled_alpha_hi_i32);
let luma_i16 = _mm_and_si128(pixels, luma_mask);
+53 -86
View File
@@ -1,7 +1,6 @@
use std::arch::wasm32::*;
use crate::pixels::U8x2;
use crate::utils::foreach_with_pre_reading;
use crate::{ImageView, ImageViewMut};
use super::native;
@@ -25,23 +24,18 @@ pub(crate) unsafe fn multiply_alpha_inplace(image: &mut ImageViewMut<U8x2>) {
}
#[inline]
pub(crate) unsafe fn multiply_alpha_row(src_row: &[U8x2], dst_row: &mut [U8x2]) {
#[target_feature(enable = "simd128")]
unsafe fn multiply_alpha_row(src_row: &[U8x2], dst_row: &mut [U8x2]) {
let src_chunks = src_row.chunks_exact(8);
let src_remainder = src_chunks.remainder();
let mut dst_chunks = dst_row.chunks_exact_mut(8);
let src_dst = src_chunks.zip(&mut dst_chunks);
foreach_with_pre_reading(
src_dst,
|(src, dst)| {
let pixels = v128_load(src.as_ptr() as *const v128);
let dst_ptr = dst.as_mut_ptr() as *mut v128;
(pixels, dst_ptr)
},
|(mut pixels, dst_ptr)| {
pixels = multiplies_alpha_8_pixels(pixels);
v128_store(dst_ptr, pixels);
},
);
// A simple for-loop in this case is as fast as implementation with pre-reading
for (src, dst) in src_dst {
let src_pixels = v128_load(src.as_ptr() as *const v128);
let dst_pixels = multiplies_alpha_8_pixels(src_pixels);
v128_store(dst.as_mut_ptr() as *mut v128, dst_pixels);
}
if !src_remainder.is_empty() {
let dst_reminder = dst_chunks.into_remainder();
@@ -50,9 +44,10 @@ pub(crate) unsafe fn multiply_alpha_row(src_row: &[U8x2], dst_row: &mut [U8x2])
}
#[inline]
pub(crate) unsafe fn multiply_alpha_row_inplace(row: &mut [U8x2]) {
#[target_feature(enable = "simd128")]
unsafe fn multiply_alpha_row_inplace(row: &mut [U8x2]) {
let mut chunks = row.chunks_exact_mut(8);
// Using a simple for-loop in this case is faster than implementation with pre-reading
// Using a simple for-loop in this case is as fast as implementation with pre-reading
for chunk in &mut chunks {
let src_pixels = v128_load(chunk.as_ptr() as *const v128);
let dst_pixels = multiplies_alpha_8_pixels(src_pixels);
@@ -66,41 +61,32 @@ pub(crate) unsafe fn multiply_alpha_row_inplace(row: &mut [U8x2]) {
}
#[inline]
#[target_feature(enable = "simd128")]
unsafe fn multiplies_alpha_8_pixels(pixels: v128) -> v128 {
let zero = i64x2_splat(0);
let half = i16x8_splat(128);
const MAX_A: i16 = 0xff00u16 as i16;
let max_alpha = i16x8_splat(MAX_A);
let half = u16x8_splat(128);
let max_alpha = u16x8_splat(0xff00);
/*
|L A | |L A | |L A | |L A | |L A | |L A | |L A | |L A |
|00 01| |02 03| |04 05| |06 07| |08 09| |10 11| |12 13| |14 15|
*/
const FACTOR_MASK: v128 = i8x16(1, 1, 3, 3, 5, 5, 7, 7, 9, 9, 11, 11, 13, 13, 15, 15);
let factor_pixels = i8x16_swizzle(pixels, FACTOR_MASK);
let factor_pixels = u8x16_swizzle(pixels, FACTOR_MASK);
let factor_pixels = v128_or(factor_pixels, max_alpha);
let src_i16_lo =
i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(pixels, zero);
let factors = i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(
factor_pixels,
zero,
);
let src_i16_lo = i16x8_add(i16x8_mul(src_i16_lo, factors), half);
let dst_i16_lo = i16x8_add(src_i16_lo, u16x8_shr(src_i16_lo, 8));
let dst_i16_lo = u16x8_shr(dst_i16_lo, 8);
let src_u16_lo = u16x8_extend_low_u8x16(pixels);
let factors = u16x8_extend_low_u8x16(factor_pixels);
let mut dst_u16_lo = u16x8_add(u16x8_mul(src_u16_lo, factors), half);
dst_u16_lo = u16x8_add(dst_u16_lo, u16x8_shr(dst_u16_lo, 8));
dst_u16_lo = u16x8_shr(dst_u16_lo, 8);
let src_i16_hi =
i8x16_shuffle::<8, 24, 9, 25, 10, 26, 11, 27, 12, 28, 13, 29, 14, 30, 15, 31>(pixels, zero);
let factors = i8x16_shuffle::<8, 24, 9, 25, 10, 26, 11, 27, 12, 28, 13, 29, 14, 30, 15, 31>(
factor_pixels,
zero,
);
let src_i16_hi = i16x8_add(i16x8_mul(src_i16_hi, factors), half);
let dst_i16_hi = i16x8_add(src_i16_hi, u16x8_shr(src_i16_hi, 8));
let dst_i16_hi = u16x8_shr(dst_i16_hi, 8);
let src_u16_hi = u16x8_extend_high_u8x16(pixels);
let factors = u16x8_extend_high_u8x16(factor_pixels);
let mut dst_u16_hi = u16x8_add(u16x8_mul(src_u16_hi, factors), half);
dst_u16_hi = u16x8_add(dst_u16_hi, u16x8_shr(dst_u16_hi, 8));
dst_u16_hi = u16x8_shr(dst_u16_hi, 8);
u8x16_narrow_i16x8(dst_i16_lo, dst_i16_hi)
u8x16_narrow_i16x8(dst_u16_lo, dst_u16_hi)
}
// Divide
@@ -121,23 +107,18 @@ pub(crate) unsafe fn divide_alpha_inplace(image: &mut ImageViewMut<U8x2>) {
}
#[inline]
pub(crate) unsafe fn divide_alpha_row(src_row: &[U8x2], dst_row: &mut [U8x2]) {
#[target_feature(enable = "simd128")]
unsafe fn divide_alpha_row(src_row: &[U8x2], dst_row: &mut [U8x2]) {
let src_chunks = src_row.chunks_exact(8);
let src_remainder = src_chunks.remainder();
let mut dst_chunks = dst_row.chunks_exact_mut(8);
let src_dst = src_chunks.zip(&mut dst_chunks);
foreach_with_pre_reading(
src_dst,
|(src, dst)| {
let pixels = v128_load(src.as_ptr() as *const v128);
let dst_ptr = dst.as_mut_ptr() as *mut v128;
(pixels, dst_ptr)
},
|(mut pixels, dst_ptr)| {
pixels = divide_alpha_8_pixels(pixels);
v128_store(dst_ptr, pixels);
},
);
// Using a simple for-loop in this case is as fast as implementation with pre-reading
for (src, dst) in src_dst {
let src_pixels = v128_load(src.as_ptr() as *const v128);
let dst_pixels = divide_alpha_8_pixels(src_pixels);
v128_store(dst.as_mut_ptr() as *mut v128, dst_pixels);
}
if !src_remainder.is_empty() {
let dst_reminder = dst_chunks.into_remainder();
@@ -160,20 +141,15 @@ pub(crate) unsafe fn divide_alpha_row(src_row: &[U8x2], dst_row: &mut [U8x2]) {
}
#[inline]
pub(crate) unsafe fn divide_alpha_row_inplace(row: &mut [U8x2]) {
#[target_feature(enable = "simd128")]
unsafe fn divide_alpha_row_inplace(row: &mut [U8x2]) {
let mut chunks = row.chunks_exact_mut(8);
foreach_with_pre_reading(
&mut chunks,
|chunk| {
let pixels = v128_load(chunk.as_ptr() as *const v128);
let dst_ptr = chunk.as_mut_ptr() as *mut v128;
(pixels, dst_ptr)
},
|(mut pixels, dst_ptr)| {
pixels = divide_alpha_8_pixels(pixels);
v128_store(dst_ptr, pixels);
},
);
// Using a simple for-loop in this case is as fast as implementation with pre-reading
for chunk in &mut chunks {
let src_pixels = v128_load(chunk.as_ptr() as *const v128);
let dst_pixels = divide_alpha_8_pixels(src_pixels);
v128_store(chunk.as_mut_ptr() as *mut v128, dst_pixels);
}
let reminder = chunks.into_remainder();
if !reminder.is_empty() {
@@ -193,6 +169,7 @@ pub(crate) unsafe fn divide_alpha_row_inplace(row: &mut [U8x2]) {
}
#[inline]
#[target_feature(enable = "simd128")]
unsafe fn divide_alpha_8_pixels(pixels: v128) -> v128 {
let alpha_mask = i16x8_splat(0xff00u16 as i16);
let luma_mask = i16x8_splat(0xff);
@@ -201,33 +178,23 @@ unsafe fn divide_alpha_8_pixels(pixels: v128) -> v128 {
9, -1, -1, -1, 11, -1, -1, -1, 13, -1, -1, -1, 15, -1, -1, -1,
);
let alpha_scale = f32x4_splat(255.0 * 256.0);
// sse4 _mm_cvtps_epi32 converts inf to i32::MIN or 2147483648f32 u32.
// wasm32 u32x4_trunc_sat_f32x4 on AVX systems converts inf to u32::MAX.
// Tests pass without capping inf from dividing by zero, but scaled values will not match sse4,
// and other potential test cases will (probably?) break.
let alpha_scale_max = f32x4_splat(2147483648f32);
let alpha_lo_f32 = f32x4_convert_u32x4(i8x16_swizzle(pixels, ALPHA32_SH_LO));
// trunc_sat will always round down. Adding f32x4_nearest would match _mm_cvtps_epi32 exactly,
// but would add extra instructions.
let scaled_alpha_lo_u32 = u32x4_trunc_sat_f32x4(f32x4_pmin(
f32x4_div(alpha_scale, alpha_lo_f32),
alpha_scale_max,
));
let alpha_lo_f32 = f32x4_convert_u32x4(u8x16_swizzle(pixels, ALPHA32_SH_LO));
// In case of zero division the result will be u32::MAX or 0.
let scaled_alpha_lo_u32 = u32x4_trunc_sat_f32x4(f32x4_div(alpha_scale, alpha_lo_f32));
let alpha_hi_f32 = f32x4_convert_u32x4(i8x16_swizzle(pixels, ALPHA32_SH_HI));
let scaled_alpha_hi_u32 = u32x4_trunc_sat_f32x4(f32x4_pmin(
f32x4_div(alpha_scale, alpha_hi_f32),
alpha_scale_max,
));
let scaled_alpha_hi_u32 = u32x4_trunc_sat_f32x4(f32x4_div(alpha_scale, alpha_hi_f32));
// All u32::MAX values in arguments will interpreted as -1i32.
// u16x8_narrow_i32x4() converts all negative values into 0.
let scaled_alpha_u16 = u16x8_narrow_i32x4(scaled_alpha_lo_u32, scaled_alpha_hi_u32);
let luma_u16 = v128_and(pixels, luma_mask);
let scaled_luma_u16 = u16x8_mul(luma_u16, scaled_alpha_u16);
let scaled_luma_u16 = u16x8_shr(scaled_luma_u16, 8);
// Blend scaled luma with original alpha channel.
let alpha = v128_and(pixels, alpha_mask);
u8x16_shuffle::<0, 17, 2, 19, 4, 21, 6, 23, 8, 25, 10, 27, 12, 29, 14, 31>(
scaled_luma_u16,
alpha,
)
v128_or(scaled_luma_u16, alpha)
}
+4 -4
View File
@@ -28,7 +28,7 @@ impl AlphaMulDiv for U8x4 {
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => unsafe { neon::multiply_alpha(src_image, dst_image) },
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => unsafe { wasm32::multiply_alpha(src_image, dst_image) },
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha(src_image, dst_image) },
_ => native::multiply_alpha(src_image, dst_image),
}
}
@@ -42,7 +42,7 @@ impl AlphaMulDiv for U8x4 {
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => unsafe { neon::multiply_alpha_inplace(image) },
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => unsafe { wasm32::multiply_alpha_inplace(image) },
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha_inplace(image) },
_ => native::multiply_alpha_inplace(image),
}
}
@@ -60,7 +60,7 @@ impl AlphaMulDiv for U8x4 {
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => unsafe { neon::divide_alpha(src_image, dst_image) },
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => unsafe { wasm32::divide_alpha(src_image, dst_image) },
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha(src_image, dst_image) },
_ => native::divide_alpha(src_image, dst_image),
}
}
@@ -74,7 +74,7 @@ impl AlphaMulDiv for U8x4 {
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => unsafe { neon::divide_alpha_inplace(image) },
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => unsafe { wasm32::divide_alpha_inplace(image) },
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha_inplace(image) },
_ => native::divide_alpha_inplace(image),
}
}
+64 -88
View File
@@ -1,7 +1,6 @@
use std::arch::wasm32::*;
use crate::pixels::U8x4;
use crate::utils::foreach_with_pre_reading;
use crate::wasm32_utils;
use crate::{ImageView, ImageViewMut};
@@ -26,23 +25,18 @@ pub(crate) unsafe fn multiply_alpha_inplace(image: &mut ImageViewMut<U8x4>) {
}
#[inline]
pub(crate) unsafe fn multiply_alpha_row(src_row: &[U8x4], dst_row: &mut [U8x4]) {
#[target_feature(enable = "simd128")]
unsafe fn multiply_alpha_row(src_row: &[U8x4], dst_row: &mut [U8x4]) {
let src_chunks = src_row.chunks_exact(4);
let src_remainder = src_chunks.remainder();
let mut dst_chunks = dst_row.chunks_exact_mut(4);
let src_dst = src_chunks.zip(&mut dst_chunks);
foreach_with_pre_reading(
src_dst,
|(src, dst)| {
let pixels = v128_load(src.as_ptr() as *const v128);
let dst_ptr = dst.as_mut_ptr() as *mut v128;
(pixels, dst_ptr)
},
|(mut pixels, dst_ptr)| {
pixels = multiply_alpha_4_pixels(pixels);
v128_store(dst_ptr, pixels);
},
);
// A simple for-loop in this case is as fast as implementation with pre-reading
for (src, dst) in src_dst {
let mut pixels = v128_load(src.as_ptr() as *const v128);
pixels = multiply_alpha_4_pixels(pixels);
v128_store(dst.as_mut_ptr() as *mut v128, pixels);
}
if !src_remainder.is_empty() {
let dst_reminder = dst_chunks.into_remainder();
@@ -51,9 +45,10 @@ pub(crate) unsafe fn multiply_alpha_row(src_row: &[U8x4], dst_row: &mut [U8x4])
}
#[inline]
pub(crate) unsafe fn multiply_alpha_row_inplace(row: &mut [U8x4]) {
#[target_feature(enable = "simd128")]
unsafe fn multiply_alpha_row_inplace(row: &mut [U8x4]) {
let mut chunks = row.chunks_exact_mut(4);
// Using a simple for-loop in this case is faster than implementation with pre-reading
// A simple for-loop in this case is as fast as implementation with pre-reading
for chunk in &mut chunks {
let mut pixels = v128_load(chunk.as_ptr() as *const v128);
pixels = multiply_alpha_4_pixels(pixels);
@@ -67,38 +62,28 @@ pub(crate) unsafe fn multiply_alpha_row_inplace(row: &mut [U8x4]) {
}
#[inline]
#[target_feature(enable = "simd128")]
unsafe fn multiply_alpha_4_pixels(pixels: v128) -> v128 {
let zero = i64x2_splat(0);
let half = i16x8_splat(128);
const MAX_A: u32 = 0xff000000u32;
let max_alpha = u32x4_splat(MAX_A);
let half = u16x8_splat(128);
let max_alpha = u32x4_splat(0xff000000);
const FACTOR_MASK: v128 = i8x16(3, 3, 3, 3, 7, 7, 7, 7, 11, 11, 11, 11, 15, 15, 15, 15);
let factor_pixels = u8x16_swizzle(pixels, FACTOR_MASK);
let factor_pixels = v128_or(factor_pixels, max_alpha);
let pix1 =
i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(pixels, zero);
let factors = i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(
factor_pixels,
zero,
);
let pix1 = i16x8_add(i16x8_mul(pix1, factors), half);
let pix1 = i16x8_add(pix1, u16x8_shr(pix1, 8));
let pix1 = u16x8_shr(pix1, 8);
let src_u16_lo = u16x8_extend_low_u8x16(pixels);
let factors = u16x8_extend_low_u8x16(factor_pixels);
let mut dst_u16_lo = u16x8_add(u16x8_mul(src_u16_lo, factors), half);
dst_u16_lo = u16x8_add(dst_u16_lo, u16x8_shr(dst_u16_lo, 8));
dst_u16_lo = u16x8_shr(dst_u16_lo, 8);
let pix2 =
i8x16_shuffle::<8, 24, 9, 25, 10, 26, 11, 27, 12, 28, 13, 29, 14, 30, 15, 31>(pixels, zero);
let factors = i8x16_shuffle::<8, 24, 9, 25, 10, 26, 11, 27, 12, 28, 13, 29, 14, 30, 15, 31>(
factor_pixels,
zero,
);
let pix2 = i16x8_add(i16x8_mul(pix2, factors), half);
let pix2 = i16x8_add(pix2, u16x8_shr(pix2, 8));
let pix2 = u16x8_shr(pix2, 8);
let src_u16_hi = u16x8_extend_high_u8x16(pixels);
let factors = u16x8_extend_high_u8x16(factor_pixels);
let mut dst_u16_hi = u16x8_add(u16x8_mul(src_u16_hi, factors), half);
dst_u16_hi = u16x8_add(dst_u16_hi, u16x8_shr(dst_u16_hi, 8));
dst_u16_hi = u16x8_shr(dst_u16_hi, 8);
u8x16_narrow_i16x8(pix1, pix2)
u8x16_narrow_i16x8(dst_u16_lo, dst_u16_hi)
}
// Divide
@@ -118,23 +103,18 @@ pub(crate) unsafe fn divide_alpha_inplace(image: &mut ImageViewMut<U8x4>) {
}
#[inline]
pub(crate) unsafe fn divide_alpha_row(src_row: &[U8x4], dst_row: &mut [U8x4]) {
#[target_feature(enable = "simd128")]
unsafe fn divide_alpha_row(src_row: &[U8x4], dst_row: &mut [U8x4]) {
let src_chunks = src_row.chunks_exact(4);
let src_remainder = src_chunks.remainder();
let mut dst_chunks = dst_row.chunks_exact_mut(4);
let src_dst = src_chunks.zip(&mut dst_chunks);
foreach_with_pre_reading(
src_dst,
|(src, dst)| {
let pixels = v128_load(src.as_ptr() as *const v128);
let dst_ptr = dst.as_mut_ptr() as *mut v128;
(pixels, dst_ptr)
},
|(mut pixels, dst_ptr)| {
pixels = divide_alpha_4_pixels(pixels);
v128_store(dst_ptr, pixels);
},
);
// A simple for-loop in this case is faster than implementation with pre-reading
for (src, dst) in src_dst {
let mut pixels = v128_load(src.as_ptr() as *const v128);
pixels = divide_alpha_4_pixels(pixels);
v128_store(dst.as_mut_ptr() as *mut v128, pixels);
}
if !src_remainder.is_empty() {
let dst_reminder = dst_chunks.into_remainder();
@@ -157,20 +137,15 @@ pub(crate) unsafe fn divide_alpha_row(src_row: &[U8x4], dst_row: &mut [U8x4]) {
}
#[inline]
pub(crate) unsafe fn divide_alpha_row_inplace(row: &mut [U8x4]) {
#[target_feature(enable = "simd128")]
unsafe fn divide_alpha_row_inplace(row: &mut [U8x4]) {
let mut chunks = row.chunks_exact_mut(4);
foreach_with_pre_reading(
&mut chunks,
|chunk| {
let pixels = v128_load(chunk.as_ptr() as *const v128);
let dst_ptr = chunk.as_mut_ptr() as *mut v128;
(pixels, dst_ptr)
},
|(mut pixels, dst_ptr)| {
pixels = divide_alpha_4_pixels(pixels);
v128_store(dst_ptr, pixels);
},
);
// A simple for-loop in this case is faster than implementation with pre-reading
for chunk in &mut chunks {
let mut pixels = v128_load(chunk.as_ptr() as *const v128);
pixels = divide_alpha_4_pixels(pixels);
v128_store(chunk.as_mut_ptr() as *mut v128, pixels);
}
let tail = chunks.into_remainder();
if !tail.is_empty() {
@@ -190,31 +165,32 @@ pub(crate) unsafe fn divide_alpha_row_inplace(row: &mut [U8x4]) {
}
#[inline]
unsafe fn divide_alpha_4_pixels(src_pixels: v128) -> v128 {
let zero = i64x2_splat(0);
let alpha_mask = i32x4_splat(0xff000000u32 as i32);
const SHUFFLE1: v128 = i8x16(0, 1, 0, 1, 0, 1, 0, 1, 4, 5, 4, 5, 4, 5, 4, 5);
const SHUFFLE2: v128 = i8x16(8, 9, 8, 9, 8, 9, 8, 9, 12, 13, 12, 13, 12, 13, 12, 13);
#[target_feature(enable = "simd128")]
unsafe fn divide_alpha_4_pixels(pixels: v128) -> v128 {
const FACTOR_LO_SHUFFLE: v128 = i8x16(0, 1, 0, 1, 0, 1, -1, -1, 2, 3, 2, 3, 2, 3, -1, -1);
const FACTOR_HI_SHUFFLE: v128 = i8x16(4, 5, 4, 5, 4, 5, -1, -1, 6, 7, 6, 7, 6, 7, -1, -1);
let alpha_mask = u32x4_splat(0xff000000);
let alpha_scale = f32x4_splat(255.0 * 256.0);
let alpha_scale_max = f32x4_splat(2147483648f32);
let alpha_f32 = f32x4_convert_i32x4(u32x4_shr(src_pixels, 24));
let scaled_alpha_f32 = f32x4_div(alpha_scale, alpha_f32);
let scaled_alpha_u32 = u32x4_trunc_sat_f32x4(f32x4_pmin(scaled_alpha_f32, alpha_scale_max));
let mma0 = u8x16_swizzle(scaled_alpha_u32, SHUFFLE1);
let mma1 = u8x16_swizzle(scaled_alpha_u32, SHUFFLE2);
let alpha_f32 = f32x4_convert_i32x4(u32x4_shr(pixels, 24));
// In case of zero division the result will be u32::MAX or 0.
let scaled_alpha_u32 = u32x4_trunc_sat_f32x4(f32x4_div(alpha_scale, alpha_f32));
// All u32::MAX values in arguments will interpreted as -1i32.
// u16x8_narrow_i32x4() converts all negative values into 0.
let scaled_alpha_u16 = u16x8_narrow_i32x4(scaled_alpha_u32, scaled_alpha_u32);
let factor_lo_u16x8 = u8x16_swizzle(scaled_alpha_u16, FACTOR_LO_SHUFFLE);
let factor_hi_u16x8 = u8x16_swizzle(scaled_alpha_u16, FACTOR_HI_SHUFFLE);
let pix0 =
u8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(zero, src_pixels);
let pix1 = u8x16_shuffle::<8, 24, 9, 25, 10, 26, 11, 27, 12, 28, 13, 29, 14, 30, 15, 31>(
zero, src_pixels,
);
// alpha_mask's first byte is 0
let src_u16_lo =
u8x16_shuffle::<0, 16, 0, 17, 0, 18, 0, 19, 0, 20, 0, 21, 0, 22, 0, 23>(alpha_mask, pixels);
let src_u16_hi =
u8x16_shuffle::<0, 24, 0, 25, 0, 26, 0, 27, 0, 28, 0, 29, 0, 30, 0, 31>(alpha_mask, pixels);
let pix0 = wasm32_utils::u16x8_mul_hi(pix0, mma0);
let pix1 = wasm32_utils::u16x8_mul_hi(pix1, mma1);
let dst_lo = wasm32_utils::u16x8_mul_shr16(src_u16_lo, factor_lo_u16x8);
let dst_hi = wasm32_utils::u16x8_mul_shr16(src_u16_hi, factor_hi_u16x8);
let alpha = v128_and(src_pixels, alpha_mask);
let rgb = u8x16_narrow_i16x8(pix0, pix1);
u8x16_shuffle::<0, 1, 2, 19, 4, 5, 6, 23, 8, 9, 10, 27, 12, 13, 14, 31>(rgb, alpha)
let alpha = v128_and(pixels, alpha_mask);
let rgb = u8x16_narrow_i16x8(dst_lo, dst_hi);
v128_or(rgb, alpha)
}
+1 -1
View File
@@ -31,7 +31,7 @@ impl Convolution for U16 {
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => neon::horiz_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => {
CpuExtensions::Simd128 => {
wasm32::horiz_convolution(src_image, dst_image, offset, coeffs)
}
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
+1 -1
View File
@@ -31,7 +31,7 @@ impl Convolution for U16x2 {
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => neon::horiz_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => {
CpuExtensions::Simd128 => {
wasm32::horiz_convolution(src_image, dst_image, offset, coeffs)
}
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
+1 -1
View File
@@ -31,7 +31,7 @@ impl Convolution for U16x3 {
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => neon::horiz_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => {
CpuExtensions::Simd128 => {
wasm32::horiz_convolution(src_image, dst_image, offset, coeffs)
}
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
+1 -1
View File
@@ -31,7 +31,7 @@ impl Convolution for U16x4 {
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => neon::horiz_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => {
CpuExtensions::Simd128 => {
wasm32::horiz_convolution(src_image, dst_image, offset, coeffs)
}
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
+1 -1
View File
@@ -31,7 +31,7 @@ impl Convolution for U8 {
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => neon::horiz_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => {
CpuExtensions::Simd128 => {
wasm32::horiz_convolution(src_image, dst_image, offset, coeffs)
}
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
+1 -1
View File
@@ -31,7 +31,7 @@ impl Convolution for U8x2 {
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => neon::horiz_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => {
CpuExtensions::Simd128 => {
wasm32::horiz_convolution(src_image, dst_image, offset, coeffs)
}
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
+1 -1
View File
@@ -31,7 +31,7 @@ impl Convolution for U8x3 {
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => neon::horiz_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => {
CpuExtensions::Simd128 => {
wasm32::horiz_convolution(src_image, dst_image, offset, coeffs)
}
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
+1 -1
View File
@@ -31,7 +31,7 @@ impl Convolution for U8x4 {
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => neon::horiz_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => {
CpuExtensions::Simd128 => {
wasm32::horiz_convolution(src_image, dst_image, offset, coeffs)
}
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
+1 -1
View File
@@ -32,7 +32,7 @@ pub(crate) fn vert_convolution_u16<T: PixelExt<Component = u16>>(
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => neon::vert_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => wasm32::vert_convolution(src_image, dst_image, offset, coeffs),
CpuExtensions::Simd128 => wasm32::vert_convolution(src_image, dst_image, offset, coeffs),
_ => native::vert_convolution(src_image, dst_image, offset, coeffs),
}
}
+1 -1
View File
@@ -32,7 +32,7 @@ pub(crate) fn vert_convolution_u8<T: PixelExt<Component = u8>>(
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => neon::vert_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => wasm32::vert_convolution(src_image, dst_image, offset, coeffs),
CpuExtensions::Simd128 => wasm32::vert_convolution(src_image, dst_image, offset, coeffs),
_ => native::vert_convolution(src_image, dst_image, offset, coeffs),
}
}
+10 -3
View File
@@ -7,17 +7,24 @@ use crate::{
DifferentTypesOfPixelsError, DynamicImageView, DynamicImageViewMut, ImageView, ImageViewMut,
};
/// SIMD extension of CPU.
/// Specific variants depends from target architecture.
/// Look at source code to see all available variants.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum CpuExtensions {
None,
#[cfg(target_arch = "x86_64")]
/// SIMD extension of x86_64 architecture
Sse4_1,
#[cfg(target_arch = "x86_64")]
/// SIMD extension of x86_64 architecture
Avx2,
#[cfg(target_arch = "aarch64")]
/// SIMD extension of Arm64 architecture
Neon,
#[cfg(target_arch = "wasm32")]
Wasm32,
/// SIMD extension of Wasm32 architecture
Simd128,
}
impl CpuExtensions {
@@ -31,7 +38,7 @@ impl CpuExtensions {
#[cfg(target_arch = "aarch64")]
Self::Neon => true,
#[cfg(target_arch = "wasm32")]
Self::Wasm32 => true,
Self::Simd128 => true,
Self::None => true,
}
}
@@ -60,7 +67,7 @@ impl Default for CpuExtensions {
}
#[cfg(target_arch = "wasm32")]
fn default() -> Self {
Self::Wasm32
Self::Simd128
}
#[cfg(not(any(
+42 -30
View File
@@ -1,73 +1,85 @@
use crate::pixels::{U8x3, U8x4};
use std::arch::wasm32::*;
use std::intrinsics::transmute;
#[inline(always)]
pub unsafe fn load_v128<T>(buf: &[T], index: usize) -> v128 {
use crate::pixels::{U8x3, U8x4};
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn load_v128<T>(buf: &[T], index: usize) -> v128 {
v128_load(buf.get_unchecked(index..).as_ptr() as *const v128)
}
#[inline(always)]
pub unsafe fn loadl_i64<T>(buf: &[T], index: usize) -> v128 {
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn loadl_i64<T>(buf: &[T], index: usize) -> v128 {
let i = buf.get_unchecked(index..).as_ptr() as *const i64;
i64x2(*i, 0)
}
#[inline(always)]
pub unsafe fn loadl_i32<T>(buf: &[T], index: usize) -> v128 {
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn loadl_i32<T>(buf: &[T], index: usize) -> v128 {
let i = buf.get_unchecked(index..).as_ptr() as *const i32;
i32x4(*i, 0, 0, 0)
}
#[inline(always)]
pub unsafe fn loadl_i16<T>(buf: &[T], index: usize) -> v128 {
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn loadl_i16<T>(buf: &[T], index: usize) -> v128 {
let i = buf.get_unchecked(index..).as_ptr() as *const i16;
i16x8(*i, 0, 0, 0, 0, 0, 0, 0)
}
#[inline(always)]
pub unsafe fn ptr_i16_to_set1_i64(buf: &[i16], index: usize) -> v128 {
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn ptr_i16_to_set1_i64(buf: &[i16], index: usize) -> v128 {
i64x2_splat(*(buf.get_unchecked(index..).as_ptr() as *const i64))
}
#[inline(always)]
pub unsafe fn ptr_i16_to_set1_i32(buf: &[i16], index: usize) -> v128 {
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn ptr_i16_to_set1_i32(buf: &[i16], index: usize) -> v128 {
i32x4_splat(*(buf.get_unchecked(index..).as_ptr() as *const i32))
}
#[inline(always)]
pub unsafe fn i32x4_extend_low_ptr_u8(buf: &[u8], index: usize) -> v128 {
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn i32x4_extend_low_ptr_u8(buf: &[u8], index: usize) -> v128 {
let ptr = buf.get_unchecked(index..).as_ptr() as *const v128;
u32x4_extend_low_u16x8(i16x8_extend_low_u8x16(v128_load(ptr)))
}
#[inline(always)]
pub unsafe fn i32x4_extend_low_ptr_u8x4(buf: &[U8x4], index: usize) -> v128 {
let v: u32 = transmute(buf.get_unchecked(index).0);
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn i32x4_extend_low_ptr_u8x4(buf: &[U8x4], index: usize) -> v128 {
let v: u32 = buf.get_unchecked(index).0;
u32x4_extend_low_u16x8(i16x8_extend_low_u8x16(u32x4(v, 0, 0, 0)))
}
#[inline(always)]
pub unsafe fn i32x4_extend_low_ptr_u8x3(buf: &[U8x3], index: usize) -> v128 {
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn i32x4_extend_low_ptr_u8x3(buf: &[U8x3], index: usize) -> v128 {
let pixel = buf.get_unchecked(index).0;
i32x4(pixel[0] as i32, pixel[1] as i32, pixel[2] as i32, 0)
}
#[inline(always)]
pub unsafe fn i32x4_v128_from_u8(buf: &[u8], index: usize) -> v128 {
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn i32x4_v128_from_u8(buf: &[u8], index: usize) -> v128 {
let ptr = buf.get_unchecked(index..).as_ptr() as *const i32;
i32x4(*ptr, 0, 0, 0)
}
#[inline(always)]
pub unsafe fn u16x8_mul_hi(a: v128, b: v128) -> v128 {
let lo = u32x4_extmul_low_u16x8(a, b);
let hi = u32x4_extmul_high_u16x8(a, b);
i16x8_shuffle::<1, 3, 5, 7, 9, 11, 13, 15>(lo, hi)
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn u16x8_mul_shr16(a_u16x8: v128, b_u16x8: v128) -> v128 {
let lo_u32x4 = u32x4_extmul_low_u16x8(a_u16x8, b_u16x8);
let hi_u32x4 = u32x4_extmul_high_u16x8(a_u16x8, b_u16x8);
i16x8_shuffle::<1, 3, 5, 7, 9, 11, 13, 15>(lo_u32x4, hi_u32x4)
}
#[inline(always)]
pub unsafe fn i64x2_mul_lo(a: v128, b: v128) -> v128 {
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn i64x2_mul_lo(a: v128, b: v128) -> v128 {
const SHUFFLE: v128 = i8x16(0, 1, 2, 3, 8, 9, 10, 11, -1, -1, -1, -1, -1, -1, -1, -1);
i64x2_extmul_low_i32x4(i8x16_swizzle(a, SHUFFLE), i8x16_swizzle(b, SHUFFLE))
}
+5 -5
View File
@@ -254,7 +254,7 @@ pub fn save_result(image: &Image, name: &str) {
_ => panic!("Unsupported type of pixels"),
};
image::save_buffer(
&path,
path,
image.buffer(),
image.width().get(),
image.height().get(),
@@ -263,16 +263,16 @@ pub fn save_result(image: &Image, name: &str) {
.unwrap();
}
pub fn cpu_ext_into_str(cpu_extensions: CpuExtensions) -> &'static str {
pub const fn cpu_ext_into_str(cpu_extensions: CpuExtensions) -> &'static str {
match cpu_extensions {
CpuExtensions::None => "native",
CpuExtensions::None => "rust",
#[cfg(target_arch = "x86_64")]
CpuExtensions::Sse4_1 => "sse41",
CpuExtensions::Sse4_1 => "sse4.1",
#[cfg(target_arch = "x86_64")]
CpuExtensions::Avx2 => "avx2",
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => "neon",
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => "wasm32",
CpuExtensions::Simd128 => "simd128",
}
}
+14 -14
View File
@@ -94,8 +94,7 @@ fn mul_div_alpha_test<P, const N: usize>(
{
assert_eq!(
r, *e,
"failed test: src={:?}, result={:?}, expected_result={:?}",
s, r, e
"failed test: src={s:?}, result={r:?}, expected_result={e:?}",
);
}
@@ -119,8 +118,7 @@ fn mul_div_alpha_test<P, const N: usize>(
for ((s, r), e) in src_pixels.iter().zip(src_pixels_clone).zip(expected_pixels) {
assert_eq!(
r, e,
"failed inplace test: src={:?}, result={:?}, expected_result={:?}",
s, r, e
"failed inplace test: src={s:?}, result={r:?}, expected_result={e:?}",
);
}
}
@@ -157,7 +155,7 @@ mod multiply_alpha_u8x4 {
#[cfg(target_arch = "wasm32")]
#[test]
fn wasm32_test() {
mul_div_alpha_test(Oper::Mul, SRC_PIXELS, RES_PIXELS, CpuExtensions::Wasm32);
mul_div_alpha_test(Oper::Mul, SRC_PIXELS, RES_PIXELS, CpuExtensions::Simd128);
}
#[test]
@@ -215,7 +213,7 @@ mod multiply_alpha_u8x2 {
#[cfg(target_arch = "wasm32")]
#[test]
fn wasm32_test() {
mul_div_alpha_test(OPER, SRC_PIXELS, RES_PIXELS, CpuExtensions::Wasm32);
mul_div_alpha_test(OPER, SRC_PIXELS, RES_PIXELS, CpuExtensions::Simd128);
}
#[test]
@@ -226,9 +224,10 @@ mod multiply_alpha_u8x2 {
#[cfg(test)]
mod multiply_alpha_u16x2 {
use super::*;
use fast_image_resize::pixels::U16x2;
use super::*;
const SRC_PIXELS: [U16x2; 9] = [
U16x2::new([0xffff, 0x8000]),
U16x2::new([0x8000, 0x8000]),
@@ -273,7 +272,7 @@ mod multiply_alpha_u16x2 {
#[cfg(target_arch = "wasm32")]
#[test]
fn wasm32_test() {
mul_div_alpha_test(Oper::Mul, SRC_PIXELS, RES_PIXELS, CpuExtensions::Wasm32);
mul_div_alpha_test(Oper::Mul, SRC_PIXELS, RES_PIXELS, CpuExtensions::Simd128);
}
#[test]
@@ -284,9 +283,10 @@ mod multiply_alpha_u16x2 {
#[cfg(test)]
mod multiply_alpha_u16x4 {
use super::*;
use fast_image_resize::pixels::U16x4;
use super::*;
const SRC_PIXELS: [U16x4; 3] = [
U16x4::new([0xffff, 0x8000, 0, 0x8000]),
U16x4::new([0xffff, 0x8000, 0, 0xffff]),
@@ -319,7 +319,7 @@ mod multiply_alpha_u16x4 {
#[cfg(target_arch = "wasm32")]
#[test]
fn wasm32_test() {
mul_div_alpha_test(Oper::Mul, SRC_PIXELS, RES_PIXELS, CpuExtensions::Wasm32);
mul_div_alpha_test(Oper::Mul, SRC_PIXELS, RES_PIXELS, CpuExtensions::Simd128);
}
#[test]
@@ -363,7 +363,7 @@ mod divide_alpha_u8x4 {
#[cfg(target_arch = "wasm32")]
#[test]
fn wasm32_test() {
mul_div_alpha_test(OPER, SRC_PIXELS, RES_PIXELS, CpuExtensions::Wasm32);
mul_div_alpha_test(OPER, SRC_PIXELS, RES_PIXELS, CpuExtensions::Simd128);
}
#[test]
@@ -421,7 +421,7 @@ mod divide_alpha_u8x2 {
#[cfg(target_arch = "wasm32")]
#[test]
fn wasm32_test() {
mul_div_alpha_test(OPER, SRC_PIXELS, RES_PIXELS, CpuExtensions::Wasm32);
mul_div_alpha_test(OPER, SRC_PIXELS, RES_PIXELS, CpuExtensions::Simd128);
}
#[test]
@@ -490,7 +490,7 @@ mod divide_alpha_u16x2 {
#[cfg(target_arch = "wasm32")]
#[test]
fn wasm32_test() {
mul_div_alpha_test(OPER, SRC_PIXELS, RES_PIXELS, CpuExtensions::Wasm32);
mul_div_alpha_test(OPER, SRC_PIXELS, RES_PIXELS, CpuExtensions::Simd128);
}
#[test]
@@ -541,7 +541,7 @@ mod divide_alpha_u16x4 {
#[cfg(target_arch = "wasm32")]
#[test]
fn wasm32_test() {
mul_div_alpha_test(OPER, SRC_PIXELS, RES_PIXELS, CpuExtensions::Wasm32);
mul_div_alpha_test(OPER, SRC_PIXELS, RES_PIXELS, CpuExtensions::Simd128);
}
#[test]
+17 -17
View File
@@ -205,7 +205,7 @@ fn resize_to_same_width_after_cropping() {
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
if !cpu_extensions.is_supported() {
@@ -382,7 +382,7 @@ fn downscale_u8() {
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
P::downscale_test(
@@ -410,7 +410,7 @@ fn upscale_u8() {
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(
@@ -438,7 +438,7 @@ fn downscale_u8x2() {
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
P::downscale_test(
@@ -470,7 +470,7 @@ fn upscale_u8x2() {
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(
@@ -502,7 +502,7 @@ fn downscale_u8x3() {
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
P::downscale_test(
@@ -534,7 +534,7 @@ fn upscale_u8x3() {
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(
@@ -566,7 +566,7 @@ fn downscale_u8x4() {
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
P::downscale_test(
@@ -604,7 +604,7 @@ fn upscale_u8x4() {
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(
@@ -632,7 +632,7 @@ fn downscale_u16() {
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
P::downscale_test(
@@ -660,7 +660,7 @@ fn upscale_u16() {
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(
@@ -692,7 +692,7 @@ fn downscale_u16x2() {
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
P::downscale_test(
@@ -724,7 +724,7 @@ fn upscale_u16x2() {
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(
@@ -756,7 +756,7 @@ fn downscale_u16x3() {
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
P::downscale_test(
@@ -788,7 +788,7 @@ fn upscale_u16x3() {
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(
@@ -820,7 +820,7 @@ fn downscale_u16x4() {
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
P::downscale_test(
@@ -852,7 +852,7 @@ fn upscale_u16x4() {
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(