- Added optimisation of multiplying and dividing image by alpha channel with helps of `SSE4.1` instructions.

- Improved performance of dividing image by alpha channel without forced SIMD instructions.
- Deleted variant `SSE2` from enum ``CpuExtensions``.
This commit is contained in:
Kirill Kuzminykh
2022-01-12 23:38:32 +03:00
parent 8ed04a546f
commit d4ead15856
22 changed files with 675 additions and 526 deletions
+9
View File
@@ -1,3 +1,12 @@
## [Unreleased] - ReleaseDate
- Added optimisation of multiplying and dividing image by alpha channel with helps
of ``SSE4.1`` instructions.
- Improved performance of dividing image by alpha channel without forced
SIMD instructions.
- Breaking changes:
- Deleted variant `SSE2` from enum ``CpuExtensions``.
## [0.5.3] - 2021-12-14
- Added optimisation of convolution U8x3 images with helps of ``AVX2`` instructions.
Generated
+54 -85
View File
@@ -33,9 +33,9 @@ dependencies = [
[[package]]
name = "anyhow"
version = "1.0.51"
version = "1.0.52"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8b26702f315f53b6071259e15dd9d64528213b44d61de1ec926eca7715d62203"
checksum = "84450d0b4a8bd1ba4144ce8ce718fbc5d071358b1e5384bace6536b3d1f2d5b3"
[[package]]
name = "argh"
@@ -98,9 +98,9 @@ dependencies = [
[[package]]
name = "bytemuck"
version = "1.7.2"
version = "1.7.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "72957246c41db82b8ef88a5486143830adeb8227ef9837740bdec67724cf2c5b"
checksum = "439989e6b8c38d1b6570a384ef1e49c8848128f5a97f3914baef02920842712f"
[[package]]
name = "byteorder"
@@ -178,9 +178,9 @@ dependencies = [
[[package]]
name = "crossbeam-channel"
version = "0.5.1"
version = "0.5.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "06ed27e177f16d65f0f0c22a213e17c696ace5dd64b14258b52f9417ccb52db4"
checksum = "e54ea8bc3fb1ee042f5aace6e3c6e025d3874866da222930f70ce62aceba0bfa"
dependencies = [
"cfg-if",
"crossbeam-utils",
@@ -199,9 +199,9 @@ dependencies = [
[[package]]
name = "crossbeam-epoch"
version = "0.9.5"
version = "0.9.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "4ec02e091aa634e2c3ada4a392989e7c3116673ef0ac5b72232439094d73b7fd"
checksum = "97242a70df9b89a65d0b6df3c4bf5b9ce03c5b7309019777fbde37e7537f8762"
dependencies = [
"cfg-if",
"crossbeam-utils",
@@ -212,9 +212,9 @@ dependencies = [
[[package]]
name = "crossbeam-queue"
version = "0.3.2"
version = "0.3.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9b10ddc024425c88c2ad148c1b0fd53f4c6d38db9697c9f1588381212fa657c9"
checksum = "b979d76c9fcb84dffc80a73f7290da0f83e4c95773494674cb44b76d13a7a110"
dependencies = [
"cfg-if",
"crossbeam-utils",
@@ -222,9 +222,9 @@ dependencies = [
[[package]]
name = "crossbeam-utils"
version = "0.8.5"
version = "0.8.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d82cfc11ce7f2c3faef78d8a684447b40d503d9681acebed6cb728d45940c4db"
checksum = "cfcae03edb34f947e64acdb1c33ec169824e20657e9ecb61cef6c8c74dcb8120"
dependencies = [
"cfg-if",
"lazy_static",
@@ -263,7 +263,7 @@ checksum = "22813a6dc45b335f9bade10bf7271dc477e81113e89eb251a0bc2a8a81c536e1"
dependencies = [
"bstr",
"csv-core",
"itoa",
"itoa 0.4.8",
"ryu",
"serde",
]
@@ -349,9 +349,9 @@ checksum = "7360491ce676a36bf9bb3c56c1aa791658183a54d2744120f27285738d90465a"
[[package]]
name = "fallible_collections"
version = "0.4.3"
version = "0.4.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "eaefd4190151d458f16f0793d3452d7f13aeb3701566a4cefc4c37598876cc00"
checksum = "52db5973b6a19247baf19b30f41c23a1bfffc2e9ce0a5db2f60e3cd5dc8895f7"
dependencies = [
"hashbrown 0.11.2",
]
@@ -368,6 +368,15 @@ dependencies = [
"thiserror",
]
[[package]]
name = "fastrand"
version = "1.6.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "779d043b6a0b90cc4c0ed7ee380a6504394cee7efd7db050e3774eee387324b2"
dependencies = [
"instant",
]
[[package]]
name = "form_urlencoded"
version = "1.0.1"
@@ -525,6 +534,12 @@ version = "0.4.8"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b71991ff56294aa922b450139ee08b3bfc70982c6b2c7562771375cf73542dd4"
[[package]]
name = "itoa"
version = "1.0.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1aab8fc367588b89dcee83ab0fd66b72b50b72fa1904d7095045ace2b0c81c35"
[[package]]
name = "jobserver"
version = "0.1.24"
@@ -731,9 +746,9 @@ dependencies = [
[[package]]
name = "num_cpus"
version = "1.13.0"
version = "1.13.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "05499f3756671c15885fee9034446956fff3f243d6077b91e5767df161f766b3"
checksum = "19e64526ebdee182341572e50e9ad03965aa510cd94427a4549448f285e957a1"
dependencies = [
"hermit-abi",
"libc",
@@ -741,9 +756,9 @@ dependencies = [
[[package]]
name = "once_cell"
version = "1.8.0"
version = "1.9.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "692fcb63b64b1758029e0a96ee63e049ce8c5948587f2f7208df04625e5f6b56"
checksum = "da32515d9f6e6e489d7bc9d84c71b060db7247dc035bbe44eac88cf87486d8d5"
[[package]]
name = "open"
@@ -810,70 +825,24 @@ dependencies = [
"miniz_oxide 0.3.7",
]
[[package]]
name = "ppv-lite86"
version = "0.2.15"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ed0cfbc8191465bed66e1718596ee0b0b35d5ee1f41c5df2189d0fe8bde535ba"
[[package]]
name = "proc-macro2"
version = "1.0.33"
version = "1.0.36"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "fb37d2df5df740e582f28f8560cf425f52bb267d872fe58358eadb554909f07a"
checksum = "c7342d5883fbccae1cc37a2353b09c87c9b0f3afd73f5fb9bba687a1f733b029"
dependencies = [
"unicode-xid",
]
[[package]]
name = "quote"
version = "1.0.10"
version = "1.0.14"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "38bc8cc6a5f2e3655e0899c1b848643b2562f853f114bfec7be120678e3ace05"
checksum = "47aa80447ce4daf1717500037052af176af5d38cc3e571d9ec1c7353fc10c87d"
dependencies = [
"proc-macro2",
]
[[package]]
name = "rand"
version = "0.8.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "2e7573632e6454cf6b99d7aac4ccca54be06da05aca2ef7423d22d27d4d4bcd8"
dependencies = [
"libc",
"rand_chacha",
"rand_core",
"rand_hc",
]
[[package]]
name = "rand_chacha"
version = "0.3.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e6c10a63a0fa32252be49d21e7709d4d4baf8d231c2dbce1eaa8141b9b127d88"
dependencies = [
"ppv-lite86",
"rand_core",
]
[[package]]
name = "rand_core"
version = "0.6.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d34f1408f55294453790c48b2f1ebbb1c5b4b7563eb1f418bcfcfdbb06ebb4e7"
dependencies = [
"getrandom",
]
[[package]]
name = "rand_hc"
version = "0.3.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d51e9f596de227fda2ea6c84607f5558e196eeaf43c986b724ba4fb8fdf497e7"
dependencies = [
"rand_core",
]
[[package]]
name = "rayon"
version = "1.5.1"
@@ -945,9 +914,9 @@ dependencies = [
[[package]]
name = "rgb"
version = "0.8.30"
version = "0.8.31"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "08a9852b34c4628f8ad76797a933577059163651ec5a7dace462adc365bee66c"
checksum = "9a374af9a0e5fdcdd98c1c7b64f05004f9ea2555b6c75f211daa81268a3c50f1"
dependencies = [
"bytemuck",
]
@@ -987,18 +956,18 @@ checksum = "d29ab0c6d3fc0ee92fe66e2d99f700eab17a8d57d1c1d3b748380fb20baa78cd"
[[package]]
name = "serde"
version = "1.0.131"
version = "1.0.133"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b4ad69dfbd3e45369132cc64e6748c2d65cdfb001a2b1c232d128b4ad60561c1"
checksum = "97565067517b60e2d1ea8b268e59ce036de907ac523ad83a0475da04e818989a"
dependencies = [
"serde_derive",
]
[[package]]
name = "serde_derive"
version = "1.0.131"
version = "1.0.133"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b710a83c4e0dff6a3d511946b95274ad9ca9e5d3ae497b63fda866ac955358d2"
checksum = "ed201699328568d8d08208fdd080e3ff594e6c422e438b6705905da01005d537"
dependencies = [
"proc-macro2",
"quote",
@@ -1007,11 +976,11 @@ dependencies = [
[[package]]
name = "serde_json"
version = "1.0.72"
version = "1.0.74"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d0ffa0837f2dfa6fb90868c2b5468cad482e175f7dad97e7421951e663f2b527"
checksum = "ee2bb9cd061c5865d345bb02ca49fcef1391741b672b54a0bf7b679badec3142"
dependencies = [
"itoa",
"itoa 1.0.1",
"ryu",
"serde",
]
@@ -1050,9 +1019,9 @@ checksum = "3bdb25a4593d6656239319426f4025f7a658157e25e89f0e0319d7516d46042d"
[[package]]
name = "syn"
version = "1.0.82"
version = "1.0.85"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8daf5dd0bb60cbd4137b1b587d2fc0ae729bc07cf01cd70b36a1ed5ade3b9d59"
checksum = "a684ac3dcd8913827e18cd09a68384ee66c1de24157e3c556c9ab16d85695fb7"
dependencies = [
"proc-macro2",
"quote",
@@ -1061,13 +1030,13 @@ dependencies = [
[[package]]
name = "tempfile"
version = "3.2.0"
version = "3.3.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "dac1c663cfc93810f88aed9b8941d48cabf856a1b111c29a40439018d870eb22"
checksum = "5cdb1ef4eaeeaddc8fbd371e5017057064af0911902ef36b39801f67cc6d79e4"
dependencies = [
"cfg-if",
"fastrand",
"libc",
"rand",
"redox_syscall",
"remove_dir_all",
"winapi",
@@ -1196,9 +1165,9 @@ checksum = "accd4ea62f7bb7a82fe23066fb0957d48ef677f6eeb8215f372f52e48bb32426"
[[package]]
name = "version_check"
version = "0.9.3"
version = "0.9.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "5fecdca9a5291cc2b8dcf7dc02453fee791a280f3743cb0905f8822ae463b3fe"
checksum = "49874b5167b65d7193b8aba1567f5c7d93d001cafc34600cee003eda787e483f"
[[package]]
name = "wasi"
+1 -1
View File
@@ -22,7 +22,7 @@ thiserror = "1.0.30"
glassbench = "0.3.1"
image = "0.23.14"
resize = "0.7.2"
rgb = "0.8.30"
rgb = "0.8.31"
[[bench]]
+23 -17
View File
@@ -2,6 +2,12 @@
Rust library for fast image resizing with using of SIMD instructions.
_Note: This library does not support converting image color spaces.
If it is important for you to resize images with a non-linear color space
(e.g. sRGB) correctly, then you need to convert it to a linear color space
before resizing. [Read more](https://legacy.imagemagick.org/Usage/resize/#resize_colorspace)
about resizing with respect to color space._
[CHANGELOG](https://github.com/Cykooz/fast_image_resize/blob/main/CHANGELOG.md)
Supported pixel formats and available optimisations:
@@ -28,7 +34,7 @@ Environment:
- RAM: DDR4 3000 MHz
- Ubuntu 20.04 (linux 5.11)
- Rust 1.57.0
- fast_image_resize = "0.5.3"
- fast_image_resize = "0.6.0"
- glassbench = "0.3.1"
- `rustflags = ["-C", "llvm-args=-x86-branches-within-32B-boundaries"]`
@@ -53,11 +59,11 @@ Pipeline:
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|------------|:-------:|:--------:|:----------:|:--------:|
| image | 91.896 | 176.959 | 256.786 | 341.548 |
| resize | 15.453 | 71.340 | 131.232 | 191.469 |
| fir rust | 0.481 | 53.733 | 90.519 | 121.121 |
| fir sse4.1 | - | 43.113 | 53.484 | 75.127 |
| fir avx2 | - | 10.765 | 14.131 | 19.827 |
| image | 92.300 | 180.063 | 257.043 | 342.268 |
| resize | 15.429 | 71.589 | 131.447 | 191.048 |
| fir rust | 0.481 | 55.347 | 89.512 | 121.270 |
| fir sse4.1 | - | 44.841 | 54.630 | 75.284 |
| fir avx2 | - | 10.814 | 14.408 | 20.897 |
### Resize RGBA image (U8x4) 4928x3279 => 852x567
@@ -65,16 +71,16 @@ Pipeline:
`src_image => multiply by alpha => resize => divide by alpha => dst_image`
- Source image [nasa-4928x3279.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279.png)
- Source image [nasa-4928x3279-rgba.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279-rgba.png)
- Numbers in table is mean duration of image resizing in milliseconds.
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|------------|:-------:|:--------:|:----------:|:--------:|
| image | 98.113 | 177.039 | 254.666 | 337.147 |
| resize | 17.875 | 79.014 | 148.691 | 218.400 |
| fir rust | 13.188 | 63.942 | 89.681 | 119.664 |
| fir sse4.1 | 11.868 | 22.957 | 29.164 | 36.799 |
| fir avx2 | 6.949 | 14.854 | 18.399 | 23.772 |
| image | 98.598 | 177.427 | 253.149 | 336.999 |
| resize | 17.988 | 79.891 | 149.178 | 218.593 |
| fir rust | 11.952 | 63.214 | 89.397 | 118.714 |
| fir sse4.1 | 9.169 | 20.664 | 26.442 | 34.326 |
| fir avx2 | 7.353 | 15.514 | 18.936 | 24.624 |
### Resize grayscale image (U8) 4928x3279 => 852x567
@@ -88,10 +94,10 @@ Pipeline:
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|----------|:-------:|:--------:|:----------:|:--------:|
| image | 76.981 | 126.595 | 166.765 | 209.593 |
| resize | 9.632 | 24.332 | 47.533 | 80.667 |
| fir rust | 0.197 | 21.773 | 24.476 | 34.909 |
| fir avx2 | - | 9.467 | 7.691 | 11.776 |
| image | 77.209 | 128.704 | 166.905 | 210.503 |
| resize | 9.694 | 24.477 | 47.722 | 80.824 |
| fir rust | 0.198 | 21.457 | 24.256 | 36.893 |
| fir avx2 | - | 9.758 | 7.787 | 12.099 |
## Examples
@@ -162,7 +168,7 @@ fn resize_image_example() {
### Change CPU extensions used by resizer
```ignore
```rust, ignore
use fast_image_resize as fr;
fn main() {
+9 -8
View File
@@ -40,7 +40,7 @@ fn multiplies_alpha_avx2(bench: &mut Bench) {
}
#[cfg(target_arch = "x86_64")]
fn multiplies_alpha_sse2(bench: &mut Bench) {
fn multiplies_alpha_sse4(bench: &mut Bench) {
let width = NonZeroU32::new(4096).unwrap();
let height = NonZeroU32::new(2048).unwrap();
let src_data = get_src_image(width, height, p(255, 128, 0, 128));
@@ -49,10 +49,10 @@ fn multiplies_alpha_sse2(bench: &mut Bench) {
let mut dst_view = dst_data.view_mut();
let mut alpha_mul_div: MulDiv = Default::default();
unsafe {
alpha_mul_div.set_cpu_extensions(CpuExtensions::Sse2);
alpha_mul_div.set_cpu_extensions(CpuExtensions::Sse4_1);
}
bench.task("Multiplies alpha SSE2", |task| {
bench.task("Multiplies alpha SSE4.1", |task| {
task.iter(|| {
alpha_mul_div
.multiply_alpha(&src_view, &mut dst_view)
@@ -105,7 +105,7 @@ fn divides_alpha_avx2(bench: &mut Bench) {
}
#[cfg(target_arch = "x86_64")]
fn divides_alpha_sse2(bench: &mut Bench) {
fn divides_alpha_sse4(bench: &mut Bench) {
let width = NonZeroU32::new(4096).unwrap();
let height = NonZeroU32::new(2048).unwrap();
let src_data = get_src_image(width, height, p(128, 64, 0, 128));
@@ -114,10 +114,10 @@ fn divides_alpha_sse2(bench: &mut Bench) {
let mut dst_view = dst_data.view_mut();
let mut alpha_mul_div: MulDiv = Default::default();
unsafe {
alpha_mul_div.set_cpu_extensions(CpuExtensions::Sse2);
alpha_mul_div.set_cpu_extensions(CpuExtensions::Sse4_1);
}
bench.task("Divides alpha SSE2", |task| {
bench.task("Divides alpha SSE4.1", |task| {
task.iter(|| {
alpha_mul_div
.divide_alpha(&src_view, &mut dst_view)
@@ -149,6 +149,7 @@ fn divides_alpha_native(bench: &mut Bench) {
pub fn main() {
use glassbench::*;
let name = env!("CARGO_CRATE_NAME");
let cmd = Command::read();
if cmd.include_bench(name) {
@@ -156,13 +157,13 @@ pub fn main() {
#[cfg(target_arch = "x86_64")]
{
multiplies_alpha_avx2(&mut bench);
multiplies_alpha_sse2(&mut bench);
multiplies_alpha_sse4(&mut bench);
}
multiplies_alpha_native(&mut bench);
#[cfg(target_arch = "x86_64")]
{
divides_alpha_avx2(&mut bench);
divides_alpha_sse2(&mut bench);
divides_alpha_sse4(&mut bench);
}
divides_alpha_native(&mut bench);
if let Err(e) = after_bench(&mut bench, &cmd) {
Binary file not shown.

After

Width:  |  Height:  |  Size: 675 KiB

+36 -41
View File
@@ -1,79 +1,74 @@
use std::arch::x86_64::*;
use crate::alpha::native;
use crate::alpha::sse4;
use crate::image_view::{TypedImageView, TypedImageViewMut};
use crate::pixels::U8x4;
use crate::simd_utils;
pub(crate) fn divide_alpha_avx2(
#[target_feature(enable = "avx2")]
pub(crate) unsafe fn divide_alpha_avx2(
src_image: TypedImageView<U8x4>,
mut dst_image: TypedImageViewMut<U8x4>,
) {
let width = src_image.width().get();
let src_rows = src_image.iter_rows(0);
let dst_rows = dst_image.iter_rows_mut();
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
divide_alpha_row_avx2(src_row, dst_row, width as usize);
}
}
}
pub(crate) fn divide_alpha_inplace_avx2(mut image: TypedImageViewMut<U8x4>) {
let width = image.width().get() as usize;
for dst_row in image.iter_rows_mut() {
unsafe {
let src_row = std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len());
divide_alpha_row_avx2(src_row, dst_row, width);
}
divide_alpha_row_avx2(src_row, dst_row);
}
}
#[target_feature(enable = "avx2")]
unsafe fn divide_alpha_row_avx2(src_row: &[U8x4], dst_row: &mut [U8x4], width: usize) {
let mut x: usize = 0;
pub(crate) unsafe fn divide_alpha_inplace_avx2(mut image: TypedImageViewMut<U8x4>) {
for dst_row in image.iter_rows_mut() {
let src_row = std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len());
divide_alpha_row_avx2(src_row, dst_row);
}
}
#[target_feature(enable = "avx2")]
unsafe fn divide_alpha_row_avx2(src_row: &[U8x4], dst_row: &mut [U8x4]) {
let zero = _mm256_setzero_si256();
let alpha_mask = _mm256_set1_epi32(0xff000000u32 as i32);
#[rustfmt::skip]
let alpha_mask = _mm256_set1_epi32(0xff000000u32 as i32);
#[rustfmt::skip]
let shuffle1 = _mm256_set_epi8(
let shuffle1 = _mm256_set_epi8(
5, 4, 5, 4, 5, 4, 5, 4, 1, 0, 1, 0, 1, 0, 1, 0,
5, 4, 5, 4, 5, 4, 5, 4, 1, 0, 1, 0, 1, 0, 1, 0,
);
#[rustfmt::skip]
let shuffle2 = _mm256_set_epi8(
let shuffle2 = _mm256_set_epi8(
13, 12, 13, 12, 13, 12, 13, 12, 9, 8, 9, 8, 9, 8, 9, 8,
13, 12, 13, 12, 13, 12, 13, 12, 9, 8, 9, 8, 9, 8, 9, 8,
);
let alpha_scale = _mm256_set1_ps(255.0 * 256.0);
while x < width.saturating_sub(7) {
let mut source = simd_utils::loadu_si256(src_row, x);
let src_chunks = src_row.chunks_exact(8);
let src_remainder = src_chunks.remainder();
let mut dst_chunks = dst_row.chunks_exact_mut(8);
let alpha_f32 = _mm256_cvtepi32_ps(_mm256_srli_epi32::<24>(source));
let scaled_alpha_f32 = _mm256_mul_ps(alpha_scale, _mm256_rcp_ps(alpha_f32));
for (src, dst) in src_chunks.zip(&mut dst_chunks) {
let src_pixels = _mm256_loadu_si256(src.as_ptr() as *const __m256i);
let alpha_f32 = _mm256_cvtepi32_ps(_mm256_srli_epi32::<24>(src_pixels));
let scaled_alpha_f32 = _mm256_div_ps(alpha_scale, alpha_f32);
let scaled_alpha_i32 = _mm256_cvtps_epi32(scaled_alpha_f32);
let mma0 = _mm256_shuffle_epi8(scaled_alpha_i32, shuffle1);
let mma1 = _mm256_shuffle_epi8(scaled_alpha_i32, shuffle2);
let mut pix0 = _mm256_unpacklo_epi8(zero, source);
let mut pix1 = _mm256_unpackhi_epi8(zero, source);
let pix0 = _mm256_unpacklo_epi8(zero, src_pixels);
let pix1 = _mm256_unpackhi_epi8(zero, src_pixels);
pix0 = _mm256_mulhi_epu16(pix0, mma0);
pix1 = _mm256_mulhi_epu16(pix1, mma1);
let pix0 = _mm256_mulhi_epu16(pix0, mma0);
let pix1 = _mm256_mulhi_epu16(pix1, mma1);
let alpha = _mm256_and_si256(source, alpha_mask);
source = _mm256_packus_epi16(pix0, pix1);
source = _mm256_blendv_epi8(source, alpha, alpha_mask);
let alpha = _mm256_and_si256(src_pixels, alpha_mask);
let rgb = _mm256_packus_epi16(pix0, pix1);
let dst_pixels = _mm256_blendv_epi8(rgb, alpha, alpha_mask);
let dst_ptr = dst_row.get_unchecked_mut(x..).as_mut_ptr() as *mut __m256i;
_mm256_storeu_si256(dst_ptr, source);
x += 8;
_mm256_storeu_si256(dst.as_mut_ptr() as *mut __m256i, dst_pixels);
}
let src_tail = &src_row[x..];
let dst_tail = &mut dst_row[x..];
native::divide_alpha_row_native(src_tail, dst_tail);
if !src_remainder.is_empty() {
let dst_reminder = dst_chunks.into_remainder();
sse4::div::divide_alpha_row_sse4(src_remainder, dst_reminder);
}
}
+2 -5
View File
@@ -1,5 +1,2 @@
pub(crate) use div::{divide_alpha_avx2, divide_alpha_inplace_avx2};
pub(crate) use mul::{multiply_alpha_avx2, multiply_alpha_inplace_avx2};
mod div;
mod mul;
pub(crate) mod div;
pub(crate) mod mul;
+32 -32
View File
@@ -5,7 +5,8 @@ use crate::image_view::{TypedImageView, TypedImageViewMut};
use crate::pixels::U8x4;
use crate::simd_utils;
pub(crate) fn multiply_alpha_avx2(
#[target_feature(enable = "avx2")]
pub(crate) unsafe fn multiply_alpha_avx2(
src_image: TypedImageView<U8x4>,
mut dst_image: TypedImageViewMut<U8x4>,
) {
@@ -14,62 +15,61 @@ pub(crate) fn multiply_alpha_avx2(
let dst_rows = dst_image.iter_rows_mut();
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
multiply_alpha_row_avx2(src_row, dst_row, width);
}
multiply_alpha_row_avx2(src_row, dst_row, width);
}
}
pub(crate) fn multiply_alpha_inplace_avx2(mut image: TypedImageViewMut<U8x4>) {
#[target_feature(enable = "avx2")]
pub(crate) unsafe fn multiply_alpha_inplace_avx2(mut image: TypedImageViewMut<U8x4>) {
let width = image.width().get() as usize;
for dst_row in image.iter_rows_mut() {
unsafe {
let src_row = std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len());
multiply_alpha_row_avx2(src_row, dst_row, width);
}
let src_row = std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len());
multiply_alpha_row_avx2(src_row, dst_row, width);
}
}
/// https://github.com/Wizermil/premultiply_alpha/blob/master/premultiply_alpha/premultiply_alpha.hpp#L232
#[inline]
#[target_feature(enable = "avx2")]
unsafe fn multiply_alpha_row_avx2(src_row: &[U8x4], dst_row: &mut [U8x4], width: usize) {
let mask_alpha_color_odd_255 = _mm256_set1_epi32(0xff000000u32 as i32);
let div_255 = _mm256_set1_epi16(0x8081u16 as i16);
let zero = _mm256_setzero_si256();
let half = _mm256_set1_epi16(128);
const MAX_A: i32 = 0xff000000u32 as i32;
let max_alpha = _mm256_set1_epi32(MAX_A);
#[rustfmt::skip]
let mask_shuffle_alpha = _mm256_set_epi8(
15, -1, 15, -1, 11, -1, 11, -1, 7, -1, 7, -1, 3, -1, 3, -1,
15, -1, 15, -1, 11, -1, 11, -1, 7, -1, 7, -1, 3, -1, 3, -1,
);
#[rustfmt::skip]
let mask_shuffle_color_odd = _mm256_set_epi8(
-1, -1, 13, -1, -1, -1, 9, -1, -1, -1, 5, -1, -1, -1, 1, -1,
-1, -1, 13, -1, -1, -1, 9, -1, -1, -1, 5, -1, -1, -1, 1, -1,
let factor_mask = _mm256_set_epi8(
15, 15, 15, 15, 11, 11, 11, 11, 7, 7, 7, 7, 3, 3, 3, 3,
15, 15, 15, 15, 11, 11, 11, 11, 7, 7, 7, 7, 3, 3, 3, 3,
);
let mut x: usize = 0;
while x < width.saturating_sub(7) {
let mut color = simd_utils::loadu_si256(src_row, x);
let src_pixels = simd_utils::loadu_si256(src_row, x);
let alpha = _mm256_shuffle_epi8(color, mask_shuffle_alpha);
let mut color_even = _mm256_slli_epi16::<8>(color);
let mut color_odd = _mm256_shuffle_epi8(color, mask_shuffle_color_odd);
color_odd = _mm256_or_si256(color_odd, mask_alpha_color_odd_255);
let factor_pixels = _mm256_shuffle_epi8(src_pixels, factor_mask);
let factor_pixels = _mm256_or_si256(factor_pixels, max_alpha);
color_odd = _mm256_mulhi_epu16(color_odd, alpha);
color_even = _mm256_mulhi_epu16(color_even, alpha);
let pix1 = _mm256_unpacklo_epi8(src_pixels, zero);
let factors = _mm256_unpacklo_epi8(factor_pixels, zero);
let pix1 = _mm256_add_epi16(_mm256_mullo_epi16(pix1, factors), half);
let pix1 = _mm256_add_epi16(pix1, _mm256_srli_epi16::<8>(pix1));
let pix1 = _mm256_srli_epi16::<8>(pix1);
color_odd = _mm256_srli_epi16::<7>(_mm256_mulhi_epu16(color_odd, div_255));
color_even = _mm256_srli_epi16::<7>(_mm256_mulhi_epu16(color_even, div_255));
let pix2 = _mm256_unpackhi_epi8(src_pixels, zero);
let factors = _mm256_unpackhi_epi8(factor_pixels, zero);
let pix2 = _mm256_add_epi16(_mm256_mullo_epi16(pix2, factors), half);
let pix2 = _mm256_add_epi16(pix2, _mm256_srli_epi16::<8>(pix2));
let pix2 = _mm256_srli_epi16::<8>(pix2);
color = _mm256_or_si256(color_even, _mm256_slli_epi16::<8>(color_odd));
let dst_pixels = _mm256_packus_epi16(pix1, pix2);
let dst_ptr = dst_row.get_unchecked_mut(x..).as_mut_ptr() as *mut __m256i;
_mm256_storeu_si256(dst_ptr, color);
_mm256_storeu_si256(dst_ptr, dst_pixels);
x += 8;
}
let src_tail = &src_row[x..];
let dst_tail = &mut dst_row[x..];
native::multiply_alpha_row_native(src_tail, dst_tail);
native::mul::multiply_alpha_row_native(src_tail, dst_tail);
}
+25 -26
View File
@@ -1,15 +1,16 @@
pub use errors::*;
use crate::image_view::{TypedImageView, TypedImageViewMut};
use crate::pixels::U8x4;
use crate::CpuExtensions;
use crate::{ImageView, ImageViewMut};
pub use errors::*;
#[cfg(target_arch = "x86_64")]
mod avx2;
mod errors;
mod native;
#[cfg(target_arch = "x86_64")]
mod sse2;
mod sse4;
/// Methods of this structure used to multiply or divide RGB-channels
/// by alpha-channel.
@@ -60,13 +61,14 @@ impl MulDiv {
let (src_image_u8x4, dst_image_u8x4) = assert_images(src_image, dst_image)?;
match self.cpu_extensions {
#[cfg(target_arch = "x86_64")]
CpuExtensions::Avx2 => avx2::multiply_alpha_avx2(src_image_u8x4, dst_image_u8x4),
// WARNING: SSE2 implementation is drastically slower than native version
// #[cfg(target_arch = "x86_64")]
// CpuExtensions::Sse4_1 | CpuExtensions::Sse2 => {
// sse2::multiply_alpha_sse2(src_image, dst_image)
// }
_ => native::multiply_alpha_native(src_image_u8x4, dst_image_u8x4),
CpuExtensions::Avx2 => unsafe {
avx2::mul::multiply_alpha_avx2(src_image_u8x4, dst_image_u8x4)
},
#[cfg(target_arch = "x86_64")]
CpuExtensions::Sse4_1 => unsafe {
sse4::mul::multiply_alpha_sse4(src_image_u8x4, dst_image_u8x4)
},
_ => native::mul::multiply_alpha_native(src_image_u8x4, dst_image_u8x4),
}
Ok(())
}
@@ -76,13 +78,10 @@ impl MulDiv {
let image_u8x4 = assert_image(image)?;
match self.cpu_extensions {
#[cfg(target_arch = "x86_64")]
CpuExtensions::Avx2 => avx2::multiply_alpha_inplace_avx2(image_u8x4),
// WARNING: SSE2 implementation is drastically slower than native version
// #[cfg(target_arch = "x86_64")]
// CpuExtensions::Sse4_1 | CpuExtensions::Sse2 => {
// sse2::multiply_alpha_sse2(src_image, dst_image)
// }
_ => native::multiply_alpha_inplace_native(image_u8x4),
CpuExtensions::Avx2 => unsafe { avx2::mul::multiply_alpha_inplace_avx2(image_u8x4) },
#[cfg(target_arch = "x86_64")]
CpuExtensions::Sse4_1 => unsafe { sse4::mul::multiply_alpha_inplace_sse4(image_u8x4) },
_ => native::mul::multiply_alpha_inplace_native(image_u8x4),
}
Ok(())
}
@@ -97,12 +96,14 @@ impl MulDiv {
let (src_image_u8x4, dst_image_u8x4) = assert_images(src_image, dst_image)?;
match self.cpu_extensions {
#[cfg(target_arch = "x86_64")]
CpuExtensions::Avx2 => avx2::divide_alpha_avx2(src_image_u8x4, dst_image_u8x4),
CpuExtensions::Avx2 => unsafe {
avx2::div::divide_alpha_avx2(src_image_u8x4, dst_image_u8x4)
},
#[cfg(target_arch = "x86_64")]
CpuExtensions::Sse4_1 | CpuExtensions::Sse2 => {
sse2::divide_alpha_sse2(src_image_u8x4, dst_image_u8x4)
}
_ => native::divide_alpha_native(src_image_u8x4, dst_image_u8x4),
CpuExtensions::Sse4_1 => unsafe {
sse4::div::divide_alpha_sse4(src_image_u8x4, dst_image_u8x4)
},
_ => native::div::divide_alpha_native(src_image_u8x4, dst_image_u8x4),
}
Ok(())
}
@@ -112,12 +113,10 @@ impl MulDiv {
let image_u8x4 = assert_image(image)?;
match self.cpu_extensions {
#[cfg(target_arch = "x86_64")]
CpuExtensions::Avx2 => avx2::divide_alpha_inplace_avx2(image_u8x4),
CpuExtensions::Avx2 => unsafe { avx2::div::divide_alpha_inplace_avx2(image_u8x4) },
#[cfg(target_arch = "x86_64")]
CpuExtensions::Sse4_1 | CpuExtensions::Sse2 => {
sse2::divide_alpha_inplace_sse2(image_u8x4)
}
_ => native::divide_alpha_inplace_native(image_u8x4),
CpuExtensions::Sse4_1 => unsafe { sse4::div::divide_alpha_inplace_sse4(image_u8x4) },
_ => native::div::divide_alpha_inplace_native(image_u8x4),
}
Ok(())
}
+67 -7
View File
@@ -1,6 +1,7 @@
use crate::image_view::{TypedImageView, TypedImageViewMut};
use crate::pixels::U8x4;
#[inline]
pub(crate) fn divide_alpha_native(
src_image: TypedImageView<U8x4>,
mut dst_image: TypedImageViewMut<U8x4>,
@@ -13,6 +14,7 @@ pub(crate) fn divide_alpha_native(
}
}
#[inline]
pub(crate) fn divide_alpha_inplace_native(mut image: TypedImageViewMut<U8x4>) {
for dst_row in image.iter_rows_mut() {
let src_row = unsafe { std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len()) };
@@ -20,12 +22,6 @@ pub(crate) fn divide_alpha_inplace_native(mut image: TypedImageViewMut<U8x4>) {
}
}
#[inline(always)]
pub(crate) fn div_and_clip(v: u8, recip_alpha: f32) -> u8 {
let res = v as f32 * recip_alpha;
res.min(255.) as u8
}
#[inline(always)]
pub(crate) fn divide_alpha_row_native(src_row: &[U8x4], dst_row: &mut [U8x4]) {
src_row
@@ -34,7 +30,7 @@ pub(crate) fn divide_alpha_row_native(src_row: &[U8x4], dst_row: &mut [U8x4]) {
.for_each(|(src_pixel, dst_pixel)| {
let components: [u8; 4] = src_pixel.0.to_le_bytes();
let alpha = components[3];
let recip_alpha = if alpha == 0 { 0. } else { 255. / alpha as f32 };
let recip_alpha = RECIP_ALPHA[alpha as usize];
dst_pixel.0 = u32::from_le_bytes([
div_and_clip(components[0], recip_alpha),
div_and_clip(components[1], recip_alpha),
@@ -43,3 +39,67 @@ pub(crate) fn divide_alpha_row_native(src_row: &[U8x4], dst_row: &mut [U8x4]) {
]);
});
}
const fn recip_alpha_array(precision: u32) -> [u32; 256] {
let mut res = [0; 256];
let scale = 1 << (precision + 1);
let mut i: usize = 1;
while i < 256 {
res[i] = (((255 * scale / i as u32) + 1) >> 1) as u32;
i += 1;
}
res
}
const PRECISION: u32 = 8;
#[inline(always)]
fn div_and_clip(v: u8, recip_alpha: u32) -> u8 {
((v as u32 * recip_alpha) >> PRECISION).min(255) as u8
}
const RECIP_ALPHA: [u32; 256] = recip_alpha_array(PRECISION);
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn test_recip_alpha_array() {
for alpha in 0..=255u8 {
let expected = if alpha == 0 {
0
} else {
let scale = (1 << PRECISION) as f64;
(255.0 * scale / alpha as f64).round() as u32
};
let recip_alpha = RECIP_ALPHA[alpha as usize];
assert_eq!(expected, recip_alpha, "alpha {}", alpha);
}
}
#[test]
fn test_div_and_clip() {
let mut err_sum: i32 = 0;
for alpha in 0..=255u8 {
for color in 0..=255u8 {
let multiplied_color = (color as f64 * alpha as f64 / 255.).round().min(255.) as u8;
let expected_color = if alpha == 0 {
0
} else {
let recip_alpha = 255. / alpha as f64;
let res = multiplied_color as f64 * recip_alpha;
res.min(255.) as u8
};
let recip_alpha = RECIP_ALPHA[alpha as usize];
let result_color = div_and_clip(multiplied_color, recip_alpha);
let delta = result_color as i32 - expected_color as i32;
err_sum += delta.abs();
}
}
assert_eq!(err_sum, 3468);
}
}
+2 -7
View File
@@ -1,7 +1,2 @@
pub(crate) use div::{divide_alpha_inplace_native, divide_alpha_native, divide_alpha_row_native};
pub(crate) use mul::{
multiply_alpha_inplace_native, multiply_alpha_native, multiply_alpha_row_native,
};
mod div;
mod mul;
pub(crate) mod div;
pub(crate) mod mul;
-69
View File
@@ -1,69 +0,0 @@
use std::arch::x86_64::*;
use crate::alpha::native;
use crate::image_view::{TypedImageView, TypedImageViewMut};
use crate::pixels::U8x4;
use crate::simd_utils;
pub(crate) fn divide_alpha_sse2(
src_image: TypedImageView<U8x4>,
mut dst_image: TypedImageViewMut<U8x4>,
) {
let width = src_image.width().get() as usize;
let src_rows = src_image.iter_rows(0);
let dst_rows = dst_image.iter_rows_mut();
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
divide_alpha_row_sse2(src_row, dst_row, width);
}
}
}
pub(crate) fn divide_alpha_inplace_sse2(mut image: TypedImageViewMut<U8x4>) {
let width = image.width().get() as usize;
for dst_row in image.iter_rows_mut() {
unsafe {
let src_row = std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len());
divide_alpha_row_sse2(src_row, dst_row, width);
}
}
}
#[target_feature(enable = "sse2")]
unsafe fn divide_alpha_row_sse2(src_row: &[U8x4], dst_row: &mut [U8x4], width: usize) {
let zero = _mm_setzero_si128();
let alpha_mask = _mm_set1_epi32(0xff000000u32 as i32);
let shuffle0 = _mm_set_epi8(5, 4, 5, 4, 5, 4, 5, 4, 1, 0, 1, 0, 1, 0, 1, 0);
let shuffle1 = _mm_set_epi8(13, 12, 13, 12, 13, 12, 13, 12, 9, 8, 9, 8, 9, 8, 9, 8);
let alpha_scale = _mm_set1_ps(255.0 * 256.0);
let mut x: usize = 0;
while x < width.saturating_sub(3) {
let mut source = simd_utils::loadu_si128(src_row, x);
let alpha = _mm_and_si128(source, alpha_mask);
let alpha_f32 = _mm_cvtepi32_ps(_mm_srli_epi32::<24>(source));
let scaled_recip_alpha_f32 = _mm_mul_ps(alpha_scale, _mm_rcp_ps(alpha_f32));
let scaled_recip_alpha_i32 = _mm_cvtps_epi32(scaled_recip_alpha_f32);
let mma0 = _mm_shuffle_epi8(scaled_recip_alpha_i32, shuffle0);
let mma1 = _mm_shuffle_epi8(scaled_recip_alpha_i32, shuffle1);
let mut pix0 = _mm_unpacklo_epi8(zero, source);
let mut pix1 = _mm_unpackhi_epi8(zero, source);
pix0 = _mm_mulhi_epu16(pix0, mma0);
pix1 = _mm_mulhi_epu16(pix1, mma1);
source = _mm_packus_epi16(pix0, pix1);
source = _mm_blendv_epi8(source, alpha, alpha_mask);
let dst_ptr = dst_row.get_unchecked_mut(x..).as_mut_ptr() as *mut __m128i;
_mm_storeu_si128(dst_ptr, source);
x += 4;
}
let src_tail = &src_row[x..];
let dst_tail = &mut dst_row[x..];
native::divide_alpha_row_native(src_tail, dst_tail);
}
-4
View File
@@ -1,4 +0,0 @@
pub(crate) use div::{divide_alpha_inplace_sse2, divide_alpha_sse2};
mod div;
mod mul;
-63
View File
@@ -1,63 +0,0 @@
use std::arch::x86_64::*;
use crate::alpha::native;
use crate::image_view::{TypedImageView, TypedImageViewMut};
use crate::pixels::U8x4;
use crate::simd_utils;
#[allow(dead_code)]
pub(crate) fn multiply_alpha_sse2(
src_image: TypedImageView<U8x4>,
mut dst_image: TypedImageViewMut<U8x4>,
) {
let width = src_image.width().get() as usize;
let src_rows = src_image.iter_rows(0);
let dst_rows = dst_image.iter_rows_mut();
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
multiply_alpha_row_sse2(src_row, dst_row, width);
}
}
}
/// https://github.com/Wizermil/premultiply_alpha/blob/master/premultiply_alpha/premultiply_alpha.hpp#L108
/// This implementation is twice slowly than native version.
#[allow(dead_code)]
#[target_feature(enable = "sse2")]
unsafe fn multiply_alpha_row_sse2(src_row: &[U8x4], dst_row: &mut [U8x4], width: usize) {
let mask_alpha_color_odd_255 = _mm_set1_epi32(0xff000000u32 as i32);
let div_255 = _mm_set1_epi16(0x8081u16 as i16);
let mask_shuffle_alpha =
_mm_set_epi8(15, -1, 15, -1, 11, -1, 11, -1, 7, -1, 7, -1, 3, -1, 3, -1);
let mask_shuffle_color_odd =
_mm_set_epi8(-1, -1, 13, -1, -1, -1, 9, -1, -1, -1, 5, -1, -1, -1, 1, -1);
let mut x: usize = 0;
while x < width.saturating_sub(3) {
let mut color = simd_utils::loadu_si128(src_row, x);
let alpha = _mm_shuffle_epi8(color, mask_shuffle_alpha);
let mut color_even = _mm_slli_epi16::<8>(color);
let mut color_odd = _mm_shuffle_epi8(color, mask_shuffle_color_odd);
color_odd = _mm_or_si128(color_odd, mask_alpha_color_odd_255);
color_odd = _mm_mulhi_epu16(color_odd, alpha);
color_even = _mm_mulhi_epu16(color_even, alpha);
color_odd = _mm_srli_epi16::<7>(_mm_mulhi_epu16(color_odd, div_255));
color_even = _mm_srli_epi16::<7>(_mm_mulhi_epu16(color_even, div_255));
color = _mm_or_si128(color_even, _mm_slli_epi16::<8>(color_odd));
let dst_ptr = dst_row.get_unchecked_mut(x..).as_mut_ptr() as *mut __m128i;
_mm_storeu_si128(dst_ptr, color);
x += 4;
}
let src_tail = &src_row[x..];
let dst_tail = &mut dst_row[x..];
native::multiply_alpha_row_native(src_tail, dst_tail);
}
+83
View File
@@ -0,0 +1,83 @@
use std::arch::x86_64::*;
use crate::image_view::{TypedImageView, TypedImageViewMut};
use crate::pixels::U8x4;
#[target_feature(enable = "sse4.1")]
pub(crate) unsafe fn divide_alpha_sse4(
src_image: TypedImageView<U8x4>,
mut dst_image: TypedImageViewMut<U8x4>,
) {
let src_rows = src_image.iter_rows(0);
let dst_rows = dst_image.iter_rows_mut();
for (src_row, dst_row) in src_rows.zip(dst_rows) {
divide_alpha_row_sse4(src_row, dst_row);
}
}
#[target_feature(enable = "sse4.1")]
pub(crate) unsafe fn divide_alpha_inplace_sse4(mut image: TypedImageViewMut<U8x4>) {
for dst_row in image.iter_rows_mut() {
let src_row = std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len());
divide_alpha_row_sse4(src_row, dst_row);
}
}
#[target_feature(enable = "sse4.1")]
pub(crate) unsafe fn divide_alpha_row_sse4(src_row: &[U8x4], dst_row: &mut [U8x4]) {
let src_chunks = src_row.chunks_exact(4);
let src_remainder = src_chunks.remainder();
let mut dst_chunks = dst_row.chunks_exact_mut(4);
for (src, dst) in src_chunks.zip(&mut dst_chunks) {
divide_alpha(src.as_ptr(), dst.as_mut_ptr());
}
if !src_remainder.is_empty() {
let dst_reminder = dst_chunks.into_remainder();
let mut src_pixels = [U8x4(0); 4];
src_pixels
.iter_mut()
.zip(src_remainder)
.for_each(|(d, s)| *d = *s);
let mut dst_pixels = [U8x4(0); 4];
divide_alpha(src_pixels.as_ptr(), dst_pixels.as_mut_ptr());
dst_pixels
.iter()
.zip(dst_reminder)
.for_each(|(s, d)| *d = *s);
}
}
#[target_feature(enable = "sse4.1")]
unsafe fn divide_alpha(src: *const U8x4, dst: *mut U8x4) {
let zero = _mm_setzero_si128();
let alpha_mask = _mm_set1_epi32(0xff000000u32 as i32);
let shuffle1 = _mm_set_epi8(5, 4, 5, 4, 5, 4, 5, 4, 1, 0, 1, 0, 1, 0, 1, 0);
let shuffle2 = _mm_set_epi8(13, 12, 13, 12, 13, 12, 13, 12, 9, 8, 9, 8, 9, 8, 9, 8);
let alpha_scale = _mm_set1_ps(255.0 * 256.0);
let src_pixels = _mm_loadu_si128(src as *const __m128i);
let alpha_f32 = _mm_cvtepi32_ps(_mm_srli_epi32::<24>(src_pixels));
let scaled_alpha_f32 = _mm_div_ps(alpha_scale, alpha_f32);
// let scaled_alpha_f32 = _mm_mul_ps(alpha_scale, _mm_rcp_ps(alpha_f32));
let scaled_alpha_i32 = _mm_cvtps_epi32(scaled_alpha_f32);
let mma0 = _mm_shuffle_epi8(scaled_alpha_i32, shuffle1);
let mma1 = _mm_shuffle_epi8(scaled_alpha_i32, shuffle2);
let pix0 = _mm_unpacklo_epi8(zero, src_pixels);
let pix1 = _mm_unpackhi_epi8(zero, src_pixels);
let pix0 = _mm_mulhi_epu16(pix0, mma0);
let pix1 = _mm_mulhi_epu16(pix1, mma1);
let alpha = _mm_and_si128(src_pixels, alpha_mask);
let rgb = _mm_packus_epi16(pix0, pix1);
let dst_pixels = _mm_blendv_epi8(rgb, alpha, alpha_mask);
_mm_storeu_si128(dst as *mut __m128i, dst_pixels);
}
+2
View File
@@ -0,0 +1,2 @@
pub(crate) mod div;
pub(crate) mod mul;
+69
View File
@@ -0,0 +1,69 @@
use std::arch::x86_64::*;
use crate::alpha::native;
use crate::image_view::{TypedImageView, TypedImageViewMut};
use crate::pixels::U8x4;
#[target_feature(enable = "sse4.1")]
pub(crate) unsafe fn multiply_alpha_sse4(
src_image: TypedImageView<U8x4>,
mut dst_image: TypedImageViewMut<U8x4>,
) {
let src_rows = src_image.iter_rows(0);
let dst_rows = dst_image.iter_rows_mut();
for (src_row, dst_row) in src_rows.zip(dst_rows) {
multiply_alpha_row_sse4(src_row, dst_row);
}
}
#[target_feature(enable = "sse4.1")]
pub(crate) unsafe fn multiply_alpha_inplace_sse4(mut image: TypedImageViewMut<U8x4>) {
for dst_row in image.iter_rows_mut() {
let src_row = std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len());
multiply_alpha_row_sse4(src_row, dst_row);
}
}
#[inline]
#[target_feature(enable = "sse4.1")]
unsafe fn multiply_alpha_row_sse4(src_row: &[U8x4], dst_row: &mut [U8x4]) {
let zero = _mm_setzero_si128();
let half = _mm_set1_epi16(128);
const MAX_A: i32 = 0xff000000u32 as i32;
let max_alpha = _mm_set1_epi32(MAX_A);
let factor_mask = _mm_set_epi8(15, 15, 15, 15, 11, 11, 11, 11, 7, 7, 7, 7, 3, 3, 3, 3);
let src_chunks = src_row.chunks_exact(4);
let src_remainder = src_chunks.remainder();
let mut dst_chunks = dst_row.chunks_exact_mut(4);
for (src, dst) in src_chunks.zip(&mut dst_chunks) {
let src_pixels = _mm_loadu_si128(src.as_ptr() as *const __m128i);
let factor_pixels = _mm_shuffle_epi8(src_pixels, factor_mask);
let factor_pixels = _mm_or_si128(factor_pixels, max_alpha);
let pix1 = _mm_unpacklo_epi8(src_pixels, zero);
let factors = _mm_unpacklo_epi8(factor_pixels, zero);
let pix1 = _mm_add_epi16(_mm_mullo_epi16(pix1, factors), half);
let pix1 = _mm_add_epi16(pix1, _mm_srli_epi16::<8>(pix1));
let pix1 = _mm_srli_epi16::<8>(pix1);
let pix2 = _mm_unpackhi_epi8(src_pixels, zero);
let factors = _mm_unpackhi_epi8(factor_pixels, zero);
let pix2 = _mm_add_epi16(_mm_mullo_epi16(pix2, factors), half);
let pix2 = _mm_add_epi16(pix2, _mm_srli_epi16::<8>(pix2));
let pix2 = _mm_srli_epi16::<8>(pix2);
let dst_pixels = _mm_packus_epi16(pix1, pix2);
_mm_storeu_si128(dst.as_mut_ptr() as *mut __m128i, dst_pixels);
}
if !src_remainder.is_empty() {
let dst_reminder = dst_chunks.into_remainder();
native::mul::multiply_alpha_row_native(src_remainder, dst_reminder);
}
}
-4
View File
@@ -10,8 +10,6 @@ use crate::pixels::{Pixel, PixelType};
pub enum CpuExtensions {
None,
#[cfg(target_arch = "x86_64")]
Sse2,
#[cfg(target_arch = "x86_64")]
Sse4_1,
#[cfg(target_arch = "x86_64")]
Avx2,
@@ -24,8 +22,6 @@ impl Default for CpuExtensions {
Self::Avx2
} else if is_x86_feature_detected!("sse4.1") {
Self::Sse4_1
} else if is_x86_feature_detected!("sse2") {
Self::Sse2
} else {
Self::None
}
+86 -4
View File
@@ -4,6 +4,9 @@ use fast_image_resize::pixels::U8x4;
use fast_image_resize::{
CpuExtensions, Image, ImageRows, ImageRowsMut, ImageView, ImageViewMut, MulDiv, PixelType,
};
use utils::{cpu_ext_into_str, image_checksum};
mod utils;
const fn p(r: u8, g: u8, b: u8, a: u8) -> U8x4 {
U8x4(u32::from_le_bytes([r, g, b, a]))
@@ -83,8 +86,8 @@ fn multiply_alpha_avx2_test() {
#[cfg(target_arch = "x86_64")]
#[test]
fn multiply_alpha_sse2_test() {
multiply_alpha_test(CpuExtensions::Sse2);
fn multiply_alpha_sse4_test() {
multiply_alpha_test(CpuExtensions::Sse4_1);
}
#[test]
@@ -164,11 +167,90 @@ fn divide_alpha_avx2_test() {
#[cfg(target_arch = "x86_64")]
#[test]
fn divide_alpha_sse2_test() {
divide_alpha_test(CpuExtensions::Sse2);
fn divide_alpha_sse4_test() {
divide_alpha_test(CpuExtensions::Sse4_1);
}
#[test]
fn divide_alpha_native_test() {
divide_alpha_test(CpuExtensions::None);
}
#[test]
fn multiply_alpha_real_image_test() {
let mut pixels = vec![0u32; 256 * 256];
let mut i: usize = 0;
for alpha in 0..=255u8 {
for color in 0..=255u8 {
let pixel = u32::from_le_bytes([color, color, color, alpha]);
pixels[i] = pixel;
i += 1;
}
}
let size = NonZeroU32::new(256).unwrap();
let src_image = Image::from_vec_u32(size, size, pixels, PixelType::U8x4).unwrap();
let mut dst_image = Image::new(size, size, PixelType::U8x4);
let mut alpha_mul_div: MulDiv = Default::default();
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
for cpu_extensions in cpu_extensions_vec {
unsafe {
alpha_mul_div.set_cpu_extensions(cpu_extensions);
}
alpha_mul_div
.multiply_alpha(&src_image.view(), &mut dst_image.view_mut())
.unwrap();
let name = format!("multiple_alpha-{}", cpu_ext_into_str(cpu_extensions));
utils::save_result(&dst_image, &name);
let checksum = image_checksum::<4>(dst_image.buffer());
assert_eq!(checksum, [4177920, 4177920, 4177920, 8355840]);
}
}
#[test]
fn divide_alpha_real_image_test() {
let mut pixels = vec![0u32; 256 * 256];
let mut i: usize = 0;
for alpha in 0..=255u8 {
for color in 0..=255u8 {
let multiplied_color = (color as f64 * (alpha as f64 / 255.)).round().min(255.) as u8;
let pixel =
u32::from_le_bytes([multiplied_color, multiplied_color, multiplied_color, alpha]);
pixels[i] = pixel;
i += 1;
}
}
let size = NonZeroU32::new(256).unwrap();
let src_image = Image::from_vec_u32(size, size, pixels, PixelType::U8x4).unwrap();
let mut dst_image = Image::new(size, size, PixelType::U8x4);
let mut alpha_mul_div: MulDiv = Default::default();
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
for cpu_extensions in cpu_extensions_vec {
unsafe {
alpha_mul_div.set_cpu_extensions(cpu_extensions);
}
alpha_mul_div
.divide_alpha(&src_image.view(), &mut dst_image.view_mut())
.unwrap();
let name = format!("divide_alpha-{}", cpu_ext_into_str(cpu_extensions));
utils::save_result(&dst_image, &name);
let checksum = image_checksum::<4>(dst_image.buffer());
assert_eq!(checksum, [8292504, 8292504, 8292504, 8355840]);
}
}
+30 -153
View File
@@ -1,15 +1,13 @@
use std::fs::File;
use std::num::NonZeroU32;
use image::codecs::png::PngEncoder;
use image::io::Reader as ImageReader;
use image::{ColorType, DynamicImage, GenericImageView};
use fast_image_resize::pixels::*;
use fast_image_resize::{
CpuExtensions, DifferentTypesOfPixelsError, FilterType, Image, ImageView, PixelType, ResizeAlg,
Resizer,
};
use utils::{cpu_ext_into_str, PixelExt};
mod utils;
fn get_new_height(src_image: &ImageView, new_width: u32) -> u32 {
let scale = new_width as f32 / src_image.width().get() as f32;
@@ -19,36 +17,6 @@ fn get_new_height(src_image: &ImageView, new_width: u32) -> u32 {
const NEW_WIDTH: u32 = 255;
const NEW_BIG_WIDTH: u32 = 5016;
fn save_result(image: &Image, name: &str) {
if std::env::var("DONT_SAVE_RESULT").unwrap_or_else(|_| "".to_owned()) == "1" {
return;
}
std::fs::create_dir_all("./data/result").unwrap();
let mut file = File::create(format!("./data/result/{}.png", name)).unwrap();
let color_type = match image.pixel_type() {
PixelType::U8x3 => ColorType::Rgb8,
PixelType::U8x4 => ColorType::Rgba8,
PixelType::U8 => ColorType::L8,
_ => panic!("Unsupported type of pixels"),
};
PngEncoder::new(&mut file)
.encode(
image.buffer(),
image.width().get(),
image.height().get(),
color_type,
)
.unwrap();
}
fn image_checksum<const N: usize>(buffer: &[u8]) -> [u32; N] {
let mut res = [0u32; N];
for pixel in buffer.chunks_exact(N) {
res.iter_mut().zip(pixel).for_each(|(d, &s)| *d += s as u32);
}
res
}
#[test]
fn try_resize_to_other_pixel_type() {
let src_image = U8x4::load_big_src_image();
@@ -64,88 +32,6 @@ fn try_resize_to_other_pixel_type() {
));
}
trait PixelExt: Pixel {
fn pixel_type_str() -> &'static str {
match Self::pixel_type() {
PixelType::U8 => "u8",
PixelType::U8x3 => "u8x3",
PixelType::U8x4 => "u8x4",
PixelType::I32 => "i32",
PixelType::F32 => "f32",
}
}
fn load_big_src_image() -> Image<'static> {
let img = ImageReader::open("./data/nasa-4928x3279.png")
.unwrap()
.decode()
.unwrap();
Image::from_vec_u8(
NonZeroU32::new(img.width()).unwrap(),
NonZeroU32::new(img.height()).unwrap(),
Self::img_into_bytes(img),
Self::pixel_type(),
)
.unwrap()
}
fn load_small_src_image() -> Image<'static> {
let img = ImageReader::open("./data/nasa-852x567.png")
.unwrap()
.decode()
.unwrap();
Image::from_vec_u8(
NonZeroU32::new(img.width()).unwrap(),
NonZeroU32::new(img.height()).unwrap(),
Self::img_into_bytes(img),
Self::pixel_type(),
)
.unwrap()
}
fn img_into_bytes(img: DynamicImage) -> Vec<u8>;
}
impl PixelExt for U8 {
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
img.to_luma8().into_raw()
}
}
impl PixelExt for U8x3 {
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
img.to_rgb8().into_raw()
}
}
impl PixelExt for U8x4 {
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
img.to_rgba8().into_raw()
}
}
impl PixelExt for I32 {
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
img.to_luma16()
.as_raw()
.iter()
.map(|&p| p as u32 * (i16::MAX as u32 + 1))
.flat_map(|val| val.to_le_bytes())
.collect()
}
}
impl PixelExt for F32 {
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
img.to_luma16()
.as_raw()
.iter()
.map(|&p| p as f32 * (i16::MAX as f32 + 1.0))
.flat_map(|val| val.to_le_bytes())
.collect()
}
}
fn downscale_test<P: PixelExt>(resize_alg: ResizeAlg, cpu_extensions: CpuExtensions) -> Vec<u8> {
let image = P::load_big_src_image();
assert_eq!(image.pixel_type(), P::pixel_type());
@@ -179,23 +65,13 @@ fn downscale_test<P: PixelExt>(resize_alg: ResizeAlg, cpu_extensions: CpuExtensi
_ => "unknown",
};
let ext_name = match cpu_extensions {
CpuExtensions::None => "native",
#[cfg(target_arch = "x86_64")]
CpuExtensions::Sse2 => "sse2",
#[cfg(target_arch = "x86_64")]
CpuExtensions::Sse4_1 => "sse41",
#[cfg(target_arch = "x86_64")]
CpuExtensions::Avx2 => "avx2",
};
let name = format!(
"downscale-{}-{}-{}",
P::pixel_type_str(),
alg_name,
ext_name
cpu_ext_into_str(cpu_extensions),
);
save_result(&result, &name);
utils::save_result(&result, &name);
result.buffer().to_owned()
}
@@ -232,18 +108,13 @@ fn upscale_test<P: PixelExt>(resize_alg: ResizeAlg, cpu_extensions: CpuExtension
_ => "unknown",
};
let ext_name = match cpu_extensions {
CpuExtensions::None => "native",
#[cfg(target_arch = "x86_64")]
CpuExtensions::Sse2 => "sse2",
#[cfg(target_arch = "x86_64")]
CpuExtensions::Sse4_1 => "sse41",
#[cfg(target_arch = "x86_64")]
CpuExtensions::Avx2 => "avx2",
};
let name = format!("upscale-{}-{}-{}", P::pixel_type_str(), alg_name, ext_name);
save_result(&result, &name);
let name = format!(
"upscale-{}-{}-{}",
P::pixel_type_str(),
alg_name,
cpu_ext_into_str(cpu_extensions),
);
utils::save_result(&result, &name);
result.buffer().to_owned()
}
@@ -251,7 +122,7 @@ fn upscale_test<P: PixelExt>(resize_alg: ResizeAlg, cpu_extensions: CpuExtension
fn downscale_u8() {
type P = U8;
let buffer = downscale_test::<P>(ResizeAlg::Nearest, CpuExtensions::None);
assert_eq!(image_checksum::<1>(&buffer), [2920317]);
assert_eq!(utils::image_checksum::<1>(&buffer), [2920317]);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
@@ -261,7 +132,7 @@ fn downscale_u8() {
for cpu_extensions in cpu_extensions_vec {
let buffer =
downscale_test::<P>(ResizeAlg::Convolution(FilterType::Lanczos3), cpu_extensions);
assert_eq!(image_checksum::<1>(&buffer), [2923520]);
assert_eq!(utils::image_checksum::<1>(&buffer), [2923520]);
}
}
@@ -269,7 +140,7 @@ fn downscale_u8() {
fn upscale_u8() {
type P = U8;
let buffer = upscale_test::<P>(ResizeAlg::Nearest, CpuExtensions::None);
assert_eq!(image_checksum::<1>(&buffer), [1148750539]);
assert_eq!(utils::image_checksum::<1>(&buffer), [1148750539]);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
@@ -279,7 +150,7 @@ fn upscale_u8() {
for cpu_extensions in cpu_extensions_vec {
let buffer =
upscale_test::<P>(ResizeAlg::Convolution(FilterType::Lanczos3), cpu_extensions);
assert_eq!(image_checksum::<1>(&buffer), [1148808058]);
assert_eq!(utils::image_checksum::<1>(&buffer), [1148808058]);
}
}
@@ -287,7 +158,10 @@ fn upscale_u8() {
fn downscale_u8x3() {
type P = U8x3;
let buffer = downscale_test::<P>(ResizeAlg::Nearest, CpuExtensions::None);
assert_eq!(image_checksum::<3>(&buffer), [2937940, 2945380, 2882679]);
assert_eq!(
utils::image_checksum::<3>(&buffer),
[2937940, 2945380, 2882679]
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
@@ -298,7 +172,10 @@ fn downscale_u8x3() {
for cpu_extensions in cpu_extensions_vec {
let buffer =
downscale_test::<P>(ResizeAlg::Convolution(FilterType::Lanczos3), cpu_extensions);
assert_eq!(image_checksum::<3>(&buffer), [2942479, 2947850, 2885072]);
assert_eq!(
utils::image_checksum::<3>(&buffer),
[2942479, 2947850, 2885072]
);
}
}
@@ -307,7 +184,7 @@ fn upscale_u8x3() {
type P = U8x3;
let buffer = upscale_test::<P>(ResizeAlg::Nearest, CpuExtensions::None);
assert_eq!(
image_checksum::<3>(&buffer),
utils::image_checksum::<3>(&buffer),
[1156008260, 1158417906, 1135087540]
);
@@ -321,7 +198,7 @@ fn upscale_u8x3() {
let buffer =
upscale_test::<P>(ResizeAlg::Convolution(FilterType::Lanczos3), cpu_extensions);
assert_eq!(
image_checksum::<3>(&buffer),
utils::image_checksum::<3>(&buffer),
[1156107005, 1158443335, 1135101759]
);
}
@@ -332,7 +209,7 @@ fn downscale_u8x4() {
type P = U8x4;
let buffer = downscale_test::<P>(ResizeAlg::Nearest, CpuExtensions::None);
assert_eq!(
image_checksum::<4>(&buffer),
utils::image_checksum::<4>(&buffer),
[2937940, 2945380, 2882679, 11054250]
);
@@ -346,7 +223,7 @@ fn downscale_u8x4() {
let buffer =
downscale_test::<P>(ResizeAlg::Convolution(FilterType::Lanczos3), cpu_extensions);
assert_eq!(
image_checksum::<4>(&buffer),
utils::image_checksum::<4>(&buffer),
[2942479, 2947850, 2885072, 11054250]
);
@@ -362,7 +239,7 @@ fn upscale_u8x4() {
type P = U8x4;
let buffer = upscale_test::<P>(ResizeAlg::Nearest, CpuExtensions::None);
assert_eq!(
image_checksum::<4>(&buffer),
utils::image_checksum::<4>(&buffer),
[1156008260, 1158417906, 1135087540, 4269569040]
);
@@ -376,7 +253,7 @@ fn upscale_u8x4() {
let buffer =
upscale_test::<P>(ResizeAlg::Convolution(FilterType::Lanczos3), cpu_extensions);
assert_eq!(
image_checksum::<4>(&buffer),
utils::image_checksum::<4>(&buffer),
[1156107005, 1158443335, 1135101759, 4269569040]
);
}
+145
View File
@@ -0,0 +1,145 @@
use std::fs::File;
use std::num::NonZeroU32;
use image::codecs::png::PngEncoder;
use image::io::Reader as ImageReader;
use image::{ColorType, DynamicImage, GenericImageView};
use fast_image_resize::pixels::*;
use fast_image_resize::{CpuExtensions, Image, PixelType};
pub fn image_checksum<const N: usize>(buffer: &[u8]) -> [u32; N] {
let mut res = [0u32; N];
for pixel in buffer.chunks_exact(N) {
res.iter_mut().zip(pixel).for_each(|(d, &s)| *d += s as u32);
}
res
}
pub trait PixelExt: Pixel {
fn pixel_type_str() -> &'static str {
match Self::pixel_type() {
PixelType::U8 => "u8",
PixelType::U8x3 => "u8x3",
PixelType::U8x4 => "u8x4",
PixelType::I32 => "i32",
PixelType::F32 => "f32",
}
}
fn load_big_src_image() -> Image<'static> {
let img = ImageReader::open("./data/nasa-4928x3279.png")
.unwrap()
.decode()
.unwrap();
Image::from_vec_u8(
NonZeroU32::new(img.width()).unwrap(),
NonZeroU32::new(img.height()).unwrap(),
Self::img_into_bytes(img),
Self::pixel_type(),
)
.unwrap()
}
fn load_small_src_image() -> Image<'static> {
let img = ImageReader::open("./data/nasa-852x567.png")
.unwrap()
.decode()
.unwrap();
Image::from_vec_u8(
NonZeroU32::new(img.width()).unwrap(),
NonZeroU32::new(img.height()).unwrap(),
Self::img_into_bytes(img),
Self::pixel_type(),
)
.unwrap()
}
fn load_small_rgba_image() -> Image<'static> {
let img = ImageReader::open("./data/nasa-852x567-rgba.png")
.unwrap()
.decode()
.unwrap();
Image::from_vec_u8(
NonZeroU32::new(img.width()).unwrap(),
NonZeroU32::new(img.height()).unwrap(),
Self::img_into_bytes(img),
Self::pixel_type(),
)
.unwrap()
}
fn img_into_bytes(img: DynamicImage) -> Vec<u8>;
}
impl PixelExt for U8 {
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
img.to_luma8().into_raw()
}
}
impl PixelExt for U8x3 {
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
img.to_rgb8().into_raw()
}
}
impl PixelExt for U8x4 {
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
img.to_rgba8().into_raw()
}
}
impl PixelExt for I32 {
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
img.to_luma16()
.as_raw()
.iter()
.map(|&p| p as u32 * (i16::MAX as u32 + 1))
.flat_map(|val| val.to_le_bytes())
.collect()
}
}
impl PixelExt for F32 {
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
img.to_luma16()
.as_raw()
.iter()
.map(|&p| p as f32 * (i16::MAX as f32 + 1.0))
.flat_map(|val| val.to_le_bytes())
.collect()
}
}
pub fn save_result(image: &Image, name: &str) {
if std::env::var("DONT_SAVE_RESULT").unwrap_or_else(|_| "".to_owned()) == "1" {
return;
}
std::fs::create_dir_all("./data/result").unwrap();
let mut file = File::create(format!("./data/result/{}.png", name)).unwrap();
let color_type = match image.pixel_type() {
PixelType::U8x3 => ColorType::Rgb8,
PixelType::U8x4 => ColorType::Rgba8,
PixelType::U8 => ColorType::L8,
_ => panic!("Unsupported type of pixels"),
};
PngEncoder::new(&mut file)
.encode(
image.buffer(),
image.width().get(),
image.height().get(),
color_type,
)
.unwrap();
}
pub fn cpu_ext_into_str(cpu_extensions: CpuExtensions) -> &'static str {
match cpu_extensions {
CpuExtensions::None => "native",
#[cfg(target_arch = "x86_64")]
CpuExtensions::Sse4_1 => "sse41",
#[cfg(target_arch = "x86_64")]
CpuExtensions::Avx2 => "avx2",
}
}