mirror of
https://github.com/Cykooz/fast_image_resize.git
synced 2026-10-07 17:01:09 +00:00
- Added optimisation of multiplying and dividing image by alpha channel with helps of `SSE4.1` instructions.
- Improved performance of dividing image by alpha channel without forced SIMD instructions. - Deleted variant `SSE2` from enum ``CpuExtensions``.
This commit is contained in:
@@ -1,3 +1,12 @@
|
||||
## [Unreleased] - ReleaseDate
|
||||
|
||||
- Added optimisation of multiplying and dividing image by alpha channel with helps
|
||||
of ``SSE4.1`` instructions.
|
||||
- Improved performance of dividing image by alpha channel without forced
|
||||
SIMD instructions.
|
||||
- Breaking changes:
|
||||
- Deleted variant `SSE2` from enum ``CpuExtensions``.
|
||||
|
||||
## [0.5.3] - 2021-12-14
|
||||
|
||||
- Added optimisation of convolution U8x3 images with helps of ``AVX2`` instructions.
|
||||
|
||||
Generated
+54
-85
@@ -33,9 +33,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "anyhow"
|
||||
version = "1.0.51"
|
||||
version = "1.0.52"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "8b26702f315f53b6071259e15dd9d64528213b44d61de1ec926eca7715d62203"
|
||||
checksum = "84450d0b4a8bd1ba4144ce8ce718fbc5d071358b1e5384bace6536b3d1f2d5b3"
|
||||
|
||||
[[package]]
|
||||
name = "argh"
|
||||
@@ -98,9 +98,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "bytemuck"
|
||||
version = "1.7.2"
|
||||
version = "1.7.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "72957246c41db82b8ef88a5486143830adeb8227ef9837740bdec67724cf2c5b"
|
||||
checksum = "439989e6b8c38d1b6570a384ef1e49c8848128f5a97f3914baef02920842712f"
|
||||
|
||||
[[package]]
|
||||
name = "byteorder"
|
||||
@@ -178,9 +178,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "crossbeam-channel"
|
||||
version = "0.5.1"
|
||||
version = "0.5.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "06ed27e177f16d65f0f0c22a213e17c696ace5dd64b14258b52f9417ccb52db4"
|
||||
checksum = "e54ea8bc3fb1ee042f5aace6e3c6e025d3874866da222930f70ce62aceba0bfa"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"crossbeam-utils",
|
||||
@@ -199,9 +199,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "crossbeam-epoch"
|
||||
version = "0.9.5"
|
||||
version = "0.9.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "4ec02e091aa634e2c3ada4a392989e7c3116673ef0ac5b72232439094d73b7fd"
|
||||
checksum = "97242a70df9b89a65d0b6df3c4bf5b9ce03c5b7309019777fbde37e7537f8762"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"crossbeam-utils",
|
||||
@@ -212,9 +212,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "crossbeam-queue"
|
||||
version = "0.3.2"
|
||||
version = "0.3.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9b10ddc024425c88c2ad148c1b0fd53f4c6d38db9697c9f1588381212fa657c9"
|
||||
checksum = "b979d76c9fcb84dffc80a73f7290da0f83e4c95773494674cb44b76d13a7a110"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"crossbeam-utils",
|
||||
@@ -222,9 +222,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "crossbeam-utils"
|
||||
version = "0.8.5"
|
||||
version = "0.8.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d82cfc11ce7f2c3faef78d8a684447b40d503d9681acebed6cb728d45940c4db"
|
||||
checksum = "cfcae03edb34f947e64acdb1c33ec169824e20657e9ecb61cef6c8c74dcb8120"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"lazy_static",
|
||||
@@ -263,7 +263,7 @@ checksum = "22813a6dc45b335f9bade10bf7271dc477e81113e89eb251a0bc2a8a81c536e1"
|
||||
dependencies = [
|
||||
"bstr",
|
||||
"csv-core",
|
||||
"itoa",
|
||||
"itoa 0.4.8",
|
||||
"ryu",
|
||||
"serde",
|
||||
]
|
||||
@@ -349,9 +349,9 @@ checksum = "7360491ce676a36bf9bb3c56c1aa791658183a54d2744120f27285738d90465a"
|
||||
|
||||
[[package]]
|
||||
name = "fallible_collections"
|
||||
version = "0.4.3"
|
||||
version = "0.4.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "eaefd4190151d458f16f0793d3452d7f13aeb3701566a4cefc4c37598876cc00"
|
||||
checksum = "52db5973b6a19247baf19b30f41c23a1bfffc2e9ce0a5db2f60e3cd5dc8895f7"
|
||||
dependencies = [
|
||||
"hashbrown 0.11.2",
|
||||
]
|
||||
@@ -368,6 +368,15 @@ dependencies = [
|
||||
"thiserror",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "fastrand"
|
||||
version = "1.6.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "779d043b6a0b90cc4c0ed7ee380a6504394cee7efd7db050e3774eee387324b2"
|
||||
dependencies = [
|
||||
"instant",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "form_urlencoded"
|
||||
version = "1.0.1"
|
||||
@@ -525,6 +534,12 @@ version = "0.4.8"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b71991ff56294aa922b450139ee08b3bfc70982c6b2c7562771375cf73542dd4"
|
||||
|
||||
[[package]]
|
||||
name = "itoa"
|
||||
version = "1.0.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "1aab8fc367588b89dcee83ab0fd66b72b50b72fa1904d7095045ace2b0c81c35"
|
||||
|
||||
[[package]]
|
||||
name = "jobserver"
|
||||
version = "0.1.24"
|
||||
@@ -731,9 +746,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "num_cpus"
|
||||
version = "1.13.0"
|
||||
version = "1.13.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "05499f3756671c15885fee9034446956fff3f243d6077b91e5767df161f766b3"
|
||||
checksum = "19e64526ebdee182341572e50e9ad03965aa510cd94427a4549448f285e957a1"
|
||||
dependencies = [
|
||||
"hermit-abi",
|
||||
"libc",
|
||||
@@ -741,9 +756,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "once_cell"
|
||||
version = "1.8.0"
|
||||
version = "1.9.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "692fcb63b64b1758029e0a96ee63e049ce8c5948587f2f7208df04625e5f6b56"
|
||||
checksum = "da32515d9f6e6e489d7bc9d84c71b060db7247dc035bbe44eac88cf87486d8d5"
|
||||
|
||||
[[package]]
|
||||
name = "open"
|
||||
@@ -810,70 +825,24 @@ dependencies = [
|
||||
"miniz_oxide 0.3.7",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "ppv-lite86"
|
||||
version = "0.2.15"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ed0cfbc8191465bed66e1718596ee0b0b35d5ee1f41c5df2189d0fe8bde535ba"
|
||||
|
||||
[[package]]
|
||||
name = "proc-macro2"
|
||||
version = "1.0.33"
|
||||
version = "1.0.36"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "fb37d2df5df740e582f28f8560cf425f52bb267d872fe58358eadb554909f07a"
|
||||
checksum = "c7342d5883fbccae1cc37a2353b09c87c9b0f3afd73f5fb9bba687a1f733b029"
|
||||
dependencies = [
|
||||
"unicode-xid",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "quote"
|
||||
version = "1.0.10"
|
||||
version = "1.0.14"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "38bc8cc6a5f2e3655e0899c1b848643b2562f853f114bfec7be120678e3ace05"
|
||||
checksum = "47aa80447ce4daf1717500037052af176af5d38cc3e571d9ec1c7353fc10c87d"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rand"
|
||||
version = "0.8.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "2e7573632e6454cf6b99d7aac4ccca54be06da05aca2ef7423d22d27d4d4bcd8"
|
||||
dependencies = [
|
||||
"libc",
|
||||
"rand_chacha",
|
||||
"rand_core",
|
||||
"rand_hc",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rand_chacha"
|
||||
version = "0.3.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e6c10a63a0fa32252be49d21e7709d4d4baf8d231c2dbce1eaa8141b9b127d88"
|
||||
dependencies = [
|
||||
"ppv-lite86",
|
||||
"rand_core",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rand_core"
|
||||
version = "0.6.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d34f1408f55294453790c48b2f1ebbb1c5b4b7563eb1f418bcfcfdbb06ebb4e7"
|
||||
dependencies = [
|
||||
"getrandom",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rand_hc"
|
||||
version = "0.3.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d51e9f596de227fda2ea6c84607f5558e196eeaf43c986b724ba4fb8fdf497e7"
|
||||
dependencies = [
|
||||
"rand_core",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rayon"
|
||||
version = "1.5.1"
|
||||
@@ -945,9 +914,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rgb"
|
||||
version = "0.8.30"
|
||||
version = "0.8.31"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "08a9852b34c4628f8ad76797a933577059163651ec5a7dace462adc365bee66c"
|
||||
checksum = "9a374af9a0e5fdcdd98c1c7b64f05004f9ea2555b6c75f211daa81268a3c50f1"
|
||||
dependencies = [
|
||||
"bytemuck",
|
||||
]
|
||||
@@ -987,18 +956,18 @@ checksum = "d29ab0c6d3fc0ee92fe66e2d99f700eab17a8d57d1c1d3b748380fb20baa78cd"
|
||||
|
||||
[[package]]
|
||||
name = "serde"
|
||||
version = "1.0.131"
|
||||
version = "1.0.133"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b4ad69dfbd3e45369132cc64e6748c2d65cdfb001a2b1c232d128b4ad60561c1"
|
||||
checksum = "97565067517b60e2d1ea8b268e59ce036de907ac523ad83a0475da04e818989a"
|
||||
dependencies = [
|
||||
"serde_derive",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "serde_derive"
|
||||
version = "1.0.131"
|
||||
version = "1.0.133"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b710a83c4e0dff6a3d511946b95274ad9ca9e5d3ae497b63fda866ac955358d2"
|
||||
checksum = "ed201699328568d8d08208fdd080e3ff594e6c422e438b6705905da01005d537"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
@@ -1007,11 +976,11 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "serde_json"
|
||||
version = "1.0.72"
|
||||
version = "1.0.74"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d0ffa0837f2dfa6fb90868c2b5468cad482e175f7dad97e7421951e663f2b527"
|
||||
checksum = "ee2bb9cd061c5865d345bb02ca49fcef1391741b672b54a0bf7b679badec3142"
|
||||
dependencies = [
|
||||
"itoa",
|
||||
"itoa 1.0.1",
|
||||
"ryu",
|
||||
"serde",
|
||||
]
|
||||
@@ -1050,9 +1019,9 @@ checksum = "3bdb25a4593d6656239319426f4025f7a658157e25e89f0e0319d7516d46042d"
|
||||
|
||||
[[package]]
|
||||
name = "syn"
|
||||
version = "1.0.82"
|
||||
version = "1.0.85"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "8daf5dd0bb60cbd4137b1b587d2fc0ae729bc07cf01cd70b36a1ed5ade3b9d59"
|
||||
checksum = "a684ac3dcd8913827e18cd09a68384ee66c1de24157e3c556c9ab16d85695fb7"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
@@ -1061,13 +1030,13 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "tempfile"
|
||||
version = "3.2.0"
|
||||
version = "3.3.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "dac1c663cfc93810f88aed9b8941d48cabf856a1b111c29a40439018d870eb22"
|
||||
checksum = "5cdb1ef4eaeeaddc8fbd371e5017057064af0911902ef36b39801f67cc6d79e4"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"fastrand",
|
||||
"libc",
|
||||
"rand",
|
||||
"redox_syscall",
|
||||
"remove_dir_all",
|
||||
"winapi",
|
||||
@@ -1196,9 +1165,9 @@ checksum = "accd4ea62f7bb7a82fe23066fb0957d48ef677f6eeb8215f372f52e48bb32426"
|
||||
|
||||
[[package]]
|
||||
name = "version_check"
|
||||
version = "0.9.3"
|
||||
version = "0.9.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "5fecdca9a5291cc2b8dcf7dc02453fee791a280f3743cb0905f8822ae463b3fe"
|
||||
checksum = "49874b5167b65d7193b8aba1567f5c7d93d001cafc34600cee003eda787e483f"
|
||||
|
||||
[[package]]
|
||||
name = "wasi"
|
||||
|
||||
+1
-1
@@ -22,7 +22,7 @@ thiserror = "1.0.30"
|
||||
glassbench = "0.3.1"
|
||||
image = "0.23.14"
|
||||
resize = "0.7.2"
|
||||
rgb = "0.8.30"
|
||||
rgb = "0.8.31"
|
||||
|
||||
|
||||
[[bench]]
|
||||
|
||||
@@ -2,6 +2,12 @@
|
||||
|
||||
Rust library for fast image resizing with using of SIMD instructions.
|
||||
|
||||
_Note: This library does not support converting image color spaces.
|
||||
If it is important for you to resize images with a non-linear color space
|
||||
(e.g. sRGB) correctly, then you need to convert it to a linear color space
|
||||
before resizing. [Read more](https://legacy.imagemagick.org/Usage/resize/#resize_colorspace)
|
||||
about resizing with respect to color space._
|
||||
|
||||
[CHANGELOG](https://github.com/Cykooz/fast_image_resize/blob/main/CHANGELOG.md)
|
||||
|
||||
Supported pixel formats and available optimisations:
|
||||
@@ -28,7 +34,7 @@ Environment:
|
||||
- RAM: DDR4 3000 MHz
|
||||
- Ubuntu 20.04 (linux 5.11)
|
||||
- Rust 1.57.0
|
||||
- fast_image_resize = "0.5.3"
|
||||
- fast_image_resize = "0.6.0"
|
||||
- glassbench = "0.3.1"
|
||||
- `rustflags = ["-C", "llvm-args=-x86-branches-within-32B-boundaries"]`
|
||||
|
||||
@@ -53,11 +59,11 @@ Pipeline:
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|------------|:-------:|:--------:|:----------:|:--------:|
|
||||
| image | 91.896 | 176.959 | 256.786 | 341.548 |
|
||||
| resize | 15.453 | 71.340 | 131.232 | 191.469 |
|
||||
| fir rust | 0.481 | 53.733 | 90.519 | 121.121 |
|
||||
| fir sse4.1 | - | 43.113 | 53.484 | 75.127 |
|
||||
| fir avx2 | - | 10.765 | 14.131 | 19.827 |
|
||||
| image | 92.300 | 180.063 | 257.043 | 342.268 |
|
||||
| resize | 15.429 | 71.589 | 131.447 | 191.048 |
|
||||
| fir rust | 0.481 | 55.347 | 89.512 | 121.270 |
|
||||
| fir sse4.1 | - | 44.841 | 54.630 | 75.284 |
|
||||
| fir avx2 | - | 10.814 | 14.408 | 20.897 |
|
||||
|
||||
### Resize RGBA image (U8x4) 4928x3279 => 852x567
|
||||
|
||||
@@ -65,16 +71,16 @@ Pipeline:
|
||||
|
||||
`src_image => multiply by alpha => resize => divide by alpha => dst_image`
|
||||
|
||||
- Source image [nasa-4928x3279.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279.png)
|
||||
- Source image [nasa-4928x3279-rgba.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279-rgba.png)
|
||||
- Numbers in table is mean duration of image resizing in milliseconds.
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|------------|:-------:|:--------:|:----------:|:--------:|
|
||||
| image | 98.113 | 177.039 | 254.666 | 337.147 |
|
||||
| resize | 17.875 | 79.014 | 148.691 | 218.400 |
|
||||
| fir rust | 13.188 | 63.942 | 89.681 | 119.664 |
|
||||
| fir sse4.1 | 11.868 | 22.957 | 29.164 | 36.799 |
|
||||
| fir avx2 | 6.949 | 14.854 | 18.399 | 23.772 |
|
||||
| image | 98.598 | 177.427 | 253.149 | 336.999 |
|
||||
| resize | 17.988 | 79.891 | 149.178 | 218.593 |
|
||||
| fir rust | 11.952 | 63.214 | 89.397 | 118.714 |
|
||||
| fir sse4.1 | 9.169 | 20.664 | 26.442 | 34.326 |
|
||||
| fir avx2 | 7.353 | 15.514 | 18.936 | 24.624 |
|
||||
|
||||
### Resize grayscale image (U8) 4928x3279 => 852x567
|
||||
|
||||
@@ -88,10 +94,10 @@ Pipeline:
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|----------|:-------:|:--------:|:----------:|:--------:|
|
||||
| image | 76.981 | 126.595 | 166.765 | 209.593 |
|
||||
| resize | 9.632 | 24.332 | 47.533 | 80.667 |
|
||||
| fir rust | 0.197 | 21.773 | 24.476 | 34.909 |
|
||||
| fir avx2 | - | 9.467 | 7.691 | 11.776 |
|
||||
| image | 77.209 | 128.704 | 166.905 | 210.503 |
|
||||
| resize | 9.694 | 24.477 | 47.722 | 80.824 |
|
||||
| fir rust | 0.198 | 21.457 | 24.256 | 36.893 |
|
||||
| fir avx2 | - | 9.758 | 7.787 | 12.099 |
|
||||
|
||||
## Examples
|
||||
|
||||
@@ -162,7 +168,7 @@ fn resize_image_example() {
|
||||
|
||||
### Change CPU extensions used by resizer
|
||||
|
||||
```ignore
|
||||
```rust, ignore
|
||||
use fast_image_resize as fr;
|
||||
|
||||
fn main() {
|
||||
|
||||
@@ -40,7 +40,7 @@ fn multiplies_alpha_avx2(bench: &mut Bench) {
|
||||
}
|
||||
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
fn multiplies_alpha_sse2(bench: &mut Bench) {
|
||||
fn multiplies_alpha_sse4(bench: &mut Bench) {
|
||||
let width = NonZeroU32::new(4096).unwrap();
|
||||
let height = NonZeroU32::new(2048).unwrap();
|
||||
let src_data = get_src_image(width, height, p(255, 128, 0, 128));
|
||||
@@ -49,10 +49,10 @@ fn multiplies_alpha_sse2(bench: &mut Bench) {
|
||||
let mut dst_view = dst_data.view_mut();
|
||||
let mut alpha_mul_div: MulDiv = Default::default();
|
||||
unsafe {
|
||||
alpha_mul_div.set_cpu_extensions(CpuExtensions::Sse2);
|
||||
alpha_mul_div.set_cpu_extensions(CpuExtensions::Sse4_1);
|
||||
}
|
||||
|
||||
bench.task("Multiplies alpha SSE2", |task| {
|
||||
bench.task("Multiplies alpha SSE4.1", |task| {
|
||||
task.iter(|| {
|
||||
alpha_mul_div
|
||||
.multiply_alpha(&src_view, &mut dst_view)
|
||||
@@ -105,7 +105,7 @@ fn divides_alpha_avx2(bench: &mut Bench) {
|
||||
}
|
||||
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
fn divides_alpha_sse2(bench: &mut Bench) {
|
||||
fn divides_alpha_sse4(bench: &mut Bench) {
|
||||
let width = NonZeroU32::new(4096).unwrap();
|
||||
let height = NonZeroU32::new(2048).unwrap();
|
||||
let src_data = get_src_image(width, height, p(128, 64, 0, 128));
|
||||
@@ -114,10 +114,10 @@ fn divides_alpha_sse2(bench: &mut Bench) {
|
||||
let mut dst_view = dst_data.view_mut();
|
||||
let mut alpha_mul_div: MulDiv = Default::default();
|
||||
unsafe {
|
||||
alpha_mul_div.set_cpu_extensions(CpuExtensions::Sse2);
|
||||
alpha_mul_div.set_cpu_extensions(CpuExtensions::Sse4_1);
|
||||
}
|
||||
|
||||
bench.task("Divides alpha SSE2", |task| {
|
||||
bench.task("Divides alpha SSE4.1", |task| {
|
||||
task.iter(|| {
|
||||
alpha_mul_div
|
||||
.divide_alpha(&src_view, &mut dst_view)
|
||||
@@ -149,6 +149,7 @@ fn divides_alpha_native(bench: &mut Bench) {
|
||||
|
||||
pub fn main() {
|
||||
use glassbench::*;
|
||||
|
||||
let name = env!("CARGO_CRATE_NAME");
|
||||
let cmd = Command::read();
|
||||
if cmd.include_bench(name) {
|
||||
@@ -156,13 +157,13 @@ pub fn main() {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
{
|
||||
multiplies_alpha_avx2(&mut bench);
|
||||
multiplies_alpha_sse2(&mut bench);
|
||||
multiplies_alpha_sse4(&mut bench);
|
||||
}
|
||||
multiplies_alpha_native(&mut bench);
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
{
|
||||
divides_alpha_avx2(&mut bench);
|
||||
divides_alpha_sse2(&mut bench);
|
||||
divides_alpha_sse4(&mut bench);
|
||||
}
|
||||
divides_alpha_native(&mut bench);
|
||||
if let Err(e) = after_bench(&mut bench, &cmd) {
|
||||
|
||||
Binary file not shown.
|
After Width: | Height: | Size: 675 KiB |
+36
-41
@@ -1,79 +1,74 @@
|
||||
use std::arch::x86_64::*;
|
||||
|
||||
use crate::alpha::native;
|
||||
use crate::alpha::sse4;
|
||||
use crate::image_view::{TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::U8x4;
|
||||
use crate::simd_utils;
|
||||
|
||||
pub(crate) fn divide_alpha_avx2(
|
||||
#[target_feature(enable = "avx2")]
|
||||
pub(crate) unsafe fn divide_alpha_avx2(
|
||||
src_image: TypedImageView<U8x4>,
|
||||
mut dst_image: TypedImageViewMut<U8x4>,
|
||||
) {
|
||||
let width = src_image.width().get();
|
||||
let src_rows = src_image.iter_rows(0);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
divide_alpha_row_avx2(src_row, dst_row, width as usize);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn divide_alpha_inplace_avx2(mut image: TypedImageViewMut<U8x4>) {
|
||||
let width = image.width().get() as usize;
|
||||
for dst_row in image.iter_rows_mut() {
|
||||
unsafe {
|
||||
let src_row = std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len());
|
||||
divide_alpha_row_avx2(src_row, dst_row, width);
|
||||
}
|
||||
divide_alpha_row_avx2(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "avx2")]
|
||||
unsafe fn divide_alpha_row_avx2(src_row: &[U8x4], dst_row: &mut [U8x4], width: usize) {
|
||||
let mut x: usize = 0;
|
||||
pub(crate) unsafe fn divide_alpha_inplace_avx2(mut image: TypedImageViewMut<U8x4>) {
|
||||
for dst_row in image.iter_rows_mut() {
|
||||
let src_row = std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len());
|
||||
divide_alpha_row_avx2(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "avx2")]
|
||||
unsafe fn divide_alpha_row_avx2(src_row: &[U8x4], dst_row: &mut [U8x4]) {
|
||||
let zero = _mm256_setzero_si256();
|
||||
let alpha_mask = _mm256_set1_epi32(0xff000000u32 as i32);
|
||||
#[rustfmt::skip]
|
||||
let alpha_mask = _mm256_set1_epi32(0xff000000u32 as i32);
|
||||
#[rustfmt::skip]
|
||||
let shuffle1 = _mm256_set_epi8(
|
||||
let shuffle1 = _mm256_set_epi8(
|
||||
5, 4, 5, 4, 5, 4, 5, 4, 1, 0, 1, 0, 1, 0, 1, 0,
|
||||
5, 4, 5, 4, 5, 4, 5, 4, 1, 0, 1, 0, 1, 0, 1, 0,
|
||||
);
|
||||
#[rustfmt::skip]
|
||||
let shuffle2 = _mm256_set_epi8(
|
||||
let shuffle2 = _mm256_set_epi8(
|
||||
13, 12, 13, 12, 13, 12, 13, 12, 9, 8, 9, 8, 9, 8, 9, 8,
|
||||
13, 12, 13, 12, 13, 12, 13, 12, 9, 8, 9, 8, 9, 8, 9, 8,
|
||||
);
|
||||
let alpha_scale = _mm256_set1_ps(255.0 * 256.0);
|
||||
|
||||
while x < width.saturating_sub(7) {
|
||||
let mut source = simd_utils::loadu_si256(src_row, x);
|
||||
let src_chunks = src_row.chunks_exact(8);
|
||||
let src_remainder = src_chunks.remainder();
|
||||
let mut dst_chunks = dst_row.chunks_exact_mut(8);
|
||||
|
||||
let alpha_f32 = _mm256_cvtepi32_ps(_mm256_srli_epi32::<24>(source));
|
||||
let scaled_alpha_f32 = _mm256_mul_ps(alpha_scale, _mm256_rcp_ps(alpha_f32));
|
||||
for (src, dst) in src_chunks.zip(&mut dst_chunks) {
|
||||
let src_pixels = _mm256_loadu_si256(src.as_ptr() as *const __m256i);
|
||||
|
||||
let alpha_f32 = _mm256_cvtepi32_ps(_mm256_srli_epi32::<24>(src_pixels));
|
||||
let scaled_alpha_f32 = _mm256_div_ps(alpha_scale, alpha_f32);
|
||||
let scaled_alpha_i32 = _mm256_cvtps_epi32(scaled_alpha_f32);
|
||||
let mma0 = _mm256_shuffle_epi8(scaled_alpha_i32, shuffle1);
|
||||
let mma1 = _mm256_shuffle_epi8(scaled_alpha_i32, shuffle2);
|
||||
|
||||
let mut pix0 = _mm256_unpacklo_epi8(zero, source);
|
||||
let mut pix1 = _mm256_unpackhi_epi8(zero, source);
|
||||
let pix0 = _mm256_unpacklo_epi8(zero, src_pixels);
|
||||
let pix1 = _mm256_unpackhi_epi8(zero, src_pixels);
|
||||
|
||||
pix0 = _mm256_mulhi_epu16(pix0, mma0);
|
||||
pix1 = _mm256_mulhi_epu16(pix1, mma1);
|
||||
let pix0 = _mm256_mulhi_epu16(pix0, mma0);
|
||||
let pix1 = _mm256_mulhi_epu16(pix1, mma1);
|
||||
|
||||
let alpha = _mm256_and_si256(source, alpha_mask);
|
||||
source = _mm256_packus_epi16(pix0, pix1);
|
||||
source = _mm256_blendv_epi8(source, alpha, alpha_mask);
|
||||
let alpha = _mm256_and_si256(src_pixels, alpha_mask);
|
||||
let rgb = _mm256_packus_epi16(pix0, pix1);
|
||||
let dst_pixels = _mm256_blendv_epi8(rgb, alpha, alpha_mask);
|
||||
|
||||
let dst_ptr = dst_row.get_unchecked_mut(x..).as_mut_ptr() as *mut __m256i;
|
||||
_mm256_storeu_si256(dst_ptr, source);
|
||||
|
||||
x += 8;
|
||||
_mm256_storeu_si256(dst.as_mut_ptr() as *mut __m256i, dst_pixels);
|
||||
}
|
||||
|
||||
let src_tail = &src_row[x..];
|
||||
let dst_tail = &mut dst_row[x..];
|
||||
native::divide_alpha_row_native(src_tail, dst_tail);
|
||||
if !src_remainder.is_empty() {
|
||||
let dst_reminder = dst_chunks.into_remainder();
|
||||
sse4::div::divide_alpha_row_sse4(src_remainder, dst_reminder);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,5 +1,2 @@
|
||||
pub(crate) use div::{divide_alpha_avx2, divide_alpha_inplace_avx2};
|
||||
pub(crate) use mul::{multiply_alpha_avx2, multiply_alpha_inplace_avx2};
|
||||
|
||||
mod div;
|
||||
mod mul;
|
||||
pub(crate) mod div;
|
||||
pub(crate) mod mul;
|
||||
|
||||
+32
-32
@@ -5,7 +5,8 @@ use crate::image_view::{TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::U8x4;
|
||||
use crate::simd_utils;
|
||||
|
||||
pub(crate) fn multiply_alpha_avx2(
|
||||
#[target_feature(enable = "avx2")]
|
||||
pub(crate) unsafe fn multiply_alpha_avx2(
|
||||
src_image: TypedImageView<U8x4>,
|
||||
mut dst_image: TypedImageViewMut<U8x4>,
|
||||
) {
|
||||
@@ -14,62 +15,61 @@ pub(crate) fn multiply_alpha_avx2(
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
multiply_alpha_row_avx2(src_row, dst_row, width);
|
||||
}
|
||||
multiply_alpha_row_avx2(src_row, dst_row, width);
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn multiply_alpha_inplace_avx2(mut image: TypedImageViewMut<U8x4>) {
|
||||
#[target_feature(enable = "avx2")]
|
||||
pub(crate) unsafe fn multiply_alpha_inplace_avx2(mut image: TypedImageViewMut<U8x4>) {
|
||||
let width = image.width().get() as usize;
|
||||
for dst_row in image.iter_rows_mut() {
|
||||
unsafe {
|
||||
let src_row = std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len());
|
||||
multiply_alpha_row_avx2(src_row, dst_row, width);
|
||||
}
|
||||
let src_row = std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len());
|
||||
multiply_alpha_row_avx2(src_row, dst_row, width);
|
||||
}
|
||||
}
|
||||
|
||||
/// https://github.com/Wizermil/premultiply_alpha/blob/master/premultiply_alpha/premultiply_alpha.hpp#L232
|
||||
#[inline]
|
||||
#[target_feature(enable = "avx2")]
|
||||
unsafe fn multiply_alpha_row_avx2(src_row: &[U8x4], dst_row: &mut [U8x4], width: usize) {
|
||||
let mask_alpha_color_odd_255 = _mm256_set1_epi32(0xff000000u32 as i32);
|
||||
let div_255 = _mm256_set1_epi16(0x8081u16 as i16);
|
||||
let zero = _mm256_setzero_si256();
|
||||
let half = _mm256_set1_epi16(128);
|
||||
|
||||
const MAX_A: i32 = 0xff000000u32 as i32;
|
||||
let max_alpha = _mm256_set1_epi32(MAX_A);
|
||||
#[rustfmt::skip]
|
||||
let mask_shuffle_alpha = _mm256_set_epi8(
|
||||
15, -1, 15, -1, 11, -1, 11, -1, 7, -1, 7, -1, 3, -1, 3, -1,
|
||||
15, -1, 15, -1, 11, -1, 11, -1, 7, -1, 7, -1, 3, -1, 3, -1,
|
||||
);
|
||||
#[rustfmt::skip]
|
||||
let mask_shuffle_color_odd = _mm256_set_epi8(
|
||||
-1, -1, 13, -1, -1, -1, 9, -1, -1, -1, 5, -1, -1, -1, 1, -1,
|
||||
-1, -1, 13, -1, -1, -1, 9, -1, -1, -1, 5, -1, -1, -1, 1, -1,
|
||||
let factor_mask = _mm256_set_epi8(
|
||||
15, 15, 15, 15, 11, 11, 11, 11, 7, 7, 7, 7, 3, 3, 3, 3,
|
||||
15, 15, 15, 15, 11, 11, 11, 11, 7, 7, 7, 7, 3, 3, 3, 3,
|
||||
);
|
||||
|
||||
let mut x: usize = 0;
|
||||
while x < width.saturating_sub(7) {
|
||||
let mut color = simd_utils::loadu_si256(src_row, x);
|
||||
let src_pixels = simd_utils::loadu_si256(src_row, x);
|
||||
|
||||
let alpha = _mm256_shuffle_epi8(color, mask_shuffle_alpha);
|
||||
let mut color_even = _mm256_slli_epi16::<8>(color);
|
||||
let mut color_odd = _mm256_shuffle_epi8(color, mask_shuffle_color_odd);
|
||||
color_odd = _mm256_or_si256(color_odd, mask_alpha_color_odd_255);
|
||||
let factor_pixels = _mm256_shuffle_epi8(src_pixels, factor_mask);
|
||||
let factor_pixels = _mm256_or_si256(factor_pixels, max_alpha);
|
||||
|
||||
color_odd = _mm256_mulhi_epu16(color_odd, alpha);
|
||||
color_even = _mm256_mulhi_epu16(color_even, alpha);
|
||||
let pix1 = _mm256_unpacklo_epi8(src_pixels, zero);
|
||||
let factors = _mm256_unpacklo_epi8(factor_pixels, zero);
|
||||
let pix1 = _mm256_add_epi16(_mm256_mullo_epi16(pix1, factors), half);
|
||||
let pix1 = _mm256_add_epi16(pix1, _mm256_srli_epi16::<8>(pix1));
|
||||
let pix1 = _mm256_srli_epi16::<8>(pix1);
|
||||
|
||||
color_odd = _mm256_srli_epi16::<7>(_mm256_mulhi_epu16(color_odd, div_255));
|
||||
color_even = _mm256_srli_epi16::<7>(_mm256_mulhi_epu16(color_even, div_255));
|
||||
let pix2 = _mm256_unpackhi_epi8(src_pixels, zero);
|
||||
let factors = _mm256_unpackhi_epi8(factor_pixels, zero);
|
||||
let pix2 = _mm256_add_epi16(_mm256_mullo_epi16(pix2, factors), half);
|
||||
let pix2 = _mm256_add_epi16(pix2, _mm256_srli_epi16::<8>(pix2));
|
||||
let pix2 = _mm256_srli_epi16::<8>(pix2);
|
||||
|
||||
color = _mm256_or_si256(color_even, _mm256_slli_epi16::<8>(color_odd));
|
||||
let dst_pixels = _mm256_packus_epi16(pix1, pix2);
|
||||
|
||||
let dst_ptr = dst_row.get_unchecked_mut(x..).as_mut_ptr() as *mut __m256i;
|
||||
_mm256_storeu_si256(dst_ptr, color);
|
||||
_mm256_storeu_si256(dst_ptr, dst_pixels);
|
||||
|
||||
x += 8;
|
||||
}
|
||||
|
||||
let src_tail = &src_row[x..];
|
||||
let dst_tail = &mut dst_row[x..];
|
||||
native::multiply_alpha_row_native(src_tail, dst_tail);
|
||||
native::mul::multiply_alpha_row_native(src_tail, dst_tail);
|
||||
}
|
||||
|
||||
+25
-26
@@ -1,15 +1,16 @@
|
||||
pub use errors::*;
|
||||
|
||||
use crate::image_view::{TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::U8x4;
|
||||
use crate::CpuExtensions;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
pub use errors::*;
|
||||
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod avx2;
|
||||
mod errors;
|
||||
mod native;
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod sse2;
|
||||
mod sse4;
|
||||
|
||||
/// Methods of this structure used to multiply or divide RGB-channels
|
||||
/// by alpha-channel.
|
||||
@@ -60,13 +61,14 @@ impl MulDiv {
|
||||
let (src_image_u8x4, dst_image_u8x4) = assert_images(src_image, dst_image)?;
|
||||
match self.cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::multiply_alpha_avx2(src_image_u8x4, dst_image_u8x4),
|
||||
// WARNING: SSE2 implementation is drastically slower than native version
|
||||
// #[cfg(target_arch = "x86_64")]
|
||||
// CpuExtensions::Sse4_1 | CpuExtensions::Sse2 => {
|
||||
// sse2::multiply_alpha_sse2(src_image, dst_image)
|
||||
// }
|
||||
_ => native::multiply_alpha_native(src_image_u8x4, dst_image_u8x4),
|
||||
CpuExtensions::Avx2 => unsafe {
|
||||
avx2::mul::multiply_alpha_avx2(src_image_u8x4, dst_image_u8x4)
|
||||
},
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe {
|
||||
sse4::mul::multiply_alpha_sse4(src_image_u8x4, dst_image_u8x4)
|
||||
},
|
||||
_ => native::mul::multiply_alpha_native(src_image_u8x4, dst_image_u8x4),
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -76,13 +78,10 @@ impl MulDiv {
|
||||
let image_u8x4 = assert_image(image)?;
|
||||
match self.cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::multiply_alpha_inplace_avx2(image_u8x4),
|
||||
// WARNING: SSE2 implementation is drastically slower than native version
|
||||
// #[cfg(target_arch = "x86_64")]
|
||||
// CpuExtensions::Sse4_1 | CpuExtensions::Sse2 => {
|
||||
// sse2::multiply_alpha_sse2(src_image, dst_image)
|
||||
// }
|
||||
_ => native::multiply_alpha_inplace_native(image_u8x4),
|
||||
CpuExtensions::Avx2 => unsafe { avx2::mul::multiply_alpha_inplace_avx2(image_u8x4) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::mul::multiply_alpha_inplace_sse4(image_u8x4) },
|
||||
_ => native::mul::multiply_alpha_inplace_native(image_u8x4),
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -97,12 +96,14 @@ impl MulDiv {
|
||||
let (src_image_u8x4, dst_image_u8x4) = assert_images(src_image, dst_image)?;
|
||||
match self.cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::divide_alpha_avx2(src_image_u8x4, dst_image_u8x4),
|
||||
CpuExtensions::Avx2 => unsafe {
|
||||
avx2::div::divide_alpha_avx2(src_image_u8x4, dst_image_u8x4)
|
||||
},
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 | CpuExtensions::Sse2 => {
|
||||
sse2::divide_alpha_sse2(src_image_u8x4, dst_image_u8x4)
|
||||
}
|
||||
_ => native::divide_alpha_native(src_image_u8x4, dst_image_u8x4),
|
||||
CpuExtensions::Sse4_1 => unsafe {
|
||||
sse4::div::divide_alpha_sse4(src_image_u8x4, dst_image_u8x4)
|
||||
},
|
||||
_ => native::div::divide_alpha_native(src_image_u8x4, dst_image_u8x4),
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -112,12 +113,10 @@ impl MulDiv {
|
||||
let image_u8x4 = assert_image(image)?;
|
||||
match self.cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::divide_alpha_inplace_avx2(image_u8x4),
|
||||
CpuExtensions::Avx2 => unsafe { avx2::div::divide_alpha_inplace_avx2(image_u8x4) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 | CpuExtensions::Sse2 => {
|
||||
sse2::divide_alpha_inplace_sse2(image_u8x4)
|
||||
}
|
||||
_ => native::divide_alpha_inplace_native(image_u8x4),
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::div::divide_alpha_inplace_sse4(image_u8x4) },
|
||||
_ => native::div::divide_alpha_inplace_native(image_u8x4),
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
+67
-7
@@ -1,6 +1,7 @@
|
||||
use crate::image_view::{TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::U8x4;
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn divide_alpha_native(
|
||||
src_image: TypedImageView<U8x4>,
|
||||
mut dst_image: TypedImageViewMut<U8x4>,
|
||||
@@ -13,6 +14,7 @@ pub(crate) fn divide_alpha_native(
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn divide_alpha_inplace_native(mut image: TypedImageViewMut<U8x4>) {
|
||||
for dst_row in image.iter_rows_mut() {
|
||||
let src_row = unsafe { std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len()) };
|
||||
@@ -20,12 +22,6 @@ pub(crate) fn divide_alpha_inplace_native(mut image: TypedImageViewMut<U8x4>) {
|
||||
}
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) fn div_and_clip(v: u8, recip_alpha: f32) -> u8 {
|
||||
let res = v as f32 * recip_alpha;
|
||||
res.min(255.) as u8
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) fn divide_alpha_row_native(src_row: &[U8x4], dst_row: &mut [U8x4]) {
|
||||
src_row
|
||||
@@ -34,7 +30,7 @@ pub(crate) fn divide_alpha_row_native(src_row: &[U8x4], dst_row: &mut [U8x4]) {
|
||||
.for_each(|(src_pixel, dst_pixel)| {
|
||||
let components: [u8; 4] = src_pixel.0.to_le_bytes();
|
||||
let alpha = components[3];
|
||||
let recip_alpha = if alpha == 0 { 0. } else { 255. / alpha as f32 };
|
||||
let recip_alpha = RECIP_ALPHA[alpha as usize];
|
||||
dst_pixel.0 = u32::from_le_bytes([
|
||||
div_and_clip(components[0], recip_alpha),
|
||||
div_and_clip(components[1], recip_alpha),
|
||||
@@ -43,3 +39,67 @@ pub(crate) fn divide_alpha_row_native(src_row: &[U8x4], dst_row: &mut [U8x4]) {
|
||||
]);
|
||||
});
|
||||
}
|
||||
|
||||
const fn recip_alpha_array(precision: u32) -> [u32; 256] {
|
||||
let mut res = [0; 256];
|
||||
let scale = 1 << (precision + 1);
|
||||
let mut i: usize = 1;
|
||||
while i < 256 {
|
||||
res[i] = (((255 * scale / i as u32) + 1) >> 1) as u32;
|
||||
i += 1;
|
||||
}
|
||||
res
|
||||
}
|
||||
|
||||
const PRECISION: u32 = 8;
|
||||
|
||||
#[inline(always)]
|
||||
fn div_and_clip(v: u8, recip_alpha: u32) -> u8 {
|
||||
((v as u32 * recip_alpha) >> PRECISION).min(255) as u8
|
||||
}
|
||||
|
||||
const RECIP_ALPHA: [u32; 256] = recip_alpha_array(PRECISION);
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn test_recip_alpha_array() {
|
||||
for alpha in 0..=255u8 {
|
||||
let expected = if alpha == 0 {
|
||||
0
|
||||
} else {
|
||||
let scale = (1 << PRECISION) as f64;
|
||||
(255.0 * scale / alpha as f64).round() as u32
|
||||
};
|
||||
|
||||
let recip_alpha = RECIP_ALPHA[alpha as usize];
|
||||
assert_eq!(expected, recip_alpha, "alpha {}", alpha);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_div_and_clip() {
|
||||
let mut err_sum: i32 = 0;
|
||||
for alpha in 0..=255u8 {
|
||||
for color in 0..=255u8 {
|
||||
let multiplied_color = (color as f64 * alpha as f64 / 255.).round().min(255.) as u8;
|
||||
|
||||
let expected_color = if alpha == 0 {
|
||||
0
|
||||
} else {
|
||||
let recip_alpha = 255. / alpha as f64;
|
||||
let res = multiplied_color as f64 * recip_alpha;
|
||||
res.min(255.) as u8
|
||||
};
|
||||
|
||||
let recip_alpha = RECIP_ALPHA[alpha as usize];
|
||||
let result_color = div_and_clip(multiplied_color, recip_alpha);
|
||||
let delta = result_color as i32 - expected_color as i32;
|
||||
err_sum += delta.abs();
|
||||
}
|
||||
}
|
||||
assert_eq!(err_sum, 3468);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,7 +1,2 @@
|
||||
pub(crate) use div::{divide_alpha_inplace_native, divide_alpha_native, divide_alpha_row_native};
|
||||
pub(crate) use mul::{
|
||||
multiply_alpha_inplace_native, multiply_alpha_native, multiply_alpha_row_native,
|
||||
};
|
||||
|
||||
mod div;
|
||||
mod mul;
|
||||
pub(crate) mod div;
|
||||
pub(crate) mod mul;
|
||||
|
||||
@@ -1,69 +0,0 @@
|
||||
use std::arch::x86_64::*;
|
||||
|
||||
use crate::alpha::native;
|
||||
use crate::image_view::{TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::U8x4;
|
||||
use crate::simd_utils;
|
||||
|
||||
pub(crate) fn divide_alpha_sse2(
|
||||
src_image: TypedImageView<U8x4>,
|
||||
mut dst_image: TypedImageViewMut<U8x4>,
|
||||
) {
|
||||
let width = src_image.width().get() as usize;
|
||||
let src_rows = src_image.iter_rows(0);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
divide_alpha_row_sse2(src_row, dst_row, width);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn divide_alpha_inplace_sse2(mut image: TypedImageViewMut<U8x4>) {
|
||||
let width = image.width().get() as usize;
|
||||
for dst_row in image.iter_rows_mut() {
|
||||
unsafe {
|
||||
let src_row = std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len());
|
||||
divide_alpha_row_sse2(src_row, dst_row, width);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "sse2")]
|
||||
unsafe fn divide_alpha_row_sse2(src_row: &[U8x4], dst_row: &mut [U8x4], width: usize) {
|
||||
let zero = _mm_setzero_si128();
|
||||
let alpha_mask = _mm_set1_epi32(0xff000000u32 as i32);
|
||||
let shuffle0 = _mm_set_epi8(5, 4, 5, 4, 5, 4, 5, 4, 1, 0, 1, 0, 1, 0, 1, 0);
|
||||
let shuffle1 = _mm_set_epi8(13, 12, 13, 12, 13, 12, 13, 12, 9, 8, 9, 8, 9, 8, 9, 8);
|
||||
let alpha_scale = _mm_set1_ps(255.0 * 256.0);
|
||||
|
||||
let mut x: usize = 0;
|
||||
while x < width.saturating_sub(3) {
|
||||
let mut source = simd_utils::loadu_si128(src_row, x);
|
||||
|
||||
let alpha = _mm_and_si128(source, alpha_mask);
|
||||
let alpha_f32 = _mm_cvtepi32_ps(_mm_srli_epi32::<24>(source));
|
||||
let scaled_recip_alpha_f32 = _mm_mul_ps(alpha_scale, _mm_rcp_ps(alpha_f32));
|
||||
let scaled_recip_alpha_i32 = _mm_cvtps_epi32(scaled_recip_alpha_f32);
|
||||
let mma0 = _mm_shuffle_epi8(scaled_recip_alpha_i32, shuffle0);
|
||||
let mma1 = _mm_shuffle_epi8(scaled_recip_alpha_i32, shuffle1);
|
||||
|
||||
let mut pix0 = _mm_unpacklo_epi8(zero, source);
|
||||
let mut pix1 = _mm_unpackhi_epi8(zero, source);
|
||||
|
||||
pix0 = _mm_mulhi_epu16(pix0, mma0);
|
||||
pix1 = _mm_mulhi_epu16(pix1, mma1);
|
||||
|
||||
source = _mm_packus_epi16(pix0, pix1);
|
||||
source = _mm_blendv_epi8(source, alpha, alpha_mask);
|
||||
let dst_ptr = dst_row.get_unchecked_mut(x..).as_mut_ptr() as *mut __m128i;
|
||||
_mm_storeu_si128(dst_ptr, source);
|
||||
|
||||
x += 4;
|
||||
}
|
||||
|
||||
let src_tail = &src_row[x..];
|
||||
let dst_tail = &mut dst_row[x..];
|
||||
native::divide_alpha_row_native(src_tail, dst_tail);
|
||||
}
|
||||
@@ -1,4 +0,0 @@
|
||||
pub(crate) use div::{divide_alpha_inplace_sse2, divide_alpha_sse2};
|
||||
|
||||
mod div;
|
||||
mod mul;
|
||||
@@ -1,63 +0,0 @@
|
||||
use std::arch::x86_64::*;
|
||||
|
||||
use crate::alpha::native;
|
||||
use crate::image_view::{TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::U8x4;
|
||||
use crate::simd_utils;
|
||||
|
||||
#[allow(dead_code)]
|
||||
pub(crate) fn multiply_alpha_sse2(
|
||||
src_image: TypedImageView<U8x4>,
|
||||
mut dst_image: TypedImageViewMut<U8x4>,
|
||||
) {
|
||||
let width = src_image.width().get() as usize;
|
||||
let src_rows = src_image.iter_rows(0);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
multiply_alpha_row_sse2(src_row, dst_row, width);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// https://github.com/Wizermil/premultiply_alpha/blob/master/premultiply_alpha/premultiply_alpha.hpp#L108
|
||||
/// This implementation is twice slowly than native version.
|
||||
#[allow(dead_code)]
|
||||
#[target_feature(enable = "sse2")]
|
||||
unsafe fn multiply_alpha_row_sse2(src_row: &[U8x4], dst_row: &mut [U8x4], width: usize) {
|
||||
let mask_alpha_color_odd_255 = _mm_set1_epi32(0xff000000u32 as i32);
|
||||
let div_255 = _mm_set1_epi16(0x8081u16 as i16);
|
||||
|
||||
let mask_shuffle_alpha =
|
||||
_mm_set_epi8(15, -1, 15, -1, 11, -1, 11, -1, 7, -1, 7, -1, 3, -1, 3, -1);
|
||||
let mask_shuffle_color_odd =
|
||||
_mm_set_epi8(-1, -1, 13, -1, -1, -1, 9, -1, -1, -1, 5, -1, -1, -1, 1, -1);
|
||||
|
||||
let mut x: usize = 0;
|
||||
while x < width.saturating_sub(3) {
|
||||
let mut color = simd_utils::loadu_si128(src_row, x);
|
||||
let alpha = _mm_shuffle_epi8(color, mask_shuffle_alpha);
|
||||
|
||||
let mut color_even = _mm_slli_epi16::<8>(color);
|
||||
let mut color_odd = _mm_shuffle_epi8(color, mask_shuffle_color_odd);
|
||||
color_odd = _mm_or_si128(color_odd, mask_alpha_color_odd_255);
|
||||
|
||||
color_odd = _mm_mulhi_epu16(color_odd, alpha);
|
||||
color_even = _mm_mulhi_epu16(color_even, alpha);
|
||||
|
||||
color_odd = _mm_srli_epi16::<7>(_mm_mulhi_epu16(color_odd, div_255));
|
||||
color_even = _mm_srli_epi16::<7>(_mm_mulhi_epu16(color_even, div_255));
|
||||
|
||||
color = _mm_or_si128(color_even, _mm_slli_epi16::<8>(color_odd));
|
||||
|
||||
let dst_ptr = dst_row.get_unchecked_mut(x..).as_mut_ptr() as *mut __m128i;
|
||||
_mm_storeu_si128(dst_ptr, color);
|
||||
|
||||
x += 4;
|
||||
}
|
||||
|
||||
let src_tail = &src_row[x..];
|
||||
let dst_tail = &mut dst_row[x..];
|
||||
native::multiply_alpha_row_native(src_tail, dst_tail);
|
||||
}
|
||||
@@ -0,0 +1,83 @@
|
||||
use std::arch::x86_64::*;
|
||||
|
||||
use crate::image_view::{TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::U8x4;
|
||||
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
pub(crate) unsafe fn divide_alpha_sse4(
|
||||
src_image: TypedImageView<U8x4>,
|
||||
mut dst_image: TypedImageViewMut<U8x4>,
|
||||
) {
|
||||
let src_rows = src_image.iter_rows(0);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
divide_alpha_row_sse4(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
pub(crate) unsafe fn divide_alpha_inplace_sse4(mut image: TypedImageViewMut<U8x4>) {
|
||||
for dst_row in image.iter_rows_mut() {
|
||||
let src_row = std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len());
|
||||
divide_alpha_row_sse4(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
pub(crate) unsafe fn divide_alpha_row_sse4(src_row: &[U8x4], dst_row: &mut [U8x4]) {
|
||||
let src_chunks = src_row.chunks_exact(4);
|
||||
let src_remainder = src_chunks.remainder();
|
||||
let mut dst_chunks = dst_row.chunks_exact_mut(4);
|
||||
|
||||
for (src, dst) in src_chunks.zip(&mut dst_chunks) {
|
||||
divide_alpha(src.as_ptr(), dst.as_mut_ptr());
|
||||
}
|
||||
|
||||
if !src_remainder.is_empty() {
|
||||
let dst_reminder = dst_chunks.into_remainder();
|
||||
let mut src_pixels = [U8x4(0); 4];
|
||||
src_pixels
|
||||
.iter_mut()
|
||||
.zip(src_remainder)
|
||||
.for_each(|(d, s)| *d = *s);
|
||||
|
||||
let mut dst_pixels = [U8x4(0); 4];
|
||||
divide_alpha(src_pixels.as_ptr(), dst_pixels.as_mut_ptr());
|
||||
|
||||
dst_pixels
|
||||
.iter()
|
||||
.zip(dst_reminder)
|
||||
.for_each(|(s, d)| *d = *s);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
unsafe fn divide_alpha(src: *const U8x4, dst: *mut U8x4) {
|
||||
let zero = _mm_setzero_si128();
|
||||
let alpha_mask = _mm_set1_epi32(0xff000000u32 as i32);
|
||||
let shuffle1 = _mm_set_epi8(5, 4, 5, 4, 5, 4, 5, 4, 1, 0, 1, 0, 1, 0, 1, 0);
|
||||
let shuffle2 = _mm_set_epi8(13, 12, 13, 12, 13, 12, 13, 12, 9, 8, 9, 8, 9, 8, 9, 8);
|
||||
let alpha_scale = _mm_set1_ps(255.0 * 256.0);
|
||||
|
||||
let src_pixels = _mm_loadu_si128(src as *const __m128i);
|
||||
|
||||
let alpha_f32 = _mm_cvtepi32_ps(_mm_srli_epi32::<24>(src_pixels));
|
||||
let scaled_alpha_f32 = _mm_div_ps(alpha_scale, alpha_f32);
|
||||
// let scaled_alpha_f32 = _mm_mul_ps(alpha_scale, _mm_rcp_ps(alpha_f32));
|
||||
let scaled_alpha_i32 = _mm_cvtps_epi32(scaled_alpha_f32);
|
||||
let mma0 = _mm_shuffle_epi8(scaled_alpha_i32, shuffle1);
|
||||
let mma1 = _mm_shuffle_epi8(scaled_alpha_i32, shuffle2);
|
||||
|
||||
let pix0 = _mm_unpacklo_epi8(zero, src_pixels);
|
||||
let pix1 = _mm_unpackhi_epi8(zero, src_pixels);
|
||||
|
||||
let pix0 = _mm_mulhi_epu16(pix0, mma0);
|
||||
let pix1 = _mm_mulhi_epu16(pix1, mma1);
|
||||
|
||||
let alpha = _mm_and_si128(src_pixels, alpha_mask);
|
||||
let rgb = _mm_packus_epi16(pix0, pix1);
|
||||
let dst_pixels = _mm_blendv_epi8(rgb, alpha, alpha_mask);
|
||||
|
||||
_mm_storeu_si128(dst as *mut __m128i, dst_pixels);
|
||||
}
|
||||
@@ -0,0 +1,2 @@
|
||||
pub(crate) mod div;
|
||||
pub(crate) mod mul;
|
||||
@@ -0,0 +1,69 @@
|
||||
use std::arch::x86_64::*;
|
||||
|
||||
use crate::alpha::native;
|
||||
use crate::image_view::{TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::U8x4;
|
||||
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
pub(crate) unsafe fn multiply_alpha_sse4(
|
||||
src_image: TypedImageView<U8x4>,
|
||||
mut dst_image: TypedImageViewMut<U8x4>,
|
||||
) {
|
||||
let src_rows = src_image.iter_rows(0);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
multiply_alpha_row_sse4(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
pub(crate) unsafe fn multiply_alpha_inplace_sse4(mut image: TypedImageViewMut<U8x4>) {
|
||||
for dst_row in image.iter_rows_mut() {
|
||||
let src_row = std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len());
|
||||
multiply_alpha_row_sse4(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
unsafe fn multiply_alpha_row_sse4(src_row: &[U8x4], dst_row: &mut [U8x4]) {
|
||||
let zero = _mm_setzero_si128();
|
||||
let half = _mm_set1_epi16(128);
|
||||
|
||||
const MAX_A: i32 = 0xff000000u32 as i32;
|
||||
let max_alpha = _mm_set1_epi32(MAX_A);
|
||||
let factor_mask = _mm_set_epi8(15, 15, 15, 15, 11, 11, 11, 11, 7, 7, 7, 7, 3, 3, 3, 3);
|
||||
|
||||
let src_chunks = src_row.chunks_exact(4);
|
||||
let src_remainder = src_chunks.remainder();
|
||||
let mut dst_chunks = dst_row.chunks_exact_mut(4);
|
||||
|
||||
for (src, dst) in src_chunks.zip(&mut dst_chunks) {
|
||||
let src_pixels = _mm_loadu_si128(src.as_ptr() as *const __m128i);
|
||||
|
||||
let factor_pixels = _mm_shuffle_epi8(src_pixels, factor_mask);
|
||||
let factor_pixels = _mm_or_si128(factor_pixels, max_alpha);
|
||||
|
||||
let pix1 = _mm_unpacklo_epi8(src_pixels, zero);
|
||||
let factors = _mm_unpacklo_epi8(factor_pixels, zero);
|
||||
let pix1 = _mm_add_epi16(_mm_mullo_epi16(pix1, factors), half);
|
||||
let pix1 = _mm_add_epi16(pix1, _mm_srli_epi16::<8>(pix1));
|
||||
let pix1 = _mm_srli_epi16::<8>(pix1);
|
||||
|
||||
let pix2 = _mm_unpackhi_epi8(src_pixels, zero);
|
||||
let factors = _mm_unpackhi_epi8(factor_pixels, zero);
|
||||
let pix2 = _mm_add_epi16(_mm_mullo_epi16(pix2, factors), half);
|
||||
let pix2 = _mm_add_epi16(pix2, _mm_srli_epi16::<8>(pix2));
|
||||
let pix2 = _mm_srli_epi16::<8>(pix2);
|
||||
|
||||
let dst_pixels = _mm_packus_epi16(pix1, pix2);
|
||||
|
||||
_mm_storeu_si128(dst.as_mut_ptr() as *mut __m128i, dst_pixels);
|
||||
}
|
||||
|
||||
if !src_remainder.is_empty() {
|
||||
let dst_reminder = dst_chunks.into_remainder();
|
||||
native::mul::multiply_alpha_row_native(src_remainder, dst_reminder);
|
||||
}
|
||||
}
|
||||
@@ -10,8 +10,6 @@ use crate::pixels::{Pixel, PixelType};
|
||||
pub enum CpuExtensions {
|
||||
None,
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
Sse2,
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
Sse4_1,
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
Avx2,
|
||||
@@ -24,8 +22,6 @@ impl Default for CpuExtensions {
|
||||
Self::Avx2
|
||||
} else if is_x86_feature_detected!("sse4.1") {
|
||||
Self::Sse4_1
|
||||
} else if is_x86_feature_detected!("sse2") {
|
||||
Self::Sse2
|
||||
} else {
|
||||
Self::None
|
||||
}
|
||||
|
||||
+86
-4
@@ -4,6 +4,9 @@ use fast_image_resize::pixels::U8x4;
|
||||
use fast_image_resize::{
|
||||
CpuExtensions, Image, ImageRows, ImageRowsMut, ImageView, ImageViewMut, MulDiv, PixelType,
|
||||
};
|
||||
use utils::{cpu_ext_into_str, image_checksum};
|
||||
|
||||
mod utils;
|
||||
|
||||
const fn p(r: u8, g: u8, b: u8, a: u8) -> U8x4 {
|
||||
U8x4(u32::from_le_bytes([r, g, b, a]))
|
||||
@@ -83,8 +86,8 @@ fn multiply_alpha_avx2_test() {
|
||||
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
#[test]
|
||||
fn multiply_alpha_sse2_test() {
|
||||
multiply_alpha_test(CpuExtensions::Sse2);
|
||||
fn multiply_alpha_sse4_test() {
|
||||
multiply_alpha_test(CpuExtensions::Sse4_1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -164,11 +167,90 @@ fn divide_alpha_avx2_test() {
|
||||
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
#[test]
|
||||
fn divide_alpha_sse2_test() {
|
||||
divide_alpha_test(CpuExtensions::Sse2);
|
||||
fn divide_alpha_sse4_test() {
|
||||
divide_alpha_test(CpuExtensions::Sse4_1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn divide_alpha_native_test() {
|
||||
divide_alpha_test(CpuExtensions::None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn multiply_alpha_real_image_test() {
|
||||
let mut pixels = vec![0u32; 256 * 256];
|
||||
let mut i: usize = 0;
|
||||
for alpha in 0..=255u8 {
|
||||
for color in 0..=255u8 {
|
||||
let pixel = u32::from_le_bytes([color, color, color, alpha]);
|
||||
pixels[i] = pixel;
|
||||
i += 1;
|
||||
}
|
||||
}
|
||||
let size = NonZeroU32::new(256).unwrap();
|
||||
let src_image = Image::from_vec_u32(size, size, pixels, PixelType::U8x4).unwrap();
|
||||
let mut dst_image = Image::new(size, size, PixelType::U8x4);
|
||||
|
||||
let mut alpha_mul_div: MulDiv = Default::default();
|
||||
|
||||
let mut cpu_extensions_vec = vec![CpuExtensions::None];
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Avx2);
|
||||
}
|
||||
for cpu_extensions in cpu_extensions_vec {
|
||||
unsafe {
|
||||
alpha_mul_div.set_cpu_extensions(cpu_extensions);
|
||||
}
|
||||
alpha_mul_div
|
||||
.multiply_alpha(&src_image.view(), &mut dst_image.view_mut())
|
||||
.unwrap();
|
||||
|
||||
let name = format!("multiple_alpha-{}", cpu_ext_into_str(cpu_extensions));
|
||||
utils::save_result(&dst_image, &name);
|
||||
|
||||
let checksum = image_checksum::<4>(dst_image.buffer());
|
||||
assert_eq!(checksum, [4177920, 4177920, 4177920, 8355840]);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn divide_alpha_real_image_test() {
|
||||
let mut pixels = vec![0u32; 256 * 256];
|
||||
let mut i: usize = 0;
|
||||
for alpha in 0..=255u8 {
|
||||
for color in 0..=255u8 {
|
||||
let multiplied_color = (color as f64 * (alpha as f64 / 255.)).round().min(255.) as u8;
|
||||
let pixel =
|
||||
u32::from_le_bytes([multiplied_color, multiplied_color, multiplied_color, alpha]);
|
||||
pixels[i] = pixel;
|
||||
i += 1;
|
||||
}
|
||||
}
|
||||
let size = NonZeroU32::new(256).unwrap();
|
||||
let src_image = Image::from_vec_u32(size, size, pixels, PixelType::U8x4).unwrap();
|
||||
let mut dst_image = Image::new(size, size, PixelType::U8x4);
|
||||
|
||||
let mut alpha_mul_div: MulDiv = Default::default();
|
||||
|
||||
let mut cpu_extensions_vec = vec![CpuExtensions::None];
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
|
||||
cpu_extensions_vec.push(CpuExtensions::Avx2);
|
||||
}
|
||||
for cpu_extensions in cpu_extensions_vec {
|
||||
unsafe {
|
||||
alpha_mul_div.set_cpu_extensions(cpu_extensions);
|
||||
}
|
||||
alpha_mul_div
|
||||
.divide_alpha(&src_image.view(), &mut dst_image.view_mut())
|
||||
.unwrap();
|
||||
|
||||
let name = format!("divide_alpha-{}", cpu_ext_into_str(cpu_extensions));
|
||||
utils::save_result(&dst_image, &name);
|
||||
|
||||
let checksum = image_checksum::<4>(dst_image.buffer());
|
||||
assert_eq!(checksum, [8292504, 8292504, 8292504, 8355840]);
|
||||
}
|
||||
}
|
||||
|
||||
+30
-153
@@ -1,15 +1,13 @@
|
||||
use std::fs::File;
|
||||
use std::num::NonZeroU32;
|
||||
|
||||
use image::codecs::png::PngEncoder;
|
||||
use image::io::Reader as ImageReader;
|
||||
use image::{ColorType, DynamicImage, GenericImageView};
|
||||
|
||||
use fast_image_resize::pixels::*;
|
||||
use fast_image_resize::{
|
||||
CpuExtensions, DifferentTypesOfPixelsError, FilterType, Image, ImageView, PixelType, ResizeAlg,
|
||||
Resizer,
|
||||
};
|
||||
use utils::{cpu_ext_into_str, PixelExt};
|
||||
|
||||
mod utils;
|
||||
|
||||
fn get_new_height(src_image: &ImageView, new_width: u32) -> u32 {
|
||||
let scale = new_width as f32 / src_image.width().get() as f32;
|
||||
@@ -19,36 +17,6 @@ fn get_new_height(src_image: &ImageView, new_width: u32) -> u32 {
|
||||
const NEW_WIDTH: u32 = 255;
|
||||
const NEW_BIG_WIDTH: u32 = 5016;
|
||||
|
||||
fn save_result(image: &Image, name: &str) {
|
||||
if std::env::var("DONT_SAVE_RESULT").unwrap_or_else(|_| "".to_owned()) == "1" {
|
||||
return;
|
||||
}
|
||||
std::fs::create_dir_all("./data/result").unwrap();
|
||||
let mut file = File::create(format!("./data/result/{}.png", name)).unwrap();
|
||||
let color_type = match image.pixel_type() {
|
||||
PixelType::U8x3 => ColorType::Rgb8,
|
||||
PixelType::U8x4 => ColorType::Rgba8,
|
||||
PixelType::U8 => ColorType::L8,
|
||||
_ => panic!("Unsupported type of pixels"),
|
||||
};
|
||||
PngEncoder::new(&mut file)
|
||||
.encode(
|
||||
image.buffer(),
|
||||
image.width().get(),
|
||||
image.height().get(),
|
||||
color_type,
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
fn image_checksum<const N: usize>(buffer: &[u8]) -> [u32; N] {
|
||||
let mut res = [0u32; N];
|
||||
for pixel in buffer.chunks_exact(N) {
|
||||
res.iter_mut().zip(pixel).for_each(|(d, &s)| *d += s as u32);
|
||||
}
|
||||
res
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn try_resize_to_other_pixel_type() {
|
||||
let src_image = U8x4::load_big_src_image();
|
||||
@@ -64,88 +32,6 @@ fn try_resize_to_other_pixel_type() {
|
||||
));
|
||||
}
|
||||
|
||||
trait PixelExt: Pixel {
|
||||
fn pixel_type_str() -> &'static str {
|
||||
match Self::pixel_type() {
|
||||
PixelType::U8 => "u8",
|
||||
PixelType::U8x3 => "u8x3",
|
||||
PixelType::U8x4 => "u8x4",
|
||||
PixelType::I32 => "i32",
|
||||
PixelType::F32 => "f32",
|
||||
}
|
||||
}
|
||||
|
||||
fn load_big_src_image() -> Image<'static> {
|
||||
let img = ImageReader::open("./data/nasa-4928x3279.png")
|
||||
.unwrap()
|
||||
.decode()
|
||||
.unwrap();
|
||||
Image::from_vec_u8(
|
||||
NonZeroU32::new(img.width()).unwrap(),
|
||||
NonZeroU32::new(img.height()).unwrap(),
|
||||
Self::img_into_bytes(img),
|
||||
Self::pixel_type(),
|
||||
)
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
fn load_small_src_image() -> Image<'static> {
|
||||
let img = ImageReader::open("./data/nasa-852x567.png")
|
||||
.unwrap()
|
||||
.decode()
|
||||
.unwrap();
|
||||
Image::from_vec_u8(
|
||||
NonZeroU32::new(img.width()).unwrap(),
|
||||
NonZeroU32::new(img.height()).unwrap(),
|
||||
Self::img_into_bytes(img),
|
||||
Self::pixel_type(),
|
||||
)
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
fn img_into_bytes(img: DynamicImage) -> Vec<u8>;
|
||||
}
|
||||
|
||||
impl PixelExt for U8 {
|
||||
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
|
||||
img.to_luma8().into_raw()
|
||||
}
|
||||
}
|
||||
|
||||
impl PixelExt for U8x3 {
|
||||
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
|
||||
img.to_rgb8().into_raw()
|
||||
}
|
||||
}
|
||||
|
||||
impl PixelExt for U8x4 {
|
||||
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
|
||||
img.to_rgba8().into_raw()
|
||||
}
|
||||
}
|
||||
|
||||
impl PixelExt for I32 {
|
||||
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
|
||||
img.to_luma16()
|
||||
.as_raw()
|
||||
.iter()
|
||||
.map(|&p| p as u32 * (i16::MAX as u32 + 1))
|
||||
.flat_map(|val| val.to_le_bytes())
|
||||
.collect()
|
||||
}
|
||||
}
|
||||
|
||||
impl PixelExt for F32 {
|
||||
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
|
||||
img.to_luma16()
|
||||
.as_raw()
|
||||
.iter()
|
||||
.map(|&p| p as f32 * (i16::MAX as f32 + 1.0))
|
||||
.flat_map(|val| val.to_le_bytes())
|
||||
.collect()
|
||||
}
|
||||
}
|
||||
|
||||
fn downscale_test<P: PixelExt>(resize_alg: ResizeAlg, cpu_extensions: CpuExtensions) -> Vec<u8> {
|
||||
let image = P::load_big_src_image();
|
||||
assert_eq!(image.pixel_type(), P::pixel_type());
|
||||
@@ -179,23 +65,13 @@ fn downscale_test<P: PixelExt>(resize_alg: ResizeAlg, cpu_extensions: CpuExtensi
|
||||
_ => "unknown",
|
||||
};
|
||||
|
||||
let ext_name = match cpu_extensions {
|
||||
CpuExtensions::None => "native",
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse2 => "sse2",
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => "sse41",
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => "avx2",
|
||||
};
|
||||
|
||||
let name = format!(
|
||||
"downscale-{}-{}-{}",
|
||||
P::pixel_type_str(),
|
||||
alg_name,
|
||||
ext_name
|
||||
cpu_ext_into_str(cpu_extensions),
|
||||
);
|
||||
save_result(&result, &name);
|
||||
utils::save_result(&result, &name);
|
||||
result.buffer().to_owned()
|
||||
}
|
||||
|
||||
@@ -232,18 +108,13 @@ fn upscale_test<P: PixelExt>(resize_alg: ResizeAlg, cpu_extensions: CpuExtension
|
||||
_ => "unknown",
|
||||
};
|
||||
|
||||
let ext_name = match cpu_extensions {
|
||||
CpuExtensions::None => "native",
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse2 => "sse2",
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => "sse41",
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => "avx2",
|
||||
};
|
||||
|
||||
let name = format!("upscale-{}-{}-{}", P::pixel_type_str(), alg_name, ext_name);
|
||||
save_result(&result, &name);
|
||||
let name = format!(
|
||||
"upscale-{}-{}-{}",
|
||||
P::pixel_type_str(),
|
||||
alg_name,
|
||||
cpu_ext_into_str(cpu_extensions),
|
||||
);
|
||||
utils::save_result(&result, &name);
|
||||
result.buffer().to_owned()
|
||||
}
|
||||
|
||||
@@ -251,7 +122,7 @@ fn upscale_test<P: PixelExt>(resize_alg: ResizeAlg, cpu_extensions: CpuExtension
|
||||
fn downscale_u8() {
|
||||
type P = U8;
|
||||
let buffer = downscale_test::<P>(ResizeAlg::Nearest, CpuExtensions::None);
|
||||
assert_eq!(image_checksum::<1>(&buffer), [2920317]);
|
||||
assert_eq!(utils::image_checksum::<1>(&buffer), [2920317]);
|
||||
|
||||
let mut cpu_extensions_vec = vec![CpuExtensions::None];
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
@@ -261,7 +132,7 @@ fn downscale_u8() {
|
||||
for cpu_extensions in cpu_extensions_vec {
|
||||
let buffer =
|
||||
downscale_test::<P>(ResizeAlg::Convolution(FilterType::Lanczos3), cpu_extensions);
|
||||
assert_eq!(image_checksum::<1>(&buffer), [2923520]);
|
||||
assert_eq!(utils::image_checksum::<1>(&buffer), [2923520]);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -269,7 +140,7 @@ fn downscale_u8() {
|
||||
fn upscale_u8() {
|
||||
type P = U8;
|
||||
let buffer = upscale_test::<P>(ResizeAlg::Nearest, CpuExtensions::None);
|
||||
assert_eq!(image_checksum::<1>(&buffer), [1148750539]);
|
||||
assert_eq!(utils::image_checksum::<1>(&buffer), [1148750539]);
|
||||
|
||||
let mut cpu_extensions_vec = vec![CpuExtensions::None];
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
@@ -279,7 +150,7 @@ fn upscale_u8() {
|
||||
for cpu_extensions in cpu_extensions_vec {
|
||||
let buffer =
|
||||
upscale_test::<P>(ResizeAlg::Convolution(FilterType::Lanczos3), cpu_extensions);
|
||||
assert_eq!(image_checksum::<1>(&buffer), [1148808058]);
|
||||
assert_eq!(utils::image_checksum::<1>(&buffer), [1148808058]);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -287,7 +158,10 @@ fn upscale_u8() {
|
||||
fn downscale_u8x3() {
|
||||
type P = U8x3;
|
||||
let buffer = downscale_test::<P>(ResizeAlg::Nearest, CpuExtensions::None);
|
||||
assert_eq!(image_checksum::<3>(&buffer), [2937940, 2945380, 2882679]);
|
||||
assert_eq!(
|
||||
utils::image_checksum::<3>(&buffer),
|
||||
[2937940, 2945380, 2882679]
|
||||
);
|
||||
|
||||
let mut cpu_extensions_vec = vec![CpuExtensions::None];
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
@@ -298,7 +172,10 @@ fn downscale_u8x3() {
|
||||
for cpu_extensions in cpu_extensions_vec {
|
||||
let buffer =
|
||||
downscale_test::<P>(ResizeAlg::Convolution(FilterType::Lanczos3), cpu_extensions);
|
||||
assert_eq!(image_checksum::<3>(&buffer), [2942479, 2947850, 2885072]);
|
||||
assert_eq!(
|
||||
utils::image_checksum::<3>(&buffer),
|
||||
[2942479, 2947850, 2885072]
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -307,7 +184,7 @@ fn upscale_u8x3() {
|
||||
type P = U8x3;
|
||||
let buffer = upscale_test::<P>(ResizeAlg::Nearest, CpuExtensions::None);
|
||||
assert_eq!(
|
||||
image_checksum::<3>(&buffer),
|
||||
utils::image_checksum::<3>(&buffer),
|
||||
[1156008260, 1158417906, 1135087540]
|
||||
);
|
||||
|
||||
@@ -321,7 +198,7 @@ fn upscale_u8x3() {
|
||||
let buffer =
|
||||
upscale_test::<P>(ResizeAlg::Convolution(FilterType::Lanczos3), cpu_extensions);
|
||||
assert_eq!(
|
||||
image_checksum::<3>(&buffer),
|
||||
utils::image_checksum::<3>(&buffer),
|
||||
[1156107005, 1158443335, 1135101759]
|
||||
);
|
||||
}
|
||||
@@ -332,7 +209,7 @@ fn downscale_u8x4() {
|
||||
type P = U8x4;
|
||||
let buffer = downscale_test::<P>(ResizeAlg::Nearest, CpuExtensions::None);
|
||||
assert_eq!(
|
||||
image_checksum::<4>(&buffer),
|
||||
utils::image_checksum::<4>(&buffer),
|
||||
[2937940, 2945380, 2882679, 11054250]
|
||||
);
|
||||
|
||||
@@ -346,7 +223,7 @@ fn downscale_u8x4() {
|
||||
let buffer =
|
||||
downscale_test::<P>(ResizeAlg::Convolution(FilterType::Lanczos3), cpu_extensions);
|
||||
assert_eq!(
|
||||
image_checksum::<4>(&buffer),
|
||||
utils::image_checksum::<4>(&buffer),
|
||||
[2942479, 2947850, 2885072, 11054250]
|
||||
);
|
||||
|
||||
@@ -362,7 +239,7 @@ fn upscale_u8x4() {
|
||||
type P = U8x4;
|
||||
let buffer = upscale_test::<P>(ResizeAlg::Nearest, CpuExtensions::None);
|
||||
assert_eq!(
|
||||
image_checksum::<4>(&buffer),
|
||||
utils::image_checksum::<4>(&buffer),
|
||||
[1156008260, 1158417906, 1135087540, 4269569040]
|
||||
);
|
||||
|
||||
@@ -376,7 +253,7 @@ fn upscale_u8x4() {
|
||||
let buffer =
|
||||
upscale_test::<P>(ResizeAlg::Convolution(FilterType::Lanczos3), cpu_extensions);
|
||||
assert_eq!(
|
||||
image_checksum::<4>(&buffer),
|
||||
utils::image_checksum::<4>(&buffer),
|
||||
[1156107005, 1158443335, 1135101759, 4269569040]
|
||||
);
|
||||
}
|
||||
|
||||
+145
@@ -0,0 +1,145 @@
|
||||
use std::fs::File;
|
||||
use std::num::NonZeroU32;
|
||||
|
||||
use image::codecs::png::PngEncoder;
|
||||
use image::io::Reader as ImageReader;
|
||||
use image::{ColorType, DynamicImage, GenericImageView};
|
||||
|
||||
use fast_image_resize::pixels::*;
|
||||
use fast_image_resize::{CpuExtensions, Image, PixelType};
|
||||
|
||||
pub fn image_checksum<const N: usize>(buffer: &[u8]) -> [u32; N] {
|
||||
let mut res = [0u32; N];
|
||||
for pixel in buffer.chunks_exact(N) {
|
||||
res.iter_mut().zip(pixel).for_each(|(d, &s)| *d += s as u32);
|
||||
}
|
||||
res
|
||||
}
|
||||
|
||||
pub trait PixelExt: Pixel {
|
||||
fn pixel_type_str() -> &'static str {
|
||||
match Self::pixel_type() {
|
||||
PixelType::U8 => "u8",
|
||||
PixelType::U8x3 => "u8x3",
|
||||
PixelType::U8x4 => "u8x4",
|
||||
PixelType::I32 => "i32",
|
||||
PixelType::F32 => "f32",
|
||||
}
|
||||
}
|
||||
|
||||
fn load_big_src_image() -> Image<'static> {
|
||||
let img = ImageReader::open("./data/nasa-4928x3279.png")
|
||||
.unwrap()
|
||||
.decode()
|
||||
.unwrap();
|
||||
Image::from_vec_u8(
|
||||
NonZeroU32::new(img.width()).unwrap(),
|
||||
NonZeroU32::new(img.height()).unwrap(),
|
||||
Self::img_into_bytes(img),
|
||||
Self::pixel_type(),
|
||||
)
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
fn load_small_src_image() -> Image<'static> {
|
||||
let img = ImageReader::open("./data/nasa-852x567.png")
|
||||
.unwrap()
|
||||
.decode()
|
||||
.unwrap();
|
||||
Image::from_vec_u8(
|
||||
NonZeroU32::new(img.width()).unwrap(),
|
||||
NonZeroU32::new(img.height()).unwrap(),
|
||||
Self::img_into_bytes(img),
|
||||
Self::pixel_type(),
|
||||
)
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
fn load_small_rgba_image() -> Image<'static> {
|
||||
let img = ImageReader::open("./data/nasa-852x567-rgba.png")
|
||||
.unwrap()
|
||||
.decode()
|
||||
.unwrap();
|
||||
Image::from_vec_u8(
|
||||
NonZeroU32::new(img.width()).unwrap(),
|
||||
NonZeroU32::new(img.height()).unwrap(),
|
||||
Self::img_into_bytes(img),
|
||||
Self::pixel_type(),
|
||||
)
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
fn img_into_bytes(img: DynamicImage) -> Vec<u8>;
|
||||
}
|
||||
|
||||
impl PixelExt for U8 {
|
||||
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
|
||||
img.to_luma8().into_raw()
|
||||
}
|
||||
}
|
||||
|
||||
impl PixelExt for U8x3 {
|
||||
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
|
||||
img.to_rgb8().into_raw()
|
||||
}
|
||||
}
|
||||
|
||||
impl PixelExt for U8x4 {
|
||||
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
|
||||
img.to_rgba8().into_raw()
|
||||
}
|
||||
}
|
||||
|
||||
impl PixelExt for I32 {
|
||||
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
|
||||
img.to_luma16()
|
||||
.as_raw()
|
||||
.iter()
|
||||
.map(|&p| p as u32 * (i16::MAX as u32 + 1))
|
||||
.flat_map(|val| val.to_le_bytes())
|
||||
.collect()
|
||||
}
|
||||
}
|
||||
|
||||
impl PixelExt for F32 {
|
||||
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
|
||||
img.to_luma16()
|
||||
.as_raw()
|
||||
.iter()
|
||||
.map(|&p| p as f32 * (i16::MAX as f32 + 1.0))
|
||||
.flat_map(|val| val.to_le_bytes())
|
||||
.collect()
|
||||
}
|
||||
}
|
||||
|
||||
pub fn save_result(image: &Image, name: &str) {
|
||||
if std::env::var("DONT_SAVE_RESULT").unwrap_or_else(|_| "".to_owned()) == "1" {
|
||||
return;
|
||||
}
|
||||
std::fs::create_dir_all("./data/result").unwrap();
|
||||
let mut file = File::create(format!("./data/result/{}.png", name)).unwrap();
|
||||
let color_type = match image.pixel_type() {
|
||||
PixelType::U8x3 => ColorType::Rgb8,
|
||||
PixelType::U8x4 => ColorType::Rgba8,
|
||||
PixelType::U8 => ColorType::L8,
|
||||
_ => panic!("Unsupported type of pixels"),
|
||||
};
|
||||
PngEncoder::new(&mut file)
|
||||
.encode(
|
||||
image.buffer(),
|
||||
image.width().get(),
|
||||
image.height().get(),
|
||||
color_type,
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
pub fn cpu_ext_into_str(cpu_extensions: CpuExtensions) -> &'static str {
|
||||
match cpu_extensions {
|
||||
CpuExtensions::None => "native",
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => "sse41",
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => "avx2",
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user