Added support for optimization with help of SSE4.1 and AVX2 for the F32x4 pixel type (#30).

This commit is contained in:
Kirill Kuzminykh
2024-06-23 21:57:54 +03:00
parent f42b3b6edc
commit 84d6c17b98
10 changed files with 299 additions and 145 deletions
+1 -2
View File
@@ -4,9 +4,8 @@
- Added support for optimization with help of `SSE4.1` and `AVX2` for
the `F32` pixel type.
- Added support for new pixel types `F32x2` and `F32x3` with
- Added support for new pixel types `F32x2`, `F32x3` and `F32x4` with
optimizations for `SSE4.1` and `AVX2` (#30).
- Added basic support for the new pixel type `F32x4` (#30).
## [4.0.0] - 2024-05-13
Generated
+24 -24
View File
@@ -113,7 +113,7 @@ checksum = "0ae92a5119aa49cdbcf6b9f893fe4e1d98b04ccbf82ee0584ad948a44a734dea"
dependencies = [
"proc-macro2",
"quote",
"syn 2.0.66",
"syn 2.0.67",
]
[[package]]
@@ -171,9 +171,9 @@ checksum = "cf4b9d6a944f767f8e5e0db018570623c85f3d925ac718db4e06d0187adb21c1"
[[package]]
name = "bitstream-io"
version = "2.4.2"
version = "2.3.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "415f8399438eb5e4b2f73ed3152a3448b98149dda642a957ee704e1daa5cf1d8"
checksum = "7c12d1856e42f0d817a835fe55853957c85c8c8a470114029143d3f12671446e"
[[package]]
name = "block-buffer"
@@ -208,9 +208,9 @@ checksum = "79296716171880943b8470b5f8d03aa55eb2e645a4874bdbb28adb49162e012c"
[[package]]
name = "bytemuck"
version = "1.16.0"
version = "1.16.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "78834c15cb5d5efe3452d58b1e8ba890dd62d21907f867f383358198e56ebca5"
checksum = "b236fc92302c97ed75b38da1f4917b5cdda4984745740f153a5d3059e48d725e"
[[package]]
name = "byteorder"
@@ -365,7 +365,7 @@ dependencies = [
"heck",
"proc-macro2",
"quote",
"syn 2.0.66",
"syn 2.0.67",
]
[[package]]
@@ -808,7 +808,7 @@ checksum = "c34819042dc3d3971c46c2190835914dfbe0c3c13f61449b2997f4e9722dfa60"
dependencies = [
"proc-macro2",
"quote",
"syn 2.0.66",
"syn 2.0.67",
]
[[package]]
@@ -887,9 +887,9 @@ dependencies = [
[[package]]
name = "lazy_static"
version = "1.4.0"
version = "1.5.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e2abad23fbc42b3700f2f279844dc832adb2b2eb069b2df918f455c4e18cc646"
checksum = "bbd2bcb4c963f2ddae06a2efc7e9f3591312473c50c6685e1f298068316e66fe"
[[package]]
name = "lebe"
@@ -985,9 +985,9 @@ checksum = "68354c5c6bd36d73ff3feceb05efa59b6acb7626617f4962be322a825e61f79a"
[[package]]
name = "miniz_oxide"
version = "0.7.3"
version = "0.7.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "87dfd01fe195c66b572b37921ad8803d010623c0aca821bea2302239d155cdae"
checksum = "b8a240ddb74feaf34a79a7add65a741f3167852fba007066dcac1ca548d89c08"
dependencies = [
"adler",
"simd-adler32",
@@ -1056,7 +1056,7 @@ checksum = "ed3955f1a9c7c0c15e092f9c887db08b1fc683305fdf6eb6684f22555355e202"
dependencies = [
"proc-macro2",
"quote",
"syn 2.0.66",
"syn 2.0.67",
]
[[package]]
@@ -1152,7 +1152,7 @@ dependencies = [
"pest_meta",
"proc-macro2",
"quote",
"syn 2.0.66",
"syn 2.0.67",
]
[[package]]
@@ -1231,9 +1231,9 @@ checksum = "5b40af805b3121feab8a3c29f04d8ad262fa8e0561883e7653e024ae4479e6de"
[[package]]
name = "proc-macro2"
version = "1.0.85"
version = "1.0.86"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "22244ce15aa966053a896d1accb3a6e68469b97c7f33f284b99f0d576879fc23"
checksum = "5e719e8df665df0d1c8fbfd238015744736151d4445ec0836b8e628aae103b77"
dependencies = [
"unicode-ident",
]
@@ -1254,7 +1254,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8021cf59c8ec9c432cfc2526ac6b8aa508ecaf29cd415f271b8406c1b851c3fd"
dependencies = [
"quote",
"syn 2.0.66",
"syn 2.0.67",
]
[[package]]
@@ -1348,9 +1348,9 @@ dependencies = [
[[package]]
name = "ravif"
version = "0.11.7"
version = "0.11.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "67376f469e7e7840d0040bbf4b9b3334005bb167f814621326e4c7ab8cd6e944"
checksum = "bc13288f5ab39e6d7c9d501759712e6969fcc9734220846fc9ed26cae2cc4234"
dependencies = [
"avif-serialize",
"imgref",
@@ -1481,7 +1481,7 @@ checksum = "500cbc0ebeb6f46627f50f3f5811ccf6bf00643be300b4c3eabc0ef55dc5b5ba"
dependencies = [
"proc-macro2",
"quote",
"syn 2.0.66",
"syn 2.0.67",
]
[[package]]
@@ -1580,9 +1580,9 @@ dependencies = [
[[package]]
name = "syn"
version = "2.0.66"
version = "2.0.67"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c42f3f41a2de00b01c0aaad383c5a45241efc8b2d1eda5661812fda5f3cdcff5"
checksum = "ff8655ed1d86f3af4ee3fd3263786bc14245ad17c4c7e85ba7187fb3ae028c90"
dependencies = [
"proc-macro2",
"quote",
@@ -1655,7 +1655,7 @@ checksum = "46c3384250002a6d5af4d114f2845d37b57521033f30d5c3f46c4d70e1197533"
dependencies = [
"proc-macro2",
"quote",
"syn 2.0.66",
"syn 2.0.67",
]
[[package]]
@@ -1847,7 +1847,7 @@ dependencies = [
"once_cell",
"proc-macro2",
"quote",
"syn 2.0.66",
"syn 2.0.67",
"wasm-bindgen-shared",
]
@@ -1869,7 +1869,7 @@ checksum = "e94f17b526d0a461a191c78ea52bbce64071ed5c04c9ffe424dcb38f74171bb7"
dependencies = [
"proc-macro2",
"quote",
"syn 2.0.66",
"syn 2.0.67",
"wasm-bindgen-backend",
"wasm-bindgen-shared",
]
+18 -24
View File
@@ -24,7 +24,7 @@ Supported pixel formats and available optimizations:
| F32 | One `f32` component per pixel (e.g. L) | + | + | - | - |
| F32x2 | Two `f32` components per pixel (e.g. LA32F) | + | + | - | - |
| F32x3 | Three `f32` components per pixel (e.g. RGB32F) | + | + | - | - |
| F32x4 | Four `f32` components per pixel (e.g. RGBA32F) | - | - | - | - |
| F32x4 | Four `f32` components per pixel (e.g. RGBA32F) | + | + | - | - |
## Colorspace
@@ -58,7 +58,6 @@ Other libraries used to compare of resizing speed:
- libvips (single-threaded mode, cache disabled)
<!-- bench_compare_rgb start -->
### Resize RGB8 image (U8x3) 4928x3279 => 852x567
Pipeline:
@@ -70,17 +69,15 @@ Pipeline:
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
| image | 28.28 | - | 82.31 | 134.61 | 186.22 |
| resize | 7.83 | 26.85 | 53.70 | 97.59 | 144.79 |
| libvips | 7.76 | 59.65 | 19.83 | 30.73 | 39.80 |
| fir rust | 0.28 | 12.77 | 17.94 | 27.94 | 40.05 |
| fir sse4.1 | 0.28 | 3.99 | 5.66 | 9.86 | 15.42 |
| fir avx2 | 0.28 | 3.06 | 3.89 | 6.85 | 13.25 |
| image | 32.06 | - | 94.81 | 153.34 | 212.23 |
| resize | 9.12 | 26.76 | 52.27 | 97.57 | 144.36 |
| libvips | 7.75 | 59.45 | 19.98 | 30.79 | 39.94 |
| fir rust | 0.29 | 11.91 | 16.56 | 26.11 | 37.89 |
| fir sse4.1 | 0.29 | 4.09 | 5.68 | 9.79 | 15.45 |
| fir avx2 | 0.29 | 3.10 | 3.98 | 6.82 | 13.30 |
<!-- bench_compare_rgb end -->
<!-- bench_compare_rgba start -->
### Resize RGBA8 image (U8x4) 4928x3279 => 852x567
Pipeline:
@@ -94,16 +91,14 @@ Pipeline:
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|------------|:-------:|:------:|:--------:|:-------:|:--------:|
| resize | 11.34 | 42.76 | 85.43 | 147.41 | 211.31 |
| libvips | 8.97 | 120.82 | 190.82 | 341.07 | 500.87 |
| fir rust | 0.19 | 21.79 | 28.34 | 42.24 | 56.72 |
| fir sse4.1 | 0.19 | 10.14 | 12.26 | 18.43 | 24.22 |
| fir avx2 | 0.19 | 7.44 | 8.48 | 14.10 | 21.31 |
| resize | 11.45 | 42.73 | 85.20 | 147.41 | 211.63 |
| libvips | 9.77 | 120.67 | 189.16 | 336.71 | 500.29 |
| fir rust | 0.19 | 21.84 | 26.36 | 36.95 | 50.38 |
| fir sse4.1 | 0.19 | 10.22 | 12.42 | 17.84 | 24.86 |
| fir avx2 | 0.19 | 7.85 | 8.87 | 13.82 | 22.10 |
<!-- bench_compare_rgba end -->
<!-- bench_compare_l start -->
### Resize L8 image (U8) 4928x3279 => 852x567
Pipeline:
@@ -116,13 +111,12 @@ Pipeline:
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
| image | 25.53 | - | 56.37 | 84.03 | 111.89 |
| resize | 5.60 | 10.85 | 18.72 | 37.97 | 65.65 |
| libvips | 4.67 | 25.02 | 9.80 | 13.73 | 18.17 |
| fir rust | 0.15 | 4.94 | 6.37 | 8.97 | 13.36 |
| fir sse4.1 | 0.15 | 1.66 | 2.10 | 3.28 | 5.57 |
| fir avx2 | 0.15 | 1.76 | 1.96 | 2.37 | 4.41 |
| image | 28.75 | - | 60.74 | 89.30 | 117.34 |
| resize | 6.83 | 11.03 | 20.67 | 43.74 | 67.86 |
| libvips | 4.66 | 25.00 | 9.74 | 13.19 | 17.94 |
| fir rust | 0.15 | 4.54 | 5.69 | 8.05 | 12.22 |
| fir sse4.1 | 0.15 | 1.70 | 2.16 | 3.34 | 5.64 |
| fir avx2 | 0.15 | 1.74 | 1.90 | 2.31 | 4.28 |
<!-- bench_compare_l end -->
## Examples
+1 -1
View File
@@ -11,7 +11,7 @@ Environment:
- Ubuntu 22.04 (linux 6.5.0)
- Rust 1.79
- criterion = "0.5.1"
- fast_image_resize = "4.0.0"
- fast_image_resize = "4.1.0"
{% if arch_id == "wasm32" -%}
- wasmtime = "20.0.0"
{% endif %}
+71 -67
View File
@@ -9,7 +9,7 @@ Environment:
- Ubuntu 22.04 (linux 6.5.0)
- Rust 1.79
- criterion = "0.5.1"
- fast_image_resize = "4.0.0"
- fast_image_resize = "4.1.0"
Other libraries used to compare of resizing speed:
@@ -40,12 +40,12 @@ Pipeline:
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
| image | 28.28 | - | 82.31 | 134.61 | 186.22 |
| resize | 7.83 | 26.85 | 53.70 | 97.59 | 144.79 |
| libvips | 7.76 | 59.65 | 19.83 | 30.73 | 39.80 |
| fir rust | 0.28 | 12.77 | 17.94 | 27.94 | 40.05 |
| fir sse4.1 | 0.28 | 3.99 | 5.66 | 9.86 | 15.42 |
| fir avx2 | 0.28 | 3.06 | 3.89 | 6.85 | 13.25 |
| image | 32.06 | - | 94.81 | 153.34 | 212.23 |
| resize | 9.12 | 26.76 | 52.27 | 97.57 | 144.36 |
| libvips | 7.75 | 59.45 | 19.98 | 30.79 | 39.94 |
| fir rust | 0.29 | 11.91 | 16.56 | 26.11 | 37.89 |
| fir sse4.1 | 0.29 | 4.09 | 5.68 | 9.79 | 15.45 |
| fir avx2 | 0.29 | 3.10 | 3.98 | 6.82 | 13.30 |
<!-- bench_compare_rgb end -->
@@ -64,11 +64,11 @@ Pipeline:
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|------------|:-------:|:------:|:--------:|:-------:|:--------:|
| resize | 11.34 | 42.76 | 85.43 | 147.41 | 211.31 |
| libvips | 8.97 | 120.82 | 190.82 | 341.07 | 500.87 |
| fir rust | 0.19 | 21.79 | 28.34 | 42.24 | 56.72 |
| fir sse4.1 | 0.19 | 10.14 | 12.26 | 18.43 | 24.22 |
| fir avx2 | 0.19 | 7.44 | 8.48 | 14.10 | 21.31 |
| resize | 11.45 | 42.73 | 85.20 | 147.41 | 211.63 |
| libvips | 9.77 | 120.67 | 189.16 | 336.71 | 500.29 |
| fir rust | 0.19 | 21.84 | 26.36 | 36.95 | 50.38 |
| fir sse4.1 | 0.19 | 10.22 | 12.42 | 17.84 | 24.86 |
| fir avx2 | 0.19 | 7.85 | 8.87 | 13.82 | 22.10 |
<!-- bench_compare_rgba end -->
@@ -86,12 +86,12 @@ Pipeline:
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
| image | 25.53 | - | 56.37 | 84.03 | 111.89 |
| resize | 5.60 | 10.85 | 18.72 | 37.97 | 65.65 |
| libvips | 4.67 | 25.02 | 9.80 | 13.73 | 18.17 |
| fir rust | 0.15 | 4.94 | 6.37 | 8.97 | 13.36 |
| fir sse4.1 | 0.15 | 1.66 | 2.10 | 3.28 | 5.57 |
| fir avx2 | 0.15 | 1.76 | 1.96 | 2.37 | 4.41 |
| image | 28.75 | - | 60.74 | 89.30 | 117.34 |
| resize | 6.83 | 11.03 | 20.67 | 43.74 | 67.86 |
| libvips | 4.66 | 25.00 | 9.74 | 13.19 | 17.94 |
| fir rust | 0.15 | 4.54 | 5.69 | 8.05 | 12.22 |
| fir sse4.1 | 0.15 | 1.70 | 2.16 | 3.34 | 5.64 |
| fir avx2 | 0.15 | 1.74 | 1.90 | 2.31 | 4.28 |
<!-- bench_compare_l end -->
@@ -105,17 +105,17 @@ Pipeline:
- Source image
[nasa-4928x3279-rgba.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279-rgba.png)
has converted into grayscale image with alpha channel (two bytes per pixel).
has converted into grayscale image with an alpha channel (two bytes per pixel).
- Numbers in the table mean a duration of image resizing in milliseconds.
- The `image` crate does not support multiplying and dividing by alpha channel.
- The `resize` crate does not support this pixel format.
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
| libvips | 6.45 | 73.39 | 117.87 | 205.86 | 293.32 |
| fir rust | 0.17 | 16.83 | 18.97 | 24.14 | 31.45 |
| fir sse4.1 | 0.17 | 6.15 | 7.17 | 9.64 | 13.40 |
| fir avx2 | 0.17 | 4.30 | 4.89 | 6.50 | 9.73 |
| libvips | 6.45 | 72.42 | 117.32 | 205.22 | 293.11 |
| fir rust | 0.17 | 18.09 | 20.67 | 25.58 | 32.87 |
| fir sse4.1 | 0.17 | 6.18 | 7.20 | 9.61 | 13.48 |
| fir avx2 | 0.17 | 4.34 | 4.91 | 6.53 | 9.59 |
<!-- bench_compare_la end -->
@@ -133,12 +133,12 @@ Pipeline:
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
| image | 28.09 | - | 82.63 | 134.66 | 185.48 |
| resize | 8.06 | 26.77 | 51.13 | 97.40 | 144.26 |
| libvips | 16.00 | 63.19 | 54.32 | 103.03 | 125.58 |
| fir rust | 0.33 | 27.24 | 42.15 | 72.69 | 104.45 |
| fir sse4.1 | 0.33 | 15.98 | 23.43 | 38.31 | 54.65 |
| fir avx2 | 0.33 | 13.90 | 19.49 | 29.92 | 36.67 |
| image | 31.04 | - | 87.07 | 139.57 | 190.73 |
| resize | 8.09 | 26.30 | 49.97 | 96.79 | 143.58 |
| libvips | 16.00 | 62.91 | 54.53 | 102.92 | 125.23 |
| fir rust | 0.35 | 30.77 | 47.74 | 81.36 | 115.72 |
| fir sse4.1 | 0.35 | 16.16 | 23.62 | 38.31 | 54.64 |
| fir avx2 | 0.35 | 13.85 | 19.15 | 29.74 | 36.50 |
<!-- bench_compare_rgb16 end -->
@@ -157,11 +157,11 @@ Pipeline:
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|------------|:-------:|:------:|:--------:|:-------:|:--------:|
| resize | 12.17 | 42.81 | 83.83 | 144.13 | 206.21 |
| libvips | 22.58 | 130.14 | 209.14 | 367.87 | 536.49 |
| fir rust | 0.37 | 59.65 | 78.35 | 116.41 | 155.49 |
| fir sse4.1 | 0.37 | 31.77 | 42.27 | 63.73 | 85.90 |
| fir avx2 | 0.37 | 20.39 | 26.03 | 36.49 | 47.81 |
| resize | 12.27 | 43.48 | 83.80 | 144.81 | 207.30 |
| libvips | 22.68 | 128.29 | 205.54 | 364.11 | 533.24 |
| fir rust | 0.39 | 62.43 | 82.36 | 122.70 | 164.59 |
| fir sse4.1 | 0.39 | 31.92 | 42.43 | 63.89 | 85.88 |
| fir avx2 | 0.39 | 20.45 | 26.17 | 36.72 | 48.24 |
<!-- bench_compare_rgba16 end -->
@@ -179,12 +179,12 @@ Pipeline:
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
| image | 25.68 | - | 56.85 | 85.61 | 114.66 |
| resize | 6.30 | 9.95 | 16.04 | 33.38 | 58.32 |
| libvips | 7.42 | 26.00 | 21.69 | 36.26 | 46.04 |
| fir rust | 0.17 | 13.56 | 19.27 | 28.71 | 40.87 |
| fir sse4.1 | 0.17 | 5.26 | 7.44 | 12.86 | 18.78 |
| fir avx2 | 0.17 | 5.44 | 6.42 | 8.56 | 13.65 |
| image | 28.43 | - | 61.77 | 90.52 | 119.06 |
| resize | 6.37 | 11.09 | 20.38 | 42.28 | 67.60 |
| libvips | 7.63 | 26.04 | 21.73 | 36.34 | 45.95 |
| fir rust | 0.17 | 14.63 | 20.81 | 30.42 | 43.41 |
| fir sse4.1 | 0.17 | 5.31 | 7.48 | 12.85 | 18.79 |
| fir avx2 | 0.17 | 5.55 | 6.45 | 8.50 | 13.61 |
<!-- bench_compare_l16 end -->
@@ -198,17 +198,17 @@ Pipeline:
- Source image
[nasa-4928x3279-rgba.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279-rgba.png)
has converted into grayscale image with alpha channel (four bytes per pixel).
has converted into grayscale image with an alpha channel (four bytes per pixel).
- Numbers in the table mean a duration of image resizing in milliseconds.
- The `image` crate does not support multiplying and dividing by alpha channel.
- The `resize` crate does not support this pixel format.
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
| libvips | 12.53 | 80.02 | 134.11 | 231.61 | 328.41 |
| fir rust | 0.19 | 25.16 | 32.76 | 51.72 | 70.89 |
| fir sse4.1 | 0.19 | 14.92 | 21.27 | 33.49 | 45.88 |
| fir avx2 | 0.19 | 11.72 | 15.02 | 21.87 | 29.07 |
| libvips | 12.15 | 79.49 | 133.83 | 232.22 | 327.14 |
| fir rust | 0.19 | 31.57 | 40.69 | 62.09 | 84.65 |
| fir sse4.1 | 0.19 | 14.86 | 21.03 | 33.10 | 45.59 |
| fir avx2 | 0.19 | 11.57 | 14.84 | 21.66 | 28.89 |
<!-- bench_compare_la16 end -->
@@ -226,18 +226,18 @@ Pipeline:
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
| image | 24.16 | - | 52.33 | 82.88 | 111.10 |
| resize | 4.99 | 8.85 | 13.38 | 30.00 | 45.71 |
| libvips | 7.37 | 25.94 | 20.30 | 40.97 | 70.76 |
| fir rust | 0.18 | 9.56 | 15.05 | 29.59 | 52.03 |
| fir sse4.1 | 0.18 | 5.03 | 7.30 | 11.55 | 16.90 |
| fir avx2 | 0.18 | 4.65 | 5.41 | 7.14 | 10.78 |
| image | 24.45 | - | 49.71 | 78.69 | 105.12 |
| resize | 5.02 | 8.82 | 13.42 | 29.90 | 45.04 |
| libvips | 6.48 | 25.81 | 21.50 | 42.64 | 70.48 |
| fir rust | 0.19 | 9.60 | 15.08 | 29.70 | 51.94 |
| fir sse4.1 | 0.19 | 5.05 | 7.28 | 11.54 | 16.87 |
| fir avx2 | 0.19 | 4.64 | 5.52 | 7.12 | 10.63 |
<!-- bench_compare_l32f end -->
<!-- bench_compare_la32f start -->
### Resize LA32F (luma with alpha channel) image (F32x2) 4928x3279 => 852x567
### Resize LA-F32 (luma with alpha channel) image (F32x2) 4928x3279 => 852x567
Pipeline:
@@ -245,17 +245,17 @@ Pipeline:
- Source image
[nasa-4928x3279-rgba.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279-rgba.png)
has converted into grayscale image with alpha channel (two `f32` values per pixel).
has converted into grayscale image with an alpha channel (two `f32` values per pixel).
- Numbers in the table mean a duration of image resizing in milliseconds.
- The `image` crate does not support multiplying and dividing by alpha channel.
- The `resize` crate does not support this pixel format.
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
| libvips | 11.85 | 70.31 | 101.80 | 177.59 | 254.22 |
| fir rust | 0.38 | 21.26 | 28.82 | 47.50 | 70.34 |
| fir sse4.1 | 0.38 | 16.23 | 20.91 | 30.35 | 40.12 |
| fir avx2 | 0.38 | 15.05 | 17.18 | 22.47 | 27.89 |
| libvips | 11.97 | 69.97 | 101.99 | 177.48 | 252.49 |
| fir rust | 0.39 | 21.50 | 29.51 | 47.46 | 70.36 |
| fir sse4.1 | 0.39 | 16.46 | 21.12 | 30.52 | 40.27 |
| fir avx2 | 0.39 | 15.32 | 17.33 | 22.63 | 28.11 |
<!-- bench_compare_la32f end -->
@@ -271,12 +271,14 @@ Pipeline:
has converted into RGB32F image.
- Numbers in the table mean a duration of image resizing in milliseconds.
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|----------|:-------:|:-----:|:--------:|:-------:|:--------:|
| image | 26.52 | - | 62.98 | 105.56 | 147.74 |
| resize | 8.76 | 14.03 | 23.62 | 47.69 | 70.36 |
| libvips | 11.91 | 60.28 | 53.55 | 112.62 | 200.59 |
| fir rust | 0.87 | 17.12 | 27.34 | 51.51 | 75.67 |
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
| image | 26.87 | - | 64.40 | 106.65 | 147.88 |
| resize | 8.89 | 14.23 | 23.78 | 47.82 | 70.56 |
| libvips | 11.71 | 60.78 | 53.26 | 114.11 | 193.66 |
| fir rust | 0.88 | 16.43 | 26.93 | 50.27 | 74.94 |
| fir sse4.1 | 0.88 | 12.63 | 19.38 | 32.27 | 46.71 |
| fir avx2 | 0.88 | 10.95 | 13.96 | 20.31 | 29.14 |
<!-- bench_compare_rgb32f end -->
@@ -296,9 +298,11 @@ Pipeline:
- The `resize` crate does not support multiplying and dividing by alpha channel
for this pixel format.
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|----------|:-------:|:------:|:--------:|:-------:|:--------:|
| libvips | 23.29 | 111.87 | 140.37 | 251.12 | 381.52 |
| fir rust | 0.98 | 35.65 | 45.42 | 70.24 | 92.90 |
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|------------|:-------:|:------:|:--------:|:-------:|:--------:|
| libvips | 23.22 | 111.20 | 140.22 | 250.45 | 386.72 |
| fir rust | 1.06 | 36.13 | 45.93 | 70.53 | 93.30 |
| fir sse4.1 | 1.06 | 32.10 | 40.35 | 58.59 | 77.88 |
| fir avx2 | 1.06 | 29.64 | 31.75 | 41.44 | 51.22 |
<!-- bench_compare_rgba32f end -->
+91
View File
@@ -0,0 +1,91 @@
use std::arch::x86_64::*;
use crate::convolution::{Coefficients, CoefficientsChunk};
use crate::pixels::F32x4;
use crate::{simd_utils, ImageView, ImageViewMut};
#[inline]
pub(crate) fn horiz_convolution(
src_view: &impl ImageView<Pixel = F32x4>,
dst_view: &mut impl ImageViewMut<Pixel = F32x4>,
offset: u32,
coeffs: Coefficients,
) {
let coefficients_chunks = coeffs.get_chunks();
let dst_height = dst_view.height();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_rows(src_rows, dst_rows, &coefficients_chunks);
}
}
let yy = dst_height - dst_height % 4;
let src_rows = src_view.iter_rows(yy + offset);
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_rows([src_row], [dst_row], &coefficients_chunks);
}
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - length of all rows in src_rows must be equal
/// - length of all rows in dst_rows must be equal
/// - coefficients_chunks.len() == dst_rows.0.len()
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "avx2")]
unsafe fn horiz_convolution_rows<const ROWS_COUNT: usize>(
src_rows: [&[F32x4]; ROWS_COUNT],
dst_rows: [&mut [F32x4]; ROWS_COUNT],
coefficients_chunks: &[CoefficientsChunk],
) {
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
let mut rgba_sums = [_mm256_set1_pd(0.); ROWS_COUNT];
let mut coeffs = coeffs_chunk.values;
let coeffs_by_2 = coeffs.chunks_exact(2);
coeffs = coeffs_by_2.remainder();
for k in coeffs_by_2 {
let coeff0_f64x4 = _mm256_set1_pd(k[0]);
let coeff1_f64x4 = _mm256_set1_pd(k[1]);
for r in 0..ROWS_COUNT {
let pixel01 = simd_utils::loadu_ps256(src_rows[r], x);
let pixel0_f64x4 = _mm256_cvtps_pd(_mm256_extractf128_ps::<0>(pixel01));
rgba_sums[r] =
_mm256_add_pd(rgba_sums[r], _mm256_mul_pd(pixel0_f64x4, coeff0_f64x4));
let pixels1_f64x4 = _mm256_cvtps_pd(_mm256_extractf128_ps::<1>(pixel01));
rgba_sums[r] =
_mm256_add_pd(rgba_sums[r], _mm256_mul_pd(pixels1_f64x4, coeff1_f64x4));
}
x += 2;
}
if let Some(&k) = coeffs.first() {
let coeff0_f64x4 = _mm256_set1_pd(k);
for r in 0..ROWS_COUNT {
let pixel0 = simd_utils::loadu_ps(src_rows[r], x);
let pixel0_f64x4 = _mm256_cvtps_pd(pixel0);
rgba_sums[r] =
_mm256_add_pd(rgba_sums[r], _mm256_mul_pd(pixel0_f64x4, coeff0_f64x4));
}
}
for r in 0..ROWS_COUNT {
let dst_pixel = dst_rows[r].get_unchecked_mut(dst_x);
let rgba_f32x4 = _mm256_cvtpd_ps(rgba_sums[r]);
_mm_storeu_ps(dst_pixel.0.as_mut_ptr(), rgba_f32x4);
}
}
}
+12 -12
View File
@@ -5,13 +5,13 @@ use crate::{ImageView, ImageViewMut};
use super::{Coefficients, Convolution};
// #[cfg(target_arch = "x86_64")]
// mod avx2;
#[cfg(target_arch = "x86_64")]
mod avx2;
mod native;
// #[cfg(target_arch = "aarch64")]
// mod neon;
// #[cfg(target_arch = "x86_64")]
// mod sse4;
#[cfg(target_arch = "x86_64")]
mod sse4;
// #[cfg(target_arch = "wasm32")]
// mod wasm32;
@@ -24,14 +24,14 @@ impl Convolution for F32x4 {
cpu_extensions: CpuExtensions,
) {
match cpu_extensions {
// #[cfg(target_arch = "x86_64")]
// CpuExtensions::Avx2 => avx2::horiz_convolution(src_view, dst_view, offset, coeffs),
// #[cfg(target_arch = "x86_64")]
// CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, coeffs),
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
#[cfg(target_arch = "x86_64")]
CpuExtensions::Avx2 => avx2::horiz_convolution(src_view, dst_view, offset, coeffs),
#[cfg(target_arch = "x86_64")]
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, coeffs),
// #[cfg(target_arch = "aarch64")]
// CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, coeffs),
// #[cfg(target_arch = "wasm32")]
// CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
_ => native::horiz_convolution(src_view, dst_view, offset, coeffs),
}
}
+80
View File
@@ -0,0 +1,80 @@
use std::arch::x86_64::*;
use crate::convolution::{Coefficients, CoefficientsChunk};
use crate::pixels::F32x4;
use crate::{simd_utils, ImageView, ImageViewMut};
#[inline]
pub(crate) fn horiz_convolution(
src_view: &impl ImageView<Pixel = F32x4>,
dst_view: &mut impl ImageViewMut<Pixel = F32x4>,
offset: u32,
coeffs: Coefficients,
) {
let coefficients_chunks = coeffs.get_chunks();
let dst_height = dst_view.height();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_rows(src_rows, dst_rows, &coefficients_chunks);
}
}
let yy = dst_height - dst_height % 4;
let src_rows = src_view.iter_rows(yy + offset);
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_rows([src_row], [dst_row], &coefficients_chunks);
}
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - length of all rows in src_rows must be equal
/// - length of all rows in dst_rows must be equal
/// - coefficients_chunks.len() == dst_rows.0.len()
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "sse4.1")]
unsafe fn horiz_convolution_rows<const ROWS_COUNT: usize>(
src_rows: [&[F32x4]; ROWS_COUNT],
dst_rows: [&mut [F32x4]; ROWS_COUNT],
coefficients_chunks: &[CoefficientsChunk],
) {
let mut rg_buf = [0f64; 2];
let mut ba_buf = [0f64; 2];
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
let mut rg_sums = [_mm_set1_pd(0.); ROWS_COUNT];
let mut ba_sums = [_mm_set1_pd(0.); ROWS_COUNT];
for &k in coeffs_chunk.values {
let coeffs_f64x2 = _mm_set1_pd(k);
for r in 0..ROWS_COUNT {
let pixel = simd_utils::loadu_ps(src_rows[r], x);
let rg_f64x2 = _mm_cvtps_pd(pixel);
rg_sums[r] = _mm_add_pd(rg_sums[r], _mm_mul_pd(rg_f64x2, coeffs_f64x2));
let ba_f64x2 = _mm_cvtps_pd(_mm_movehl_ps(pixel, pixel));
ba_sums[r] = _mm_add_pd(ba_sums[r], _mm_mul_pd(ba_f64x2, coeffs_f64x2));
}
x += 1;
}
for i in 0..ROWS_COUNT {
_mm_storeu_pd(rg_buf.as_mut_ptr(), rg_sums[i]);
_mm_storeu_pd(ba_buf.as_mut_ptr(), ba_sums[i]);
let dst_pixel = dst_rows[i].get_unchecked_mut(dst_x);
dst_pixel.0 = [
rg_buf[0] as f32,
rg_buf[1] as f32,
ba_buf[0] as f32,
ba_buf[1] as f32,
];
}
}
}
+1 -15
View File
@@ -26,21 +26,7 @@ pub(crate) fn vert_convolution<T>(
let mut x_src = src_x_initial;
let dst_components = T::components_mut(dst_row);
let (_, dst_chunks, tail) = unsafe { dst_components.align_to_mut::<[u8; 32]>() };
x_src = convolution_by_chunks(
src_image,
&normalizer,
initial,
dst_chunks,
x_src,
first_y_src,
ks,
);
if tail.is_empty() {
continue;
}
let (_, dst_chunks, tail) = unsafe { tail.align_to_mut::<[u8; 16]>() };
let (_, dst_chunks, tail) = unsafe { dst_components.align_to_mut::<[u8; 16]>() };
x_src = convolution_by_chunks(
src_image,
&normalizer,