- Added support of the new type of pixels - U8.

- Added benchmarks for resizing of U8 image.
This commit is contained in:
Kirill Kuzminykh
2021-10-23 12:10:33 +03:00
parent 8b8fa0381b
commit 61b94a05fc
49 changed files with 2938 additions and 2128 deletions
+9
View File
@@ -1,3 +1,12 @@
## [Unreleased] - ReleaseDate
- Added support of new type of pixels `U8` (without forced SIMD).
- Breaking changes:
- ``ImageData`` renamed into ``Image``.
- ``SrcImageView`` and ``DstImageView`` replaced by ``ImageView``
and ``ImageViewMut``.
- Method ``Resizer.resize()`` now returns ``Result<(), DifferentTypesOfPixelsError>``.
## [0.3.1] - 2021-10-09
- Added support of compilation for architectures other than x86_64.
Generated
+63 -63
View File
@@ -22,9 +22,9 @@ checksum = "739f4a8db6605981345c5654f3a85b056ce52f37a39d34da03f25bf2151ea16e"
[[package]]
name = "ahash"
version = "0.7.4"
version = "0.7.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "43bb833f0bf979d8475d38fbf09ed3b8a55e1885fe93ad3f93239fc6a4f17b98"
checksum = "fcb51a0695d8f838b1ee009b3fbf66bda078cd64590202a864a8f3e8c4315c47"
dependencies = [
"getrandom",
"once_cell",
@@ -33,15 +33,15 @@ dependencies = [
[[package]]
name = "anyhow"
version = "1.0.43"
version = "1.0.44"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "28ae2b3dec75a406790005a200b1bd89785afc02517a00ca99ecfe093ee9e6cf"
checksum = "61604a8f862e1d5c3229fdd78f8b02c68dcf73a4c4b05fd636d12240aaa242c1"
[[package]]
name = "argh"
version = "0.1.5"
version = "0.1.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "2e7317a549bc17c5278d9e72bb6e62c6aa801ac2567048e39ebc1c194249323e"
checksum = "f023c76cd7975f9969f8e29f0e461decbdc7f51048ce43427107a3d192f1c9bf"
dependencies = [
"argh_derive",
"argh_shared",
@@ -49,9 +49,9 @@ dependencies = [
[[package]]
name = "argh_derive"
version = "0.1.5"
version = "0.1.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "60949c42375351e9442e354434b0cba2ac402c1237edf673cac3a4bf983b8d3c"
checksum = "48ad219abc0c06ca788aface2e3a1970587e3413ab70acd20e54b6ec524c1f8f"
dependencies = [
"argh_shared",
"heck",
@@ -62,9 +62,9 @@ dependencies = [
[[package]]
name = "argh_shared"
version = "0.1.5"
version = "0.1.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8a61eb019cb8f415d162cb9f12130ee6bbe9168b7d953c17f4ad049e4051ca00"
checksum = "38de00daab4eac7d753e97697066238d67ce9d7e2d823ab4f72fe14af29f3f33"
[[package]]
name = "autocfg"
@@ -86,9 +86,9 @@ checksum = "bef38d45163c2f1dde094a7dfd33ccf595c92905c8f8f4fdc18d06fb1037718a"
[[package]]
name = "bstr"
version = "0.2.16"
version = "0.2.17"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "90682c8d613ad3373e66de8c6411e0ae2ab2571e879d2efbf73558cc66f21279"
checksum = "ba3569f383e8f1598449f1a423e72e99569137b47740b1da11ef19af3d5c3223"
dependencies = [
"lazy_static",
"memchr",
@@ -110,9 +110,9 @@ checksum = "14c189c53d098945499cdfa7ecc63567cf3886b3332b312a5b4585d8d3a6a610"
[[package]]
name = "cc"
version = "1.0.69"
version = "1.0.71"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e70cc2f62c6ce1868963827bd677764c62d07c3d9a3e1fb1177ee1a9ab199eb2"
checksum = "79c2681d6594606957bbb8631c4b90a7fcaaa72cdb714743a437b156d6a7eedd"
dependencies = [
"jobserver",
]
@@ -390,9 +390,9 @@ dependencies = [
[[package]]
name = "gif"
version = "0.11.2"
version = "0.11.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "5a668f699973d0f573d15749b7002a9ac9e1f9c6b220e7b165601334c173d8de"
checksum = "c3a7187e78088aead22ceedeee99779455b23fc231fe13ec443f99bb71694e5b"
dependencies = [
"color_quant",
"weezl",
@@ -400,9 +400,9 @@ dependencies = [
[[package]]
name = "git2"
version = "0.13.21"
version = "0.13.23"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "659cd14835e75b64d9dba5b660463506763cf0aa6cb640aeeb0e98d841093490"
checksum = "2a8057932925d3a9d9e4434ea016570d37420ddb1ceed45a174d577f24ed6700"
dependencies = [
"bitflags",
"libc",
@@ -449,7 +449,7 @@ version = "0.11.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ab5ef0d4909ef3724cc8cce6ccc8572c5c817592e9285f5464f8e86f8bd3726e"
dependencies = [
"ahash 0.7.4",
"ahash 0.7.6",
]
[[package]]
@@ -511,9 +511,9 @@ dependencies = [
[[package]]
name = "instant"
version = "0.1.10"
version = "0.1.12"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "bee0328b1209d157ef001c94dd85b4f8f64139adb0eac2659f4b08382b2f474d"
checksum = "7a5bbe824c507c5da5956355e86a746d82e0e1464f65d862cc5e71da70e94b2c"
dependencies = [
"cfg-if",
]
@@ -550,15 +550,15 @@ checksum = "e2abad23fbc42b3700f2f279844dc832adb2b2eb069b2df918f455c4e18cc646"
[[package]]
name = "libc"
version = "0.2.101"
version = "0.2.104"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "3cb00336871be5ed2c8ed44b60ae9959dc5b9f08539422ed43f09e34ecaeba21"
checksum = "7b2f96d100e1cf1929e7719b7edb3b90ab5298072638fccd77be9ce942ecdfce"
[[package]]
name = "libgit2-sys"
version = "0.12.22+1.1.0"
version = "0.12.24+1.3.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "89c53ac117c44f7042ad8d8f5681378dfbc6010e49ec2c0d1f11dfedc7a4a1c3"
checksum = "ddbd6021eef06fb289a8f54b3c2acfdd85ff2a585dfbb24b8576325373d2152c"
dependencies = [
"cc",
"libc",
@@ -591,9 +591,9 @@ dependencies = [
[[package]]
name = "lock_api"
version = "0.4.4"
version = "0.4.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "0382880606dff6d15c9476c416d18690b72742aa7b605bb6dd6ec9030fbf07eb"
checksum = "712a4d093c9976e24e7dbca41db895dabcbac38eb5f4045393d17a95bdfb1109"
dependencies = [
"scopeguard",
]
@@ -658,9 +658,9 @@ dependencies = [
[[package]]
name = "mio"
version = "0.7.13"
version = "0.7.14"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8c2bdb6314ec10835cd3293dd268473a835c02b7b352e788be788b3c6ca6bb16"
checksum = "8067b404fe97c70829f082dec8bcf4f71225d7eaea1d8645349cb76fa06205cc"
dependencies = [
"libc",
"log",
@@ -756,9 +756,9 @@ dependencies = [
[[package]]
name = "parking_lot"
version = "0.11.1"
version = "0.11.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "6d7744ac029df22dca6284efe4e898991d28e3085c706c972bcd7da4a27a15eb"
checksum = "7d17b78036a60663b797adeaee46f5c9dfebb86948d1255007a1d6be0271ff99"
dependencies = [
"instant",
"lock_api",
@@ -767,9 +767,9 @@ dependencies = [
[[package]]
name = "parking_lot_core"
version = "0.8.3"
version = "0.8.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "fa7a782938e745763fe6907fc6ba86946d72f49fe7e21de074e08128a99fb018"
checksum = "d76e8e1493bcac0d2766c42737f34458f1c8c50c0d23bcb24ea953affb273216"
dependencies = [
"cfg-if",
"instant",
@@ -781,9 +781,9 @@ dependencies = [
[[package]]
name = "pathdiff"
version = "0.2.0"
version = "0.2.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "877630b3de15c0b64cc52f659345724fbf6bdad9bd9566699fc53688f3c34a34"
checksum = "8835116a5c179084a830efb3adc117ab007512b535bc1a21c991d3b32a6b44dd"
[[package]]
name = "percent-encoding"
@@ -793,9 +793,9 @@ checksum = "d4fd5641d01c8f18a23da7b6fe29298ff4b55afcccdf78973b24cf3175fee32e"
[[package]]
name = "pkg-config"
version = "0.3.19"
version = "0.3.20"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "3831453b3449ceb48b6d9c7ad7c96d5ea673e9b470a1dc578c2ce6521230884c"
checksum = "7c9b1041b4387893b91ee6746cddfc28516aff326a3519fb2adf820932c5e6cb"
[[package]]
name = "png"
@@ -811,24 +811,24 @@ dependencies = [
[[package]]
name = "ppv-lite86"
version = "0.2.10"
version = "0.2.14"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ac74c624d6b2d21f425f752262f42188365d7b8ff1aff74c82e45136510a4857"
checksum = "c3ca011bd0129ff4ae15cd04c4eef202cadf6c51c21e47aba319b4e0501db741"
[[package]]
name = "proc-macro2"
version = "1.0.28"
version = "1.0.30"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "5c7ed8b8c7b886ea3ed7dde405212185f423ab44682667c8c6dd14aa1d9f6612"
checksum = "edc3358ebc67bc8b7fa0c007f945b0b18226f78437d61bec735a9eb96b61ee70"
dependencies = [
"unicode-xid",
]
[[package]]
name = "quote"
version = "1.0.9"
version = "1.0.10"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c3d0b9745dc2debf507c8422de05d7226cc1f0644216dfdfead988f9b1ab32a7"
checksum = "38bc8cc6a5f2e3655e0899c1b848643b2562f853f114bfec7be120678e3ace05"
dependencies = [
"proc-macro2",
]
@@ -986,18 +986,18 @@ checksum = "d29ab0c6d3fc0ee92fe66e2d99f700eab17a8d57d1c1d3b748380fb20baa78cd"
[[package]]
name = "serde"
version = "1.0.129"
version = "1.0.130"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d1f72836d2aa753853178eda473a3b9d8e4eefdaf20523b919677e6de489f8f1"
checksum = "f12d06de37cf59146fbdecab66aa99f9fe4f78722e3607577a5375d66bd0c913"
dependencies = [
"serde_derive",
]
[[package]]
name = "serde_derive"
version = "1.0.129"
version = "1.0.130"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e57ae87ad533d9a56427558b516d0adac283614e347abf85b0dc0cbbf0a249f3"
checksum = "d7bc1a1ab1961464eae040d96713baa5a724a8152c1222492465b54322ec508b"
dependencies = [
"proc-macro2",
"quote",
@@ -1006,9 +1006,9 @@ dependencies = [
[[package]]
name = "serde_json"
version = "1.0.66"
version = "1.0.68"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "336b10da19a12ad094b59d870ebde26a45402e5b470add4b5fd03c5048a32127"
checksum = "0f690853975602e1bfe1ccbf50504d67174e3bcf340f23b5ea9992e0587a52d8"
dependencies = [
"itoa",
"ryu",
@@ -1037,9 +1037,9 @@ dependencies = [
[[package]]
name = "smallvec"
version = "1.6.1"
version = "1.7.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "fe0f37c9e8f3c5a4a66ad655a93c74daac4ad00c441533bf5c6e7990bb42604e"
checksum = "1ecab6c735a6bb4139c0caafd0cc3635748bbb3acf4550e8138122099251f309"
[[package]]
name = "svg"
@@ -1049,9 +1049,9 @@ checksum = "3bdb25a4593d6656239319426f4025f7a658157e25e89f0e0319d7516d46042d"
[[package]]
name = "syn"
version = "1.0.75"
version = "1.0.80"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b7f58f7e8eaa0009c5fec437aabf511bd9933e4b2d7407bd05273c01a8906ea7"
checksum = "d010a1623fbd906d51d650a9916aaefc05ffa0e4053ff7fe601167f3e715d194"
dependencies = [
"proc-macro2",
"quote",
@@ -1088,18 +1088,18 @@ dependencies = [
[[package]]
name = "thiserror"
version = "1.0.26"
version = "1.0.30"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "93119e4feac1cbe6c798c34d3a53ea0026b0b1de6a120deef895137c0529bfe2"
checksum = "854babe52e4df1653706b98fcfc05843010039b406875930a70e4d9644e5c417"
dependencies = [
"thiserror-impl",
]
[[package]]
name = "thiserror-impl"
version = "1.0.26"
version = "1.0.30"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "060d69a0afe7796bf42e9e2ff91f5ee691fb15c53d38b4b62a9a53eb23164745"
checksum = "aa32fd3f627f367fe16f893e2597ae3c05020f8bba2666a4e6ea73d377e5714b"
dependencies = [
"proc-macro2",
"quote",
@@ -1129,9 +1129,9 @@ dependencies = [
[[package]]
name = "tinyvec"
version = "1.3.1"
version = "1.5.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "848a1e1181b9f6753b5e96a092749e29b11d19ede67dfbbd6c7dc7e0f49b5338"
checksum = "f83b2a3d4d9091d0abd7eba4dc2710b1718583bd4d8992e2190720ea38f391f7"
dependencies = [
"tinyvec_macros",
]
@@ -1144,9 +1144,9 @@ checksum = "cda74da7e1a664f795bb1f8a87ec406fb89a02522cf6e50620d016add6dbbf5c"
[[package]]
name = "unicode-bidi"
version = "0.3.6"
version = "0.3.7"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "246f4c42e67e7a4e3c6106ff716a5d067d4132a642840b242e357e468a2a0085"
checksum = "1a01404663e3db436ed2746d9fefef640d868edae3cceb81c3b8d5732fda678f"
[[package]]
name = "unicode-normalization"
@@ -1165,9 +1165,9 @@ checksum = "8895849a949e7845e06bd6dc1aa51731a103c42707010a5b591c0038fb73385b"
[[package]]
name = "unicode-width"
version = "0.1.8"
version = "0.1.9"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9337591893a19b88d8d87f2cec1e73fad5cdfd10e5a6f349f498ad6ea2ffb1e3"
checksum = "3ed742d4ea2bd1176e236172c8429aaf54486e7ac098db29ffe6529e0ce50973"
[[package]]
name = "unicode-xid"
+6 -1
View File
@@ -15,7 +15,7 @@ exclude = ["/data"]
[dependencies]
num-traits = "0.2.14"
thiserror = "1.0.26"
thiserror = "1.0.30"
[dev-dependencies]
@@ -45,6 +45,11 @@ name = "bench_compare_rgba"
harness = false
[[bench]]
name = "bench_compare_u8"
harness = false
[profile.dev.package.'*']
opt-level = 3
+43 -47
View File
@@ -6,25 +6,27 @@ Rust library for fast image resizing with using of SIMD instructions.
Supported pixel formats and available optimisations:
- `U8x4` - four `u8` components per pixel:
- native Rust-code without forced SIMD
- SSE4.1
- AVX2
- native Rust-code without forced SIMD
- SSE4.1
- AVX2
- `I32` - one `i32` component per pixel:
- native Rust-code without forced SIMD
- native Rust-code without forced SIMD
- `F32` - one `f32` component per pixel:
- native Rust-code without forced SIMD
- native Rust-code without forced SIMD
- `U8` - one `u8` component per pixel:
- native Rust-code without forced SIMD
## Benchmarks
Environment:
- CPU: Intel(R) Core(TM) i7-6700K CPU @ 4.00GHz
- RAM: DDR4 3000 MHz
- Ubuntu 20.04 (linux 5.8)
- Rust 1.54
- fast_image_resize = "0.3"
- RAM: DDR4 3000 MHz
- Ubuntu 20.04 (linux 5.11)
- Rust 1.56
- fast_image_resize = "0.4"
- glassbench = "0.3.0"
Other Rust libraries used to compare of resizing speed:
Other Rust libraries used to compare of resizing speed:
- image = "0.23.14" (<https://crates.io/crates/image>)
- resize = "0.7.2" (<https://crates.io/crates/resize>)
@@ -36,7 +38,7 @@ Resize algorithms:
### Resize RGB image 4928x3279 => 852x567
Pipeline:
Pipeline:
`src_image => resize => dst_image`
@@ -45,25 +47,15 @@ Pipeline:
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|------------|:-------:|:--------:|:----------:|:--------:|
| image | 106.303 | 199.011 | 291.423 | 382.793 |
| resize | 16.099 | 71.947 | 132.481 | 205.961 |
| fir rust | 0.478 | 55.376 | 84.567 | 117.269 |
| fir sse4.1 | - | 11.847 | 17.823 | 25.459 |
| fir avx2 | - | 8.598 | 11.864 | 17.549 |
Compiled with `rustflags = ["-C", "target-cpu=native"]`
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|------------|:-------:|:--------:|:----------:|:--------:|
| image | 91.787 | 182.502 | 282.821 | 385.492 |
| resize | 15.908 | 62.011 | 114.741 | 167.815 |
| fir rust | 0.471 | 56.330 | 61.589 | 85.443 |
| fir sse4.1 | - | 11.204 | 16.510 | 23.442 |
| fir avx2 | - | 8.402 | 11.550 | 16.987 |
| image | 107.950 | 198.726 | 288.085 | 380.573 |
| resize | 15.573 | 72.009 | 132.181 | 192.426 |
| fir rust | 0.473 | 56.476 | 86.983 | 120.115 |
| fir sse4.1 | - | 11.856 | 17.748 | 25.288 |
| fir avx2 | - | 9.052 | 12.027 | 17.477 |
### Resize RGBA image 4928x3279 => 852x567
Pipeline:
Pipeline:
`src_image => multiply by alpha => resize => divide by alpha => dst_image`
@@ -72,21 +64,27 @@ Pipeline:
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|------------|:-------:|:--------:|:----------:|:--------:|
| image | 102.564 | 205.981 | 309.851 | 423.252 |
| resize | 19.003 | 92.075 | 169.876 | 247.993 |
| fir rust | 13.619 | 67.097 | 97.157 | 125.865 |
| fir sse4.1 | 12.182 | 23.555 | 29.506 | 37.118 |
| fir avx2 | 6.931 | 15.052 | 18.391 | 23.812 |
| image | 107.165 | 191.338 | 281.272 | 372.183 |
| resize | 18.099 | 79.512 | 149.128 | 225.173 |
| fir rust | 13.265 | 69.358 | 99.794 | 132.545 |
| fir sse4.1 | 11.739 | 23.080 | 29.013 | 36.556 |
| fir avx2 | 6.958 | 15.590 | 18.610 | 24.219 |
Compiled with `rustflags = ["-C", "target-cpu=native"]`
### Resize gray image (U8) 4928x3279 => 852x567
Pipeline:
`src_image => resize => dst_image`
- Source image [nasa-4928x3279.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279.png)
has converted into grayscale image with one byte per pixel.
- Numbers in table is mean duration of image resizing in milliseconds.
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|------------|:-------:|:--------:|:----------:|:--------:|
| image | 90.421 | 193.379 | 306.635 | 422.674 |
| resize | 19.572 | 69.091 | 130.327 | 192.235 |
| fir rust | 9.987 | 69.005 | 76.551 | 103.646 |
| fir sse4.1 | 7.860 | 18.375 | 23.601 | 30.603 |
| fir avx2 | 6.897 | 14.907 | 18.034 | 23.677 |
| image | 96.792 | 143.195 | 188.121 | 240.504 |
| resize | 10.582 | 26.577 | 53.537 | 81.599 |
| fir rust | 0.203 | 24.832 | 30.962 | 46.958 |
## Examples
@@ -111,7 +109,7 @@ fn resize_image_example() {
.unwrap();
let width = NonZeroU32::new(img.width()).unwrap();
let height = NonZeroU32::new(img.height()).unwrap();
let mut src_image = fr::ImageData::from_vec_u8(
let mut src_image = fr::Image::from_vec_u8(
width,
height,
img.to_rgba8().into_raw(),
@@ -123,31 +121,30 @@ fn resize_image_example() {
let alpha_mul_div: fr::MulDiv = Default::default();
// Multiple RGB channels of source image by alpha channel
alpha_mul_div
.multiply_alpha_inplace(&mut src_image.dst_view())
.multiply_alpha_inplace(&mut src_image.view_mut())
.unwrap();
// Create wrapper that own data of destination image
let dst_width = NonZeroU32::new(1024).unwrap();
let dst_height = NonZeroU32::new(768).unwrap();
let mut dst_image = fr::ImageData::new(dst_width, dst_height, src_image.pixel_type());
let mut dst_image = fr::Image::new(dst_width, dst_height, src_image.pixel_type());
// Get mutable view of destination image data
let mut dst_view = dst_image.dst_view();
let mut dst_view = dst_image.view_mut();
// Create Resizer instance and resize source image
// into buffer of destination image
let mut resizer = fr::Resizer::new(fr::ResizeAlg::Convolution(fr::FilterType::Lanczos3));
resizer.resize(&src_image.src_view(), &mut dst_view);
resizer.resize(&src_image.view(), &mut dst_view).unwrap();
// Divide RGB channels of destination image by alpha
alpha_mul_div.divide_alpha_inplace(&mut dst_view).unwrap();
// Write destination image as PNG-file
let mut result_buf = BufWriter::new(Vec::new());
let encoder = PngEncoder::new(&mut result_buf);
encoder
PngEncoder::new(&mut result_buf)
.encode(
dst_image.get_buffer(),
dst_image.buffer(),
dst_width.get(),
dst_height.get(),
ColorType::Rgba8,
@@ -166,6 +163,5 @@ fn main() {
unsafe {
resizer.set_cpu_extensions(fr::CpuExtensions::Sse4_1);
}
// ...
}
```
+22 -21
View File
@@ -2,7 +2,8 @@ use std::num::NonZeroU32;
use glassbench::*;
use fast_image_resize::{CpuExtensions, ImageData, MulDiv, PixelType};
use fast_image_resize::PixelType;
use fast_image_resize::{CpuExtensions, Image, MulDiv};
const fn p(r: u8, g: u8, b: u8, a: u8) -> u32 {
u32::from_le_bytes([r, g, b, a])
@@ -10,10 +11,10 @@ const fn p(r: u8, g: u8, b: u8, a: u8) -> u32 {
// Multiplies by alpha
fn get_src_image(width: NonZeroU32, height: NonZeroU32, pixel: u32) -> ImageData<'static> {
fn get_src_image(width: NonZeroU32, height: NonZeroU32, pixel: u32) -> Image<'static> {
let buf_size = (width.get() * height.get()) as usize;
let buffer = vec![pixel; buf_size];
ImageData::from_vec_u32(width, height, buffer, PixelType::U8x4).unwrap()
Image::from_vec_u32(width, height, buffer, PixelType::U8x4).unwrap()
}
#[cfg(target_arch = "x86_64")]
@@ -21,9 +22,9 @@ fn multiplies_alpha_avx2(bench: &mut Bench) {
let width = NonZeroU32::new(4096).unwrap();
let height = NonZeroU32::new(2048).unwrap();
let src_data = get_src_image(width, height, p(255, 128, 0, 128));
let mut dst_data = ImageData::new(width, height, PixelType::U8x4);
let src_view = src_data.src_view();
let mut dst_view = dst_data.dst_view();
let mut dst_data = Image::new(width, height, PixelType::U8x4);
let src_view = src_data.view();
let mut dst_view = dst_data.view_mut();
let mut alpha_mul_div: MulDiv = Default::default();
unsafe {
alpha_mul_div.set_cpu_extensions(CpuExtensions::Avx2);
@@ -43,9 +44,9 @@ fn multiplies_alpha_sse2(bench: &mut Bench) {
let width = NonZeroU32::new(4096).unwrap();
let height = NonZeroU32::new(2048).unwrap();
let src_data = get_src_image(width, height, p(255, 128, 0, 128));
let mut dst_data = ImageData::new(width, height, PixelType::U8x4);
let src_view = src_data.src_view();
let mut dst_view = dst_data.dst_view();
let mut dst_data = Image::new(width, height, PixelType::U8x4);
let src_view = src_data.view();
let mut dst_view = dst_data.view_mut();
let mut alpha_mul_div: MulDiv = Default::default();
unsafe {
alpha_mul_div.set_cpu_extensions(CpuExtensions::Sse2);
@@ -64,9 +65,9 @@ fn multiplies_alpha_native(bench: &mut Bench) {
let width = NonZeroU32::new(4096).unwrap();
let height = NonZeroU32::new(2048).unwrap();
let src_data = get_src_image(width, height, p(255, 128, 0, 128));
let mut dst_data = ImageData::new(width, height, PixelType::U8x4);
let src_view = src_data.src_view();
let mut dst_view = dst_data.dst_view();
let mut dst_data = Image::new(width, height, PixelType::U8x4);
let src_view = src_data.view();
let mut dst_view = dst_data.view_mut();
let mut alpha_mul_div: MulDiv = Default::default();
unsafe {
alpha_mul_div.set_cpu_extensions(CpuExtensions::None);
@@ -86,9 +87,9 @@ fn divides_alpha_avx2(bench: &mut Bench) {
let width = NonZeroU32::new(4096).unwrap();
let height = NonZeroU32::new(2048).unwrap();
let src_data = get_src_image(width, height, p(128, 64, 0, 128));
let mut dst_data = ImageData::new(width, height, PixelType::U8x4);
let src_view = src_data.src_view();
let mut dst_view = dst_data.dst_view();
let mut dst_data = Image::new(width, height, PixelType::U8x4);
let src_view = src_data.view();
let mut dst_view = dst_data.view_mut();
let mut alpha_mul_div: MulDiv = Default::default();
unsafe {
alpha_mul_div.set_cpu_extensions(CpuExtensions::Avx2);
@@ -108,9 +109,9 @@ fn divides_alpha_sse2(bench: &mut Bench) {
let width = NonZeroU32::new(4096).unwrap();
let height = NonZeroU32::new(2048).unwrap();
let src_data = get_src_image(width, height, p(128, 64, 0, 128));
let mut dst_data = ImageData::new(width, height, PixelType::U8x4);
let src_view = src_data.src_view();
let mut dst_view = dst_data.dst_view();
let mut dst_data = Image::new(width, height, PixelType::U8x4);
let src_view = src_data.view();
let mut dst_view = dst_data.view_mut();
let mut alpha_mul_div: MulDiv = Default::default();
unsafe {
alpha_mul_div.set_cpu_extensions(CpuExtensions::Sse2);
@@ -129,9 +130,9 @@ fn divides_alpha_native(bench: &mut Bench) {
let width = NonZeroU32::new(4096).unwrap();
let height = NonZeroU32::new(2048).unwrap();
let src_data = get_src_image(width, height, p(128, 64, 0, 128));
let mut dst_data = ImageData::new(width, height, PixelType::U8x4);
let src_view = src_data.src_view();
let mut dst_view = dst_data.dst_view();
let mut dst_data = Image::new(width, height, PixelType::U8x4);
let src_view = src_data.view();
let mut dst_view = dst_data.view_mut();
let mut alpha_mul_div: MulDiv = Default::default();
unsafe {
alpha_mul_div.set_cpu_extensions(CpuExtensions::None);
+6 -6
View File
@@ -5,7 +5,7 @@ use image::imageops;
use resize::Pixel::RGB8;
use rgb::{FromSlice, RGB};
use fast_image_resize::ImageData;
use fast_image_resize::Image;
use fast_image_resize::{CpuExtensions, FilterType, PixelType, ResizeAlg, Resizer};
mod utils;
@@ -72,16 +72,16 @@ pub fn bench_downscale_rgb(bench: &mut Bench) {
for (cpu_ext, ext_name) in cpu_ext_and_name {
for alg_name in alg_names {
let src_rgba_image = utils::get_big_rgba_image();
let src_image_data = ImageData::from_vec_u8(
let src_image_data = Image::from_vec_u8(
NonZeroU32::new(src_image.width()).unwrap(),
NonZeroU32::new(src_image.height()).unwrap(),
src_rgba_image.into_raw(),
PixelType::U8x4,
)
.unwrap();
let src_view = src_image_data.src_view();
let mut dst_image = ImageData::new(new_width, new_height, PixelType::U8x4);
let mut dst_view = dst_image.dst_view();
let src_view = src_image_data.view();
let mut dst_image = Image::new(new_width, new_height, PixelType::U8x4);
let mut dst_view = dst_image.view_mut();
let resize_alg = match alg_name {
"Nearest" => ResizeAlg::Nearest,
@@ -99,7 +99,7 @@ pub fn bench_downscale_rgb(bench: &mut Bench) {
bench.task(format!("fir {} - {}", ext_name, alg_name), |task| {
task.iter(|| {
fast_resizer.resize(&src_view, &mut dst_view);
fast_resizer.resize(&src_view, &mut dst_view).unwrap();
})
});
}
+10 -8
View File
@@ -7,7 +7,7 @@ use resize::Pixel::RGBA8;
use rgb::FromSlice;
use fast_image_resize::{CpuExtensions, FilterType, PixelType, ResizeAlg, Resizer};
use fast_image_resize::{ImageData, MulDiv};
use fast_image_resize::{Image, MulDiv};
mod utils;
@@ -79,21 +79,21 @@ pub fn bench_downscale_rgba(bench: &mut Bench) {
"Lanczos3" => ResizeAlg::Convolution(FilterType::Lanczos3),
_ => return,
};
let src_image_data = ImageData::from_vec_u8(
let src_image_data = Image::from_vec_u8(
NonZeroU32::new(src_image.width()).unwrap(),
NonZeroU32::new(src_image.height()).unwrap(),
src_image.as_raw().clone(),
PixelType::U8x4,
)
.unwrap();
let src_view = src_image_data.src_view();
let mut premultiplied_src_image = ImageData::new(
let src_view = src_image_data.view();
let mut premultiplied_src_image = Image::new(
NonZeroU32::new(src_image.width()).unwrap(),
NonZeroU32::new(src_image.height()).unwrap(),
PixelType::U8x4,
);
let mut dst_image = ImageData::new(new_width, new_height, PixelType::U8x4);
let mut dst_view = dst_image.dst_view();
let mut dst_image = Image::new(new_width, new_height, PixelType::U8x4);
let mut dst_view = dst_image.view_mut();
let mut mul_div = MulDiv::default();
let mut fast_resizer = Resizer::new(resize_alg);
@@ -107,9 +107,11 @@ pub fn bench_downscale_rgba(bench: &mut Bench) {
bench.task(format!("fir {} - {}", ext_name, alg_name), |task| {
task.iter(|| {
mul_div
.multiply_alpha(&src_view, &mut premultiplied_src_image.dst_view())
.multiply_alpha(&src_view, &mut premultiplied_src_image.view_mut())
.unwrap();
fast_resizer
.resize(&premultiplied_src_image.view(), &mut dst_view)
.unwrap();
fast_resizer.resize(&premultiplied_src_image.src_view(), &mut dst_view);
mul_div.divide_alpha_inplace(&mut dst_view).unwrap();
})
});
+107
View File
@@ -0,0 +1,107 @@
use std::num::NonZeroU32;
use glassbench::*;
use image::imageops;
use resize::Pixel::Gray8;
use rgb::alt::Gray;
use rgb::FromSlice;
use fast_image_resize::Image;
use fast_image_resize::{CpuExtensions, FilterType, PixelType, ResizeAlg, Resizer};
mod utils;
pub fn bench_downscale_u8(bench: &mut Bench) {
let src_image = utils::get_big_luma8_image();
let new_width = NonZeroU32::new(852).unwrap();
let new_height = NonZeroU32::new(567).unwrap();
let alg_names = ["Nearest", "Bilinear", "CatmullRom", "Lanczos3"];
// image crate
// https://crates.io/crates/image
for alg_name in alg_names {
let filter = match alg_name {
"Nearest" => imageops::Nearest,
"Bilinear" => imageops::Triangle,
"CatmullRom" => imageops::CatmullRom,
"Lanczos3" => imageops::Lanczos3,
_ => continue,
};
bench.task(format!("image - {}", alg_name), |task| {
task.iter(|| {
imageops::resize(&src_image, new_width.get(), new_height.get(), filter);
})
});
}
// resize crate
// https://crates.io/crates/resize
for alg_name in alg_names {
let resize_src_image = src_image.as_raw().as_gray();
let mut dst = vec![Gray(0u8); (new_width.get() * new_height.get()) as usize];
bench.task(format!("resize - {}", alg_name), |task| {
let filter = match alg_name {
"Nearest" => resize::Type::Point,
"Bilinear" => resize::Type::Triangle,
"CatmullRom" => resize::Type::Catrom,
"Lanczos3" => resize::Type::Lanczos3,
_ => return,
};
let mut resize = resize::new(
src_image.width() as usize,
src_image.height() as usize,
new_width.get() as usize,
new_height.get() as usize,
Gray8,
filter,
)
.unwrap();
task.iter(|| {
resize.resize(resize_src_image, &mut dst).unwrap();
})
});
}
// fast_image_resize crate;
let mut cpu_ext_and_name = vec![(CpuExtensions::None, "rust")];
for (cpu_ext, ext_name) in cpu_ext_and_name {
for alg_name in alg_names {
let src_rgba_image = utils::get_big_luma8_image();
let src_image_data = Image::from_vec_u8(
NonZeroU32::new(src_image.width()).unwrap(),
NonZeroU32::new(src_image.height()).unwrap(),
src_rgba_image.into_raw(),
PixelType::U8,
)
.unwrap();
let src_view = src_image_data.view();
let mut dst_image = Image::new(new_width, new_height, PixelType::U8);
let mut dst_view = dst_image.view_mut();
let resize_alg = match alg_name {
"Nearest" => ResizeAlg::Nearest,
"Bilinear" => ResizeAlg::Convolution(FilterType::Bilinear),
"CatmullRom" => ResizeAlg::Convolution(FilterType::CatmullRom),
"Lanczos3" => ResizeAlg::Convolution(FilterType::Lanczos3),
_ => return,
};
let mut fast_resizer = Resizer::new(resize_alg);
unsafe {
fast_resizer.reset_internal_buffers();
fast_resizer.set_cpu_extensions(cpu_ext);
}
bench.task(format!("fir {} - {}", ext_name, alg_name), |task| {
task.iter(|| {
fast_resizer.resize(&src_view, &mut dst_view).unwrap();
})
});
}
}
utils::print_md_table(bench);
}
glassbench!("Compare resize of U8 image", bench_downscale_u8,);
+96 -41
View File
@@ -2,7 +2,7 @@ use std::num::NonZeroU32;
use glassbench::*;
use fast_image_resize::ImageData;
use fast_image_resize::Image;
use fast_image_resize::{CpuExtensions, FilterType, PixelType, ResizeAlg, Resizer};
mod utils;
@@ -13,11 +13,11 @@ const NEW_HEIGHT: u32 = 567;
const NEW_BIG_WIDTH: u32 = 4928;
const NEW_BIG_HEIGHT: u32 = 3279;
fn get_big_source_image() -> ImageData<'static> {
fn get_big_source_image() -> Image<'static> {
let img = utils::get_big_rgba_image();
let width = img.width();
let height = img.height();
ImageData::from_vec_u8(
Image::from_vec_u8(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
img.into_raw(),
@@ -26,8 +26,8 @@ fn get_big_source_image() -> ImageData<'static> {
.unwrap()
}
fn get_big_i32_image() -> ImageData<'static> {
let img = utils::get_big_luma_image();
fn get_big_i32_image() -> Image<'static> {
let img = utils::get_big_luma16_image();
let img_data: Vec<u32> = img
.as_raw()
.iter()
@@ -35,7 +35,7 @@ fn get_big_i32_image() -> ImageData<'static> {
.collect();
let width = img.width();
let height = img.height();
ImageData::from_vec_u32(
Image::from_vec_u32(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
img_data,
@@ -44,11 +44,24 @@ fn get_big_i32_image() -> ImageData<'static> {
.unwrap()
}
fn get_small_source_image() -> ImageData<'static> {
fn get_big_u8_image() -> Image<'static> {
let img = utils::get_big_luma8_image();
let width = img.width();
let height = img.height();
Image::from_vec_u8(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
img.into_raw(),
PixelType::U8,
)
.unwrap()
}
fn get_small_source_image() -> Image<'static> {
let img = utils::get_small_rgba_image();
let width = img.width();
let height = img.height();
ImageData::from_vec_u8(
Image::from_vec_u8(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
img.into_raw(),
@@ -57,42 +70,42 @@ fn get_small_source_image() -> ImageData<'static> {
.unwrap()
}
fn nearest_wo_simd_bench(bench: &mut Bench) {
fn native_nearest_bench(bench: &mut Bench) {
let image = get_big_source_image();
let mut res_image = ImageData::new(
let mut res_image = Image::new(
NonZeroU32::new(NEW_WIDTH).unwrap(),
NonZeroU32::new(NEW_HEIGHT).unwrap(),
image.pixel_type(),
);
let src_image = image.src_view();
let mut dst_image = res_image.dst_view();
let src_image = image.view();
let mut dst_image = res_image.view_mut();
let mut resizer = Resizer::new(ResizeAlg::Nearest);
unsafe {
resizer.set_cpu_extensions(CpuExtensions::None);
}
bench.task("nearest wo SIMD", |task| {
task.iter(|| {
resizer.resize(&src_image, &mut dst_image);
resizer.resize(&src_image, &mut dst_image).unwrap();
})
});
}
fn lanczos3_wo_simd_bench(bench: &mut Bench) {
fn native_lanczos3_bench(bench: &mut Bench) {
let image = get_big_source_image();
let mut res_image = ImageData::new(
let mut res_image = Image::new(
NonZeroU32::new(NEW_WIDTH).unwrap(),
NonZeroU32::new(NEW_HEIGHT).unwrap(),
image.pixel_type(),
);
let src_image = image.src_view();
let mut dst_image = res_image.dst_view();
let src_image = image.view();
let mut dst_image = res_image.view_mut();
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
unsafe {
resizer.set_cpu_extensions(CpuExtensions::None);
}
bench.task("lanczos3 wo SIMD", |task| {
task.iter(|| {
resizer.resize(&src_image, &mut dst_image);
resizer.resize(&src_image, &mut dst_image).unwrap();
})
});
}
@@ -100,20 +113,20 @@ fn lanczos3_wo_simd_bench(bench: &mut Bench) {
#[cfg(target_arch = "x86_64")]
fn sse4_lanczos3_bench(bench: &mut Bench) {
let image = get_big_source_image();
let mut res_image = ImageData::new(
let mut res_image = Image::new(
NonZeroU32::new(NEW_WIDTH).unwrap(),
NonZeroU32::new(NEW_HEIGHT).unwrap(),
image.pixel_type(),
);
let src_image = image.src_view();
let mut dst_image = res_image.dst_view();
let src_image = image.view();
let mut dst_image = res_image.view_mut();
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
unsafe {
resizer.set_cpu_extensions(CpuExtensions::Sse4_1);
}
bench.task("sse4 lanczos3", |task| {
task.iter(|| {
resizer.resize(&src_image, &mut dst_image);
resizer.resize(&src_image, &mut dst_image).unwrap();
})
});
}
@@ -121,20 +134,20 @@ fn sse4_lanczos3_bench(bench: &mut Bench) {
#[cfg(target_arch = "x86_64")]
fn avx2_lanczos3_bench(bench: &mut Bench) {
let image = get_big_source_image();
let mut res_image = ImageData::new(
let mut res_image = Image::new(
NonZeroU32::new(NEW_WIDTH).unwrap(),
NonZeroU32::new(NEW_HEIGHT).unwrap(),
image.pixel_type(),
);
let src_image = image.src_view();
let mut dst_image = res_image.dst_view();
let src_image = image.view();
let mut dst_image = res_image.view_mut();
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
unsafe {
resizer.set_cpu_extensions(CpuExtensions::Avx2);
}
bench.task("avx2 lanczos3", |task| {
task.iter(|| {
resizer.resize(&src_image, &mut dst_image);
resizer.resize(&src_image, &mut dst_image).unwrap();
})
});
}
@@ -142,20 +155,20 @@ fn avx2_lanczos3_bench(bench: &mut Bench) {
#[cfg(target_arch = "x86_64")]
fn avx2_supersampling_lanczos3_bench(bench: &mut Bench) {
let image = get_big_source_image();
let mut res_image = ImageData::new(
let mut res_image = Image::new(
NonZeroU32::new(NEW_WIDTH).unwrap(),
NonZeroU32::new(NEW_HEIGHT).unwrap(),
image.pixel_type(),
);
let src_image = image.src_view();
let mut dst_image = res_image.dst_view();
let src_image = image.view();
let mut dst_image = res_image.view_mut();
let mut resizer = Resizer::new(ResizeAlg::SuperSampling(FilterType::Lanczos3, 2));
unsafe {
resizer.set_cpu_extensions(CpuExtensions::Avx2);
}
bench.task("avx2 supersampling lanczos3", |task| {
task.iter(|| {
resizer.resize(&src_image, &mut dst_image);
resizer.resize(&src_image, &mut dst_image).unwrap();
})
});
}
@@ -163,40 +176,80 @@ fn avx2_supersampling_lanczos3_bench(bench: &mut Bench) {
#[cfg(target_arch = "x86_64")]
fn avx2_lanczos3_upscale_bench(bench: &mut Bench) {
let image = get_small_source_image();
let mut res_image = ImageData::new(
let mut res_image = Image::new(
NonZeroU32::new(NEW_BIG_WIDTH).unwrap(),
NonZeroU32::new(NEW_BIG_HEIGHT).unwrap(),
image.pixel_type(),
);
let src_image = image.src_view();
let mut dst_image = res_image.dst_view();
let src_image = image.view();
let mut dst_image = res_image.view_mut();
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
unsafe {
resizer.set_cpu_extensions(CpuExtensions::Avx2);
}
bench.task("avx2 lanczos3 upscale", |task| {
task.iter(|| {
resizer.resize(&src_image, &mut dst_image);
resizer.resize(&src_image, &mut dst_image).unwrap();
})
});
}
fn native_lanczos3_i32_bench(bench: &mut Bench) {
let image = get_big_i32_image();
let mut res_image = ImageData::new(
let mut res_image = Image::new(
NonZeroU32::new(NEW_WIDTH).unwrap(),
NonZeroU32::new(NEW_HEIGHT).unwrap(),
image.pixel_type(),
);
let src_image = image.src_view();
let mut dst_image = res_image.dst_view();
let src_image = image.view();
let mut dst_image = res_image.view_mut();
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
unsafe {
resizer.set_cpu_extensions(CpuExtensions::None);
}
bench.task("i32 lanczos3 wo SIMD", |task| {
task.iter(|| {
resizer.resize(&src_image, &mut dst_image);
resizer.resize(&src_image, &mut dst_image).unwrap();
})
});
}
fn native_lanczos3_u8_bench(bench: &mut Bench) {
let image = get_big_u8_image();
let mut res_image = Image::new(
NonZeroU32::new(NEW_WIDTH).unwrap(),
NonZeroU32::new(NEW_HEIGHT).unwrap(),
image.pixel_type(),
);
let src_image = image.view();
let mut dst_image = res_image.view_mut();
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
unsafe {
resizer.set_cpu_extensions(CpuExtensions::None);
}
bench.task("u8 lanczos3 wo SIMD", |task| {
task.iter(|| {
resizer.resize(&src_image, &mut dst_image).unwrap();
})
});
}
fn native_nearest_u8_bench(bench: &mut Bench) {
let image = get_big_u8_image();
let mut res_image = Image::new(
NonZeroU32::new(NEW_WIDTH).unwrap(),
NonZeroU32::new(NEW_HEIGHT).unwrap(),
image.pixel_type(),
);
let src_image = image.view();
let mut dst_image = res_image.view_mut();
let mut resizer = Resizer::new(ResizeAlg::Nearest);
unsafe {
resizer.set_cpu_extensions(CpuExtensions::None);
}
bench.task("u8 nearest wo SIMD", |task| {
task.iter(|| {
resizer.resize(&src_image, &mut dst_image).unwrap();
})
});
}
@@ -209,14 +262,16 @@ pub fn main() {
let mut bench = create_bench(name, "Resize", &cmd);
#[cfg(target_arch = "x86_64")]
{
sse4_lanczos3_bench(&mut bench);
avx2_lanczos3_bench(&mut bench);
avx2_supersampling_lanczos3_bench(&mut bench);
avx2_lanczos3_upscale_bench(&mut bench);
sse4_lanczos3_bench(&mut bench);
}
nearest_wo_simd_bench(&mut bench);
lanczos3_wo_simd_bench(&mut bench);
native_nearest_bench(&mut bench);
native_lanczos3_bench(&mut bench);
native_lanczos3_i32_bench(&mut bench);
native_lanczos3_u8_bench(&mut bench);
native_nearest_u8_bench(&mut bench);
if let Err(e) = after_bench(&mut bench, &cmd) {
eprintln!("{:?}", e);
}
+11 -2
View File
@@ -3,7 +3,7 @@ use std::env;
use glassbench::*;
use image::io::Reader;
use image::{ImageBuffer, Luma, RgbImage, RgbaImage};
use image::{GrayImage, ImageBuffer, Luma, RgbImage, RgbaImage};
pub fn get_big_rgb_image() -> RgbImage {
let cur_dir = env::current_dir().unwrap();
@@ -23,7 +23,7 @@ pub fn get_big_rgba_image() -> RgbaImage {
img.to_rgba8()
}
pub fn get_big_luma_image() -> ImageBuffer<Luma<u16>, Vec<u16>> {
pub fn get_big_luma16_image() -> ImageBuffer<Luma<u16>, Vec<u16>> {
let cur_dir = env::current_dir().unwrap();
let img = Reader::open(cur_dir.join("data/nasa-4928x3279.png"))
.unwrap()
@@ -32,6 +32,15 @@ pub fn get_big_luma_image() -> ImageBuffer<Luma<u16>, Vec<u16>> {
img.to_luma16()
}
pub fn get_big_luma8_image() -> GrayImage {
let cur_dir = env::current_dir().unwrap();
let img = Reader::open(cur_dir.join("data/nasa-4928x3279.png"))
.unwrap()
.decode()
.unwrap();
img.to_luma8()
}
pub fn get_small_rgba_image() -> RgbaImage {
let cur_dir = env::current_dir().unwrap();
let img = Reader::open(cur_dir.join("data/nasa-852x567.png"))
Binary file not shown.

Before

Width:  |  Height:  |  Size: 15 MiB

+7 -3
View File
@@ -1,10 +1,14 @@
use std::arch::x86_64::*;
use crate::alpha::native;
use crate::image_view::{TypedImageView, TypedImageViewMut};
use crate::pixels::U8x4;
use crate::simd_utils;
use crate::{DstImageView, SrcImageView};
pub(crate) fn divide_alpha_avx2(src_image: &SrcImageView, dst_image: &mut DstImageView) {
pub(crate) fn divide_alpha_avx2(
src_image: TypedImageView<U8x4>,
mut dst_image: TypedImageViewMut<U8x4>,
) {
let width = src_image.width().get();
let src_rows = src_image.iter_rows(0, src_image.height().get());
let dst_rows = dst_image.iter_rows_mut();
@@ -16,7 +20,7 @@ pub(crate) fn divide_alpha_avx2(src_image: &SrcImageView, dst_image: &mut DstIma
}
}
pub(crate) fn divide_alpha_inplace_avx2(image: &mut DstImageView) {
pub(crate) fn divide_alpha_inplace_avx2(mut image: TypedImageViewMut<U8x4>) {
let width = image.width().get() as usize;
for dst_row in image.iter_rows_mut() {
unsafe {
+8 -3
View File
@@ -1,9 +1,14 @@
use std::arch::x86_64::*;
use crate::alpha::native;
use crate::{simd_utils, DstImageView, SrcImageView};
use crate::image_view::{TypedImageView, TypedImageViewMut};
use crate::pixels::U8x4;
use crate::simd_utils;
pub(crate) fn multiply_alpha_avx2(src_image: &SrcImageView, dst_image: &mut DstImageView) {
pub(crate) fn multiply_alpha_avx2(
src_image: TypedImageView<U8x4>,
mut dst_image: TypedImageViewMut<U8x4>,
) {
let width = src_image.width().get() as usize;
let src_rows = src_image.iter_rows(0, src_image.height().get());
let dst_rows = dst_image.iter_rows_mut();
@@ -15,7 +20,7 @@ pub(crate) fn multiply_alpha_avx2(src_image: &SrcImageView, dst_image: &mut DstI
}
}
pub(crate) fn multiply_alpha_inplace_avx2(image: &mut DstImageView) {
pub(crate) fn multiply_alpha_inplace_avx2(mut image: TypedImageViewMut<U8x4>) {
let width = image.width().get() as usize;
for dst_row in image.iter_rows_mut() {
unsafe {
+19
View File
@@ -0,0 +1,19 @@
use thiserror::Error;
#[derive(Error, Debug, Clone, Copy)]
#[non_exhaustive]
pub enum MulDivImagesError {
#[error("Size of source image does not match to destination image")]
SizeIsDifferent,
#[error("Pixel type of source image does not match to destination image")]
PixelTypeIsDifferent,
#[error("Pixel type of image is not supported")]
UnsupportedPixelType,
}
#[derive(Error, Debug, Clone, Copy)]
#[non_exhaustive]
pub enum MulDivImageError {
#[error("Pixel type of image is not supported")]
UnsupportedPixelType,
}
+67 -73
View File
@@ -1,33 +1,17 @@
use thiserror::Error;
use crate::{CpuExtensions, PixelType};
use crate::{DstImageView, SrcImageView};
use crate::image_view::{TypedImageView, TypedImageViewMut};
use crate::pixels::U8x4;
use crate::CpuExtensions;
use crate::{ImageView, ImageViewMut};
pub use errors::*;
#[cfg(target_arch = "x86_64")]
mod avx2;
mod errors;
mod native;
#[cfg(target_arch = "x86_64")]
mod sse2;
#[derive(Error, Debug, Clone, Copy)]
#[non_exhaustive]
pub enum MulDivImagesError {
#[error("Size of source image does not match to destination image")]
SizeIsDifferent,
#[error("Pixel type of source image does not match to destination image")]
PixelTypeIsDifferent,
#[error("Pixel type of image is not supported")]
UnsupportedPixelType,
}
#[derive(Error, Debug, Clone, Copy)]
#[non_exhaustive]
pub enum MulDivImageError {
#[error("Pixel type of image is not supported")]
UnsupportedPixelType,
}
/// Methods of this structure used to multiplies or divides RGB-channels
/// Methods of this structure used to multiply or divide RGB-channels
/// by alpha-channel.
///
/// By default, instance of `MulDiv` created with best CPU-extensions provided by your CPU.
@@ -37,15 +21,15 @@ pub enum MulDivImageError {
///
/// ```
/// use std::num::NonZeroU32;
/// use fast_image_resize::{ImageData, MulDiv, PixelType};
/// use fast_image_resize::{Image, MulDiv, PixelType};
///
/// let width = NonZeroU32::new(10).unwrap();
/// let height = NonZeroU32::new(7).unwrap();
/// let src_image = ImageData::new(width, height, PixelType::U8x4);
/// let mut dst_image = ImageData::new(width, height, PixelType::U8x4);
/// let src_image = Image::new(width, height, PixelType::U8x4);
/// let mut dst_image = Image::new(width, height, PixelType::U8x4);
///
/// let mul_div = MulDiv::default();
/// mul_div.multiply_alpha(&src_image.src_view(), &mut dst_image.dst_view()).unwrap();
/// mul_div.multiply_alpha(&src_image.view(), &mut dst_image.view_mut()).unwrap();
/// ```
#[derive(Default, Debug, Clone)]
pub struct MulDiv {
@@ -69,35 +53,35 @@ impl MulDiv {
/// result into destination image.
pub fn multiply_alpha(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
src_image: &ImageView,
dst_image: &mut ImageViewMut,
) -> Result<(), MulDivImagesError> {
self.assert_images(src_image, dst_image)?;
let (src_image_u8x4, dst_image_u8x4) = assert_images(src_image, dst_image)?;
match self.cpu_extensions {
#[cfg(target_arch = "x86_64")]
CpuExtensions::Avx2 => avx2::multiply_alpha_avx2(src_image, dst_image),
CpuExtensions::Avx2 => avx2::multiply_alpha_avx2(src_image_u8x4, dst_image_u8x4),
// WARNING: SSE2 implementation is drastically slower than native version
// #[cfg(target_arch = "x86_64")]
// CpuExtensions::Sse4_1 | CpuExtensions::Sse2 => {
// sse2::multiply_alpha_sse2(src_image, dst_image)
// }
_ => native::multiply_alpha_native(src_image, dst_image),
_ => native::multiply_alpha_native(src_image_u8x4, dst_image_u8x4),
}
Ok(())
}
/// Multiplies RGB-channels of image by alpha-channel inplace.
pub fn multiply_alpha_inplace(&self, image: &mut DstImageView) -> Result<(), MulDivImageError> {
self.assert_image(image)?;
pub fn multiply_alpha_inplace(&self, image: &mut ImageViewMut) -> Result<(), MulDivImageError> {
let image_u8x4 = assert_image(image)?;
match self.cpu_extensions {
#[cfg(target_arch = "x86_64")]
CpuExtensions::Avx2 => avx2::multiply_alpha_inplace_avx2(image),
CpuExtensions::Avx2 => avx2::multiply_alpha_inplace_avx2(image_u8x4),
// WARNING: SSE2 implementation is drastically slower than native version
// #[cfg(target_arch = "x86_64")]
// CpuExtensions::Sse4_1 | CpuExtensions::Sse2 => {
// sse2::multiply_alpha_sse2(src_image, dst_image)
// }
_ => native::multiply_alpha_inplace_native(image),
_ => native::multiply_alpha_inplace_native(image_u8x4),
}
Ok(())
}
@@ -106,58 +90,68 @@ impl MulDiv {
/// result into destination image.
pub fn divide_alpha(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
src_image: &ImageView,
dst_image: &mut ImageViewMut,
) -> Result<(), MulDivImagesError> {
self.assert_images(src_image, dst_image)?;
let (src_image_u8x4, dst_image_u8x4) = assert_images(src_image, dst_image)?;
match self.cpu_extensions {
#[cfg(target_arch = "x86_64")]
CpuExtensions::Avx2 => avx2::divide_alpha_avx2(src_image, dst_image),
CpuExtensions::Avx2 => avx2::divide_alpha_avx2(src_image_u8x4, dst_image_u8x4),
#[cfg(target_arch = "x86_64")]
CpuExtensions::Sse4_1 | CpuExtensions::Sse2 => {
sse2::divide_alpha_sse2(src_image, dst_image)
sse2::divide_alpha_sse2(src_image_u8x4, dst_image_u8x4)
}
_ => native::divide_alpha_native(src_image, dst_image),
_ => native::divide_alpha_native(src_image_u8x4, dst_image_u8x4),
}
Ok(())
}
/// Divides RGB-channels of image by alpha-channel inplace.
pub fn divide_alpha_inplace(&self, image: &mut DstImageView) -> Result<(), MulDivImageError> {
self.assert_image(image)?;
pub fn divide_alpha_inplace(&self, image: &mut ImageViewMut) -> Result<(), MulDivImageError> {
let image_u8x4 = assert_image(image)?;
match self.cpu_extensions {
#[cfg(target_arch = "x86_64")]
CpuExtensions::Avx2 => avx2::divide_alpha_inplace_avx2(image),
CpuExtensions::Avx2 => avx2::divide_alpha_inplace_avx2(image_u8x4),
#[cfg(target_arch = "x86_64")]
CpuExtensions::Sse4_1 | CpuExtensions::Sse2 => sse2::divide_alpha_inplace_sse2(image),
_ => native::divide_alpha_inplace_native(image),
}
Ok(())
}
#[inline]
fn assert_images(
&self,
src_image: &SrcImageView,
dst_image: &DstImageView,
) -> Result<(), MulDivImagesError> {
if src_image.width() != dst_image.width() || src_image.height() != dst_image.height() {
return Err(MulDivImagesError::SizeIsDifferent);
}
if src_image.pixel_type() != PixelType::U8x4 {
return Err(MulDivImagesError::UnsupportedPixelType);
}
if src_image.pixel_type() != dst_image.pixel_type() {
return Err(MulDivImagesError::PixelTypeIsDifferent);
}
Ok(())
}
#[inline]
fn assert_image(&self, image: &DstImageView) -> Result<(), MulDivImageError> {
if image.pixel_type() != PixelType::U8x4 {
return Err(MulDivImageError::UnsupportedPixelType);
CpuExtensions::Sse4_1 | CpuExtensions::Sse2 => {
sse2::divide_alpha_inplace_sse2(image_u8x4)
}
_ => native::divide_alpha_inplace_native(image_u8x4),
}
Ok(())
}
}
#[inline]
fn assert_images<'s, 'd, 'da>(
src_image: &'s ImageView<'s>,
dst_image: &'d mut ImageViewMut<'da>,
) -> Result<
(
TypedImageView<'s, 's, U8x4>,
TypedImageViewMut<'d, 'da, U8x4>,
),
MulDivImagesError,
> {
let src_image_u8x4 = src_image
.u32_image()
.ok_or(MulDivImagesError::UnsupportedPixelType)?;
let dst_image_u8x4 = dst_image
.u32_image()
.ok_or(MulDivImagesError::UnsupportedPixelType)?;
if src_image_u8x4.width() != dst_image_u8x4.width()
|| src_image_u8x4.height() != dst_image_u8x4.height()
{
return Err(MulDivImagesError::SizeIsDifferent);
}
Ok((src_image_u8x4, dst_image_u8x4))
}
#[inline]
fn assert_image<'a, 'b>(
image: &'a mut ImageViewMut<'b>,
) -> Result<TypedImageViewMut<'a, 'b, U8x4>, MulDivImageError> {
image
.u32_image()
.ok_or(MulDivImageError::UnsupportedPixelType)
}
+7 -3
View File
@@ -1,6 +1,10 @@
use crate::{DstImageView, SrcImageView};
use crate::image_view::{TypedImageView, TypedImageViewMut};
use crate::pixels::U8x4;
pub(crate) fn divide_alpha_native(src_image: &SrcImageView, dst_image: &mut DstImageView) {
pub(crate) fn divide_alpha_native(
src_image: TypedImageView<U8x4>,
mut dst_image: TypedImageViewMut<U8x4>,
) {
let src_rows = src_image.iter_rows(0, src_image.height().get());
let dst_rows = dst_image.iter_rows_mut();
@@ -9,7 +13,7 @@ pub(crate) fn divide_alpha_native(src_image: &SrcImageView, dst_image: &mut DstI
}
}
pub(crate) fn divide_alpha_inplace_native(image: &mut DstImageView) {
pub(crate) fn divide_alpha_inplace_native(mut image: TypedImageViewMut<U8x4>) {
for dst_row in image.iter_rows_mut() {
let src_row = unsafe { std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len()) };
divide_alpha_row_native(src_row, dst_row);
+7 -3
View File
@@ -1,6 +1,10 @@
use crate::{DstImageView, SrcImageView};
use crate::image_view::{TypedImageView, TypedImageViewMut};
use crate::pixels::U8x4;
pub(crate) fn multiply_alpha_native(src_image: &SrcImageView, dst_image: &mut DstImageView) {
pub(crate) fn multiply_alpha_native(
src_image: TypedImageView<U8x4>,
mut dst_image: TypedImageViewMut<U8x4>,
) {
let src_rows = src_image.iter_rows(0, src_image.height().get());
let dst_rows = dst_image.iter_rows_mut();
@@ -9,7 +13,7 @@ pub(crate) fn multiply_alpha_native(src_image: &SrcImageView, dst_image: &mut Ds
}
}
pub(crate) fn multiply_alpha_inplace_native(image: &mut DstImageView) {
pub(crate) fn multiply_alpha_inplace_native(mut image: TypedImageViewMut<U8x4>) {
for dst_row in image.iter_rows_mut() {
let src_row = unsafe { std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len()) };
multiply_alpha_row_native(src_row, dst_row);
+7 -3
View File
@@ -1,10 +1,14 @@
use std::arch::x86_64::*;
use crate::alpha::native;
use crate::image_view::{TypedImageView, TypedImageViewMut};
use crate::pixels::U8x4;
use crate::simd_utils;
use crate::{DstImageView, SrcImageView};
pub(crate) fn divide_alpha_sse2(src_image: &SrcImageView, dst_image: &mut DstImageView) {
pub(crate) fn divide_alpha_sse2(
src_image: TypedImageView<U8x4>,
mut dst_image: TypedImageViewMut<U8x4>,
) {
let width = src_image.width().get() as usize;
let src_rows = src_image.iter_rows(0, src_image.height().get());
let dst_rows = dst_image.iter_rows_mut();
@@ -16,7 +20,7 @@ pub(crate) fn divide_alpha_sse2(src_image: &SrcImageView, dst_image: &mut DstIma
}
}
pub(crate) fn divide_alpha_inplace_sse2(image: &mut DstImageView) {
pub(crate) fn divide_alpha_inplace_sse2(mut image: TypedImageViewMut<U8x4>) {
let width = image.width().get() as usize;
for dst_row in image.iter_rows_mut() {
unsafe {
-1
View File
@@ -1,5 +1,4 @@
pub(crate) use div::{divide_alpha_inplace_sse2, divide_alpha_sse2};
pub(crate) use mul::multiply_alpha_sse2;
mod div;
mod mul;
+6 -2
View File
@@ -1,11 +1,15 @@
use std::arch::x86_64::*;
use crate::alpha::native;
use crate::image_view::{TypedImageView, TypedImageViewMut};
use crate::pixels::U8x4;
use crate::simd_utils;
use crate::{DstImageView, SrcImageView};
#[allow(dead_code)]
pub(crate) fn multiply_alpha_sse2(src_image: &SrcImageView, dst_image: &mut DstImageView) {
pub(crate) fn multiply_alpha_sse2(
src_image: TypedImageView<U8x4>,
mut dst_image: TypedImageViewMut<U8x4>,
) {
let width = src_image.width().get() as usize;
let src_rows = src_image.iter_rows(0, src_image.height().get());
let dst_rows = dst_image.iter_rows_mut();
-3
View File
@@ -1,3 +0,0 @@
pub use u8x4::Avx2U8x4;
mod u8x4;
-521
View File
@@ -1,521 +0,0 @@
use std::arch::x86_64::*;
use std::intrinsics::transmute;
use crate::convolution::optimisations::CoefficientsI16Chunk;
use crate::convolution::{optimisations, Bound, Coefficients, Convolution};
use crate::image_view::{DstImageView, FourRows, FourRowsMut, SrcImageView};
use crate::simd_utils;
pub struct Avx2U8x4;
// This code is based on C-implementation from Pillow-SIMD package for Python
// https://github.com/uploadcare/pillow-simd
impl Avx2U8x4 {
/// For safety, it is necessary to ensure the following conditions:
/// - length of all rows in src_rows must be equal
/// - length of all rows in dst_rows must be equal
/// - coefficients_chunks.len() == dst_rows.0.len()
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
#[inline]
#[target_feature(enable = "avx2")]
unsafe fn horiz_convolution_8u4x(
&self,
src_rows: FourRows,
dst_rows: FourRowsMut,
coefficients_chunks: &[CoefficientsI16Chunk],
precision: u8,
) {
let (s_row0, s_row1, s_row2, s_row3) = src_rows;
let (d_row0, d_row1, d_row2, d_row3) = dst_rows;
let zero = _mm256_setzero_si256();
let initial = _mm256_set1_epi32(1 << (precision - 1));
#[rustfmt::skip]
let sh1 = _mm256_set_epi8(
-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0,
-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0,
);
#[rustfmt::skip]
let sh2 = _mm256_set_epi8(
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
);
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let x_start = coeffs_chunk.start as usize;
let mut x: usize = 0;
let mut sss0 = initial;
let mut sss1 = initial;
let coeffs = coeffs_chunk.values;
let coeffs_by_4 = coeffs.chunks_exact(4);
let reminder1 = coeffs_by_4.remainder();
for k in coeffs_by_4 {
let mmk0 = simd_utils::ptr_i16_to_256set1_epi32(k, 0);
let mmk1 = simd_utils::ptr_i16_to_256set1_epi32(k, 2);
let mut source = _mm256_inserti128_si256::<1>(
_mm256_castsi128_si256(simd_utils::loadu_si128(s_row0, x + x_start)),
simd_utils::loadu_si128(s_row1, x + x_start),
);
let mut pix = _mm256_shuffle_epi8(source, sh1);
sss0 = _mm256_add_epi32(sss0, _mm256_madd_epi16(pix, mmk0));
pix = _mm256_shuffle_epi8(source, sh2);
sss0 = _mm256_add_epi32(sss0, _mm256_madd_epi16(pix, mmk1));
source = _mm256_inserti128_si256::<1>(
_mm256_castsi128_si256(simd_utils::loadu_si128(s_row2, x + x_start)),
simd_utils::loadu_si128(s_row3, x + x_start),
);
pix = _mm256_shuffle_epi8(source, sh1);
sss1 = _mm256_add_epi32(sss1, _mm256_madd_epi16(pix, mmk0));
pix = _mm256_shuffle_epi8(source, sh2);
sss1 = _mm256_add_epi32(sss1, _mm256_madd_epi16(pix, mmk1));
x += 4;
}
let coeffs_by_2 = reminder1.chunks_exact(2);
let reminder2 = coeffs_by_2.remainder();
for k in coeffs_by_2 {
let mmk = simd_utils::ptr_i16_to_256set1_epi32(k, 0);
let mut pix = _mm256_inserti128_si256::<1>(
_mm256_castsi128_si256(simd_utils::loadl_epi64(s_row0, x + x_start)),
simd_utils::loadl_epi64(s_row1, x + x_start),
);
pix = _mm256_shuffle_epi8(pix, sh1);
sss0 = _mm256_add_epi32(sss0, _mm256_madd_epi16(pix, mmk));
pix = _mm256_inserti128_si256::<1>(
_mm256_castsi128_si256(simd_utils::loadl_epi64(s_row2, x + x_start)),
simd_utils::loadl_epi64(s_row3, x + x_start),
);
pix = _mm256_shuffle_epi8(pix, sh1);
sss1 = _mm256_add_epi32(sss1, _mm256_madd_epi16(pix, mmk));
x += 2;
}
for &k in reminder2 {
// [16] xx k0 xx k0 xx k0 xx k0 xx k0 xx k0 xx k0 xx k0
let mmk = _mm256_set1_epi32(k as i32);
// [16] xx a0 xx b0 xx g0 xx r0 xx a0 xx b0 xx g0 xx r0
let mut pix = _mm256_inserti128_si256::<1>(
_mm256_castsi128_si256(simd_utils::mm_cvtepu8_epi32(s_row0, x + x_start)),
simd_utils::mm_cvtepu8_epi32(s_row1, x + x_start),
);
sss0 = _mm256_add_epi32(sss0, _mm256_madd_epi16(pix, mmk));
pix = _mm256_inserti128_si256::<1>(
_mm256_castsi128_si256(simd_utils::mm_cvtepu8_epi32(s_row2, x + x_start)),
simd_utils::mm_cvtepu8_epi32(s_row3, x + x_start),
);
sss1 = _mm256_add_epi32(sss1, _mm256_madd_epi16(pix, mmk));
x += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss0 = _mm256_srai_epi32::<$imm8>(sss0);
sss1 = _mm256_srai_epi32::<$imm8>(sss1);
}};
}
constify_imm8!(precision, call);
sss0 = _mm256_packs_epi32(sss0, zero);
sss1 = _mm256_packs_epi32(sss1, zero);
sss0 = _mm256_packus_epi16(sss0, zero);
sss1 = _mm256_packus_epi16(sss1, zero);
*d_row0.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm256_extracti128_si256::<0>(sss0)));
*d_row1.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm256_extracti128_si256::<1>(sss0)));
*d_row2.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm256_extracti128_si256::<0>(sss1)));
*d_row3.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm256_extracti128_si256::<1>(sss1)));
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - bounds.len() == dst_row.len()
/// - coeffs.len() == dst_rows.0.len() * window_size
/// - max(bound.start + bound.size for bound in bounds) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
#[inline]
#[target_feature(enable = "avx2")]
unsafe fn horiz_convolution_8u(
&self,
src_row: &[u32],
dst_row: &mut [u32],
coefficients_chunks: &[CoefficientsI16Chunk],
precision: u8,
) {
#[rustfmt::skip]
let sh1 = _mm256_set_epi8(
-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0,
-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0,
);
#[rustfmt::skip]
let sh2 = _mm256_set_epi8(
11, 10, 9, 8, 11, 10, 9, 8, 11, 10, 9, 8, 11, 10, 9, 8,
3, 2, 1, 0, 3, 2, 1, 0, 3, 2, 1, 0, 3, 2, 1, 0,
);
#[rustfmt::skip]
let sh3 = _mm256_set_epi8(
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
);
#[rustfmt::skip]
let sh4 = _mm256_set_epi8(
15, 14, 13, 12, 15, 14, 13, 12, 15, 14, 13, 12, 15, 14, 13, 12,
7, 6, 5, 4, 7, 6, 5, 4, 7, 6, 5, 4, 7, 6, 5, 4,
);
#[rustfmt::skip]
let sh5 = _mm256_set_epi8(
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0,
);
#[rustfmt::skip]
let sh6 = _mm256_set_epi8(
7, 6, 5, 4, 7, 6, 5, 4, 7, 6, 5, 4, 7, 6, 5, 4,
3, 2, 1, 0, 3, 2, 1, 0, 3, 2, 1, 0, 3, 2, 1, 0,
);
let sh7 = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let x_start = coeffs_chunk.start as usize;
let mut x: usize = 0;
let mut coeffs = coeffs_chunk.values;
let mut sss: __m128i = if coeffs.len() < 8 {
_mm_set1_epi32(1 << (precision - 1))
} else {
// Lower part will be added to higher, use only half of the error
let mut sss256 = _mm256_set1_epi32(1 << (precision - 2));
let coeffs_by_8 = coeffs.chunks_exact(8);
let reminder1 = coeffs_by_8.remainder();
for k in coeffs_by_8 {
let tmp = simd_utils::loadu_si128(k, 0);
let ksource = _mm256_insertf128_si256::<1>(_mm256_castsi128_si256(tmp), tmp);
let source = simd_utils::loadu_si256(src_row, x + x_start);
let mut pix = _mm256_shuffle_epi8(source, sh1);
let mut mmk = _mm256_shuffle_epi8(ksource, sh2);
sss256 = _mm256_add_epi32(sss256, _mm256_madd_epi16(pix, mmk));
pix = _mm256_shuffle_epi8(source, sh3);
mmk = _mm256_shuffle_epi8(ksource, sh4);
sss256 = _mm256_add_epi32(sss256, _mm256_madd_epi16(pix, mmk));
x += 8;
}
let coeffs_by_4 = reminder1.chunks_exact(4);
coeffs = coeffs_by_4.remainder();
for k in coeffs_by_4 {
let tmp = simd_utils::loadl_epi64(k, 0);
let ksource = _mm256_insertf128_si256::<1>(_mm256_castsi128_si256(tmp), tmp);
let tmp = simd_utils::loadu_si128(src_row, x + x_start);
let source = _mm256_insertf128_si256::<1>(_mm256_castsi128_si256(tmp), tmp);
let pix = _mm256_shuffle_epi8(source, sh5);
let mmk = _mm256_shuffle_epi8(ksource, sh6);
sss256 = _mm256_add_epi32(sss256, _mm256_madd_epi16(pix, mmk));
x += 4;
}
_mm_add_epi32(
_mm256_extracti128_si256::<0>(sss256),
_mm256_extracti128_si256::<1>(sss256),
)
};
let coeffs_by_2 = coeffs.chunks_exact(2);
let reminder1 = coeffs_by_2.remainder();
for k in coeffs_by_2 {
let mmk = simd_utils::ptr_i16_to_set1_epi32(k, 0);
let source = simd_utils::loadl_epi64(src_row, x + x_start);
let pix = _mm_shuffle_epi8(source, sh7);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
x += 2
}
for &k in reminder1 {
let pix = simd_utils::mm_cvtepu8_epi32(src_row, x + x_start);
let mmk = _mm_set1_epi32(k as i32);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
x += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss = _mm_srai_epi32::<$imm8>(sss);
}};
}
constify_imm8!(precision, call);
sss = _mm_packs_epi32(sss, sss);
*dst_row.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss, sss)));
}
}
#[inline]
#[target_feature(enable = "avx2")]
pub unsafe fn vert_convolution_8u(
&self,
src_img: &SrcImageView,
dst_row: &mut [u32],
coeffs: &[i16],
bound: Bound,
precision: u8,
) {
let src_width = src_img.width().get() as usize;
let y_start = bound.start;
let y_size = bound.size;
let initial = _mm_set1_epi32(1 << (precision - 1));
let initial_256 = _mm256_set1_epi32(1 << (precision - 1));
let mut x: usize = 0;
while x < src_width.saturating_sub(7) {
let mut sss0 = initial_256;
let mut sss1 = initial_256;
let mut sss2 = initial_256;
let mut sss3 = initial_256;
let mut y: u32 = 0;
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
// Load two coefficients at once
let mmk = simd_utils::ptr_i16_to_256set1_epi32(coeffs, y as usize);
let source1 = simd_utils::loadu_si256(s_row1, x); // top line
let source2 = simd_utils::loadu_si256(s_row2, x); // bottom line
let mut source = _mm256_unpacklo_epi8(source1, source2);
let mut pix = _mm256_unpacklo_epi8(source, _mm256_setzero_si256());
sss0 = _mm256_add_epi32(sss0, _mm256_madd_epi16(pix, mmk));
pix = _mm256_unpackhi_epi8(source, _mm256_setzero_si256());
sss1 = _mm256_add_epi32(sss1, _mm256_madd_epi16(pix, mmk));
source = _mm256_unpackhi_epi8(source1, source2);
pix = _mm256_unpacklo_epi8(source, _mm256_setzero_si256());
sss2 = _mm256_add_epi32(sss2, _mm256_madd_epi16(pix, mmk));
pix = _mm256_unpackhi_epi8(source, _mm256_setzero_si256());
sss3 = _mm256_add_epi32(sss3, _mm256_madd_epi16(pix, mmk));
y += 2;
}
for s_row in src_img.iter_rows(y_start + y, y_start + y_size) {
let mmk = _mm256_set1_epi32(coeffs[y as usize] as i32);
let source1 = simd_utils::loadu_si256(s_row, x); // top line
let source2 = _mm256_setzero_si256(); // bottom line is empty
let mut source = _mm256_unpacklo_epi8(source1, source2);
let mut pix = _mm256_unpacklo_epi8(source, _mm256_setzero_si256());
sss0 = _mm256_add_epi32(sss0, _mm256_madd_epi16(pix, mmk));
pix = _mm256_unpackhi_epi8(source, _mm256_setzero_si256());
sss1 = _mm256_add_epi32(sss1, _mm256_madd_epi16(pix, mmk));
source = _mm256_unpackhi_epi8(source1, _mm256_setzero_si256());
pix = _mm256_unpacklo_epi8(source, _mm256_setzero_si256());
sss2 = _mm256_add_epi32(sss2, _mm256_madd_epi16(pix, mmk));
pix = _mm256_unpackhi_epi8(source, _mm256_setzero_si256());
sss3 = _mm256_add_epi32(sss3, _mm256_madd_epi16(pix, mmk));
y += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss0 = _mm256_srai_epi32::<$imm8>(sss0);
sss1 = _mm256_srai_epi32::<$imm8>(sss1);
sss2 = _mm256_srai_epi32::<$imm8>(sss2);
sss3 = _mm256_srai_epi32::<$imm8>(sss3);
}};
}
constify_imm8!(precision, call);
sss0 = _mm256_packs_epi32(sss0, sss1);
sss2 = _mm256_packs_epi32(sss2, sss3);
sss0 = _mm256_packus_epi16(sss0, sss2);
let dst_ptr = dst_row.get_unchecked_mut(x..).as_mut_ptr() as *mut __m256i;
_mm256_storeu_si256(dst_ptr, sss0);
x += 8;
}
while x < src_width.saturating_sub(1) {
let mut sss0 = initial; // left row
let mut sss1 = initial; // right row
let mut y: u32 = 0;
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
// Load two coefficients at once
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
let source1 = simd_utils::loadl_epi64(s_row1, x); // top line
let source2 = simd_utils::loadl_epi64(s_row2, x); // bottom line
let source = _mm_unpacklo_epi8(source1, source2);
let mut pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
y += 2;
}
for s_row in src_img.iter_rows(y_start + y, y_start + y_size) {
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
let source1 = simd_utils::loadl_epi64(s_row, x); // top line
let source2 = _mm_setzero_si128(); // bottom line is empty
let source = _mm_unpacklo_epi8(source1, source2);
let mut pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
y += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss0 = _mm_srai_epi32::<$imm8>(sss0);
sss1 = _mm_srai_epi32::<$imm8>(sss1);
}};
}
constify_imm8!(precision, call);
sss0 = _mm_packs_epi32(sss0, sss1);
sss0 = _mm_packus_epi16(sss0, sss0);
let dst_ptr = dst_row.get_unchecked_mut(x..).as_mut_ptr() as *mut __m128i;
_mm_storel_epi64(dst_ptr, sss0);
x += 2;
}
while x < src_width {
let mut sss = initial;
let mut y: u32 = 0;
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
// Load two coefficients at once
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
let source1 = simd_utils::mm_cvtsi32_si128(s_row1, x); // top line
let source2 = simd_utils::mm_cvtsi32_si128(s_row2, x); // bottom line
let source = _mm_unpacklo_epi8(source1, source2);
let pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
y += 2;
}
for s_row in src_img.iter_rows(y_start + y, y_start + y_size) {
let pix = simd_utils::mm_cvtepu8_epi32(s_row, x);
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
y += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss = _mm_srai_epi32::<$imm8>(sss);
}};
}
constify_imm8!(precision, call);
sss = _mm_packs_epi32(sss, sss);
*dst_row.get_unchecked_mut(x) =
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss, sss)));
x += 1;
}
}
}
impl Convolution for Avx2U8x4 {
#[inline]
fn horiz_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
offset: u32,
coeffs: Coefficients,
) {
let (values, window_size, bounds_per_pixel) =
(coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coefficients_chunks =
normalizer_guard.normalized_i16_chunks(window_size, &bounds_per_pixel);
let dst_height = dst_image.height().get();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_image.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
self.horiz_convolution_8u4x(src_rows, dst_rows, &coefficients_chunks, precision);
}
}
let mut yy = dst_height - dst_height % 4;
while yy < dst_height {
unsafe {
self.horiz_convolution_8u(
src_image.get_row(yy + offset).unwrap(),
dst_image.get_row_mut(yy).unwrap(),
&coefficients_chunks,
precision,
);
}
yy += 1;
}
}
#[inline]
fn vert_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coeffs_i16 = normalizer_guard.normalized_i16();
let coeffs_chunks = coeffs_i16.chunks(window_size);
let dst_rows = dst_image.iter_rows_mut();
for ((&bound, k), dst_row) in bounds.iter().zip(coeffs_chunks).zip(dst_rows) {
unsafe {
self.vert_convolution_8u(src_image, dst_row, k, bound, precision);
}
}
}
}
+31
View File
@@ -0,0 +1,31 @@
use super::{Coefficients, Convolution};
use crate::image_view::{TypedImageView, TypedImageViewMut};
use crate::pixels::F32;
use crate::CpuExtensions;
mod native;
impl Convolution for F32 {
fn horiz_convolution(
src_image: TypedImageView<Self>,
dst_image: TypedImageViewMut<Self>,
offset: u32,
coeffs: Coefficients,
cpu_extensions: CpuExtensions,
) {
match cpu_extensions {
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
}
}
fn vert_convolution(
src_image: TypedImageView<Self>,
dst_image: TypedImageViewMut<Self>,
coeffs: Coefficients,
cpu_extensions: CpuExtensions,
) {
match cpu_extensions {
_ => native::vert_convolution(src_image, dst_image, coeffs),
}
}
}
+48
View File
@@ -0,0 +1,48 @@
use crate::convolution::Coefficients;
use crate::image_view::{TypedImageView, TypedImageViewMut};
use crate::pixels::F32;
pub(crate) fn horiz_convolution(
src_image: TypedImageView<F32>,
mut dst_image: TypedImageViewMut<F32>,
offset: u32,
coeffs: Coefficients,
) {
let coefficients_chunks = coeffs.get_chunks();
let mut y_src = offset;
for out_row in dst_image.iter_rows_mut() {
for (out_pixel, coeffs_chunk) in out_row.iter_mut().zip(&coefficients_chunks) {
let first_x_src = coeffs_chunk.start;
let mut ss = 0.;
let pixels = src_image.iter_horiz(first_x_src, y_src);
for (&k, &pixel) in coeffs_chunk.values.iter().zip(pixels) {
ss += pixel as f64 * k;
}
*out_pixel = ss.round() as f32;
}
y_src += 1;
}
}
pub(crate) fn vert_convolution(
src_image: TypedImageView<F32>,
mut dst_image: TypedImageViewMut<F32>,
coeffs: Coefficients,
) {
let coefficients_chunks = coeffs.get_chunks();
for (out_row, coeffs_chunk) in dst_image.iter_rows_mut().zip(coefficients_chunks) {
let first_y_src = coeffs_chunk.start;
for (x_src, out_pixel) in out_row.iter_mut().enumerate() {
let mut ss = 0.;
let mut y_src = first_y_src;
for &k in coeffs_chunk.values.iter() {
let pixel = src_image.get_pixel(x_src as u32, y_src);
ss += pixel as f64 * k;
y_src += 1;
}
*out_pixel = ss.round() as f32;
}
}
}
+31
View File
@@ -0,0 +1,31 @@
use super::{Coefficients, Convolution};
use crate::image_view::{TypedImageView, TypedImageViewMut};
use crate::pixels::I32;
use crate::CpuExtensions;
mod native;
impl Convolution for I32 {
fn horiz_convolution(
src_image: TypedImageView<Self>,
dst_image: TypedImageViewMut<Self>,
offset: u32,
coeffs: Coefficients,
cpu_extensions: CpuExtensions,
) {
match cpu_extensions {
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
}
}
fn vert_convolution(
src_image: TypedImageView<Self>,
dst_image: TypedImageViewMut<Self>,
coeffs: Coefficients,
cpu_extensions: CpuExtensions,
) {
match cpu_extensions {
_ => native::vert_convolution(src_image, dst_image, coeffs),
}
}
}
+48
View File
@@ -0,0 +1,48 @@
use crate::convolution::Coefficients;
use crate::image_view::{TypedImageView, TypedImageViewMut};
use crate::pixels::I32;
pub(crate) fn horiz_convolution(
src_image: TypedImageView<I32>,
mut dst_image: TypedImageViewMut<I32>,
offset: u32,
coeffs: Coefficients,
) {
let coefficients_chunks = coeffs.get_chunks();
let mut y_src = offset;
for out_row in dst_image.iter_rows_mut() {
for (out_pixel, coeffs_chunk) in out_row.iter_mut().zip(&coefficients_chunks) {
let first_x_src = coeffs_chunk.start;
let mut ss = 0.;
let pixels = src_image.iter_horiz(first_x_src, y_src);
for (&k, &pixel) in coeffs_chunk.values.iter().zip(pixels) {
ss += pixel as f64 * k;
}
*out_pixel = ss.round() as i32;
}
y_src += 1;
}
}
pub(crate) fn vert_convolution(
src_image: TypedImageView<I32>,
mut dst_image: TypedImageViewMut<I32>,
coeffs: Coefficients,
) {
let coefficients_chunks = coeffs.get_chunks();
for (out_row, coeffs_chunk) in dst_image.iter_rows_mut().zip(coefficients_chunks) {
let first_y_src = coeffs_chunk.start;
for (x_src, out_pixel) in out_row.iter_mut().enumerate() {
let mut ss = 0.;
let mut y_src = first_y_src;
for &k in coeffs_chunk.values.iter() {
let pixel = src_image.get_pixel(x_src as u32, y_src);
ss += pixel as f64 * k;
y_src += 1;
}
*out_pixel = ss.round() as i32;
}
}
}
+17 -19
View File
@@ -1,39 +1,37 @@
use std::num::NonZeroU32;
#[cfg(target_arch = "x86_64")]
pub use avx2::Avx2U8x4;
use crate::image_view::{TypedImageView, TypedImageViewMut};
use crate::pixels::Pixel;
use crate::CpuExtensions;
pub use filters::{get_filter_func, FilterType};
pub use native::{NativeF32, NativeI32, NativeU8x4};
#[cfg(target_arch = "x86_64")]
pub use sse4::Sse4U8x4;
use crate::image_view::{DstImageView, SrcImageView};
#[macro_use]
mod macros;
#[cfg(target_arch = "x86_64")]
mod avx2;
mod f32x1;
mod filters;
mod native;
mod i32x1;
mod optimisations;
#[cfg(target_arch = "x86_64")]
mod sse4;
mod u8x1;
mod u8x4;
pub trait Convolution {
pub(crate) trait Convolution
where
Self: Pixel + Sized,
{
fn horiz_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
src_image: TypedImageView<Self>,
dst_image: TypedImageViewMut<Self>,
offset: u32,
coeffs: Coefficients,
cpu_extensions: CpuExtensions,
);
fn vert_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
src_image: TypedImageView<Self>,
dst_image: TypedImageViewMut<Self>,
coeffs: Coefficients,
cpu_extensions: CpuExtensions,
);
}
-75
View File
@@ -1,75 +0,0 @@
use std::slice;
use crate::convolution::{Coefficients, Convolution};
use crate::{DstImageView, SrcImageView};
pub struct NativeF32;
impl Convolution for NativeF32 {
fn horiz_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
offset: u32,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
for y_dst in 0..dst_image.height().get() {
let y_src = y_dst + offset;
if let Some(out_row) = dst_image.get_row_mut(y_dst) {
let out_row_f32 = unsafe {
let len = out_row.len();
let ptr = out_row.as_mut_ptr();
slice::from_raw_parts_mut(ptr as *mut f32, len)
};
for (x_dst, (&bound, out_pixel)) in bounds.iter().zip(out_row_f32).enumerate() {
let first_x_src = bound.start;
let start_index = window_size * x_dst;
let end_index = start_index + bound.size as usize;
let ks = &values[start_index..end_index];
let mut ss = 0.;
let pixels = src_image.iter_horiz_f32(first_x_src, y_src);
for (&k, &pixel) in ks.iter().zip(pixels) {
ss += pixel as f64 * k;
}
*out_pixel = ss as f32;
}
}
}
}
fn vert_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
for (y_dst, &bound) in bounds.iter().enumerate() {
let first_y_src = bound.start;
let start_index = window_size * y_dst;
let end_index = start_index + bound.size as usize;
let ks = &values[start_index..end_index];
if let Some(out_row) = dst_image.get_row_mut(y_dst as u32) {
let out_row_f32 = unsafe {
let len = out_row.len();
let ptr = out_row.as_mut_ptr();
slice::from_raw_parts_mut(ptr as *mut f32, len)
};
for (x_src, out_pixel) in out_row_f32.iter_mut().enumerate() {
let mut ss = 0.;
for (dy, &k) in ks.iter().enumerate() {
let pixel = src_image.get_pixel_f32(x_src as u32, first_y_src + dy as u32);
ss += pixel as f64 * k;
}
*out_pixel = ss as f32;
}
}
}
}
}
-53
View File
@@ -1,53 +0,0 @@
use crate::convolution::{Coefficients, Convolution};
use crate::{DstImageView, SrcImageView};
pub struct NativeI32;
impl Convolution for NativeI32 {
fn horiz_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
offset: u32,
coeffs: Coefficients,
) {
let coefficients_chunks = coeffs.get_chunks();
let mut y_src = offset;
for out_row in dst_image.iter_rows_mut() {
for (out_pixel, coeffs_chunk) in out_row.iter_mut().zip(&coefficients_chunks) {
let first_x_src = coeffs_chunk.start;
let mut ss = 0.;
let pixels = src_image.iter_horiz_i32(first_x_src, y_src);
for (&k, &pixel) in coeffs_chunk.values.iter().zip(pixels) {
ss += pixel as f64 * k;
}
*out_pixel = ss.round() as i32 as u32;
}
y_src += 1;
}
}
fn vert_convolution(
&self,
image: &SrcImageView,
out_image: &mut DstImageView,
coeffs: Coefficients,
) {
let coefficients_chunks = coeffs.get_chunks();
for (out_row, coeffs_chunk) in out_image.iter_rows_mut().zip(coefficients_chunks) {
let first_y_src = coeffs_chunk.start;
for (x_src, out_pixel) in out_row.iter_mut().enumerate() {
let mut ss = 0.;
let mut y_src = first_y_src;
for &k in coeffs_chunk.values.iter() {
let pixel = image.get_pixel_i32(x_src as u32, y_src);
ss += pixel as f64 * k;
y_src += 1;
}
*out_pixel = ss.round() as i32 as u32;
}
}
}
}
-7
View File
@@ -1,7 +0,0 @@
pub use f32x1::NativeF32;
pub use i32x1::NativeI32;
pub use u8x4::NativeU8x4;
mod f32x1;
mod i32x1;
mod u8x4;
-95
View File
@@ -1,95 +0,0 @@
use crate::convolution::{optimisations, Coefficients, Convolution};
use crate::{DstImageView, SrcImageView};
pub struct NativeU8x4;
impl Convolution for NativeU8x4 {
fn horiz_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
offset: u32,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coefficients_chunks = normalizer_guard.normalized_i16_chunks(window_size, &bounds);
let dst_rows = dst_image.iter_rows_mut();
for (y_dst, dst_row) in dst_rows.enumerate() {
let y_src = y_dst as u32 + offset;
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
let first_x_src = coeffs_chunk.start;
let ks = coeffs_chunk.values;
let mut ss0 = 1 << (precision - 1);
let mut ss1 = ss0;
let mut ss2 = ss0;
let mut ss3 = ss0;
let src_pixels = src_image.iter_horiz(first_x_src, y_src);
for (&k, &src_pixel) in ks.iter().zip(src_pixels) {
let components: [u8; 4] = src_pixel.to_le_bytes();
ss0 += components[0] as i32 * (k as i32);
ss1 += components[1] as i32 * (k as i32);
ss2 += components[2] as i32 * (k as i32);
ss3 += components[3] as i32 * (k as i32);
}
let t: [u8; 4] = unsafe {
[
optimisations::clip8(ss0, precision),
optimisations::clip8(ss1, precision),
optimisations::clip8(ss2, precision),
optimisations::clip8(ss3, precision),
]
};
*dst_pixel = u32::from_le_bytes(t);
}
}
}
fn vert_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coefficients_chunks = normalizer_guard.normalized_i16_chunks(window_size, &bounds);
let dst_rows = dst_image.iter_rows_mut();
for (&coeffs_chunk, dst_row) in coefficients_chunks.iter().zip(dst_rows) {
let first_y_src = coeffs_chunk.start;
let ks = coeffs_chunk.values;
for (x_src, out_pixel) in dst_row.iter_mut().enumerate() {
let mut ss0 = 1 << (precision - 1);
let mut ss1 = ss0;
let mut ss2 = ss0;
let mut ss3 = ss0;
for (dy, &k) in ks.iter().enumerate() {
let pixel = src_image.get_pixel_u32(x_src as u32, first_y_src + dy as u32);
let components: [u8; 4] = pixel.to_le_bytes();
ss0 += components[0] as i32 * (k as i32);
ss1 += components[1] as i32 * (k as i32);
ss2 += components[2] as i32 * (k as i32);
ss3 += components[3] as i32 * (k as i32);
}
let t: [u8; 4] = unsafe {
[
optimisations::clip8(ss0, precision),
optimisations::clip8(ss1, precision),
optimisations::clip8(ss2, precision),
optimisations::clip8(ss3, precision),
]
};
*out_pixel = u32::from_le_bytes(t);
}
}
}
}
-3
View File
@@ -1,3 +0,0 @@
pub use u8x4::Sse4U8x4;
mod u8x4;
-547
View File
@@ -1,547 +0,0 @@
use std::arch::x86_64::*;
use std::intrinsics::transmute;
use crate::convolution::optimisations::CoefficientsI16Chunk;
use crate::convolution::{optimisations, Bound, Coefficients, Convolution};
use crate::image_view::{DstImageView, FourRows, FourRowsMut, SrcImageView};
use crate::simd_utils;
pub struct Sse4U8x4;
// This code is based on C-implementation from Pillow-SIMD package for Python
// https://github.com/uploadcare/pillow-simd
impl Sse4U8x4 {
/// For safety, it is necessary to ensure the following conditions:
/// - length of all rows in src_rows must be equal
/// - length of all rows in dst_rows must be equal
/// - coefficients_chunks.len() == dst_rows.0.len()
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "sse4.1")]
unsafe fn horiz_convolution_8u4x(
&self,
src_rows: FourRows,
dst_rows: FourRowsMut,
coefficients_chunks: &[CoefficientsI16Chunk],
precision: u8,
) {
let (s_row0, s_row1, s_row2, s_row3) = src_rows;
let (d_row0, d_row1, d_row2, d_row3) = dst_rows;
let initial = _mm_set1_epi32(1 << (precision - 1));
let mask_lo = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
let mask_hi = _mm_set_epi8(-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8);
let mask = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let x_start = coeffs_chunk.start as usize;
let mut x: usize = 0;
let mut sss0 = initial;
let mut sss1 = initial;
let mut sss2 = initial;
let mut sss3 = initial;
let coeffs = coeffs_chunk.values;
let coeffs_by_4 = coeffs.chunks_exact(4);
let reminder1 = coeffs_by_4.remainder();
for k in coeffs_by_4 {
let mmk_lo = simd_utils::ptr_i16_to_set1_epi32(k, 0);
let mmk_hi = simd_utils::ptr_i16_to_set1_epi32(k, 2);
// [8] a3 b3 g3 r3 a2 b2 g2 r2 a1 b1 g1 r1 a0 b0 g0 r0
let mut source = simd_utils::loadu_si128(s_row0, x + x_start);
// [16] a1 a0 b1 b0 g1 g0 r1 r0
let mut pix = _mm_shuffle_epi8(source, mask_lo);
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk_lo));
// [16] a3 a2 b3 b2 g3 g2 r3 r2
pix = _mm_shuffle_epi8(source, mask_hi);
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk_hi));
source = simd_utils::loadu_si128(s_row1, x + x_start);
pix = _mm_shuffle_epi8(source, mask_lo);
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk_lo));
pix = _mm_shuffle_epi8(source, mask_hi);
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk_hi));
source = simd_utils::loadu_si128(s_row2, x + x_start);
pix = _mm_shuffle_epi8(source, mask_lo);
sss2 = _mm_add_epi32(sss2, _mm_madd_epi16(pix, mmk_lo));
pix = _mm_shuffle_epi8(source, mask_hi);
sss2 = _mm_add_epi32(sss2, _mm_madd_epi16(pix, mmk_hi));
source = simd_utils::loadu_si128(s_row3, x + x_start);
pix = _mm_shuffle_epi8(source, mask_lo);
sss3 = _mm_add_epi32(sss3, _mm_madd_epi16(pix, mmk_lo));
pix = _mm_shuffle_epi8(source, mask_hi);
sss3 = _mm_add_epi32(sss3, _mm_madd_epi16(pix, mmk_hi));
x += 4;
}
let coeffs_by_2 = reminder1.chunks_exact(2);
let reminder2 = coeffs_by_2.remainder();
for k in coeffs_by_2 {
// [16] k1 k0 k1 k0 k1 k0 k1 k0
let mmk = simd_utils::ptr_i16_to_set1_epi32(k, 0);
// [8] x x x x x x x x a1 b1 g1 r1 a0 b0 g0 r0
let mut pix = simd_utils::loadl_epi64(s_row0, x + x_start);
// [16] a1 a0 b1 b0 g1 g0 r1 r0
pix = _mm_shuffle_epi8(pix, mask);
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
pix = simd_utils::loadl_epi64(s_row1, x + x_start);
pix = _mm_shuffle_epi8(pix, mask);
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
pix = simd_utils::loadl_epi64(s_row2, x + x_start);
pix = _mm_shuffle_epi8(pix, mask);
sss2 = _mm_add_epi32(sss2, _mm_madd_epi16(pix, mmk));
pix = simd_utils::loadl_epi64(s_row3, x + x_start);
pix = _mm_shuffle_epi8(pix, mask);
sss3 = _mm_add_epi32(sss3, _mm_madd_epi16(pix, mmk));
x += 2;
}
for &k in reminder2 {
// [16] xx k0 xx k0 xx k0 xx k0
let mmk = _mm_set1_epi32(k as i32);
// [16] xx a0 xx b0 xx g0 xx r0
let mut pix = simd_utils::mm_cvtepu8_epi32(s_row0, x);
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
pix = simd_utils::mm_cvtepu8_epi32(s_row1, x);
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
pix = simd_utils::mm_cvtepu8_epi32(s_row2, x);
sss2 = _mm_add_epi32(sss2, _mm_madd_epi16(pix, mmk));
pix = simd_utils::mm_cvtepu8_epi32(s_row3, x);
sss3 = _mm_add_epi32(sss3, _mm_madd_epi16(pix, mmk));
x += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss0 = _mm_srai_epi32::<$imm8>(sss0);
sss1 = _mm_srai_epi32::<$imm8>(sss1);
sss2 = _mm_srai_epi32::<$imm8>(sss2);
sss3 = _mm_srai_epi32::<$imm8>(sss3);
}};
}
constify_imm8!(precision, call);
sss0 = _mm_packs_epi32(sss0, sss0);
sss1 = _mm_packs_epi32(sss1, sss1);
sss2 = _mm_packs_epi32(sss2, sss2);
sss3 = _mm_packs_epi32(sss3, sss3);
*d_row0.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss0, sss0)));
*d_row1.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss1, sss1)));
*d_row2.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss2, sss2)));
*d_row3.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss3, sss3)));
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - bounds.len() == dst_row.len()
/// - coefficients_chunks.len() == dst_row.len()
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "sse4.1")]
unsafe fn horiz_convolution_8u(
&self,
src_row: &[u32],
dst_row: &mut [u32],
coefficients_chunks: &[CoefficientsI16Chunk],
precision: u8,
) {
let initial = _mm_set1_epi32(1 << (precision - 1));
let sh1 = _mm_set_epi8(-1, 11, -1, 3, -1, 10, -1, 2, -1, 9, -1, 1, -1, 8, -1, 0);
let sh2 = _mm_set_epi8(5, 4, 1, 0, 5, 4, 1, 0, 5, 4, 1, 0, 5, 4, 1, 0);
let sh3 = _mm_set_epi8(-1, 15, -1, 7, -1, 14, -1, 6, -1, 13, -1, 5, -1, 12, -1, 4);
let sh4 = _mm_set_epi8(7, 6, 3, 2, 7, 6, 3, 2, 7, 6, 3, 2, 7, 6, 3, 2);
let sh5 = _mm_set_epi8(13, 12, 9, 8, 13, 12, 9, 8, 13, 12, 9, 8, 13, 12, 9, 8);
let sh6 = _mm_set_epi8(
15, 14, 11, 10, 15, 14, 11, 10, 15, 14, 11, 10, 15, 14, 11, 10,
);
let sh7 = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
// for (dst_x, (&bound, k)) in bounds.iter().zip(coeffs_chunks).enumerate() {
let x_start = coeffs_chunk.start as usize;
let mut x: usize = 0;
let mut coeffs = coeffs_chunk.values;
let mut sss = initial;
let coeffs_by_8 = coeffs.chunks_exact(8);
let reminder1 = coeffs_by_8.remainder();
for k in coeffs_by_8 {
let ksource = simd_utils::loadu_si128(k, 0);
let mut source = simd_utils::loadu_si128(src_row, x + x_start);
let mut pix = _mm_shuffle_epi8(source, sh1);
let mut mmk = _mm_shuffle_epi8(ksource, sh2);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
pix = _mm_shuffle_epi8(source, sh3);
mmk = _mm_shuffle_epi8(ksource, sh4);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
source = simd_utils::loadu_si128(src_row, x + 4 + x_start);
pix = _mm_shuffle_epi8(source, sh1);
mmk = _mm_shuffle_epi8(ksource, sh5);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
pix = _mm_shuffle_epi8(source, sh3);
mmk = _mm_shuffle_epi8(ksource, sh6);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
x += 8;
}
let coeffs_by_4 = reminder1.chunks_exact(4);
coeffs = coeffs_by_4.remainder();
for k in coeffs_by_4 {
let source = simd_utils::loadu_si128(src_row, x + x_start);
let ksource = simd_utils::loadl_epi64(k, 0);
let mut pix = _mm_shuffle_epi8(source, sh1);
let mut mmk = _mm_shuffle_epi8(ksource, sh2);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
pix = _mm_shuffle_epi8(source, sh3);
mmk = _mm_shuffle_epi8(ksource, sh4);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
x += 4;
}
let coeffs_by_2 = coeffs.chunks_exact(2);
let reminder1 = coeffs_by_2.remainder();
for k in coeffs_by_2 {
let mmk = simd_utils::ptr_i16_to_set1_epi32(k, 0);
let source = simd_utils::loadl_epi64(src_row, x + x_start);
let pix = _mm_shuffle_epi8(source, sh7);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
x += 2
}
for &k in reminder1 {
let pix = simd_utils::mm_cvtepu8_epi32(src_row, x + x_start);
let mmk = _mm_set1_epi32(k as i32);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
x += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss = _mm_srai_epi32::<$imm8>(sss);
}};
}
constify_imm8!(precision, call);
sss = _mm_packs_epi32(sss, sss);
*dst_row.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss, sss)));
}
}
#[target_feature(enable = "sse4.1")]
pub unsafe fn vert_convolution_8u(
&self,
src_img: &SrcImageView,
dst_row: &mut [u32],
coeffs: &[i16],
bound: Bound,
precision: u8,
) {
let mut xx: usize = 0;
let src_width = src_img.width().get() as usize;
let y_start = bound.start;
let y_size = bound.size;
let initial = _mm_set1_epi32(1 << (precision - 1));
while xx < src_width.saturating_sub(7) {
let mut sss0 = initial;
let mut sss1 = initial;
let mut sss2 = initial;
let mut sss3 = initial;
let mut sss4 = initial;
let mut sss5 = initial;
let mut sss6 = initial;
let mut sss7 = initial;
let mut y: u32 = 0;
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
// Load two coefficients at once
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
let mut source1 = simd_utils::loadu_si128(s_row1, xx); // top line
let mut source2 = simd_utils::loadu_si128(s_row2, xx); // bottom line
let mut source = _mm_unpacklo_epi8(source1, source2);
let mut pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
source = _mm_unpackhi_epi8(source1, source2);
pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss2 = _mm_add_epi32(sss2, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss3 = _mm_add_epi32(sss3, _mm_madd_epi16(pix, mmk));
source1 = simd_utils::loadu_si128(s_row1, xx + 4); // top line
source2 = simd_utils::loadu_si128(s_row2, xx + 4); // bottom line
source = _mm_unpacklo_epi8(source1, source2);
pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss4 = _mm_add_epi32(sss4, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss5 = _mm_add_epi32(sss5, _mm_madd_epi16(pix, mmk));
source = _mm_unpackhi_epi8(source1, source2);
pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss6 = _mm_add_epi32(sss6, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss7 = _mm_add_epi32(sss7, _mm_madd_epi16(pix, mmk));
y += 2;
}
for s_row in src_img.iter_rows(y_start + y, y_start + y_size) {
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
let mut source1 = simd_utils::loadu_si128(s_row, xx); // top line
let mut source = _mm_unpacklo_epi8(source1, _mm_setzero_si128());
let mut pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
source = _mm_unpackhi_epi8(source1, _mm_setzero_si128());
pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss2 = _mm_add_epi32(sss2, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss3 = _mm_add_epi32(sss3, _mm_madd_epi16(pix, mmk));
source1 = simd_utils::loadu_si128(s_row, xx + 4); // top line
source = _mm_unpacklo_epi8(source1, _mm_setzero_si128());
pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss4 = _mm_add_epi32(sss4, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss5 = _mm_add_epi32(sss5, _mm_madd_epi16(pix, mmk));
source = _mm_unpackhi_epi8(source1, _mm_setzero_si128());
pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss6 = _mm_add_epi32(sss6, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss7 = _mm_add_epi32(sss7, _mm_madd_epi16(pix, mmk));
y += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss0 = _mm_srai_epi32::<$imm8>(sss0);
sss1 = _mm_srai_epi32::<$imm8>(sss1);
sss2 = _mm_srai_epi32::<$imm8>(sss2);
sss3 = _mm_srai_epi32::<$imm8>(sss3);
sss4 = _mm_srai_epi32::<$imm8>(sss4);
sss5 = _mm_srai_epi32::<$imm8>(sss5);
sss6 = _mm_srai_epi32::<$imm8>(sss6);
sss7 = _mm_srai_epi32::<$imm8>(sss7);
}};
}
constify_imm8!(precision, call);
sss0 = _mm_packs_epi32(sss0, sss1);
sss2 = _mm_packs_epi32(sss2, sss3);
sss0 = _mm_packus_epi16(sss0, sss2);
let dst_ptr = dst_row.get_unchecked_mut(xx..).as_mut_ptr() as *mut __m128i;
_mm_storeu_si128(dst_ptr, sss0);
sss4 = _mm_packs_epi32(sss4, sss5);
sss6 = _mm_packs_epi32(sss6, sss7);
sss4 = _mm_packus_epi16(sss4, sss6);
let dst_ptr = dst_row.get_unchecked_mut(xx + 4..).as_mut_ptr() as *mut __m128i;
_mm_storeu_si128(dst_ptr, sss4);
xx += 8;
}
while xx < src_width.saturating_sub(1) {
let mut sss0 = initial; // left row
let mut sss1 = initial; // right row
let mut y: u32 = 0;
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
// Load two coefficients at once
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
let source1 = simd_utils::loadl_epi64(s_row1, xx); // top line
let source2 = simd_utils::loadl_epi64(s_row2, xx); // bottom line
let source = _mm_unpacklo_epi8(source1, source2);
let mut pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
y += 2;
}
for s_row1 in src_img.iter_rows(y_start + y, y_start + y_size) {
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
let source1 = simd_utils::loadl_epi64(s_row1, xx); // top line
let source = _mm_unpacklo_epi8(source1, _mm_setzero_si128());
let mut pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
y += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss0 = _mm_srai_epi32::<$imm8>(sss0);
sss1 = _mm_srai_epi32::<$imm8>(sss1);
}};
}
constify_imm8!(precision, call);
sss0 = _mm_packs_epi32(sss0, sss1);
sss0 = _mm_packus_epi16(sss0, sss0);
let dst_ptr = dst_row.get_unchecked_mut(xx..).as_mut_ptr() as *mut __m128i;
_mm_storel_epi64(dst_ptr, sss0);
//
xx += 2;
}
while xx < src_width {
let mut sss = initial;
let mut y: u32 = 0;
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
// Load two coefficients at once
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
let source1 = simd_utils::mm_cvtsi32_si128(s_row1, xx); // top line
let source2 = simd_utils::mm_cvtsi32_si128(s_row2, xx); // bottom line
let source = _mm_unpacklo_epi8(source1, source2);
let pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
y += 2;
}
for s_row in src_img.iter_rows(y_start + y, y_start + y_size) {
let pix = simd_utils::mm_cvtepu8_epi32(s_row, xx);
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
y += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss = _mm_srai_epi32::<$imm8>(sss);
}};
}
constify_imm8!(precision, call);
sss = _mm_packs_epi32(sss, sss);
*dst_row.get_unchecked_mut(xx) =
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss, sss)));
xx += 1;
}
}
}
impl Convolution for Sse4U8x4 {
#[inline]
fn horiz_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
offset: u32,
coeffs: Coefficients,
) {
let (values, window_size, bounds_per_pixel) =
(coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coefficients_chunks =
normalizer_guard.normalized_i16_chunks(window_size, &bounds_per_pixel);
let dst_height = dst_image.height().get();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_image.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
self.horiz_convolution_8u4x(src_rows, dst_rows, &coefficients_chunks, precision);
}
}
let mut yy = dst_height - dst_height % 4;
while yy < dst_height {
unsafe {
self.horiz_convolution_8u(
src_image.get_row(yy + offset).unwrap(),
dst_image.get_row_mut(yy).unwrap(),
&coefficients_chunks,
precision,
);
}
yy += 1;
}
}
#[inline]
fn vert_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coeffs_i16 = normalizer_guard.normalized_i16();
let coeffs_chunks = coeffs_i16.chunks(window_size);
let dst_rows = dst_image.iter_rows_mut();
for ((&bound, k), dst_row) in bounds.iter().zip(coeffs_chunks).zip(dst_rows) {
unsafe {
self.vert_convolution_8u(src_image, dst_row, k, bound, precision);
}
}
}
}
+27
View File
@@ -0,0 +1,27 @@
use super::{Coefficients, Convolution};
use crate::image_view::{TypedImageView, TypedImageViewMut};
use crate::pixels::U8;
use crate::CpuExtensions;
mod native;
impl Convolution for U8 {
fn horiz_convolution(
src_image: TypedImageView<Self>,
dst_image: TypedImageViewMut<Self>,
offset: u32,
coeffs: Coefficients,
_cpu_extensions: CpuExtensions,
) {
native::horiz_convolution(src_image, dst_image, offset, coeffs);
}
fn vert_convolution(
src_image: TypedImageView<Self>,
dst_image: TypedImageViewMut<Self>,
coeffs: Coefficients,
_cpu_extensions: CpuExtensions,
) {
native::vert_convolution(src_image, dst_image, coeffs);
}
}
+60
View File
@@ -0,0 +1,60 @@
use crate::convolution::{optimisations, Coefficients};
use crate::image_view::{TypedImageView, TypedImageViewMut};
use crate::pixels::U8;
pub(crate) fn horiz_convolution(
src_image: TypedImageView<U8>,
mut dst_image: TypedImageViewMut<U8>,
offset: u32,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coefficients_chunks = normalizer_guard.normalized_i16_chunks(window_size, &bounds);
let dst_rows = dst_image.iter_rows_mut();
for (y_dst, dst_row) in dst_rows.enumerate() {
let y_src = y_dst as u32 + offset;
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
let first_x_src = coeffs_chunk.start;
let ks = coeffs_chunk.values;
let mut ss0 = 1 << (precision - 1);
let src_pixels = src_image.iter_horiz(first_x_src, y_src);
for (&k, &src_pixel) in ks.iter().zip(src_pixels) {
ss0 += src_pixel as i32 * (k as i32);
}
*dst_pixel = unsafe { optimisations::clip8(ss0, precision) };
}
}
}
pub(crate) fn vert_convolution(
src_image: TypedImageView<U8>,
mut dst_image: TypedImageViewMut<U8>,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coefficients_chunks = normalizer_guard.normalized_i16_chunks(window_size, &bounds);
let dst_rows = dst_image.iter_rows_mut();
for (&coeffs_chunk, dst_row) in coefficients_chunks.iter().zip(dst_rows) {
let first_y_src = coeffs_chunk.start;
let ks = coeffs_chunk.values;
for (x_src, dst_pixel) in dst_row.iter_mut().enumerate() {
let mut ss0 = 1 << (precision - 1);
for (dy, &k) in ks.iter().enumerate() {
let src_pixel = src_image.get_pixel(x_src as u32, first_y_src + dy as u32);
ss0 += src_pixel as i32 * (k as i32);
}
*dst_pixel = unsafe { optimisations::clip8(ss0, precision) };
}
}
}
+511
View File
@@ -0,0 +1,511 @@
use std::arch::x86_64::*;
use std::intrinsics::transmute;
use crate::convolution::optimisations::CoefficientsI16Chunk;
use crate::convolution::{optimisations, Bound, Coefficients};
use crate::image_view::{FourRows, FourRowsMut, TypedImageView, TypedImageViewMut};
use crate::pixels::U8x4;
use crate::simd_utils;
// This code is based on C-implementation from Pillow-SIMD package for Python
// https://github.com/uploadcare/pillow-simd
#[inline]
pub(crate) fn horiz_convolution(
src_image: TypedImageView<U8x4>,
mut dst_image: TypedImageViewMut<U8x4>,
offset: u32,
coeffs: Coefficients,
) {
let (values, window_size, bounds_per_pixel) =
(coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coefficients_chunks =
normalizer_guard.normalized_i16_chunks(window_size, &bounds_per_pixel);
let dst_height = dst_image.height().get();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_image.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_8u4x(src_rows, dst_rows, &coefficients_chunks, precision);
}
}
let mut yy = dst_height - dst_height % 4;
while yy < dst_height {
unsafe {
horiz_convolution_8u(
src_image.get_row(yy + offset).unwrap(),
dst_image.get_row_mut(yy).unwrap(),
&coefficients_chunks,
precision,
);
}
yy += 1;
}
}
#[inline]
pub(crate) fn vert_convolution(
src_image: TypedImageView<U8x4>,
mut dst_image: TypedImageViewMut<U8x4>,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coeffs_i16 = normalizer_guard.normalized_i16();
let coeffs_chunks = coeffs_i16.chunks(window_size);
let dst_rows = dst_image.iter_rows_mut();
for ((&bound, k), dst_row) in bounds.iter().zip(coeffs_chunks).zip(dst_rows) {
unsafe {
vert_convolution_8u(&src_image, dst_row, k, bound, precision);
}
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - length of all rows in src_rows must be equal
/// - length of all rows in dst_rows must be equal
/// - coefficients_chunks.len() == dst_rows.0.len()
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
#[inline]
#[target_feature(enable = "avx2")]
unsafe fn horiz_convolution_8u4x(
src_rows: FourRows<u32>,
dst_rows: FourRowsMut<u32>,
coefficients_chunks: &[CoefficientsI16Chunk],
precision: u8,
) {
let (s_row0, s_row1, s_row2, s_row3) = src_rows;
let (d_row0, d_row1, d_row2, d_row3) = dst_rows;
let zero = _mm256_setzero_si256();
let initial = _mm256_set1_epi32(1 << (precision - 1));
#[rustfmt::skip]
let sh1 = _mm256_set_epi8(
-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0,
-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0,
);
#[rustfmt::skip]
let sh2 = _mm256_set_epi8(
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
);
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let x_start = coeffs_chunk.start as usize;
let mut x: usize = 0;
let mut sss0 = initial;
let mut sss1 = initial;
let coeffs = coeffs_chunk.values;
let coeffs_by_4 = coeffs.chunks_exact(4);
let reminder1 = coeffs_by_4.remainder();
for k in coeffs_by_4 {
let mmk0 = simd_utils::ptr_i16_to_256set1_epi32(k, 0);
let mmk1 = simd_utils::ptr_i16_to_256set1_epi32(k, 2);
let mut source = _mm256_inserti128_si256::<1>(
_mm256_castsi128_si256(simd_utils::loadu_si128(s_row0, x + x_start)),
simd_utils::loadu_si128(s_row1, x + x_start),
);
let mut pix = _mm256_shuffle_epi8(source, sh1);
sss0 = _mm256_add_epi32(sss0, _mm256_madd_epi16(pix, mmk0));
pix = _mm256_shuffle_epi8(source, sh2);
sss0 = _mm256_add_epi32(sss0, _mm256_madd_epi16(pix, mmk1));
source = _mm256_inserti128_si256::<1>(
_mm256_castsi128_si256(simd_utils::loadu_si128(s_row2, x + x_start)),
simd_utils::loadu_si128(s_row3, x + x_start),
);
pix = _mm256_shuffle_epi8(source, sh1);
sss1 = _mm256_add_epi32(sss1, _mm256_madd_epi16(pix, mmk0));
pix = _mm256_shuffle_epi8(source, sh2);
sss1 = _mm256_add_epi32(sss1, _mm256_madd_epi16(pix, mmk1));
x += 4;
}
let coeffs_by_2 = reminder1.chunks_exact(2);
let reminder2 = coeffs_by_2.remainder();
for k in coeffs_by_2 {
let mmk = simd_utils::ptr_i16_to_256set1_epi32(k, 0);
let mut pix = _mm256_inserti128_si256::<1>(
_mm256_castsi128_si256(simd_utils::loadl_epi64(s_row0, x + x_start)),
simd_utils::loadl_epi64(s_row1, x + x_start),
);
pix = _mm256_shuffle_epi8(pix, sh1);
sss0 = _mm256_add_epi32(sss0, _mm256_madd_epi16(pix, mmk));
pix = _mm256_inserti128_si256::<1>(
_mm256_castsi128_si256(simd_utils::loadl_epi64(s_row2, x + x_start)),
simd_utils::loadl_epi64(s_row3, x + x_start),
);
pix = _mm256_shuffle_epi8(pix, sh1);
sss1 = _mm256_add_epi32(sss1, _mm256_madd_epi16(pix, mmk));
x += 2;
}
for &k in reminder2 {
// [16] xx k0 xx k0 xx k0 xx k0 xx k0 xx k0 xx k0 xx k0
let mmk = _mm256_set1_epi32(k as i32);
// [16] xx a0 xx b0 xx g0 xx r0 xx a0 xx b0 xx g0 xx r0
let mut pix = _mm256_inserti128_si256::<1>(
_mm256_castsi128_si256(simd_utils::mm_cvtepu8_epi32(s_row0, x + x_start)),
simd_utils::mm_cvtepu8_epi32(s_row1, x + x_start),
);
sss0 = _mm256_add_epi32(sss0, _mm256_madd_epi16(pix, mmk));
pix = _mm256_inserti128_si256::<1>(
_mm256_castsi128_si256(simd_utils::mm_cvtepu8_epi32(s_row2, x + x_start)),
simd_utils::mm_cvtepu8_epi32(s_row3, x + x_start),
);
sss1 = _mm256_add_epi32(sss1, _mm256_madd_epi16(pix, mmk));
x += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss0 = _mm256_srai_epi32::<$imm8>(sss0);
sss1 = _mm256_srai_epi32::<$imm8>(sss1);
}};
}
constify_imm8!(precision, call);
sss0 = _mm256_packs_epi32(sss0, zero);
sss1 = _mm256_packs_epi32(sss1, zero);
sss0 = _mm256_packus_epi16(sss0, zero);
sss1 = _mm256_packus_epi16(sss1, zero);
*d_row0.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm256_extracti128_si256::<0>(sss0)));
*d_row1.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm256_extracti128_si256::<1>(sss0)));
*d_row2.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm256_extracti128_si256::<0>(sss1)));
*d_row3.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm256_extracti128_si256::<1>(sss1)));
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - bounds.len() == dst_row.len()
/// - coeffs.len() == dst_rows.0.len() * window_size
/// - max(bound.start + bound.size for bound in bounds) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
#[inline]
#[target_feature(enable = "avx2")]
unsafe fn horiz_convolution_8u(
src_row: &[u32],
dst_row: &mut [u32],
coefficients_chunks: &[CoefficientsI16Chunk],
precision: u8,
) {
#[rustfmt::skip]
let sh1 = _mm256_set_epi8(
-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0,
-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0,
);
#[rustfmt::skip]
let sh2 = _mm256_set_epi8(
11, 10, 9, 8, 11, 10, 9, 8, 11, 10, 9, 8, 11, 10, 9, 8,
3, 2, 1, 0, 3, 2, 1, 0, 3, 2, 1, 0, 3, 2, 1, 0,
);
#[rustfmt::skip]
let sh3 = _mm256_set_epi8(
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
);
#[rustfmt::skip]
let sh4 = _mm256_set_epi8(
15, 14, 13, 12, 15, 14, 13, 12, 15, 14, 13, 12, 15, 14, 13, 12,
7, 6, 5, 4, 7, 6, 5, 4, 7, 6, 5, 4, 7, 6, 5, 4,
);
#[rustfmt::skip]
let sh5 = _mm256_set_epi8(
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0,
);
#[rustfmt::skip]
let sh6 = _mm256_set_epi8(
7, 6, 5, 4, 7, 6, 5, 4, 7, 6, 5, 4, 7, 6, 5, 4,
3, 2, 1, 0, 3, 2, 1, 0, 3, 2, 1, 0, 3, 2, 1, 0,
);
let sh7 = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let x_start = coeffs_chunk.start as usize;
let mut x: usize = 0;
let mut coeffs = coeffs_chunk.values;
let mut sss: __m128i = if coeffs.len() < 8 {
_mm_set1_epi32(1 << (precision - 1))
} else {
// Lower part will be added to higher, use only half of the error
let mut sss256 = _mm256_set1_epi32(1 << (precision - 2));
let coeffs_by_8 = coeffs.chunks_exact(8);
let reminder1 = coeffs_by_8.remainder();
for k in coeffs_by_8 {
let tmp = simd_utils::loadu_si128(k, 0);
let ksource = _mm256_insertf128_si256::<1>(_mm256_castsi128_si256(tmp), tmp);
let source = simd_utils::loadu_si256(src_row, x + x_start);
let mut pix = _mm256_shuffle_epi8(source, sh1);
let mut mmk = _mm256_shuffle_epi8(ksource, sh2);
sss256 = _mm256_add_epi32(sss256, _mm256_madd_epi16(pix, mmk));
pix = _mm256_shuffle_epi8(source, sh3);
mmk = _mm256_shuffle_epi8(ksource, sh4);
sss256 = _mm256_add_epi32(sss256, _mm256_madd_epi16(pix, mmk));
x += 8;
}
let coeffs_by_4 = reminder1.chunks_exact(4);
coeffs = coeffs_by_4.remainder();
for k in coeffs_by_4 {
let tmp = simd_utils::loadl_epi64(k, 0);
let ksource = _mm256_insertf128_si256::<1>(_mm256_castsi128_si256(tmp), tmp);
let tmp = simd_utils::loadu_si128(src_row, x + x_start);
let source = _mm256_insertf128_si256::<1>(_mm256_castsi128_si256(tmp), tmp);
let pix = _mm256_shuffle_epi8(source, sh5);
let mmk = _mm256_shuffle_epi8(ksource, sh6);
sss256 = _mm256_add_epi32(sss256, _mm256_madd_epi16(pix, mmk));
x += 4;
}
_mm_add_epi32(
_mm256_extracti128_si256::<0>(sss256),
_mm256_extracti128_si256::<1>(sss256),
)
};
let coeffs_by_2 = coeffs.chunks_exact(2);
let reminder1 = coeffs_by_2.remainder();
for k in coeffs_by_2 {
let mmk = simd_utils::ptr_i16_to_set1_epi32(k, 0);
let source = simd_utils::loadl_epi64(src_row, x + x_start);
let pix = _mm_shuffle_epi8(source, sh7);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
x += 2
}
for &k in reminder1 {
let pix = simd_utils::mm_cvtepu8_epi32(src_row, x + x_start);
let mmk = _mm_set1_epi32(k as i32);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
x += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss = _mm_srai_epi32::<$imm8>(sss);
}};
}
constify_imm8!(precision, call);
sss = _mm_packs_epi32(sss, sss);
*dst_row.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss, sss)));
}
}
#[inline]
#[target_feature(enable = "avx2")]
pub(crate) unsafe fn vert_convolution_8u(
src_img: &TypedImageView<U8x4>,
dst_row: &mut [u32],
coeffs: &[i16],
bound: Bound,
precision: u8,
) {
let src_width = src_img.width().get() as usize;
let y_start = bound.start;
let y_size = bound.size;
let initial = _mm_set1_epi32(1 << (precision - 1));
let initial_256 = _mm256_set1_epi32(1 << (precision - 1));
let mut x: usize = 0;
while x < src_width.saturating_sub(7) {
let mut sss0 = initial_256;
let mut sss1 = initial_256;
let mut sss2 = initial_256;
let mut sss3 = initial_256;
let mut y: u32 = 0;
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
// Load two coefficients at once
let mmk = simd_utils::ptr_i16_to_256set1_epi32(coeffs, y as usize);
let source1 = simd_utils::loadu_si256(s_row1, x); // top line
let source2 = simd_utils::loadu_si256(s_row2, x); // bottom line
let mut source = _mm256_unpacklo_epi8(source1, source2);
let mut pix = _mm256_unpacklo_epi8(source, _mm256_setzero_si256());
sss0 = _mm256_add_epi32(sss0, _mm256_madd_epi16(pix, mmk));
pix = _mm256_unpackhi_epi8(source, _mm256_setzero_si256());
sss1 = _mm256_add_epi32(sss1, _mm256_madd_epi16(pix, mmk));
source = _mm256_unpackhi_epi8(source1, source2);
pix = _mm256_unpacklo_epi8(source, _mm256_setzero_si256());
sss2 = _mm256_add_epi32(sss2, _mm256_madd_epi16(pix, mmk));
pix = _mm256_unpackhi_epi8(source, _mm256_setzero_si256());
sss3 = _mm256_add_epi32(sss3, _mm256_madd_epi16(pix, mmk));
y += 2;
}
for s_row in src_img.iter_rows(y_start + y, y_start + y_size) {
let mmk = _mm256_set1_epi32(coeffs[y as usize] as i32);
let source1 = simd_utils::loadu_si256(s_row, x); // top line
let source2 = _mm256_setzero_si256(); // bottom line is empty
let mut source = _mm256_unpacklo_epi8(source1, source2);
let mut pix = _mm256_unpacklo_epi8(source, _mm256_setzero_si256());
sss0 = _mm256_add_epi32(sss0, _mm256_madd_epi16(pix, mmk));
pix = _mm256_unpackhi_epi8(source, _mm256_setzero_si256());
sss1 = _mm256_add_epi32(sss1, _mm256_madd_epi16(pix, mmk));
source = _mm256_unpackhi_epi8(source1, _mm256_setzero_si256());
pix = _mm256_unpacklo_epi8(source, _mm256_setzero_si256());
sss2 = _mm256_add_epi32(sss2, _mm256_madd_epi16(pix, mmk));
pix = _mm256_unpackhi_epi8(source, _mm256_setzero_si256());
sss3 = _mm256_add_epi32(sss3, _mm256_madd_epi16(pix, mmk));
y += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss0 = _mm256_srai_epi32::<$imm8>(sss0);
sss1 = _mm256_srai_epi32::<$imm8>(sss1);
sss2 = _mm256_srai_epi32::<$imm8>(sss2);
sss3 = _mm256_srai_epi32::<$imm8>(sss3);
}};
}
constify_imm8!(precision, call);
sss0 = _mm256_packs_epi32(sss0, sss1);
sss2 = _mm256_packs_epi32(sss2, sss3);
sss0 = _mm256_packus_epi16(sss0, sss2);
let dst_ptr = dst_row.get_unchecked_mut(x..).as_mut_ptr() as *mut __m256i;
_mm256_storeu_si256(dst_ptr, sss0);
x += 8;
}
while x < src_width.saturating_sub(1) {
let mut sss0 = initial; // left row
let mut sss1 = initial; // right row
let mut y: u32 = 0;
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
// Load two coefficients at once
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
let source1 = simd_utils::loadl_epi64(s_row1, x); // top line
let source2 = simd_utils::loadl_epi64(s_row2, x); // bottom line
let source = _mm_unpacklo_epi8(source1, source2);
let mut pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
y += 2;
}
for s_row in src_img.iter_rows(y_start + y, y_start + y_size) {
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
let source1 = simd_utils::loadl_epi64(s_row, x); // top line
let source2 = _mm_setzero_si128(); // bottom line is empty
let source = _mm_unpacklo_epi8(source1, source2);
let mut pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
y += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss0 = _mm_srai_epi32::<$imm8>(sss0);
sss1 = _mm_srai_epi32::<$imm8>(sss1);
}};
}
constify_imm8!(precision, call);
sss0 = _mm_packs_epi32(sss0, sss1);
sss0 = _mm_packus_epi16(sss0, sss0);
let dst_ptr = dst_row.get_unchecked_mut(x..).as_mut_ptr() as *mut __m128i;
_mm_storel_epi64(dst_ptr, sss0);
x += 2;
}
while x < src_width {
let mut sss = initial;
let mut y: u32 = 0;
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
// Load two coefficients at once
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
let source1 = simd_utils::mm_cvtsi32_si128(s_row1, x); // top line
let source2 = simd_utils::mm_cvtsi32_si128(s_row2, x); // bottom line
let source = _mm_unpacklo_epi8(source1, source2);
let pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
y += 2;
}
for s_row in src_img.iter_rows(y_start + y, y_start + y_size) {
let pix = simd_utils::mm_cvtepu8_epi32(s_row, x);
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
y += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss = _mm_srai_epi32::<$imm8>(sss);
}};
}
constify_imm8!(precision, call);
sss = _mm_packs_epi32(sss, sss);
*dst_row.get_unchecked_mut(x) = transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss, sss)));
x += 1;
}
}
+43
View File
@@ -0,0 +1,43 @@
use super::{Coefficients, Convolution};
use crate::image_view::{TypedImageView, TypedImageViewMut};
use crate::pixels::U8x4;
use crate::CpuExtensions;
#[cfg(target_arch = "x86_64")]
mod avx2;
mod native;
#[cfg(target_arch = "x86_64")]
mod sse4;
impl Convolution for U8x4 {
fn horiz_convolution(
src_image: TypedImageView<Self>,
dst_image: TypedImageViewMut<Self>,
offset: u32,
coeffs: Coefficients,
cpu_extensions: CpuExtensions,
) {
match cpu_extensions {
#[cfg(target_arch = "x86_64")]
CpuExtensions::Avx2 => avx2::horiz_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "x86_64")]
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_image, dst_image, offset, coeffs),
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
}
}
fn vert_convolution(
src_image: TypedImageView<Self>,
dst_image: TypedImageViewMut<Self>,
coeffs: Coefficients,
cpu_extensions: CpuExtensions,
) {
match cpu_extensions {
#[cfg(target_arch = "x86_64")]
CpuExtensions::Avx2 => avx2::vert_convolution(src_image, dst_image, coeffs),
#[cfg(target_arch = "x86_64")]
CpuExtensions::Sse4_1 => sse4::vert_convolution(src_image, dst_image, coeffs),
_ => native::vert_convolution(src_image, dst_image, coeffs),
}
}
}
+90
View File
@@ -0,0 +1,90 @@
use crate::convolution::{optimisations, Coefficients};
use crate::image_view::{TypedImageView, TypedImageViewMut};
use crate::pixels::U8x4;
pub(crate) fn horiz_convolution(
src_image: TypedImageView<U8x4>,
mut dst_image: TypedImageViewMut<U8x4>,
offset: u32,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coefficients_chunks = normalizer_guard.normalized_i16_chunks(window_size, &bounds);
let dst_rows = dst_image.iter_rows_mut();
for (y_dst, dst_row) in dst_rows.enumerate() {
let y_src = y_dst as u32 + offset;
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
let first_x_src = coeffs_chunk.start;
let ks = coeffs_chunk.values;
let mut ss0 = 1 << (precision - 1);
let mut ss1 = ss0;
let mut ss2 = ss0;
let mut ss3 = ss0;
let src_pixels = src_image.iter_horiz(first_x_src, y_src);
for (&k, &src_pixel) in ks.iter().zip(src_pixels) {
let components: [u8; 4] = src_pixel.to_le_bytes();
ss0 += components[0] as i32 * (k as i32);
ss1 += components[1] as i32 * (k as i32);
ss2 += components[2] as i32 * (k as i32);
ss3 += components[3] as i32 * (k as i32);
}
let t: [u8; 4] = unsafe {
[
optimisations::clip8(ss0, precision),
optimisations::clip8(ss1, precision),
optimisations::clip8(ss2, precision),
optimisations::clip8(ss3, precision),
]
};
*dst_pixel = u32::from_le_bytes(t);
}
}
}
pub(crate) fn vert_convolution(
src_image: TypedImageView<U8x4>,
mut dst_image: TypedImageViewMut<U8x4>,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coefficients_chunks = normalizer_guard.normalized_i16_chunks(window_size, &bounds);
let dst_rows = dst_image.iter_rows_mut();
for (&coeffs_chunk, dst_row) in coefficients_chunks.iter().zip(dst_rows) {
let first_y_src = coeffs_chunk.start;
let ks = coeffs_chunk.values;
for (x_src, out_pixel) in dst_row.iter_mut().enumerate() {
let mut ss0 = 1 << (precision - 1);
let mut ss1 = ss0;
let mut ss2 = ss0;
let mut ss3 = ss0;
for (dy, &k) in ks.iter().enumerate() {
let pixel = src_image.get_pixel(x_src as u32, first_y_src + dy as u32);
let components: [u8; 4] = pixel.to_le_bytes();
ss0 += components[0] as i32 * (k as i32);
ss1 += components[1] as i32 * (k as i32);
ss2 += components[2] as i32 * (k as i32);
ss3 += components[3] as i32 * (k as i32);
}
let t: [u8; 4] = unsafe {
[
optimisations::clip8(ss0, precision),
optimisations::clip8(ss1, precision),
optimisations::clip8(ss2, precision),
optimisations::clip8(ss3, precision),
]
};
*out_pixel = u32::from_le_bytes(t);
}
}
}
+537
View File
@@ -0,0 +1,537 @@
use std::arch::x86_64::*;
use std::intrinsics::transmute;
use crate::convolution::optimisations::CoefficientsI16Chunk;
use crate::convolution::{optimisations, Bound, Coefficients};
use crate::image_view::{FourRows, FourRowsMut, TypedImageView, TypedImageViewMut};
use crate::pixels::U8x4;
use crate::simd_utils;
// This code is based on C-implementation from Pillow-SIMD package for Python
// https://github.com/uploadcare/pillow-simd
#[inline]
pub(crate) fn horiz_convolution(
src_image: TypedImageView<U8x4>,
mut dst_image: TypedImageViewMut<U8x4>,
offset: u32,
coeffs: Coefficients,
) {
let (values, window_size, bounds_per_pixel) =
(coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coefficients_chunks =
normalizer_guard.normalized_i16_chunks(window_size, &bounds_per_pixel);
let dst_height = dst_image.height().get();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_image.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_8u4x(src_rows, dst_rows, &coefficients_chunks, precision);
}
}
let mut yy = dst_height - dst_height % 4;
while yy < dst_height {
unsafe {
horiz_convolution_8u(
src_image.get_row(yy + offset).unwrap(),
dst_image.get_row_mut(yy).unwrap(),
&coefficients_chunks,
precision,
);
}
yy += 1;
}
}
#[inline]
pub(crate) fn vert_convolution(
src_image: TypedImageView<U8x4>,
mut dst_image: TypedImageViewMut<U8x4>,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coeffs_i16 = normalizer_guard.normalized_i16();
let coeffs_chunks = coeffs_i16.chunks(window_size);
let dst_rows = dst_image.iter_rows_mut();
for ((&bound, k), dst_row) in bounds.iter().zip(coeffs_chunks).zip(dst_rows) {
unsafe {
vert_convolution_8u(&src_image, dst_row, k, bound, precision);
}
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - length of all rows in src_rows must be equal
/// - length of all rows in dst_rows must be equal
/// - coefficients_chunks.len() == dst_rows.0.len()
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "sse4.1")]
unsafe fn horiz_convolution_8u4x(
src_rows: FourRows<u32>,
dst_rows: FourRowsMut<u32>,
coefficients_chunks: &[CoefficientsI16Chunk],
precision: u8,
) {
let (s_row0, s_row1, s_row2, s_row3) = src_rows;
let (d_row0, d_row1, d_row2, d_row3) = dst_rows;
let initial = _mm_set1_epi32(1 << (precision - 1));
let mask_lo = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
let mask_hi = _mm_set_epi8(-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8);
let mask = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let x_start = coeffs_chunk.start as usize;
let mut x: usize = 0;
let mut sss0 = initial;
let mut sss1 = initial;
let mut sss2 = initial;
let mut sss3 = initial;
let coeffs = coeffs_chunk.values;
let coeffs_by_4 = coeffs.chunks_exact(4);
let reminder1 = coeffs_by_4.remainder();
for k in coeffs_by_4 {
let mmk_lo = simd_utils::ptr_i16_to_set1_epi32(k, 0);
let mmk_hi = simd_utils::ptr_i16_to_set1_epi32(k, 2);
// [8] a3 b3 g3 r3 a2 b2 g2 r2 a1 b1 g1 r1 a0 b0 g0 r0
let mut source = simd_utils::loadu_si128(s_row0, x + x_start);
// [16] a1 a0 b1 b0 g1 g0 r1 r0
let mut pix = _mm_shuffle_epi8(source, mask_lo);
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk_lo));
// [16] a3 a2 b3 b2 g3 g2 r3 r2
pix = _mm_shuffle_epi8(source, mask_hi);
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk_hi));
source = simd_utils::loadu_si128(s_row1, x + x_start);
pix = _mm_shuffle_epi8(source, mask_lo);
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk_lo));
pix = _mm_shuffle_epi8(source, mask_hi);
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk_hi));
source = simd_utils::loadu_si128(s_row2, x + x_start);
pix = _mm_shuffle_epi8(source, mask_lo);
sss2 = _mm_add_epi32(sss2, _mm_madd_epi16(pix, mmk_lo));
pix = _mm_shuffle_epi8(source, mask_hi);
sss2 = _mm_add_epi32(sss2, _mm_madd_epi16(pix, mmk_hi));
source = simd_utils::loadu_si128(s_row3, x + x_start);
pix = _mm_shuffle_epi8(source, mask_lo);
sss3 = _mm_add_epi32(sss3, _mm_madd_epi16(pix, mmk_lo));
pix = _mm_shuffle_epi8(source, mask_hi);
sss3 = _mm_add_epi32(sss3, _mm_madd_epi16(pix, mmk_hi));
x += 4;
}
let coeffs_by_2 = reminder1.chunks_exact(2);
let reminder2 = coeffs_by_2.remainder();
for k in coeffs_by_2 {
// [16] k1 k0 k1 k0 k1 k0 k1 k0
let mmk = simd_utils::ptr_i16_to_set1_epi32(k, 0);
// [8] x x x x x x x x a1 b1 g1 r1 a0 b0 g0 r0
let mut pix = simd_utils::loadl_epi64(s_row0, x + x_start);
// [16] a1 a0 b1 b0 g1 g0 r1 r0
pix = _mm_shuffle_epi8(pix, mask);
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
pix = simd_utils::loadl_epi64(s_row1, x + x_start);
pix = _mm_shuffle_epi8(pix, mask);
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
pix = simd_utils::loadl_epi64(s_row2, x + x_start);
pix = _mm_shuffle_epi8(pix, mask);
sss2 = _mm_add_epi32(sss2, _mm_madd_epi16(pix, mmk));
pix = simd_utils::loadl_epi64(s_row3, x + x_start);
pix = _mm_shuffle_epi8(pix, mask);
sss3 = _mm_add_epi32(sss3, _mm_madd_epi16(pix, mmk));
x += 2;
}
for &k in reminder2 {
// [16] xx k0 xx k0 xx k0 xx k0
let mmk = _mm_set1_epi32(k as i32);
// [16] xx a0 xx b0 xx g0 xx r0
let mut pix = simd_utils::mm_cvtepu8_epi32(s_row0, x);
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
pix = simd_utils::mm_cvtepu8_epi32(s_row1, x);
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
pix = simd_utils::mm_cvtepu8_epi32(s_row2, x);
sss2 = _mm_add_epi32(sss2, _mm_madd_epi16(pix, mmk));
pix = simd_utils::mm_cvtepu8_epi32(s_row3, x);
sss3 = _mm_add_epi32(sss3, _mm_madd_epi16(pix, mmk));
x += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss0 = _mm_srai_epi32::<$imm8>(sss0);
sss1 = _mm_srai_epi32::<$imm8>(sss1);
sss2 = _mm_srai_epi32::<$imm8>(sss2);
sss3 = _mm_srai_epi32::<$imm8>(sss3);
}};
}
constify_imm8!(precision, call);
sss0 = _mm_packs_epi32(sss0, sss0);
sss1 = _mm_packs_epi32(sss1, sss1);
sss2 = _mm_packs_epi32(sss2, sss2);
sss3 = _mm_packs_epi32(sss3, sss3);
*d_row0.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss0, sss0)));
*d_row1.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss1, sss1)));
*d_row2.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss2, sss2)));
*d_row3.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss3, sss3)));
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - bounds.len() == dst_row.len()
/// - coefficients_chunks.len() == dst_row.len()
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "sse4.1")]
unsafe fn horiz_convolution_8u(
src_row: &[u32],
dst_row: &mut [u32],
coefficients_chunks: &[CoefficientsI16Chunk],
precision: u8,
) {
let initial = _mm_set1_epi32(1 << (precision - 1));
let sh1 = _mm_set_epi8(-1, 11, -1, 3, -1, 10, -1, 2, -1, 9, -1, 1, -1, 8, -1, 0);
let sh2 = _mm_set_epi8(5, 4, 1, 0, 5, 4, 1, 0, 5, 4, 1, 0, 5, 4, 1, 0);
let sh3 = _mm_set_epi8(-1, 15, -1, 7, -1, 14, -1, 6, -1, 13, -1, 5, -1, 12, -1, 4);
let sh4 = _mm_set_epi8(7, 6, 3, 2, 7, 6, 3, 2, 7, 6, 3, 2, 7, 6, 3, 2);
let sh5 = _mm_set_epi8(13, 12, 9, 8, 13, 12, 9, 8, 13, 12, 9, 8, 13, 12, 9, 8);
let sh6 = _mm_set_epi8(
15, 14, 11, 10, 15, 14, 11, 10, 15, 14, 11, 10, 15, 14, 11, 10,
);
let sh7 = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
// for (dst_x, (&bound, k)) in bounds.iter().zip(coeffs_chunks).enumerate() {
let x_start = coeffs_chunk.start as usize;
let mut x: usize = 0;
let mut coeffs = coeffs_chunk.values;
let mut sss = initial;
let coeffs_by_8 = coeffs.chunks_exact(8);
let reminder1 = coeffs_by_8.remainder();
for k in coeffs_by_8 {
let ksource = simd_utils::loadu_si128(k, 0);
let mut source = simd_utils::loadu_si128(src_row, x + x_start);
let mut pix = _mm_shuffle_epi8(source, sh1);
let mut mmk = _mm_shuffle_epi8(ksource, sh2);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
pix = _mm_shuffle_epi8(source, sh3);
mmk = _mm_shuffle_epi8(ksource, sh4);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
source = simd_utils::loadu_si128(src_row, x + 4 + x_start);
pix = _mm_shuffle_epi8(source, sh1);
mmk = _mm_shuffle_epi8(ksource, sh5);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
pix = _mm_shuffle_epi8(source, sh3);
mmk = _mm_shuffle_epi8(ksource, sh6);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
x += 8;
}
let coeffs_by_4 = reminder1.chunks_exact(4);
coeffs = coeffs_by_4.remainder();
for k in coeffs_by_4 {
let source = simd_utils::loadu_si128(src_row, x + x_start);
let ksource = simd_utils::loadl_epi64(k, 0);
let mut pix = _mm_shuffle_epi8(source, sh1);
let mut mmk = _mm_shuffle_epi8(ksource, sh2);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
pix = _mm_shuffle_epi8(source, sh3);
mmk = _mm_shuffle_epi8(ksource, sh4);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
x += 4;
}
let coeffs_by_2 = coeffs.chunks_exact(2);
let reminder1 = coeffs_by_2.remainder();
for k in coeffs_by_2 {
let mmk = simd_utils::ptr_i16_to_set1_epi32(k, 0);
let source = simd_utils::loadl_epi64(src_row, x + x_start);
let pix = _mm_shuffle_epi8(source, sh7);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
x += 2
}
for &k in reminder1 {
let pix = simd_utils::mm_cvtepu8_epi32(src_row, x + x_start);
let mmk = _mm_set1_epi32(k as i32);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
x += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss = _mm_srai_epi32::<$imm8>(sss);
}};
}
constify_imm8!(precision, call);
sss = _mm_packs_epi32(sss, sss);
*dst_row.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss, sss)));
}
}
#[target_feature(enable = "sse4.1")]
pub(crate) unsafe fn vert_convolution_8u(
src_img: &TypedImageView<U8x4>,
dst_row: &mut [u32],
coeffs: &[i16],
bound: Bound,
precision: u8,
) {
let mut xx: usize = 0;
let src_width = src_img.width().get() as usize;
let y_start = bound.start;
let y_size = bound.size;
let initial = _mm_set1_epi32(1 << (precision - 1));
while xx < src_width.saturating_sub(7) {
let mut sss0 = initial;
let mut sss1 = initial;
let mut sss2 = initial;
let mut sss3 = initial;
let mut sss4 = initial;
let mut sss5 = initial;
let mut sss6 = initial;
let mut sss7 = initial;
let mut y: u32 = 0;
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
// Load two coefficients at once
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
let mut source1 = simd_utils::loadu_si128(s_row1, xx); // top line
let mut source2 = simd_utils::loadu_si128(s_row2, xx); // bottom line
let mut source = _mm_unpacklo_epi8(source1, source2);
let mut pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
source = _mm_unpackhi_epi8(source1, source2);
pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss2 = _mm_add_epi32(sss2, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss3 = _mm_add_epi32(sss3, _mm_madd_epi16(pix, mmk));
source1 = simd_utils::loadu_si128(s_row1, xx + 4); // top line
source2 = simd_utils::loadu_si128(s_row2, xx + 4); // bottom line
source = _mm_unpacklo_epi8(source1, source2);
pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss4 = _mm_add_epi32(sss4, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss5 = _mm_add_epi32(sss5, _mm_madd_epi16(pix, mmk));
source = _mm_unpackhi_epi8(source1, source2);
pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss6 = _mm_add_epi32(sss6, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss7 = _mm_add_epi32(sss7, _mm_madd_epi16(pix, mmk));
y += 2;
}
for s_row in src_img.iter_rows(y_start + y, y_start + y_size) {
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
let mut source1 = simd_utils::loadu_si128(s_row, xx); // top line
let mut source = _mm_unpacklo_epi8(source1, _mm_setzero_si128());
let mut pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
source = _mm_unpackhi_epi8(source1, _mm_setzero_si128());
pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss2 = _mm_add_epi32(sss2, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss3 = _mm_add_epi32(sss3, _mm_madd_epi16(pix, mmk));
source1 = simd_utils::loadu_si128(s_row, xx + 4); // top line
source = _mm_unpacklo_epi8(source1, _mm_setzero_si128());
pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss4 = _mm_add_epi32(sss4, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss5 = _mm_add_epi32(sss5, _mm_madd_epi16(pix, mmk));
source = _mm_unpackhi_epi8(source1, _mm_setzero_si128());
pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss6 = _mm_add_epi32(sss6, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss7 = _mm_add_epi32(sss7, _mm_madd_epi16(pix, mmk));
y += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss0 = _mm_srai_epi32::<$imm8>(sss0);
sss1 = _mm_srai_epi32::<$imm8>(sss1);
sss2 = _mm_srai_epi32::<$imm8>(sss2);
sss3 = _mm_srai_epi32::<$imm8>(sss3);
sss4 = _mm_srai_epi32::<$imm8>(sss4);
sss5 = _mm_srai_epi32::<$imm8>(sss5);
sss6 = _mm_srai_epi32::<$imm8>(sss6);
sss7 = _mm_srai_epi32::<$imm8>(sss7);
}};
}
constify_imm8!(precision, call);
sss0 = _mm_packs_epi32(sss0, sss1);
sss2 = _mm_packs_epi32(sss2, sss3);
sss0 = _mm_packus_epi16(sss0, sss2);
let dst_ptr = dst_row.get_unchecked_mut(xx..).as_mut_ptr() as *mut __m128i;
_mm_storeu_si128(dst_ptr, sss0);
sss4 = _mm_packs_epi32(sss4, sss5);
sss6 = _mm_packs_epi32(sss6, sss7);
sss4 = _mm_packus_epi16(sss4, sss6);
let dst_ptr = dst_row.get_unchecked_mut(xx + 4..).as_mut_ptr() as *mut __m128i;
_mm_storeu_si128(dst_ptr, sss4);
xx += 8;
}
while xx < src_width.saturating_sub(1) {
let mut sss0 = initial; // left row
let mut sss1 = initial; // right row
let mut y: u32 = 0;
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
// Load two coefficients at once
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
let source1 = simd_utils::loadl_epi64(s_row1, xx); // top line
let source2 = simd_utils::loadl_epi64(s_row2, xx); // bottom line
let source = _mm_unpacklo_epi8(source1, source2);
let mut pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
y += 2;
}
for s_row1 in src_img.iter_rows(y_start + y, y_start + y_size) {
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
let source1 = simd_utils::loadl_epi64(s_row1, xx); // top line
let source = _mm_unpacklo_epi8(source1, _mm_setzero_si128());
let mut pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
y += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss0 = _mm_srai_epi32::<$imm8>(sss0);
sss1 = _mm_srai_epi32::<$imm8>(sss1);
}};
}
constify_imm8!(precision, call);
sss0 = _mm_packs_epi32(sss0, sss1);
sss0 = _mm_packus_epi16(sss0, sss0);
let dst_ptr = dst_row.get_unchecked_mut(xx..).as_mut_ptr() as *mut __m128i;
_mm_storel_epi64(dst_ptr, sss0);
//
xx += 2;
}
while xx < src_width {
let mut sss = initial;
let mut y: u32 = 0;
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
// Load two coefficients at once
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
let source1 = simd_utils::mm_cvtsi32_si128(s_row1, xx); // top line
let source2 = simd_utils::mm_cvtsi32_si128(s_row2, xx); // bottom line
let source = _mm_unpacklo_epi8(source1, source2);
let pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
y += 2;
}
for s_row in src_img.iter_rows(y_start + y, y_start + y_size) {
let pix = simd_utils::mm_cvtepu8_epi32(s_row, xx);
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
y += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss = _mm_srai_epi32::<$imm8>(sss);
}};
}
constify_imm8!(precision, call);
sss = _mm_packs_epi32(sss, sss);
*dst_row.get_unchecked_mut(xx) = transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss, sss)));
xx += 1;
}
}
+4
View File
@@ -27,3 +27,7 @@ pub enum CropBoxError {
#[error("Size of the crop box is out of the image boundaries")]
SizeIsOutOfImageBoundaries,
}
#[derive(Error, Debug, Clone, Copy)]
#[error("Type of pixels of the source image is not equal to pixel type of the destination image.")]
pub struct DifferentTypesOfPixelsError;
+235
View File
@@ -0,0 +1,235 @@
use std::num::NonZeroU32;
use crate::image_view::{ImageRows, ImageRowsMut, TypedImageView, TypedImageViewMut};
use crate::pixels::Pixel;
use crate::{ImageBufferError, ImageView, ImageViewMut, InvalidBufferSizeError, PixelType};
#[derive(Debug)]
enum PixelsContainer<'a> {
MutU32(&'a mut [u32]),
MutU8(&'a mut [u8]),
VecU32(Vec<u32>),
VecU8(Vec<u8>),
}
/// Simple image container.
#[derive(Debug)]
pub struct Image<'a> {
width: NonZeroU32,
height: NonZeroU32,
pixels: PixelsContainer<'a>,
pixel_type: PixelType,
}
impl<'a> Image<'a> {
/// Create empty image with given dimensions and pixel type.
pub fn new(width: NonZeroU32, height: NonZeroU32, pixel_type: PixelType) -> Self {
let size = (width.get() * height.get()) as usize;
let pixels = if let PixelType::U8 = pixel_type {
PixelsContainer::VecU8(vec![0; size])
} else {
PixelsContainer::VecU32(vec![0; size])
};
Self {
width,
height,
pixels,
pixel_type,
}
}
pub fn from_vec_u32(
width: NonZeroU32,
height: NonZeroU32,
buffer: Vec<u32>,
pixel_type: PixelType,
) -> Result<Self, InvalidBufferSizeError> {
let size = (width.get() * height.get()) as usize;
if buffer.len() != size {
return Err(InvalidBufferSizeError);
}
Ok(Self {
width,
height,
pixels: PixelsContainer::VecU32(buffer),
pixel_type,
})
}
pub fn from_vec_u8(
width: NonZeroU32,
height: NonZeroU32,
buffer: Vec<u8>,
pixel_type: PixelType,
) -> Result<Self, ImageBufferError> {
let size = (width.get() * height.get()) as usize * pixel_type.size();
if buffer.len() != size {
return Err(ImageBufferError::InvalidBufferSize);
}
if !pixel_type.is_aligned(&buffer) {
return Err(ImageBufferError::InvalidBufferAlignment);
}
Ok(Self {
width,
height,
pixels: PixelsContainer::VecU8(buffer),
pixel_type,
})
}
pub fn from_slice_u32(
width: NonZeroU32,
height: NonZeroU32,
buffer: &'a mut [u32],
pixel_type: PixelType,
) -> Result<Self, InvalidBufferSizeError> {
let size = (width.get() * height.get()) as usize;
if buffer.len() != size {
return Err(InvalidBufferSizeError);
}
Ok(Self {
width,
height,
pixels: PixelsContainer::MutU32(buffer),
pixel_type,
})
}
pub fn from_slice_u8(
width: NonZeroU32,
height: NonZeroU32,
buffer: &'a mut [u8],
pixel_type: PixelType,
) -> Result<Self, ImageBufferError> {
let size = (width.get() * height.get()) as usize * pixel_type.size();
if buffer.len() != size {
return Err(ImageBufferError::InvalidBufferSize);
}
if !pixel_type.is_aligned(buffer) {
return Err(ImageBufferError::InvalidBufferAlignment);
}
Ok(Self {
width,
height,
pixels: PixelsContainer::MutU8(buffer),
pixel_type,
})
}
#[inline(always)]
pub fn pixel_type(&self) -> PixelType {
self.pixel_type
}
#[inline(always)]
pub fn width(&self) -> NonZeroU32 {
self.width
}
#[inline(always)]
pub fn height(&self) -> NonZeroU32 {
self.height
}
/// Buffer with image pixels.
#[inline(always)]
pub fn buffer(&self) -> &[u8] {
match &self.pixels {
PixelsContainer::MutU32(p) => unsafe { p.align_to::<u8>().1 },
PixelsContainer::MutU8(p) => *p,
PixelsContainer::VecU32(v) => unsafe { v.align_to::<u8>().1 },
PixelsContainer::VecU8(v) => v,
}
}
#[inline(always)]
fn buffer_mut(&mut self) -> &mut [u8] {
match &mut self.pixels {
PixelsContainer::MutU32(p) => unsafe { p.align_to_mut::<u8>().1 },
PixelsContainer::MutU8(p) => p,
PixelsContainer::VecU32(ref mut v) => unsafe { v.align_to_mut::<u8>().1 },
PixelsContainer::VecU8(ref mut v) => v.as_mut_slice(),
}
}
#[inline(always)]
pub fn view(&self) -> ImageView {
let buffer = self.buffer();
let rows = match self.pixel_type {
PixelType::U8x4 => {
let pixels = unsafe { buffer.align_to::<u32>().1 };
ImageRows::U8x4(pixels.chunks(self.width.get() as usize).collect())
}
PixelType::I32 => {
let pixels = unsafe { buffer.align_to::<i32>().1 };
ImageRows::I32(pixels.chunks(self.width.get() as usize).collect())
}
PixelType::F32 => {
let pixels = unsafe { buffer.align_to::<f32>().1 };
ImageRows::F32(pixels.chunks(self.width.get() as usize).collect())
}
PixelType::U8 => ImageRows::U8(buffer.chunks(self.width.get() as usize).collect()),
};
ImageView::new(self.width, self.height, rows).unwrap()
}
#[inline(always)]
pub fn view_mut(&mut self) -> ImageViewMut {
let pixel_type = self.pixel_type;
let width = self.width;
let height = self.height;
let buffer = self.buffer_mut();
let rows = match pixel_type {
PixelType::U8x4 => {
let pixels = unsafe { buffer.align_to_mut::<u32>().1 };
ImageRowsMut::U8x4(pixels.chunks_mut(width.get() as usize).collect())
}
PixelType::I32 => {
let pixels = unsafe { buffer.align_to_mut::<i32>().1 };
ImageRowsMut::I32(pixels.chunks_mut(width.get() as usize).collect())
}
PixelType::F32 => {
let pixels = unsafe { buffer.align_to_mut::<f32>().1 };
ImageRowsMut::F32(pixels.chunks_mut(width.get() as usize).collect())
}
PixelType::U8 => ImageRowsMut::U8(buffer.chunks_mut(width.get() as usize).collect()),
};
ImageViewMut::new(width, height, rows).unwrap()
}
}
/// Generic image container for internal purposes.
pub(crate) struct InnerImage<'a, P>
where
P: Pixel,
{
width: NonZeroU32,
height: NonZeroU32,
rows: Vec<&'a mut [P::Type]>,
}
impl<'a, P> InnerImage<'a, P>
where
P: Pixel,
{
pub fn new(width: NonZeroU32, height: NonZeroU32, pixels: &'a mut [P::Type]) -> Self {
let rows = pixels.chunks_mut(width.get() as usize).collect();
Self {
width,
height,
rows,
}
}
#[inline(always)]
pub fn src_view<'s>(&'s self) -> TypedImageView<'s, 'a, P> {
let rows = self.rows.as_slice();
let rows: &[&[P::Type]] = unsafe { std::mem::transmute(rows) };
TypedImageView::new(self.width, self.height, rows)
}
#[inline(always)]
pub fn dst_view<'s>(&'s mut self) -> TypedImageViewMut<'s, 'a, P> {
TypedImageViewMut::new(self.width, self.height, self.rows.as_mut_slice())
}
}
-162
View File
@@ -1,162 +0,0 @@
use std::num::NonZeroU32;
use crate::{DstImageView, ImageBufferError, InvalidBufferSizeError, PixelType, SrcImageView};
#[derive(Debug)]
enum PixelsContainer<'a> {
Mut(&'a mut [u32]),
VecU32(Vec<u32>),
VecU8(Vec<u8>),
}
#[derive(Debug)]
pub struct ImageData<'a> {
width: NonZeroU32,
height: NonZeroU32,
pixels: PixelsContainer<'a>,
pixel_type: PixelType,
}
impl<'a> ImageData<'a> {
pub fn new(width: NonZeroU32, height: NonZeroU32, pixel_type: PixelType) -> Self {
let size = (width.get() * height.get()) as usize;
let pixels = vec![0; size];
Self {
width,
height,
pixels: PixelsContainer::VecU32(pixels),
pixel_type,
}
}
pub fn from_vec_u32(
width: NonZeroU32,
height: NonZeroU32,
pixels: Vec<u32>,
pixel_type: PixelType,
) -> Result<Self, InvalidBufferSizeError> {
let size = (width.get() * height.get()) as usize;
if pixels.len() != size {
return Err(InvalidBufferSizeError);
}
Ok(Self {
width,
height,
pixels: PixelsContainer::VecU32(pixels),
pixel_type,
})
}
pub fn from_vec_u8(
width: NonZeroU32,
height: NonZeroU32,
mut pixels: Vec<u8>,
pixel_type: PixelType,
) -> Result<Self, ImageBufferError> {
let size = (width.get() * height.get()) as usize * 4;
if pixels.len() != size {
return Err(ImageBufferError::InvalidBufferSize);
}
let (head, _, _) = unsafe { &pixels.align_to_mut::<u32>() };
if !head.is_empty() {
return Err(ImageBufferError::InvalidBufferAlignment);
}
Ok(Self {
width,
height,
pixels: PixelsContainer::VecU8(pixels),
pixel_type,
})
}
pub fn from_slice_u32(
width: NonZeroU32,
height: NonZeroU32,
pixels: &'a mut [u32],
pixel_type: PixelType,
) -> Result<Self, InvalidBufferSizeError> {
let size = (width.get() * height.get()) as usize;
if pixels.len() != size {
return Err(InvalidBufferSizeError);
}
Ok(Self {
width,
height,
pixels: PixelsContainer::Mut(pixels),
pixel_type,
})
}
pub fn from_slice_u8(
width: NonZeroU32,
height: NonZeroU32,
buffer: &'a mut [u8],
pixel_type: PixelType,
) -> Result<Self, ImageBufferError> {
let size = (width.get() * height.get()) as usize * 4;
if buffer.len() != size {
return Err(ImageBufferError::InvalidBufferSize);
}
let (head, pixels, _) = unsafe { buffer.align_to_mut::<u32>() };
if !head.is_empty() {
return Err(ImageBufferError::InvalidBufferAlignment);
}
Ok(Self {
width,
height,
pixels: PixelsContainer::Mut(pixels),
pixel_type,
})
}
#[inline(always)]
pub fn pixel_type(&self) -> PixelType {
self.pixel_type
}
#[inline(always)]
pub fn width(&self) -> NonZeroU32 {
self.width
}
#[inline(always)]
pub fn height(&self) -> NonZeroU32 {
self.height
}
#[inline(always)]
pub fn get_pixels(&self) -> &[u32] {
match &self.pixels {
PixelsContainer::Mut(p) => p,
PixelsContainer::VecU32(v) => v,
PixelsContainer::VecU8(v) => unsafe { v.align_to::<u32>().1 },
}
}
#[inline(always)]
pub fn get_buffer(&self) -> &[u8] {
let pixels = self.get_pixels();
let (_, buffer, _) = unsafe { pixels.align_to::<u8>() };
buffer
}
#[inline(always)]
pub fn src_view(&self) -> SrcImageView {
let pixels = self.get_pixels();
let rows = pixels.chunks(self.width.get() as usize).collect();
SrcImageView::from_rows(self.width, self.height, rows, self.pixel_type).unwrap()
}
#[inline(always)]
pub fn dst_view(&mut self) -> DstImageView {
let rows = match &mut self.pixels {
PixelsContainer::Mut(p) => p.chunks_mut(self.width.get() as usize).collect(),
PixelsContainer::VecU32(v) => v.chunks_mut(self.width.get() as usize).collect(),
PixelsContainer::VecU8(v) => {
let p = unsafe { v.align_to_mut::<u32>().1 };
p.chunks_mut(self.width.get() as usize).collect()
}
};
DstImageView::from_rows(self.width, self.height, rows, self.pixel_type).unwrap()
}
}
+449 -212
View File
@@ -1,26 +1,20 @@
use std::mem::transmute;
use std::num::NonZeroU32;
use std::slice;
use crate::errors::{CropBoxError, ImageBufferError, ImageRowsError, InvalidBufferSizeError};
use crate::errors::{CropBoxError, ImageBufferError, ImageRowsError};
use crate::pixels::{Pixel, PixelType, U8x4, F32, I32, U8};
pub(crate) type TwoRows<'a> = (&'a [u32], &'a [u32]);
pub(crate) type FourRows<'a> = (&'a [u32], &'a [u32], &'a [u32], &'a [u32]);
pub(crate) type RowMut<'a, 'b> = &'b mut &'a mut [u32];
pub(crate) type FourRowsMut<'a, 'b> = (
&'b mut &'a mut [u32],
&'b mut &'a mut [u32],
&'b mut &'a mut [u32],
&'b mut &'a mut [u32],
pub(crate) type RowMut<'a, 'b, T> = &'a mut &'b mut [T];
pub(crate) type TwoRows<'a, T> = (&'a [T], &'a [T]);
pub(crate) type FourRows<'a, T> = (&'a [T], &'a [T], &'a [T], &'a [T]);
pub(crate) type FourRowsMut<'a, 'b, T> = (
&'a mut &'b mut [T],
&'a mut &'b mut [T],
&'a mut &'b mut [T],
&'a mut &'b mut [T],
);
#[derive(Debug, Clone, Copy, PartialEq)]
pub enum PixelType {
U8x4,
I32,
F32,
}
/// Parameters of crop box that may be used with [`ImageView`]
#[derive(Debug, Clone, Copy)]
pub struct CropBox {
pub left: u32,
@@ -29,39 +23,88 @@ pub struct CropBox {
pub height: NonZeroU32,
}
/// An immutable rows of image.
#[derive(Debug, Clone)]
pub enum ImageRows<'a> {
U8x4(Vec<&'a [u32]>),
I32(Vec<&'a [i32]>),
F32(Vec<&'a [f32]>),
U8(Vec<&'a [u8]>),
}
impl<'a> ImageRows<'a> {
pub(crate) fn check_size(
&self,
width: NonZeroU32,
height: NonZeroU32,
) -> Result<(), ImageRowsError> {
match self {
ImageRows::U8x4(rows) => check_rows_count_and_size(width, height, rows),
ImageRows::I32(rows) => check_rows_count_and_size(width, height, rows),
ImageRows::F32(rows) => check_rows_count_and_size(width, height, rows),
ImageRows::U8(rows) => check_rows_count_and_size(width, height, rows),
}
}
pub fn pixel_type(&self) -> PixelType {
match self {
Self::U8x4(_) => PixelType::U8x4,
Self::I32(_) => PixelType::I32,
Self::F32(_) => PixelType::F32,
Self::U8(_) => PixelType::U8,
}
}
}
/// A mutable rows of image.
#[derive(Debug)]
pub enum ImageRowsMut<'a> {
U8x4(Vec<&'a mut [u32]>),
I32(Vec<&'a mut [i32]>),
F32(Vec<&'a mut [f32]>),
U8(Vec<&'a mut [u8]>),
}
impl<'a> ImageRowsMut<'a> {
pub(crate) fn check_size(
&self,
width: NonZeroU32,
height: NonZeroU32,
) -> Result<(), ImageRowsError> {
match self {
Self::U8x4(rows) => check_rows_count_and_size(width, height, rows),
Self::I32(rows) => check_rows_count_and_size(width, height, rows),
Self::F32(rows) => check_rows_count_and_size(width, height, rows),
Self::U8(rows) => check_rows_count_and_size(width, height, rows),
}
}
pub fn pixel_type(&self) -> PixelType {
match self {
Self::U8x4(_) => PixelType::U8x4,
Self::I32(_) => PixelType::I32,
Self::F32(_) => PixelType::F32,
Self::U8(_) => PixelType::U8,
}
}
}
/// An immutable view of image data used by resizer as source image.
#[derive(Debug, Clone)]
pub struct SrcImageView<'a> {
pub struct ImageView<'a> {
width: NonZeroU32,
height: NonZeroU32,
crop_box: CropBox,
rows: Vec<&'a [u32]>,
pixel_type: PixelType,
rows: ImageRows<'a>,
}
/// An mutable view of image data used by resizer as destination image.
#[derive(Debug)]
pub struct DstImageView<'a> {
width: NonZeroU32,
height: NonZeroU32,
rows: Vec<&'a mut [u32]>,
pixel_type: PixelType,
}
impl<'a> SrcImageView<'a> {
pub fn from_rows(
impl<'a> ImageView<'a> {
pub fn new(
width: NonZeroU32,
height: NonZeroU32,
rows: Vec<&'a [u32]>,
pixel_type: PixelType,
rows: ImageRows<'a>,
) -> Result<Self, ImageRowsError> {
if rows.len() != height.get() as usize {
return Err(ImageRowsError::InvalidRowsCount);
}
let row_size = width.get() as usize;
if rows.iter().any(|row| row.len() != row_size) {
return Err(ImageRowsError::InvalidRowSize);
}
rows.check_size(width, height)?;
Ok(Self {
width,
height,
@@ -72,7 +115,6 @@ impl<'a> SrcImageView<'a> {
height,
},
rows,
pixel_type,
})
}
@@ -82,36 +124,41 @@ impl<'a> SrcImageView<'a> {
buffer: &'a [u8],
pixel_type: PixelType,
) -> Result<Self, ImageBufferError> {
let size = (width.get() * height.get()) as usize * 4;
let size = (width.get() * height.get()) as usize * pixel_type.size();
if buffer.len() != size {
return Err(ImageBufferError::InvalidBufferSize);
}
let (head, pixels, _) = unsafe { buffer.align_to::<u32>() };
if !head.is_empty() {
return Err(ImageBufferError::InvalidBufferAlignment);
}
let rows = pixels.chunks(width.get() as usize).collect();
Ok(Self::from_rows(width, height, rows, pixel_type).unwrap())
}
pub fn from_pixels(
width: NonZeroU32,
height: NonZeroU32,
pixels: &'a [u32],
pixel_type: PixelType,
) -> Result<Self, InvalidBufferSizeError> {
let size = (width.get() * height.get()) as usize;
if pixels.len() != size {
return Err(InvalidBufferSizeError);
}
let rows = pixels.chunks(width.get() as usize).collect();
Ok(Self::from_rows(width, height, rows, pixel_type).unwrap())
let rows = match pixel_type {
PixelType::U8x4 => {
let pixels = align_buffer_to(buffer)?;
ImageRows::U8x4(pixels.chunks(width.get() as usize).collect())
}
PixelType::I32 => {
let pixels = align_buffer_to(buffer)?;
ImageRows::I32(pixels.chunks(width.get() as usize).collect())
}
PixelType::F32 => {
let pixels = align_buffer_to(buffer)?;
ImageRows::F32(pixels.chunks(width.get() as usize).collect())
}
PixelType::U8 => ImageRows::U8(buffer.chunks(width.get() as usize).collect()),
};
Ok(Self {
width,
height,
crop_box: CropBox {
left: 0,
top: 0,
width,
height,
},
rows,
})
}
#[inline(always)]
pub fn pixel_type(&self) -> PixelType {
self.pixel_type
self.rows.pixel_type()
}
#[inline(always)]
@@ -149,10 +196,10 @@ impl<'a> SrcImageView<'a> {
/// center cropping (e.g. if cropping the width, take 50% off
/// of the left side, and therefore 50% off the right side).
/// (0.0, 0.0) will crop from the top left corner (i.e. if
/// cropping the width, take all of the crop off of the right
/// cropping the width, take all the crop off of the right
/// side, and if cropping the height, take all of it off the
/// bottom). (1.0, 0.0) will crop from the bottom left
/// corner, etc. (i.e. if cropping the width, take all of the
/// corner, etc. (i.e. if cropping the width, take all the
/// crop off the left side, and if cropping the height take
/// none from the top, and therefore all off the bottom).
pub fn set_crop_box_to_fit_dst_size(
@@ -204,163 +251,86 @@ impl<'a> SrcImageView<'a> {
.unwrap();
}
#[inline(always)]
pub fn get_buffer(&self) -> Vec<u8> {
let row_size = self.width.get() as usize;
self.rows
.iter()
.map(|row| unsafe { row[0..row_size].align_to::<u8>().1 })
.flatten()
.copied()
.collect()
}
#[inline]
pub(crate) fn get_pixel_u32(&self, x: u32, y: u32) -> u32 {
self.rows[y as usize][x as usize]
}
#[inline(always)]
pub(crate) fn get_pixel_i32(&self, x: u32, y: u32) -> i32 {
unsafe { transmute(self.get_pixel_u32(x, y)) }
}
#[inline(always)]
pub(crate) fn get_pixel_f32(&self, x: u32, y: u32) -> f32 {
f32::from_bits(self.get_pixel_u32(x, y))
}
#[inline(always)]
pub(crate) fn iter_4_rows(
&'a self,
start_y: u32,
max_y: u32,
) -> impl Iterator<Item = FourRows<'a>> {
let start_y = start_y as usize;
let max_y = max_y.min(self.height.get()) as usize;
let rows = self.rows.get(start_y..max_y).unwrap_or_else(|| &[]);
rows.chunks_exact(4).map(|rows| match *rows {
[r0, r1, r2, r3] => (r0, r1, r2, r3),
_ => unreachable!(),
})
}
#[inline(always)]
pub(crate) fn iter_2_rows(
&'a self,
start_y: u32,
max_y: u32,
) -> impl Iterator<Item = TwoRows<'a>> {
let start_y = start_y as usize;
let max_y = max_y.min(self.height.get()) as usize;
let rows = self.rows.get(start_y..max_y).unwrap_or_else(|| &[]);
rows.chunks_exact(2).map(|rows| match *rows {
[r0, r1] => (r0, r1),
_ => unreachable!(),
})
}
#[inline(always)]
pub(crate) fn iter_rows(&'a self, start_y: u32, max_y: u32) -> impl Iterator<Item = &'a [u32]> {
let start_y = start_y as usize;
let max_y = max_y.min(self.height.get()) as usize;
let rows = self.rows.get(start_y..max_y).unwrap_or_else(|| &[]);
rows.iter().copied()
}
#[inline(always)]
pub(crate) fn iter_horiz(&self, x: u32, y: u32) -> &[u32] {
if let Some(&row) = self.rows.get(y as usize) {
let start_pos = x as usize;
if let Some(res) = row.get(start_pos..) {
return res;
}
pub(crate) fn u32_image(&self) -> Option<TypedImageView<U8x4>> {
if let ImageRows::U8x4(ref rows) = self.rows {
Some(TypedImageView {
width: self.width,
height: self.height,
crop_box: self.crop_box,
rows,
})
} else {
None
}
&[]
}
#[inline]
pub(crate) fn iter_horiz_i32(&self, x: u32, y: u32) -> &[i32] {
let row = self.iter_horiz(x, y);
let ptr = row.as_ptr();
unsafe { slice::from_raw_parts(ptr as *const i32, row.len()) }
pub(crate) fn i32_image(&self) -> Option<TypedImageView<I32>> {
if let ImageRows::I32(ref rows) = self.rows {
Some(TypedImageView {
width: self.width,
height: self.height,
crop_box: self.crop_box,
rows,
})
} else {
None
}
}
#[inline]
pub(crate) fn iter_horiz_f32(&self, x: u32, y: u32) -> &[f32] {
let row = self.iter_horiz(x, y);
let ptr = row.as_ptr();
unsafe { slice::from_raw_parts(ptr as *const f32, row.len()) }
pub(crate) fn f32_image(&self) -> Option<TypedImageView<F32>> {
if let ImageRows::F32(ref rows) = self.rows {
Some(TypedImageView {
width: self.width,
height: self.height,
crop_box: self.crop_box,
rows,
})
} else {
None
}
}
#[inline(always)]
pub(crate) fn get_row(&self, y: u32) -> Option<&[u32]> {
self.rows.get(y as usize).copied()
}
#[inline(always)]
pub(crate) fn iter_rows_with_step(
&self,
mut y: f64,
step: f64,
max_count: usize,
) -> impl Iterator<Item = &[u32]> {
let steps = (self.height.get() as f64 - y) / step;
let steps = (steps.max(0.).ceil() as usize).min(max_count);
(0..steps).map(move |_| {
// Safety of value of y guaranteed by calculation of steps count
let row = unsafe { *self.rows.get_unchecked(y as usize) };
y += step;
row
})
pub(crate) fn u8_image(&self) -> Option<TypedImageView<U8>> {
if let ImageRows::U8(ref rows) = self.rows {
Some(TypedImageView {
width: self.width,
height: self.height,
crop_box: self.crop_box,
rows,
})
} else {
None
}
}
}
impl<'a> DstImageView<'a> {
#[inline(always)]
pub fn from_rows(
width: NonZeroU32,
height: NonZeroU32,
rows: Vec<&'a mut [u32]>,
pixel_type: PixelType,
) -> Result<Self, ImageRowsError> {
if rows.len() != height.get() as usize {
return Err(ImageRowsError::InvalidRowsCount);
}
let row_size = width.get() as usize;
if rows.iter().any(|row| row.len() != row_size) {
return Err(ImageRowsError::InvalidRowSize);
}
Ok(Self {
/// Generic immutable image view.
pub(crate) struct TypedImageView<'a, 'b, P>
where
P: Pixel,
{
width: NonZeroU32,
height: NonZeroU32,
crop_box: CropBox,
rows: &'a [&'b [P::Type]],
}
impl<'a, 'b, P> TypedImageView<'a, 'b, P>
where
P: Pixel,
{
pub fn new(width: NonZeroU32, height: NonZeroU32, rows: &'a [&'b [P::Type]]) -> Self {
Self {
width,
height,
crop_box: CropBox {
left: 0,
top: 0,
width,
height,
},
rows,
pixel_type,
})
}
pub fn from_buffer(
width: NonZeroU32,
height: NonZeroU32,
buffer: &'a mut [u8],
pixel_type: PixelType,
) -> Result<Self, ImageBufferError> {
let size = (width.get() * height.get()) as usize * 4;
if buffer.len() != size {
return Err(ImageBufferError::InvalidBufferSize);
}
let (head, pixels, _) = unsafe { buffer.align_to_mut::<u32>() };
if !head.is_empty() {
return Err(ImageBufferError::InvalidBufferAlignment);
}
let rows = pixels.chunks_mut(width.get() as usize).collect();
Ok(Self::from_rows(width, height, rows, pixel_type).unwrap())
}
#[inline(always)]
pub fn pixel_type(&self) -> PixelType {
self.pixel_type
}
#[inline(always)]
@@ -374,12 +344,248 @@ impl<'a> DstImageView<'a> {
}
#[inline(always)]
pub(crate) fn iter_rows_mut(&mut self) -> slice::IterMut<&'a mut [u32]> {
pub fn crop_box(&self) -> CropBox {
self.crop_box
}
#[inline]
pub(crate) fn get_pixel(&self, x: u32, y: u32) -> P::Type {
self.rows[y as usize][x as usize]
}
#[inline(always)]
pub(crate) fn iter_4_rows<'s>(
&'s self,
start_y: u32,
max_y: u32,
) -> impl Iterator<Item = FourRows<'b, P::Type>> + 's {
let start_y = start_y as usize;
let max_y = max_y.min(self.height.get()) as usize;
let rows = self.rows.get(start_y..max_y).unwrap_or_else(|| &[]);
rows.chunks_exact(4).map(|rows| match *rows {
[r0, r1, r2, r3] => (r0, r1, r2, r3),
_ => unreachable!(),
})
}
#[inline(always)]
pub(crate) fn iter_2_rows<'s>(
&'s self,
start_y: u32,
max_y: u32,
) -> impl Iterator<Item = TwoRows<'b, P::Type>> + 's {
let start_y = start_y as usize;
let max_y = max_y.min(self.height.get()) as usize;
let rows = self.rows.get(start_y..max_y).unwrap_or_else(|| &[]);
rows.chunks_exact(2).map(|rows| match *rows {
[r0, r1] => (r0, r1),
_ => unreachable!(),
})
}
#[inline(always)]
pub(crate) fn iter_rows<'s>(
&'s self,
start_y: u32,
max_y: u32,
) -> impl Iterator<Item = &'b [P::Type]> + 's {
let start_y = start_y as usize;
let max_y = max_y.min(self.height.get()) as usize;
let rows = self.rows.get(start_y..max_y).unwrap_or_else(|| &[]);
rows.iter().copied()
}
#[inline(always)]
pub(crate) fn iter_horiz(&self, x: u32, y: u32) -> &'b [P::Type] {
if let Some(&row) = self.rows.get(y as usize) {
let start_pos = x as usize;
if let Some(res) = row.get(start_pos..) {
return res;
}
}
&[]
}
#[inline(always)]
pub(crate) fn get_row(&self, y: u32) -> Option<&'b [P::Type]> {
self.rows.get(y as usize).copied()
}
#[inline(always)]
pub(crate) fn iter_rows_with_step<'s>(
&'s self,
mut y: f64,
step: f64,
max_count: usize,
) -> impl Iterator<Item = &'b [P::Type]> + 's {
let steps = (self.height.get() as f64 - y) / step;
let steps = (steps.max(0.).ceil() as usize).min(max_count);
(0..steps).map(move |_| {
// Safety of value of y guaranteed by calculation of steps count
let row = unsafe { *self.rows.get_unchecked(y as usize) };
y += step;
row
})
}
}
/// A mutable view of image data used by resizer as destination image.
#[derive(Debug)]
pub struct ImageViewMut<'a> {
width: NonZeroU32,
height: NonZeroU32,
rows: ImageRowsMut<'a>,
}
impl<'a> ImageViewMut<'a> {
pub fn new(
width: NonZeroU32,
height: NonZeroU32,
rows: ImageRowsMut<'a>,
) -> Result<Self, ImageRowsError> {
rows.check_size(width, height)?;
Ok(Self {
width,
height,
rows,
})
}
pub fn from_buffer(
width: NonZeroU32,
height: NonZeroU32,
buffer: &'a mut [u8],
pixel_type: PixelType,
) -> Result<Self, ImageBufferError> {
let size = (width.get() * height.get()) as usize * pixel_type.size();
if buffer.len() != size {
return Err(ImageBufferError::InvalidBufferSize);
}
let rows = match pixel_type {
PixelType::U8x4 => {
let pixels = align_buffer_to_mut(buffer)?;
ImageRowsMut::U8x4(pixels.chunks_mut(width.get() as usize).collect())
}
PixelType::I32 => {
let pixels = align_buffer_to_mut(buffer)?;
ImageRowsMut::I32(pixels.chunks_mut(width.get() as usize).collect())
}
PixelType::F32 => {
let pixels = align_buffer_to_mut(buffer)?;
ImageRowsMut::F32(pixels.chunks_mut(width.get() as usize).collect())
}
PixelType::U8 => ImageRowsMut::U8(buffer.chunks_mut(width.get() as usize).collect()),
};
Ok(Self {
width,
height,
rows,
})
}
#[inline(always)]
pub fn pixel_type(&self) -> PixelType {
self.rows.pixel_type()
}
#[inline(always)]
pub fn width(&self) -> NonZeroU32 {
self.width
}
#[inline(always)]
pub fn height(&self) -> NonZeroU32 {
self.height
}
pub(crate) fn u32_image<'s>(&'s mut self) -> Option<TypedImageViewMut<'s, 'a, U8x4>> {
if let ImageRowsMut::U8x4(rows) = &mut self.rows {
Some(TypedImageViewMut {
width: self.width,
height: self.height,
rows,
})
} else {
None
}
}
pub(crate) fn i32_image<'s>(&'s mut self) -> Option<TypedImageViewMut<'s, 'a, I32>> {
if let ImageRowsMut::I32(rows) = &mut self.rows {
Some(TypedImageViewMut {
width: self.width,
height: self.height,
rows,
})
} else {
None
}
}
pub(crate) fn f32_image<'s>(&'s mut self) -> Option<TypedImageViewMut<'s, 'a, F32>> {
if let ImageRowsMut::F32(rows) = &mut self.rows {
Some(TypedImageViewMut {
width: self.width,
height: self.height,
rows,
})
} else {
None
}
}
pub(crate) fn u8_image<'s>(&'s mut self) -> Option<TypedImageViewMut<'s, 'a, U8>> {
if let ImageRowsMut::U8(rows) = &mut self.rows {
Some(TypedImageViewMut {
width: self.width,
height: self.height,
rows,
})
} else {
None
}
}
}
/// Generic mutable image view.
pub(crate) struct TypedImageViewMut<'a, 'b, P>
where
P: Pixel,
{
width: NonZeroU32,
height: NonZeroU32,
rows: &'a mut [&'b mut [P::Type]],
}
impl<'a, 'b, P> TypedImageViewMut<'a, 'b, P>
where
P: Pixel,
{
pub fn new(width: NonZeroU32, height: NonZeroU32, rows: &'a mut [&'b mut [P::Type]]) -> Self {
Self {
width,
height,
rows,
}
}
#[inline(always)]
pub fn width(&self) -> NonZeroU32 {
self.width
}
#[inline(always)]
pub fn height(&self) -> NonZeroU32 {
self.height
}
#[inline(always)]
pub fn iter_rows_mut(&mut self) -> slice::IterMut<&'b mut [P::Type]> {
self.rows.iter_mut()
}
#[inline(always)]
pub(crate) fn iter_4_rows_mut(&mut self) -> impl Iterator<Item = FourRowsMut<'a, '_>> {
pub fn iter_4_rows_mut<'s>(&'s mut self) -> impl Iterator<Item = FourRowsMut<'s, 'b, P::Type>> {
self.rows.chunks_exact_mut(4).map(|rows| match rows {
[a, b, c, d] => (a, b, c, d),
_ => unreachable!(),
@@ -387,7 +593,38 @@ impl<'a> DstImageView<'a> {
}
#[inline(always)]
pub(crate) fn get_row_mut(&mut self, y: u32) -> Option<RowMut<'a, '_>> {
pub fn get_row_mut<'s>(&'s mut self, y: u32) -> Option<RowMut<'s, 'b, P::Type>> {
self.rows.get_mut(y as usize)
}
}
fn check_rows_count_and_size<T>(
width: NonZeroU32,
height: NonZeroU32,
rows: &[impl AsRef<[T]>],
) -> Result<(), ImageRowsError> {
if rows.len() != height.get() as usize {
return Err(ImageRowsError::InvalidRowsCount);
}
let row_size = width.get() as usize;
if rows.iter().any(|row| row.as_ref().len() != row_size) {
return Err(ImageRowsError::InvalidRowSize);
}
Ok(())
}
fn align_buffer_to<T>(buffer: &[u8]) -> Result<&[T], ImageBufferError> {
let (head, pixels, _) = unsafe { buffer.align_to::<T>() };
if !head.is_empty() {
return Err(ImageBufferError::InvalidBufferAlignment);
}
Ok(pixels)
}
fn align_buffer_to_mut<T>(buffer: &mut [u8]) -> Result<&mut [T], ImageBufferError> {
let (head, pixels, _) = unsafe { buffer.align_to_mut::<T>() };
if !head.is_empty() {
return Err(ImageBufferError::InvalidBufferAlignment);
}
Ok(pixels)
}
+7 -4
View File
@@ -2,16 +2,19 @@
pub use alpha::{MulDiv, MulDivImageError, MulDivImagesError};
pub use convolution::FilterType;
pub use errors::{CropBoxError, ImageBufferError, ImageRowsError, InvalidBufferSizeError};
pub use image_data::ImageData;
pub use image_view::{CropBox, DstImageView, PixelType, SrcImageView};
pub use errors::*;
pub use image_view::{CropBox, ImageRows, ImageRowsMut, ImageView, ImageViewMut};
pub use pixels::PixelType;
pub use resizer::{CpuExtensions, ResizeAlg, Resizer};
pub use crate::image::Image;
mod alpha;
mod convolution;
mod errors;
mod image_data;
mod image;
mod image_view;
mod pixels;
mod resizer;
#[cfg(target_arch = "x86_64")]
mod simd_utils;
+56
View File
@@ -0,0 +1,56 @@
use std::mem::size_of;
#[derive(Debug, Clone, Copy, PartialEq)]
pub enum PixelType {
U8x4,
I32,
F32,
U8,
}
impl PixelType {
pub(crate) fn size(&self) -> usize {
match self {
Self::U8 => 1,
_ => 4,
}
}
pub(crate) fn is_aligned(&self, buffer: &[u8]) -> bool {
match self {
Self::U8x4 => unsafe { buffer.align_to::<u32>().0.is_empty() },
Self::I32 => unsafe { buffer.align_to::<i32>().0.is_empty() },
Self::F32 => unsafe { buffer.align_to::<f32>().0.is_empty() },
Self::U8 => true,
}
}
}
pub(crate) trait Pixel {
type Type: Copy;
fn size() -> usize {
size_of::<Self::Type>()
}
fn pixel_type() -> PixelType;
}
macro_rules! pixel_struct {
($name:ident, $type:tt, $pixel_type:expr) => {
pub struct $name;
impl Pixel for $name {
type Type = $type;
fn pixel_type() -> PixelType {
$pixel_type
}
}
};
}
pixel_struct!(U8x4, u32, PixelType::U8x4);
pixel_struct!(I32, i32, PixelType::I32);
pixel_struct!(F32, f32, PixelType::F32);
pixel_struct!(U8, u8, PixelType::U8);
+96 -69
View File
@@ -1,8 +1,10 @@
use std::num::NonZeroU32;
use crate::convolution::{self, Convolution, FilterType};
use crate::image_data::ImageData;
use crate::image_view::{DstImageView, PixelType, SrcImageView};
use crate::errors::DifferentTypesOfPixelsError;
use crate::image::InnerImage;
use crate::image_view::{ImageView, ImageViewMut, TypedImageView, TypedImageViewMut};
use crate::pixels::{Pixel, PixelType};
#[derive(Debug, Clone, Copy)]
pub enum CpuExtensions {
@@ -35,32 +37,6 @@ impl Default for CpuExtensions {
}
}
impl CpuExtensions {
#[cfg(target_arch = "x86_64")]
#[inline]
fn get_resampler(&self, pixel_type: PixelType) -> &dyn Convolution {
match pixel_type {
PixelType::U8x4 => match self {
Self::Sse4_1 => &convolution::Sse4U8x4,
Self::Avx2 => &convolution::Avx2U8x4,
_ => &convolution::NativeU8x4,
},
PixelType::I32 => &convolution::NativeI32,
PixelType::F32 => &convolution::NativeF32,
}
}
#[cfg(not(target_arch = "x86_64"))]
#[inline]
fn get_resampler(&self, pixel_type: PixelType) -> &dyn Convolution {
match pixel_type {
PixelType::U8x4 => &convolution::NativeU8x4,
PixelType::I32 => &convolution::NativeI32,
PixelType::F32 => &convolution::NativeF32,
}
}
}
#[derive(Debug, Clone, Copy)]
#[non_exhaustive]
pub enum ResizeAlg {
@@ -80,8 +56,8 @@ impl Default for ResizeAlg {
pub struct Resizer {
pub algorithm: ResizeAlg,
cpu_extensions: CpuExtensions,
convolution_buffer: Vec<u32>,
super_sampling_buffer: Vec<u32>,
convolution_buffer: Vec<u8>,
super_sampling_buffer: Vec<u8>,
}
impl Resizer {
@@ -101,10 +77,54 @@ impl Resizer {
///
/// This method doesn't multiply source image and doesn't divide
/// destination image by alpha channel.
/// You must use [MulDiv](crate::MulDiv) for this actions.
pub fn resize(&mut self, src_image: &SrcImageView, dst_image: &mut DstImageView) {
/// You must use [MulDiv](crate::MulDiv) for these actions.
pub fn resize(
&mut self,
src_image: &ImageView,
dst_image: &mut ImageViewMut,
) -> Result<(), DifferentTypesOfPixelsError> {
if src_image.pixel_type() != dst_image.pixel_type() {
return Err(DifferentTypesOfPixelsError);
}
match src_image.pixel_type() {
PixelType::U8x4 => {
if let Some(src_rows) = src_image.u32_image() {
if let Some(dst_rows) = dst_image.u32_image() {
self.resize_inner(src_rows, dst_rows);
}
}
}
PixelType::I32 => {
if let Some(src_rows) = src_image.i32_image() {
if let Some(dst_rows) = dst_image.i32_image() {
self.resize_inner(src_rows, dst_rows);
}
}
}
PixelType::F32 => {
if let Some(src_rows) = src_image.f32_image() {
if let Some(dst_rows) = dst_image.f32_image() {
self.resize_inner(src_rows, dst_rows);
}
}
}
PixelType::U8 => {
if let Some(src_rows) = src_image.u8_image() {
if let Some(dst_rows) = dst_image.u8_image() {
self.resize_inner(src_rows, dst_rows);
}
}
}
}
Ok(())
}
fn resize_inner<P>(&mut self, src_image: TypedImageView<P>, dst_image: TypedImageViewMut<P>)
where
P: Convolution,
{
match self.algorithm {
ResizeAlg::Nearest => resample_nearest(&src_image, dst_image),
ResizeAlg::Nearest => resample_nearest(src_image, dst_image),
ResizeAlg::Convolution(filter_type) => {
let convolution_buffer = &mut self.convolution_buffer;
resample_convolution(
@@ -135,7 +155,7 @@ impl Resizer {
/// intermediate resizing steps.
pub fn size_of_internal_buffers(&self) -> usize {
(self.convolution_buffer.capacity() + self.super_sampling_buffer.capacity())
* std::mem::size_of::<u32>()
* std::mem::size_of::<u8>()
}
/// Deallocates the internal buffers used to store the results of
@@ -162,21 +182,25 @@ impl Resizer {
}
}
fn get_temp_image_from_buffer(
buffer: &mut Vec<u32>,
fn get_temp_image_from_buffer<P: Pixel>(
buffer: &mut Vec<u8>,
width: NonZeroU32,
height: NonZeroU32,
pixel_type: PixelType,
) -> ImageData {
let buf_size = (width.get() * height.get()) as usize;
) -> InnerImage<P> {
let pixels_count = (width.get() * height.get()) as usize;
// Add pixel size as gap for alignment of resulted buffer.
let buf_size = pixels_count * P::size() + P::size();
if buffer.len() < buf_size {
buffer.resize(buf_size, 0);
}
let pixels = &mut buffer[0..buf_size];
ImageData::from_slice_u32(width, height, pixels, pixel_type).unwrap()
let pixels = unsafe { buffer.align_to_mut::<P::Type>().1 };
InnerImage::new(width, height, &mut pixels[0..pixels_count])
}
fn resample_nearest(src_image: &SrcImageView, dst_image: &mut DstImageView) {
fn resample_nearest<P>(src_image: TypedImageView<P>, mut dst_image: TypedImageViewMut<P>)
where
P: Pixel,
{
let crop_box = src_image.crop_box();
let dst_width = dst_image.width().get();
let x_scale = crop_box.width.get() as f64 / dst_width as f64;
@@ -202,18 +226,19 @@ fn resample_nearest(src_image: &SrcImageView, dst_image: &mut DstImageView) {
}
}
fn resample_convolution(
src_image: &SrcImageView,
dst_image: &mut DstImageView,
fn resample_convolution<P>(
src_image: TypedImageView<P>,
dst_image: TypedImageViewMut<P>,
filter_type: FilterType,
cpu_extensions: CpuExtensions,
temp_buffer: &mut Vec<u32>,
) {
temp_buffer: &mut Vec<u8>,
) where
P: Convolution,
{
let crop_box = src_image.crop_box();
let dst_width = dst_image.width();
let dst_height = dst_image.height();
let (filter_fn, filter_support) = convolution::get_filter_func(filter_type);
let resampler = cpu_extensions.get_resampler(src_image.pixel_type());
let need_horizontal = dst_width != src_image.width() || crop_box.width != src_image.width();
let need_vertical = dst_height != src_image.height() || crop_box.height != src_image.height();
@@ -246,17 +271,13 @@ fn resample_convolution(
let y_last = last_y_bound.start + last_y_bound.size;
let temp_height = NonZeroU32::new(y_last - y_first).unwrap();
let mut temp_image = get_temp_image_from_buffer(
temp_buffer,
dst_width,
temp_height,
src_image.pixel_type(),
);
resampler.horiz_convolution(
let mut temp_image = get_temp_image_from_buffer(temp_buffer, dst_width, temp_height);
P::horiz_convolution(
src_image,
&mut temp_image.dst_view(),
temp_image.dst_view(),
y_first,
horiz_coeffs,
cpu_extensions,
);
// Shift bounds for vertical pass
@@ -264,24 +285,31 @@ fn resample_convolution(
.bounds
.iter_mut()
.for_each(|b| b.start -= y_first);
resampler.vert_convolution(&temp_image.src_view(), dst_image, vert_coeffs);
P::vert_convolution(
temp_image.src_view(),
dst_image,
vert_coeffs,
cpu_extensions,
);
} else {
resampler.horiz_convolution(src_image, dst_image, y_first, horiz_coeffs);
P::horiz_convolution(src_image, dst_image, y_first, horiz_coeffs, cpu_extensions);
}
} else if need_vertical {
resampler.vert_convolution(src_image, dst_image, vert_coeffs);
P::vert_convolution(src_image, dst_image, vert_coeffs, cpu_extensions);
}
}
fn resample_super_sampling(
src_image: &SrcImageView,
dst_image: &mut DstImageView,
fn resample_super_sampling<P>(
src_image: TypedImageView<P>,
dst_image: TypedImageViewMut<P>,
filter_type: FilterType,
multiplicity: u8,
cpu_extensions: CpuExtensions,
temp_buffer: &mut Vec<u32>,
convolution_temp_buffer: &mut Vec<u32>,
) {
temp_buffer: &mut Vec<u8>,
convolution_temp_buffer: &mut Vec<u8>,
) where
P: Convolution,
{
let crop_box = src_image.crop_box();
let dst_width = dst_image.width().get();
let dst_height = dst_image.height().get();
@@ -299,12 +327,11 @@ fn resample_super_sampling(
let tmp_height =
NonZeroU32::new((crop_box.height.get() as f32 / factor).round() as u32).unwrap();
let mut tmp_img =
get_temp_image_from_buffer(temp_buffer, tmp_width, tmp_height, src_image.pixel_type());
resample_nearest(src_image, &mut tmp_img.dst_view());
let mut tmp_img = get_temp_image_from_buffer(temp_buffer, tmp_width, tmp_height);
resample_nearest(src_image, tmp_img.dst_view());
// Second step is resizing the temporary image with a convolution.
resample_convolution(
&tmp_img.src_view(),
tmp_img.src_view(),
dst_image,
filter_type,
cpu_extensions,
+18 -19
View File
@@ -1,6 +1,9 @@
use fast_image_resize::{CpuExtensions, DstImageView, ImageData, MulDiv, PixelType, SrcImageView};
use std::num::NonZeroU32;
use fast_image_resize::{
CpuExtensions, Image, ImageRows, ImageRowsMut, ImageView, ImageViewMut, MulDiv, PixelType,
};
const fn p(r: u8, g: u8, b: u8, a: u8) -> u32 {
u32::from_le_bytes([r, g, b, a])
}
@@ -21,20 +24,19 @@ fn multiply_alpha_test(cpu_extensions: CpuExtensions) {
];
let rows: Vec<&[u32]> = src_rows.iter().map(|r| r.as_ref()).collect();
let src_image_view = SrcImageView::from_rows(
let src_image_view = ImageView::new(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
rows,
PixelType::U8x4,
ImageRows::U8x4(rows),
)
.unwrap();
let mut dst_image = ImageData::new(
let mut dst_image = Image::new(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
PixelType::U8x4,
);
let mut dst_image_view = dst_image.dst_view();
let mut dst_image_view = dst_image.view_mut();
let mut alpha_mul_div: MulDiv = Default::default();
unsafe {
@@ -45,7 +47,7 @@ fn multiply_alpha_test(cpu_extensions: CpuExtensions) {
.multiply_alpha(&src_image_view, &mut dst_image_view)
.unwrap();
let dst_pixels = dst_image.get_pixels();
let dst_pixels = unsafe { dst_image.buffer().align_to::<u32>().1 };
let dst_rows = dst_pixels.chunks_exact(width as usize);
for (row, &valid_pixel) in dst_rows.zip(res_pixels.iter()) {
for &pixel in row.iter() {
@@ -55,11 +57,10 @@ fn multiply_alpha_test(cpu_extensions: CpuExtensions) {
// Inplace
let rows: Vec<&mut [u32]> = src_rows.iter_mut().map(|r| r.as_mut()).collect();
let mut image_view = DstImageView::from_rows(
let mut image_view = ImageViewMut::new(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
rows,
PixelType::U8x4,
ImageRowsMut::U8x4(rows),
)
.unwrap();
alpha_mul_div
@@ -104,20 +105,19 @@ fn divide_alpha_test(cpu_extensions: CpuExtensions) {
];
let rows: Vec<&[u32]> = src_rows.iter().map(|r| r.as_ref()).collect();
let src_image_view = SrcImageView::from_rows(
let src_image_view = ImageView::new(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
rows,
PixelType::U8x4,
ImageRows::U8x4(rows),
)
.unwrap();
let mut dst_image = ImageData::new(
let mut dst_image = Image::new(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
PixelType::U8x4,
);
let mut dst_image_view = dst_image.dst_view();
let mut dst_image_view = dst_image.view_mut();
let mut alpha_mul_div: MulDiv = Default::default();
unsafe {
@@ -128,7 +128,7 @@ fn divide_alpha_test(cpu_extensions: CpuExtensions) {
.divide_alpha(&src_image_view, &mut dst_image_view)
.unwrap();
let dst_pixels = dst_image.get_pixels();
let dst_pixels = unsafe { dst_image.buffer().align_to::<u32>().1 };
let dst_rows = dst_pixels.chunks_exact(width as usize);
for (row, &valid_pixel) in dst_rows.zip(res_pixels.iter()) {
for &pixel in row.iter() {
@@ -138,11 +138,10 @@ fn divide_alpha_test(cpu_extensions: CpuExtensions) {
// Inplace
let rows: Vec<&mut [u32]> = src_rows.iter_mut().map(|r| r.as_mut()).collect();
let mut image_view = DstImageView::from_rows(
let mut image_view = ImageViewMut::new(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
rows,
PixelType::U8x4,
ImageRowsMut::U8x4(rows),
)
.unwrap();
alpha_mul_div.divide_alpha_inplace(&mut image_view).unwrap();
+7 -9
View File
@@ -16,7 +16,7 @@ fn resize_image_example() {
.unwrap();
let width = NonZeroU32::new(img.width()).unwrap();
let height = NonZeroU32::new(img.height()).unwrap();
let mut src_image = fr::ImageData::from_vec_u8(
let mut src_image = fr::Image::from_vec_u8(
width,
height,
img.to_rgba8().into_raw(),
@@ -28,31 +28,30 @@ fn resize_image_example() {
let alpha_mul_div: fr::MulDiv = Default::default();
// Multiple RGB channels of source image by alpha channel
alpha_mul_div
.multiply_alpha_inplace(&mut src_image.dst_view())
.multiply_alpha_inplace(&mut src_image.view_mut())
.unwrap();
// Create wrapper that own data of destination image
let dst_width = NonZeroU32::new(1024).unwrap();
let dst_height = NonZeroU32::new(768).unwrap();
let mut dst_image = fr::ImageData::new(dst_width, dst_height, src_image.pixel_type());
let mut dst_image = fr::Image::new(dst_width, dst_height, src_image.pixel_type());
// Get mutable view of destination image data
let mut dst_view = dst_image.dst_view();
let mut dst_view = dst_image.view_mut();
// Create Resizer instance and resize source image
// into buffer of destination image
let mut resizer = fr::Resizer::new(fr::ResizeAlg::Convolution(fr::FilterType::Lanczos3));
resizer.resize(&src_image.src_view(), &mut dst_view);
resizer.resize(&src_image.view(), &mut dst_view).unwrap();
// Divide RGB channels of destination image by alpha
alpha_mul_div.divide_alpha_inplace(&mut dst_view).unwrap();
// Write destination image as PNG-file
let mut result_buf = BufWriter::new(Vec::new());
let encoder = PngEncoder::new(&mut result_buf);
encoder
PngEncoder::new(&mut result_buf)
.encode(
dst_image.get_buffer(),
dst_image.buffer(),
dst_width.get(),
dst_height.get(),
ColorType::Rgba8,
@@ -65,5 +64,4 @@ fn main() {
unsafe {
resizer.set_cpu_extensions(fr::CpuExtensions::Sse4_1);
}
// ...
}
+122 -50
View File
@@ -6,17 +6,18 @@ use image::io::Reader as ImageReader;
use image::{ColorType, GenericImageView};
use fast_image_resize::{
CpuExtensions, FilterType, ImageData, PixelType, ResizeAlg, Resizer, SrcImageView,
CpuExtensions, DifferentTypesOfPixelsError, FilterType, Image, ImageView, PixelType, ResizeAlg,
Resizer,
};
fn get_source_image() -> ImageData<'static> {
fn get_source_image_u8x4() -> Image<'static> {
let img = ImageReader::open("./data/nasa-4928x3279.png")
.unwrap()
.decode()
.unwrap();
let width = img.width();
let height = img.height();
ImageData::from_vec_u8(
Image::from_vec_u8(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
img.to_rgba8().into_raw(),
@@ -25,14 +26,30 @@ fn get_source_image() -> ImageData<'static> {
.unwrap()
}
fn get_small_source_image() -> ImageData<'static> {
fn get_source_image_u8x1() -> Image<'static> {
let img = ImageReader::open("./data/nasa-4928x3279.png")
.unwrap()
.decode()
.unwrap();
let width = img.width();
let height = img.height();
Image::from_vec_u8(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
img.to_luma8().into_raw(),
PixelType::U8,
)
.unwrap()
}
fn get_small_source_image() -> Image<'static> {
let img = ImageReader::open("./data/nasa-852x567.png")
.unwrap()
.decode()
.unwrap();
let width = img.width();
let height = img.height();
ImageData::from_vec_u8(
Image::from_vec_u8(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
img.to_rgba8().into_raw(),
@@ -41,7 +58,7 @@ fn get_small_source_image() -> ImageData<'static> {
.unwrap()
}
fn get_new_height(src_image: &SrcImageView, new_width: u32) -> u32 {
fn get_new_height(src_image: &ImageView, new_width: u32) -> u32 {
let scale = new_width as f32 / src_image.width().get() as f32;
(src_image.height().get() as f32 * scale).round() as u32
}
@@ -49,69 +66,79 @@ fn get_new_height(src_image: &SrcImageView, new_width: u32) -> u32 {
const NEW_WIDTH: u32 = 255;
const NEW_BIG_WIDTH: u32 = 5016;
fn save_result(image: &SrcImageView, name: &str) {
fn save_result(image: &Image, name: &str) {
std::fs::create_dir_all("./data/result").unwrap();
let mut file = File::create(format!("./data/result/{}.png", name)).unwrap();
let encoder = PngEncoder::new(&mut file);
encoder
let color_type = match image.pixel_type() {
PixelType::U8x4 => ColorType::Rgba8,
PixelType::U8 => ColorType::L8,
_ => panic!("Unsupported type of pixels"),
};
PngEncoder::new(&mut file)
.encode(
&image.get_buffer(),
image.buffer(),
image.width().get(),
image.height().get(),
ColorType::Rgba8,
color_type,
)
.unwrap();
}
#[test]
fn resize_wo_simd_lanczos3_test() {
let image = get_source_image();
let image = get_source_image_u8x4();
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
unsafe {
resizer.set_cpu_extensions(CpuExtensions::None);
}
let new_height = get_new_height(&image.src_view(), NEW_WIDTH);
let mut result = ImageData::new(
let new_height = get_new_height(&image.view(), NEW_WIDTH);
let mut result = Image::new(
NonZeroU32::new(NEW_WIDTH).unwrap(),
NonZeroU32::new(new_height).unwrap(),
image.pixel_type(),
);
resizer.resize(&image.src_view(), &mut result.dst_view());
save_result(&result.src_view(), "lanczos3_wo_simd");
assert!(resizer
.resize(&image.view(), &mut result.view_mut())
.is_ok());
save_result(&result, "u8x4-lanczos3-native");
}
#[test]
fn resize_sse4_lanczos3_test() {
let image = get_source_image();
let image = get_source_image_u8x4();
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
unsafe {
resizer.set_cpu_extensions(CpuExtensions::Sse4_1);
}
let new_height = get_new_height(&image.src_view(), NEW_WIDTH);
let mut result = ImageData::new(
let new_height = get_new_height(&image.view(), NEW_WIDTH);
let mut result = Image::new(
NonZeroU32::new(NEW_WIDTH).unwrap(),
NonZeroU32::new(new_height).unwrap(),
image.pixel_type(),
);
resizer.resize(&image.src_view(), &mut result.dst_view());
save_result(&result.src_view(), "lanczos3_sse4");
assert!(resizer
.resize(&image.view(), &mut result.view_mut())
.is_ok());
save_result(&result, "u8x4-lanczos3-sse4");
}
#[test]
fn resize_avx2_lanczos3_test() {
let image = get_source_image();
let image = get_source_image_u8x4();
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
unsafe {
resizer.set_cpu_extensions(CpuExtensions::Avx2);
}
let new_height = get_new_height(&image.src_view(), NEW_WIDTH);
let mut result = ImageData::new(
let new_height = get_new_height(&image.view(), NEW_WIDTH);
let mut result = Image::new(
NonZeroU32::new(NEW_WIDTH).unwrap(),
NonZeroU32::new(new_height).unwrap(),
image.pixel_type(),
);
resizer.resize(&image.src_view(), &mut result.dst_view());
save_result(&result.src_view(), "lanczos3_avx2");
assert!(resizer
.resize(&image.view(), &mut result.view_mut())
.is_ok());
save_result(&result, "u8x4-lanczos3-avx2");
}
#[test]
@@ -121,64 +148,109 @@ fn resize_avx2_lanczos3_upscale_test() {
unsafe {
resizer.set_cpu_extensions(CpuExtensions::Avx2);
}
let new_height = get_new_height(&image.src_view(), NEW_BIG_WIDTH);
let mut result = ImageData::new(
let new_height = get_new_height(&image.view(), NEW_BIG_WIDTH);
let mut result = Image::new(
NonZeroU32::new(NEW_BIG_WIDTH).unwrap(),
NonZeroU32::new(new_height).unwrap(),
image.pixel_type(),
);
resizer.resize(&image.src_view(), &mut result.dst_view());
save_result(&result.src_view(), "lanczos3_avx2_upscale");
assert!(resizer
.resize(&image.view(), &mut result.view_mut())
.is_ok());
save_result(&result, "u8x4-lanczos3_upscale-avx2");
}
#[test]
fn resize_nearest_test() {
let image = get_source_image();
let image = get_source_image_u8x4();
let mut resizer = Resizer::new(ResizeAlg::Nearest);
unsafe {
resizer.set_cpu_extensions(CpuExtensions::None);
}
let new_height = get_new_height(&image.src_view(), NEW_WIDTH);
let mut result = ImageData::new(
let new_height = get_new_height(&image.view(), NEW_WIDTH);
let mut result = Image::new(
NonZeroU32::new(NEW_WIDTH).unwrap(),
NonZeroU32::new(new_height).unwrap(),
image.pixel_type(),
);
resizer.resize(&image.src_view(), &mut result.dst_view());
save_result(&result.src_view(), "nearest_wo_simd");
assert!(resizer
.resize(&image.view(), &mut result.view_mut())
.is_ok());
save_result(&result, "u8x4-nearest-native");
}
#[test]
fn resize_super_sampling_test() {
let image = get_source_image();
let image = get_source_image_u8x4();
let mut resizer = Resizer::new(ResizeAlg::SuperSampling(FilterType::Lanczos3, 2));
unsafe {
resizer.set_cpu_extensions(CpuExtensions::Avx2);
}
let new_height = get_new_height(&image.src_view(), NEW_WIDTH);
let mut result = ImageData::new(
let new_height = get_new_height(&image.view(), NEW_WIDTH);
let mut result = Image::new(
NonZeroU32::new(NEW_WIDTH).unwrap(),
NonZeroU32::new(new_height).unwrap(),
image.pixel_type(),
);
resizer.resize(&image.src_view(), &mut result.dst_view());
save_result(&result.src_view(), "super_sampling_avx2");
assert!(resizer
.resize(&image.view(), &mut result.view_mut())
.is_ok());
save_result(&result, "u8x4-super_sampling-avx2");
}
#[test]
fn resize_with_cropping() {
let src_image = get_source_image();
fn try_resize_to_other_pixel_type() {
let src_image = get_source_image_u8x4();
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
let mut dst_image = Image::new(
NonZeroU32::new(1024).unwrap(),
NonZeroU32::new(256).unwrap(),
PixelType::U8,
);
assert!(matches!(
resizer.resize(&src_image.view(), &mut dst_image.view_mut()),
Err(DifferentTypesOfPixelsError)
));
}
#[test]
fn resize_nearest_u8x1() {
let image = get_source_image_u8x1();
assert!(matches!(image.pixel_type(), PixelType::U8));
let mut resizer = Resizer::new(ResizeAlg::Nearest);
unsafe {
resizer.set_cpu_extensions(CpuExtensions::None);
}
let new_height = get_new_height(&image.view(), NEW_WIDTH);
let mut result = Image::new(
NonZeroU32::new(NEW_WIDTH).unwrap(),
NonZeroU32::new(new_height).unwrap(),
image.pixel_type(),
);
assert!(resizer
.resize(&image.view(), &mut result.view_mut())
.is_ok());
save_result(&result, "u8x1-nearest-native");
}
#[test]
fn resize_lanczos3_u8x1() {
let image = get_source_image_u8x1();
assert!(matches!(image.pixel_type(), PixelType::U8));
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
unsafe {
resizer.set_cpu_extensions(CpuExtensions::None);
}
let mut dst_image = ImageData::new(
NonZeroU32::new(1024).unwrap(),
NonZeroU32::new(256).unwrap(),
src_image.pixel_type(),
let new_height = get_new_height(&image.view(), NEW_WIDTH);
let mut result = Image::new(
NonZeroU32::new(NEW_WIDTH).unwrap(),
NonZeroU32::new(new_height).unwrap(),
image.pixel_type(),
);
let mut src_view = src_image.src_view();
src_view.set_crop_box_to_fit_dst_size(dst_image.width(), dst_image.height(), None);
resizer.resize(&src_view, &mut dst_image.dst_view());
save_result(&dst_image.src_view(), "cropping_lanczos3");
assert!(resizer
.resize(&image.view(), &mut result.view_mut())
.is_ok());
save_result(&result, "u8x1-lanczos3-native");
}