Added support for the new pixel type PixelType::F32x2 with optimizations for SSE4.1 and AVX2 (#30).

This commit is contained in:
Kirill Kuzminykh
2024-06-18 01:16:32 +03:00
parent c503d288e8
commit ad2610b91e
44 changed files with 2254 additions and 868 deletions
+14
View File
@@ -1,5 +1,19 @@
## [Unreleased] - ReleaseDate
### Added
- Added support for the new pixel type `PixelType::F32x2` with
optimizations for SSE4.1 and AVX2 (#30).
## [4.0.0] - 2024-05-13
| | rust | sse4.1 | avx2 |
|--------------------------------|:----:|:------:|:----:|
| Multiplies alpha U16x2 | 6.10 | 3.02 | 2.25 |
| Multiplies alpha inplace U16x2 | 5.47 | 2.74 | 1.59 |
| Multiplies alpha F32x2 | 4.75 | 4.08 | 4.06 |
| Multiplies alpha inplace F32x2 | 7.30 | 6.63 | 5.79 |
### Added
- Added Gaussian filter for convolution algorithm.
Generated
+95 -86
View File
@@ -76,9 +76,9 @@ dependencies = [
[[package]]
name = "anstyle-query"
version = "1.0.3"
version = "1.1.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "a64c907d4e79225ac72e2a354c9ce84d50ebb4586dee56c82b3ee73004f537f5"
checksum = "ad186efb764318d35165f1758e7dcef3b10628e26d41a44bc5550652e6804391"
dependencies = [
"windows-sys",
]
@@ -95,9 +95,9 @@ dependencies = [
[[package]]
name = "anyhow"
version = "1.0.83"
version = "1.0.86"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "25bdb32cbbdce2b519a9cd7df3a678443100e265d5e25ca763b7572a5104f5f3"
checksum = "b3d1d046238990b9cf5bcde22a3fb3584ee5cf65fb2765f454ed428c7a0063da"
[[package]]
name = "arbitrary"
@@ -113,7 +113,7 @@ checksum = "0ae92a5119aa49cdbcf6b9f893fe4e1d98b04ccbf82ee0584ad948a44a734dea"
dependencies = [
"proc-macro2",
"quote",
"syn 2.0.63",
"syn 2.0.66",
]
[[package]]
@@ -171,9 +171,9 @@ checksum = "cf4b9d6a944f767f8e5e0db018570623c85f3d925ac718db4e06d0187adb21c1"
[[package]]
name = "bitstream-io"
version = "2.3.0"
version = "2.4.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "7c12d1856e42f0d817a835fe55853957c85c8c8a470114029143d3f12671446e"
checksum = "415f8399438eb5e4b2f73ed3152a3448b98149dda642a957ee704e1daa5cf1d8"
[[package]]
name = "block-buffer"
@@ -196,9 +196,9 @@ dependencies = [
[[package]]
name = "built"
version = "0.7.2"
version = "0.7.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "41bfbdb21256b87a8b5e80fab81a8eed158178e812fd7ba451907518b2742f16"
checksum = "c6a6c0b39c38fd754ac338b00a88066436389c0f029da5d37d1e01091d9b7c17"
[[package]]
name = "bumpalo"
@@ -232,9 +232,9 @@ checksum = "37b2a672a2cb129a2e41c10b1224bb368f9f37a2b16b612598138befd7b37eb5"
[[package]]
name = "cc"
version = "1.0.97"
version = "1.0.99"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "099a5357d84c4c61eb35fc8eafa9a79a902c2f76911e5747ced4e032edd8d9b4"
checksum = "96c51067fd44124faa7f870b4b1c969379ad32b2ba805aa959430ceaa384f695"
dependencies = [
"jobserver",
"libc",
@@ -259,9 +259,9 @@ checksum = "baf1de4339761588bc0619e3cbc0120ee582ebb74b53b4efbf79117bd2da40fd"
[[package]]
name = "cfg_aliases"
version = "0.1.1"
version = "0.2.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "fd16c4719339c4530435d38e511904438d07cce7950afa3718a84ac36c10e89e"
checksum = "613afe47fcd5fac7ccf1db93babcb082c5994d996f20b8b159f2ad1658eb5724"
[[package]]
name = "chrono"
@@ -277,9 +277,9 @@ dependencies = [
[[package]]
name = "chrono-tz"
version = "0.8.6"
version = "0.9.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d59ae0466b83e838b81a54256c39d5d7c20b9d7daa10510a242d9b75abd5936e"
checksum = "93698b29de5e97ad0ae26447b344c482a7284c737d9ddc5f9e52b74a336671bb"
dependencies = [
"chrono",
"chrono-tz-build",
@@ -288,9 +288,9 @@ dependencies = [
[[package]]
name = "chrono-tz-build"
version = "0.2.1"
version = "0.3.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "433e39f13c9a060046954e0592a8d0a4bcb1040125cbf91cb8ee58964cfb350f"
checksum = "0c088aee841df9c3041febbb73934cfc39708749bf96dc827e3359cd39ef11b1"
dependencies = [
"parse-zoneinfo",
"phf",
@@ -326,9 +326,9 @@ dependencies = [
[[package]]
name = "clap"
version = "4.5.4"
version = "4.5.7"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "90bc066a67923782aa8515dbaea16946c5bcc5addbd668bb80af688e53e548a0"
checksum = "5db83dced34638ad474f39f250d7fea9598bdd239eaced1bdf45d597da0f433f"
dependencies = [
"clap_builder",
"clap_derive",
@@ -346,9 +346,9 @@ dependencies = [
[[package]]
name = "clap_builder"
version = "4.5.2"
version = "4.5.7"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ae129e2e766ae0ec03484e609954119f123cc1fe650337e155d03b022f24f7b4"
checksum = "f7e204572485eb3fbf28f871612191521df159bc3e15a9f5064c66dba3a8c05f"
dependencies = [
"anstream",
"anstyle",
@@ -358,21 +358,21 @@ dependencies = [
[[package]]
name = "clap_derive"
version = "4.5.4"
version = "4.5.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "528131438037fd55894f62d6e9f068b8f45ac57ffa77517819645d10aed04f64"
checksum = "c780290ccf4fb26629baa7a1081e68ced113f1d3ec302fa5948f1c381ebf06c6"
dependencies = [
"heck",
"proc-macro2",
"quote",
"syn 2.0.63",
"syn 2.0.66",
]
[[package]]
name = "clap_lex"
version = "0.7.0"
version = "0.7.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "98cc8fbded0c607b7ba9dd60cd98df59af97e84d24e49c8557331cfc26d301ce"
checksum = "4b82cf0babdbd58558212896d1a4272303a57bdb245c2bf1147185fb45640e70"
[[package]]
name = "color_quant"
@@ -403,9 +403,9 @@ dependencies = [
[[package]]
name = "crc32fast"
version = "1.4.0"
version = "1.4.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b3855a8a784b474f333699ef2bbca9db2c4a1f6d9088a90a2d25b1eb53111eaa"
checksum = "a97769d94ddab943e4510d138150169a2758b5ef3eb191a9ee688de3e23ef7b3"
dependencies = [
"cfg-if",
]
@@ -465,9 +465,9 @@ dependencies = [
[[package]]
name = "crossbeam-utils"
version = "0.8.19"
version = "0.8.20"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "248e3bacc7dc6baa3b21e405ee045c3047101a49145e7e9eca583ab4c2ca5345"
checksum = "22ec99545bb0ed0ea7bb9b8e1e9122ea386ff8a48c0922e43f36d45ab09e0e80"
[[package]]
name = "crunchy"
@@ -512,9 +512,9 @@ dependencies = [
[[package]]
name = "either"
version = "1.11.0"
version = "1.12.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "a47c1c47d2f5964e29c61246e81db715514cd532db6b5116a25ea3c03d6780a2"
checksum = "3dca9240753cf90908d7e4aac30f630662b02aebaa1b58a3cadabdb23385b58b"
[[package]]
name = "env_filter"
@@ -571,7 +571,7 @@ dependencies = [
"document-features",
"fast_image_resize",
"image",
"itertools 0.12.1",
"itertools 0.13.0",
"libvips",
"nix",
"num-traits",
@@ -660,11 +660,11 @@ dependencies = [
[[package]]
name = "globwalk"
version = "0.8.1"
version = "0.9.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "93e3af942408868f6934a7b85134a3230832b9977cf66125df2f9edcfce4ddcc"
checksum = "0bf760ebf69878d9fd8f110c89703d90ce35095324d1f1edcb595c63945ee757"
dependencies = [
"bitflags 1.3.2",
"bitflags 2.5.0",
"ignore",
"walkdir",
]
@@ -808,7 +808,7 @@ checksum = "c34819042dc3d3971c46c2190835914dfbe0c3c13f61449b2997f4e9722dfa60"
dependencies = [
"proc-macro2",
"quote",
"syn 2.0.63",
"syn 2.0.66",
]
[[package]]
@@ -846,6 +846,15 @@ dependencies = [
"either",
]
[[package]]
name = "itertools"
version = "0.13.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "413ee7dfc52ee1a4949ceeb7dbc8a33f2d6c088194d9f922fb8318faf1f01186"
dependencies = [
"either",
]
[[package]]
name = "itoa"
version = "1.0.11"
@@ -890,9 +899,9 @@ checksum = "03087c2bad5e1034e8cace5926dec053fb3790248370865f5117a7d0213354c8"
[[package]]
name = "libc"
version = "0.2.154"
version = "0.2.155"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ae743338b92ff9146ce83992f766a31066a91a8c84a45e0e9f21e7cf6de6d346"
checksum = "97b3888a4aecf77e811145cadf6eef5901f4782c53886191b2f693f24761847c"
[[package]]
name = "libfuzzer-sys"
@@ -964,9 +973,9 @@ dependencies = [
[[package]]
name = "memchr"
version = "2.7.2"
version = "2.7.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "6c8640c5d730cb13ebd907d8d04b52f55ac9a2eec55b440c8892f40d56c76c1d"
checksum = "78ca9ab1a0babb1e7d5695e3530886289c18cf2f87ec19a575a0abdce112e3a3"
[[package]]
name = "minimal-lexical"
@@ -976,9 +985,9 @@ checksum = "68354c5c6bd36d73ff3feceb05efa59b6acb7626617f4962be322a825e61f79a"
[[package]]
name = "miniz_oxide"
version = "0.7.2"
version = "0.7.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9d811f3e15f28568be3407c8e7fdb6514c1cda3cb30683f15b6a1a1dc4ea14a7"
checksum = "87dfd01fe195c66b572b37921ad8803d010623c0aca821bea2302239d155cdae"
dependencies = [
"adler",
"simd-adler32",
@@ -992,9 +1001,9 @@ checksum = "650eef8c711430f1a879fdd01d4745a7deea475becfb90269c06775983bbf086"
[[package]]
name = "nix"
version = "0.28.0"
version = "0.29.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ab2156c4fce2f8df6c499cc1c763e4394b7482525bf2a9701c9d79d215f519e4"
checksum = "71e2746dc3a24dd78b3cfcb7be93368c6de9963d30f43a6a73998a9cf4b17b46"
dependencies = [
"bitflags 2.5.0",
"cfg-if",
@@ -1047,7 +1056,7 @@ checksum = "ed3955f1a9c7c0c15e092f9c887db08b1fc683305fdf6eb6684f22555355e202"
dependencies = [
"proc-macro2",
"quote",
"syn 2.0.63",
"syn 2.0.66",
]
[[package]]
@@ -1143,7 +1152,7 @@ dependencies = [
"pest_meta",
"proc-macro2",
"quote",
"syn 2.0.63",
"syn 2.0.66",
]
[[package]]
@@ -1222,9 +1231,9 @@ checksum = "5b40af805b3121feab8a3c29f04d8ad262fa8e0561883e7653e024ae4479e6de"
[[package]]
name = "proc-macro2"
version = "1.0.82"
version = "1.0.85"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8ad3d49ab951a01fbaafe34f2ec74122942fe18a3f9814c3268f1bb72042131b"
checksum = "22244ce15aa966053a896d1accb3a6e68469b97c7f33f284b99f0d576879fc23"
dependencies = [
"unicode-ident",
]
@@ -1245,7 +1254,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8021cf59c8ec9c432cfc2526ac6b8aa508ecaf29cd415f271b8406c1b851c3fd"
dependencies = [
"quote",
"syn 2.0.63",
"syn 2.0.66",
]
[[package]]
@@ -1339,9 +1348,9 @@ dependencies = [
[[package]]
name = "ravif"
version = "0.11.5"
version = "0.11.7"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "bc13288f5ab39e6d7c9d501759712e6969fcc9734220846fc9ed26cae2cc4234"
checksum = "67376f469e7e7840d0040bbf4b9b3334005bb167f814621326e4c7ab8cd6e944"
dependencies = [
"avif-serialize",
"imgref",
@@ -1374,9 +1383,9 @@ dependencies = [
[[package]]
name = "regex"
version = "1.10.4"
version = "1.10.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c117dbdfde9c8308975b6a18d71f3f385c89461f7b3fb054288ecf2a2058ba4c"
checksum = "b91213439dad192326a0d7c6ee3955910425f441d7038e0d6933b0aec5c4517f"
dependencies = [
"aho-corasick",
"memchr",
@@ -1386,9 +1395,9 @@ dependencies = [
[[package]]
name = "regex-automata"
version = "0.4.6"
version = "0.4.7"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "86b83b8b9847f9bf95ef68afb0b8e6cdb80f498442f5179a29fad448fcc1eaea"
checksum = "38caf58cc5ef2fed281f89292ef23f6365465ed9a41b7a7754eb4e26496c92df"
dependencies = [
"aho-corasick",
"memchr",
@@ -1397,9 +1406,9 @@ dependencies = [
[[package]]
name = "regex-syntax"
version = "0.8.3"
version = "0.8.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "adad44e29e4c806119491a7f06f03de4d1af22c3a680dd47f1e6e179439d1f56"
checksum = "7a66a03ae7c801facd77a29370b4faec201768915ac14a721ba36f20bc9c209b"
[[package]]
name = "resize"
@@ -1457,22 +1466,22 @@ checksum = "94143f37725109f92c262ed2cf5e59bce7498c01bcc1502d7b9afe439a4e9f49"
[[package]]
name = "serde"
version = "1.0.201"
version = "1.0.203"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "780f1cebed1629e4753a1a38a3c72d30b97ec044f0aef68cb26650a3c5cf363c"
checksum = "7253ab4de971e72fb7be983802300c30b5a7f0c2e56fab8abfc6a214307c0094"
dependencies = [
"serde_derive",
]
[[package]]
name = "serde_derive"
version = "1.0.201"
version = "1.0.203"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c5e405930b9796f1c00bee880d03fc7e0bb4b9a11afc776885ffe84320da2865"
checksum = "500cbc0ebeb6f46627f50f3f5811ccf6bf00643be300b4c3eabc0ef55dc5b5ba"
dependencies = [
"proc-macro2",
"quote",
"syn 2.0.63",
"syn 2.0.66",
]
[[package]]
@@ -1488,9 +1497,9 @@ dependencies = [
[[package]]
name = "serde_spanned"
version = "0.6.5"
version = "0.6.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "eb3622f419d1296904700073ea6cc23ad690adbd66f13ea683df73298736f0c1"
checksum = "79e674e01f999af37c49f70a6ede167a8a60b2503e56c5599532a65baa5969a0"
dependencies = [
"serde",
]
@@ -1571,9 +1580,9 @@ dependencies = [
[[package]]
name = "syn"
version = "2.0.63"
version = "2.0.66"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "bf5be731623ca1a1fb7d8be6f261a3be6d3e2337b8a1f97be944d020c8fcb704"
checksum = "c42f3f41a2de00b01c0aaad383c5a45241efc8b2d1eda5661812fda5f3cdcff5"
dependencies = [
"proc-macro2",
"quote",
@@ -1601,9 +1610,9 @@ checksum = "e1fc403891a21bcfb7c37834ba66a547a8f402146eba7265b5a6d88059c9ff2f"
[[package]]
name = "tera"
version = "1.19.1"
version = "1.20.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "970dff17c11e884a4a09bc76e3a17ef71e01bb13447a11e85226e254fe6d10b8"
checksum = "ab9d851b45e865f178319da0abdbfe6acbc4328759ff18dafc3a41c16b4cd2ee"
dependencies = [
"chrono",
"chrono-tz",
@@ -1631,22 +1640,22 @@ dependencies = [
[[package]]
name = "thiserror"
version = "1.0.60"
version = "1.0.61"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "579e9083ca58dd9dcf91a9923bb9054071b9ebbd800b342194c9feb0ee89fc18"
checksum = "c546c80d6be4bc6a00c0f01730c08df82eaa7a7a61f11d656526506112cc1709"
dependencies = [
"thiserror-impl",
]
[[package]]
name = "thiserror-impl"
version = "1.0.60"
version = "1.0.61"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e2470041c06ec3ac1ab38d0356a6119054dedaea53e12fbefc0de730a1c08524"
checksum = "46c3384250002a6d5af4d114f2845d37b57521033f30d5c3f46c4d70e1197533"
dependencies = [
"proc-macro2",
"quote",
"syn 2.0.63",
"syn 2.0.66",
]
[[package]]
@@ -1672,9 +1681,9 @@ dependencies = [
[[package]]
name = "toml"
version = "0.8.12"
version = "0.8.14"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e9dd1545e8208b4a5af1aa9bbd0b4cf7e9ea08fabc5d0a5c67fcaafa17433aa3"
checksum = "6f49eb2ab21d2f26bd6db7bf383edc527a7ebaee412d17af4d40fdccd442f335"
dependencies = [
"serde",
"serde_spanned",
@@ -1684,18 +1693,18 @@ dependencies = [
[[package]]
name = "toml_datetime"
version = "0.6.5"
version = "0.6.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "3550f4e9685620ac18a50ed434eb3aec30db8ba93b0287467bca5826ea25baf1"
checksum = "4badfd56924ae69bcc9039335b2e017639ce3f9b001c393c1b2d1ef846ce2cbf"
dependencies = [
"serde",
]
[[package]]
name = "toml_edit"
version = "0.22.12"
version = "0.22.14"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d3328d4f68a705b2a4498da1d580585d39a6510f98318a2cec3018a7ec61ddef"
checksum = "f21c7aaf97f1bd9ca9d4f9e73b0a6c74bd5afef56f2bc931943a6e1c37e04e38"
dependencies = [
"indexmap",
"serde",
@@ -1774,9 +1783,9 @@ checksum = "3354b9ac3fae1ff6755cb6db53683adb661634f67557942dea4facebec0fee4b"
[[package]]
name = "utf8parse"
version = "0.2.1"
version = "0.2.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "711b9620af191e0cdc7468a8d14e709c3dcdb115b36f838e601583af800a370a"
checksum = "06abde3611657adf66d383f00b093d7faecc7fa57071cce2578660c9f1010821"
[[package]]
name = "v_frame"
@@ -1838,7 +1847,7 @@ dependencies = [
"once_cell",
"proc-macro2",
"quote",
"syn 2.0.63",
"syn 2.0.66",
"wasm-bindgen-shared",
]
@@ -1860,7 +1869,7 @@ checksum = "e94f17b526d0a461a191c78ea52bbce64071ed5c04c9ffe424dcb38f74171bb7"
dependencies = [
"proc-macro2",
"quote",
"syn 2.0.63",
"syn 2.0.66",
"wasm-bindgen-backend",
"wasm-bindgen-shared",
]
@@ -1970,9 +1979,9 @@ checksum = "bec47e5bfd1bff0eeaf6d8b485cc1074891a197ab4225d504cb7a1ab88b02bf0"
[[package]]
name = "winnow"
version = "0.6.8"
version = "0.6.13"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c3c52e9c97a68071b23e836c9380edae937f17b9c4667bd021973efc689f618d"
checksum = "59b5e5f6c299a3c7890b876a2a587f3115162487e704907d9b6cd29473052ba1"
dependencies = [
"memchr",
]
+7 -3
View File
@@ -46,14 +46,14 @@ png = "0.17.13"
serde = { version = "1.0", features = ["serde_derive"] }
serde_json = "1.0"
walkdir = "2.5"
itertools = "0.12.1"
itertools = "0.13.0"
criterion = { version = "0.5.1", default-features = false, features = ["cargo_bench_support"] }
tera = "1.19"
tera = "1.20"
testing = { path = "testing" }
[target.'cfg(not(target_arch = "wasm32"))'.dev-dependencies]
nix = { version = "0.28.0", default-features = false, features = ["sched"] }
nix = { version = "0.29.0", default-features = false, features = ["sched"] }
[target.'cfg(all(not(target_arch = "wasm32"), not(target_os = "windows")))'.dev-dependencies]
@@ -110,6 +110,10 @@ name = "bench_compare_la16"
harness = false
[[bench]]
name = "bench_compare_la_f32"
harness = false
[[bench]]
name = "bench_color_mapper"
harness = false
+3 -2
View File
@@ -20,8 +20,9 @@ Supported pixel formats and available optimisations:
| U16x2 | Two `u16` components per pixel (e.g. LA16) | + | + | + | + |
| U16x3 | Three `u16` components per pixel (e.g. RGB16) | + | + | + | + |
| U16x4 | Four `u16` components per pixel (e.g. RGBA16, RGBx16, CMYK16) | + | + | + | + |
| I32 | One `i32` component per pixel | - | - | - | - |
| F32 | One `f32` component per pixel | - | - | - | - |
| I32 | One `i32` component per pixel (e.g. L) | - | - | - | - |
| F32 | One `f32` component per pixel (e.g. L) | - | - | - | - |
| F32x2 | Two `f32` components per pixel (e.g. LA) | + | + | - | - |
## Colorspace
+7
View File
@@ -1,3 +1,5 @@
use num_traits::ToBytes;
use fast_image_resize::images::Image;
use fast_image_resize::CpuExtensions;
use fast_image_resize::MulDiv;
@@ -24,11 +26,13 @@ fn multiplies_alpha(
let sample_size = 100;
let width = 4096;
let height = 2048;
let f32x2_bytes: Vec<u8> = [1.0, 0.5].iter().flat_map(|v| v.to_le_bytes()).collect();
let pixel: &[u8] = match pixel_type {
PixelType::U8x4 => &[255, 128, 0, 128],
PixelType::U8x2 => &[255, 128],
PixelType::U16x2 => &[255, 255, 0, 128],
PixelType::U16x4 => &[0, 255, 0, 128, 0, 0, 0, 128],
PixelType::F32x2 => &f32x2_bytes,
_ => unreachable!(),
};
let src_data = get_src_image(width, height, pixel_type, pixel);
@@ -75,11 +79,13 @@ fn divides_alpha(
let sample_size = 100;
let width = 4095;
let height = 2048;
let f32x2_bytes: Vec<u8> = [0.5, 0.5].iter().flat_map(|v| v.to_le_bytes()).collect();
let pixel: &[u8] = match pixel_type {
PixelType::U8x4 => &[128, 64, 0, 128],
PixelType::U8x2 => &[128, 128],
PixelType::U16x2 => &[0, 128, 0, 128],
PixelType::U16x4 => &[0, 128, 0, 64, 0, 0, 0, 128],
PixelType::F32x2 => &f32x2_bytes,
_ => unreachable!(),
};
let src_data = get_src_image(width, height, pixel_type, pixel);
@@ -124,6 +130,7 @@ fn bench_alpha(bench_group: &mut utils::BenchGroup) {
PixelType::U8x4,
PixelType::U16x2,
PixelType::U16x4,
PixelType::F32x2,
];
let mut cpu_extensions = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
+1 -1
View File
@@ -8,7 +8,7 @@ mod utils;
pub fn bench_compare_l(bench_group: &mut utils::BenchGroup) {
type P = U8;
let src_image = P::load_big_image().to_luma8();
let src_image = P::load_big_image();
utils::image_resize(bench_group, &src_image);
utils::resize_resize(
bench_group,
+1 -1
View File
@@ -8,7 +8,7 @@ mod utils;
pub fn bench_downscale_l16(bench_group: &mut utils::BenchGroup) {
type P = U16;
let src_image = P::load_big_image().to_luma16();
let src_image = P::load_big_image();
utils::image_resize(bench_group, &src_image);
utils::resize_resize(
bench_group,
+14
View File
@@ -0,0 +1,14 @@
use fast_image_resize::pixels::F32x2;
mod utils;
pub fn bench_downscale_la_f32(bench_group: &mut utils::BenchGroup) {
type P = F32x2;
utils::libvips_resize::<P>(bench_group, true);
utils::fir_resize::<P>(bench_group, true);
}
fn main() {
let res = utils::run_bench(bench_downscale_la_f32, "Compare resize of LA-F32 image");
utils::print_and_write_compare_result(&res);
}
+1 -1
View File
@@ -8,7 +8,7 @@ mod utils;
pub fn bench_downscale_rgb(bench_group: &mut utils::BenchGroup) {
type P = U8x3;
let src_image = P::load_big_image().to_rgb8();
let src_image = P::load_big_image();
utils::image_resize(bench_group, &src_image);
utils::resize_resize(
bench_group,
+1 -1
View File
@@ -8,7 +8,7 @@ mod utils;
pub fn bench_downscale_rgb16(bench_group: &mut utils::BenchGroup) {
type P = U16x3;
let src_image = P::load_big_image().to_rgb16();
let src_image = P::load_big_image();
utils::image_resize(bench_group, &src_image);
utils::resize_resize(
bench_group,
+1 -1
View File
@@ -8,7 +8,7 @@ mod utils;
pub fn bench_downscale_rgba(bench_group: &mut utils::BenchGroup) {
type P = U8x4;
let src_image = P::load_big_image().to_rgba8();
let src_image = P::load_big_image();
utils::resize_resize(
bench_group,
RGBA8P,
+1 -1
View File
@@ -8,7 +8,7 @@ mod utils;
pub fn bench_downscale_rgba16(bench_group: &mut utils::BenchGroup) {
type P = U16x4;
let src_image = P::load_big_image().to_rgba16();
let src_image = P::load_big_image();
utils::resize_resize(
bench_group,
RGBA16P,
+3
View File
@@ -115,6 +115,7 @@ pub fn resize_in_one_dimension_bench(bench_group: &mut utils::BenchGroup) {
PixelType::U16x3 => U16x3::load_big_square_src_image(),
PixelType::U16x4 => U16x4::load_big_square_src_image(),
PixelType::I32 => I32::load_big_square_src_image(),
PixelType::F32x2 => F32x2::load_big_square_src_image(),
_ => unreachable!(),
};
downscale_bench(
@@ -150,6 +151,7 @@ pub fn resize_bench(bench_group: &mut utils::BenchGroup) {
PixelType::U16x3,
PixelType::U16x4,
PixelType::I32,
PixelType::F32x2,
];
let mut cpu_extensions = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
@@ -181,6 +183,7 @@ pub fn resize_bench(bench_group: &mut utils::BenchGroup) {
PixelType::U16x3 => U16x3::load_big_square_src_image(),
PixelType::U16x4 => U16x4::load_big_square_src_image(),
PixelType::I32 => I32::load_big_square_src_image(),
PixelType::F32x2 => F32x2::load_big_square_src_image(),
_ => unreachable!(),
};
downscale_bench(
+1 -1
View File
@@ -6,7 +6,7 @@ Pipeline:
- Source image
[nasa-4928x3279-rgba.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279-rgba.png)
has converted into grayscale image with alpha channel (two bytes per pixel).
has converted into grayscale image with an alpha channel (two bytes per pixel).
- Numbers in the table mean a duration of image resizing in milliseconds.
- The `image` crate does not support multiplying and dividing by alpha channel.
- The `resize` crate does not support this pixel format.
+1 -1
View File
@@ -6,7 +6,7 @@ Pipeline:
- Source image
[nasa-4928x3279-rgba.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279-rgba.png)
has converted into grayscale image with alpha channel (four bytes per pixel).
has converted into grayscale image with an alpha channel (four bytes per pixel).
- Numbers in the table mean a duration of image resizing in milliseconds.
- The `image` crate does not support multiplying and dividing by alpha channel.
- The `resize` crate does not support this pixel format.
@@ -0,0 +1,14 @@
### Resize LA-F32 (luma with alpha channel) image (F32x2) 4928x3279 => 852x567
Pipeline:
`src_image => multiply by alpha => resize => divide by alpha => dst_image`
- Source image
[nasa-4928x3279-rgba.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279-rgba.png)
has converted into grayscale image with an alpha channel (two `f32` values per pixel).
- Numbers in the table mean a duration of image resizing in milliseconds.
- The `image` crate does not support multiplying and dividing by alpha channel.
- The `resize` crate does not support this pixel format.
{{ compare_results -}}
+1 -1
View File
@@ -9,7 +9,7 @@ Environment:
- RAM: DDR4 4000 MHz
{% endif -%}
- Ubuntu 22.04 (linux 6.5.0)
- Rust 1.78
- Rust 1.79
- criterion = "0.5.1"
- fast_image_resize = "4.0.0"
{% if arch_id == "wasm32" -%}
+5 -4
View File
@@ -104,10 +104,11 @@ mod vips {
let src_image_data = P::load_big_src_image();
let src_width = src_image_data.width() as i32;
let src_height = src_image_data.height() as i32;
let band_format = if P::count_of_component_values() > 256 {
BandFormat::Ushort
} else {
BandFormat::Uchar
let band_format = match P::count_of_component_values() {
0x100 => BandFormat::Uchar,
0x10000 => BandFormat::Ushort,
0 => BandFormat::Float,
_ => panic!("Unknown type of pixel"),
};
let src_vips_image = VipsImage::new_from_memory(
src_image_data.buffer(),
+25 -3
View File
@@ -1,5 +1,4 @@
<!-- introduction start -->
## Benchmarks of fast_image_resize crate for x86_64 architecture
Environment:
@@ -7,16 +6,18 @@ Environment:
- CPU: AMD Ryzen 9 5950X
- RAM: DDR4 4000 MHz
- Ubuntu 22.04 (linux 6.5.0)
- Rust 1.78
- Rust 1.79
- criterion = "0.5.1"
- fast_image_resize = "4.0.0"
Other libraries used to compare of resizing speed:
- image = "0.25.1" (<https://crates.io/crates/image>)
- resize = "0.8.4" (<https://crates.io/crates/resize>)
- libvips = "8.12.1" (single-threaded mode, cache disabled)
Resize algorithms:
- Nearest
@@ -24,7 +25,6 @@ Resize algorithms:
- Bilinear - convolution with minimal kernel size 2x2 px
- Bicubic (CatmullRom) - convolution with minimal kernel size 4x4 px
- Lanczos3 - convolution with minimal kernel size 6x6 px
<!-- introduction end -->
<!-- bench_compare_rgb start -->
@@ -211,3 +211,25 @@ Pipeline:
| fir avx2 | 0.19 | 11.72 | 15.02 | 21.87 | 29.07 |
<!-- bench_compare_la16 end -->
<!-- bench_compare_la_f32 start -->
### Resize LA-F32 (luma with alpha channel) image (F32x2) 4928x3279 => 852x567
Pipeline:
`src_image => multiply by alpha => resize => divide by alpha => dst_image`
- Source image
[nasa-4928x3279-rgba.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279-rgba.png)
has converted into grayscale image with alpha channel (two `f32` values per pixel).
- Numbers in the table mean a duration of image resizing in milliseconds.
- The `image` crate does not support multiplying and dividing by alpha channel.
- The `resize` crate does not support this pixel format.
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
| libvips | 11.85 | 70.31 | 101.80 | 177.59 | 254.22 |
| fir rust | 0.38 | 21.26 | 28.82 | 47.50 | 70.34 |
| fir sse4.1 | 0.38 | 16.23 | 20.91 | 30.35 | 40.12 |
| fir avx2 | 0.39 | 15.05 | 17.18 | 22.47 | 27.89 |
<!-- bench_compare_la_f32 end -->
+151
View File
@@ -0,0 +1,151 @@
use std::arch::x86_64::*;
use crate::pixels::F32x2;
use crate::{ImageView, ImageViewMut};
use super::sse4;
#[target_feature(enable = "avx2")]
pub(crate) unsafe fn multiply_alpha(
src_view: &impl ImageView<Pixel = F32x2>,
dst_view: &mut impl ImageViewMut<Pixel = F32x2>,
) {
let src_rows = src_view.iter_rows(0);
let dst_rows = dst_view.iter_rows_mut(0);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
multiply_alpha_row(src_row, dst_row);
}
}
#[target_feature(enable = "avx2")]
pub(crate) unsafe fn multiply_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = F32x2>) {
for row in image_view.iter_rows_mut(0) {
multiply_alpha_row_inplace(row);
}
}
#[inline]
#[target_feature(enable = "avx2")]
pub(crate) unsafe fn multiply_alpha_row(src_row: &[F32x2], dst_row: &mut [F32x2]) {
let src_chunks = src_row.chunks_exact(8);
let src_remainder = src_chunks.remainder();
let mut dst_chunks = dst_row.chunks_exact_mut(8);
for (src_chunk, dst_chunk) in src_chunks.zip(&mut dst_chunks) {
let src_ptr = src_chunk.as_ptr() as *const f32;
let src_pixels03 = _mm256_loadu_ps(src_ptr);
let src_pixels47 = _mm256_loadu_ps(src_ptr.add(8));
multiply_alpha_8_pixels(src_pixels03, src_pixels47, dst_chunk);
}
if !src_remainder.is_empty() {
let dst_reminder = dst_chunks.into_remainder();
sse4::multiply_alpha_row(src_remainder, dst_reminder);
}
}
#[inline]
#[target_feature(enable = "avx2")]
pub(crate) unsafe fn multiply_alpha_row_inplace(row: &mut [F32x2]) {
let mut chunks = row.chunks_exact_mut(8);
for chunk in &mut chunks {
let src_ptr = chunk.as_ptr() as *const f32;
let src_pixels01 = _mm256_loadu_ps(src_ptr);
let src_pixels23 = _mm256_loadu_ps(src_ptr.add(8));
multiply_alpha_8_pixels(src_pixels01, src_pixels23, chunk);
}
let reminder = chunks.into_remainder();
if !reminder.is_empty() {
sse4::multiply_alpha_row_inplace(reminder);
}
}
#[inline]
#[target_feature(enable = "avx2")]
unsafe fn multiply_alpha_8_pixels(pixels03: __m256, pixels47: __m256, dst_chunk: &mut [F32x2]) {
let luma07 = _mm256_shuffle_ps::<0b10_00_10_00>(pixels03, pixels47);
let alpha07 = _mm256_shuffle_ps::<0b11_01_11_01>(pixels03, pixels47);
let multiplied_luma07 = _mm256_mul_ps(luma07, alpha07);
let dst_pixel03 = _mm256_unpacklo_ps(multiplied_luma07, alpha07);
let dst_pixel47 = _mm256_unpackhi_ps(multiplied_luma07, alpha07);
let dst_ptr = dst_chunk.as_mut_ptr() as *mut f32;
_mm256_storeu_ps(dst_ptr, dst_pixel03);
_mm256_storeu_ps(dst_ptr.add(8), dst_pixel47);
}
// Divide
#[target_feature(enable = "avx2")]
pub(crate) unsafe fn divide_alpha(
src_view: &impl ImageView<Pixel = F32x2>,
dst_view: &mut impl ImageViewMut<Pixel = F32x2>,
) {
let src_rows = src_view.iter_rows(0);
let dst_rows = dst_view.iter_rows_mut(0);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
divide_alpha_row(src_row, dst_row);
}
}
#[target_feature(enable = "avx2")]
pub(crate) unsafe fn divide_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = F32x2>) {
for row in image_view.iter_rows_mut(0) {
divide_alpha_row_inplace(row);
}
}
#[target_feature(enable = "avx2")]
pub(crate) unsafe fn divide_alpha_row(src_row: &[F32x2], dst_row: &mut [F32x2]) {
let src_chunks = src_row.chunks_exact(8);
let src_remainder = src_chunks.remainder();
let mut dst_chunks = dst_row.chunks_exact_mut(8);
for (src_chunk, dst_chunk) in src_chunks.zip(&mut dst_chunks) {
let src_ptr = src_chunk.as_ptr() as *const f32;
let src_pixels03 = _mm256_loadu_ps(src_ptr);
let src_pixels47 = _mm256_loadu_ps(src_ptr.add(8));
divide_alpha_8_pixels(src_pixels03, src_pixels47, dst_chunk);
}
if !src_remainder.is_empty() {
let dst_reminder = dst_chunks.into_remainder();
sse4::divide_alpha_row(src_remainder, dst_reminder);
}
}
#[target_feature(enable = "avx2")]
pub(crate) unsafe fn divide_alpha_row_inplace(row: &mut [F32x2]) {
let mut chunks = row.chunks_exact_mut(8);
for chunk in &mut chunks {
let src_ptr = chunk.as_ptr() as *const f32;
let src_pixels01 = _mm256_loadu_ps(src_ptr);
let src_pixels23 = _mm256_loadu_ps(src_ptr.add(8));
divide_alpha_8_pixels(src_pixels01, src_pixels23, chunk);
}
let reminder = chunks.into_remainder();
if !reminder.is_empty() {
sse4::divide_alpha_row_inplace(reminder);
}
}
#[inline]
#[target_feature(enable = "avx2")]
unsafe fn divide_alpha_8_pixels(pixels03: __m256, pixels47: __m256, dst_chunk: &mut [F32x2]) {
let zero = _mm256_set1_ps(0.);
let luma07 = _mm256_shuffle_ps::<0b10_00_10_00>(pixels03, pixels47);
let alpha07 = _mm256_shuffle_ps::<0b11_01_11_01>(pixels03, pixels47);
let mut multiplied_luma07 = _mm256_div_ps(luma07, alpha07);
let mask_zero = _mm256_cmp_ps::<_CMP_NEQ_UQ>(alpha07, zero);
multiplied_luma07 = _mm256_and_ps(mask_zero, multiplied_luma07);
let dst_pixel03 = _mm256_unpacklo_ps(multiplied_luma07, alpha07);
let dst_pixel47 = _mm256_unpackhi_ps(multiplied_luma07, alpha07);
let dst_ptr = dst_chunk.as_mut_ptr() as *mut f32;
_mm256_storeu_ps(dst_ptr, dst_pixel03);
_mm256_storeu_ps(dst_ptr.add(8), dst_pixel47);
}
+87
View File
@@ -0,0 +1,87 @@
use crate::cpu_extensions::CpuExtensions;
use crate::pixels::F32x2;
use crate::{ImageError, ImageView, ImageViewMut};
use super::AlphaMulDiv;
#[cfg(target_arch = "x86_64")]
mod avx2;
mod native;
#[cfg(target_arch = "x86_64")]
mod sse4;
impl AlphaMulDiv for F32x2 {
fn multiply_alpha(
src_view: &impl ImageView<Pixel = Self>,
dst_view: &mut impl ImageViewMut<Pixel = Self>,
cpu_extensions: CpuExtensions,
) -> Result<(), ImageError> {
match cpu_extensions {
#[cfg(target_arch = "x86_64")]
CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha(src_view, dst_view) },
#[cfg(target_arch = "x86_64")]
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha(src_view, dst_view) },
// #[cfg(target_arch = "aarch64")]
// CpuExtensions::Neon => unsafe { neon::multiply_alpha(src_view, dst_view) },
// #[cfg(target_arch = "wasm32")]
// CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha(src_view, dst_view) },
_ => native::multiply_alpha(src_view, dst_view),
}
Ok(())
}
fn multiply_alpha_inplace(
image_view: &mut impl ImageViewMut<Pixel = Self>,
cpu_extensions: CpuExtensions,
) -> Result<(), ImageError> {
match cpu_extensions {
#[cfg(target_arch = "x86_64")]
CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha_inplace(image_view) },
#[cfg(target_arch = "x86_64")]
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha_inplace(image_view) },
// #[cfg(target_arch = "aarch64")]
// CpuExtensions::Neon => unsafe { neon::multiply_alpha_inplace(image_view) },
// #[cfg(target_arch = "wasm32")]
// CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha_inplace(image_view) },
_ => native::multiply_alpha_inplace(image_view),
}
Ok(())
}
fn divide_alpha(
src_view: &impl ImageView<Pixel = Self>,
dst_view: &mut impl ImageViewMut<Pixel = Self>,
cpu_extensions: CpuExtensions,
) -> Result<(), ImageError> {
match cpu_extensions {
#[cfg(target_arch = "x86_64")]
CpuExtensions::Avx2 => unsafe { avx2::divide_alpha(src_view, dst_view) },
#[cfg(target_arch = "x86_64")]
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha(src_view, dst_view) },
// #[cfg(target_arch = "aarch64")]
// CpuExtensions::Neon => unsafe { crate::alpha::u16x2::neon::divide_alpha(src_view, dst_view) },
// #[cfg(target_arch = "wasm32")]
// CpuExtensions::Simd128 => unsafe { crate::alpha::u16x2::wasm32::divide_alpha(src_view, dst_view) },
_ => native::divide_alpha(src_view, dst_view),
}
Ok(())
}
fn divide_alpha_inplace(
image_view: &mut impl ImageViewMut<Pixel = Self>,
cpu_extensions: CpuExtensions,
) -> Result<(), ImageError> {
match cpu_extensions {
#[cfg(target_arch = "x86_64")]
CpuExtensions::Avx2 => unsafe { avx2::divide_alpha_inplace(image_view) },
#[cfg(target_arch = "x86_64")]
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha_inplace(image_view) },
// #[cfg(target_arch = "aarch64")]
// CpuExtensions::Neon => unsafe { crate::alpha::u16x2::neon::divide_alpha_inplace(image_view) },
// #[cfg(target_arch = "wasm32")]
// CpuExtensions::Simd128 => unsafe { crate::alpha::u16x2::wasm32::divide_alpha_inplace(image_view) },
_ => native::divide_alpha_inplace(image_view),
}
Ok(())
}
}
+90
View File
@@ -0,0 +1,90 @@
use num_traits::Zero;
use crate::pixels::F32x2;
use crate::utils::foreach_with_pre_reading;
use crate::{ImageView, ImageViewMut};
pub(crate) fn multiply_alpha(
src_view: &impl ImageView<Pixel = F32x2>,
dst_view: &mut impl ImageViewMut<Pixel = F32x2>,
) {
let src_rows = src_view.iter_rows(0);
let dst_rows = dst_view.iter_rows_mut(0);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
multiply_alpha_row(src_row, dst_row);
}
}
pub(crate) fn multiply_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = F32x2>) {
for row in image_view.iter_rows_mut(0) {
multiply_alpha_row_inplace(row);
}
}
#[inline(always)]
pub(crate) fn multiply_alpha_row(src_row: &[F32x2], dst_row: &mut [F32x2]) {
for (src_pixel, dst_pixel) in src_row.iter().zip(dst_row) {
let components: [f32; 2] = src_pixel.0;
let alpha = components[1];
dst_pixel.0 = [components[0] * alpha, alpha];
}
}
#[inline(always)]
pub(crate) fn multiply_alpha_row_inplace(row: &mut [F32x2]) {
for pixel in row {
pixel.0[0] *= pixel.0[1];
}
}
// Divide
#[inline]
pub(crate) fn divide_alpha(
src_view: &impl ImageView<Pixel = F32x2>,
dst_view: &mut impl ImageViewMut<Pixel = F32x2>,
) {
let src_rows = src_view.iter_rows(0);
let dst_rows = dst_view.iter_rows_mut(0);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
divide_alpha_row(src_row, dst_row);
}
}
#[inline]
pub(crate) fn divide_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = F32x2>) {
for row in image_view.iter_rows_mut(0) {
divide_alpha_row_inplace(row);
}
}
#[inline(always)]
pub(crate) fn divide_alpha_row(src_row: &[F32x2], dst_row: &mut [F32x2]) {
foreach_with_pre_reading(
src_row.iter().zip(dst_row),
|(&src_pixel, dst_pixel)| (src_pixel, dst_pixel),
|(src_pixel, dst_pixel)| {
let alpha = src_pixel.0[1];
if alpha.is_zero() {
dst_pixel.0 = [0.; 2];
} else {
dst_pixel.0 = [src_pixel.0[0] / alpha, alpha];
}
},
);
}
#[inline(always)]
pub(crate) fn divide_alpha_row_inplace(row: &mut [F32x2]) {
for pixel in row {
let components: [f32; 2] = pixel.0;
let alpha = components[1];
if alpha.is_zero() {
pixel.0[0] = 0.;
} else {
pixel.0[0] = components[0] / alpha;
}
}
}
+152
View File
@@ -0,0 +1,152 @@
use std::arch::x86_64::*;
use crate::pixels::F32x2;
use crate::{ImageView, ImageViewMut};
use super::native;
#[target_feature(enable = "sse4.1")]
pub(crate) unsafe fn multiply_alpha(
src_view: &impl ImageView<Pixel = F32x2>,
dst_view: &mut impl ImageViewMut<Pixel = F32x2>,
) {
let src_rows = src_view.iter_rows(0);
let dst_rows = dst_view.iter_rows_mut(0);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
multiply_alpha_row(src_row, dst_row);
}
}
#[target_feature(enable = "sse4.1")]
pub(crate) unsafe fn multiply_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = F32x2>) {
for row in image_view.iter_rows_mut(0) {
multiply_alpha_row_inplace(row);
}
}
#[inline]
#[target_feature(enable = "sse4.1")]
pub(crate) unsafe fn multiply_alpha_row(src_row: &[F32x2], dst_row: &mut [F32x2]) {
let src_chunks = src_row.chunks_exact(4);
let src_remainder = src_chunks.remainder();
let mut dst_chunks = dst_row.chunks_exact_mut(4);
for (src_chunk, dst_chunk) in src_chunks.zip(&mut dst_chunks) {
let src_ptr = src_chunk.as_ptr() as *const f32;
let src_pixels01 = _mm_loadu_ps(src_ptr);
let src_pixels23 = _mm_loadu_ps(src_ptr.add(4));
multiply_alpha_4_pixels(src_pixels01, src_pixels23, dst_chunk);
}
if !src_remainder.is_empty() {
let dst_reminder = dst_chunks.into_remainder();
native::multiply_alpha_row(src_remainder, dst_reminder);
}
}
#[inline]
#[target_feature(enable = "sse4.1")]
pub(crate) unsafe fn multiply_alpha_row_inplace(row: &mut [F32x2]) {
let mut chunks = row.chunks_exact_mut(4);
for chunk in &mut chunks {
let src_ptr = chunk.as_ptr() as *const f32;
let src_pixels01 = _mm_loadu_ps(src_ptr);
let src_pixels23 = _mm_loadu_ps(src_ptr.add(4));
multiply_alpha_4_pixels(src_pixels01, src_pixels23, chunk);
}
let reminder = chunks.into_remainder();
if !reminder.is_empty() {
native::multiply_alpha_row_inplace(reminder);
}
}
#[inline]
#[target_feature(enable = "sse4.1")]
unsafe fn multiply_alpha_4_pixels(pixels01: __m128, pixels23: __m128, dst_chunk: &mut [F32x2]) {
let luma03 = _mm_shuffle_ps::<0b10_00_10_00>(pixels01, pixels23);
let alpha03 = _mm_shuffle_ps::<0b11_01_11_01>(pixels01, pixels23);
let multiplied_luma03 = _mm_mul_ps(luma03, alpha03);
let dst_pixel01 = _mm_unpacklo_ps(multiplied_luma03, alpha03);
let dst_pixel23 = _mm_unpackhi_ps(multiplied_luma03, alpha03);
let dst_ptr = dst_chunk.as_mut_ptr() as *mut f32;
_mm_storeu_ps(dst_ptr, dst_pixel01);
_mm_storeu_ps(dst_ptr.add(4), dst_pixel23);
}
// Divide
#[target_feature(enable = "sse4.1")]
pub(crate) unsafe fn divide_alpha(
src_view: &impl ImageView<Pixel = F32x2>,
dst_view: &mut impl ImageViewMut<Pixel = F32x2>,
) {
let src_rows = src_view.iter_rows(0);
let dst_rows = dst_view.iter_rows_mut(0);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
divide_alpha_row(src_row, dst_row);
}
}
#[target_feature(enable = "sse4.1")]
pub(crate) unsafe fn divide_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = F32x2>) {
for row in image_view.iter_rows_mut(0) {
divide_alpha_row_inplace(row);
}
}
#[target_feature(enable = "sse4.1")]
pub(crate) unsafe fn divide_alpha_row(src_row: &[F32x2], dst_row: &mut [F32x2]) {
let src_chunks = src_row.chunks_exact(4);
let src_remainder = src_chunks.remainder();
let mut dst_chunks = dst_row.chunks_exact_mut(4);
for (src_chunk, dst_chunk) in src_chunks.zip(&mut dst_chunks) {
let src_ptr = src_chunk.as_ptr() as *const f32;
let src_pixels01 = _mm_loadu_ps(src_ptr);
let src_pixels23 = _mm_loadu_ps(src_ptr.add(4));
divide_alpha_4_pixels(src_pixels01, src_pixels23, dst_chunk);
}
if !src_remainder.is_empty() {
let dst_reminder = dst_chunks.into_remainder();
native::divide_alpha_row(src_remainder, dst_reminder);
}
}
#[target_feature(enable = "sse4.1")]
pub(crate) unsafe fn divide_alpha_row_inplace(row: &mut [F32x2]) {
let mut chunks = row.chunks_exact_mut(4);
for chunk in &mut chunks {
let src_ptr = chunk.as_ptr() as *const f32;
let src_pixels01 = _mm_loadu_ps(src_ptr);
let src_pixels23 = _mm_loadu_ps(src_ptr.add(4));
divide_alpha_4_pixels(src_pixels01, src_pixels23, chunk);
}
let reminder = chunks.into_remainder();
if !reminder.is_empty() {
native::divide_alpha_row_inplace(reminder);
}
}
#[inline]
#[target_feature(enable = "sse4.1")]
unsafe fn divide_alpha_4_pixels(pixels01: __m128, pixels23: __m128, dst_chunk: &mut [F32x2]) {
let zero = _mm_set_ps1(0.);
let luma03 = _mm_shuffle_ps::<0b10_00_10_00>(pixels01, pixels23);
let alpha03 = _mm_shuffle_ps::<0b11_01_11_01>(pixels01, pixels23);
let mut multiplied_luma03 = _mm_div_ps(luma03, alpha03);
let mask_zero = _mm_cmpneq_ps(alpha03, zero);
multiplied_luma03 = _mm_and_ps(mask_zero, multiplied_luma03);
let dst_pixel01 = _mm_unpacklo_ps(multiplied_luma03, alpha03);
let dst_pixel23 = _mm_unpackhi_ps(multiplied_luma03, alpha03);
let dst_ptr = dst_chunk.as_mut_ptr() as *mut f32;
_mm_storeu_ps(dst_ptr, dst_pixel01);
_mm_storeu_ps(dst_ptr.add(4), dst_pixel23);
}
+1
View File
@@ -8,6 +8,7 @@ cfg_if::cfg_if! {
mod u16x2;
mod u16x4;
mod u8x2;
mod f32x2;
}
}
+3 -3
View File
@@ -41,7 +41,7 @@ pub(crate) unsafe fn multiply_alpha_row(src_row: &[U16x2], dst_row: &mut [U16x2]
(pixels, dst_ptr)
},
|(mut pixels, dst_ptr)| {
pixels = multiplies_alpha_4_pixels(pixels);
pixels = multiply_alpha_4_pixels(pixels);
_mm_storeu_si128(dst_ptr, pixels);
},
);
@@ -64,7 +64,7 @@ pub(crate) unsafe fn multiply_alpha_row_inplace(row: &mut [U16x2]) {
(pixels, dst_ptr)
},
|(mut pixels, dst_ptr)| {
pixels = multiplies_alpha_4_pixels(pixels);
pixels = multiply_alpha_4_pixels(pixels);
_mm_storeu_si128(dst_ptr, pixels);
},
);
@@ -77,7 +77,7 @@ pub(crate) unsafe fn multiply_alpha_row_inplace(row: &mut [U16x2]) {
#[inline]
#[target_feature(enable = "sse4.1")]
unsafe fn multiplies_alpha_4_pixels(pixels: __m128i) -> __m128i {
unsafe fn multiply_alpha_4_pixels(pixels: __m128i) -> __m128i {
let zero = _mm_setzero_si128();
let half = _mm_set1_epi32(0x8000);
+68 -19
View File
@@ -1,5 +1,5 @@
use crate::pixels::{
InnerPixel, IntoPixelComponent, U16x2, U16x3, U16x4, U8x2, U8x3, U8x4, U16, U8,
F32x2, InnerPixel, IntoPixelComponent, U16x2, U16x3, U16x4, U8x2, U8x3, U8x4, F32, I32, U16, U8,
};
use crate::{
try_pixel_type, DifferentDimensionsError, ImageView, ImageViewMut, IntoImageView,
@@ -10,18 +10,15 @@ pub fn change_type_of_pixel_components(
src_image: &impl IntoImageView,
dst_image: &mut impl IntoImageViewMut,
) -> Result<(), MappingError> {
macro_rules! map {
($value:expr, $(($low_enum:path, $low_pt:ty, $high_enum:path, $high_pt:ty)),*) => {
match $value {
macro_rules! map_dst {
(
$src_pt:ty, $dst_type:expr,
$(($dst_enum:path, $dst_pt:ty)),*
) => {
match $dst_type {
$(
($low_enum, $low_enum) =>
change_components_type::<$low_pt, $low_pt>(src_image, dst_image),
($low_enum, $high_enum) =>
change_components_type::<$low_pt, $high_pt>(src_image, dst_image),
($high_enum, $low_enum) =>
change_components_type::<$high_pt, $low_pt>(src_image, dst_image),
($high_enum, $high_enum) =>
change_components_type::<$high_pt, $high_pt>(src_image, dst_image),
$dst_enum =>
change_components_type::<$src_pt, $dst_pt>(src_image, dst_image),
)*
_ => Err(MappingError::UnsupportedCombinationOfImageTypes),
}
@@ -33,13 +30,65 @@ pub fn change_type_of_pixel_components(
use PixelType as PT;
map!(
(src_pixel_type, dst_pixel_type),
(PT::U8, U8, PT::U16, U16),
(PT::U8x2, U8x2, PT::U16x2, U16x2),
(PT::U8x3, U8x3, PT::U16x3, U16x3),
(PT::U8x4, U8x4, PT::U16x4, U16x4)
)
match src_pixel_type {
PixelType::U8 => map_dst!(
U8,
dst_pixel_type,
(PT::U8, U8),
(PT::U16, U16),
(PT::I32, I32),
(PT::F32, F32)
),
PixelType::U8x2 => map_dst!(
U8x2,
dst_pixel_type,
(PT::U8x2, U8x2),
(PT::U16x2, U16x2),
(PT::F32x2, F32x2)
),
PixelType::U8x3 => map_dst!(U8x3, dst_pixel_type, (PT::U8x3, U8x3), (PT::U16x3, U16x3)),
PixelType::U8x4 => map_dst!(U8x4, dst_pixel_type, (PT::U8x4, U8x4), (PT::U16x4, U16x4)),
PixelType::U16 => map_dst!(
U16,
dst_pixel_type,
(PT::U8, U8),
(PT::U16, U16),
(PT::I32, I32),
(PT::F32, F32)
),
PixelType::U16x2 => map_dst!(
U16x2,
dst_pixel_type,
(PT::U8x2, U8x2),
(PT::U16x2, U16x2),
(PT::F32x2, F32x2)
),
PixelType::U16x3 => map_dst!(U16x3, dst_pixel_type, (PT::U8x3, U8x3), (PT::U16x3, U16x3)),
PixelType::U16x4 => map_dst!(U16x4, dst_pixel_type, (PT::U8x4, U8x4), (PT::U16x4, U16x4)),
PixelType::I32 => map_dst!(
I32,
dst_pixel_type,
(PT::U8, U8),
(PT::U16, U16),
(PT::I32, I32),
(PT::F32, F32)
),
PixelType::F32 => map_dst!(
F32,
dst_pixel_type,
(PT::U8, U8),
(PT::U16, U16),
(PT::I32, I32),
(PT::F32, F32)
),
PixelType::F32x2 => map_dst!(
F32x2,
dst_pixel_type,
(PT::U8x2, U8x2),
(PT::U16x2, U16x2),
(PT::F32x2, F32x2)
),
}
}
#[inline(always)]
+3 -2
View File
@@ -1,3 +1,4 @@
use crate::convolution::vertical_f32::vert_convolution_f32;
use crate::cpu_extensions::CpuExtensions;
use crate::pixels::F32;
use crate::{ImageView, ImageViewMut};
@@ -22,8 +23,8 @@ impl Convolution for F32 {
dst_view: &mut impl ImageViewMut<Pixel = Self>,
offset: u32,
coeffs: Coefficients,
_cpu_extensions: CpuExtensions,
cpu_extensions: CpuExtensions,
) {
native::vert_convolution(src_view, dst_view, offset, coeffs);
vert_convolution_f32(src_view, dst_view, offset, coeffs, cpu_extensions);
}
}
+1 -26
View File
@@ -19,32 +19,7 @@ pub(crate) fn horiz_convolution(
for (&k, &pixel) in coeffs_chunk.values.iter().zip(src_pixels) {
ss += pixel.0 as f64 * k;
}
dst_pixel.0 = ss.round() as f32;
}
}
}
pub(crate) fn vert_convolution(
src_view: &impl ImageView<Pixel = F32>,
dst_view: &mut impl ImageViewMut<Pixel = F32>,
offset: u32,
coeffs: Coefficients,
) {
let coefficients_chunks = coeffs.get_chunks();
let dst_rows = dst_view.iter_rows_mut(0);
let start_src_x = offset as usize;
for (&coeffs_chunk, dst_row) in coefficients_chunks.iter().zip(dst_rows) {
let first_y_src = coeffs_chunk.start;
let mut src_x = start_src_x;
for dst_pixel in dst_row.iter_mut() {
let mut ss = 0.;
let src_rows = src_view.iter_rows(first_y_src);
for (src_row, &k) in src_rows.zip(coeffs_chunk.values) {
let src_pixel = unsafe { src_row.get_unchecked(src_x) };
ss += src_pixel.0 as f64 * k;
}
dst_pixel.0 = ss.round() as f32;
src_x += 1;
dst_pixel.0 = ss as f32;
}
}
}
+186
View File
@@ -0,0 +1,186 @@
use std::arch::x86_64::*;
use crate::convolution::{Coefficients, CoefficientsChunk};
use crate::pixels::F32x2;
use crate::{simd_utils, ImageView, ImageViewMut};
#[inline]
pub(crate) fn horiz_convolution(
src_view: &impl ImageView<Pixel = F32x2>,
dst_view: &mut impl ImageViewMut<Pixel = F32x2>,
offset: u32,
coeffs: Coefficients,
) {
let coefficients_chunks = coeffs.get_chunks();
let dst_height = dst_view.height();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks);
}
}
let yy = dst_height - dst_height % 4;
let src_rows = src_view.iter_rows(yy + offset);
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks);
}
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - length of all rows in src_rows must be equal
/// - length of all rows in dst_rows must be equal
/// - coefficients_chunks.len() == dst_rows.0.len()
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "avx2")]
unsafe fn horiz_convolution_four_rows(
src_rows: [&[F32x2]; 4],
dst_rows: [&mut [F32x2]; 4],
coefficients_chunks: &[CoefficientsChunk],
) {
const ROWS_COUNT: usize = 4;
let mut ll_buf = [0f64; 2];
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
let mut ll_sum = [_mm256_set1_pd(0.); ROWS_COUNT];
let mut coeffs = coeffs_chunk.values;
let coeffs_by_4 = coeffs.chunks_exact(4);
coeffs = coeffs_by_4.remainder();
for k in coeffs_by_4 {
let coeff0_f64x4 = _mm256_set_pd(k[1], k[1], k[0], k[0]);
let coeff1_f64x4 = _mm256_set_pd(k[3], k[3], k[2], k[2]);
for i in 0..ROWS_COUNT {
let mut sum = ll_sum[i];
let pixels04_f32x8 = simd_utils::loadu_ps256(src_rows[i], x);
let pixels01_f64x4 = _mm256_cvtps_pd(_mm256_extractf128_ps::<0>(pixels04_f32x8));
sum = _mm256_add_pd(sum, _mm256_mul_pd(pixels01_f64x4, coeff0_f64x4));
let pixels23_f64x4 = _mm256_cvtps_pd(_mm256_extractf128_ps::<1>(pixels04_f32x8));
sum = _mm256_add_pd(sum, _mm256_mul_pd(pixels23_f64x4, coeff1_f64x4));
ll_sum[i] = sum;
}
x += 4;
}
let coeffs_by_2 = coeffs.chunks_exact(2);
coeffs = coeffs_by_2.remainder();
for k in coeffs_by_2 {
let coeff_f64x4 = _mm256_set_pd(k[1], k[1], k[0], k[0]);
for i in 0..ROWS_COUNT {
let mut sum = ll_sum[i];
let pixels01_f32x4 = simd_utils::loadu_ps(src_rows[i], x);
let pixels01_f64x4 = _mm256_cvtps_pd(pixels01_f32x4);
sum = _mm256_add_pd(sum, _mm256_mul_pd(pixels01_f64x4, coeff_f64x4));
ll_sum[i] = sum;
}
x += 2;
}
if let Some(&k) = coeffs.first() {
let coeff0_f64x4 = _mm256_set1_pd(k);
for i in 0..ROWS_COUNT {
let mut sum = ll_sum[i];
let pixel = src_rows[i].get_unchecked(x);
let pixel0_f64x4 = _mm256_set_pd(0., 0., pixel.0[1] as f64, pixel.0[0] as f64);
sum = _mm256_add_pd(sum, _mm256_mul_pd(pixel0_f64x4, coeff0_f64x4));
ll_sum[i] = sum;
}
}
for i in 0..ROWS_COUNT {
let sum_f64x2 = _mm_add_pd(
_mm256_extractf128_pd::<0>(ll_sum[i]),
_mm256_extractf128_pd::<1>(ll_sum[i]),
);
_mm_storeu_pd(ll_buf.as_mut_ptr(), sum_f64x2);
let dst_pixel = dst_rows[i].get_unchecked_mut(dst_x);
dst_pixel.0 = ll_buf.map(|v| v as f32);
}
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - bounds.len() == dst_row.len()
/// - coeffs.len() == dst_rows.0.len() * window_size
/// - max(bound.start + bound.size for bound in bounds) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
#[inline]
#[target_feature(enable = "avx2")]
unsafe fn horiz_convolution_one_row(
src_row: &[F32x2],
dst_row: &mut [F32x2],
coefficients_chunks: &[CoefficientsChunk],
) {
let mut ll_buf = [0f64; 2];
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
let mut ll_sum = _mm256_set1_pd(0.);
let mut coeffs = coeffs_chunk.values;
let coeffs_by_4 = coeffs.chunks_exact(4);
coeffs = coeffs_by_4.remainder();
for k in coeffs_by_4 {
let coeff0_f64x4 = _mm256_set_pd(k[1], k[1], k[0], k[0]);
let coeff1_f64x4 = _mm256_set_pd(k[3], k[3], k[2], k[2]);
let pixels04_f32x8 = simd_utils::loadu_ps256(src_row, x);
let pixels01_f64x4 = _mm256_cvtps_pd(_mm256_extractf128_ps::<0>(pixels04_f32x8));
ll_sum = _mm256_add_pd(ll_sum, _mm256_mul_pd(pixels01_f64x4, coeff0_f64x4));
let pixels23_f64x4 = _mm256_cvtps_pd(_mm256_extractf128_ps::<1>(pixels04_f32x8));
ll_sum = _mm256_add_pd(ll_sum, _mm256_mul_pd(pixels23_f64x4, coeff1_f64x4));
x += 4;
}
let coeffs_by_2 = coeffs.chunks_exact(2);
coeffs = coeffs_by_2.remainder();
for k in coeffs_by_2 {
let coeff_f64x4 = _mm256_set_pd(k[1], k[1], k[0], k[0]);
let pixels01_f32x4 = simd_utils::loadu_ps(src_row, x);
let pixels01_f64x4 = _mm256_cvtps_pd(pixels01_f32x4);
ll_sum = _mm256_add_pd(ll_sum, _mm256_mul_pd(pixels01_f64x4, coeff_f64x4));
x += 2;
}
if let Some(&k) = coeffs.first() {
let coeff0_f64x4 = _mm256_set1_pd(k);
let pixel = src_row.get_unchecked(x);
let pixel0_f64x4 = _mm256_set_pd(0., 0., pixel.0[1] as f64, pixel.0[0] as f64);
ll_sum = _mm256_add_pd(ll_sum, _mm256_mul_pd(pixel0_f64x4, coeff0_f64x4));
}
let sum_f64x2 = _mm_add_pd(
_mm256_extractf128_pd::<0>(ll_sum),
_mm256_extractf128_pd::<1>(ll_sum),
);
_mm_storeu_pd(ll_buf.as_mut_ptr(), sum_f64x2);
let dst_pixel = dst_row.get_unchecked_mut(dst_x);
dst_pixel.0 = ll_buf.map(|v| v as f32);
}
}
+48
View File
@@ -0,0 +1,48 @@
use crate::convolution::vertical_f32::vert_convolution_f32;
use crate::cpu_extensions::CpuExtensions;
use crate::pixels::F32x2;
use crate::{ImageView, ImageViewMut};
use super::{Coefficients, Convolution};
#[cfg(target_arch = "x86_64")]
mod avx2;
mod native;
// #[cfg(target_arch = "aarch64")]
// mod neon;
#[cfg(target_arch = "x86_64")]
mod sse4;
// #[cfg(target_arch = "wasm32")]
// mod wasm32;
impl Convolution for F32x2 {
fn horiz_convolution(
src_view: &impl ImageView<Pixel = Self>,
dst_view: &mut impl ImageViewMut<Pixel = Self>,
offset: u32,
coeffs: Coefficients,
cpu_extensions: CpuExtensions,
) {
match cpu_extensions {
#[cfg(target_arch = "x86_64")]
CpuExtensions::Avx2 => avx2::horiz_convolution(src_view, dst_view, offset, coeffs),
#[cfg(target_arch = "x86_64")]
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, coeffs),
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
_ => native::horiz_convolution(src_view, dst_view, offset, coeffs),
}
}
fn vert_convolution(
src_view: &impl ImageView<Pixel = Self>,
dst_view: &mut impl ImageViewMut<Pixel = Self>,
offset: u32,
coeffs: Coefficients,
cpu_extensions: CpuExtensions,
) {
vert_convolution_f32(src_view, dst_view, offset, coeffs, cpu_extensions);
}
}
+27
View File
@@ -0,0 +1,27 @@
use crate::convolution::Coefficients;
use crate::pixels::F32x2;
use crate::{ImageView, ImageViewMut};
pub(crate) fn horiz_convolution(
src_view: &impl ImageView<Pixel = F32x2>,
dst_view: &mut impl ImageViewMut<Pixel = F32x2>,
offset: u32,
coeffs: Coefficients,
) {
let coefficients_chunks = coeffs.get_chunks();
let src_rows = src_view.iter_rows(offset);
let dst_rows = dst_view.iter_rows_mut(0);
for (dst_row, src_row) in dst_rows.zip(src_rows) {
for (dst_pixel, coeffs_chunk) in dst_row.iter_mut().zip(&coefficients_chunks) {
let first_x_src = coeffs_chunk.start as usize;
let mut ss = [0.; 2];
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
for (&k, &src_pixel) in coeffs_chunk.values.iter().zip(src_pixels) {
for (s, c) in ss.iter_mut().zip(src_pixel.0) {
*s += c as f64 * k;
}
}
dst_pixel.0 = ss.map(|v| v as f32);
}
}
}
+152
View File
@@ -0,0 +1,152 @@
use std::arch::x86_64::*;
use crate::convolution::{Coefficients, CoefficientsChunk};
use crate::pixels::F32x2;
use crate::{simd_utils, ImageView, ImageViewMut};
#[inline]
pub(crate) fn horiz_convolution(
src_view: &impl ImageView<Pixel = F32x2>,
dst_view: &mut impl ImageViewMut<Pixel = F32x2>,
offset: u32,
coeffs: Coefficients,
) {
let coefficients_chunks = coeffs.get_chunks();
let dst_height = dst_view.height();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks);
}
}
let yy = dst_height - dst_height % 4;
let src_rows = src_view.iter_rows(yy + offset);
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks);
}
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - length of all rows in src_rows must be equal
/// - length of all rows in dst_rows must be equal
/// - coefficients_chunks.len() == dst_rows.0.len()
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "sse4.1")]
unsafe fn horiz_convolution_four_rows(
src_rows: [&[F32x2]; 4],
dst_rows: [&mut [F32x2]; 4],
coefficients_chunks: &[CoefficientsChunk],
) {
const ROWS_COUNT: usize = 4;
let mut ll_buf = [0f64; 2];
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
let mut ll_sum = [_mm_set1_pd(0.); ROWS_COUNT];
let mut coeffs = coeffs_chunk.values;
let coeffs_by_2 = coeffs.chunks_exact(2);
coeffs = coeffs_by_2.remainder();
for k in coeffs_by_2 {
let coeff0_f64x2 = _mm_set1_pd(k[0]);
let coeff1_f64x2 = _mm_set1_pd(k[1]);
for i in 0..ROWS_COUNT {
let mut sum = ll_sum[i];
let source = simd_utils::loadu_ps(src_rows[i], x);
let pixel0_f64 = _mm_cvtps_pd(source);
sum = _mm_add_pd(sum, _mm_mul_pd(pixel0_f64, coeff0_f64x2));
let pixel1_f64 = _mm_cvtps_pd(_mm_movehl_ps(source, source));
sum = _mm_add_pd(sum, _mm_mul_pd(pixel1_f64, coeff1_f64x2));
ll_sum[i] = sum;
}
x += 2;
}
if let Some(&k) = coeffs.first() {
let coeff0_f64x2 = _mm_set1_pd(k);
for i in 0..ROWS_COUNT {
let mut sum = ll_sum[i];
let pixel = src_rows[i].get_unchecked(x);
let source = _mm_set_ps(0., 0., pixel.0[1], pixel.0[0]);
let pixel0_f64 = _mm_cvtps_pd(source);
sum = _mm_add_pd(sum, _mm_mul_pd(pixel0_f64, coeff0_f64x2));
ll_sum[i] = sum;
}
}
for i in 0..ROWS_COUNT {
_mm_storeu_pd(ll_buf.as_mut_ptr(), ll_sum[i]);
let dst_pixel = dst_rows[i].get_unchecked_mut(dst_x);
dst_pixel.0 = ll_buf.map(|v| v as f32);
}
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - bounds.len() == dst_row.len()
/// - coeffs.len() == dst_rows.0.len() * window_size
/// - max(bound.start + bound.size for bound in bounds) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
#[inline]
#[target_feature(enable = "sse4.1")]
unsafe fn horiz_convolution_one_row(
src_row: &[F32x2],
dst_row: &mut [F32x2],
coefficients_chunks: &[CoefficientsChunk],
) {
let mut ll_buf = [0f64; 2];
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
let mut ll_sum = _mm_set1_pd(0.);
let mut coeffs = coeffs_chunk.values;
let coeffs_by_2 = coeffs.chunks_exact(2);
coeffs = coeffs_by_2.remainder();
for k in coeffs_by_2 {
let coeff0_f64x2 = _mm_set1_pd(k[0]);
let coeff1_f64x2 = _mm_set1_pd(k[1]);
let source = simd_utils::loadu_ps(src_row, x);
let pixel0_f64 = _mm_cvtps_pd(source);
ll_sum = _mm_add_pd(ll_sum, _mm_mul_pd(pixel0_f64, coeff0_f64x2));
let pixel1_f64 = _mm_cvtps_pd(_mm_movehl_ps(source, source));
ll_sum = _mm_add_pd(ll_sum, _mm_mul_pd(pixel1_f64, coeff1_f64x2));
x += 2;
}
if let Some(&k) = coeffs.first() {
let coeff0_f64x2 = _mm_set1_pd(k);
let pixel = src_row.get_unchecked(x);
let source = _mm_set_ps(0., 0., pixel.0[1], pixel.0[0]);
let pixel0_f64 = _mm_cvtps_pd(source);
ll_sum = _mm_add_pd(ll_sum, _mm_mul_pd(pixel0_f64, coeff0_f64x2));
}
_mm_storeu_pd(ll_buf.as_mut_ptr(), ll_sum);
let dst_pixel = dst_row.get_unchecked_mut(dst_x);
dst_pixel.0 = ll_buf.map(|v| v as f32);
}
}
+2
View File
@@ -21,7 +21,9 @@ cfg_if::cfg_if! {
mod u16x4;
mod i32x1;
mod f32x1;
mod f32x2;
mod vertical_u16;
mod vertical_f32;
}
}
+128
View File
@@ -0,0 +1,128 @@
use std::arch::x86_64::*;
use crate::convolution::{Coefficients, CoefficientsChunk};
use crate::pixels::InnerPixel;
use crate::{simd_utils, ImageView, ImageViewMut};
use super::native;
pub(crate) fn vert_convolution<T>(
src_view: &impl ImageView<Pixel = T>,
dst_view: &mut impl ImageViewMut<Pixel = T>,
offset: u32,
coeffs: Coefficients,
) where
T: InnerPixel<Component = f32>,
{
let coefficients_chunks = coeffs.get_chunks();
let src_x = offset as usize * T::count_of_components();
let dst_rows = dst_view.iter_rows_mut(0);
for (dst_row, coeffs_chunk) in dst_rows.zip(coefficients_chunks) {
unsafe {
vert_convolution_into_one_row_f32(src_view, dst_row, src_x, coeffs_chunk);
}
}
}
#[target_feature(enable = "avx2")]
unsafe fn vert_convolution_into_one_row_f32<T: InnerPixel<Component = f32>>(
src_view: &impl ImageView<Pixel = T>,
dst_row: &mut [T],
mut src_x: usize,
coeffs_chunk: CoefficientsChunk,
) {
let mut c_buf = [0f64; 4];
let mut dst_f32 = T::components_mut(dst_row);
let mut dst_chunks = dst_f32.chunks_exact_mut(32);
for dst_chunk in &mut dst_chunks {
multiply_components_of_rows::<_, 8>(src_view, src_x, coeffs_chunk, dst_chunk, &mut c_buf);
src_x += 32;
}
dst_f32 = dst_chunks.into_remainder();
dst_chunks = dst_f32.chunks_exact_mut(16);
for dst_chunk in &mut dst_chunks {
multiply_components_of_rows::<_, 4>(src_view, src_x, coeffs_chunk, dst_chunk, &mut c_buf);
src_x += 16;
}
dst_f32 = dst_chunks.into_remainder();
dst_chunks = dst_f32.chunks_exact_mut(8);
for dst_chunk in &mut dst_chunks {
multiply_components_of_rows::<_, 2>(src_view, src_x, coeffs_chunk, dst_chunk, &mut c_buf);
src_x += 8;
}
dst_f32 = dst_chunks.into_remainder();
if !dst_f32.is_empty() {
let y_start = coeffs_chunk.start;
let coeffs = coeffs_chunk.values;
native::convolution_by_f32(src_view, dst_f32, src_x, y_start, coeffs);
}
}
#[inline]
#[target_feature(enable = "avx2")]
unsafe fn multiply_components_of_rows<T: InnerPixel<Component = f32>, const SUMS_COUNT: usize>(
src_view: &impl ImageView<Pixel = T>,
src_x: usize,
coeffs_chunk: CoefficientsChunk,
dst_chunk: &mut [f32],
c_buf: &mut [f64; 4],
) {
let mut sums = [_mm256_set1_pd(0.); SUMS_COUNT];
let y_start = coeffs_chunk.start;
let mut coeffs = coeffs_chunk.values;
let mut y: u32 = 0;
let max_rows = coeffs.len() as u32;
let coeffs_2 = coeffs.chunks_exact(2);
coeffs = coeffs_2.remainder();
for (src_rows, two_coeffs) in src_view.iter_2_rows(y_start, max_rows).zip(coeffs_2) {
let src_rows = src_rows.map(|row| T::components(row).get_unchecked(src_x..));
for (&coeff, src_row) in two_coeffs.iter().zip(src_rows) {
multiply_components_of_row(&mut sums, coeff, src_row);
}
y += 2;
}
if let Some(&coeff) = coeffs.first() {
if let Some(s_row) = src_view.iter_rows(y_start + y).next() {
let src_row = T::components(s_row).get_unchecked(src_x..);
multiply_components_of_row(&mut sums, coeff, src_row);
}
}
let mut dst_ptr = dst_chunk.as_mut_ptr();
for sum in sums {
_mm256_storeu_pd(c_buf.as_mut_ptr(), sum);
for &v in c_buf.iter() {
*dst_ptr = v as f32;
dst_ptr = dst_ptr.add(1);
}
}
}
#[inline]
#[target_feature(enable = "avx2")]
unsafe fn multiply_components_of_row<const SUMS_COUNT: usize>(
sums: &mut [__m256d; SUMS_COUNT],
coeff: f64,
src_row: &[f32],
) {
let coeff_f64x4 = _mm256_set1_pd(coeff);
let mut i = 0;
while i < SUMS_COUNT {
let comp07_f32x8 = simd_utils::loadu_ps256(src_row, i * 4);
let comp03_f64x4 = _mm256_cvtps_pd(_mm256_extractf128_ps::<0>(comp07_f32x8));
sums[i] = _mm256_add_pd(sums[i], _mm256_mul_pd(comp03_f64x4, coeff_f64x4));
i += 1;
let comp47_f64x4 = _mm256_cvtps_pd(_mm256_extractf128_ps::<1>(comp07_f32x8));
sums[i] = _mm256_add_pd(sums[i], _mm256_mul_pd(comp47_f64x4, coeff_f64x4));
i += 1;
}
}
+37
View File
@@ -0,0 +1,37 @@
use crate::convolution::Coefficients;
use crate::pixels::InnerPixel;
use crate::{CpuExtensions, ImageView, ImageViewMut};
#[cfg(target_arch = "x86_64")]
pub(crate) mod avx2;
pub(crate) mod native;
// #[cfg(target_arch = "aarch64")]
// mod neon;
#[cfg(target_arch = "x86_64")]
pub(crate) mod sse4;
// #[cfg(target_arch = "wasm32")]
// pub mod wasm32;
pub(crate) fn vert_convolution_f32<T: InnerPixel<Component = f32>>(
src_view: &impl ImageView<Pixel = T>,
dst_view: &mut impl ImageViewMut<Pixel = T>,
offset: u32,
coeffs: Coefficients,
cpu_extensions: CpuExtensions,
) {
// Check safety conditions
debug_assert!(src_view.width() - offset >= dst_view.width());
debug_assert_eq!(coeffs.bounds.len(), dst_view.height() as usize);
match cpu_extensions {
#[cfg(target_arch = "x86_64")]
CpuExtensions::Avx2 => avx2::vert_convolution(src_view, dst_view, offset, coeffs),
#[cfg(target_arch = "x86_64")]
CpuExtensions::Sse4_1 => sse4::vert_convolution(src_view, dst_view, offset, coeffs),
// #[cfg(target_arch = "aarch64")]
// CpuExtensions::Neon => neon::vert_convolution(src_view, dst_view, offset, coeffs),
// #[cfg(target_arch = "wasm32")]
// CpuExtensions::Simd128 => wasm32::vert_convolution(src_view, dst_view, offset, coeffs),
_ => native::vert_convolution(src_view, dst_view, offset, coeffs),
}
}
+97
View File
@@ -0,0 +1,97 @@
use crate::convolution::Coefficients;
use crate::pixels::InnerPixel;
use crate::utils::foreach_with_pre_reading;
use crate::{ImageView, ImageViewMut};
#[inline(always)]
pub(crate) fn vert_convolution<T>(
src_view: &impl ImageView<Pixel = T>,
dst_view: &mut impl ImageViewMut<Pixel = T>,
offset: u32,
coeffs: Coefficients,
) where
T: InnerPixel<Component = f32>,
{
let coefficients_chunks = coeffs.get_chunks();
let src_x_initial = offset as usize * T::count_of_components();
let dst_rows = dst_view.iter_rows_mut(0);
let coeffs_chunks_iter = coefficients_chunks.into_iter();
for (coeffs_chunk, dst_row) in coeffs_chunks_iter.zip(dst_rows) {
let first_y_src = coeffs_chunk.start;
let ks = coeffs_chunk.values;
let dst_components = T::components_mut(dst_row);
let mut x_src = src_x_initial;
let (_, dst_chunks, tail) = unsafe { dst_components.align_to_mut::<[f32; 8]>() };
x_src = convolution_by_chunks(src_view, dst_chunks, x_src, first_y_src, ks);
if !tail.is_empty() {
convolution_by_f32(src_view, tail, x_src, first_y_src, ks);
}
}
}
#[inline(always)]
pub(crate) fn convolution_by_f32<T: InnerPixel<Component = f32>>(
src_view: &impl ImageView<Pixel = T>,
dst_components: &mut [f32],
mut x_src: usize,
first_y_src: u32,
ks: &[f64],
) -> usize {
for dst_component in dst_components.iter_mut() {
let mut ss = 0.;
let src_rows = src_view.iter_rows(first_y_src);
for (&k, src_row) in ks.iter().zip(src_rows) {
// SAFETY: Alignment of src_row is greater or equal than alignment f32
// because a component of pixel type T is f32.
let src_ptr = src_row.as_ptr() as *const f32;
let src_component = unsafe { *src_ptr.add(x_src) };
ss += src_component as f64 * k;
}
*dst_component = ss as f32;
x_src += 1
}
x_src
}
#[inline(always)]
fn convolution_by_chunks<T, const CHUNK_SIZE: usize>(
src_view: &impl ImageView<Pixel = T>,
dst_chunks: &mut [[f32; CHUNK_SIZE]],
mut x_src: usize,
first_y_src: u32,
ks: &[f64],
) -> usize
where
T: InnerPixel<Component = f32>,
{
for dst_chunk in dst_chunks {
let mut ss = [0.; CHUNK_SIZE];
let src_rows = src_view.iter_rows(first_y_src);
foreach_with_pre_reading(
ks.iter().zip(src_rows),
|(&k, src_row)| {
let src_ptr = src_row.as_ptr() as *const f32;
let src_chunk = unsafe {
let ptr = src_ptr.add(x_src) as *const [f32; CHUNK_SIZE];
ptr.read_unaligned()
};
(src_chunk, k)
},
|(src_chunk, k)| {
for (s, c) in ss.iter_mut().zip(src_chunk) {
*s += c as f64 * k;
}
},
);
for (i, s) in ss.iter().copied().enumerate() {
dst_chunk[i] = s as f32;
}
x_src += CHUNK_SIZE;
}
x_src
}
+131
View File
@@ -0,0 +1,131 @@
use std::arch::x86_64::*;
use crate::convolution::{Coefficients, CoefficientsChunk};
use crate::pixels::InnerPixel;
use crate::{simd_utils, ImageView, ImageViewMut};
use super::native;
pub(crate) fn vert_convolution<T>(
src_view: &impl ImageView<Pixel = T>,
dst_view: &mut impl ImageViewMut<Pixel = T>,
offset: u32,
coeffs: Coefficients,
) where
T: InnerPixel<Component = f32>,
{
let coefficients_chunks = coeffs.get_chunks();
let src_x = offset as usize * T::count_of_components();
let dst_rows = dst_view.iter_rows_mut(0);
for (dst_row, coeffs_chunk) in dst_rows.zip(coefficients_chunks) {
unsafe {
vert_convolution_into_one_row_f32(src_view, dst_row, src_x, coeffs_chunk);
}
}
}
#[target_feature(enable = "sse4.1")]
unsafe fn vert_convolution_into_one_row_f32<T: InnerPixel<Component = f32>>(
src_view: &impl ImageView<Pixel = T>,
dst_row: &mut [T],
mut src_x: usize,
coeffs_chunk: CoefficientsChunk,
) {
let mut c_buf = [0f64; 2];
let mut dst_f32 = T::components_mut(dst_row);
let mut dst_chunks = dst_f32.chunks_exact_mut(16);
for dst_chunk in &mut dst_chunks {
multiply_components_of_rows::<_, 8>(src_view, src_x, coeffs_chunk, dst_chunk, &mut c_buf);
src_x += 16;
}
dst_f32 = dst_chunks.into_remainder();
dst_chunks = dst_f32.chunks_exact_mut(8);
for dst_chunk in &mut dst_chunks {
multiply_components_of_rows::<_, 4>(src_view, src_x, coeffs_chunk, dst_chunk, &mut c_buf);
src_x += 8;
}
dst_f32 = dst_chunks.into_remainder();
dst_chunks = dst_f32.chunks_exact_mut(4);
if let Some(dst_chunk) = dst_chunks.next() {
multiply_components_of_rows::<_, 2>(src_view, src_x, coeffs_chunk, dst_chunk, &mut c_buf);
src_x += 4;
}
dst_f32 = dst_chunks.into_remainder();
if !dst_f32.is_empty() {
let y_start = coeffs_chunk.start;
let coeffs = coeffs_chunk.values;
native::convolution_by_f32(src_view, dst_f32, src_x, y_start, coeffs);
}
}
#[inline]
#[target_feature(enable = "sse4.1")]
pub(crate) unsafe fn multiply_components_of_rows<
T: InnerPixel<Component = f32>,
const SUMS_COUNT: usize,
>(
src_view: &impl ImageView<Pixel = T>,
src_x: usize,
coeffs_chunk: CoefficientsChunk,
dst_chunk: &mut [f32],
c_buf: &mut [f64; 2],
) {
let mut sums = [_mm_set1_pd(0.); SUMS_COUNT];
let y_start = coeffs_chunk.start;
let mut coeffs = coeffs_chunk.values;
let mut y: u32 = 0;
let max_rows = coeffs.len() as u32;
let coeffs_2 = coeffs.chunks_exact(2);
coeffs = coeffs_2.remainder();
for (src_rows, two_coeffs) in src_view.iter_2_rows(y_start, max_rows).zip(coeffs_2) {
let src_rows = src_rows.map(|row| T::components(row).get_unchecked(src_x..));
for (&coeff, src_row) in two_coeffs.iter().zip(src_rows) {
multiply_components_of_row(&mut sums, coeff, src_row);
}
y += 2;
}
if let Some(&coeff) = coeffs.first() {
if let Some(s_row) = src_view.iter_rows(y_start + y).next() {
let src_row = T::components(s_row).get_unchecked(src_x..);
multiply_components_of_row(&mut sums, coeff, src_row);
}
}
let mut dst_ptr = dst_chunk.as_mut_ptr();
for sum in sums {
_mm_storeu_pd(c_buf.as_mut_ptr(), sum);
for &v in c_buf.iter() {
*dst_ptr = v as f32;
dst_ptr = dst_ptr.add(1);
}
}
}
#[inline]
#[target_feature(enable = "sse4.1")]
unsafe fn multiply_components_of_row<const SUMS_COUNT: usize>(
sums: &mut [__m128d; SUMS_COUNT],
coeff: f64,
src_row: &[f32],
) {
let coeff_f64x2 = _mm_set1_pd(coeff);
let mut i = 0;
while i < SUMS_COUNT {
let comp03_f32x4 = simd_utils::loadu_ps(src_row, i * 2);
let comp01_f64x2 = _mm_cvtps_pd(comp03_f32x4);
sums[i] = _mm_add_pd(sums[i], _mm_mul_pd(comp01_f64x2, coeff_f64x2));
i += 1;
let comp23_f64x2 = _mm_cvtps_pd(_mm_movehl_ps(comp03_f32x4, comp03_f32x4));
sums[i] = _mm_add_pd(sums[i], _mm_mul_pd(comp23_f64x2, coeff_f64x2));
i += 1;
}
}
+11 -3
View File
@@ -1,13 +1,13 @@
use crate::cpu_extensions::CpuExtensions;
use crate::image_view::{try_pixel_type, ImageViewMut, IntoImageView, IntoImageViewMut};
use crate::pixels::{U16x2, U16x4, U8x2, U8x4};
use crate::pixels::{F32x2, U16x2, U16x4, U8x2, U8x4};
use crate::{ImageError, ImageView, MulDivImagesError, PixelTrait, PixelType};
/// Methods of this structure used to multiply or divide color-channels (RGB or Luma)
/// by alpha-channel. Supported pixel types: U8x2, U8x4, U16x2 and U16x4.
///
/// By default, instance of `MulDiv` created with best CPU-extensions provided by your CPU.
/// You can change this by use method [MulDiv::set_cpu_extensions].
/// You can change this by using method [MulDiv::set_cpu_extensions].
///
/// # Examples
///
@@ -64,6 +64,7 @@ impl MulDiv {
PixelType::U8x4 => self.multiply::<U8x4>(src_image, dst_image),
PixelType::U16x2 => self.multiply::<U16x2>(src_image, dst_image),
PixelType::U16x4 => self.multiply::<U16x4>(src_image, dst_image),
PixelType::F32x2 => self.multiply::<F32x2>(src_image, dst_image),
_ => Err(MulDivImagesError::ImageError(
ImageError::UnsupportedPixelType,
)),
@@ -119,6 +120,7 @@ impl MulDiv {
PixelType::U8x4 => self.multiply_inplace::<U8x4>(image),
PixelType::U16x2 => self.multiply_inplace::<U16x2>(image),
PixelType::U16x4 => self.multiply_inplace::<U16x4>(image),
PixelType::F32x2 => self.multiply_inplace::<F32x2>(image),
_ => Err(ImageError::UnsupportedPixelType),
}
@@ -170,6 +172,7 @@ impl MulDiv {
PixelType::U8x4 => self.divide::<U8x4>(src_image, dst_image),
PixelType::U16x2 => self.divide::<U16x2>(src_image, dst_image),
PixelType::U16x4 => self.divide::<U16x4>(src_image, dst_image),
PixelType::F32x2 => self.divide::<F32x2>(src_image, dst_image),
_ => Err(MulDivImagesError::ImageError(
ImageError::UnsupportedPixelType,
)),
@@ -225,6 +228,7 @@ impl MulDiv {
PixelType::U8x4 => self.divide_inplace::<U8x4>(image),
PixelType::U16x2 => self.divide_inplace::<U16x2>(image),
PixelType::U16x4 => self.divide_inplace::<U16x4>(image),
PixelType::F32x2 => self.divide_inplace::<F32x2>(image),
_ => Err(ImageError::UnsupportedPixelType),
}
@@ -262,7 +266,11 @@ impl MulDiv {
{
matches!(
pixel_type,
PixelType::U8x2 | PixelType::U8x4 | PixelType::U16x2 | PixelType::U16x4
PixelType::U8x2
| PixelType::U8x4
| PixelType::U16x2
| PixelType::U16x4
| PixelType::F32x2
)
}
#[cfg(feature = "only_u8x4")]
+96 -8
View File
@@ -17,6 +17,7 @@ pub enum PixelType {
U16x4,
I32,
F32,
F32x2,
}
impl PixelType {
@@ -30,6 +31,7 @@ impl PixelType {
Self::U16x2 => 4,
Self::U16x3 => 6,
Self::U16x4 => 8,
Self::F32x2 => 8,
_ => 4,
}
}
@@ -47,6 +49,7 @@ impl PixelType {
Self::U16x4 => unsafe { buffer.align_to::<U16x4>().0.is_empty() },
Self::I32 => unsafe { buffer.align_to::<I32>().0.is_empty() },
Self::F32 => unsafe { buffer.align_to::<F32>().0.is_empty() },
Self::F32x2 => unsafe { buffer.align_to::<F32x2>().0.is_empty() },
}
}
}
@@ -94,15 +97,15 @@ where
}
impl PixelComponent for u8 {
type CountOfComponentValues = Values<256>;
type CountOfComponentValues = Values<0x100>;
}
impl PixelComponent for u16 {
type CountOfComponentValues = Values<65536>;
type CountOfComponentValues = Values<0x10000>;
}
impl PixelComponent for i32 {
type CountOfComponentValues = Values<0>;
type CountOfComponentValues = Values<0x100000000>;
}
impl PixelComponent for f32 {
@@ -299,6 +302,14 @@ pixel_struct!(
PixelType::F32,
"One `f32` component per pixel"
);
pixel_struct!(
F32x2,
[f32; 2],
f32,
2,
PixelType::F32x2,
"Two `f32` component per pixel (e.g. LA-F32)"
);
pub trait IntoPixelComponent<Out: PixelComponent>
where
@@ -313,14 +324,91 @@ impl<C: PixelComponent> IntoPixelComponent<C> for C {
}
}
impl IntoPixelComponent<u8> for u16 {
fn into_component(self) -> u8 {
self.to_le_bytes()[1]
}
}
// u8
impl IntoPixelComponent<u16> for u8 {
fn into_component(self) -> u16 {
u16::from_le_bytes([self, self])
}
}
impl IntoPixelComponent<i32> for u8 {
fn into_component(self) -> i32 {
(self as i32) << 23
}
}
impl IntoPixelComponent<f32> for u8 {
fn into_component(self) -> f32 {
(self as f32) / u8::MAX as f32
}
}
// u16
impl IntoPixelComponent<u8> for u16 {
fn into_component(self) -> u8 {
self.to_le_bytes()[1]
}
}
impl IntoPixelComponent<i32> for u16 {
fn into_component(self) -> i32 {
(self as i32) << 15
}
}
impl IntoPixelComponent<f32> for u16 {
fn into_component(self) -> f32 {
(self as f32) / u16::MAX as f32
}
}
// i32
impl IntoPixelComponent<u8> for i32 {
fn into_component(self) -> u8 {
(self.max(0).saturating_add(1 << 22) >> 23) as u8
}
}
impl IntoPixelComponent<u16> for i32 {
fn into_component(self) -> u16 {
(self.max(0).saturating_add(1 << 14) >> 15) as u16
}
}
impl IntoPixelComponent<f32> for i32 {
fn into_component(self) -> f32 {
if self < 0 {
(self as f32) / i32::MIN as f32
} else {
(self as f32) / i32::MAX as f32
}
}
}
// f32
impl IntoPixelComponent<u8> for f32 {
fn into_component(self) -> u8 {
(self.clamp(0., 1.) * u8::MAX as f32).round() as u8
}
}
impl IntoPixelComponent<u16> for f32 {
fn into_component(self) -> u16 {
(self.clamp(0., 1.) * u16::MAX as f32).round() as u16
}
}
impl IntoPixelComponent<i32> for f32 {
fn into_component(self) -> i32 {
let max = if self < 0. {
i32::MIN as f32
} else {
i32::MAX as f32
};
(self.clamp(-1., 1.) * max).round() as i32
}
}
+2 -1
View File
@@ -177,6 +177,7 @@ impl Resizer {
(PT::U16x4, pixels::U16x4),
(PT::I32, pixels::I32),
(PT::F32, pixels::F32),
(PT::F32x2, pixels::F32x2),
);
#[cfg(feature = "only_u8x4")]
@@ -538,7 +539,7 @@ fn resample_nearest<P: InnerPixel>(
let dst_rows = dst_view.iter_rows_mut(0);
for (out_row, in_row) in dst_rows.zip(src_rows) {
for (&x_in, out_pixel) in x_in_tab.iter().zip(out_row.iter_mut()) {
// Safety of value of x_in guaranteed by algorithm of creating of x_in_tab
// Safety of x_in value guaranteed by algorithm of creating of x_in_tab
*out_pixel = unsafe { *in_row.get_unchecked(x_in) };
}
}
+10
View File
@@ -30,6 +30,16 @@ pub unsafe fn loadl_epi64<T>(buf: &[T], index: usize) -> __m128i {
_mm_loadl_epi64(buf.get_unchecked(index..).as_ptr() as *const __m128i)
}
#[inline(always)]
pub unsafe fn loadu_ps<T>(buf: &[T], index: usize) -> __m128 {
_mm_loadu_ps(buf.get_unchecked(index..).as_ptr() as *const f32)
}
#[inline(always)]
pub unsafe fn loadu_ps256<T>(buf: &[T], index: usize) -> __m256 {
_mm256_loadu_ps(buf.get_unchecked(index..).as_ptr() as *const f32)
}
#[inline(always)]
pub unsafe fn mm_cvtepu8_epi32(buf: &[U8x4], index: usize) -> __m128i {
let v: i32 = transmute(buf.get_unchecked(index).0);
+229 -149
View File
@@ -1,11 +1,14 @@
use std::fs::File;
use std::io::BufReader;
use std::num::NonZeroU32;
use std::ops::Deref;
use image::io::Reader as ImageReader;
use image::{ColorType, DynamicImage};
use image::io::{Reader as ImageReader, Reader};
use image::{ColorType, ExtendedColorType, ImageBuffer};
use fast_image_resize::images::Image;
use fast_image_resize::pixels::*;
use fast_image_resize::{CpuExtensions, PixelTrait, PixelType};
use fast_image_resize::{change_type_of_pixel_components, CpuExtensions, PixelTrait, PixelType};
pub fn nonzero(v: u32) -> NonZeroU32 {
NonZeroU32::new(v).unwrap()
@@ -27,12 +30,23 @@ pub fn image_checksum<P: PixelTrait, const N: usize>(image: &Image) -> [u64; N]
res.iter_mut().zip(pixel).for_each(|(d, &s)| *d += s as u64);
}
}
4 => {
let buffer_u32 = unsafe { buffer.align_to::<u32>().1 };
for pixel in buffer_u32.chunks_exact(N) {
res.iter_mut()
.zip(pixel)
.for_each(|(d, &s)| *d = d.overflowing_add(s as u64).0);
}
}
_ => (),
};
res
}
pub trait PixelTestingExt: PixelTrait {
type ImagePixel: image::Pixel;
type Container: Deref<Target = [<Self::ImagePixel as image::Pixel>::Subpixel]>;
fn pixel_type_str() -> &'static str {
match Self::pixel_type() {
PixelType::U8 => "u8",
@@ -45,22 +59,68 @@ pub trait PixelTestingExt: PixelTrait {
PixelType::U16x4 => "u16x4",
PixelType::I32 => "i32",
PixelType::F32 => "f32",
PixelType::F32x2 => "f32x2",
_ => unreachable!(),
}
}
fn load_big_image() -> DynamicImage {
ImageReader::open("./data/nasa-4928x3279.png")
.unwrap()
.decode()
.unwrap()
fn cpu_extensions() -> Vec<CpuExtensions> {
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
cpu_extensions_vec
}
fn load_big_square_image() -> DynamicImage {
ImageReader::open("./data/nasa-4019x4019.png")
.unwrap()
.decode()
.unwrap()
fn img_paths() -> (&'static str, &'static str, &'static str) {
match Self::pixel_type() {
PixelType::U8
| PixelType::U8x3
| PixelType::U16
| PixelType::U16x3
| PixelType::I32
| PixelType::F32 => (
"./data/nasa-4928x3279.png",
"./data/nasa-4019x4019.png",
"./data/nasa-852x567.png",
),
PixelType::U8x2
| PixelType::U8x4
| PixelType::U16x2
| PixelType::U16x4
| PixelType::F32x2 => (
"./data/nasa-4928x3279-rgba.png",
"./data/nasa-4019x4019-rgba.png",
"./data/nasa-852x567-rgba.png",
),
_ => unreachable!(),
}
}
fn load_image_buffer(
img_reader: Reader<BufReader<File>>,
) -> ImageBuffer<Self::ImagePixel, Self::Container>;
fn load_big_image() -> ImageBuffer<Self::ImagePixel, Self::Container> {
Self::load_image_buffer(ImageReader::open(Self::img_paths().0).unwrap())
}
fn load_big_square_image() -> ImageBuffer<Self::ImagePixel, Self::Container> {
Self::load_image_buffer(ImageReader::open(Self::img_paths().1).unwrap())
}
fn load_small_image() -> ImageBuffer<Self::ImagePixel, Self::Container> {
Self::load_image_buffer(ImageReader::open(Self::img_paths().2).unwrap())
}
fn load_big_src_image() -> Image<'static> {
@@ -85,13 +145,6 @@ pub trait PixelTestingExt: PixelTrait {
.unwrap()
}
fn load_small_image() -> DynamicImage {
ImageReader::open("./data/nasa-852x567.png")
.unwrap()
.decode()
.unwrap()
}
fn load_small_src_image() -> Image<'static> {
let img = Self::load_small_image();
Image::from_vec_u8(
@@ -103,181 +156,195 @@ pub trait PixelTestingExt: PixelTrait {
.unwrap()
}
fn img_into_bytes(img: DynamicImage) -> Vec<u8>;
fn img_into_bytes(img: ImageBuffer<Self::ImagePixel, Self::Container>) -> Vec<u8>;
}
impl PixelTestingExt for U8 {
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
img.to_luma8().into_raw()
type ImagePixel = image::Luma<u8>;
type Container = Vec<u8>;
fn load_image_buffer(
img_reader: Reader<BufReader<File>>,
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
img_reader.decode().unwrap().to_luma8()
}
fn img_into_bytes(img: ImageBuffer<Self::ImagePixel, Self::Container>) -> Vec<u8> {
img.into_raw()
}
}
impl PixelTestingExt for U8x2 {
fn load_big_image() -> DynamicImage {
ImageReader::open("./data/nasa-4928x3279-rgba.png")
.unwrap()
.decode()
.unwrap()
type ImagePixel = image::LumaA<u8>;
type Container = Vec<u8>;
fn load_image_buffer(
img_reader: Reader<BufReader<File>>,
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
img_reader.decode().unwrap().to_luma_alpha8()
}
fn load_big_square_image() -> DynamicImage {
ImageReader::open("./data/nasa-4019x4019-rgba.png")
.unwrap()
.decode()
.unwrap()
}
fn load_small_image() -> DynamicImage {
ImageReader::open("./data/nasa-852x567-rgba.png")
.unwrap()
.decode()
.unwrap()
}
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
img.to_luma_alpha8().into_raw()
fn img_into_bytes(img: ImageBuffer<Self::ImagePixel, Self::Container>) -> Vec<u8> {
img.into_raw()
}
}
impl PixelTestingExt for U8x3 {
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
img.to_rgb8().into_raw()
type ImagePixel = image::Rgb<u8>;
type Container = Vec<u8>;
fn load_image_buffer(
img_reader: Reader<BufReader<File>>,
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
img_reader.decode().unwrap().to_rgb8()
}
fn img_into_bytes(img: ImageBuffer<Self::ImagePixel, Self::Container>) -> Vec<u8> {
img.into_raw()
}
}
impl PixelTestingExt for U8x4 {
fn load_big_image() -> DynamicImage {
ImageReader::open("./data/nasa-4928x3279-rgba.png")
.unwrap()
.decode()
.unwrap()
type ImagePixel = image::Rgba<u8>;
type Container = Vec<u8>;
fn load_image_buffer(
img_reader: Reader<BufReader<File>>,
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
img_reader.decode().unwrap().to_rgba8()
}
fn load_big_square_image() -> DynamicImage {
ImageReader::open("./data/nasa-4019x4019-rgba.png")
.unwrap()
.decode()
.unwrap()
}
fn load_small_image() -> DynamicImage {
ImageReader::open("./data/nasa-852x567-rgba.png")
.unwrap()
.decode()
.unwrap()
}
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
img.to_rgba8().into_raw()
fn img_into_bytes(img: ImageBuffer<Self::ImagePixel, Self::Container>) -> Vec<u8> {
img.into_raw()
}
}
impl PixelTestingExt for U16 {
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
// img.to_luma16()
// .as_raw()
type ImagePixel = image::Luma<u16>;
type Container = Vec<u16>;
fn load_image_buffer(
img_reader: Reader<BufReader<File>>,
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
img_reader.decode().unwrap().to_luma16()
}
fn img_into_bytes(img: ImageBuffer<Self::ImagePixel, Self::Container>) -> Vec<u8> {
// img.as_raw()
// .iter()
// .enumerate()
// .flat_map(|(i, &c)| ((i & 0xffff) as u16).to_le_bytes())
// .collect()
img.to_luma16()
.as_raw()
.iter()
.flat_map(|&c| c.to_le_bytes())
.collect()
img.as_raw().iter().flat_map(|&c| c.to_le_bytes()).collect()
}
}
impl PixelTestingExt for U16x2 {
fn load_big_image() -> DynamicImage {
ImageReader::open("./data/nasa-4928x3279-rgba.png")
.unwrap()
.decode()
.unwrap()
type ImagePixel = image::LumaA<u16>;
type Container = Vec<u16>;
fn load_image_buffer(
img_reader: Reader<BufReader<File>>,
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
img_reader.decode().unwrap().to_luma_alpha16()
}
fn load_big_square_image() -> DynamicImage {
ImageReader::open("./data/nasa-4019x4019-rgba.png")
.unwrap()
.decode()
.unwrap()
}
fn load_small_image() -> DynamicImage {
ImageReader::open("./data/nasa-852x567-rgba.png")
.unwrap()
.decode()
.unwrap()
}
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
img.to_luma_alpha16()
.as_raw()
.iter()
.flat_map(|&c| c.to_le_bytes())
.collect()
fn img_into_bytes(img: ImageBuffer<Self::ImagePixel, Self::Container>) -> Vec<u8> {
img.as_raw().iter().flat_map(|&c| c.to_le_bytes()).collect()
}
}
impl PixelTestingExt for U16x3 {
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
img.to_rgb8()
.as_raw()
.iter()
.flat_map(|&c| [c, c])
.collect()
type ImagePixel = image::Rgb<u16>;
type Container = Vec<u16>;
fn load_image_buffer(
img_reader: Reader<BufReader<File>>,
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
img_reader.decode().unwrap().to_rgb16()
}
fn img_into_bytes(img: ImageBuffer<Self::ImagePixel, Self::Container>) -> Vec<u8> {
img.as_raw().iter().flat_map(|&c| c.to_le_bytes()).collect()
}
}
impl PixelTestingExt for U16x4 {
fn load_big_image() -> DynamicImage {
ImageReader::open("./data/nasa-4928x3279-rgba.png")
.unwrap()
.decode()
.unwrap()
type ImagePixel = image::Rgba<u16>;
type Container = Vec<u16>;
fn load_image_buffer(
img_reader: Reader<BufReader<File>>,
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
img_reader.decode().unwrap().to_rgba16()
}
fn load_big_square_image() -> DynamicImage {
ImageReader::open("./data/nasa-4019x4019-rgba.png")
.unwrap()
.decode()
.unwrap()
}
fn load_small_image() -> DynamicImage {
ImageReader::open("./data/nasa-852x567-rgba.png")
.unwrap()
.decode()
.unwrap()
}
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
img.to_rgba16()
.as_raw()
.iter()
.flat_map(|&c| c.to_le_bytes())
.collect()
fn img_into_bytes(img: ImageBuffer<Self::ImagePixel, Self::Container>) -> Vec<u8> {
img.as_raw().iter().flat_map(|&c| c.to_le_bytes()).collect()
}
}
impl PixelTestingExt for I32 {
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
img.to_luma16()
.as_raw()
type ImagePixel = image::Luma<i32>;
type Container = Vec<i32>;
fn cpu_extensions() -> Vec<CpuExtensions> {
vec![CpuExtensions::None]
}
fn load_image_buffer(
img_reader: Reader<BufReader<File>>,
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
let image_u16 = img_reader.decode().unwrap().to_luma32f();
ImageBuffer::from_fn(image_u16.width(), image_u16.height(), |x, y| {
let pixel = image_u16.get_pixel(x, y);
image::Luma::from([(pixel.0[0] * i32::MAX as f32).round() as i32])
})
}
fn img_into_bytes(img: ImageBuffer<Self::ImagePixel, Self::Container>) -> Vec<u8> {
img.as_raw()
.iter()
.map(|&p| p as u32 * (i16::MAX as u32 + 1))
.flat_map(|val| val.to_le_bytes())
.collect()
}
}
impl PixelTestingExt for F32 {
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
img.to_luma16()
.as_raw()
type ImagePixel = image::Luma<f32>;
type Container = Vec<f32>;
fn cpu_extensions() -> Vec<CpuExtensions> {
vec![CpuExtensions::None]
}
fn load_image_buffer(
img_reader: Reader<BufReader<File>>,
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
img_reader.decode().unwrap().to_luma32f()
}
fn img_into_bytes(img: ImageBuffer<Self::ImagePixel, Self::Container>) -> Vec<u8> {
img.as_raw()
.iter()
.flat_map(|val| val.to_le_bytes())
.collect()
}
}
impl PixelTestingExt for F32x2 {
type ImagePixel = image::LumaA<f32>;
type Container = Vec<f32>;
fn load_image_buffer(
img_reader: Reader<BufReader<File>>,
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
img_reader.decode().unwrap().to_luma_alpha32f()
}
fn img_into_bytes(img: ImageBuffer<Self::ImagePixel, Self::Container>) -> Vec<u8> {
img.as_raw()
.iter()
.map(|&p| p as f32 * (i16::MAX as f32 + 1.0))
.flat_map(|val| val.to_le_bytes())
.collect()
}
@@ -289,15 +356,28 @@ pub fn save_result(image: &Image, name: &str) {
}
std::fs::create_dir_all("./data/result").unwrap();
let path = format!("./data/result/{}.png", name);
let color_type = match image.pixel_type() {
PixelType::U8 => ColorType::L8,
PixelType::U8x2 => ColorType::La8,
PixelType::U8x3 => ColorType::Rgb8,
PixelType::U8x4 => ColorType::Rgba8,
PixelType::U16 => ColorType::L16,
PixelType::U16x2 => ColorType::La16,
PixelType::U16x3 => ColorType::Rgb16,
PixelType::U16x4 => ColorType::Rgba16,
let color_type: ExtendedColorType = match image.pixel_type() {
PixelType::U8 => ColorType::L8.into(),
PixelType::U8x2 => ColorType::La8.into(),
PixelType::U8x3 => ColorType::Rgb8.into(),
PixelType::U8x4 => ColorType::Rgba8.into(),
PixelType::U16 => ColorType::L16.into(),
PixelType::U16x2 => ColorType::La16.into(),
PixelType::U16x3 => ColorType::Rgb16.into(),
PixelType::U16x4 => ColorType::Rgba16.into(),
PixelType::I32 | PixelType::F32 => {
let mut image_u16 = Image::new(image.width(), image.height(), PixelType::U16);
change_type_of_pixel_components(image, &mut image_u16).unwrap();
save_result(&image_u16, name);
return;
}
PixelType::F32x2 => {
let mut image_u16 = Image::new(image.width(), image.height(), PixelType::U16x2);
change_type_of_pixel_components(image, &mut image_u16).unwrap();
save_result(&image_u16, name);
return;
}
_ => panic!("Unsupported type of pixels"),
};
image::save_buffer(
+234 -295
View File
@@ -8,54 +8,6 @@ enum Oper {
Div,
}
struct TestCaseU16 {
pub color: u16,
pub alpha: u16,
pub expected_color: u16,
}
const fn new_case_16(c: u16, a: u16, e: u16) -> TestCaseU16 {
TestCaseU16 {
color: c,
alpha: a,
expected_color: e,
}
}
fn full_mul_div_alpha_test_u8<P: PixelTrait<Component = u8>>(
create_pixel: fn(u8, u8) -> P,
cpu_extensions: CpuExtensions,
) {
const PRECISION: u32 = 8;
const ALPHA_SCALE: u32 = 255u32 * (1 << (PRECISION + 1));
const ROUND_CORRECTION: u32 = 1 << (PRECISION - 1);
for oper in [Oper::Mul, Oper::Div] {
for color in 0u8..=255u8 {
for alpha in 0u8..=255u8 {
let result_color = if alpha == 0 {
0
} else {
match oper {
Oper::Mul => {
let tmp = color as u32 * alpha as u32 + 128;
(((tmp >> 8) + tmp) >> 8) as u8
}
Oper::Div => {
let recip_alpha = ((ALPHA_SCALE / alpha as u32) + 1) >> 1;
let tmp = (color as u32 * recip_alpha + ROUND_CORRECTION) >> PRECISION;
tmp.min(255) as u8
}
}
};
let src = [create_pixel(color, alpha)];
let res = [create_pixel(result_color, alpha)];
mul_div_alpha_test(oper, &src, &res, cpu_extensions);
}
}
}
}
fn mul_div_alpha_test<P: PixelTrait>(
oper: Oper,
src_pixels_tpl: &[P],
@@ -168,21 +120,7 @@ where
let mut alpha_mul_div: MulDiv = Default::default();
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
for cpu_extensions in P::cpu_extensions() {
if !cpu_extensions.is_supported() {
println!(
"Cpu Extensions '{}' not supported by your CPU",
@@ -227,67 +165,149 @@ where
}
}
mod u8x4 {
use fast_image_resize::pixels::U8x4;
mod u8_tests {
use super::*;
const fn new_u8x4(c: u8, a: u8) -> U8x4 {
U8x4::new([c, c, c, a])
fn full_mul_div_alpha_test_u8<P: PixelTrait<Component = u8>>(
create_pixel: fn(u8, u8) -> P,
cpu_extensions: CpuExtensions,
) {
const PRECISION: u32 = 8;
const ALPHA_SCALE: u32 = 255u32 * (1 << (PRECISION + 1));
const ROUND_CORRECTION: u32 = 1 << (PRECISION - 1);
for oper in [Oper::Mul, Oper::Div] {
for color in 0u8..=255u8 {
for alpha in 0u8..=255u8 {
let result_color = if alpha == 0 {
0
} else {
match oper {
Oper::Mul => {
let tmp = color as u32 * alpha as u32 + 128;
(((tmp >> 8) + tmp) >> 8) as u8
}
Oper::Div => {
let recip_alpha = ((ALPHA_SCALE / alpha as u32) + 1) >> 1;
let tmp =
(color as u32 * recip_alpha + ROUND_CORRECTION) >> PRECISION;
tmp.min(255) as u8
}
}
};
let src = [create_pixel(color, alpha)];
let res = [create_pixel(result_color, alpha)];
mul_div_alpha_test(oper, &src, &res, cpu_extensions);
}
}
}
}
#[test]
fn native_test() {
full_mul_div_alpha_test_u8(new_u8x4, CpuExtensions::None);
#[cfg(not(feature = "only_u8x4"))]
#[cfg(test)]
mod u8x2 {
use fast_image_resize::pixels::U8x2;
use super::*;
type P = U8x2;
const fn new_pixel(l: u8, a: u8) -> P {
P::new([l, a])
}
#[test]
fn mul_div_alpha_test() {
for cpu_extensions in P::cpu_extensions() {
full_mul_div_alpha_test_u8(new_pixel, cpu_extensions);
}
}
#[test]
fn multiply_real_image() {
run_tests_with_real_image_u8::<P, 2>(Oper::Mul, [4177920, 8355840]);
}
#[test]
fn divide_real_image() {
run_tests_with_real_image_u8::<P, 2>(Oper::Div, [12452343, 8355840]);
}
}
#[cfg(target_arch = "x86_64")]
#[test]
fn sse4_test() {
full_mul_div_alpha_test_u8(new_u8x4, CpuExtensions::Sse4_1);
}
#[cfg(test)]
mod u8x4 {
use fast_image_resize::pixels::U8x4;
#[cfg(target_arch = "x86_64")]
#[test]
fn avx2_test() {
full_mul_div_alpha_test_u8(new_u8x4, CpuExtensions::Avx2);
}
use super::*;
#[cfg(target_arch = "aarch64")]
#[test]
fn neon_test() {
full_mul_div_alpha_test_u8(new_u8x4, CpuExtensions::Neon);
}
type P = U8x4;
#[cfg(target_arch = "wasm32")]
#[test]
fn wasm32_test() {
full_mul_div_alpha_test_u8(new_u8x4, CpuExtensions::Simd128);
}
const fn new_pixel(c: u8, a: u8) -> P {
P::new([c, c, c, a])
}
#[test]
fn multiply_real_image() {
run_tests_with_real_image_u8::<U8x4, 4>(Oper::Mul, [4177920, 4177920, 4177920, 8355840]);
}
#[test]
fn mul_div_alpha_test() {
for cpu_extensions in P::cpu_extensions() {
full_mul_div_alpha_test_u8(new_pixel, cpu_extensions);
}
}
#[test]
fn divide_real_image() {
run_tests_with_real_image_u8::<U8x4, 4>(Oper::Div, [12452343, 12452343, 12452343, 8355840]);
#[test]
fn multiply_real_image() {
run_tests_with_real_image_u8::<P, 4>(Oper::Mul, [4177920, 4177920, 4177920, 8355840]);
}
#[test]
fn divide_real_image() {
run_tests_with_real_image_u8::<P, 4>(
Oper::Div,
[12452343, 12452343, 12452343, 8355840],
);
}
}
}
#[cfg(not(feature = "only_u8x4"))]
mod not_u8x4 {
use fast_image_resize::pixels::{U16x2, U16x4, U8x2};
mod u16_tests {
use super::*;
const fn new_u16x2(l: u16, a: u16) -> U16x2 {
U16x2::new([l, a])
struct TestCaseU16 {
pub color: u16,
pub alpha: u16,
pub expected_color: u16,
}
const fn new_u16x4(c: u16, a: u16) -> U16x4 {
U16x4::new([c, c, c, a])
const fn new_case_16(c: u16, a: u16, e: u16) -> TestCaseU16 {
TestCaseU16 {
color: c,
alpha: a,
expected_color: e,
}
}
fn get_mul_test_cases_u16<P>(create_pixel: fn(u16, u16) -> P) -> (Vec<P>, Vec<P>)
where
P: PixelTrait<Component = u16>,
{
let test_cases = [
new_case_16(0xffff, 0x8000, 0x8000),
new_case_16(0x8000, 0x8000, 0x4000),
new_case_16(0, 0x8000, 0),
new_case_16(0xffff, 0xffff, 0xffff),
new_case_16(0x8000, 0xffff, 0x8000),
new_case_16(0, 0xffff, 0),
new_case_16(0xffff, 0, 0),
new_case_16(0x8000, 0, 0),
new_case_16(0, 0, 0),
];
let mut scr_pixels = vec![];
let mut expected_pixels = vec![];
for case in test_cases {
scr_pixels.push(create_pixel(case.color, case.alpha));
expected_pixels.push(create_pixel(case.expected_color, case.alpha));
}
(scr_pixels, expected_pixels)
}
fn get_div_test_cases_u16<P>(create_pixel: fn(u16, u16) -> P) -> (Vec<P>, Vec<P>)
@@ -316,238 +336,157 @@ mod not_u8x4 {
}
#[cfg(test)]
mod u8x2 {
use super::*;
const fn new_u8x2(l: u8, a: u8) -> U8x2 {
U8x2::new([l, a])
}
#[test]
fn native_test() {
full_mul_div_alpha_test_u8(new_u8x2, CpuExtensions::None);
}
#[cfg(target_arch = "x86_64")]
#[test]
fn sse4_test() {
full_mul_div_alpha_test_u8(new_u8x2, CpuExtensions::Sse4_1);
}
#[cfg(target_arch = "x86_64")]
#[test]
fn avx2_test() {
full_mul_div_alpha_test_u8(new_u8x2, CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
#[test]
fn neon_test() {
full_mul_div_alpha_test_u8(new_u8x2, CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
#[test]
fn wasm32_test() {
full_mul_div_alpha_test_u8(new_u8x2, CpuExtensions::Simd128);
}
#[test]
fn multiply_real_image() {
run_tests_with_real_image_u8::<U8x2, 2>(Oper::Mul, [4177920, 8355840]);
}
#[test]
fn divide_real_image() {
run_tests_with_real_image_u8::<U8x2, 2>(Oper::Div, [12452343, 8355840]);
}
}
#[cfg(test)]
mod multiply_alpha_u16x2 {
mod u16x2 {
use fast_image_resize::pixels::U16x2;
use super::*;
const SRC_PIXELS: [U16x2; 9] = [
U16x2::new([0xffff, 0x8000]),
U16x2::new([0x8000, 0x8000]),
U16x2::new([0, 0x8000]),
U16x2::new([0xffff, 0xffff]),
U16x2::new([0x8000, 0xffff]),
U16x2::new([0, 0xffff]),
U16x2::new([0xffff, 0]),
U16x2::new([0x8000, 0]),
U16x2::new([0, 0]),
];
const RES_PIXELS: [U16x2; 9] = [
U16x2::new([0x8000, 0x8000]),
U16x2::new([0x4000, 0x8000]),
U16x2::new([0, 0x8000]),
U16x2::new([0xffff, 0xffff]),
U16x2::new([0x8000, 0xffff]),
U16x2::new([0, 0xffff]),
U16x2::new([0, 0]),
U16x2::new([0, 0]),
U16x2::new([0, 0]),
];
type P = U16x2;
#[cfg(target_arch = "x86_64")]
#[test]
fn avx2_test() {
mul_div_alpha_test(Oper::Mul, &SRC_PIXELS, &RES_PIXELS, CpuExtensions::Avx2);
}
#[cfg(target_arch = "x86_64")]
#[test]
fn sse4_test() {
mul_div_alpha_test(Oper::Mul, &SRC_PIXELS, &RES_PIXELS, CpuExtensions::Sse4_1);
}
#[cfg(target_arch = "aarch64")]
#[test]
fn neon_test() {
mul_div_alpha_test(Oper::Mul, &SRC_PIXELS, &RES_PIXELS, CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
#[test]
fn wasm32_test() {
mul_div_alpha_test(Oper::Mul, &SRC_PIXELS, &RES_PIXELS, CpuExtensions::Simd128);
const fn new_pixel(l: u16, a: u16) -> P {
P::new([l, a])
}
#[test]
fn native_test() {
mul_div_alpha_test(Oper::Mul, &SRC_PIXELS, &RES_PIXELS, CpuExtensions::None);
fn multiple_alpha_test() {
let (scr_pixels, expected_pixels) = get_mul_test_cases_u16(new_pixel);
for cpu_extensions in P::cpu_extensions() {
mul_div_alpha_test(Oper::Mul, &scr_pixels, &expected_pixels, cpu_extensions);
}
}
#[test]
fn divide_alpha_test() {
let (scr_pixels, expected_pixels) = get_div_test_cases_u16(new_pixel);
for cpu_extensions in P::cpu_extensions() {
mul_div_alpha_test(Oper::Div, &scr_pixels, &expected_pixels, cpu_extensions);
}
}
}
#[cfg(test)]
mod multiply_alpha_u16x4 {
mod u16x4 {
use fast_image_resize::pixels::U16x4;
use super::*;
const SRC_PIXELS: [U16x4; 3] = [
U16x4::new([0xffff, 0x8000, 0, 0x8000]),
U16x4::new([0xffff, 0x8000, 0, 0xffff]),
U16x4::new([0xffff, 0x8000, 0, 0]),
];
const RES_PIXELS: [U16x4; 3] = [
U16x4::new([0x8000, 0x4000, 0, 0x8000]),
U16x4::new([0xffff, 0x8000, 0, 0xffff]),
U16x4::new([0, 0, 0, 0]),
];
type P = U16x4;
#[cfg(target_arch = "x86_64")]
#[test]
fn avx2_test() {
mul_div_alpha_test(Oper::Mul, &SRC_PIXELS, &RES_PIXELS, CpuExtensions::Avx2);
}
#[cfg(target_arch = "x86_64")]
#[test]
fn sse4_test() {
mul_div_alpha_test(Oper::Mul, &SRC_PIXELS, &RES_PIXELS, CpuExtensions::Sse4_1);
}
#[cfg(target_arch = "aarch64")]
#[test]
fn neon_test() {
mul_div_alpha_test(Oper::Mul, &SRC_PIXELS, &RES_PIXELS, CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
#[test]
fn wasm32_test() {
mul_div_alpha_test(Oper::Mul, &SRC_PIXELS, &RES_PIXELS, CpuExtensions::Simd128);
const fn new_pixel(c: u16, a: u16) -> P {
P::new([c, c, c, a])
}
#[test]
fn native_test() {
mul_div_alpha_test(Oper::Mul, &SRC_PIXELS, &RES_PIXELS, CpuExtensions::None);
}
}
#[cfg(test)]
mod divide_alpha_u16x2 {
use super::*;
const OPER: Oper = Oper::Div;
#[test]
fn native_test() {
let (scr_pixels, expected_pixels) = get_div_test_cases_u16(new_u16x2);
mul_div_alpha_test(OPER, &scr_pixels, &expected_pixels, CpuExtensions::None);
fn multiple_alpha_test() {
let (scr_pixels, expected_pixels) = get_mul_test_cases_u16(new_pixel);
for cpu_extensions in P::cpu_extensions() {
mul_div_alpha_test(Oper::Mul, &scr_pixels, &expected_pixels, cpu_extensions);
}
}
#[cfg(target_arch = "x86_64")]
#[test]
fn sse4_test() {
let (scr_pixels, expected_pixels) = get_div_test_cases_u16(new_u16x2);
mul_div_alpha_test(OPER, &scr_pixels, &expected_pixels, CpuExtensions::Sse4_1);
fn divide_alpha_test() {
let (scr_pixels, expected_pixels) = get_div_test_cases_u16(new_pixel);
for cpu_extensions in P::cpu_extensions() {
mul_div_alpha_test(Oper::Div, &scr_pixels, &expected_pixels, cpu_extensions);
}
}
}
}
#[cfg(not(feature = "only_u8x4"))]
mod f32_tests {
use super::*;
#[cfg(target_arch = "x86_64")]
#[test]
fn avx2_test() {
let (scr_pixels, expected_pixels) = get_div_test_cases_u16(new_u16x2);
mul_div_alpha_test(OPER, &scr_pixels, &expected_pixels, CpuExtensions::Avx2);
struct TestCaseF32 {
pub color: f32,
pub alpha: f32,
pub expected_color: f32,
}
const fn new_case_f32(c: f32, a: f32, e: f32) -> TestCaseF32 {
TestCaseF32 {
color: c,
alpha: a,
expected_color: e,
}
}
#[cfg(target_arch = "aarch64")]
#[test]
fn neon_test() {
let (scr_pixels, expected_pixels) = get_div_test_cases_u16(new_u16x2);
mul_div_alpha_test(OPER, &scr_pixels, &expected_pixels, CpuExtensions::Neon);
fn get_mul_test_cases_f32<P>(create_pixel: fn(f32, f32) -> P) -> (Vec<P>, Vec<P>)
where
P: PixelTrait<Component = f32>,
{
let test_cases = [
new_case_f32(1., 0.5, 0.5),
new_case_f32(0.5, 0.5, 0.25),
new_case_f32(0., 0.5, 0.),
new_case_f32(1., 1., 1.),
new_case_f32(0.5, 1., 0.5),
new_case_f32(0., 1., 0.),
new_case_f32(1., 0., 0.),
new_case_f32(0.5, 0., 0.),
new_case_f32(0., 0., 0.),
];
let mut scr_pixels = vec![];
let mut expected_pixels = vec![];
for case in test_cases {
scr_pixels.push(create_pixel(case.color, case.alpha));
expected_pixels.push(create_pixel(case.expected_color, case.alpha));
}
(scr_pixels, expected_pixels)
}
#[cfg(target_arch = "wasm32")]
#[test]
fn wasm32_test() {
let (scr_pixels, expected_pixels) = get_div_test_cases_u16(new_u16x2);
mul_div_alpha_test(OPER, &scr_pixels, &expected_pixels, CpuExtensions::Simd128);
fn get_div_test_cases_f32<P>(create_pixel: fn(f32, f32) -> P) -> (Vec<P>, Vec<P>)
where
P: PixelTrait<Component = f32>,
{
let test_cases = [
new_case_f32(0.5, 0.5, 1.),
new_case_f32(0.25, 0.5, 0.5),
new_case_f32(0., 0.5, 0.),
new_case_f32(1., 1., 1.),
new_case_f32(0.5, 1., 0.5),
new_case_f32(0.00001, 0.00002, 0.00001 / 0.00002),
new_case_f32(0., 1., 0.),
new_case_f32(1., 0., 0.),
new_case_f32(0.5, 0., 0.),
new_case_f32(0., 0., 0.),
];
let mut scr_pixels = vec![];
let mut expected_pixels = vec![];
for case in test_cases {
scr_pixels.push(create_pixel(case.color, case.alpha));
expected_pixels.push(create_pixel(case.expected_color, case.alpha));
}
(scr_pixels, expected_pixels)
}
#[cfg(test)]
mod divide_alpha_u16x4 {
mod f32x2 {
use fast_image_resize::pixels::F32x2;
use super::*;
const OPER: Oper = Oper::Div;
#[test]
fn native_test() {
let (scr_pixels, expected_pixels) = get_div_test_cases_u16(new_u16x4);
mul_div_alpha_test(OPER, &scr_pixels, &expected_pixels, CpuExtensions::None);
}
#[cfg(target_arch = "x86_64")]
#[test]
fn sse4_test() {
let (scr_pixels, expected_pixels) = get_div_test_cases_u16(new_u16x4);
mul_div_alpha_test(OPER, &scr_pixels, &expected_pixels, CpuExtensions::Sse4_1);
}
type P = F32x2;
#[cfg(target_arch = "x86_64")]
#[test]
fn avx2_test() {
let (scr_pixels, expected_pixels) = get_div_test_cases_u16(new_u16x4);
mul_div_alpha_test(OPER, &scr_pixels, &expected_pixels, CpuExtensions::Avx2);
const fn new_pixel(c: f32, a: f32) -> P {
P::new([c, a])
}
#[cfg(target_arch = "aarch64")]
#[test]
fn neon_test() {
let (scr_pixels, expected_pixels) = get_div_test_cases_u16(new_u16x4);
mul_div_alpha_test(OPER, &scr_pixels, &expected_pixels, CpuExtensions::Neon);
fn multiple_alpha_test() {
let (scr_pixels, expected_pixels) = get_mul_test_cases_f32(new_pixel);
for cpu_extensions in P::cpu_extensions() {
mul_div_alpha_test(Oper::Mul, &scr_pixels, &expected_pixels, cpu_extensions);
}
}
#[cfg(target_arch = "wasm32")]
#[test]
fn wasm32_test() {
let (scr_pixels, expected_pixels) = get_div_test_cases_u16(new_u16x4);
mul_div_alpha_test(OPER, &scr_pixels, &expected_pixels, CpuExtensions::Simd128);
fn divide_alpha_test() {
let (scr_pixels, expected_pixels) = get_div_test_cases_f32(new_pixel);
for cpu_extensions in P::cpu_extensions() {
mul_div_alpha_test(Oper::Div, &scr_pixels, &expected_pixels, cpu_extensions);
}
}
}
}
+112 -255
View File
@@ -395,21 +395,7 @@ mod not_u8x4 {
type P = U8;
P::downscale_test(ResizeAlg::Nearest, CpuExtensions::None, [2920348]);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
for cpu_extensions in P::cpu_extensions() {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
@@ -423,21 +409,7 @@ mod not_u8x4 {
type P = U8;
P::upscale_test(ResizeAlg::Nearest, CpuExtensions::None, [1148754010]);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
for cpu_extensions in P::cpu_extensions() {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
@@ -451,21 +423,7 @@ mod not_u8x4 {
type P = U8x2;
P::downscale_test(ResizeAlg::Nearest, CpuExtensions::None, [2920348, 6121802]);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
for cpu_extensions in P::cpu_extensions() {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
@@ -483,21 +441,7 @@ mod not_u8x4 {
[1146218632, 2364895380],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
for cpu_extensions in P::cpu_extensions() {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
@@ -515,21 +459,7 @@ mod not_u8x4 {
[2937940, 2945380, 2882679],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
for cpu_extensions in P::cpu_extensions() {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
@@ -547,21 +477,7 @@ mod not_u8x4 {
[1156008260, 1158417906, 1135087540],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
for cpu_extensions in P::cpu_extensions() {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
@@ -575,21 +491,7 @@ mod not_u8x4 {
type P = U16;
P::downscale_test(ResizeAlg::Nearest, CpuExtensions::None, [750529436]);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
for cpu_extensions in P::cpu_extensions() {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
@@ -603,21 +505,7 @@ mod not_u8x4 {
type P = U16;
P::upscale_test(ResizeAlg::Nearest, CpuExtensions::None, [295229780570]);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
for cpu_extensions in P::cpu_extensions() {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
@@ -635,21 +523,7 @@ mod not_u8x4 {
[750529436, 1573303114],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
for cpu_extensions in P::cpu_extensions() {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
@@ -667,21 +541,7 @@ mod not_u8x4 {
[294578188424, 607778112660],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
for cpu_extensions in P::cpu_extensions() {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
@@ -699,21 +559,7 @@ mod not_u8x4 {
[755050580, 756962660, 740848503],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
for cpu_extensions in P::cpu_extensions() {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
@@ -731,21 +577,7 @@ mod not_u8x4 {
[297094122820, 297713401842, 291717497780],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
for cpu_extensions in P::cpu_extensions() {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
@@ -763,21 +595,7 @@ mod not_u8x4 {
[755050580, 756962660, 740848503, 1573303114],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
for cpu_extensions in P::cpu_extensions() {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
@@ -795,21 +613,7 @@ mod not_u8x4 {
[296859917949, 296229709231, 288684470903, 607778112660],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
for cpu_extensions in P::cpu_extensions() {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
@@ -818,6 +622,101 @@ mod not_u8x4 {
}
}
// I32
#[test]
fn downscale_i32() {
type P = I32;
P::downscale_test(ResizeAlg::Nearest, CpuExtensions::None, [24593724281554]);
for cpu_extensions in P::cpu_extensions() {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[36889044005199],
);
}
}
#[test]
fn upscale_i32() {
type P = I32;
P::upscale_test(ResizeAlg::Nearest, CpuExtensions::None, [9674237252903955]);
for cpu_extensions in P::cpu_extensions() {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[11090415545881916],
);
}
}
// F32
#[test]
fn downscale_f32() {
type P = F32;
P::downscale_test(ResizeAlg::Nearest, CpuExtensions::None, [28891951209032]);
for cpu_extensions in P::cpu_extensions() {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[41687319249443],
);
}
}
#[test]
fn upscale_f32() {
type P = F32;
P::upscale_test(ResizeAlg::Nearest, CpuExtensions::None, [11165019414549868]);
for cpu_extensions in P::cpu_extensions() {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[12506894762090128],
);
}
}
// F32x2
#[test]
fn downscale_f32x2() {
type P = F32x2;
P::downscale_test(
ResizeAlg::Nearest,
CpuExtensions::None,
[28891951209032, 26023210300788],
);
for cpu_extensions in P::cpu_extensions() {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[41687319249443, 29873206892121],
);
}
}
#[test]
fn upscale_f32x2() {
type P = F32x2;
P::upscale_test(
ResizeAlg::Nearest,
CpuExtensions::None,
[9941292360529429, 10060767588318486],
);
for cpu_extensions in P::cpu_extensions() {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[10426687457795354, 10465695788378423],
);
}
}
#[test]
fn fractional_cropping() {
let mut src_buf = [0, 0, 0, 0, 255, 0, 0, 0, 0];
@@ -899,21 +798,7 @@ mod u8x4 {
[2937940, 2945380, 2882679, 6121802],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
for cpu_extensions in P::cpu_extensions() {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Gaussian),
cpu_extensions,
@@ -942,21 +827,7 @@ mod u8x4 {
[1155096957, 1152644783, 1123285879, 2364895380],
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
for cpu_extensions in cpu_extensions_vec {
for cpu_extensions in P::cpu_extensions() {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
@@ -1038,22 +909,8 @@ mod u8x4 {
let mut resizer = Resizer::new();
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
cpu_extensions_vec.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Simd128);
}
let mut results = vec![];
for cpu_extensions in cpu_extensions_vec {
for cpu_extensions in U8x4::cpu_extensions() {
unsafe {
resizer.set_cpu_extensions(cpu_extensions);
}