mirror of
https://github.com/Cykooz/fast_image_resize.git
synced 2026-10-07 17:01:09 +00:00
Added support for multi-thread image processing with the help of rayon crate.
This commit is contained in:
@@ -5,18 +5,19 @@ on:
|
||||
branches: [ "main" ]
|
||||
pull_request:
|
||||
branches: [ "main" ]
|
||||
workflow_dispatch: {}
|
||||
workflow_dispatch: { }
|
||||
|
||||
env:
|
||||
CARGO_TERM_COLOR: always
|
||||
DONT_SAVE_RESULT: 1
|
||||
RAYON_NUM_THREADS: 4
|
||||
|
||||
jobs:
|
||||
run_tests:
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
os: [ubuntu-latest, macos-latest, windows-latest]
|
||||
os: [ ubuntu-latest, macos-latest, windows-latest ]
|
||||
|
||||
name: Test `cargo check/test` on ${{ matrix.os }}
|
||||
runs-on: ${{ matrix.os }}
|
||||
@@ -28,7 +29,12 @@ jobs:
|
||||
with:
|
||||
cache-on-failure: "true"
|
||||
|
||||
- name: Run tests
|
||||
- name: Run single-thread tests
|
||||
run: |
|
||||
cargo check
|
||||
cargo test
|
||||
|
||||
- name: Run multi-thread tests
|
||||
run: |
|
||||
cargo check --features rayon
|
||||
cargo test --features rayon
|
||||
|
||||
@@ -1,3 +1,24 @@
|
||||
## [Unreleased] - ReleaseDate
|
||||
|
||||
### Added
|
||||
|
||||
- Added support for multi-thread image processing with the help of `rayon` crate.
|
||||
You should enable `rayon` feature to turn on this behavior.
|
||||
- Added methods to split image in different directions:
|
||||
- `ImageView::split_by_height()`
|
||||
- `ImageView::split_by_width()`
|
||||
- `ImageViewMut::split_by_height_mut()`
|
||||
- `ImageViewMut::split_by_width_mut()`
|
||||
|
||||
These methods have default implementation and are used for multi-thread
|
||||
image processing.
|
||||
|
||||
## Changed
|
||||
|
||||
- **BREAKING**: Added supertraits `Send`, `Sync` and `Sized` to the `ImageView` trait.
|
||||
- Optimized convolution algorythm by deleting zero coefficients from start and
|
||||
end of bounds.
|
||||
|
||||
## [4.2.1] - 2024-07-24
|
||||
|
||||
### Fixed
|
||||
|
||||
Generated
+122
-97
@@ -8,6 +8,12 @@ version = "1.0.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f26201604c87b1e01bd3d98f8d5d9a8fcbb815e8cedb41ffccbeb4bf593a35fe"
|
||||
|
||||
[[package]]
|
||||
name = "adler2"
|
||||
version = "2.0.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "512761e0bb2578dd7380c6baaa0f4ce03e84f95e960231d1dec8bf4d7d6e2627"
|
||||
|
||||
[[package]]
|
||||
name = "aho-corasick"
|
||||
version = "1.1.3"
|
||||
@@ -95,9 +101,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "anyhow"
|
||||
version = "1.0.86"
|
||||
version = "1.0.89"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b3d1d046238990b9cf5bcde22a3fb3584ee5cf65fb2765f454ed428c7a0063da"
|
||||
checksum = "86fdf8605db99b54d3cd748a44c6d04df638eb5dafb219b135d0149bd0db01f6"
|
||||
|
||||
[[package]]
|
||||
name = "arbitrary"
|
||||
@@ -118,9 +124,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "arrayvec"
|
||||
version = "0.7.4"
|
||||
version = "0.7.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "96d30a06541fbafbc7f82ed10c06164cfbd2c401138f6addd8404629c4b16711"
|
||||
checksum = "7c02d123df017efcdfbd739ef81735b36c5ba83ec3c59c80a9d7ecc718f92e50"
|
||||
|
||||
[[package]]
|
||||
name = "autocfg"
|
||||
@@ -171,9 +177,9 @@ checksum = "b048fb63fd8b5923fc5aa7b340d8e156aec7ec02f0c78fa8a6ddc2613f6f71de"
|
||||
|
||||
[[package]]
|
||||
name = "bitstream-io"
|
||||
version = "2.5.0"
|
||||
version = "2.5.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3dcde5f311c85b8ca30c2e4198d4326bc342c76541590106f5fa4a50946ea499"
|
||||
checksum = "b81e1519b0d82120d2fd469d5bfb2919a9361c48b02d82d04befc1cdd2002452"
|
||||
|
||||
[[package]]
|
||||
name = "block-buffer"
|
||||
@@ -208,9 +214,9 @@ checksum = "79296716171880943b8470b5f8d03aa55eb2e645a4874bdbb28adb49162e012c"
|
||||
|
||||
[[package]]
|
||||
name = "bytemuck"
|
||||
version = "1.16.3"
|
||||
version = "1.18.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "102087e286b4677862ea56cf8fc58bb2cdfa8725c40ffb80fe3a008eb7f2fc83"
|
||||
checksum = "94bbb0ad554ad961ddc5da507a12a29b14e4ae5bda06b19f575a3e6079d2e2ae"
|
||||
|
||||
[[package]]
|
||||
name = "byteorder"
|
||||
@@ -232,12 +238,13 @@ checksum = "37b2a672a2cb129a2e41c10b1224bb368f9f37a2b16b612598138befd7b37eb5"
|
||||
|
||||
[[package]]
|
||||
name = "cc"
|
||||
version = "1.1.8"
|
||||
version = "1.1.21"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "504bdec147f2cc13c8b57ed9401fd8a147cc66b67ad5cb241394244f2c947549"
|
||||
checksum = "07b1695e2c7e8fc85310cde85aeaab7e3097f593c91d209d3f9df76c928100f0"
|
||||
dependencies = [
|
||||
"jobserver",
|
||||
"libc",
|
||||
"shlex",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -325,9 +332,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "clap"
|
||||
version = "4.5.13"
|
||||
version = "4.5.18"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0fbb260a053428790f3de475e304ff84cdbc4face759ea7a3e64c1edd938a7fc"
|
||||
checksum = "b0956a43b323ac1afaffc053ed5c4b7c1f1800bacd1683c353aabbb752515dd3"
|
||||
dependencies = [
|
||||
"clap_builder",
|
||||
"clap_derive",
|
||||
@@ -335,9 +342,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "clap-verbosity-flag"
|
||||
version = "2.2.1"
|
||||
version = "2.2.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "63d19864d6b68464c59f7162c9914a0b569ddc2926b4a2d71afe62a9738eff53"
|
||||
checksum = "e099138e1807662ff75e2cebe4ae2287add879245574489f9b1588eb5e5564ed"
|
||||
dependencies = [
|
||||
"clap",
|
||||
"log",
|
||||
@@ -345,9 +352,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "clap_builder"
|
||||
version = "4.5.13"
|
||||
version = "4.5.18"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "64b17d7ea74e9f833c7dbf2cbe4fb12ff26783eda4782a8975b72f895c9b4d99"
|
||||
checksum = "4d72166dd41634086d5803a47eb71ae740e61d84709c36f3c34110173db3961b"
|
||||
dependencies = [
|
||||
"anstream",
|
||||
"anstyle",
|
||||
@@ -357,9 +364,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "clap_derive"
|
||||
version = "4.5.13"
|
||||
version = "4.5.18"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "501d359d5f3dcaf6ecdeee48833ae73ec6e42723a1e52419c79abf9507eec0a0"
|
||||
checksum = "4ac6a0c7b1a9e9a5186361f67dfa1b88213572f427fb9ab038efb2bd8c582dab"
|
||||
dependencies = [
|
||||
"heck",
|
||||
"proc-macro2",
|
||||
@@ -387,15 +394,15 @@ checksum = "d3fd119d74b830634cea2a0f58bbd0d54540518a14397557951e79340abc28c0"
|
||||
|
||||
[[package]]
|
||||
name = "core-foundation-sys"
|
||||
version = "0.8.6"
|
||||
version = "0.8.7"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "06ea2b9bc92be3c2baa9334a323ebca2d6f074ff852cd1d7b11064035cd3868f"
|
||||
checksum = "773648b94d0e5d620f64f280777445740e61fe701025087ec8b57f45c791888b"
|
||||
|
||||
[[package]]
|
||||
name = "cpufeatures"
|
||||
version = "0.2.12"
|
||||
version = "0.2.14"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "53fe5e26ff1b7aef8bca9c6080520cfb8d9333c7568e1829cef191a9723e5504"
|
||||
checksum = "608697df725056feaccfa42cffdaeeec3fccc4ffc38358ecd19b243e716a78e0"
|
||||
dependencies = [
|
||||
"libc",
|
||||
]
|
||||
@@ -554,7 +561,7 @@ dependencies = [
|
||||
"flume",
|
||||
"half",
|
||||
"lebe",
|
||||
"miniz_oxide",
|
||||
"miniz_oxide 0.7.4",
|
||||
"rayon-core",
|
||||
"smallvec",
|
||||
"zune-inflate",
|
||||
@@ -575,6 +582,7 @@ dependencies = [
|
||||
"nix",
|
||||
"num-traits",
|
||||
"png",
|
||||
"rayon",
|
||||
"resize",
|
||||
"rgb",
|
||||
"serde",
|
||||
@@ -587,21 +595,21 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "fdeflate"
|
||||
version = "0.3.4"
|
||||
version = "0.3.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "4f9bfee30e4dedf0ab8b422f03af778d9612b63f502710fc500a334ebe2de645"
|
||||
checksum = "d8090f921a24b04994d9929e204f50b498a33ea6ba559ffaa05e04f7ee7fb5ab"
|
||||
dependencies = [
|
||||
"simd-adler32",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "flate2"
|
||||
version = "1.0.31"
|
||||
version = "1.0.34"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "7f211bbe8e69bbd0cfdea405084f128ae8b4aaa6b0b522fc8f2b009084797920"
|
||||
checksum = "a1b589b4dc103969ad3cf85c950899926ec64300a1a46d76c03a6072957036f0"
|
||||
dependencies = [
|
||||
"crc32fast",
|
||||
"miniz_oxide",
|
||||
"miniz_oxide 0.8.0",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -646,9 +654,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "globset"
|
||||
version = "0.4.14"
|
||||
version = "0.4.15"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "57da3b9b5b85bd66f31093f8c408b90a74431672542466497dcbdfdc02034be1"
|
||||
checksum = "15f1ce686646e7f1e19bf7d5533fe443a45dbfb990e00629110797578b42fb19"
|
||||
dependencies = [
|
||||
"aho-corasick",
|
||||
"bstr",
|
||||
@@ -692,9 +700,9 @@ checksum = "2304e00983f87ffb38b55b444b5e3b60a884b5d30c0fca7d82fe33449bbe55ea"
|
||||
|
||||
[[package]]
|
||||
name = "hermit-abi"
|
||||
version = "0.3.9"
|
||||
version = "0.4.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d231dfb89cfffdbc30e7fc41579ed6066ad03abda9e567ccafae602b97ec5024"
|
||||
checksum = "fbf6a919d6cf397374f7dfeeea91d974c7c0a7221d0d0f4f20d859d329e53fcc"
|
||||
|
||||
[[package]]
|
||||
name = "humansize"
|
||||
@@ -713,9 +721,9 @@ checksum = "9a3a5bfb195931eeb336b2a7b4d761daec841b97f947d34394601737a7bba5e4"
|
||||
|
||||
[[package]]
|
||||
name = "iana-time-zone"
|
||||
version = "0.1.60"
|
||||
version = "0.1.61"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e7ffbb5a1b541ea2561f8c41c087286cc091e21e556a4f09a8f6cbf17b69b141"
|
||||
checksum = "235e081f3925a06703c2d0117ea8b91f042756fd6e7a6e5d901e8ca1a996b220"
|
||||
dependencies = [
|
||||
"android_system_properties",
|
||||
"core-foundation-sys",
|
||||
@@ -736,9 +744,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "ignore"
|
||||
version = "0.4.22"
|
||||
version = "0.4.23"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b46810df39e66e925525d6e38ce1e7f6e1d208f72dc39757880fcb66e2c58af1"
|
||||
checksum = "6d89fd380afde86567dfba715db065673989d6253f42b88179abd3eae47bda4b"
|
||||
dependencies = [
|
||||
"crossbeam-deque",
|
||||
"globset",
|
||||
@@ -791,9 +799,9 @@ checksum = "44feda355f4159a7c757171a77de25daf6411e217b4cabd03bd6650690468126"
|
||||
|
||||
[[package]]
|
||||
name = "indexmap"
|
||||
version = "2.3.0"
|
||||
version = "2.5.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "de3fc2e30ba82dd1b3911c8de1ffc143c74a914a14e99514d7637e3099df5ea0"
|
||||
checksum = "68b900aa2f7301e21c36462b170ee99994de34dff39a4a6a528e80e7376d07e5"
|
||||
dependencies = [
|
||||
"equivalent",
|
||||
"hashbrown",
|
||||
@@ -812,9 +820,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "is-terminal"
|
||||
version = "0.4.12"
|
||||
version = "0.4.13"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f23ff5ef2b80d608d61efee834934d862cd92461afc0560dedf493e4c033738b"
|
||||
checksum = "261f68e344040fbd0edea105bef17c66edf46f984ddb1115b775ce31be948f4b"
|
||||
dependencies = [
|
||||
"hermit-abi",
|
||||
"libc",
|
||||
@@ -877,9 +885,9 @@ checksum = "f5d4a7da358eff58addd2877a45865158f0d78c911d43a5784ceb7bbf52833b0"
|
||||
|
||||
[[package]]
|
||||
name = "js-sys"
|
||||
version = "0.3.69"
|
||||
version = "0.3.70"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "29c15563dc2726973df627357ce0c9ddddbea194836909d655df6a75d2cf296d"
|
||||
checksum = "1868808506b929d7b0cfa8f75951347aa71bb21144b7791bae35d9bccfcfe37a"
|
||||
dependencies = [
|
||||
"wasm-bindgen",
|
||||
]
|
||||
@@ -898,9 +906,9 @@ checksum = "03087c2bad5e1034e8cace5926dec053fb3790248370865f5117a7d0213354c8"
|
||||
|
||||
[[package]]
|
||||
name = "libc"
|
||||
version = "0.2.155"
|
||||
version = "0.2.159"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "97b3888a4aecf77e811145cadf6eef5901f4782c53886191b2f693f24761847c"
|
||||
checksum = "561d97a539a36e26a9a5fad1ea11a3039a67714694aaa379433e580854bc3dc5"
|
||||
|
||||
[[package]]
|
||||
name = "libfuzzer-sys"
|
||||
@@ -988,6 +996,15 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b8a240ddb74feaf34a79a7add65a741f3167852fba007066dcac1ca548d89c08"
|
||||
dependencies = [
|
||||
"adler",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "miniz_oxide"
|
||||
version = "0.8.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e2d80299ef12ff69b16a84bb182e3b9df68b5a91574d3d4fa6e41b65deec4df1"
|
||||
dependencies = [
|
||||
"adler2",
|
||||
"simd-adler32",
|
||||
]
|
||||
|
||||
@@ -1110,9 +1127,9 @@ checksum = "e3148f5046208a5d56bcfc03053e3ca6334e51da8dfb19b6cdc8b306fae3283e"
|
||||
|
||||
[[package]]
|
||||
name = "pest"
|
||||
version = "2.7.11"
|
||||
version = "2.7.13"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "cd53dff83f26735fdc1ca837098ccf133605d794cdae66acfc2bfac3ec809d95"
|
||||
checksum = "fdbef9d1d47087a895abd220ed25eb4ad973a5e26f6a4367b038c25e28dfc2d9"
|
||||
dependencies = [
|
||||
"memchr",
|
||||
"thiserror",
|
||||
@@ -1121,9 +1138,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pest_derive"
|
||||
version = "2.7.11"
|
||||
version = "2.7.13"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "2a548d2beca6773b1c244554d36fcf8548a8a58e74156968211567250e48e49a"
|
||||
checksum = "4d3a6e3394ec80feb3b6393c725571754c6188490265c61aaf260810d6b95aa0"
|
||||
dependencies = [
|
||||
"pest",
|
||||
"pest_generator",
|
||||
@@ -1131,9 +1148,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pest_generator"
|
||||
version = "2.7.11"
|
||||
version = "2.7.13"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3c93a82e8d145725dcbaf44e5ea887c8a869efdcc28706df2d08c69e17077183"
|
||||
checksum = "94429506bde1ca69d1b5601962c73f4172ab4726571a59ea95931218cb0e930e"
|
||||
dependencies = [
|
||||
"pest",
|
||||
"pest_meta",
|
||||
@@ -1144,9 +1161,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pest_meta"
|
||||
version = "2.7.11"
|
||||
version = "2.7.13"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "a941429fea7e08bedec25e4f6785b6ffaacc6b755da98df5ef3e7dcf4a124c4f"
|
||||
checksum = "ac8a071862e93690b6e34e9a5fb8e33ff3734473ac0245b27232222c4906a33f"
|
||||
dependencies = [
|
||||
"once_cell",
|
||||
"pest",
|
||||
@@ -1193,21 +1210,21 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pkg-config"
|
||||
version = "0.3.30"
|
||||
version = "0.3.31"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d231b230927b5e4ad203db57bbcbee2802f6bce620b1e4a9024a07d94e2907ec"
|
||||
checksum = "953ec861398dccce10c670dfeaf3ec4911ca479e9c02154b3a215178c5f566f2"
|
||||
|
||||
[[package]]
|
||||
name = "png"
|
||||
version = "0.17.13"
|
||||
version = "0.17.14"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "06e4b0d3d1312775e782c86c91a111aa1f910cbb65e1337f9975b5f9a554b5e1"
|
||||
checksum = "52f9d46a34a05a6a57566bc2bfae066ef07585a6e3fa30fbbdff5936380623f0"
|
||||
dependencies = [
|
||||
"bitflags 1.3.2",
|
||||
"crc32fast",
|
||||
"fdeflate",
|
||||
"flate2",
|
||||
"miniz_oxide",
|
||||
"miniz_oxide 0.8.0",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -1264,9 +1281,9 @@ checksum = "a993555f31e5a609f617c12db6250dedcac1b0a85076912c436e6fc9b2c8e6a3"
|
||||
|
||||
[[package]]
|
||||
name = "quote"
|
||||
version = "1.0.36"
|
||||
version = "1.0.37"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0fa76aaf39101c457836aec0ce2316dbdc3ab723cdda1c6bd4e6ad4208acaca7"
|
||||
checksum = "b5b9d34b8991d19d98081b46eacdd8eb58c6f2b201139f7c5f643cc155a633af"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
]
|
||||
@@ -1338,9 +1355,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "ravif"
|
||||
version = "0.11.9"
|
||||
version = "0.11.10"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "5797d09f9bd33604689e87e8380df4951d4912f01b63f71205e2abd4ae25e6b6"
|
||||
checksum = "a8f0bfd976333248de2078d350bfdf182ff96e168a24d23d2436cef320dd4bdd"
|
||||
dependencies = [
|
||||
"avif-serialize",
|
||||
"imgref",
|
||||
@@ -1401,10 +1418,11 @@ checksum = "7a66a03ae7c801facd77a29370b4faec201768915ac14a721ba36f20bc9c209b"
|
||||
|
||||
[[package]]
|
||||
name = "resize"
|
||||
version = "0.8.5"
|
||||
version = "0.8.7"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "a84f5827feaf48508b264176bd88e0479695af183738cf0305fadf956c796412"
|
||||
checksum = "6eec4ee5277bcbebeac5c955c3b49811f2c033520692359f5438f8bc4113574d"
|
||||
dependencies = [
|
||||
"rayon",
|
||||
"rgb",
|
||||
]
|
||||
|
||||
@@ -1424,9 +1442,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rgb"
|
||||
version = "0.8.48"
|
||||
version = "0.8.50"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0f86ae463694029097b846d8f99fd5536740602ae00022c0c50c5600720b2f71"
|
||||
checksum = "57397d16646700483b67d2dd6511d79318f9d057fdbd21a4066aeac8b41d310a"
|
||||
dependencies = [
|
||||
"bytemuck",
|
||||
]
|
||||
@@ -1454,18 +1472,18 @@ checksum = "94143f37725109f92c262ed2cf5e59bce7498c01bcc1502d7b9afe439a4e9f49"
|
||||
|
||||
[[package]]
|
||||
name = "serde"
|
||||
version = "1.0.204"
|
||||
version = "1.0.210"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "bc76f558e0cbb2a839d37354c575f1dc3fdc6546b5be373ba43d95f231bf7c12"
|
||||
checksum = "c8e3592472072e6e22e0a54d5904d9febf8508f65fb8552499a1abc7d1078c3a"
|
||||
dependencies = [
|
||||
"serde_derive",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "serde_derive"
|
||||
version = "1.0.204"
|
||||
version = "1.0.210"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e0cd7e117be63d3c3678776753929474f3b04a43a080c744d6b0ae2a8c28e222"
|
||||
checksum = "243902eda00fad750862fc144cea25caca5e20d615af0a81bee94ca738f1df1f"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
@@ -1474,9 +1492,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "serde_json"
|
||||
version = "1.0.122"
|
||||
version = "1.0.128"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "784b6203951c57ff748476b126ccb5e8e2959a5c19e5c617ab1956be3dbc68da"
|
||||
checksum = "6ff5456707a1de34e7e37f2a6fd3d3f808c318259cbd01ab6377795054b483d8"
|
||||
dependencies = [
|
||||
"itoa",
|
||||
"memchr",
|
||||
@@ -1486,9 +1504,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "serde_spanned"
|
||||
version = "0.6.7"
|
||||
version = "0.6.8"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "eb5b1b31579f3811bf615c144393417496f152e12ac8b7663bf664f4a815306d"
|
||||
checksum = "87607cb1398ed59d48732e575a4c28a7a8ebf2454b964fe3f224f2afc07909e1"
|
||||
dependencies = [
|
||||
"serde",
|
||||
]
|
||||
@@ -1504,6 +1522,12 @@ dependencies = [
|
||||
"digest",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "shlex"
|
||||
version = "1.3.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0fda2ff0d084019ba4d7c6f371c95d8fd75ce3524c3cb8fb653a3023f6323e64"
|
||||
|
||||
[[package]]
|
||||
name = "simd-adler32"
|
||||
version = "0.3.7"
|
||||
@@ -1527,9 +1551,9 @@ checksum = "38b58827f4464d87d377d175e90bf58eb00fd8716ff0a62f80356b5e61555d0d"
|
||||
|
||||
[[package]]
|
||||
name = "slug"
|
||||
version = "0.1.5"
|
||||
version = "0.1.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3bd94acec9c8da640005f8e135a39fc0372e74535e6b368b7a04b875f784c8c4"
|
||||
checksum = "882a80f72ee45de3cc9a5afeb2da0331d58df69e4e7d8eeb5d3c7784ae67e724"
|
||||
dependencies = [
|
||||
"deunicode",
|
||||
"wasm-bindgen",
|
||||
@@ -1558,9 +1582,9 @@ checksum = "7da8b5736845d9f2fcb837ea5d9e2628564b3b043a70948a3f0b778838c5fb4f"
|
||||
|
||||
[[package]]
|
||||
name = "syn"
|
||||
version = "2.0.72"
|
||||
version = "2.0.77"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "dc4b9b9bf2add8093d3f2c0204471e951b2285580335de42f9d2534f3ae7a8af"
|
||||
checksum = "9f35bcdf61fd8e7be6caf75f429fdca8beb3ed76584befb503b1569faee373ed"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
@@ -1618,18 +1642,18 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "thiserror"
|
||||
version = "1.0.63"
|
||||
version = "1.0.64"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "c0342370b38b6a11b6cc11d6a805569958d54cfa061a29969c3b5ce2ea405724"
|
||||
checksum = "d50af8abc119fb8bb6dbabcfa89656f46f84aa0ac7688088608076ad2b459a84"
|
||||
dependencies = [
|
||||
"thiserror-impl",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "thiserror-impl"
|
||||
version = "1.0.63"
|
||||
version = "1.0.64"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "a4558b58466b9ad7ca0f102865eccc95938dca1a74a856f2b57b6629050da261"
|
||||
checksum = "08904e7672f5eb876eaaf87e0ce17857500934f4981c4a0ab2b4aa98baac7fc3"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
@@ -1680,9 +1704,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "toml_edit"
|
||||
version = "0.22.20"
|
||||
version = "0.22.22"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "583c44c02ad26b0c3f3066fe629275e50627026c51ac2e595cca4c230ce1ce1d"
|
||||
checksum = "4ae48d6208a266e853d946088ed816055e556cc6028c5e8e2b84d9fa5dd7c7f5"
|
||||
dependencies = [
|
||||
"indexmap",
|
||||
"serde",
|
||||
@@ -1755,9 +1779,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "unicode-ident"
|
||||
version = "1.0.12"
|
||||
version = "1.0.13"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3354b9ac3fae1ff6755cb6db53683adb661634f67557942dea4facebec0fee4b"
|
||||
checksum = "e91b56cd4cadaeb79bbf1a5645f6b4f8dc5bde8834ad5894a8db35fda9efa1fe"
|
||||
|
||||
[[package]]
|
||||
name = "utf8parse"
|
||||
@@ -1806,19 +1830,20 @@ checksum = "9c8d87e72b64a3b4db28d11ce29237c246188f4f51057d65a7eab63b7987e423"
|
||||
|
||||
[[package]]
|
||||
name = "wasm-bindgen"
|
||||
version = "0.2.92"
|
||||
version = "0.2.93"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "4be2531df63900aeb2bca0daaaddec08491ee64ceecbee5076636a3b026795a8"
|
||||
checksum = "a82edfc16a6c469f5f44dc7b571814045d60404b55a0ee849f9bcfa2e63dd9b5"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"once_cell",
|
||||
"wasm-bindgen-macro",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "wasm-bindgen-backend"
|
||||
version = "0.2.92"
|
||||
version = "0.2.93"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "614d787b966d3989fa7bb98a654e369c762374fd3213d212cfc0251257e747da"
|
||||
checksum = "9de396da306523044d3302746f1208fa71d7532227f15e347e2d93e4145dd77b"
|
||||
dependencies = [
|
||||
"bumpalo",
|
||||
"log",
|
||||
@@ -1831,9 +1856,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "wasm-bindgen-macro"
|
||||
version = "0.2.92"
|
||||
version = "0.2.93"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "a1f8823de937b71b9460c0c34e25f3da88250760bec0ebac694b49997550d726"
|
||||
checksum = "585c4c91a46b072c92e908d99cb1dcdf95c5218eeb6f3bf1efa991ee7a68cccf"
|
||||
dependencies = [
|
||||
"quote",
|
||||
"wasm-bindgen-macro-support",
|
||||
@@ -1841,9 +1866,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "wasm-bindgen-macro-support"
|
||||
version = "0.2.92"
|
||||
version = "0.2.93"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e94f17b526d0a461a191c78ea52bbce64071ed5c04c9ffe424dcb38f74171bb7"
|
||||
checksum = "afc340c74d9005395cf9dd098506f7f44e38f2b4a21c6aaacf9a105ea5e1e836"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
@@ -1854,9 +1879,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "wasm-bindgen-shared"
|
||||
version = "0.2.92"
|
||||
version = "0.2.93"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "af190c94f2773fdb3729c55b007a722abb5384da03bc0986df4c289bf5567e96"
|
||||
checksum = "c62a0a307cb4a311d3a07867860911ca130c3494e8c2719593806c08bc5d0484"
|
||||
|
||||
[[package]]
|
||||
name = "weezl"
|
||||
@@ -1966,9 +1991,9 @@ checksum = "589f6da84c646204747d1270a2a5661ea66ed1cced2631d546fdfb155959f9ec"
|
||||
|
||||
[[package]]
|
||||
name = "winnow"
|
||||
version = "0.6.18"
|
||||
version = "0.6.20"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "68a9bda4691f099d435ad181000724da8e5899daa10713c2d432552b9ccd3a6f"
|
||||
checksum = "36c1fec1a2bb5866f07c25f68c26e565c4c200aebb96d7e55710c19d3e8ac49b"
|
||||
dependencies = [
|
||||
"memchr",
|
||||
]
|
||||
|
||||
+30
-8
@@ -27,6 +27,7 @@ document-features = "0.2.10"
|
||||
# Optional dependencies
|
||||
image = { version = "0.25.2", optional = true, default-features = false }
|
||||
bytemuck = { version = "1.16", optional = true }
|
||||
rayon = { version = "1.10", optional = true }
|
||||
|
||||
[features]
|
||||
## Enable this feature to implement traits [IntoImageView](crate::IntoImageView) and
|
||||
@@ -34,15 +35,17 @@ bytemuck = { version = "1.16", optional = true }
|
||||
## [DynamicImage](https://docs.rs/image/latest/image/enum.DynamicImage.html)
|
||||
## type from the `image` crate.
|
||||
image = ["dep:image", "dep:bytemuck"]
|
||||
## This feature enables image processing in `rayon` thread pool.
|
||||
rayon = ["dep:rayon", "resize/rayon", "image/rayon", "testing/rayon"]
|
||||
for_testing = ["image"]
|
||||
only_u8x4 = ["testing/only_u8x4"] # This can be used to experiment with the crate's code.
|
||||
|
||||
|
||||
[dev-dependencies]
|
||||
fast_image_resize = { path = ".", features = ["for_testing"] }
|
||||
resize = { version = "0.8.5", default-features = false, features = ["std"] }
|
||||
rgb = "0.8.48"
|
||||
png = "0.17.13"
|
||||
resize = { version = "0.8.7", default-features = false, features = ["std"] }
|
||||
rgb = "0.8.50"
|
||||
png = "0.17.14"
|
||||
serde = { version = "1.0", features = ["serde_derive"] }
|
||||
serde_json = "1.0"
|
||||
walkdir = "2.5"
|
||||
@@ -134,9 +137,33 @@ name = "bench_color_mapper"
|
||||
harness = false
|
||||
|
||||
|
||||
[profile.test]
|
||||
opt-level = 3
|
||||
incremental = true
|
||||
|
||||
|
||||
# debug builds for deps
|
||||
[profile.dev.package.'*']
|
||||
opt-level = 3
|
||||
debug = false
|
||||
# Strip debug symbols as they are useless with O3 anyway.
|
||||
strip = "debuginfo"
|
||||
|
||||
|
||||
# debug builds for procmacros
|
||||
[profile.dev.build-override]
|
||||
opt-level = 2 # reasonable optimization
|
||||
codegen-units = 256 # max threading
|
||||
# When possible - this is generally scary as some procmacros
|
||||
# will fail without any feedback.
|
||||
#debug = false
|
||||
|
||||
|
||||
# release build for procmacros - same config as debug build for procmacros
|
||||
[profile.release.build-override]
|
||||
opt-level = 2
|
||||
codegen-units = 256
|
||||
debug = false # when possible
|
||||
|
||||
|
||||
[profile.release]
|
||||
@@ -159,11 +186,6 @@ codegen-units = 1
|
||||
codegen-units = 1
|
||||
|
||||
|
||||
[profile.test]
|
||||
opt-level = 3
|
||||
incremental = true
|
||||
|
||||
|
||||
[package.metadata.release]
|
||||
pre-release-replacements = [
|
||||
{ file = "CHANGELOG.md", search = "Unreleased", replace = "{{version}}" },
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
[](https://crates.io/crates/fast_image_resize)
|
||||
[](https://docs.rs/fast_image_resize)
|
||||
|
||||
Rust library for fast **single-threaded** image resizing with using of SIMD instructions.
|
||||
Rust library for fast image resizing with using of SIMD instructions.
|
||||
|
||||
[CHANGELOG](https://github.com/Cykooz/fast_image_resize/blob/main/CHANGELOG.md)
|
||||
|
||||
@@ -44,7 +44,12 @@ In addition, the crate contains functions `create_gamma_22_mapper()`
|
||||
and `create_srgb_mapper()` to create instance of `PixelComponentMapper`
|
||||
that converts images from sRGB or gamma 2.2 into linear colorspace and back.
|
||||
|
||||
## Some benchmarks for x86_64
|
||||
## Multi-threading
|
||||
|
||||
You should enable `"rayon"` feature to turn on image processing in
|
||||
[rayon](https://docs.rs/rayon/latest/rayon/) thread pool.
|
||||
|
||||
## Some benchmarks in single-threaded mode for x86_64
|
||||
|
||||
_All benchmarks:_
|
||||
[_x86_64_](https://github.com/Cykooz/fast_image_resize/blob/main/benchmarks-x86_64.md),
|
||||
@@ -70,12 +75,12 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
|
||||
| image | 32.17 | - | 94.60 | 153.14 | 211.14 |
|
||||
| resize | 9.18 | 26.68 | 49.76 | 96.06 | 141.84 |
|
||||
| libvips | 7.75 | 59.58 | 19.81 | 30.46 | 39.96 |
|
||||
| fir rust | 0.29 | 11.99 | 16.56 | 25.93 | 37.85 |
|
||||
| fir sse4.1 | 0.29 | 4.13 | 5.67 | 9.77 | 15.52 |
|
||||
| fir avx2 | 0.29 | 3.13 | 3.98 | 6.88 | 13.18 |
|
||||
| image | 34.13 | - | 88.09 | 142.99 | 191.80 |
|
||||
| resize | 8.87 | 26.97 | 53.08 | 98.26 | 145.74 |
|
||||
| libvips | 2.38 | 61.57 | 5.66 | 9.70 | 16.03 |
|
||||
| fir rust | 0.28 | 10.93 | 15.35 | 25.77 | 37.09 |
|
||||
| fir sse4.1 | 0.28 | 3.43 | 5.39 | 9.82 | 15.34 |
|
||||
| fir avx2 | 0.28 | 2.62 | 3.80 | 6.89 | 13.22 |
|
||||
|
||||
<!-- bench_compare_rgb end -->
|
||||
|
||||
@@ -94,11 +99,11 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|------------|:-------:|:------:|:--------:|:-------:|:--------:|
|
||||
| resize | 11.98 | 43.60 | 86.90 | 147.95 | 211.64 |
|
||||
| libvips | 10.06 | 122.00 | 188.57 | 336.42 | 499.80 |
|
||||
| fir rust | 0.19 | 22.02 | 26.95 | 38.41 | 51.70 |
|
||||
| fir sse4.1 | 0.19 | 10.32 | 12.59 | 18.10 | 24.95 |
|
||||
| fir avx2 | 0.19 | 7.75 | 8.88 | 13.78 | 22.12 |
|
||||
| resize | 9.90 | 37.85 | 74.20 | 133.72 | 201.29 |
|
||||
| libvips | 4.17 | 169.03 | 141.52 | 232.30 | 330.89 |
|
||||
| fir rust | 0.19 | 20.66 | 26.02 | 37.27 | 50.21 |
|
||||
| fir sse4.1 | 0.19 | 9.59 | 11.99 | 17.79 | 24.83 |
|
||||
| fir avx2 | 0.19 | 7.21 | 8.61 | 13.22 | 22.41 |
|
||||
|
||||
<!-- bench_compare_rgba end -->
|
||||
|
||||
@@ -116,12 +121,12 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
|
||||
| image | 28.76 | - | 60.68 | 89.41 | 117.41 |
|
||||
| resize | 6.40 | 11.24 | 20.84 | 42.92 | 68.93 |
|
||||
| libvips | 4.66 | 25.06 | 9.67 | 13.27 | 17.99 |
|
||||
| fir rust | 0.15 | 4.74 | 6.02 | 8.41 | 12.62 |
|
||||
| fir sse4.1 | 0.15 | 1.67 | 2.14 | 3.31 | 5.61 |
|
||||
| fir avx2 | 0.15 | 1.74 | 1.91 | 2.31 | 4.16 |
|
||||
| image | 29.07 | - | 60.25 | 89.15 | 117.51 |
|
||||
| resize | 6.42 | 11.26 | 20.87 | 42.87 | 69.50 |
|
||||
| libvips | 2.57 | 25.05 | 6.82 | 9.85 | 12.68 |
|
||||
| fir rust | 0.15 | 4.45 | 5.57 | 9.02 | 12.31 |
|
||||
| fir sse4.1 | 0.15 | 1.52 | 2.09 | 3.52 | 5.65 |
|
||||
| fir avx2 | 0.15 | 1.54 | 1.76 | 2.80 | 4.03 |
|
||||
|
||||
<!-- bench_compare_l end -->
|
||||
|
||||
|
||||
@@ -6,8 +6,6 @@ use fast_image_resize::MulDiv;
|
||||
use fast_image_resize::PixelType;
|
||||
use testing::cpu_ext_into_str;
|
||||
|
||||
use crate::utils::pin_process_to_cpu0;
|
||||
|
||||
mod utils;
|
||||
|
||||
// Multiplies by alpha
|
||||
@@ -172,7 +170,6 @@ fn bench_alpha(bench_group: &mut utils::BenchGroup) {
|
||||
}
|
||||
|
||||
fn main() {
|
||||
pin_process_to_cpu0();
|
||||
let res = utils::run_bench(bench_alpha, "Bench Alpha");
|
||||
println!("{}", utils::build_md_table(&res));
|
||||
}
|
||||
|
||||
+12
-4
@@ -4,8 +4,6 @@ use fast_image_resize::ResizeOptions;
|
||||
use fast_image_resize::{CpuExtensions, FilterType, PixelType, ResizeAlg, Resizer};
|
||||
use testing::{cpu_ext_into_str, PixelTestingExt};
|
||||
|
||||
use crate::utils::pin_process_to_cpu0;
|
||||
|
||||
mod utils;
|
||||
|
||||
const NEW_SIZE: u32 = 695;
|
||||
@@ -111,6 +109,7 @@ pub fn resize_in_one_dimension_bench(bench_group: &mut utils::BenchGroup) {
|
||||
continue;
|
||||
}
|
||||
for &cpu_extension in cpu_extensions.iter() {
|
||||
#[cfg(not(feature = "only_u8x4"))]
|
||||
let image = match pixel_type {
|
||||
PixelType::U8 => U8::load_big_square_src_image(),
|
||||
PixelType::U8x2 => U8x2::load_big_square_src_image(),
|
||||
@@ -127,6 +126,11 @@ pub fn resize_in_one_dimension_bench(bench_group: &mut utils::BenchGroup) {
|
||||
PixelType::F32x4 => F32x4::load_big_square_src_image(),
|
||||
_ => unreachable!(),
|
||||
};
|
||||
#[cfg(feature = "only_u8x4")]
|
||||
let image = match pixel_type {
|
||||
PixelType::U8x4 => U8x4::load_big_square_src_image(),
|
||||
_ => unreachable!(),
|
||||
};
|
||||
downscale_bench(
|
||||
bench_group,
|
||||
&image,
|
||||
@@ -185,6 +189,7 @@ pub fn resize_bench(bench_group: &mut utils::BenchGroup) {
|
||||
continue;
|
||||
}
|
||||
for &cpu_extension in cpu_extensions.iter() {
|
||||
#[cfg(not(feature = "only_u8x4"))]
|
||||
let image = match pixel_type {
|
||||
PixelType::U8 => U8::load_big_square_src_image(),
|
||||
PixelType::U8x2 => U8x2::load_big_square_src_image(),
|
||||
@@ -201,6 +206,11 @@ pub fn resize_bench(bench_group: &mut utils::BenchGroup) {
|
||||
PixelType::F32x4 => F32x4::load_big_square_src_image(),
|
||||
_ => unreachable!(),
|
||||
};
|
||||
#[cfg(feature = "only_u8x4")]
|
||||
let image = match pixel_type {
|
||||
PixelType::U8x4 => U8x4::load_big_square_src_image(),
|
||||
_ => unreachable!(),
|
||||
};
|
||||
downscale_bench(
|
||||
bench_group,
|
||||
&image,
|
||||
@@ -219,13 +229,11 @@ pub fn resize_bench(bench_group: &mut utils::BenchGroup) {
|
||||
}
|
||||
|
||||
fn main1() {
|
||||
pin_process_to_cpu0();
|
||||
let results = utils::run_bench(resize_bench, "Resize");
|
||||
println!("{}", utils::build_md_table(&results));
|
||||
}
|
||||
|
||||
fn main() {
|
||||
pin_process_to_cpu0();
|
||||
let results = utils::run_bench(resize_in_one_dimension_bench, "Resize one dimension");
|
||||
println!("{}", utils::build_md_table(&results));
|
||||
}
|
||||
|
||||
@@ -8,20 +8,20 @@ Environment:
|
||||
- CPU: AMD Ryzen 9 5950X
|
||||
- RAM: DDR4 4000 MHz
|
||||
{% endif -%}
|
||||
- Ubuntu 22.04 (linux 6.5.0)
|
||||
- Rust 1.79
|
||||
- Ubuntu 24.04 (linux 6.8.0)
|
||||
- Rust 1.81.0
|
||||
- criterion = "0.5.1"
|
||||
- fast_image_resize = "4.1.0"
|
||||
- fast_image_resize = "4.3.0"
|
||||
{% if arch_id == "wasm32" -%}
|
||||
- wasmtime = "22.0.0"
|
||||
- wasmtime = "25.0.1"
|
||||
{% endif %}
|
||||
|
||||
Other libraries used to compare of resizing speed:
|
||||
|
||||
- image = "0.25.1" (<https://crates.io/crates/image>)
|
||||
- resize = "0.8.4" (<https://crates.io/crates/resize>, single-threaded mode)
|
||||
- image = "0.25.2" (<https://crates.io/crates/image>)
|
||||
- resize = "0.8.7" (<https://crates.io/crates/resize>, single-threaded mode)
|
||||
{% if arch_id != "wasm32" -%}
|
||||
- libvips = "8.12.1" (single-threaded mode)
|
||||
- libvips = "8.15.1" (single-threaded mode)
|
||||
{% endif %}
|
||||
|
||||
Resize algorithms:
|
||||
|
||||
@@ -97,7 +97,12 @@ mod vips {
|
||||
has_alpha: bool,
|
||||
) {
|
||||
let app = VipsApp::new("Test Libvips", false).expect("Cannot initialize libvips");
|
||||
app.concurrency_set(1);
|
||||
let num_threads: u32 = std::env::var("RAYON_NUM_THREADS")
|
||||
.map(|s| s.parse().unwrap_or(1))
|
||||
.unwrap_or(1);
|
||||
if num_threads > 0 {
|
||||
app.concurrency_set(num_threads as i32);
|
||||
}
|
||||
app.cache_set_max(0);
|
||||
app.cache_set_max_mem(0);
|
||||
|
||||
@@ -152,7 +157,7 @@ mod vips {
|
||||
"Lanczos3" => Kernel::Lanczos3,
|
||||
_ => continue,
|
||||
};
|
||||
let options = ReduceOptions { kernel };
|
||||
let options = ReduceOptions { kernel, gap: 0. };
|
||||
bench(bench_group, SAMPLE_SIZE, "libvips", alg_name, |bencher| {
|
||||
if has_alpha && alg_name != "Nearest" {
|
||||
bencher.iter(|| {
|
||||
|
||||
+52
-52
@@ -5,16 +5,16 @@
|
||||
Environment:
|
||||
|
||||
- CPU: Neoverse-N1 2GHz (Oracle Cloud Compute, VM.Standard.A1.Flex)
|
||||
- Ubuntu 22.04 (linux 6.5.0)
|
||||
- Rust 1.79
|
||||
- Ubuntu 24.04 (linux 6.8.0)
|
||||
- Rust 1.81.0
|
||||
- criterion = "0.5.1"
|
||||
- fast_image_resize = "4.1.0"
|
||||
- fast_image_resize = "4.3.0"
|
||||
|
||||
Other libraries used to compare of resizing speed:
|
||||
|
||||
- image = "0.25.1" (<https://crates.io/crates/image>)
|
||||
- resize = "0.8.4" (<https://crates.io/crates/resize>, single-threaded mode)
|
||||
- libvips = "8.12.1" (single-threaded mode)
|
||||
- image = "0.25.2" (<https://crates.io/crates/image>)
|
||||
- resize = "0.8.7" (<https://crates.io/crates/resize>, single-threaded mode)
|
||||
- libvips = "8.15.1" (single-threaded mode)
|
||||
|
||||
Resize algorithms:
|
||||
|
||||
@@ -39,11 +39,11 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|----------|:-------:|:------:|:--------:|:-------:|:--------:|
|
||||
| image | 88.05 | - | 160.98 | 294.47 | 426.20 |
|
||||
| resize | 18.14 | 57.28 | 99.28 | 185.31 | 269.80 |
|
||||
| libvips | 26.19 | 126.94 | 107.36 | 208.17 | 277.11 |
|
||||
| fir rust | 0.84 | 21.60 | 34.95 | 81.72 | 109.98 |
|
||||
| fir neon | 0.84 | 19.70 | 29.30 | 49.19 | 73.35 |
|
||||
| image | 87.29 | - | 161.50 | 300.01 | 422.03 |
|
||||
| resize | 18.40 | 58.24 | 103.05 | 187.60 | 276.63 |
|
||||
| libvips | 10.09 | 139.03 | 27.55 | 66.93 | 88.63 |
|
||||
| fir rust | 0.87 | 20.90 | 34.27 | 84.19 | 109.98 |
|
||||
| fir neon | 0.87 | 18.68 | 27.25 | 49.81 | 73.35 |
|
||||
|
||||
<!-- bench_compare_rgb end -->
|
||||
|
||||
@@ -62,10 +62,10 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|----------|:-------:|:------:|:--------:|:-------:|:--------:|
|
||||
| resize | 25.49 | 87.09 | 124.43 | 197.94 | 292.21 |
|
||||
| libvips | 33.92 | 235.04 | 350.01 | 708.27 | 973.00 |
|
||||
| fir rust | 0.95 | 47.90 | 62.63 | 125.31 | 167.72 |
|
||||
| fir neon | 0.95 | 37.47 | 50.57 | 76.71 | 106.23 |
|
||||
| resize | 21.06 | 74.75 | 116.24 | 197.20 | 294.64 |
|
||||
| libvips | 12.78 | 324.80 | 230.82 | 453.35 | 568.10 |
|
||||
| fir rust | 0.92 | 47.50 | 59.95 | 129.03 | 166.45 |
|
||||
| fir neon | 0.92 | 35.02 | 49.27 | 77.35 | 105.99 |
|
||||
|
||||
<!-- bench_compare_rgba end -->
|
||||
|
||||
@@ -83,11 +83,11 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|----------|:-------:|:-----:|:--------:|:-------:|:--------:|
|
||||
| image | 82.40 | - | 108.23 | 176.18 | 243.14 |
|
||||
| resize | 10.63 | 27.73 | 40.17 | 63.92 | 93.05 |
|
||||
| libvips | 12.24 | 49.95 | 38.41 | 69.98 | 94.07 |
|
||||
| fir rust | 0.47 | 8.81 | 12.28 | 18.45 | 28.09 |
|
||||
| fir neon | 0.47 | 5.98 | 9.10 | 15.85 | 24.45 |
|
||||
| image | 78.41 | - | 104.80 | 174.66 | 240.78 |
|
||||
| resize | 10.63 | 27.45 | 40.33 | 63.03 | 92.52 |
|
||||
| libvips | 5.70 | 51.73 | 14.50 | 24.28 | 30.84 |
|
||||
| fir rust | 0.50 | 8.24 | 11.40 | 19.10 | 26.72 |
|
||||
| fir neon | 0.50 | 5.49 | 8.92 | 16.09 | 24.49 |
|
||||
|
||||
<!-- bench_compare_l end -->
|
||||
|
||||
@@ -108,9 +108,9 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|----------|:-------:|:------:|:--------:|:-------:|:--------:|
|
||||
| libvips | 19.00 | 144.15 | 223.38 | 402.93 | 554.41 |
|
||||
| fir rust | 0.67 | 37.10 | 45.15 | 60.54 | 71.70 |
|
||||
| fir neon | 0.67 | 18.83 | 25.06 | 37.76 | 52.57 |
|
||||
| libvips | 8.59 | 185.59 | 132.47 | 229.07 | 288.92 |
|
||||
| fir rust | 0.66 | 35.15 | 44.12 | 58.62 | 70.41 |
|
||||
| fir neon | 0.66 | 17.61 | 23.13 | 37.96 | 52.88 |
|
||||
|
||||
<!-- bench_compare_la end -->
|
||||
|
||||
@@ -128,11 +128,11 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|----------|:-------:|:------:|:--------:|:-------:|:--------:|
|
||||
| image | 93.07 | - | 169.86 | 332.89 | 475.00 |
|
||||
| resize | 19.55 | 56.39 | 96.71 | 182.61 | 270.17 |
|
||||
| libvips | 27.34 | 140.29 | 112.81 | 235.59 | 313.47 |
|
||||
| fir rust | 1.36 | 61.08 | 89.19 | 145.51 | 203.83 |
|
||||
| fir neon | 1.36 | 63.59 | 70.68 | 93.43 | 131.31 |
|
||||
| image | 89.59 | - | 161.49 | 329.96 | 478.48 |
|
||||
| resize | 20.19 | 58.37 | 99.98 | 185.89 | 272.58 |
|
||||
| libvips | 24.24 | 200.90 | 111.02 | 231.86 | 311.97 |
|
||||
| fir rust | 1.30 | 54.29 | 84.54 | 143.06 | 204.83 |
|
||||
| fir neon | 1.30 | 51.49 | 73.47 | 114.72 | 141.26 |
|
||||
|
||||
<!-- bench_compare_rgb16 end -->
|
||||
|
||||
@@ -151,10 +151,10 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|----------|:-------:|:------:|:--------:|:-------:|:--------:|
|
||||
| resize | 29.04 | 88.00 | 133.03 | 221.31 | 314.83 |
|
||||
| libvips | 36.63 | 246.57 | 369.53 | 747.34 | 1015.17 |
|
||||
| fir rust | 1.59 | 96.13 | 131.20 | 218.79 | 292.14 |
|
||||
| fir neon | 1.59 | 55.42 | 77.15 | 118.79 | 163.50 |
|
||||
| resize | 25.53 | 78.22 | 117.14 | 211.16 | 308.86 |
|
||||
| libvips | 32.82 | 324.74 | 231.14 | 460.10 | 580.86 |
|
||||
| fir rust | 1.57 | 91.95 | 126.75 | 208.18 | 285.42 |
|
||||
| fir neon | 1.57 | 52.53 | 74.02 | 114.79 | 157.37 |
|
||||
|
||||
<!-- bench_compare_rgba16 end -->
|
||||
|
||||
@@ -172,11 +172,11 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|----------|:-------:|:-----:|:--------:|:-------:|:--------:|
|
||||
| image | 82.99 | - | 113.48 | 188.98 | 258.69 |
|
||||
| resize | 11.09 | 27.46 | 43.78 | 72.15 | 98.44 |
|
||||
| libvips | 12.35 | 55.60 | 42.90 | 81.53 | 104.17 |
|
||||
| fir rust | 0.66 | 27.95 | 38.46 | 59.48 | 84.43 |
|
||||
| fir neon | 0.66 | 13.63 | 17.66 | 27.17 | 38.21 |
|
||||
| image | 80.33 | - | 109.18 | 182.67 | 252.40 |
|
||||
| resize | 11.16 | 27.22 | 43.39 | 71.07 | 96.37 |
|
||||
| libvips | 9.20 | 69.16 | 39.31 | 77.20 | 100.41 |
|
||||
| fir rust | 0.67 | 24.39 | 36.16 | 59.35 | 84.09 |
|
||||
| fir neon | 0.67 | 11.97 | 16.87 | 26.77 | 38.06 |
|
||||
|
||||
<!-- bench_compare_l16 end -->
|
||||
|
||||
@@ -197,9 +197,9 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|----------|:-------:|:------:|:--------:|:-------:|:--------:|
|
||||
| libvips | 20.74 | 149.18 | 237.81 | 430.47 | 582.17 |
|
||||
| fir rust | 1.03 | 61.38 | 80.31 | 115.02 | 153.03 |
|
||||
| fir neon | 1.03 | 27.42 | 36.49 | 55.28 | 74.79 |
|
||||
| libvips | 17.31 | 200.28 | 144.61 | 241.53 | 298.13 |
|
||||
| fir rust | 0.97 | 55.28 | 74.08 | 114.13 | 153.81 |
|
||||
| fir neon | 0.97 | 24.97 | 34.56 | 53.88 | 75.30 |
|
||||
|
||||
<!-- bench_compare_la16 end -->
|
||||
|
||||
@@ -217,10 +217,10 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|----------|:-------:|:-----:|:--------:|:-------:|:--------:|
|
||||
| image | 42.00 | - | 96.55 | 185.17 | 250.26 |
|
||||
| resize | 11.85 | 23.72 | 32.72 | 56.64 | 84.38 |
|
||||
| libvips | 10.51 | 53.28 | 44.32 | 100.09 | 124.13 |
|
||||
| fir rust | 0.96 | 21.13 | 32.32 | 53.92 | 79.45 |
|
||||
| image | 39.38 | - | 94.31 | 185.23 | 241.78 |
|
||||
| resize | 11.82 | 22.48 | 31.48 | 54.70 | 81.74 |
|
||||
| libvips | 8.05 | 66.37 | 39.86 | 92.17 | 120.92 |
|
||||
| fir rust | 0.96 | 18.62 | 30.71 | 53.79 | 77.15 |
|
||||
|
||||
<!-- bench_compare_l32f end -->
|
||||
|
||||
@@ -246,8 +246,8 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|----------|:-------:|:------:|:--------:|:-------:|:--------:|
|
||||
| libvips | 19.41 | 142.92 | 210.17 | 382.61 | 506.77 |
|
||||
| fir rust | 1.60 | 43.16 | 68.07 | 132.43 | 169.07 |
|
||||
| libvips | 16.45 | 182.62 | 127.58 | 220.52 | 278.16 |
|
||||
| fir rust | 1.56 | 41.75 | 66.81 | 123.14 | 168.42 |
|
||||
|
||||
<!-- bench_compare_la32f end -->
|
||||
|
||||
@@ -265,10 +265,10 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|----------|:-------:|:------:|:--------:|:-------:|:--------:|
|
||||
| image | 51.58 | - | 129.50 | 310.96 | 419.49 |
|
||||
| resize | 20.90 | 34.90 | 66.06 | 128.80 | 189.37 |
|
||||
| libvips | 23.94 | 141.04 | 117.50 | 280.96 | 365.64 |
|
||||
| fir rust | 2.29 | 41.90 | 77.14 | 163.38 | 224.46 |
|
||||
| image | 50.67 | - | 129.41 | 322.93 | 441.29 |
|
||||
| resize | 21.65 | 36.44 | 67.62 | 135.27 | 190.02 |
|
||||
| libvips | 19.82 | 200.45 | 114.77 | 278.40 | 355.69 |
|
||||
| fir rust | 2.24 | 38.69 | 73.20 | 152.46 | 225.81 |
|
||||
|
||||
<!-- bench_compare_rgb32f end -->
|
||||
|
||||
@@ -290,7 +290,7 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|----------|:-------:|:------:|:--------:|:-------:|:--------:|
|
||||
| libvips | 36.40 | 236.39 | 326.62 | 618.09 | 820.17 |
|
||||
| fir rust | 3.33 | 72.85 | 118.83 | 244.32 | 322.10 |
|
||||
| libvips | 30.58 | 318.17 | 223.24 | 410.56 | 530.71 |
|
||||
| fir rust | 3.12 | 68.42 | 112.09 | 213.48 | 316.09 |
|
||||
|
||||
<!-- bench_compare_rgba32f end -->
|
||||
|
||||
+42
-42
@@ -6,16 +6,16 @@ Environment:
|
||||
|
||||
- CPU: AMD Ryzen 9 5950X
|
||||
- RAM: DDR4 4000 MHz
|
||||
- Ubuntu 22.04 (linux 6.5.0)
|
||||
- Rust 1.79
|
||||
- Ubuntu 24.04 (linux 6.8.0)
|
||||
- Rust 1.81.0
|
||||
- criterion = "0.5.1"
|
||||
- fast_image_resize = "4.1.0"
|
||||
- wasmtime = "22.0.0"
|
||||
- fast_image_resize = "4.3.0"
|
||||
- wasmtime = "25.0.1"
|
||||
|
||||
Other libraries used to compare of resizing speed:
|
||||
|
||||
- image = "0.25.1" (<https://crates.io/crates/image>)
|
||||
- resize = "0.8.4" (<https://crates.io/crates/resize>, single-threaded mode)
|
||||
- image = "0.25.2" (<https://crates.io/crates/image>)
|
||||
- resize = "0.8.7" (<https://crates.io/crates/resize>, single-threaded mode)
|
||||
|
||||
Resize algorithms:
|
||||
|
||||
@@ -40,10 +40,10 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|-------------|:-------:|:-----:|:--------:|:-------:|:--------:|
|
||||
| image | 26.19 | - | 103.03 | 180.66 | 258.39 |
|
||||
| resize | 11.71 | 33.22 | 60.30 | 114.09 | 167.14 |
|
||||
| fir rust | 0.39 | 45.14 | 79.53 | 150.42 | 223.19 |
|
||||
| fir simd128 | 0.39 | 6.41 | 8.71 | 14.33 | 21.60 |
|
||||
| image | 25.61 | - | 111.45 | 199.95 | 281.24 |
|
||||
| resize | 11.32 | 33.48 | 60.90 | 113.34 | 166.20 |
|
||||
| fir rust | 0.36 | 40.56 | 76.04 | 148.68 | 223.70 |
|
||||
| fir simd128 | 0.36 | 5.19 | 7.93 | 14.07 | 20.86 |
|
||||
|
||||
<!-- bench_compare_rgb end -->
|
||||
|
||||
@@ -60,11 +60,11 @@ Pipeline:
|
||||
- Numbers in the table mean a duration of image resizing in milliseconds.
|
||||
- The `image` crate does not support multiplying and dividing by alpha channel.
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|-------------|:-------:|:------:|:--------:|:-------:|:--------:|
|
||||
| resize | 12.41 | 38.94 | 73.52 | 138.90 | 211.97 |
|
||||
| fir rust | 0.27 | 101.41 | 147.99 | 244.04 | 341.24 |
|
||||
| fir simd128 | 0.27 | 16.30 | 19.02 | 25.52 | 33.58 |
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|-------------|:-------:|:-----:|:--------:|:-------:|:--------:|
|
||||
| resize | 12.36 | 39.11 | 73.57 | 140.10 | 216.32 |
|
||||
| fir rust | 0.23 | 96.26 | 143.11 | 240.79 | 338.59 |
|
||||
| fir simd128 | 0.23 | 15.98 | 18.99 | 25.80 | 33.58 |
|
||||
|
||||
<!-- bench_compare_rgba end -->
|
||||
|
||||
@@ -82,10 +82,10 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|-------------|:-------:|:-----:|:--------:|:-------:|:--------:|
|
||||
| image | 24.25 | - | 87.19 | 149.64 | 210.43 |
|
||||
| resize | 7.87 | 17.80 | 28.64 | 53.29 | 77.57 |
|
||||
| fir rust | 0.21 | 16.50 | 28.23 | 52.74 | 78.04 |
|
||||
| fir simd128 | 0.21 | 3.02 | 3.32 | 4.66 | 7.59 |
|
||||
| image | 23.29 | - | 87.60 | 150.35 | 211.42 |
|
||||
| resize | 7.79 | 17.22 | 27.25 | 50.98 | 74.74 |
|
||||
| fir rust | 0.21 | 15.15 | 27.92 | 54.02 | 81.12 |
|
||||
| fir simd128 | 0.21 | 2.65 | 3.32 | 5.36 | 7.97 |
|
||||
|
||||
<!-- bench_compare_l end -->
|
||||
|
||||
@@ -106,8 +106,8 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|-------------|:-------:|:-----:|:--------:|:-------:|:--------:|
|
||||
| fir rust | 0.20 | 50.44 | 73.54 | 120.60 | 168.31 |
|
||||
| fir simd128 | 0.20 | 8.49 | 9.75 | 12.51 | 17.40 |
|
||||
| fir rust | 0.19 | 49.46 | 74.31 | 124.11 | 175.58 |
|
||||
| fir simd128 | 0.19 | 8.02 | 9.41 | 12.85 | 17.58 |
|
||||
|
||||
<!-- bench_compare_la end -->
|
||||
|
||||
@@ -125,10 +125,10 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|-------------|:-------:|:-----:|:--------:|:-------:|:--------:|
|
||||
| image | 27.31 | - | 106.09 | 188.35 | 270.09 |
|
||||
| resize | 12.10 | 33.83 | 60.20 | 113.85 | 168.18 |
|
||||
| fir rust | 0.41 | 39.02 | 62.29 | 107.96 | 154.74 |
|
||||
| fir simd128 | 0.41 | 32.47 | 51.34 | 89.26 | 129.12 |
|
||||
| image | 26.75 | - | 107.99 | 188.76 | 270.38 |
|
||||
| resize | 11.39 | 30.97 | 53.58 | 106.09 | 157.82 |
|
||||
| fir rust | 0.42 | 34.58 | 58.12 | 107.33 | 157.86 |
|
||||
| fir simd128 | 0.42 | 28.62 | 48.16 | 87.52 | 126.68 |
|
||||
|
||||
<!-- bench_compare_rgb16 end -->
|
||||
|
||||
@@ -147,9 +147,9 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|-------------|:-------:|:-----:|:--------:|:-------:|:--------:|
|
||||
| resize | 12.69 | 39.35 | 74.01 | 139.45 | 213.46 |
|
||||
| fir rust | 0.34 | 87.33 | 110.98 | 158.35 | 207.47 |
|
||||
| fir simd128 | 0.34 | 51.84 | 75.53 | 123.10 | 172.10 |
|
||||
| resize | 12.73 | 39.77 | 75.15 | 140.39 | 214.27 |
|
||||
| fir rust | 0.39 | 89.75 | 123.03 | 185.52 | 250.31 |
|
||||
| fir simd128 | 0.39 | 46.65 | 70.80 | 121.69 | 170.89 |
|
||||
|
||||
<!-- bench_compare_rgba16 end -->
|
||||
|
||||
@@ -167,10 +167,10 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|-------------|:-------:|:-----:|:--------:|:-------:|:--------:|
|
||||
| image | 23.98 | - | 86.94 | 149.67 | 211.09 |
|
||||
| resize | 7.88 | 17.12 | 27.73 | 52.04 | 75.31 |
|
||||
| fir rust | 0.19 | 21.90 | 30.69 | 49.16 | 69.15 |
|
||||
| fir simd128 | 0.19 | 12.02 | 17.84 | 29.47 | 42.84 |
|
||||
| image | 23.45 | - | 79.42 | 134.57 | 188.61 |
|
||||
| resize | 8.33 | 16.79 | 26.60 | 50.54 | 73.93 |
|
||||
| fir rust | 0.19 | 18.79 | 26.98 | 46.01 | 64.94 |
|
||||
| fir simd128 | 0.19 | 10.14 | 17.11 | 29.49 | 42.87 |
|
||||
|
||||
<!-- bench_compare_l16 end -->
|
||||
|
||||
@@ -191,8 +191,8 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|-------------|:-------:|:-----:|:--------:|:-------:|:--------:|
|
||||
| fir rust | 0.23 | 49.69 | 66.52 | 98.05 | 131.53 |
|
||||
| fir simd128 | 0.23 | 28.41 | 40.62 | 65.56 | 92.21 |
|
||||
| fir rust | 0.25 | 44.15 | 59.50 | 91.27 | 125.41 |
|
||||
| fir simd128 | 0.25 | 26.02 | 38.49 | 65.75 | 92.37 |
|
||||
|
||||
<!-- bench_compare_la16 end -->
|
||||
|
||||
@@ -210,9 +210,9 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|----------|:-------:|:-----:|:--------:|:-------:|:--------:|
|
||||
| image | 12.99 | - | 56.03 | 101.33 | 145.65 |
|
||||
| resize | 7.49 | 16.01 | 22.29 | 43.94 | 64.20 |
|
||||
| fir rust | 0.24 | 11.16 | 19.88 | 39.47 | 60.69 |
|
||||
| image | 13.04 | - | 63.40 | 116.33 | 167.53 |
|
||||
| resize | 7.31 | 15.27 | 22.17 | 42.93 | 65.69 |
|
||||
| fir rust | 0.25 | 10.20 | 18.49 | 35.77 | 59.88 |
|
||||
|
||||
<!-- bench_compare_l32f end -->
|
||||
|
||||
@@ -238,7 +238,7 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|----------|:-------:|:-----:|:--------:|:-------:|:--------:|
|
||||
| fir rust | 0.42 | 32.63 | 44.56 | 70.31 | 97.25 |
|
||||
| fir rust | 0.41 | 30.13 | 42.49 | 69.34 | 98.15 |
|
||||
|
||||
<!-- bench_compare_la32f end -->
|
||||
|
||||
@@ -256,9 +256,9 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|----------|:-------:|:-----:|:--------:|:-------:|:--------:|
|
||||
| image | 16.21 | - | 66.86 | 121.43 | 176.37 |
|
||||
| resize | 10.69 | 21.69 | 36.37 | 66.61 | 97.14 |
|
||||
| fir rust | 1.05 | 31.15 | 52.73 | 96.40 | 143.25 |
|
||||
| image | 16.34 | - | 67.78 | 123.04 | 180.03 |
|
||||
| resize | 10.39 | 21.49 | 35.88 | 65.69 | 95.80 |
|
||||
| fir rust | 1.02 | 25.55 | 45.73 | 86.30 | 128.94 |
|
||||
|
||||
<!-- bench_compare_rgb32f end -->
|
||||
|
||||
@@ -280,6 +280,6 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|----------|:-------:|:-----:|:--------:|:-------:|:--------:|
|
||||
| fir rust | 1.27 | 55.23 | 78.01 | 126.18 | 175.73 |
|
||||
| fir rust | 1.20 | 50.35 | 73.55 | 118.86 | 168.11 |
|
||||
|
||||
<!-- bench_compare_rgba32f end -->
|
||||
|
||||
+70
-70
@@ -6,16 +6,16 @@ Environment:
|
||||
|
||||
- CPU: AMD Ryzen 9 5950X
|
||||
- RAM: DDR4 4000 MHz
|
||||
- Ubuntu 22.04 (linux 6.5.0)
|
||||
- Rust 1.79
|
||||
- Ubuntu 24.04 (linux 6.8.0)
|
||||
- Rust 1.81.0
|
||||
- criterion = "0.5.1"
|
||||
- fast_image_resize = "4.1.0"
|
||||
- fast_image_resize = "4.3.0"
|
||||
|
||||
Other libraries used to compare of resizing speed:
|
||||
|
||||
- image = "0.25.1" (<https://crates.io/crates/image>)
|
||||
- resize = "0.8.4" (<https://crates.io/crates/resize>, single-threaded mode)
|
||||
- libvips = "8.12.1" (single-threaded mode)
|
||||
- image = "0.25.2" (<https://crates.io/crates/image>)
|
||||
- resize = "0.8.7" (<https://crates.io/crates/resize>, single-threaded mode)
|
||||
- libvips = "8.15.1" (single-threaded mode)
|
||||
|
||||
Resize algorithms:
|
||||
|
||||
@@ -40,12 +40,12 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
|
||||
| image | 32.17 | - | 94.60 | 153.14 | 211.14 |
|
||||
| resize | 9.18 | 26.68 | 49.76 | 96.06 | 141.84 |
|
||||
| libvips | 7.75 | 59.58 | 19.81 | 30.46 | 39.96 |
|
||||
| fir rust | 0.29 | 11.99 | 16.56 | 25.93 | 37.85 |
|
||||
| fir sse4.1 | 0.29 | 4.13 | 5.67 | 9.77 | 15.52 |
|
||||
| fir avx2 | 0.29 | 3.13 | 3.98 | 6.88 | 13.18 |
|
||||
| image | 34.13 | - | 88.09 | 142.99 | 191.80 |
|
||||
| resize | 8.87 | 26.97 | 53.08 | 98.26 | 145.74 |
|
||||
| libvips | 2.38 | 61.57 | 5.66 | 9.70 | 16.03 |
|
||||
| fir rust | 0.28 | 10.93 | 15.35 | 25.77 | 37.09 |
|
||||
| fir sse4.1 | 0.28 | 3.43 | 5.39 | 9.82 | 15.34 |
|
||||
| fir avx2 | 0.28 | 2.62 | 3.80 | 6.89 | 13.22 |
|
||||
|
||||
<!-- bench_compare_rgb end -->
|
||||
|
||||
@@ -64,11 +64,11 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|------------|:-------:|:------:|:--------:|:-------:|:--------:|
|
||||
| resize | 11.98 | 43.60 | 86.90 | 147.95 | 211.64 |
|
||||
| libvips | 10.06 | 122.00 | 188.57 | 336.42 | 499.80 |
|
||||
| fir rust | 0.19 | 22.02 | 26.95 | 38.41 | 51.70 |
|
||||
| fir sse4.1 | 0.19 | 10.32 | 12.59 | 18.10 | 24.95 |
|
||||
| fir avx2 | 0.19 | 7.75 | 8.88 | 13.78 | 22.12 |
|
||||
| resize | 9.90 | 37.85 | 74.20 | 133.72 | 201.29 |
|
||||
| libvips | 4.17 | 169.03 | 141.52 | 232.30 | 330.89 |
|
||||
| fir rust | 0.19 | 20.66 | 26.02 | 37.27 | 50.21 |
|
||||
| fir sse4.1 | 0.19 | 9.59 | 11.99 | 17.79 | 24.83 |
|
||||
| fir avx2 | 0.19 | 7.21 | 8.61 | 13.22 | 22.41 |
|
||||
|
||||
<!-- bench_compare_rgba end -->
|
||||
|
||||
@@ -86,12 +86,12 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
|
||||
| image | 28.76 | - | 60.68 | 89.41 | 117.41 |
|
||||
| resize | 6.40 | 11.24 | 20.84 | 42.92 | 68.93 |
|
||||
| libvips | 4.66 | 25.06 | 9.67 | 13.27 | 17.99 |
|
||||
| fir rust | 0.15 | 4.74 | 6.02 | 8.41 | 12.62 |
|
||||
| fir sse4.1 | 0.15 | 1.67 | 2.14 | 3.31 | 5.61 |
|
||||
| fir avx2 | 0.15 | 1.74 | 1.91 | 2.31 | 4.16 |
|
||||
| image | 29.07 | - | 60.25 | 89.15 | 117.51 |
|
||||
| resize | 6.42 | 11.26 | 20.87 | 42.87 | 69.50 |
|
||||
| libvips | 2.57 | 25.05 | 6.82 | 9.85 | 12.68 |
|
||||
| fir rust | 0.15 | 4.45 | 5.57 | 9.02 | 12.31 |
|
||||
| fir sse4.1 | 0.15 | 1.52 | 2.09 | 3.52 | 5.65 |
|
||||
| fir avx2 | 0.15 | 1.54 | 1.76 | 2.80 | 4.03 |
|
||||
|
||||
<!-- bench_compare_l end -->
|
||||
|
||||
@@ -112,10 +112,10 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
|
||||
| libvips | 6.42 | 73.21 | 118.10 | 205.49 | 292.37 |
|
||||
| fir rust | 0.18 | 18.67 | 21.14 | 25.86 | 32.98 |
|
||||
| fir sse4.1 | 0.17 | 6.18 | 7.21 | 9.60 | 13.48 |
|
||||
| fir avx2 | 0.17 | 4.35 | 4.92 | 6.48 | 9.60 |
|
||||
| libvips | 3.70 | 94.30 | 79.22 | 123.08 | 165.17 |
|
||||
| fir rust | 0.17 | 17.42 | 19.66 | 25.76 | 31.87 |
|
||||
| fir sse4.1 | 0.17 | 5.89 | 7.11 | 9.83 | 13.51 |
|
||||
| fir avx2 | 0.17 | 4.09 | 4.85 | 6.70 | 9.68 |
|
||||
|
||||
<!-- bench_compare_la end -->
|
||||
|
||||
@@ -133,12 +133,12 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
|
||||
| image | 31.17 | - | 86.80 | 138.46 | 191.10 |
|
||||
| resize | 8.16 | 26.79 | 50.87 | 97.41 | 144.44 |
|
||||
| libvips | 16.00 | 63.08 | 54.33 | 102.53 | 125.07 |
|
||||
| fir rust | 0.33 | 30.74 | 47.64 | 81.28 | 115.50 |
|
||||
| fir sse4.1 | 0.33 | 16.24 | 23.65 | 38.31 | 54.69 |
|
||||
| fir avx2 | 0.33 | 13.88 | 19.29 | 29.86 | 36.75 |
|
||||
| image | 31.57 | - | 87.50 | 140.87 | 192.50 |
|
||||
| resize | 8.27 | 26.82 | 50.82 | 97.83 | 144.89 |
|
||||
| libvips | 14.14 | 95.67 | 67.26 | 130.91 | 175.04 |
|
||||
| fir rust | 0.34 | 27.89 | 45.79 | 79.04 | 113.41 |
|
||||
| fir sse4.1 | 0.34 | 14.54 | 22.50 | 38.58 | 54.63 |
|
||||
| fir avx2 | 0.34 | 12.26 | 17.80 | 28.55 | 37.15 |
|
||||
|
||||
<!-- bench_compare_rgb16 end -->
|
||||
|
||||
@@ -157,11 +157,11 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|------------|:-------:|:------:|:--------:|:-------:|:--------:|
|
||||
| resize | 13.57 | 44.09 | 84.32 | 145.34 | 207.68 |
|
||||
| libvips | 22.77 | 128.72 | 205.37 | 366.05 | 537.03 |
|
||||
| fir rust | 0.40 | 63.43 | 84.58 | 127.16 | 171.87 |
|
||||
| fir sse4.1 | 0.40 | 32.01 | 42.48 | 63.86 | 86.06 |
|
||||
| fir avx2 | 0.40 | 20.60 | 26.04 | 36.66 | 47.97 |
|
||||
| resize | 11.11 | 39.37 | 71.59 | 135.94 | 204.58 |
|
||||
| libvips | 21.16 | 181.24 | 153.70 | 241.95 | 342.56 |
|
||||
| fir rust | 0.38 | 60.31 | 79.71 | 122.36 | 166.54 |
|
||||
| fir sse4.1 | 0.38 | 30.83 | 41.64 | 63.39 | 85.77 |
|
||||
| fir avx2 | 0.38 | 20.80 | 26.25 | 37.26 | 49.02 |
|
||||
|
||||
<!-- bench_compare_rgba16 end -->
|
||||
|
||||
@@ -179,12 +179,12 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
|
||||
| image | 29.50 | - | 61.96 | 91.03 | 119.53 |
|
||||
| resize | 6.40 | 11.10 | 20.93 | 43.22 | 68.96 |
|
||||
| libvips | 7.45 | 26.23 | 21.80 | 36.47 | 46.00 |
|
||||
| fir rust | 0.17 | 14.84 | 21.62 | 30.25 | 42.89 |
|
||||
| fir sse4.1 | 0.17 | 5.40 | 7.49 | 12.91 | 18.84 |
|
||||
| fir avx2 | 0.17 | 5.49 | 6.42 | 8.54 | 13.70 |
|
||||
| image | 28.99 | - | 61.22 | 91.67 | 119.54 |
|
||||
| resize | 7.02 | 11.23 | 21.15 | 44.75 | 69.40 |
|
||||
| libvips | 5.68 | 34.67 | 23.78 | 43.76 | 59.42 |
|
||||
| fir rust | 0.17 | 13.53 | 21.22 | 32.94 | 45.88 |
|
||||
| fir sse4.1 | 0.17 | 4.95 | 7.20 | 12.59 | 18.74 |
|
||||
| fir avx2 | 0.17 | 4.94 | 6.14 | 9.09 | 13.75 |
|
||||
|
||||
<!-- bench_compare_l16 end -->
|
||||
|
||||
@@ -203,12 +203,12 @@ Pipeline:
|
||||
- The `image` crate does not support multiplying and dividing by alpha channel.
|
||||
- The `resize` crate does not support this pixel format.
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
|
||||
| libvips | 12.51 | 79.48 | 133.81 | 232.12 | 326.55 |
|
||||
| fir rust | 0.19 | 32.88 | 40.97 | 62.61 | 84.70 |
|
||||
| fir sse4.1 | 0.19 | 15.01 | 21.19 | 33.27 | 45.87 |
|
||||
| fir avx2 | 0.19 | 11.84 | 15.10 | 22.03 | 29.10 |
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|------------|:-------:|:------:|:--------:|:-------:|:--------:|
|
||||
| libvips | 11.27 | 104.93 | 87.91 | 132.52 | 175.67 |
|
||||
| fir rust | 0.19 | 30.06 | 38.42 | 59.01 | 81.08 |
|
||||
| fir sse4.1 | 0.19 | 14.45 | 20.42 | 31.81 | 45.02 |
|
||||
| fir avx2 | 0.19 | 11.27 | 14.54 | 21.66 | 28.87 |
|
||||
|
||||
<!-- bench_compare_la16 end -->
|
||||
|
||||
@@ -226,12 +226,12 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
|
||||
| image | 24.21 | - | 48.61 | 79.04 | 104.73 |
|
||||
| resize | 5.26 | 8.83 | 13.71 | 30.19 | 45.85 |
|
||||
| libvips | 6.21 | 27.94 | 19.81 | 43.10 | 69.32 |
|
||||
| fir rust | 0.19 | 8.32 | 12.65 | 26.87 | 41.19 |
|
||||
| fir sse4.1 | 0.19 | 5.28 | 7.28 | 11.67 | 17.07 |
|
||||
| fir avx2 | 0.19 | 4.68 | 5.46 | 7.11 | 10.73 |
|
||||
| image | 25.15 | - | 53.26 | 84.55 | 112.61 |
|
||||
| resize | 5.18 | 8.87 | 13.79 | 30.25 | 46.06 |
|
||||
| libvips | 4.65 | 33.97 | 23.90 | 46.23 | 64.98 |
|
||||
| fir rust | 0.19 | 7.12 | 11.76 | 26.13 | 39.24 |
|
||||
| fir sse4.1 | 0.19 | 4.36 | 6.75 | 11.60 | 16.87 |
|
||||
| fir avx2 | 0.19 | 4.06 | 5.29 | 7.93 | 11.18 |
|
||||
|
||||
<!-- bench_compare_l32f end -->
|
||||
|
||||
@@ -252,10 +252,10 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
|
||||
| libvips | 13.72 | 72.74 | 100.78 | 176.30 | 252.88 |
|
||||
| fir rust | 0.35 | 21.99 | 29.09 | 47.76 | 70.52 |
|
||||
| fir sse4.1 | 0.35 | 16.25 | 20.94 | 30.33 | 40.11 |
|
||||
| fir avx2 | 0.35 | 15.73 | 17.36 | 22.92 | 27.79 |
|
||||
| libvips | 10.75 | 91.80 | 77.32 | 118.49 | 161.60 |
|
||||
| fir rust | 0.38 | 22.66 | 29.18 | 47.22 | 71.11 |
|
||||
| fir sse4.1 | 0.38 | 17.36 | 22.09 | 30.88 | 40.93 |
|
||||
| fir avx2 | 0.38 | 16.35 | 18.25 | 24.06 | 28.64 |
|
||||
|
||||
<!-- bench_compare_la32f end -->
|
||||
|
||||
@@ -273,12 +273,12 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
|
||||
| image | 26.67 | - | 64.09 | 108.10 | 155.20 |
|
||||
| resize | 9.30 | 16.45 | 24.68 | 48.28 | 72.52 |
|
||||
| libvips | 13.46 | 62.91 | 52.79 | 114.29 | 199.08 |
|
||||
| fir rust | 0.86 | 16.49 | 27.10 | 51.23 | 76.20 |
|
||||
| fir sse4.1 | 0.86 | 12.38 | 19.56 | 32.72 | 47.73 |
|
||||
| fir avx2 | 0.86 | 11.23 | 14.21 | 20.39 | 29.32 |
|
||||
| image | 27.62 | - | 64.22 | 108.90 | 148.53 |
|
||||
| resize | 9.17 | 16.34 | 24.61 | 48.55 | 72.74 |
|
||||
| libvips | 10.72 | 92.12 | 69.06 | 136.43 | 189.79 |
|
||||
| fir rust | 0.87 | 14.52 | 24.96 | 48.83 | 73.53 |
|
||||
| fir sse4.1 | 0.87 | 11.17 | 18.18 | 31.78 | 47.45 |
|
||||
| fir avx2 | 0.87 | 9.32 | 12.96 | 21.67 | 29.82 |
|
||||
|
||||
<!-- bench_compare_rgb32f end -->
|
||||
|
||||
@@ -300,9 +300,9 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|------------|:-------:|:------:|:--------:|:-------:|:--------:|
|
||||
| libvips | 24.44 | 113.81 | 138.06 | 249.74 | 379.76 |
|
||||
| fir rust | 1.01 | 36.20 | 46.05 | 71.26 | 93.95 |
|
||||
| fir sse4.1 | 1.01 | 31.81 | 39.68 | 58.07 | 77.59 |
|
||||
| fir avx2 | 1.01 | 28.64 | 30.94 | 40.70 | 49.71 |
|
||||
| libvips | 20.19 | 153.58 | 125.81 | 213.36 | 312.24 |
|
||||
| fir rust | 1.04 | 35.17 | 44.55 | 68.83 | 92.90 |
|
||||
| fir sse4.1 | 1.04 | 30.67 | 38.72 | 57.04 | 76.49 |
|
||||
| fir avx2 | 1.04 | 28.76 | 29.99 | 39.20 | 49.14 |
|
||||
|
||||
<!-- bench_compare_rgba32f end -->
|
||||
|
||||
@@ -71,5 +71,5 @@ Run benchmarks to compare with other crates for image resizing and write results
|
||||
report files, such as `./benchmarks-x86_64.md`:
|
||||
|
||||
```shell
|
||||
CARGO_TARGET_WASM32_WASI_RUNNER="wasmtime --dir=. --env WRITE_COMPARE_RESULT=1 --" cargo bench -- --color=always Compare
|
||||
CARGO_TARGET_WASM32_WASI_RUNNER="wasmtime --dir=. --env WRITE_COMPARE_RESULT=1 --" cargo bench --no-fail-fast -- --color=always Compare
|
||||
```
|
||||
|
||||
+2
-2
@@ -6,9 +6,9 @@ edition = "2021"
|
||||
|
||||
[dependencies]
|
||||
fast_image_resize = { path = "..", features = ["image"] }
|
||||
image = "0.25.1"
|
||||
image = "0.25.2"
|
||||
clap = { version = "4.5", features = ["derive"] }
|
||||
log = "0.4.21"
|
||||
log = "0.4.22"
|
||||
env_logger = "0.11.3"
|
||||
anyhow = "1.0"
|
||||
clap-verbosity-flag = "2.2"
|
||||
|
||||
@@ -52,6 +52,48 @@ pub(crate) fn div_and_clip16(v: u16, recip_alpha: u64) -> u16 {
|
||||
pub(crate) const RECIP_ALPHA: [u32; 256] = recip_alpha_array(PRECISION);
|
||||
pub(crate) static RECIP_ALPHA16: [u64; 65536] = recip_alpha16_array(PRECISION16);
|
||||
|
||||
macro_rules! process_two_images {
|
||||
{$op: ident($src_view: ident, $dst_view: ident, $($arg: ident),+);} => {
|
||||
#[allow(unused_labels)]
|
||||
'block: {
|
||||
#[cfg(feature = "rayon")]
|
||||
{
|
||||
use crate::threading::split_h_two_images_for_threading;
|
||||
use rayon::prelude::*;
|
||||
|
||||
if let Some(iter) = split_h_two_images_for_threading($src_view, $dst_view, 0) {
|
||||
iter.for_each(|(src, mut dst)| {
|
||||
$op(&src, &mut dst, $($arg),+);
|
||||
});
|
||||
break 'block;
|
||||
}
|
||||
}
|
||||
$op($src_view, $dst_view, $($arg),+);
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
macro_rules! process_one_images {
|
||||
{$op: ident($image_view: ident, $($arg: ident),+);} => {
|
||||
#[allow(unused_labels)]
|
||||
'block: {
|
||||
#[cfg(feature = "rayon")]
|
||||
{
|
||||
use crate::threading::split_h_one_image_for_threading;
|
||||
use rayon::prelude::*;
|
||||
|
||||
if let Some(iter) = split_h_one_image_for_threading($image_view) {
|
||||
iter.for_each(|mut img| {
|
||||
$op(&mut img, $($arg),+);
|
||||
});
|
||||
break 'block;
|
||||
}
|
||||
}
|
||||
$op($image_view, $($arg),+);
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
+75
-41
@@ -10,22 +10,16 @@ mod native;
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod sse4;
|
||||
|
||||
impl AlphaMulDiv for F32x2 {
|
||||
type P = F32x2;
|
||||
|
||||
impl AlphaMulDiv for P {
|
||||
fn multiply_alpha(
|
||||
src_view: &impl ImageView<Pixel = Self>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) -> Result<(), ImageError> {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha(src_view, dst_view) },
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => unsafe { neon::multiply_alpha(src_view, dst_view) },
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha(src_view, dst_view) },
|
||||
_ => native::multiply_alpha(src_view, dst_view),
|
||||
process_two_images! {
|
||||
multiple(src_view, dst_view, cpu_extensions);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -34,16 +28,8 @@ impl AlphaMulDiv for F32x2 {
|
||||
image_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) -> Result<(), ImageError> {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha_inplace(image_view) },
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => unsafe { neon::multiply_alpha_inplace(image_view) },
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha_inplace(image_view) },
|
||||
_ => native::multiply_alpha_inplace(image_view),
|
||||
process_one_images! {
|
||||
multiply_inplace(image_view, cpu_extensions);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -53,16 +39,8 @@ impl AlphaMulDiv for F32x2 {
|
||||
dst_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) -> Result<(), ImageError> {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha(src_view, dst_view) },
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => unsafe { crate::alpha::u16x2::neon::divide_alpha(src_view, dst_view) },
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => unsafe { crate::alpha::u16x2::wasm32::divide_alpha(src_view, dst_view) },
|
||||
_ => native::divide_alpha(src_view, dst_view),
|
||||
process_two_images! {
|
||||
divide(src_view, dst_view, cpu_extensions);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -71,17 +49,73 @@ impl AlphaMulDiv for F32x2 {
|
||||
image_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) -> Result<(), ImageError> {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha_inplace(image_view) },
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => unsafe { crate::alpha::u16x2::neon::divide_alpha_inplace(image_view) },
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => unsafe { crate::alpha::u16x2::wasm32::divide_alpha_inplace(image_view) },
|
||||
_ => native::divide_alpha_inplace(image_view),
|
||||
process_one_images! {
|
||||
divide_inplace(image_view, cpu_extensions);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
fn multiple(
|
||||
src_view: &impl ImageView<Pixel = P>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = P>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha(src_view, dst_view) },
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => unsafe { neon::multiply_alpha(src_view, dst_view) },
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha(src_view, dst_view) },
|
||||
_ => native::multiply_alpha(src_view, dst_view),
|
||||
}
|
||||
}
|
||||
|
||||
fn multiply_inplace(image_view: &mut impl ImageViewMut<Pixel = P>, cpu_extensions: CpuExtensions) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha_inplace(image_view) },
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => unsafe { neon::multiply_alpha_inplace(image_view) },
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha_inplace(image_view) },
|
||||
_ => native::multiply_alpha_inplace(image_view),
|
||||
}
|
||||
}
|
||||
|
||||
fn divide(
|
||||
src_view: &impl ImageView<Pixel = P>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = P>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha(src_view, dst_view) },
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => unsafe { neon::divide_alpha(src_view, dst_view) },
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha(src_view, dst_view) },
|
||||
_ => native::divide_alpha(src_view, dst_view),
|
||||
}
|
||||
}
|
||||
|
||||
fn divide_inplace(image_view: &mut impl ImageViewMut<Pixel = P>, cpu_extensions: CpuExtensions) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha_inplace(image_view) },
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => unsafe { neon::divide_alpha_inplace(image_view) },
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha_inplace(image_view) },
|
||||
_ => native::divide_alpha_inplace(image_view),
|
||||
}
|
||||
}
|
||||
|
||||
+75
-41
@@ -10,22 +10,16 @@ mod native;
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod sse4;
|
||||
|
||||
impl AlphaMulDiv for F32x4 {
|
||||
type P = F32x4;
|
||||
|
||||
impl AlphaMulDiv for P {
|
||||
fn multiply_alpha(
|
||||
src_view: &impl ImageView<Pixel = Self>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) -> Result<(), ImageError> {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha(src_view, dst_view) },
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => unsafe { neon::multiply_alpha(src_view, dst_view) },
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha(src_view, dst_view) },
|
||||
_ => native::multiply_alpha(src_view, dst_view),
|
||||
process_two_images! {
|
||||
multiple(src_view, dst_view, cpu_extensions);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -34,16 +28,8 @@ impl AlphaMulDiv for F32x4 {
|
||||
image_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) -> Result<(), ImageError> {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha_inplace(image_view) },
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => unsafe { neon::multiply_alpha_inplace(image_view) },
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha_inplace(image_view) },
|
||||
_ => native::multiply_alpha_inplace(image_view),
|
||||
process_one_images! {
|
||||
multiply_inplace(image_view, cpu_extensions);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -53,16 +39,8 @@ impl AlphaMulDiv for F32x4 {
|
||||
dst_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) -> Result<(), ImageError> {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha(src_view, dst_view) },
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => unsafe { crate::alpha::u16x2::neon::divide_alpha(src_view, dst_view) },
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => unsafe { crate::alpha::u16x2::wasm32::divide_alpha(src_view, dst_view) },
|
||||
_ => native::divide_alpha(src_view, dst_view),
|
||||
process_two_images! {
|
||||
divide(src_view, dst_view, cpu_extensions);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -71,17 +49,73 @@ impl AlphaMulDiv for F32x4 {
|
||||
image_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) -> Result<(), ImageError> {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha_inplace(image_view) },
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => unsafe { crate::alpha::u16x2::neon::divide_alpha_inplace(image_view) },
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => unsafe { crate::alpha::u16x2::wasm32::divide_alpha_inplace(image_view) },
|
||||
_ => native::divide_alpha_inplace(image_view),
|
||||
process_one_images! {
|
||||
divide_inplace(image_view, cpu_extensions);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
fn multiple(
|
||||
src_view: &impl ImageView<Pixel = P>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = P>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha(src_view, dst_view) },
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => unsafe { neon::multiply_alpha(src_view, dst_view) },
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha(src_view, dst_view) },
|
||||
_ => native::multiply_alpha(src_view, dst_view),
|
||||
}
|
||||
}
|
||||
|
||||
fn multiply_inplace(image_view: &mut impl ImageViewMut<Pixel = P>, cpu_extensions: CpuExtensions) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha_inplace(image_view) },
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => unsafe { neon::multiply_alpha_inplace(image_view) },
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha_inplace(image_view) },
|
||||
_ => native::multiply_alpha_inplace(image_view),
|
||||
}
|
||||
}
|
||||
|
||||
fn divide(
|
||||
src_view: &impl ImageView<Pixel = P>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = P>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha(src_view, dst_view) },
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => unsafe { neon::divide_alpha(src_view, dst_view) },
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha(src_view, dst_view) },
|
||||
_ => native::divide_alpha(src_view, dst_view),
|
||||
}
|
||||
}
|
||||
|
||||
fn divide_inplace(image_view: &mut impl ImageViewMut<Pixel = P>, cpu_extensions: CpuExtensions) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha_inplace(image_view) },
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => unsafe { neon::divide_alpha_inplace(image_view) },
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha_inplace(image_view) },
|
||||
_ => native::divide_alpha_inplace(image_view),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,7 +1,9 @@
|
||||
use crate::{pixels, CpuExtensions, ImageError, ImageView, ImageViewMut};
|
||||
|
||||
#[macro_use]
|
||||
mod common;
|
||||
pub(crate) mod errors;
|
||||
|
||||
mod u8x4;
|
||||
cfg_if::cfg_if! {
|
||||
if #[cfg(not(feature = "only_u8x4"))] {
|
||||
|
||||
+75
-41
@@ -13,22 +13,16 @@ mod sse4;
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
mod wasm32;
|
||||
|
||||
impl AlphaMulDiv for U16x2 {
|
||||
type P = U16x2;
|
||||
|
||||
impl AlphaMulDiv for P {
|
||||
fn multiply_alpha(
|
||||
src_view: &impl ImageView<Pixel = Self>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) -> Result<(), ImageError> {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha(src_view, dst_view) },
|
||||
_ => native::multiply_alpha(src_view, dst_view),
|
||||
process_two_images! {
|
||||
multiple(src_view, dst_view, cpu_extensions);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -37,16 +31,8 @@ impl AlphaMulDiv for U16x2 {
|
||||
image_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) -> Result<(), ImageError> {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha_inplace(image_view) },
|
||||
_ => native::multiply_alpha_inplace(image_view),
|
||||
process_one_images! {
|
||||
multiply_inplace(image_view, cpu_extensions);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -56,16 +42,8 @@ impl AlphaMulDiv for U16x2 {
|
||||
dst_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) -> Result<(), ImageError> {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha(src_view, dst_view) },
|
||||
_ => native::divide_alpha(src_view, dst_view),
|
||||
process_two_images! {
|
||||
divide(src_view, dst_view, cpu_extensions);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -74,17 +52,73 @@ impl AlphaMulDiv for U16x2 {
|
||||
image_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) -> Result<(), ImageError> {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha_inplace(image_view) },
|
||||
_ => native::divide_alpha_inplace(image_view),
|
||||
process_one_images! {
|
||||
divide_inplace(image_view, cpu_extensions);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
fn multiple(
|
||||
src_view: &impl ImageView<Pixel = P>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = P>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha(src_view, dst_view) },
|
||||
_ => native::multiply_alpha(src_view, dst_view),
|
||||
}
|
||||
}
|
||||
|
||||
fn multiply_inplace(image_view: &mut impl ImageViewMut<Pixel = P>, cpu_extensions: CpuExtensions) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha_inplace(image_view) },
|
||||
_ => native::multiply_alpha_inplace(image_view),
|
||||
}
|
||||
}
|
||||
|
||||
fn divide(
|
||||
src_view: &impl ImageView<Pixel = P>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = P>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha(src_view, dst_view) },
|
||||
_ => native::divide_alpha(src_view, dst_view),
|
||||
}
|
||||
}
|
||||
|
||||
fn divide_inplace(image_view: &mut impl ImageViewMut<Pixel = P>, cpu_extensions: CpuExtensions) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha_inplace(image_view) },
|
||||
_ => native::divide_alpha_inplace(image_view),
|
||||
}
|
||||
}
|
||||
|
||||
+75
-41
@@ -13,22 +13,16 @@ mod sse4;
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
mod wasm32;
|
||||
|
||||
impl AlphaMulDiv for U16x4 {
|
||||
type P = U16x4;
|
||||
|
||||
impl AlphaMulDiv for P {
|
||||
fn multiply_alpha(
|
||||
src_view: &impl ImageView<Pixel = Self>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) -> Result<(), ImageError> {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha(src_view, dst_view) },
|
||||
_ => native::multiply_alpha(src_view, dst_view),
|
||||
process_two_images! {
|
||||
multiple(src_view, dst_view, cpu_extensions);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -37,16 +31,8 @@ impl AlphaMulDiv for U16x4 {
|
||||
image_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) -> Result<(), ImageError> {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha_inplace(image_view) },
|
||||
_ => native::multiply_alpha_inplace(image_view),
|
||||
process_one_images! {
|
||||
multiply_inplace(image_view, cpu_extensions);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -56,16 +42,8 @@ impl AlphaMulDiv for U16x4 {
|
||||
dst_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) -> Result<(), ImageError> {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha(src_view, dst_view) },
|
||||
_ => native::divide_alpha(src_view, dst_view),
|
||||
process_two_images! {
|
||||
divide(src_view, dst_view, cpu_extensions);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -74,17 +52,73 @@ impl AlphaMulDiv for U16x4 {
|
||||
image_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) -> Result<(), ImageError> {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha_inplace(image_view) },
|
||||
_ => native::divide_alpha_inplace(image_view),
|
||||
process_one_images! {
|
||||
divide_inplace(image_view, cpu_extensions);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
fn multiple(
|
||||
src_view: &impl ImageView<Pixel = P>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = P>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha(src_view, dst_view) },
|
||||
_ => native::multiply_alpha(src_view, dst_view),
|
||||
}
|
||||
}
|
||||
|
||||
fn multiply_inplace(image_view: &mut impl ImageViewMut<Pixel = P>, cpu_extensions: CpuExtensions) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha_inplace(image_view) },
|
||||
_ => native::multiply_alpha_inplace(image_view),
|
||||
}
|
||||
}
|
||||
|
||||
fn divide(
|
||||
src_view: &impl ImageView<Pixel = P>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = P>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha(src_view, dst_view) },
|
||||
_ => native::divide_alpha(src_view, dst_view),
|
||||
}
|
||||
}
|
||||
|
||||
fn divide_inplace(image_view: &mut impl ImageViewMut<Pixel = P>, cpu_extensions: CpuExtensions) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha_inplace(image_view) },
|
||||
_ => native::divide_alpha_inplace(image_view),
|
||||
}
|
||||
}
|
||||
|
||||
+75
-41
@@ -14,22 +14,16 @@ mod sse4;
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
mod wasm32;
|
||||
|
||||
impl AlphaMulDiv for U8x2 {
|
||||
type P = U8x2;
|
||||
|
||||
impl AlphaMulDiv for P {
|
||||
fn multiply_alpha(
|
||||
src_view: &impl ImageView<Pixel = Self>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) -> Result<(), ImageError> {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha(src_view, dst_view) },
|
||||
_ => native::multiply_alpha(src_view, dst_view),
|
||||
process_two_images! {
|
||||
multiple(src_view, dst_view, cpu_extensions);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -38,16 +32,8 @@ impl AlphaMulDiv for U8x2 {
|
||||
image_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) -> Result<(), ImageError> {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha_inplace(image_view) },
|
||||
_ => native::multiply_alpha_inplace(image_view),
|
||||
process_one_images! {
|
||||
multiply_inplace(image_view, cpu_extensions);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -57,16 +43,8 @@ impl AlphaMulDiv for U8x2 {
|
||||
dst_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) -> Result<(), ImageError> {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha(src_view, dst_view) },
|
||||
_ => native::divide_alpha(src_view, dst_view),
|
||||
process_two_images! {
|
||||
divide(src_view, dst_view, cpu_extensions);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -75,17 +53,73 @@ impl AlphaMulDiv for U8x2 {
|
||||
image_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) -> Result<(), ImageError> {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha_inplace(image_view) },
|
||||
_ => native::divide_alpha_inplace(image_view),
|
||||
process_one_images! {
|
||||
divide_inplace(image_view, cpu_extensions);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
fn multiple(
|
||||
src_view: &impl ImageView<Pixel = P>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = P>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha(src_view, dst_view) },
|
||||
_ => native::multiply_alpha(src_view, dst_view),
|
||||
}
|
||||
}
|
||||
|
||||
fn multiply_inplace(image_view: &mut impl ImageViewMut<Pixel = P>, cpu_extensions: CpuExtensions) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha_inplace(image_view) },
|
||||
_ => native::multiply_alpha_inplace(image_view),
|
||||
}
|
||||
}
|
||||
|
||||
fn divide(
|
||||
src_view: &impl ImageView<Pixel = P>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = P>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha(src_view, dst_view) },
|
||||
_ => native::divide_alpha(src_view, dst_view),
|
||||
}
|
||||
}
|
||||
|
||||
fn divide_inplace(image_view: &mut impl ImageViewMut<Pixel = P>, cpu_extensions: CpuExtensions) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha_inplace(image_view) },
|
||||
_ => native::divide_alpha_inplace(image_view),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,8 +1,7 @@
|
||||
use std::arch::x86_64::*;
|
||||
|
||||
use crate::pixels::U8x4;
|
||||
use crate::utils::foreach_with_pre_reading;
|
||||
use crate::{simd_utils, ImageView, ImageViewMut};
|
||||
use std::arch::x86_64::*;
|
||||
|
||||
use super::sse4;
|
||||
|
||||
@@ -20,7 +19,8 @@ pub(crate) unsafe fn multiply_alpha(
|
||||
|
||||
#[target_feature(enable = "avx2")]
|
||||
pub(crate) unsafe fn multiply_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = U8x4>) {
|
||||
for row in image_view.iter_rows_mut(0) {
|
||||
let rows = image_view.iter_rows_mut(0);
|
||||
for row in rows {
|
||||
multiply_alpha_row_inplace(row);
|
||||
}
|
||||
}
|
||||
@@ -113,15 +113,16 @@ pub(crate) unsafe fn divide_alpha(
|
||||
) {
|
||||
let src_rows = src_view.iter_rows(0);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
let rows = src_rows.zip(dst_rows);
|
||||
for (src_row, dst_row) in rows {
|
||||
divide_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "avx2")]
|
||||
pub(crate) unsafe fn divide_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = U8x4>) {
|
||||
for row in image_view.iter_rows_mut(0) {
|
||||
let rows = image_view.iter_rows_mut(0);
|
||||
for row in rows {
|
||||
divide_alpha_row_inplace(row);
|
||||
}
|
||||
}
|
||||
|
||||
+76
-43
@@ -1,8 +1,7 @@
|
||||
use super::AlphaMulDiv;
|
||||
use crate::pixels::U8x4;
|
||||
use crate::{CpuExtensions, ImageError, ImageView, ImageViewMut};
|
||||
|
||||
use super::AlphaMulDiv;
|
||||
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod avx2;
|
||||
mod native;
|
||||
@@ -13,22 +12,16 @@ mod sse4;
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
mod wasm32;
|
||||
|
||||
impl AlphaMulDiv for U8x4 {
|
||||
type P = U8x4;
|
||||
|
||||
impl AlphaMulDiv for P {
|
||||
fn multiply_alpha(
|
||||
src_view: &impl ImageView<Pixel = Self>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) -> Result<(), ImageError> {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha(src_view, dst_view) },
|
||||
_ => native::multiply_alpha(src_view, dst_view),
|
||||
process_two_images! {
|
||||
multiple(src_view, dst_view, cpu_extensions);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -37,16 +30,8 @@ impl AlphaMulDiv for U8x4 {
|
||||
image_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) -> Result<(), ImageError> {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha_inplace(image_view) },
|
||||
_ => native::multiply_alpha_inplace(image_view),
|
||||
process_one_images! {
|
||||
multiply_inplace(image_view, cpu_extensions);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -56,16 +41,8 @@ impl AlphaMulDiv for U8x4 {
|
||||
dst_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) -> Result<(), ImageError> {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha(src_view, dst_view) },
|
||||
_ => native::divide_alpha(src_view, dst_view),
|
||||
process_two_images! {
|
||||
divide(src_view, dst_view, cpu_extensions);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -74,17 +51,73 @@ impl AlphaMulDiv for U8x4 {
|
||||
image_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) -> Result<(), ImageError> {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha_inplace(image_view) },
|
||||
_ => native::divide_alpha_inplace(image_view),
|
||||
process_one_images! {
|
||||
divide_inplace(image_view, cpu_extensions);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
fn multiple(
|
||||
src_view: &impl ImageView<Pixel = P>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = P>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha(src_view, dst_view) },
|
||||
_ => native::multiply_alpha(src_view, dst_view),
|
||||
}
|
||||
}
|
||||
|
||||
fn multiply_inplace(image_view: &mut impl ImageViewMut<Pixel = P>, cpu_extensions: CpuExtensions) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::multiply_alpha_inplace(image_view) },
|
||||
_ => native::multiply_alpha_inplace(image_view),
|
||||
}
|
||||
}
|
||||
|
||||
fn divide(
|
||||
src_view: &impl ImageView<Pixel = P>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = P>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha(src_view, dst_view) },
|
||||
_ => native::divide_alpha(src_view, dst_view),
|
||||
}
|
||||
}
|
||||
|
||||
fn divide_inplace(image_view: &mut impl ImageViewMut<Pixel = P>, cpu_extensions: CpuExtensions) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => unsafe { neon::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => unsafe { wasm32::divide_alpha_inplace(image_view) },
|
||||
_ => native::divide_alpha_inplace(image_view),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -8,8 +8,9 @@ pub(crate) fn multiply_alpha(
|
||||
) {
|
||||
let src_rows = src_view.iter_rows(0);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
let rows = src_rows.zip(dst_rows);
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
for (src_row, dst_row) in rows {
|
||||
for (src_pixel, dst_pixel) in src_row.iter().zip(dst_row.iter_mut()) {
|
||||
*dst_pixel = multiply_alpha_pixel(*src_pixel);
|
||||
}
|
||||
@@ -17,7 +18,8 @@ pub(crate) fn multiply_alpha(
|
||||
}
|
||||
|
||||
pub(crate) fn multiply_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = U8x4>) {
|
||||
for row in image_view.iter_rows_mut(0) {
|
||||
let rows = image_view.iter_rows_mut(0);
|
||||
for row in rows {
|
||||
multiply_alpha_row_inplace(row);
|
||||
}
|
||||
}
|
||||
@@ -57,15 +59,16 @@ pub(crate) fn divide_alpha(
|
||||
) {
|
||||
let src_rows = src_view.iter_rows(0);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
let rows = src_rows.zip(dst_rows);
|
||||
for (src_row, dst_row) in rows {
|
||||
divide_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn divide_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = U8x4>) {
|
||||
for row in image_view.iter_rows_mut(0) {
|
||||
let rows = image_view.iter_rows_mut(0);
|
||||
for row in rows {
|
||||
row.iter_mut().for_each(|pixel| {
|
||||
*pixel = divide_alpha_pixel(*pixel);
|
||||
});
|
||||
|
||||
@@ -13,15 +13,16 @@ pub(crate) unsafe fn multiply_alpha(
|
||||
) {
|
||||
let src_rows = src_view.iter_rows(0);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
let rows = src_rows.zip(dst_rows);
|
||||
for (src_row, dst_row) in rows {
|
||||
multiply_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
pub(crate) unsafe fn multiply_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = U8x4>) {
|
||||
for row in image_view.iter_rows_mut(0) {
|
||||
let rows = image_view.iter_rows_mut(0);
|
||||
for row in rows {
|
||||
multiply_alpha_row_inplace(row);
|
||||
}
|
||||
}
|
||||
@@ -106,14 +107,16 @@ pub(crate) unsafe fn divide_alpha(
|
||||
) {
|
||||
let src_rows = src_view.iter_rows(0);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
let rows = src_rows.zip(dst_rows);
|
||||
for (src_row, dst_row) in rows {
|
||||
divide_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
pub(crate) unsafe fn divide_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = U8x4>) {
|
||||
for row in image_view.iter_rows_mut(0) {
|
||||
let rows = image_view.iter_rows_mut(0);
|
||||
for row in rows {
|
||||
divide_alpha_row_inplace(row);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -9,7 +9,7 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = F32>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = F32>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
coeffs: &Coefficients,
|
||||
) {
|
||||
let coefficients_chunks = coeffs.get_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
@@ -1,10 +1,9 @@
|
||||
use super::{Coefficients, Convolution};
|
||||
use crate::convolution::vertical_f32::vert_convolution_f32;
|
||||
use crate::cpu_extensions::CpuExtensions;
|
||||
use crate::pixels::F32;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
use super::{Coefficients, Convolution};
|
||||
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod avx2;
|
||||
mod native;
|
||||
@@ -15,7 +14,9 @@ mod sse4;
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// mod wasm32;
|
||||
|
||||
impl Convolution for F32 {
|
||||
type P = F32;
|
||||
|
||||
impl Convolution for P {
|
||||
fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = Self>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
@@ -23,16 +24,17 @@ impl Convolution for F32 {
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
_ => native::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
debug_assert!(src_view.height() - offset >= dst_view.height());
|
||||
let coeffs_ref = &coeffs;
|
||||
|
||||
try_process_in_threads_h! {
|
||||
horiz_convolution(
|
||||
src_view,
|
||||
dst_view,
|
||||
offset,
|
||||
coeffs_ref,
|
||||
cpu_extensions,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -43,6 +45,48 @@ impl Convolution for F32 {
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
vert_convolution_f32(src_view, dst_view, offset, coeffs, cpu_extensions);
|
||||
debug_assert!(src_view.width() - offset >= dst_view.width());
|
||||
|
||||
let coeffs_ref = &coeffs;
|
||||
|
||||
try_process_in_threads_v! {
|
||||
vert_convolution(
|
||||
src_view,
|
||||
dst_view,
|
||||
offset,
|
||||
coeffs_ref,
|
||||
cpu_extensions,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = P>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = P>,
|
||||
offset: u32,
|
||||
coeffs: &Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
_ => native::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
}
|
||||
}
|
||||
|
||||
fn vert_convolution(
|
||||
src_view: &impl ImageView<Pixel = P>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = P>,
|
||||
offset: u32,
|
||||
coeffs: &Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
vert_convolution_f32(src_view, dst_view, offset, coeffs, cpu_extensions);
|
||||
}
|
||||
|
||||
@@ -6,7 +6,7 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = F32>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = F32>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
coeffs: &Coefficients,
|
||||
) {
|
||||
let coefficients_chunks = coeffs.get_chunks();
|
||||
let src_rows = src_view.iter_rows(offset);
|
||||
|
||||
@@ -9,7 +9,7 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = F32>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = F32>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
coeffs: &Coefficients,
|
||||
) {
|
||||
let coefficients_chunks = coeffs.get_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
@@ -9,7 +9,7 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = F32x2>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = F32x2>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
coeffs: &Coefficients,
|
||||
) {
|
||||
let coefficients_chunks = coeffs.get_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
@@ -15,7 +15,9 @@ mod sse4;
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// mod wasm32;
|
||||
|
||||
impl Convolution for F32x2 {
|
||||
type P = F32x2;
|
||||
|
||||
impl Convolution for P {
|
||||
fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = Self>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
@@ -23,16 +25,17 @@ impl Convolution for F32x2 {
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
_ => native::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
debug_assert!(src_view.height() - offset >= dst_view.height());
|
||||
let coeffs_ref = &coeffs;
|
||||
|
||||
try_process_in_threads_h! {
|
||||
horiz_convolution(
|
||||
src_view,
|
||||
dst_view,
|
||||
offset,
|
||||
coeffs_ref,
|
||||
cpu_extensions,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -43,6 +46,48 @@ impl Convolution for F32x2 {
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
vert_convolution_f32(src_view, dst_view, offset, coeffs, cpu_extensions);
|
||||
debug_assert!(src_view.width() - offset >= dst_view.width());
|
||||
|
||||
let coeffs_ref = &coeffs;
|
||||
|
||||
try_process_in_threads_v! {
|
||||
vert_convolution(
|
||||
src_view,
|
||||
dst_view,
|
||||
offset,
|
||||
coeffs_ref,
|
||||
cpu_extensions,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = P>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = P>,
|
||||
offset: u32,
|
||||
coeffs: &Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
_ => native::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
}
|
||||
}
|
||||
|
||||
fn vert_convolution(
|
||||
src_view: &impl ImageView<Pixel = P>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = P>,
|
||||
offset: u32,
|
||||
coeffs: &Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
vert_convolution_f32(src_view, dst_view, offset, coeffs, cpu_extensions);
|
||||
}
|
||||
|
||||
@@ -6,7 +6,7 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = F32x2>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = F32x2>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
coeffs: &Coefficients,
|
||||
) {
|
||||
let coefficients_chunks = coeffs.get_chunks();
|
||||
let src_rows = src_view.iter_rows(offset);
|
||||
|
||||
@@ -9,7 +9,7 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = F32x2>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = F32x2>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
coeffs: &Coefficients,
|
||||
) {
|
||||
let coefficients_chunks = coeffs.get_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
@@ -9,7 +9,7 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = F32x3>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = F32x3>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
coeffs: &Coefficients,
|
||||
) {
|
||||
let coefficients_chunks = coeffs.get_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
@@ -15,7 +15,9 @@ mod sse4;
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// mod wasm32;
|
||||
|
||||
impl Convolution for F32x3 {
|
||||
type P = F32x3;
|
||||
|
||||
impl Convolution for P {
|
||||
fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = Self>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
@@ -23,16 +25,17 @@ impl Convolution for F32x3 {
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
_ => native::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
debug_assert!(src_view.height() - offset >= dst_view.height());
|
||||
let coeffs_ref = &coeffs;
|
||||
|
||||
try_process_in_threads_h! {
|
||||
horiz_convolution(
|
||||
src_view,
|
||||
dst_view,
|
||||
offset,
|
||||
coeffs_ref,
|
||||
cpu_extensions,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -43,6 +46,48 @@ impl Convolution for F32x3 {
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
vert_convolution_f32(src_view, dst_view, offset, coeffs, cpu_extensions);
|
||||
debug_assert!(src_view.width() - offset >= dst_view.width());
|
||||
|
||||
let coeffs_ref = &coeffs;
|
||||
|
||||
try_process_in_threads_v! {
|
||||
vert_convolution(
|
||||
src_view,
|
||||
dst_view,
|
||||
offset,
|
||||
coeffs_ref,
|
||||
cpu_extensions,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = P>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = P>,
|
||||
offset: u32,
|
||||
coeffs: &Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
_ => native::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
}
|
||||
}
|
||||
|
||||
fn vert_convolution(
|
||||
src_view: &impl ImageView<Pixel = P>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = P>,
|
||||
offset: u32,
|
||||
coeffs: &Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
vert_convolution_f32(src_view, dst_view, offset, coeffs, cpu_extensions);
|
||||
}
|
||||
|
||||
@@ -6,7 +6,7 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = F32x3>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = F32x3>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
coeffs: &Coefficients,
|
||||
) {
|
||||
let coefficients_chunks = coeffs.get_chunks();
|
||||
let src_rows = src_view.iter_rows(offset);
|
||||
|
||||
@@ -9,7 +9,7 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = F32x3>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = F32x3>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
coeffs: &Coefficients,
|
||||
) {
|
||||
let coefficients_chunks = coeffs.get_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
@@ -9,7 +9,7 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = F32x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = F32x4>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
coeffs: &Coefficients,
|
||||
) {
|
||||
let coefficients_chunks = coeffs.get_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
@@ -15,7 +15,9 @@ mod sse4;
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// mod wasm32;
|
||||
|
||||
impl Convolution for F32x4 {
|
||||
type P = F32x4;
|
||||
|
||||
impl Convolution for P {
|
||||
fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = Self>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
@@ -23,16 +25,17 @@ impl Convolution for F32x4 {
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
_ => native::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
debug_assert!(src_view.height() - offset >= dst_view.height());
|
||||
let coeffs_ref = &coeffs;
|
||||
|
||||
try_process_in_threads_h! {
|
||||
horiz_convolution(
|
||||
src_view,
|
||||
dst_view,
|
||||
offset,
|
||||
coeffs_ref,
|
||||
cpu_extensions,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -43,6 +46,48 @@ impl Convolution for F32x4 {
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
vert_convolution_f32(src_view, dst_view, offset, coeffs, cpu_extensions);
|
||||
debug_assert!(src_view.width() - offset >= dst_view.width());
|
||||
|
||||
let coeffs_ref = &coeffs;
|
||||
|
||||
try_process_in_threads_v! {
|
||||
vert_convolution(
|
||||
src_view,
|
||||
dst_view,
|
||||
offset,
|
||||
coeffs_ref,
|
||||
cpu_extensions,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = P>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = P>,
|
||||
offset: u32,
|
||||
coeffs: &Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
_ => native::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
}
|
||||
}
|
||||
|
||||
fn vert_convolution(
|
||||
src_view: &impl ImageView<Pixel = P>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = P>,
|
||||
offset: u32,
|
||||
coeffs: &Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
vert_convolution_f32(src_view, dst_view, offset, coeffs, cpu_extensions);
|
||||
}
|
||||
|
||||
@@ -6,7 +6,7 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = F32x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = F32x4>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
coeffs: &Coefficients,
|
||||
) {
|
||||
let coefficients_chunks = coeffs.get_chunks();
|
||||
let src_rows = src_view.iter_rows(offset);
|
||||
|
||||
@@ -9,7 +9,7 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = F32x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = F32x4>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
coeffs: &Coefficients,
|
||||
) {
|
||||
let coefficients_chunks = coeffs.get_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
@@ -1,11 +1,12 @@
|
||||
use super::{Coefficients, Convolution};
|
||||
use crate::pixels::I32;
|
||||
use crate::{CpuExtensions, ImageView, ImageViewMut};
|
||||
|
||||
use super::{Coefficients, Convolution};
|
||||
|
||||
mod native;
|
||||
|
||||
impl Convolution for I32 {
|
||||
type P = I32;
|
||||
|
||||
impl Convolution for P {
|
||||
fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = Self>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
@@ -13,7 +14,17 @@ impl Convolution for I32 {
|
||||
coeffs: Coefficients,
|
||||
_cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
native::horiz_convolution(src_view, dst_view, offset, coeffs);
|
||||
debug_assert!(src_view.height() - offset >= dst_view.height());
|
||||
let coeffs_ref = &coeffs;
|
||||
|
||||
try_process_in_threads_h! {
|
||||
horiz_convolution(
|
||||
src_view,
|
||||
dst_view,
|
||||
offset,
|
||||
coeffs_ref,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
fn vert_convolution(
|
||||
@@ -23,6 +34,37 @@ impl Convolution for I32 {
|
||||
coeffs: Coefficients,
|
||||
_cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
native::vert_convolution(src_view, dst_view, offset, coeffs);
|
||||
debug_assert!(src_view.width() - offset >= dst_view.width());
|
||||
|
||||
let coeffs_ref = &coeffs;
|
||||
|
||||
try_process_in_threads_v! {
|
||||
vert_convolution(
|
||||
src_view,
|
||||
dst_view,
|
||||
offset,
|
||||
coeffs_ref,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = P>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = P>,
|
||||
offset: u32,
|
||||
coefficients: &Coefficients,
|
||||
) {
|
||||
native::horiz_convolution(src_view, dst_view, offset, coefficients);
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
fn vert_convolution(
|
||||
src_view: &impl ImageView<Pixel = P>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = P>,
|
||||
offset: u32,
|
||||
coefficients: &Coefficients,
|
||||
) {
|
||||
native::vert_convolution(src_view, dst_view, offset, coefficients);
|
||||
}
|
||||
|
||||
@@ -6,7 +6,7 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = I32>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = I32>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
coeffs: &Coefficients,
|
||||
) {
|
||||
let coefficients_chunks = coeffs.get_chunks();
|
||||
let src_rows = src_view.iter_rows(offset);
|
||||
@@ -28,7 +28,7 @@ pub(crate) fn vert_convolution(
|
||||
src_view: &impl ImageView<Pixel = I32>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = I32>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
coeffs: &Coefficients,
|
||||
) {
|
||||
let coefficients_chunks = coeffs.get_chunks();
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
|
||||
@@ -7,6 +7,7 @@ use crate::{CpuExtensions, ImageView, ImageViewMut};
|
||||
mod macros;
|
||||
|
||||
mod filters;
|
||||
#[macro_use]
|
||||
mod optimisations;
|
||||
mod u8x4;
|
||||
mod vertical_u8;
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
use crate::convolution::Coefficients;
|
||||
|
||||
// This code is based on C-implementation from Pillow-SIMD package for Python
|
||||
// https://github.com/uploadcare/pillow-simd
|
||||
|
||||
@@ -28,12 +27,6 @@ const PRECISION_BITS: u8 = 32 - 8 - 2;
|
||||
// We use i16 type to store coefficients.
|
||||
const MAX_COEFFS_PRECISION: u8 = 16 - 1;
|
||||
|
||||
/// Converts `Vec<f64>` into `Vec<i16>`.
|
||||
pub(crate) struct Normalizer16 {
|
||||
precision: u8,
|
||||
chunks: Vec<CoefficientsI16Chunk>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
pub(crate) struct CoefficientsI16Chunk {
|
||||
pub start: u32,
|
||||
@@ -47,6 +40,11 @@ impl CoefficientsI16Chunk {
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) struct Normalizer16 {
|
||||
precision: u8,
|
||||
chunks: Vec<CoefficientsI16Chunk>,
|
||||
}
|
||||
|
||||
impl Normalizer16 {
|
||||
#[inline]
|
||||
pub fn new(coefficients: Coefficients) -> Self {
|
||||
@@ -89,13 +87,17 @@ impl Normalizer16 {
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub fn coefficients(&self) -> &[CoefficientsI16Chunk] {
|
||||
pub fn precision(&self) -> u8 {
|
||||
self.precision
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub fn chunks(&self) -> &[CoefficientsI16Chunk] {
|
||||
&self.chunks
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn precision(&self) -> u8 {
|
||||
self.precision
|
||||
pub fn chunks_len(&self) -> usize {
|
||||
self.chunks.len()
|
||||
}
|
||||
|
||||
/// # Safety
|
||||
@@ -122,7 +124,7 @@ const MAX_COEFFS_PRECISION16: u8 = 32 - 1;
|
||||
#[derive(Debug, Clone)]
|
||||
pub(crate) struct CoefficientsI32Chunk {
|
||||
pub start: u32,
|
||||
pub values: Vec<i32>,
|
||||
values: Vec<i32>,
|
||||
}
|
||||
|
||||
impl CoefficientsI32Chunk {
|
||||
@@ -179,31 +181,78 @@ impl Normalizer32 {
|
||||
Self { precision, chunks }
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub fn coefficients(&self) -> &[CoefficientsI32Chunk] {
|
||||
&self.chunks
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn precision(&self) -> u8 {
|
||||
self.precision
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub fn chunks(&self) -> &[CoefficientsI32Chunk] {
|
||||
&self.chunks
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub fn chunks_len(&self) -> usize {
|
||||
self.chunks.len()
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub fn clip(&self, v: i64) -> u16 {
|
||||
(v >> self.precision).min(u16::MAX as i64).max(0) as u16
|
||||
}
|
||||
}
|
||||
|
||||
macro_rules! try_process_in_threads_h {
|
||||
{$op: ident($src_view: ident, $dst_view: ident, $offset: ident, $($arg: ident),+$(,)?);} => {
|
||||
#[allow(unused_labels)]
|
||||
'block: {
|
||||
#[cfg(feature = "rayon")]
|
||||
{
|
||||
use crate::threading::split_h_two_images_for_threading;
|
||||
use rayon::prelude::*;
|
||||
|
||||
if let Some(iter) = split_h_two_images_for_threading($src_view, $dst_view, $offset) {
|
||||
iter.for_each(|(src, mut dst)| {
|
||||
$op(&src, &mut dst, 0, $($arg),+);
|
||||
});
|
||||
break 'block;
|
||||
}
|
||||
}
|
||||
$op($src_view, $dst_view, $offset, $($arg),+);
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
macro_rules! try_process_in_threads_v {
|
||||
{$op: ident($src_view: ident, $dst_view: ident, $offset: ident, $($arg: ident),+$(,)?);} => {
|
||||
#[allow(unused_labels)]
|
||||
'block: {
|
||||
#[cfg(feature = "rayon")]
|
||||
{
|
||||
use crate::threading::split_v_two_images_for_threading;
|
||||
use rayon::prelude::*;
|
||||
|
||||
if let Some(iter) = split_v_two_images_for_threading($src_view, $dst_view, $offset) {
|
||||
iter.for_each(|(src, mut dst)| {
|
||||
$op(&src, &mut dst, 0, $($arg),+);
|
||||
});
|
||||
break 'block;
|
||||
}
|
||||
}
|
||||
$op($src_view, $dst_view, $offset, $($arg),+);
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
use crate::convolution::Bound;
|
||||
fn get_coefficients(value: f64) -> Coefficients {
|
||||
Coefficients {
|
||||
values: vec![value],
|
||||
window_size: 0,
|
||||
bounds: vec![],
|
||||
window_size: 1,
|
||||
bounds: vec![Bound { start: 0, size: 1 }],
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use std::arch::x86_64::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer32;
|
||||
use crate::pixels::U16;
|
||||
use crate::{simd_utils, ImageView, ImageViewMut};
|
||||
|
||||
@@ -9,16 +9,15 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U16>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -27,7 +26,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &normalizer);
|
||||
horiz_convolution_one_row(src_row, dst_row, normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -41,12 +40,11 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U16]; 4],
|
||||
dst_rows: [&mut [U16]; 4],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 4];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|L0 | |L1 | |L2 | |L3 | |L4 | |L5 | |L6 | |L7 |
|
||||
@@ -86,11 +84,11 @@ unsafe fn horiz_convolution_four_rows(
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
for (dst_x, chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x = chunk.start as usize;
|
||||
let mut ll_sum = [_mm256_set1_epi64x(0); 2];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
let mut coeffs = chunk.values();
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
coeffs = coeffs_by_8.remainder();
|
||||
@@ -200,12 +198,11 @@ unsafe fn horiz_convolution_four_rows(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16],
|
||||
dst_row: &mut [U16],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 4];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|L0 | |L1 | |L2 | |L3 | |L4 | |L5 | |L6 | |L7 |
|
||||
@@ -245,10 +242,10 @@ unsafe fn horiz_convolution_one_row(
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
for (dst_x, chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x = chunk.start as usize;
|
||||
let mut ll_sum = _mm256_set1_epi64x(0);
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
let mut coeffs = chunk.values();
|
||||
|
||||
let coeffs_by_16 = coeffs.chunks_exact(16);
|
||||
coeffs = coeffs_by_16.remainder();
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
use super::{Coefficients, Convolution};
|
||||
use crate::convolution::optimisations::Normalizer32;
|
||||
use crate::convolution::vertical_u16::vert_convolution_u16;
|
||||
use crate::pixels::U16;
|
||||
use crate::{CpuExtensions, ImageView, ImageViewMut};
|
||||
|
||||
use super::{Coefficients, Convolution};
|
||||
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod avx2;
|
||||
mod native;
|
||||
@@ -14,7 +14,9 @@ mod sse4;
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
mod wasm32;
|
||||
|
||||
impl Convolution for U16 {
|
||||
type P = U16;
|
||||
|
||||
impl Convolution for P {
|
||||
fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = Self>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
@@ -22,16 +24,19 @@ impl Convolution for U16 {
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
_ => native::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
debug_assert!(src_view.height() - offset >= dst_view.height());
|
||||
|
||||
let normalizer = Normalizer32::new(coeffs);
|
||||
let normalizer_ref = &normalizer;
|
||||
|
||||
try_process_in_threads_h! {
|
||||
horiz_convolution(
|
||||
src_view,
|
||||
dst_view,
|
||||
offset,
|
||||
normalizer_ref,
|
||||
cpu_extensions,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -42,6 +47,39 @@ impl Convolution for U16 {
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
vert_convolution_u16(src_view, dst_view, offset, coeffs, cpu_extensions);
|
||||
debug_assert!(src_view.width() - offset >= dst_view.width());
|
||||
|
||||
let normalizer = Normalizer32::new(coeffs);
|
||||
let normalizer_ref = &normalizer;
|
||||
|
||||
try_process_in_threads_v! {
|
||||
vert_convolution_u16(
|
||||
src_view,
|
||||
dst_view,
|
||||
offset,
|
||||
normalizer_ref,
|
||||
cpu_extensions,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = P>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = P>,
|
||||
offset: u32,
|
||||
normalizer: &Normalizer32,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
_ => native::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer32;
|
||||
use crate::pixels::U16;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
@@ -7,11 +7,10 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U16>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
let coefficients_chunks = normalizer.chunks();
|
||||
let initial = 1i64 << (precision - 1);
|
||||
|
||||
let src_rows = src_view.iter_rows(offset);
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
use std::arch::aarch64::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::{CoefficientsI32Chunk, Normalizer32};
|
||||
|
||||
use crate::neon_utils;
|
||||
use crate::pixels::U16;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
@@ -10,9 +11,8 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U16>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
|
||||
macro_rules! call {
|
||||
@@ -27,16 +27,16 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
src_view: &impl ImageView<Pixel = U16>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16>,
|
||||
offset: u32,
|
||||
normalizer: optimisations::Normalizer32,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let coefficients_chunks = normalizer.chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &coefficients_chunks);
|
||||
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, coefficients_chunks);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -45,7 +45,7 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &coefficients_chunks);
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, coefficients_chunks);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -60,7 +60,7 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
src_rows: [&[U16]; 4],
|
||||
dst_rows: [&mut [U16]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
coefficients_chunks: &[CoefficientsI32Chunk],
|
||||
) {
|
||||
let initial = vdupq_n_s64(1i64 << (PRECISION - 2));
|
||||
let zero_u16x4 = vdup_n_u16(0);
|
||||
@@ -68,7 +68,7 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut sss_a = [initial; 4];
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
coeffs = coeffs_by_4.remainder();
|
||||
@@ -140,16 +140,16 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
src_row: &[U16],
|
||||
dst_row: &mut [U16],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
coefficients_chunks: &[CoefficientsI32Chunk],
|
||||
) {
|
||||
let initial = vdupq_n_s64(1i64 << (PRECISION - 2));
|
||||
let zero_u16x8 = vdupq_n_u16(0);
|
||||
let zero_u16x4 = vdup_n_u16(0);
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut sss = initial;
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
coeffs = coeffs_by_8.remainder();
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use std::arch::x86_64::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer32;
|
||||
use crate::pixels::U16;
|
||||
use crate::{simd_utils, ImageView, ImageViewMut};
|
||||
|
||||
@@ -9,16 +9,15 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U16>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -27,7 +26,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &normalizer);
|
||||
horiz_convolution_one_row(src_row, dst_row, normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -41,12 +40,11 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U16]; 4],
|
||||
dst_rows: [&mut [U16]; 4],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 2];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|L0 | |L1 | |L2 | |L3 | |L4 | |L5 | |L6 | |L7 |
|
||||
@@ -72,11 +70,11 @@ unsafe fn horiz_convolution_four_rows(
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
for (dst_x, chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x = chunk.start as usize;
|
||||
let mut ll_sum = [_mm_set1_epi64x(0); 4];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
let mut coeffs = chunk.values();
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
coeffs = coeffs_by_8.remainder();
|
||||
@@ -168,12 +166,11 @@ unsafe fn horiz_convolution_four_rows(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16],
|
||||
dst_row: &mut [U16],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 2];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|L0 | |L1 | |L2 | |L3 | |L4 | |L5 | |L6 | |L7 |
|
||||
@@ -199,10 +196,10 @@ unsafe fn horiz_convolution_one_row(
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
for (dst_x, chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x = chunk.start as usize;
|
||||
let mut ll_sum = _mm_set1_epi64x(0);
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
let mut coeffs = chunk.values();
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
coeffs = coeffs_by_8.remainder();
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use std::arch::wasm32::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer32;
|
||||
use crate::pixels::U16;
|
||||
use crate::wasm32_utils;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
@@ -10,17 +10,15 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U16>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -29,7 +27,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_one_row(src_row, dst_row, normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -43,8 +41,7 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U16]; 4],
|
||||
dst_rows: [&mut [U16]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
@@ -74,11 +71,11 @@ unsafe fn horiz_convolution_four_rows(
|
||||
12, 13, -1, -1, -1, -1, -1, -1, 14, 15, -1, -1, -1, -1, -1, -1,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
for (dst_x, coeffs_chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut ll_sum: [v128; 4] = [i64x2_splat(0i64); 4];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
coeffs = coeffs_by_8.remainder();
|
||||
@@ -173,8 +170,7 @@ unsafe fn horiz_convolution_four_rows(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16],
|
||||
dst_row: &mut [U16],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
@@ -204,10 +200,10 @@ unsafe fn horiz_convolution_one_row(
|
||||
12, 13, -1, -1, -1, -1, -1, -1, 14, 15, -1, -1, -1, -1, -1, -1,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
for (dst_x, coeffs_chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut ll_sum = i64x2_splat(0);
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
coeffs = coeffs_by_8.remainder();
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use std::arch::x86_64::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer32;
|
||||
use crate::pixels::U16x2;
|
||||
use crate::{simd_utils, ImageView, ImageViewMut};
|
||||
|
||||
@@ -9,16 +9,15 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U16x2>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16x2>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -27,7 +26,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &normalizer);
|
||||
horiz_convolution_one_row(src_row, dst_row, normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -41,12 +40,11 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U16x2]; 4],
|
||||
dst_rows: [&mut [U16x2]; 4],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 4];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|L0 A0 | |L1 A1 | |L2 A2 | |L3 A3 |
|
||||
@@ -86,11 +84,11 @@ unsafe fn horiz_convolution_four_rows(
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
for (dst_x, chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x = chunk.start as usize;
|
||||
let mut ll_sum = [_mm256_set1_epi64x(half_error); 2];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
let mut coeffs = chunk.values();
|
||||
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
coeffs = coeffs_by_4.remainder();
|
||||
@@ -177,12 +175,11 @@ unsafe fn horiz_convolution_four_rows(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16x2],
|
||||
dst_row: &mut [U16x2],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 4];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|L0 A0 | |L1 A1 | |L2 A2 | |L3 A3 |
|
||||
@@ -222,10 +219,10 @@ unsafe fn horiz_convolution_one_row(
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
for (dst_x, chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x = chunk.start as usize;
|
||||
let mut ll_sum = _mm256_setzero_si256();
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
let mut coeffs = chunk.values();
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
coeffs = coeffs_by_8.remainder();
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
use super::{Coefficients, Convolution};
|
||||
use crate::convolution::optimisations::Normalizer32;
|
||||
use crate::convolution::vertical_u16::vert_convolution_u16;
|
||||
use crate::pixels::U16x2;
|
||||
use crate::{CpuExtensions, ImageView, ImageViewMut};
|
||||
|
||||
use super::{Coefficients, Convolution};
|
||||
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod avx2;
|
||||
mod native;
|
||||
@@ -14,7 +14,9 @@ mod sse4;
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
mod wasm32;
|
||||
|
||||
impl Convolution for U16x2 {
|
||||
type P = U16x2;
|
||||
|
||||
impl Convolution for P {
|
||||
fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = Self>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
@@ -22,16 +24,19 @@ impl Convolution for U16x2 {
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
_ => native::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
debug_assert!(src_view.height() - offset >= dst_view.height());
|
||||
|
||||
let normalizer = Normalizer32::new(coeffs);
|
||||
let normalizer_ref = &normalizer;
|
||||
|
||||
try_process_in_threads_h! {
|
||||
horiz_convolution(
|
||||
src_view,
|
||||
dst_view,
|
||||
offset,
|
||||
normalizer_ref,
|
||||
cpu_extensions,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -42,6 +47,39 @@ impl Convolution for U16x2 {
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
vert_convolution_u16(src_view, dst_view, offset, coeffs, cpu_extensions);
|
||||
debug_assert!(src_view.width() - offset >= dst_view.width());
|
||||
|
||||
let normalizer = Normalizer32::new(coeffs);
|
||||
let normalizer_ref = &normalizer;
|
||||
|
||||
try_process_in_threads_v! {
|
||||
vert_convolution_u16(
|
||||
src_view,
|
||||
dst_view,
|
||||
offset,
|
||||
normalizer_ref,
|
||||
cpu_extensions,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = P>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = P>,
|
||||
offset: u32,
|
||||
normalizer: &Normalizer32,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
_ => native::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer32;
|
||||
use crate::pixels::U16x2;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
@@ -7,11 +7,10 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U16x2>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16x2>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
let coefficients_chunks = normalizer.chunks();
|
||||
let initial: i64 = 1 << (precision - 1);
|
||||
|
||||
let src_rows = src_view.iter_rows(offset);
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use std::arch::aarch64::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::{CoefficientsI32Chunk, Normalizer32};
|
||||
use crate::neon_utils;
|
||||
use crate::pixels::U16x2;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
@@ -10,9 +10,8 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U16x2>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16x2>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
|
||||
macro_rules! call {
|
||||
@@ -27,16 +26,16 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
src_view: &impl ImageView<Pixel = U16x2>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16x2>,
|
||||
offset: u32,
|
||||
normalizer: optimisations::Normalizer32,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let coefficients_chunks = normalizer.chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &coefficients_chunks);
|
||||
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, coefficients_chunks);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -45,7 +44,7 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &coefficients_chunks);
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, coefficients_chunks);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -60,7 +59,7 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
src_rows: [&[U16x2]; 4],
|
||||
dst_rows: [&mut [U16x2]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
coefficients_chunks: &[CoefficientsI32Chunk],
|
||||
) {
|
||||
let initial = vdupq_n_s64(1i64 << (PRECISION - 1));
|
||||
// let zero_u16x8 = vdupq_n_u16(0);
|
||||
@@ -69,7 +68,7 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut sss_a = [initial; 4];
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_2 = coeffs.chunks_exact(2);
|
||||
coeffs = coeffs_by_2.remainder();
|
||||
@@ -127,16 +126,16 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
src_row: &[U16x2],
|
||||
dst_row: &mut [U16x2],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
coefficients_chunks: &[CoefficientsI32Chunk],
|
||||
) {
|
||||
let initial = vdupq_n_s64(1i64 << (PRECISION - 1));
|
||||
let zero_u16x8 = vdupq_n_u16(0);
|
||||
let zero_u16x4 = vdup_n_u16(0);
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut sss = initial;
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
coeffs = coeffs_by_4.remainder();
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use std::arch::x86_64::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer32;
|
||||
use crate::pixels::U16x2;
|
||||
use crate::{simd_utils, ImageView, ImageViewMut};
|
||||
|
||||
@@ -9,16 +9,15 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U16x2>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16x2>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -27,7 +26,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &normalizer);
|
||||
horiz_convolution_one_row(src_row, dst_row, normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -42,12 +41,11 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U16x2]; 4],
|
||||
dst_rows: [&mut [U16x2]; 4],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 2];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|L0 A0 | |L1 A1 | |L2 A2 | |L3 A3 |
|
||||
@@ -73,11 +71,11 @@ unsafe fn horiz_convolution_four_rows(
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
for (dst_x, chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x = chunk.start as usize;
|
||||
let mut ll_sum = [_mm_set1_epi64x(half_error); 4];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
let mut coeffs = chunk.values();
|
||||
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
coeffs = coeffs_by_4.remainder();
|
||||
@@ -158,12 +156,11 @@ unsafe fn horiz_convolution_four_rows(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16x2],
|
||||
dst_row: &mut [U16x2],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 2];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|L0 A0 | |L1 A1 | |L2 A2 | |L3 A3 |
|
||||
@@ -189,10 +186,10 @@ unsafe fn horiz_convolution_one_row(
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
for (dst_x, chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x = chunk.start as usize;
|
||||
let mut ll_sum = _mm_set1_epi64x(half_error);
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
let mut coeffs = chunk.values();
|
||||
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
coeffs = coeffs_by_4.remainder();
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use std::arch::wasm32::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer32;
|
||||
use crate::pixels::U16x2;
|
||||
use crate::wasm32_utils;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
@@ -10,17 +10,15 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U16x2>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16x2>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -29,7 +27,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_one_row(src_row, dst_row, normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -44,8 +42,7 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U16x2]; 4],
|
||||
dst_rows: [&mut [U16x2]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
@@ -75,11 +72,11 @@ unsafe fn horiz_convolution_four_rows(
|
||||
12, 13, -1, -1, -1, -1, -1, -1, 14, 15, -1, -1, -1, -1, -1, -1,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
for (dst_x, coeffs_chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut ll_sum = [i64x2_splat(half_error); 4];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
coeffs = coeffs_by_4.remainder();
|
||||
@@ -160,8 +157,7 @@ unsafe fn horiz_convolution_four_rows(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16x2],
|
||||
dst_row: &mut [U16x2],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
@@ -191,10 +187,10 @@ unsafe fn horiz_convolution_one_row(
|
||||
12, 13, -1, -1, -1, -1, -1, -1, 14, 15, -1, -1, -1, -1, -1, -1,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
for (dst_x, coeffs_chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut ll_sum = i64x2_splat(half_error);
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
coeffs = coeffs_by_4.remainder();
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use std::arch::x86_64::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer32;
|
||||
use crate::pixels::U16x3;
|
||||
use crate::{simd_utils, ImageView, ImageViewMut};
|
||||
|
||||
@@ -9,16 +9,15 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U16x3>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16x3>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -27,7 +26,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &normalizer);
|
||||
horiz_convolution_one_row(src_row, dst_row, normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -41,15 +40,13 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U16x3]; 4],
|
||||
dst_rows: [&mut [U16x3]; 4],
|
||||
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut rg_buf = [0i64; 4];
|
||||
let mut rg_bb_buf = [0i64; 4];
|
||||
let mut bbb_buf = [0i64; 4];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|R G B | |R G B | |R G | - |B | |R G B | |R G B | |R |
|
||||
@@ -95,13 +92,13 @@ unsafe fn horiz_convolution_four_rows(
|
||||
|
||||
let width = src_rows[0].len();
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
for (dst_x, chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x = chunk.start as usize;
|
||||
let mut rg_sum = [_mm256_set1_epi8(0); 4];
|
||||
let mut rg_bb_sum = [_mm256_set1_epi8(0); 4];
|
||||
let mut bbb_sum = [_mm256_set1_epi8(0); 4];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
let mut coeffs = chunk.values();
|
||||
let end_x = x + coeffs.len();
|
||||
|
||||
if width - end_x >= 1 {
|
||||
@@ -177,14 +174,13 @@ unsafe fn horiz_convolution_four_rows(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16x3],
|
||||
dst_row: &mut [U16x3],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut rg_buf = [0i64; 4];
|
||||
let mut rg_bb_buf = [0i64; 4];
|
||||
let mut bbb_buf = [0i64; 4];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|R G B | |R G B | |R G | - |B | |R G B | |R G B | |R |
|
||||
@@ -232,13 +228,13 @@ unsafe fn horiz_convolution_one_row(
|
||||
|
||||
let width = src_row.len();
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
for (dst_x, chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x = chunk.start as usize;
|
||||
let mut rg_sum = zero_i64x4;
|
||||
let mut rg_bb_sum = zero_i64x4;
|
||||
let mut bbb_sum = zero_i64x4;
|
||||
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
let mut coeffs = chunk.values();
|
||||
let end_x = x + coeffs.len();
|
||||
|
||||
if width - end_x >= 1 {
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
use super::{Coefficients, Convolution};
|
||||
use crate::convolution::optimisations::Normalizer32;
|
||||
use crate::convolution::vertical_u16::vert_convolution_u16;
|
||||
use crate::pixels::U16x3;
|
||||
use crate::{CpuExtensions, ImageView, ImageViewMut};
|
||||
|
||||
use super::{Coefficients, Convolution};
|
||||
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod avx2;
|
||||
mod native;
|
||||
@@ -14,7 +14,9 @@ mod sse4;
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
mod wasm32;
|
||||
|
||||
impl Convolution for U16x3 {
|
||||
type P = U16x3;
|
||||
|
||||
impl Convolution for P {
|
||||
fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = Self>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
@@ -22,16 +24,19 @@ impl Convolution for U16x3 {
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
_ => native::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
debug_assert!(src_view.height() - offset >= dst_view.height());
|
||||
|
||||
let normalizer = Normalizer32::new(coeffs);
|
||||
let normalizer_ref = &normalizer;
|
||||
|
||||
try_process_in_threads_h! {
|
||||
horiz_convolution(
|
||||
src_view,
|
||||
dst_view,
|
||||
offset,
|
||||
normalizer_ref,
|
||||
cpu_extensions,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -42,6 +47,39 @@ impl Convolution for U16x3 {
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
vert_convolution_u16(src_view, dst_view, offset, coeffs, cpu_extensions);
|
||||
debug_assert!(src_view.width() - offset >= dst_view.width());
|
||||
|
||||
let normalizer = Normalizer32::new(coeffs);
|
||||
let normalizer_ref = &normalizer;
|
||||
|
||||
try_process_in_threads_v! {
|
||||
vert_convolution_u16(
|
||||
src_view,
|
||||
dst_view,
|
||||
offset,
|
||||
normalizer_ref,
|
||||
cpu_extensions,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = P>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = P>,
|
||||
offset: u32,
|
||||
normalizer: &Normalizer32,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
_ => native::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer32;
|
||||
use crate::pixels::U16x3;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
@@ -7,11 +7,10 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U16x3>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16x3>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
let coefficients_chunks = normalizer.chunks();
|
||||
let initial = 1i64 << (precision - 1);
|
||||
|
||||
let src_rows = src_view.iter_rows(offset);
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use std::arch::aarch64::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::{CoefficientsI32Chunk, Normalizer32};
|
||||
use crate::neon_utils;
|
||||
use crate::pixels::U16x3;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
@@ -10,9 +10,8 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U16x3>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16x3>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
|
||||
macro_rules! call {
|
||||
@@ -27,15 +26,15 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
src_view: &impl ImageView<Pixel = U16x3>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16x3>,
|
||||
offset: u32,
|
||||
normalizer: optimisations::Normalizer32,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let coefficients_chunks = normalizer.chunks();
|
||||
|
||||
let src_iter = src_view.iter_rows(offset);
|
||||
let dst_iter = dst_view.iter_rows_mut(0);
|
||||
for (src_row, dst_row) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &coefficients_chunks);
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, coefficients_chunks);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -49,16 +48,16 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
src_row: &[U16x3],
|
||||
dst_row: &mut [U16x3],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
coefficients_chunks: &[CoefficientsI32Chunk],
|
||||
) {
|
||||
let initial = vdupq_n_s64(1i64 << (PRECISION - 2));
|
||||
let zero_u16x8 = vdupq_n_u16(0);
|
||||
let zero_u16x4 = vdup_n_u16(0);
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut sss = [initial; 3];
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
coeffs = coeffs_by_8.remainder();
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use std::arch::x86_64::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer32;
|
||||
use crate::pixels::U16x3;
|
||||
use crate::{simd_utils, ImageView, ImageViewMut};
|
||||
|
||||
@@ -9,16 +9,15 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U16x3>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16x3>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -27,7 +26,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &normalizer);
|
||||
horiz_convolution_one_row(src_row, dst_row, normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -42,13 +41,12 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U16x3]; 4],
|
||||
dst_rows: [&mut [U16x3]; 4],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut rg_buf = [0i64; 2];
|
||||
let mut bb_buf = [0i64; 2];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|R G B | |R G B | |R G |
|
||||
@@ -71,12 +69,12 @@ unsafe fn horiz_convolution_four_rows(
|
||||
|
||||
let width = src_rows[0].len();
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
for (dst_x, chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x = chunk.start as usize;
|
||||
let mut rg_sum = [_mm_set1_epi8(0); 4];
|
||||
let mut bb_sum = [_mm_set1_epi8(0); 4];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
let mut coeffs = chunk.values();
|
||||
let end_x = x + coeffs.len();
|
||||
|
||||
if width - end_x >= 1 {
|
||||
@@ -137,12 +135,11 @@ unsafe fn horiz_convolution_four_rows(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16x3],
|
||||
dst_row: &mut [U16x3],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let rg_initial = _mm_set1_epi64x(1 << (precision - 1));
|
||||
let bb_initial = _mm_set1_epi64x(1 << (precision - 2));
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|R G B | |R G B | |R G |
|
||||
@@ -167,13 +164,13 @@ unsafe fn horiz_convolution_one_row(
|
||||
|
||||
let width = src_row.len();
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
for (dst_x, chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x = chunk.start as usize;
|
||||
|
||||
let mut rg_sum = rg_initial;
|
||||
let mut bb_sum = bb_initial;
|
||||
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
let mut coeffs = chunk.values();
|
||||
let end_x = x + coeffs.len();
|
||||
|
||||
if width - end_x >= 1 {
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use std::arch::wasm32::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer32;
|
||||
use crate::pixels::U16x3;
|
||||
use crate::wasm32_utils;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
@@ -10,17 +10,15 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U16x3>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16x3>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -29,7 +27,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_one_row(src_row, dst_row, normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -44,8 +42,7 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U16x3]; 4],
|
||||
dst_rows: [&mut [U16x3]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
const ZERO: v128 = i64x2(0, 0);
|
||||
let precision = normalizer.precision();
|
||||
@@ -74,12 +71,12 @@ unsafe fn horiz_convolution_four_rows(
|
||||
|
||||
let width = src_rows[0].len();
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
for (dst_x, coeffs_chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut rg_sum = [ZERO; 4];
|
||||
let mut bb_sum = [ZERO; 4];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
let end_x = x + coeffs.len();
|
||||
|
||||
if width - end_x >= 1 {
|
||||
@@ -147,8 +144,7 @@ unsafe fn horiz_convolution_four_rows(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16x3],
|
||||
dst_row: &mut [U16x3],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let rg_initial = i64x2_splat(1 << (precision - 1));
|
||||
@@ -177,13 +173,13 @@ unsafe fn horiz_convolution_one_row(
|
||||
|
||||
let width = src_row.len();
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
for (dst_x, coeffs_chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
|
||||
let mut rg_sum = rg_initial;
|
||||
let mut bb_sum = bb_initial;
|
||||
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
let end_x = x + coeffs.len();
|
||||
|
||||
if width - end_x >= 1 {
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use std::arch::x86_64::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer32;
|
||||
use crate::pixels::U16x4;
|
||||
use crate::{simd_utils, ImageView, ImageViewMut};
|
||||
|
||||
@@ -9,16 +9,15 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U16x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16x4>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -27,7 +26,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &normalizer);
|
||||
horiz_convolution_one_row(src_row, dst_row, normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -42,13 +41,12 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U16x4]; 4],
|
||||
dst_rows: [&mut [U16x4]; 4],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut rg_buf = [0i64; 4];
|
||||
let mut ba_buf = [0i64; 4];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|R0 G0 B0 A0 | |R1 G1 B1 A1 |
|
||||
@@ -87,13 +85,12 @@ unsafe fn horiz_convolution_four_rows(
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
for (dst_x, chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x = chunk.start as usize;
|
||||
let mut coeffs = chunk.values();
|
||||
let mut rg_sum = [_mm256_set1_epi64x(half_error); 2];
|
||||
let mut ba_sum = [_mm256_set1_epi64x(half_error); 2];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_2 = coeffs.chunks_exact(2);
|
||||
coeffs = coeffs_by_2.remainder();
|
||||
|
||||
@@ -178,13 +175,12 @@ unsafe fn horiz_convolution_four_rows(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16x4],
|
||||
dst_row: &mut [U16x4],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut rg_buf = [0i64; 4];
|
||||
let mut ba_buf = [0i64; 4];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|R0 G0 B0 A0 | |R1 G1 B1 A1 |
|
||||
@@ -224,9 +220,9 @@ unsafe fn horiz_convolution_one_row(
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
for (dst_x, chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x = chunk.start as usize;
|
||||
let mut coeffs = chunk.values();
|
||||
let mut rg_sum = _mm256_setzero_si256();
|
||||
let mut ba_sum = _mm256_setzero_si256();
|
||||
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
use super::{Coefficients, Convolution};
|
||||
use crate::convolution::optimisations::Normalizer32;
|
||||
use crate::convolution::vertical_u16::vert_convolution_u16;
|
||||
use crate::pixels::U16x4;
|
||||
use crate::{CpuExtensions, ImageView, ImageViewMut};
|
||||
|
||||
use super::{Coefficients, Convolution};
|
||||
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod avx2;
|
||||
mod native;
|
||||
@@ -14,7 +14,9 @@ mod sse4;
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
mod wasm32;
|
||||
|
||||
impl Convolution for U16x4 {
|
||||
type P = U16x4;
|
||||
|
||||
impl Convolution for P {
|
||||
fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = Self>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
@@ -22,16 +24,19 @@ impl Convolution for U16x4 {
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
_ => native::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
debug_assert!(src_view.height() - offset >= dst_view.height());
|
||||
|
||||
let normalizer = Normalizer32::new(coeffs);
|
||||
let normalizer_ref = &normalizer;
|
||||
|
||||
try_process_in_threads_h! {
|
||||
horiz_convolution(
|
||||
src_view,
|
||||
dst_view,
|
||||
offset,
|
||||
normalizer_ref,
|
||||
cpu_extensions,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -42,6 +47,39 @@ impl Convolution for U16x4 {
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
vert_convolution_u16(src_view, dst_view, offset, coeffs, cpu_extensions);
|
||||
debug_assert!(src_view.width() - offset >= dst_view.width());
|
||||
|
||||
let normalizer = Normalizer32::new(coeffs);
|
||||
let normalizer_ref = &normalizer;
|
||||
|
||||
try_process_in_threads_v! {
|
||||
vert_convolution_u16(
|
||||
src_view,
|
||||
dst_view,
|
||||
offset,
|
||||
normalizer_ref,
|
||||
cpu_extensions,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = P>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = P>,
|
||||
offset: u32,
|
||||
normalizer: &Normalizer32,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
_ => native::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer32;
|
||||
use crate::pixels::U16x4;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
@@ -7,11 +7,10 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U16x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16x4>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
let coefficients_chunks = normalizer.chunks();
|
||||
let initial: i64 = 1 << (precision - 1);
|
||||
|
||||
let src_rows = src_view.iter_rows(offset);
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use std::arch::aarch64::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::{CoefficientsI32Chunk, Normalizer32};
|
||||
use crate::neon_utils;
|
||||
use crate::pixels::U16x4;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
@@ -10,9 +10,8 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U16x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16x4>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
|
||||
macro_rules! call {
|
||||
@@ -27,16 +26,16 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
src_view: &impl ImageView<Pixel = U16x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16x4>,
|
||||
offset: u32,
|
||||
normalizer: optimisations::Normalizer32,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let coefficients_chunks = normalizer.chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &coefficients_chunks);
|
||||
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, coefficients_chunks);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -45,7 +44,7 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &coefficients_chunks);
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, coefficients_chunks);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -60,7 +59,7 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
src_rows: [&[U16x4]; 4],
|
||||
dst_rows: [&mut [U16x4]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
coefficients_chunks: &[CoefficientsI32Chunk],
|
||||
) {
|
||||
let initial = vdupq_n_s64(1i64 << (PRECISION - 1));
|
||||
let zero_u16x8 = vdupq_n_u16(0);
|
||||
@@ -71,7 +70,7 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
|
||||
let mut sss_a = [int64x2x2_t(initial, initial); 4];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(4);
|
||||
coeffs = coeffs_by_8.remainder();
|
||||
@@ -179,16 +178,16 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
src_row: &[U16x4],
|
||||
dst_row: &mut [U16x4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
coefficients_chunks: &[CoefficientsI32Chunk],
|
||||
) {
|
||||
let initial = vdupq_n_s64(1i64 << (PRECISION - 1));
|
||||
let zero_u16x8 = vdupq_n_u16(0);
|
||||
let zero_u16x4 = vdup_n_u16(0);
|
||||
|
||||
for (&coeffs_chunk, dst_pix) in coefficients_chunks.iter().zip(dst_row) {
|
||||
for (coeffs_chunk, dst_pix) in coefficients_chunks.iter().zip(dst_row) {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut sss = int64x2x2_t(initial, initial);
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
coeffs = coeffs_by_4.remainder();
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use std::arch::x86_64::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer32;
|
||||
use crate::pixels::U16x4;
|
||||
use crate::{simd_utils, ImageView, ImageViewMut};
|
||||
|
||||
@@ -9,16 +9,15 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U16x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16x4>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -27,7 +26,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &normalizer);
|
||||
horiz_convolution_one_row(src_row, dst_row, normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -42,13 +41,12 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U16x4]; 4],
|
||||
dst_rows: [&mut [U16x4]; 4],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut rg_buf = [0i64; 2];
|
||||
let mut ba_buf = [0i64; 2];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|R0 G0 B0 A0 | |R1 G1 B1 A1 |
|
||||
@@ -74,13 +72,12 @@ unsafe fn horiz_convolution_four_rows(
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
for (dst_x, chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x = chunk.start as usize;
|
||||
let mut coeffs = chunk.values();
|
||||
let mut rg_sum = [_mm_set1_epi64x(half_error); 4];
|
||||
let mut ba_sum = [_mm_set1_epi64x(half_error); 4];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_2 = coeffs.chunks_exact(2);
|
||||
coeffs = coeffs_by_2.remainder();
|
||||
|
||||
@@ -142,13 +139,12 @@ unsafe fn horiz_convolution_four_rows(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16x4],
|
||||
dst_row: &mut [U16x4],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut rg_buf = [0i64; 2];
|
||||
let mut ba_buf = [0i64; 2];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|R0 G0 B0 A0 | |R1 G1 B1 A1 |
|
||||
@@ -174,9 +170,9 @@ unsafe fn horiz_convolution_one_row(
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
for (dst_x, chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x = chunk.start as usize;
|
||||
let mut coeffs = chunk.values();
|
||||
let mut rg_sum = _mm_set1_epi64x(half_error);
|
||||
let mut ba_sum = _mm_set1_epi64x(half_error);
|
||||
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use std::arch::wasm32::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer32;
|
||||
use crate::pixels::U16x4;
|
||||
use crate::wasm32_utils;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
@@ -10,17 +10,15 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U16x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16x4>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -29,7 +27,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_one_row(src_row, dst_row, normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -44,8 +42,7 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U16x4]; 4],
|
||||
dst_rows: [&mut [U16x4]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
@@ -76,12 +73,12 @@ unsafe fn horiz_convolution_four_rows(
|
||||
12, 13, -1, -1, -1, -1, -1, -1, 14, 15, -1, -1, -1, -1, -1, -1,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
for (dst_x, coeffs_chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut rg_sum = [i64x2_splat(half_error); 4];
|
||||
let mut ba_sum = [i64x2_splat(half_error); 4];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_2 = coeffs.chunks_exact(2);
|
||||
coeffs = coeffs_by_2.remainder();
|
||||
@@ -150,8 +147,7 @@ unsafe fn horiz_convolution_four_rows(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16x4],
|
||||
dst_row: &mut [U16x4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
normalizer: &Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
@@ -182,9 +178,9 @@ unsafe fn horiz_convolution_one_row(
|
||||
12, 13, -1, -1, -1, -1, -1, -1, 14, 15, -1, -1, -1, -1, -1, -1,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
for (dst_x, coeffs_chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
let mut rg_sum = i64x2_splat(half_error);
|
||||
let mut ba_sum = i64x2_splat(half_error);
|
||||
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use std::arch::x86_64::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer16;
|
||||
use crate::pixels::U8;
|
||||
use crate::{simd_utils, ImageView, ImageViewMut};
|
||||
|
||||
@@ -9,16 +9,15 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U8>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -27,7 +26,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &normalizer);
|
||||
horiz_convolution_one_row(src_row, dst_row, normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -43,19 +42,17 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U8]; 4],
|
||||
dst_rows: [&mut [U8]; 4],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let zero = _mm_setzero_si128();
|
||||
// 8 components will be added, use only 1/8 of the error
|
||||
let initial = _mm256_set1_epi32(1 << (normalizer.precision() - 4));
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let coeffs = coeffs_chunk.values();
|
||||
let mut x = coeffs_chunk.start as usize;
|
||||
for (dst_x, chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x = chunk.start as usize;
|
||||
let mut result_i32x8x4 = [initial, initial, initial, initial];
|
||||
|
||||
let coeffs_by_16 = coeffs.chunks_exact(16);
|
||||
let coeffs_by_16 = chunk.values().chunks_exact(16);
|
||||
let reminder16 = coeffs_by_16.remainder();
|
||||
for k in coeffs_by_16 {
|
||||
let coeffs_i16x16 = _mm256_loadu_si256(k.as_ptr() as *const __m256i);
|
||||
@@ -109,22 +106,16 @@ unsafe fn horiz_convolution_four_rows(
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[inline]
|
||||
#[target_feature(enable = "avx2")]
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U8],
|
||||
dst_row: &mut [U8],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
unsafe fn horiz_convolution_one_row(src_row: &[U8], dst_row: &mut [U8], normalizer: &Normalizer16) {
|
||||
let zero = _mm_setzero_si128();
|
||||
// 8 components will be added, use only 1/8 of the error
|
||||
let initial = _mm256_set1_epi32(1 << (normalizer.precision() - 4));
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let coeffs = coeffs_chunk.values();
|
||||
let mut x = coeffs_chunk.start as usize;
|
||||
for (dst_x, chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x = chunk.start as usize;
|
||||
let mut result_i32x8 = initial;
|
||||
|
||||
let coeffs_by_16 = coeffs.chunks_exact(16);
|
||||
let coeffs_by_16 = chunk.values().chunks_exact(16);
|
||||
let reminder16 = coeffs_by_16.remainder();
|
||||
for k in coeffs_by_16 {
|
||||
let coeffs_i16x16 = _mm256_loadu_si256(k.as_ptr() as *const __m256i);
|
||||
|
||||
+52
-14
@@ -1,9 +1,9 @@
|
||||
use super::{Coefficients, Convolution};
|
||||
use crate::convolution::optimisations::Normalizer16;
|
||||
use crate::convolution::vertical_u8::vert_convolution_u8;
|
||||
use crate::pixels::U8;
|
||||
use crate::{CpuExtensions, ImageView, ImageViewMut};
|
||||
|
||||
use super::{Coefficients, Convolution};
|
||||
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod avx2;
|
||||
mod native;
|
||||
@@ -14,7 +14,9 @@ mod sse4;
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
mod wasm32;
|
||||
|
||||
impl Convolution for U8 {
|
||||
type P = U8;
|
||||
|
||||
impl Convolution for P {
|
||||
fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = Self>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
@@ -22,16 +24,19 @@ impl Convolution for U8 {
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
_ => native::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
debug_assert!(src_view.height() - offset >= dst_view.height());
|
||||
|
||||
let normalizer = Normalizer16::new(coeffs);
|
||||
let normalizer_ref = &normalizer;
|
||||
|
||||
try_process_in_threads_h! {
|
||||
horiz_convolution(
|
||||
src_view,
|
||||
dst_view,
|
||||
offset,
|
||||
normalizer_ref,
|
||||
cpu_extensions,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -42,6 +47,39 @@ impl Convolution for U8 {
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
vert_convolution_u8(src_view, dst_view, offset, coeffs, cpu_extensions);
|
||||
debug_assert!(src_view.width() - offset >= dst_view.width());
|
||||
|
||||
let normalizer = Normalizer16::new(coeffs);
|
||||
let normalizer_ref = &normalizer;
|
||||
|
||||
try_process_in_threads_v! {
|
||||
vert_convolution_u8(
|
||||
src_view,
|
||||
dst_view,
|
||||
offset,
|
||||
normalizer_ref,
|
||||
cpu_extensions,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = P>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = P>,
|
||||
offset: u32,
|
||||
normalizer: &Normalizer16,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
_ => native::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer16;
|
||||
use crate::pixels::U8;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
@@ -7,17 +7,16 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U8>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
let initial = 1i32 << (precision - 1);
|
||||
let initial = 1 << (precision - 1);
|
||||
let coefficients = normalizer.chunks();
|
||||
|
||||
let src_rows = src_view.iter_rows(offset);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
for (dst_row, src_row) in dst_rows.zip(src_rows) {
|
||||
for (coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
|
||||
for (coeffs_chunk, dst_pixel) in coefficients.iter().zip(dst_row.iter_mut()) {
|
||||
let first_x_src = coeffs_chunk.start as usize;
|
||||
let mut ss = initial;
|
||||
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use std::arch::aarch64::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer16;
|
||||
use crate::neon_utils;
|
||||
use crate::pixels::U8;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
@@ -10,17 +10,15 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U8>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -29,7 +27,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_one_row(src_row, dst_row, normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -44,18 +42,17 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U8]; 4],
|
||||
dst_rows: [&mut [U8]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let initial = vdupq_n_s32(1 << (precision - 3));
|
||||
let zero_u8x16 = vdupq_n_u8(0);
|
||||
let zero_u8x8 = vdup_n_u8(0);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
for (dst_x, coeffs_chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x = coeffs_chunk.start as usize;
|
||||
let mut sss_a = [initial; 4];
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_16 = coeffs.chunks_exact(16);
|
||||
coeffs = coeffs_by_16.remainder();
|
||||
@@ -150,21 +147,16 @@ unsafe fn horiz_convolution_four_rows(
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[target_feature(enable = "neon")]
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U8],
|
||||
dst_row: &mut [U8],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
unsafe fn horiz_convolution_one_row(src_row: &[U8], dst_row: &mut [U8], normalizer: &Normalizer16) {
|
||||
let precision = normalizer.precision();
|
||||
let initial = vdupq_n_s32(1 << (precision - 3));
|
||||
let zero_u8x16 = vdupq_n_u8(0);
|
||||
let zero_u8x8 = vdup_n_u8(0);
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
for (dst_x, coeffs_chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x = coeffs_chunk.start as usize;
|
||||
let mut sss = initial;
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_16 = coeffs.chunks_exact(16);
|
||||
coeffs = coeffs_by_16.remainder();
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use std::arch::x86_64::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer16;
|
||||
use crate::pixels::U8;
|
||||
use crate::{simd_utils, ImageView, ImageViewMut};
|
||||
|
||||
@@ -9,16 +9,15 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U8>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -27,7 +26,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &normalizer);
|
||||
horiz_convolution_one_row(src_row, dst_row, normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -43,19 +42,17 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U8]; 4],
|
||||
dst_rows: [&mut [U8]; 4],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let zero = _mm_setzero_si128();
|
||||
let initial = 1 << (normalizer.precision() - 1);
|
||||
let mut buf = [0, 0, 0, 0, initial];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let coeffs = coeffs_chunk.values();
|
||||
let mut x = coeffs_chunk.start as usize;
|
||||
for (dst_x, chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x = chunk.start as usize;
|
||||
let mut result_i32x4 = [zero, zero, zero, zero];
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
let coeffs_by_8 = chunk.values().chunks_exact(8);
|
||||
let reminder8 = coeffs_by_8.remainder();
|
||||
for k in coeffs_by_8 {
|
||||
let coeffs_i16x8 = _mm_loadu_si128(k.as_ptr() as *const __m128i);
|
||||
@@ -108,22 +105,16 @@ unsafe fn horiz_convolution_four_rows(
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[inline]
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U8],
|
||||
dst_row: &mut [U8],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
unsafe fn horiz_convolution_one_row(src_row: &[U8], dst_row: &mut [U8], normalizer: &Normalizer16) {
|
||||
let zero = _mm_setzero_si128();
|
||||
let initial = 1 << (normalizer.precision() - 1);
|
||||
let mut buf = [0, 0, 0, 0, initial];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let coeffs = coeffs_chunk.values();
|
||||
let mut x = coeffs_chunk.start as usize;
|
||||
for (dst_x, chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x = chunk.start as usize;
|
||||
let mut result_i32x4 = zero;
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
let coeffs_by_8 = chunk.values().chunks_exact(8);
|
||||
let reminder8 = coeffs_by_8.remainder();
|
||||
for k in coeffs_by_8 {
|
||||
let coeffs_i16x8 = _mm_loadu_si128(k.as_ptr() as *const __m128i);
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use std::arch::wasm32::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer16;
|
||||
use crate::pixels::U8;
|
||||
use crate::wasm32_utils;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
@@ -10,17 +10,15 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U8>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -29,7 +27,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_one_row(src_row, dst_row, normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -45,15 +43,15 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U8]; 4],
|
||||
dst_rows: [&mut [U8]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
const ZERO: v128 = i64x2(0, 0);
|
||||
let initial = 1 << (normalizer.precision() - 1);
|
||||
let mut buf = [0, 0, 0, 0, initial];
|
||||
let coefficients_chunks = normalizer.chunks();
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let coeffs = coeffs_chunk.values;
|
||||
let coeffs = coeffs_chunk.values();
|
||||
let mut x = coeffs_chunk.start as usize;
|
||||
let mut result_i32x4 = [ZERO, ZERO, ZERO, ZERO];
|
||||
|
||||
@@ -110,18 +108,14 @@ unsafe fn horiz_convolution_four_rows(
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[inline]
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U8],
|
||||
dst_row: &mut [U8],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
unsafe fn horiz_convolution_one_row(src_row: &[U8], dst_row: &mut [U8], normalizer: &Normalizer16) {
|
||||
const ZERO: v128 = i64x2(0, 0);
|
||||
let initial = 1 << (normalizer.precision() - 1);
|
||||
let mut buf = [0, 0, 0, 0, initial];
|
||||
let coefficients_chunks = normalizer.chunks();
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let coeffs = coeffs_chunk.values;
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let coeffs = coeffs_chunk.values();
|
||||
let mut x = coeffs_chunk.start as usize;
|
||||
let mut result_i32x4 = ZERO;
|
||||
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use std::arch::x86_64::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer16;
|
||||
use crate::pixels::U8x2;
|
||||
use crate::{simd_utils, ImageView, ImageViewMut};
|
||||
|
||||
@@ -9,16 +9,15 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U8x2>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x2>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -27,7 +26,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &normalizer);
|
||||
horiz_convolution_one_row(src_row, dst_row, normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -43,11 +42,10 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U8x2]; 4],
|
||||
dst_rows: [&mut [U8x2]; 4],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let initial = _mm256_set1_epi32(1 << (precision - 2));
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|L A | |L A | |L A | |L A | |L A | |L A | |L A | |L A |
|
||||
@@ -73,12 +71,12 @@ unsafe fn horiz_convolution_four_rows(
|
||||
-1, 15, -1, 13, -1, 11, -1, 9, -1, 14, -1, 12, -1, 10, -1, 8,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x = coeffs_chunk.start as usize;
|
||||
for (dst_x, chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x = chunk.start as usize;
|
||||
|
||||
let mut sss0 = initial;
|
||||
let mut sss1 = initial;
|
||||
let coeffs = coeffs_chunk.values();
|
||||
let coeffs = chunk.values();
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
let reminder = coeffs_by_8.remainder();
|
||||
@@ -188,12 +186,7 @@ unsafe fn horiz_convolution_four_rows(
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "avx2")]
|
||||
unsafe fn set_dst_pixel(
|
||||
raw: __m128i,
|
||||
d_row: &mut [U8x2],
|
||||
dst_x: usize,
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
unsafe fn set_dst_pixel(raw: __m128i, d_row: &mut [U8x2], dst_x: usize, normalizer: &Normalizer16) {
|
||||
let l32x2 = _mm_extract_epi64::<0>(raw);
|
||||
let a32x2 = _mm_extract_epi64::<1>(raw);
|
||||
let l32 = ((l32x2 >> 32) as i32).saturating_add((l32x2 & 0xffffffff) as i32);
|
||||
@@ -213,7 +206,7 @@ unsafe fn set_dst_pixel(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U8x2],
|
||||
dst_row: &mut [U8x2],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
/*
|
||||
@@ -313,11 +306,10 @@ unsafe fn horiz_convolution_one_row(
|
||||
L: |-1 02| |-1 00|
|
||||
*/
|
||||
let pix_sh4 = _mm_set_epi8(-1, 7, -1, 5, -1, 6, -1, 4, -1, 3, -1, 1, -1, 2, -1, 0);
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x = coeffs_chunk.start as usize;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
for (dst_x, chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x = chunk.start as usize;
|
||||
let mut coeffs = chunk.values();
|
||||
|
||||
let mut sss = if coeffs.len() < 16 {
|
||||
// Lower part will be added to higher, use only half of the error
|
||||
|
||||
+52
-14
@@ -1,9 +1,9 @@
|
||||
use super::{Coefficients, Convolution};
|
||||
use crate::convolution::optimisations::Normalizer16;
|
||||
use crate::convolution::vertical_u8::vert_convolution_u8;
|
||||
use crate::pixels::U8x2;
|
||||
use crate::{CpuExtensions, ImageView, ImageViewMut};
|
||||
|
||||
use super::{Coefficients, Convolution};
|
||||
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod avx2;
|
||||
mod native;
|
||||
@@ -14,7 +14,9 @@ mod sse4;
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
mod wasm32;
|
||||
|
||||
impl Convolution for U8x2 {
|
||||
type P = U8x2;
|
||||
|
||||
impl Convolution for P {
|
||||
fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = Self>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
@@ -22,16 +24,19 @@ impl Convolution for U8x2 {
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
_ => native::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
debug_assert!(src_view.height() - offset >= dst_view.height());
|
||||
|
||||
let normalizer = Normalizer16::new(coeffs);
|
||||
let normalizer_ref = &normalizer;
|
||||
|
||||
try_process_in_threads_h! {
|
||||
horiz_convolution(
|
||||
src_view,
|
||||
dst_view,
|
||||
offset,
|
||||
normalizer_ref,
|
||||
cpu_extensions,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -42,6 +47,39 @@ impl Convolution for U8x2 {
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
vert_convolution_u8(src_view, dst_view, offset, coeffs, cpu_extensions);
|
||||
debug_assert!(src_view.width() - offset >= dst_view.width());
|
||||
|
||||
let normalizer = Normalizer16::new(coeffs);
|
||||
let normalizer_ref = &normalizer;
|
||||
|
||||
try_process_in_threads_v! {
|
||||
vert_convolution_u8(
|
||||
src_view,
|
||||
dst_view,
|
||||
offset,
|
||||
normalizer_ref,
|
||||
cpu_extensions,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = P>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = P>,
|
||||
offset: u32,
|
||||
normalizer: &Normalizer16,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
_ => native::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer16;
|
||||
use crate::pixels::U8x2;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
@@ -6,11 +6,10 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U8x2>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x2>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
let coefficients_chunks = normalizer.chunks();
|
||||
let initial = 1 << (precision - 1);
|
||||
|
||||
let src_rows = src_view.iter_rows(offset);
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use std::arch::aarch64::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::{CoefficientsI16Chunk, Normalizer16};
|
||||
use crate::pixels::U8x2;
|
||||
use crate::{neon_utils, ImageView, ImageViewMut};
|
||||
|
||||
@@ -9,9 +9,8 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U8x2>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x2>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
|
||||
macro_rules! call {
|
||||
@@ -26,16 +25,16 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
src_view: &impl ImageView<Pixel = U8x2>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x2>,
|
||||
offset: u32,
|
||||
normalizer: optimisations::Normalizer16,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let coefficients_chunks = normalizer.chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &coefficients_chunks);
|
||||
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, coefficients_chunks);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -44,7 +43,7 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &coefficients_chunks);
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, coefficients_chunks);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -59,7 +58,7 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
src_rows: [&[U8x2]; 4],
|
||||
dst_rows: [&mut [U8x2]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
coefficients_chunks: &[CoefficientsI16Chunk],
|
||||
) {
|
||||
let initial = vdupq_n_s32(1 << (PRECISION - 2));
|
||||
let zero_u8x16 = vdupq_n_u8(0);
|
||||
@@ -68,7 +67,7 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut sss_a = [initial; 4];
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
coeffs = coeffs_by_8.remainder();
|
||||
@@ -176,16 +175,16 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
src_row: &[U8x2],
|
||||
dst_row: &mut [U8x2],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
coefficients_chunks: &[CoefficientsI16Chunk],
|
||||
) {
|
||||
let initial = vdupq_n_s32(1 << (PRECISION - 2));
|
||||
let zero_u8x16 = vdupq_n_u8(0);
|
||||
let zero_u8x8 = vdup_n_u8(0);
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut sss = initial;
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
coeffs = coeffs_by_8.remainder();
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use std::arch::x86_64::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer16;
|
||||
use crate::pixels::U8x2;
|
||||
use crate::{simd_utils, ImageView, ImageViewMut};
|
||||
|
||||
@@ -9,16 +9,15 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U8x2>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x2>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -27,7 +26,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &normalizer);
|
||||
horiz_convolution_one_row(src_row, dst_row, normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -43,11 +42,10 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U8x2]; 4],
|
||||
dst_rows: [&mut [U8x2]; 4],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let initial = _mm_set1_epi32(1 << (precision - 2));
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|L A | |L A | |L A | |L A | |L A | |L A | |L A | |L A |
|
||||
@@ -71,11 +69,11 @@ unsafe fn horiz_convolution_four_rows(
|
||||
-1, 15, -1, 13, -1, 11, -1, 9, -1, 14, -1, 12, -1, 10, -1, 8,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x = coeffs_chunk.start as usize;
|
||||
for (dst_x, chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x = chunk.start as usize;
|
||||
let coeffs = chunk.values();
|
||||
|
||||
let mut sss: [__m128i; 4] = [initial; 4];
|
||||
let coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
let reminder = coeffs_by_8.remainder();
|
||||
@@ -140,12 +138,7 @@ unsafe fn horiz_convolution_four_rows(
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
unsafe fn set_dst_pixel(
|
||||
raw: __m128i,
|
||||
d_row: &mut [U8x2],
|
||||
dst_x: usize,
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
unsafe fn set_dst_pixel(raw: __m128i, d_row: &mut [U8x2], dst_x: usize, normalizer: &Normalizer16) {
|
||||
let l32x2 = _mm_extract_epi64::<0>(raw);
|
||||
let a32x2 = _mm_extract_epi64::<1>(raw);
|
||||
let l32 = ((l32x2 >> 32) as i32).saturating_add((l32x2 & 0xffffffff) as i32);
|
||||
@@ -165,7 +158,7 @@ unsafe fn set_dst_pixel(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U8x2],
|
||||
dst_row: &mut [U8x2],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
/*
|
||||
@@ -242,11 +235,10 @@ unsafe fn horiz_convolution_one_row(
|
||||
L: |-1 02| |-1 00|
|
||||
*/
|
||||
let pix_sh3 = _mm_set_epi8(-1, 7, -1, 5, -1, 6, -1, 4, -1, 3, -1, 1, -1, 2, -1, 0);
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x = coeffs_chunk.start as usize;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
for (dst_x, chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x = chunk.start as usize;
|
||||
let mut coeffs = chunk.values();
|
||||
|
||||
// Lower part will be added to higher, use only half of the error
|
||||
let mut sss = _mm_set1_epi32(1 << (precision - 2));
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use std::arch::wasm32::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer16;
|
||||
use crate::pixels::U8x2;
|
||||
use crate::wasm32_utils;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
@@ -10,17 +10,15 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U8x2>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x2>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -29,7 +27,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_one_row(src_row, dst_row, normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -45,8 +43,7 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U8x2]; 4],
|
||||
dst_rows: [&mut [U8x2]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let initial = i32x4_splat(1 << (precision - 2));
|
||||
@@ -67,11 +64,11 @@ unsafe fn horiz_convolution_four_rows(
|
||||
*/
|
||||
const SH2: v128 = i8x16(8, -1, 10, -1, 12, -1, 14, -1, 9, -1, 11, -1, 13, -1, 15, -1);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
for (dst_x, coeffs_chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x = coeffs_chunk.start as usize;
|
||||
|
||||
let mut sss: [v128; 4] = [initial; 4];
|
||||
let coeffs = coeffs_chunk.values;
|
||||
let coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
let reminder = coeffs_by_8.remainder();
|
||||
@@ -142,8 +139,7 @@ unsafe fn horiz_convolution_four_rows(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U8x2],
|
||||
dst_row: &mut [U8x2],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
const SH1: v128 = i8x16(0, -1, 2, -1, 4, -1, 6, -1, 1, -1, 3, -1, 5, -1, 7, -1);
|
||||
/*
|
||||
@@ -156,9 +152,9 @@ unsafe fn horiz_convolution_one_row(
|
||||
let precision = normalizer.precision();
|
||||
let initial = i32x4_splat(1 << (precision - 2));
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
for (dst_x, coeffs_chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x = coeffs_chunk.start as usize;
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let mut sss = initial;
|
||||
|
||||
@@ -214,12 +210,7 @@ unsafe fn horiz_convolution_one_row(
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn set_dst_pixel(
|
||||
raw: v128,
|
||||
d_row: &mut [U8x2],
|
||||
dst_x: usize,
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
unsafe fn set_dst_pixel(raw: v128, d_row: &mut [U8x2], dst_x: usize, normalizer: &Normalizer16) {
|
||||
let mut buf = [0i32; 4];
|
||||
v128_store(buf.as_mut_ptr() as _, raw);
|
||||
let l32 = buf[0].saturating_add(buf[1]);
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
use std::arch::x86_64::*;
|
||||
use std::intrinsics::transmute;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer16;
|
||||
use crate::pixels::U8x3;
|
||||
use crate::{simd_utils, ImageView, ImageViewMut};
|
||||
|
||||
@@ -10,9 +10,8 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U8x3>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x3>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
|
||||
macro_rules! call {
|
||||
@@ -27,7 +26,7 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
src_view: &impl ImageView<Pixel = U8x3>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x3>,
|
||||
offset: u32,
|
||||
normalizer: optimisations::Normalizer16,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
@@ -35,7 +34,7 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &normalizer);
|
||||
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -44,7 +43,7 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &normalizer);
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -60,12 +59,11 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
src_rows: [&[U8x3]; 4],
|
||||
dst_rows: [&mut [U8x3]; 4],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let zero = _mm256_setzero_si256();
|
||||
let initial = _mm256_set1_epi32(1 << (PRECISION - 1));
|
||||
let src_width = src_rows[0].len();
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|R G B | |R G B | |R G B | |R G B | |R G B | |R |
|
||||
@@ -96,13 +94,13 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
-1, -1, -1, -1, -1, 11, -1, 8, -1, 10, -1, 7, -1, 9, -1, 6,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let x_start = coeffs_chunk.start as usize;
|
||||
for (dst_x, chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let x_start = chunk.start as usize;
|
||||
let mut x = x_start;
|
||||
|
||||
let mut sss0 = initial;
|
||||
let mut sss1 = initial;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
let mut coeffs = chunk.values();
|
||||
|
||||
// (16 bytes) / (3 bytes per pixel) = 5 whole pixels + 1 byte
|
||||
let max_x = src_width.saturating_sub(5);
|
||||
@@ -224,7 +222,7 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
src_row: &[U8x3],
|
||||
dst_row: &mut [U8x3],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
#[rustfmt::skip]
|
||||
let sh1 = _mm256_set_epi8(
|
||||
@@ -271,12 +269,11 @@ unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
*/
|
||||
let sh7 = _mm_set_epi8(-1, -1, -1, -1, -1, 5, -1, 2, -1, 4, -1, 1, -1, 3, -1, 0);
|
||||
let src_width = src_row.len();
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let x_start = coeffs_chunk.start as usize;
|
||||
for (dst_x, chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let x_start = chunk.start as usize;
|
||||
let mut x = x_start;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
let mut coeffs = chunk.values();
|
||||
|
||||
// (16 bytes) / (3 bytes per pixel) = 5 whole pixels + 1 bytes
|
||||
// 4 + 5 = 9
|
||||
|
||||
+52
-14
@@ -1,9 +1,9 @@
|
||||
use super::{Coefficients, Convolution};
|
||||
use crate::convolution::optimisations::Normalizer16;
|
||||
use crate::convolution::vertical_u8::vert_convolution_u8;
|
||||
use crate::pixels::U8x3;
|
||||
use crate::{CpuExtensions, ImageView, ImageViewMut};
|
||||
|
||||
use super::{Coefficients, Convolution};
|
||||
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod avx2;
|
||||
mod native;
|
||||
@@ -14,7 +14,9 @@ mod sse4;
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
mod wasm32;
|
||||
|
||||
impl Convolution for U8x3 {
|
||||
type P = U8x3;
|
||||
|
||||
impl Convolution for P {
|
||||
fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = Self>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
@@ -22,16 +24,19 @@ impl Convolution for U8x3 {
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
_ => native::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
debug_assert!(src_view.height() - offset >= dst_view.height());
|
||||
|
||||
let normalizer = Normalizer16::new(coeffs);
|
||||
let normalizer_ref = &normalizer;
|
||||
|
||||
try_process_in_threads_h! {
|
||||
horiz_convolution(
|
||||
src_view,
|
||||
dst_view,
|
||||
offset,
|
||||
normalizer_ref,
|
||||
cpu_extensions,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -42,6 +47,39 @@ impl Convolution for U8x3 {
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
vert_convolution_u8(src_view, dst_view, offset, coeffs, cpu_extensions);
|
||||
debug_assert!(src_view.width() - offset >= dst_view.width());
|
||||
|
||||
let normalizer = Normalizer16::new(coeffs);
|
||||
let normalizer_ref = &normalizer;
|
||||
|
||||
try_process_in_threads_v! {
|
||||
vert_convolution_u8(
|
||||
src_view,
|
||||
dst_view,
|
||||
offset,
|
||||
normalizer_ref,
|
||||
cpu_extensions,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = P>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = P>,
|
||||
offset: u32,
|
||||
normalizer: &Normalizer16,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
_ => native::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer16;
|
||||
use crate::pixels::U8x3;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
@@ -7,11 +7,10 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U8x3>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x3>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients = normalizer.coefficients();
|
||||
let coefficients = normalizer.chunks();
|
||||
let initial = 1i32 << (precision - 1);
|
||||
|
||||
let src_rows = src_view.iter_rows(offset);
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use std::arch::aarch64::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::{CoefficientsI16Chunk, Normalizer16};
|
||||
use crate::neon_utils;
|
||||
use crate::pixels::U8x3;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
@@ -10,9 +10,8 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U8x3>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x3>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
|
||||
macro_rules! call {
|
||||
@@ -27,9 +26,9 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
src_view: &impl ImageView<Pixel = U8x3>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x3>,
|
||||
offset: u32,
|
||||
normalizer: optimisations::Normalizer16,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let coefficients_chunks = normalizer.chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
@@ -60,7 +59,7 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
src_rows: [&[U8x3]; 4],
|
||||
dst_rows: [&mut [U8x3]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
coefficients_chunks: &[CoefficientsI16Chunk],
|
||||
) {
|
||||
let initial = vdupq_n_s32(1 << (PRECISION - 1));
|
||||
let zero_u8x8 = vdup_n_u8(0);
|
||||
@@ -68,7 +67,7 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut sss_a = [initial; 4];
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
coeffs = coeffs_by_8.remainder();
|
||||
@@ -129,15 +128,15 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
src_row: &[U8x3],
|
||||
dst_row: &mut [U8x3],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
coefficients_chunks: &[CoefficientsI16Chunk],
|
||||
) {
|
||||
let initial = vdupq_n_s32(1 << (PRECISION - 1));
|
||||
let zero_u8x8 = vdup_n_u8(0);
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut sss = initial;
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
coeffs = coeffs_by_8.remainder();
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
use std::arch::x86_64::*;
|
||||
use std::intrinsics::transmute;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer16;
|
||||
use crate::pixels::U8x3;
|
||||
use crate::{simd_utils, ImageView, ImageViewMut};
|
||||
|
||||
@@ -10,9 +10,8 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U8x3>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x3>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
|
||||
macro_rules! call {
|
||||
@@ -27,7 +26,7 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
src_view: &impl ImageView<Pixel = U8x3>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x3>,
|
||||
offset: u32,
|
||||
normalizer: optimisations::Normalizer16,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
@@ -35,7 +34,7 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &normalizer);
|
||||
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -44,7 +43,7 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &normalizer);
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -60,12 +59,11 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
src_rows: [&[U8x3]; 4],
|
||||
dst_rows: [&mut [U8x3]; 4],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let zero = _mm_setzero_si128();
|
||||
let initial = _mm_set1_epi32(1 << (PRECISION - 1));
|
||||
let src_width = src_rows[0].len();
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|R G B | |R G B | |R G B | |R G B | |R G B | |R |
|
||||
@@ -94,16 +92,16 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
-1, -1, -1, -1, -1, 11, -1, 8, -1, 10, -1, 7, -1, 9, -1, 6,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let x_start = coeffs_chunk.start as usize;
|
||||
for (dst_x, chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let x_start = chunk.start as usize;
|
||||
let mut x = x_start;
|
||||
|
||||
let mut sss_a = [initial; 4];
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
let mut coeffs = chunk.values();
|
||||
|
||||
// Next block of code will be load source pixels by 16 bytes per time.
|
||||
// We must guarantee what this process will not go beyond
|
||||
// the one row of image.
|
||||
// The next block of code will load source pixels by 16 bytes per time.
|
||||
// We must guarantee that this process won't go beyond
|
||||
// the one row of the image.
|
||||
// (16 bytes) / (3 bytes per pixel) = 5 whole pixels + 1 byte
|
||||
let max_x = src_width.saturating_sub(5);
|
||||
if x < max_x {
|
||||
@@ -128,9 +126,9 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
}
|
||||
}
|
||||
|
||||
// Next block of code will be load source pixels by 8 bytes per time.
|
||||
// We must guarantee what this process will not go beyond
|
||||
// the one row of image.
|
||||
// The next block of code will load source pixels by 8 bytes per time.
|
||||
// We must guarantee that this process won't go beyond
|
||||
// the one row of the image.
|
||||
// (8 bytes) / (3 bytes per pixel) = 2 whole pixels + 2 bytes
|
||||
let max_x = src_width.saturating_sub(2);
|
||||
if x < max_x {
|
||||
@@ -172,7 +170,8 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
let sss = _mm_packs_epi32(sss_a[i], zero);
|
||||
let pixel: u32 = transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss, zero)));
|
||||
let bytes = pixel.to_le_bytes();
|
||||
dst_rows[i].get_unchecked_mut(dst_x).0 = [bytes[0], bytes[1], bytes[2]];
|
||||
let dst_pixel = dst_rows[i].get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = [bytes[0], bytes[1], bytes[2]];
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -187,7 +186,7 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
src_row: &[U8x3],
|
||||
dst_row: &mut [U8x3],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
#[rustfmt::skip]
|
||||
let pix_sh1 = _mm_set_epi8(
|
||||
@@ -219,18 +218,18 @@ unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
R: |-1 03| |-1 00|
|
||||
*/
|
||||
let src_width = src_row.len();
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let x_start = coeffs_chunk.start as usize;
|
||||
for (dst_x, chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let x_start = chunk.start as usize;
|
||||
let mut x = x_start;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let mut coeffs = chunk.values();
|
||||
let mut sss = _mm_set1_epi32(1 << (PRECISION - 1));
|
||||
|
||||
// Next block of code will be load source pixels by 16 bytes per time.
|
||||
// We must guarantee what this process will not go beyond
|
||||
// the one row of image.
|
||||
// (16 bytes) / (3 bytes per pixel) = 5 whole pixels + 1 bytes
|
||||
// The next block of code will load source pixels by 16 bytes per time.
|
||||
// We must guarantee that this process won't go beyond
|
||||
// the one row of the image.
|
||||
// (16 bytes) / (3 bytes per pixel) = 5 whole pixels + 1 byte
|
||||
let max_x = src_width.saturating_sub(5);
|
||||
if x < max_x {
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
@@ -253,9 +252,9 @@ unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
}
|
||||
}
|
||||
|
||||
// Next block of code will be load source pixels by 8 bytes per time.
|
||||
// We must guarantee what this process will not go beyond
|
||||
// the one row of image.
|
||||
// The next block of code will load source pixels by 8 bytes per time.
|
||||
// We must guarantee that this process won't go beyond
|
||||
// the one row of the image.
|
||||
// (8 bytes) / (3 bytes per pixel) = 2 whole pixels + 2 bytes
|
||||
let max_x = src_width.saturating_sub(2);
|
||||
if x < max_x {
|
||||
@@ -286,6 +285,7 @@ unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
sss = _mm_packs_epi32(sss, sss);
|
||||
let pixel: u32 = transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss, sss)));
|
||||
let bytes = pixel.to_le_bytes();
|
||||
dst_row.get_unchecked_mut(dst_x).0 = [bytes[0], bytes[1], bytes[2]];
|
||||
let dst_pixel = dst_row.get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = [bytes[0], bytes[1], bytes[2]];
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
use std::arch::wasm32::*;
|
||||
use std::intrinsics::transmute;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer16;
|
||||
use crate::pixels::U8x3;
|
||||
use crate::wasm32_utils;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
@@ -11,18 +11,15 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U8x3>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x3>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision() as u32;
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, precision);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -31,7 +28,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, precision);
|
||||
horiz_convolution_one_row(src_row, dst_row, normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -47,10 +44,10 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U8x3]; 4],
|
||||
dst_rows: [&mut [U8x3]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
precision: u32,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
const ZERO: v128 = i64x2(0, 0);
|
||||
let precision = normalizer.precision() as u32;
|
||||
let initial = i32x4_splat(1 << (precision - 1));
|
||||
let src_width = src_rows[0].len();
|
||||
|
||||
@@ -81,12 +78,12 @@ unsafe fn horiz_convolution_four_rows(
|
||||
6, -1, 9, -1, 7, -1, 10, -1, 8, -1, 11, -1, -1, -1, -1, -1
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
for (dst_x, coeffs_chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let x_start = coeffs_chunk.start as usize;
|
||||
let mut x = x_start;
|
||||
|
||||
let mut sss_a = [initial; 4];
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
// Next block of code will be load source pixels by 16 bytes per time.
|
||||
// We must guarantee what this process will not go beyond
|
||||
@@ -174,8 +171,7 @@ unsafe fn horiz_convolution_four_rows(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U8x3],
|
||||
dst_row: &mut [U8x3],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
precision: u32,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
#[rustfmt::skip]
|
||||
const PIX_SH1: v128 = i8x16(
|
||||
@@ -206,13 +202,14 @@ unsafe fn horiz_convolution_one_row(
|
||||
G: |-1 04| |-1 01|
|
||||
R: |-1 03| |-1 00|
|
||||
*/
|
||||
let precision = normalizer.precision() as u32;
|
||||
let src_width = src_row.len();
|
||||
let initial = i32x4_splat(1 << (precision - 1));
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
for (dst_x, coeffs_chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let x_start = coeffs_chunk.start as usize;
|
||||
let mut x = x_start;
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
let mut sss = initial;
|
||||
|
||||
// Next block of code will be load source pixels by 16 bytes per time.
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
use std::arch::x86_64::*;
|
||||
use std::intrinsics::transmute;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer16;
|
||||
use crate::pixels::U8x4;
|
||||
use crate::{simd_utils, ImageView, ImageViewMut};
|
||||
|
||||
@@ -13,9 +13,8 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U8x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x4>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
|
||||
macro_rules! call {
|
||||
@@ -30,15 +29,14 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
src_view: &impl ImageView<Pixel = U8x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x4>,
|
||||
offset: u32,
|
||||
normalizer: optimisations::Normalizer16,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &normalizer);
|
||||
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -47,7 +45,7 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &normalizer);
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -63,7 +61,7 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
src_rows: [&[U8x4]; 4],
|
||||
dst_rows: [&mut [U8x4]; 4],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let zero = _mm256_setzero_si256();
|
||||
let initial = _mm256_set1_epi32(1 << (PRECISION - 1));
|
||||
@@ -79,7 +77,7 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
|
||||
);
|
||||
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
let coefficients_chunks = normalizer.chunks();
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x = coeffs_chunk.start as usize;
|
||||
@@ -185,7 +183,7 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
src_row: &[U8x4],
|
||||
dst_row: &mut [U8x4],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
#[rustfmt::skip]
|
||||
let sh1 = _mm256_set_epi8(
|
||||
@@ -219,7 +217,7 @@ unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
);
|
||||
let sh7 = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
|
||||
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
let coefficients_chunks = normalizer.chunks();
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x = coeffs_chunk.start as usize;
|
||||
|
||||
+52
-14
@@ -1,9 +1,9 @@
|
||||
use super::{Coefficients, Convolution};
|
||||
use crate::convolution::optimisations::Normalizer16;
|
||||
use crate::convolution::vertical_u8::vert_convolution_u8;
|
||||
use crate::pixels::U8x4;
|
||||
use crate::{CpuExtensions, ImageView, ImageViewMut};
|
||||
|
||||
use super::{Coefficients, Convolution};
|
||||
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod avx2;
|
||||
mod native;
|
||||
@@ -14,7 +14,9 @@ mod sse4;
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
mod wasm32;
|
||||
|
||||
impl Convolution for U8x4 {
|
||||
type P = U8x4;
|
||||
|
||||
impl Convolution for P {
|
||||
fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = Self>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
@@ -22,16 +24,19 @@ impl Convolution for U8x4 {
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
_ => native::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
debug_assert!(src_view.height() - offset >= dst_view.height());
|
||||
|
||||
let normalizer = Normalizer16::new(coeffs);
|
||||
let normalizer_ref = &normalizer;
|
||||
|
||||
try_process_in_threads_h! {
|
||||
horiz_convolution(
|
||||
src_view,
|
||||
dst_view,
|
||||
offset,
|
||||
normalizer_ref,
|
||||
cpu_extensions,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -42,6 +47,39 @@ impl Convolution for U8x4 {
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
vert_convolution_u8(src_view, dst_view, offset, coeffs, cpu_extensions);
|
||||
debug_assert!(src_view.width() - offset >= dst_view.width());
|
||||
|
||||
let normalizer = Normalizer16::new(coeffs);
|
||||
let normalizer_ref = &normalizer;
|
||||
|
||||
try_process_in_threads_v! {
|
||||
vert_convolution_u8(
|
||||
src_view,
|
||||
dst_view,
|
||||
offset,
|
||||
normalizer_ref,
|
||||
cpu_extensions,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = P>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = P>,
|
||||
offset: u32,
|
||||
normalizer: &Normalizer16,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
_ => native::horiz_convolution(src_view, dst_view, offset, normalizer),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer16;
|
||||
use crate::image_view::{ImageView, ImageViewMut};
|
||||
use crate::pixels::U8x4;
|
||||
|
||||
@@ -7,20 +7,20 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U8x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x4>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients = normalizer.coefficients();
|
||||
let initial = 1 << (precision - 1);
|
||||
|
||||
let coefficients = normalizer.chunks();
|
||||
let src_rows = src_view.iter_rows(offset);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
for (dst_row, src_row) in dst_rows.zip(src_rows) {
|
||||
for (chunk, dst_pixel) in coefficients.iter().zip(dst_row.iter_mut()) {
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
for (coeffs_chunk, dst_pixel) in coefficients.iter().zip(dst_row.iter_mut()) {
|
||||
let first_x_src = coeffs_chunk.start as usize;
|
||||
let mut ss = [initial; 4];
|
||||
let src_pixels = unsafe { src_row.get_unchecked(chunk.start as usize..) };
|
||||
for (&k, &src_pixel) in chunk.values().iter().zip(src_pixels) {
|
||||
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
|
||||
for (&k, &src_pixel) in coeffs_chunk.values().iter().zip(src_pixels) {
|
||||
for (i, s) in ss.iter_mut().enumerate() {
|
||||
*s += src_pixel.0[i] as i32 * (k as i32);
|
||||
}
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use std::arch::aarch64::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::{CoefficientsI16Chunk, Normalizer16};
|
||||
use crate::neon_utils;
|
||||
use crate::pixels::U8x4;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
@@ -10,9 +10,8 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U8x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x4>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
|
||||
macro_rules! call {
|
||||
@@ -27,9 +26,9 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
src_view: &impl ImageView<Pixel = U8x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x4>,
|
||||
offset: u32,
|
||||
normalizer: optimisations::Normalizer16,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let coefficients_chunks = normalizer.chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
@@ -60,7 +59,7 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
src_rows: [&[U8x4]; 4],
|
||||
dst_rows: [&mut [U8x4]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
coefficients_chunks: &[CoefficientsI16Chunk],
|
||||
) {
|
||||
let initial = vdupq_n_s32(1 << (PRECISION - 1));
|
||||
let zero_u8x16 = vdupq_n_u8(0);
|
||||
@@ -71,7 +70,7 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
|
||||
let mut sss_a = [initial; 4];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
coeffs = coeffs_by_8.remainder();
|
||||
@@ -204,16 +203,16 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
src_row: &[U8x4],
|
||||
dst_row: &mut [U8x4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
coefficients_chunks: &[CoefficientsI16Chunk],
|
||||
) {
|
||||
let initial = vdupq_n_s32(1 << (PRECISION - 1));
|
||||
let zero_u8x16 = vdupq_n_u8(0);
|
||||
let zero_u8x8 = vdup_n_u8(0);
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut sss = initial;
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
coeffs = coeffs_by_8.remainder();
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
use std::arch::x86_64::*;
|
||||
use std::intrinsics::transmute;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer16;
|
||||
use crate::pixels::U8x4;
|
||||
use crate::{simd_utils, ImageView, ImageViewMut};
|
||||
|
||||
@@ -13,9 +13,8 @@ pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U8x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x4>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
|
||||
macro_rules! call {
|
||||
@@ -30,7 +29,7 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
src_view: &impl ImageView<Pixel = U8x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x4>,
|
||||
offset: u32,
|
||||
normalizer: optimisations::Normalizer16,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
@@ -38,7 +37,7 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &normalizer);
|
||||
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -47,7 +46,7 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &normalizer);
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -62,15 +61,15 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
src_rows: [&[U8x4]; 4],
|
||||
dst_rows: [&mut [U8x4]; 4],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let initial = _mm_set1_epi32(1 << (PRECISION - 1));
|
||||
let mask_lo = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
|
||||
let mask_hi = _mm_set_epi8(-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8);
|
||||
let mask = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
|
||||
|
||||
for (dst_x, chunk) in normalizer.coefficients().iter().enumerate() {
|
||||
let mut x: usize = chunk.start as usize;
|
||||
for (dst_x, chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x = chunk.start as usize;
|
||||
|
||||
let mut sss0 = initial;
|
||||
let mut sss1 = initial;
|
||||
@@ -168,13 +167,13 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
sss2 = _mm_packs_epi32(sss2, sss2);
|
||||
sss3 = _mm_packs_epi32(sss3, sss3);
|
||||
*dst_rows[0].get_unchecked_mut(dst_x) =
|
||||
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss0, sss0)));
|
||||
transmute::<i32, U8x4>(_mm_cvtsi128_si32(_mm_packus_epi16(sss0, sss0)));
|
||||
*dst_rows[1].get_unchecked_mut(dst_x) =
|
||||
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss1, sss1)));
|
||||
transmute::<i32, U8x4>(_mm_cvtsi128_si32(_mm_packus_epi16(sss1, sss1)));
|
||||
*dst_rows[2].get_unchecked_mut(dst_x) =
|
||||
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss2, sss2)));
|
||||
transmute::<i32, U8x4>(_mm_cvtsi128_si32(_mm_packus_epi16(sss2, sss2)));
|
||||
*dst_rows[3].get_unchecked_mut(dst_x) =
|
||||
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss3, sss3)));
|
||||
transmute::<i32, U8x4>(_mm_cvtsi128_si32(_mm_packus_epi16(sss3, sss3)));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -187,7 +186,7 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
src_row: &[U8x4],
|
||||
dst_row: &mut [U8x4],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let initial = _mm_set1_epi32(1 << (PRECISION - 1));
|
||||
let sh1 = _mm_set_epi8(-1, 11, -1, 3, -1, 10, -1, 2, -1, 9, -1, 1, -1, 8, -1, 0);
|
||||
@@ -200,8 +199,8 @@ unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
);
|
||||
let sh7 = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
|
||||
|
||||
for (dst_x, chunk) in normalizer.coefficients().iter().enumerate() {
|
||||
let mut x: usize = chunk.start as usize;
|
||||
for (dst_x, chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x = chunk.start as usize;
|
||||
let mut sss = initial;
|
||||
|
||||
let coeffs_by_8 = chunk.values().chunks_exact(8);
|
||||
|
||||
@@ -1,31 +1,25 @@
|
||||
use std::arch::wasm32::*;
|
||||
use std::intrinsics::transmute;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer16;
|
||||
use crate::pixels::U8x4;
|
||||
use crate::wasm32_utils;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
// This code is based on C-implementation from Pillow-SIMD package for Python
|
||||
// https://github.com/uploadcare/pillow-simd
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = U8x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x4>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision() as u32;
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, precision);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -34,7 +28,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, precision);
|
||||
horiz_convolution_one_row(src_row, dst_row, normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -49,15 +43,15 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U8x4]; 4],
|
||||
dst_rows: [&mut [U8x4]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
precision: u32,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let precision = normalizer.precision() as u32;
|
||||
let initial = i32x4_splat(1 << (precision - 1));
|
||||
const MASK_LO: v128 = i8x16(0, -1, 4, -1, 1, -1, 5, -1, 2, -1, 6, -1, 3, -1, 7, -1);
|
||||
const MASK_HI: v128 = i8x16(8, -1, 12, -1, 9, -1, 13, -1, 10, -1, 14, -1, 11, -1, 15, -1);
|
||||
const MASK: v128 = i8x16(0, -1, 4, -1, 1, -1, 5, -1, 2, -1, 6, -1, 3, -1, 7, -1);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
for (dst_x, coeffs_chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
|
||||
let mut sss0 = initial;
|
||||
@@ -65,7 +59,7 @@ unsafe fn horiz_convolution_four_rows(
|
||||
let mut sss2 = initial;
|
||||
let mut sss3 = initial;
|
||||
|
||||
let coeffs = coeffs_chunk.values;
|
||||
let coeffs = coeffs_chunk.values();
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
let reminder1 = coeffs_by_4.remainder();
|
||||
|
||||
@@ -176,9 +170,9 @@ unsafe fn horiz_convolution_four_rows(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U8x4],
|
||||
dst_row: &mut [U8x4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
precision: u32,
|
||||
normalizer: &Normalizer16,
|
||||
) {
|
||||
let precision = normalizer.precision() as u32;
|
||||
let initial = i32x4_splat(1 << (precision - 1));
|
||||
const SH1: v128 = i8x16(0, -1, 8, -1, 1, -1, 9, -1, 2, -1, 10, -1, 3, -1, 11, -1);
|
||||
const SH2: v128 = i8x16(0, 1, 4, 5, 0, 1, 4, 5, 0, 1, 4, 5, 0, 1, 4, 5);
|
||||
@@ -190,11 +184,11 @@ unsafe fn horiz_convolution_one_row(
|
||||
);
|
||||
const SH7: v128 = i8x16(0, -1, 4, -1, 1, -1, 5, -1, 2, -1, 6, -1, 3, -1, 7, -1);
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
for (dst_x, coeffs_chunk) in normalizer.chunks().iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut sss = initial;
|
||||
|
||||
let coeffs_by_8 = coeffs_chunk.values.chunks_exact(8);
|
||||
let coeffs_by_8 = coeffs_chunk.values().chunks_exact(8);
|
||||
let reminder8 = coeffs_by_8.remainder();
|
||||
|
||||
for k in coeffs_by_8 {
|
||||
|
||||
@@ -10,7 +10,7 @@ pub(crate) fn vert_convolution<T>(
|
||||
src_view: &impl ImageView<Pixel = T>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = T>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
coeffs: &Coefficients,
|
||||
) where
|
||||
T: InnerPixel<Component = f32>,
|
||||
{
|
||||
|
||||
@@ -16,7 +16,7 @@ pub(crate) fn vert_convolution_f32<T: InnerPixel<Component = f32>>(
|
||||
src_view: &impl ImageView<Pixel = T>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = T>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
coeffs: &Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
// Check safety conditions
|
||||
|
||||
@@ -8,7 +8,7 @@ pub(crate) fn vert_convolution<T>(
|
||||
src_view: &impl ImageView<Pixel = T>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = T>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
coeffs: &Coefficients,
|
||||
) where
|
||||
T: InnerPixel<Component = f32>,
|
||||
{
|
||||
|
||||
@@ -10,7 +10,7 @@ pub(crate) fn vert_convolution<T>(
|
||||
src_view: &impl ImageView<Pixel = T>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = T>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
coeffs: &Coefficients,
|
||||
) where
|
||||
T: InnerPixel<Component = f32>,
|
||||
{
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use std::arch::x86_64::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::{CoefficientsI32Chunk, Normalizer32};
|
||||
use crate::pixels::InnerPixel;
|
||||
use crate::{simd_utils, ImageView, ImageViewMut};
|
||||
|
||||
@@ -8,18 +8,17 @@ pub(crate) fn vert_convolution<T>(
|
||||
src_view: &impl ImageView<Pixel = T>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = T>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer32,
|
||||
) where
|
||||
T: InnerPixel<Component = u16>,
|
||||
{
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
let coefficients_chunks = normalizer.chunks();
|
||||
let src_x = offset as usize * T::count_of_components();
|
||||
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
for (dst_row, coeffs_chunk) in dst_rows.zip(coefficients_chunks) {
|
||||
unsafe {
|
||||
vert_convolution_into_one_row_u16(src_view, dst_row, src_x, coeffs_chunk, &normalizer);
|
||||
vert_convolution_into_one_row_u16(src_view, dst_row, src_x, coeffs_chunk, normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -29,8 +28,8 @@ pub(crate) unsafe fn vert_convolution_into_one_row_u16<T>(
|
||||
src_view: &impl ImageView<Pixel = T>,
|
||||
dst_row: &mut [T],
|
||||
mut src_x: usize,
|
||||
coeffs_chunk: &optimisations::CoefficientsI32Chunk,
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
coeffs_chunk: &CoefficientsI32Chunk,
|
||||
normalizer: &Normalizer32,
|
||||
) where
|
||||
T: InnerPixel<Component = u16>,
|
||||
{
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
use crate::convolution::Coefficients;
|
||||
use crate::convolution::optimisations::Normalizer32;
|
||||
use crate::pixels::InnerPixel;
|
||||
use crate::{CpuExtensions, ImageView, ImageViewMut};
|
||||
|
||||
@@ -16,22 +16,22 @@ pub(crate) fn vert_convolution_u16<T: InnerPixel<Component = u16>>(
|
||||
src_view: &impl ImageView<Pixel = T>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = T>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer32,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
// Check safety conditions
|
||||
debug_assert!(src_view.width() - offset >= dst_view.width());
|
||||
debug_assert_eq!(coeffs.bounds.len(), dst_view.height() as usize);
|
||||
debug_assert_eq!(normalizer.chunks_len(), dst_view.height() as usize);
|
||||
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::vert_convolution(src_view, dst_view, offset, coeffs),
|
||||
CpuExtensions::Avx2 => avx2::vert_convolution(src_view, dst_view, offset, normalizer),
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => sse4::vert_convolution(src_view, dst_view, offset, coeffs),
|
||||
CpuExtensions::Sse4_1 => sse4::vert_convolution(src_view, dst_view, offset, normalizer),
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => neon::vert_convolution(src_view, dst_view, offset, coeffs),
|
||||
CpuExtensions::Neon => neon::vert_convolution(src_view, dst_view, offset, normalizer),
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => wasm32::vert_convolution(src_view, dst_view, offset, coeffs),
|
||||
_ => native::vert_convolution(src_view, dst_view, offset, coeffs),
|
||||
CpuExtensions::Simd128 => wasm32::vert_convolution(src_view, dst_view, offset, normalizer),
|
||||
_ => native::vert_convolution(src_view, dst_view, offset, normalizer),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::convolution::optimisations::Normalizer32;
|
||||
use crate::pixels::InnerPixel;
|
||||
use crate::utils::foreach_with_pre_reading;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
@@ -8,18 +8,17 @@ pub(crate) fn vert_convolution<T>(
|
||||
src_view: &impl ImageView<Pixel = T>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = T>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
normalizer: &Normalizer32,
|
||||
) where
|
||||
T: InnerPixel<Component = u16>,
|
||||
{
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
let coefficients_chunks = normalizer.chunks();
|
||||
let precision = normalizer.precision();
|
||||
let initial: i64 = 1 << (precision - 1);
|
||||
let src_x_initial = offset as usize * T::count_of_components();
|
||||
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
let coeffs_chunks_iter = coefficients_chunks.into_iter();
|
||||
let coeffs_chunks_iter = coefficients_chunks.iter();
|
||||
for (coeffs_chunk, dst_row) in coeffs_chunks_iter.zip(dst_rows) {
|
||||
let first_y_src = coeffs_chunk.start;
|
||||
let ks = coeffs_chunk.values();
|
||||
@@ -29,7 +28,7 @@ pub(crate) fn vert_convolution<T>(
|
||||
let (_, dst_chunks, tail) = unsafe { dst_components.align_to_mut::<[u16; 16]>() };
|
||||
x_src = convolution_by_chunks(
|
||||
src_view,
|
||||
&normalizer,
|
||||
normalizer,
|
||||
initial,
|
||||
dst_chunks,
|
||||
x_src,
|
||||
@@ -38,7 +37,7 @@ pub(crate) fn vert_convolution<T>(
|
||||
);
|
||||
|
||||
if !tail.is_empty() {
|
||||
convolution_by_u16(src_view, &normalizer, initial, tail, x_src, first_y_src, ks);
|
||||
convolution_by_u16(src_view, normalizer, initial, tail, x_src, first_y_src, ks);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -46,7 +45,7 @@ pub(crate) fn vert_convolution<T>(
|
||||
#[inline(always)]
|
||||
pub(crate) fn convolution_by_u16<T: InnerPixel<Component = u16>>(
|
||||
src_view: &impl ImageView<Pixel = T>,
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
normalizer: &Normalizer32,
|
||||
initial: i64,
|
||||
dst_components: &mut [u16],
|
||||
mut x_src: usize,
|
||||
@@ -72,7 +71,7 @@ pub(crate) fn convolution_by_u16<T: InnerPixel<Component = u16>>(
|
||||
#[inline(always)]
|
||||
fn convolution_by_chunks<T, const CHUNK_SIZE: usize>(
|
||||
src_view: &impl ImageView<Pixel = T>,
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
normalizer: &Normalizer32,
|
||||
initial: i64,
|
||||
dst_chunks: &mut [[u16; CHUNK_SIZE]],
|
||||
mut x_src: usize,
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user