mirror of
https://github.com/Cykooz/fast_image_resize.git
synced 2026-10-08 01:11:09 +00:00
Added support of new type of pixels PixelType::U16x2
This commit is contained in:
@@ -1,3 +1,8 @@
|
||||
## [Unreleased] - ReleaseDate
|
||||
|
||||
- Added support of new type of pixels `PixelType::U16x2`
|
||||
(e.g. luma with alpha channel).
|
||||
|
||||
## [0.9.3] - 2022-05-31
|
||||
|
||||
- Added support of new type of pixels `PixelType::U16`.
|
||||
|
||||
Generated
+149
-104
@@ -33,9 +33,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "anyhow"
|
||||
version = "1.0.57"
|
||||
version = "1.0.58"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "08f9b8508dccb7687a1d6c4ce66b2b0ecef467c94667de27d8d7fe1f8d2a9cdc"
|
||||
checksum = "bb07d2053ccdbe10e2af2995a2f116c1330396493dc1269f6a91d0ae82e19704"
|
||||
|
||||
[[package]]
|
||||
name = "argh"
|
||||
@@ -104,9 +104,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "bumpalo"
|
||||
version = "3.9.1"
|
||||
version = "3.10.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "a4a45a46ab1f2412e53d3a0ade76ffad2025804294569aae387231a0cd6e0899"
|
||||
checksum = "37ccbd214614c6783386c1af30caf03192f17891059cecc394b4fb119e363de3"
|
||||
|
||||
[[package]]
|
||||
name = "bytemuck"
|
||||
@@ -165,6 +165,15 @@ version = "1.1.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3d7b894f5411737b7867f4827955924d7c254fc9f4d91a6aad6b097804b1018b"
|
||||
|
||||
[[package]]
|
||||
name = "coolor"
|
||||
version = "0.5.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "af4d7a805ca0d92f8c61a31c809d4323fdaa939b0b440e544d21db7797c5aaad"
|
||||
dependencies = [
|
||||
"crossterm",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "crc32fast"
|
||||
version = "1.3.2"
|
||||
@@ -190,9 +199,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "crossbeam-channel"
|
||||
version = "0.5.4"
|
||||
version = "0.5.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "5aaa7bd5fb665c6864b5f963dd9097905c54125909c7aa94c9e18507cdbe6c53"
|
||||
checksum = "4c02a4d71819009c192cf4872265391563fd6a84c81ff2c0f2a7026ca4c1d85c"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"crossbeam-utils",
|
||||
@@ -211,15 +220,15 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "crossbeam-epoch"
|
||||
version = "0.9.8"
|
||||
version = "0.9.9"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "1145cf131a2c6ba0615079ab6a638f7e1973ac9c2634fcbeaaad6114246efe8c"
|
||||
checksum = "07db9d94cbd326813772c968ccd25999e5f8ae22f4f8d1b11effa37ef6ce281d"
|
||||
dependencies = [
|
||||
"autocfg",
|
||||
"cfg-if",
|
||||
"crossbeam-utils",
|
||||
"lazy_static",
|
||||
"memoffset",
|
||||
"once_cell",
|
||||
"scopeguard",
|
||||
]
|
||||
|
||||
@@ -235,35 +244,35 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "crossbeam-utils"
|
||||
version = "0.8.8"
|
||||
version = "0.8.9"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0bf124c720b7686e3c2663cf54062ab0f68a88af2fb6a030e87e30bf721fcb38"
|
||||
checksum = "8ff1f980957787286a554052d03c7aee98d99cc32e09f6d45f0a814133c87978"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"lazy_static",
|
||||
"once_cell",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "crossterm"
|
||||
version = "0.19.0"
|
||||
version = "0.23.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "7c36c10130df424b2f3552fcc2ddcd9b28a27b1e54b358b45874f88d1ca6888c"
|
||||
checksum = "a2102ea4f781910f8a5b98dd061f4c2023f479ce7bb1236330099ceb5a93cf17"
|
||||
dependencies = [
|
||||
"bitflags",
|
||||
"crossterm_winapi",
|
||||
"lazy_static",
|
||||
"libc",
|
||||
"mio",
|
||||
"parking_lot",
|
||||
"signal-hook",
|
||||
"signal-hook-mio",
|
||||
"winapi",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "crossterm_winapi"
|
||||
version = "0.7.0"
|
||||
version = "0.9.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0da8964ace4d3e4a044fd027919b2237000b24315a37c916f61809f1ff2140b9"
|
||||
checksum = "2ae1b35a484aa10e07fe0638d02301c5ad24de82d310ccbd2f3693da5f09bf1c"
|
||||
dependencies = [
|
||||
"winapi",
|
||||
]
|
||||
@@ -292,9 +301,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "csv2svg"
|
||||
version = "0.1.6"
|
||||
version = "0.1.7"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "a8e2619b51a7edc3570447c19b821b4f8e17b4e0ddab688582f6ceb2a83acf07"
|
||||
checksum = "5eede96fa4be061f1b49b92c756f8d150de9fbb5b11e09c6c2b3d670e32fd9f5"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"argh",
|
||||
@@ -410,21 +419,19 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "flate2"
|
||||
version = "1.0.23"
|
||||
version = "1.0.24"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b39522e96686d38f4bc984b9198e3a0613264abaebaff2c5c918bfa6b6da09af"
|
||||
checksum = "f82b0f4c27ad9f8bfd1f3208d882da2b09c301bc1c828fd3a00d0216d2fbbff6"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"crc32fast",
|
||||
"libc",
|
||||
"miniz_oxide",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "flume"
|
||||
version = "0.10.12"
|
||||
version = "0.10.13"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "843c03199d0c0ca54bc1ea90ac0d507274c28abcc4f691ae8b4eaa375087c76a"
|
||||
checksum = "1ceeb589a3157cac0ab8cc585feb749bd2cea5cb55a6ee802ad72d9fd38303da"
|
||||
dependencies = [
|
||||
"futures-core",
|
||||
"futures-sink",
|
||||
@@ -457,14 +464,14 @@ checksum = "21163e139fa306126e6eedaf49ecdb4588f939600f0b1e770f4205ee4b7fa868"
|
||||
|
||||
[[package]]
|
||||
name = "getrandom"
|
||||
version = "0.2.6"
|
||||
version = "0.2.7"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9be70c98951c83b8d2f8f60d7065fa6d5146873094452a1008da8c2f1e4205ad"
|
||||
checksum = "4eb1a864a501629691edf6c15a593b7a51eebaa1e8468e9ddc623de7c9b58ec6"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"js-sys",
|
||||
"libc",
|
||||
"wasi",
|
||||
"wasi 0.11.0+wasi-snapshot-preview1",
|
||||
"wasm-bindgen",
|
||||
]
|
||||
|
||||
@@ -480,9 +487,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "git2"
|
||||
version = "0.13.25"
|
||||
version = "0.14.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f29229cc1b24c0e6062f6e742aa3e256492a5323365e5ed3413599f8a5eff7d6"
|
||||
checksum = "d0155506aab710a86160ddb504a480d2964d7ab5b9e62419be69e0032bc5931c"
|
||||
dependencies = [
|
||||
"bitflags",
|
||||
"libc",
|
||||
@@ -493,9 +500,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "glassbench"
|
||||
version = "0.3.1"
|
||||
version = "0.3.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "18232f8d0a65b776d32b887f2aa3ab0230a761ea2c82e4739ae824ec5f2a27a9"
|
||||
checksum = "b56c8dd194e65fc268f379096640af9443f5f79386392131b75d66faff5d59f4"
|
||||
dependencies = [
|
||||
"base64",
|
||||
"chrono",
|
||||
@@ -503,7 +510,6 @@ dependencies = [
|
||||
"csv2svg",
|
||||
"git2",
|
||||
"lazy_static",
|
||||
"minimad",
|
||||
"open",
|
||||
"rusqlite",
|
||||
"serde",
|
||||
@@ -646,9 +652,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "js-sys"
|
||||
version = "0.3.57"
|
||||
version = "0.3.58"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "671a26f820db17c2a2750743f1dd03bafd15b98c9f30c7c2628c024c05d73397"
|
||||
checksum = "c3fac17f7123a73ca62df411b1bf727ccc805daa070338fda671c86dac1bdc27"
|
||||
dependencies = [
|
||||
"wasm-bindgen",
|
||||
]
|
||||
@@ -673,9 +679,9 @@ checksum = "349d5a591cd28b49e1d1037471617a32ddcda5731b99419008085f72d5a53836"
|
||||
|
||||
[[package]]
|
||||
name = "libgit2-sys"
|
||||
version = "0.12.26+1.3.0"
|
||||
version = "0.13.4+1.4.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "19e1c899248e606fbfe68dcb31d8b0176ebab833b103824af31bddf4b7457494"
|
||||
checksum = "d0fa6563431ede25f5cc7f6d803c6afbc1c5d3ad3d4925d12c882bf2b526f5d1"
|
||||
dependencies = [
|
||||
"cc",
|
||||
"libc",
|
||||
@@ -748,42 +754,32 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "minimad"
|
||||
version = "0.7.1"
|
||||
version = "0.9.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "c6176dd156ab0154ee95322efe2da93796cc338a8a63a6338d64c79419bcbe13"
|
||||
checksum = "cd37b2e65fbd459544194d8f52ed84027e031684335a062c708774c09d172b0b"
|
||||
dependencies = [
|
||||
"lazy_static",
|
||||
"once_cell",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "miniz_oxide"
|
||||
version = "0.5.1"
|
||||
version = "0.5.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d2b29bd4bc3f33391105ebee3589c19197c4271e3e5a9ec9bfe8127eeff8f082"
|
||||
checksum = "6f5c75688da582b8ffc1f1799e9db273f32133c49e048f614d22ec3256773ccc"
|
||||
dependencies = [
|
||||
"adler",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "mio"
|
||||
version = "0.7.14"
|
||||
version = "0.8.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "8067b404fe97c70829f082dec8bcf4f71225d7eaea1d8645349cb76fa06205cc"
|
||||
checksum = "57ee1c23c7c63b0c9250c339ffdc69255f110b298b901b9f6c82547b7b87caaf"
|
||||
dependencies = [
|
||||
"libc",
|
||||
"log",
|
||||
"miow",
|
||||
"ntapi",
|
||||
"winapi",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "miow"
|
||||
version = "0.3.7"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b9f1c5b025cda876f66ef43a113f91ebc9f4ccef34843000e0adf6ebbab84e21"
|
||||
dependencies = [
|
||||
"winapi",
|
||||
"wasi 0.11.0+wasi-snapshot-preview1",
|
||||
"windows-sys",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -806,15 +802,6 @@ dependencies = [
|
||||
"libc",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "ntapi"
|
||||
version = "0.3.7"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "c28774a7fd2fbb4f0babd8237ce554b73af68021b5f695a3cebd6c59bac0980f"
|
||||
dependencies = [
|
||||
"winapi",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "num-integer"
|
||||
version = "0.1.45"
|
||||
@@ -884,27 +871,25 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "parking_lot"
|
||||
version = "0.11.2"
|
||||
version = "0.12.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "7d17b78036a60663b797adeaee46f5c9dfebb86948d1255007a1d6be0271ff99"
|
||||
checksum = "3742b2c103b9f06bc9fff0a37ff4912935851bee6d36f3c02bcc755bcfec228f"
|
||||
dependencies = [
|
||||
"instant",
|
||||
"lock_api",
|
||||
"parking_lot_core",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "parking_lot_core"
|
||||
version = "0.8.5"
|
||||
version = "0.9.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d76e8e1493bcac0d2766c42737f34458f1c8c50c0d23bcb24ea953affb273216"
|
||||
checksum = "09a279cbf25cb0757810394fbc1e359949b59e348145c643a939a525692e6929"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"instant",
|
||||
"libc",
|
||||
"redox_syscall",
|
||||
"smallvec",
|
||||
"winapi",
|
||||
"windows-sys",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -959,18 +944,18 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "proc-macro2"
|
||||
version = "1.0.39"
|
||||
version = "1.0.40"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "c54b25569025b7fc9651de43004ae593a75ad88543b17178aa5e1b9c4f15f56f"
|
||||
checksum = "dd96a1e8ed2596c337f8eae5f24924ec83f5ad5ab21ea8e455d3566c69fbcaf7"
|
||||
dependencies = [
|
||||
"unicode-ident",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "quote"
|
||||
version = "1.0.18"
|
||||
version = "1.0.20"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "a1feb54ed693b93a84e14094943b84b7c4eae204c512b7ccb95ab0c66d278ad1"
|
||||
checksum = "3bcdf212e9776fbcb2d23ab029360416bb1706b1aea2d1a5ba002727cbcab804"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
]
|
||||
@@ -1036,9 +1021,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "resize"
|
||||
version = "0.7.2"
|
||||
version = "0.7.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "a2a32135a3cd4df690ee74dc0e1a77ffe2cdd4611b4e58b84df26a856d92b4ef"
|
||||
checksum = "d05ed0e778666d123be79444a5ddb81506fda18ea6d7c05a8b6a701a6215d310"
|
||||
dependencies = [
|
||||
"fallible_collections",
|
||||
"rgb",
|
||||
@@ -1046,9 +1031,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rgb"
|
||||
version = "0.8.32"
|
||||
version = "0.8.33"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e74fdc210d8f24a7dbfedc13b04ba5764f5232754ccebfdf5fff1bad791ccbc6"
|
||||
checksum = "c3b221de559e4a29df3b957eec92bc0de6bc8eaf6ca9cfed43e5e1d67ff65a34"
|
||||
dependencies = [
|
||||
"bytemuck",
|
||||
]
|
||||
@@ -1119,13 +1104,23 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "signal-hook"
|
||||
version = "0.1.17"
|
||||
version = "0.3.14"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "7e31d442c16f047a671b5a71e2161d6e68814012b7f5379d269ebd915fac2729"
|
||||
checksum = "a253b5e89e2698464fc26b545c9edceb338e18a89effeeecfea192c3025be29d"
|
||||
dependencies = [
|
||||
"libc",
|
||||
"signal-hook-registry",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "signal-hook-mio"
|
||||
version = "0.2.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "29ad2e15f37ec9a6cc544097b78a1ec90001e9f71b81338ca39f430adaca99af"
|
||||
dependencies = [
|
||||
"libc",
|
||||
"mio",
|
||||
"signal-hook-registry",
|
||||
"signal-hook",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -1160,9 +1155,9 @@ checksum = "3bdb25a4593d6656239319426f4025f7a658157e25e89f0e0319d7516d46042d"
|
||||
|
||||
[[package]]
|
||||
name = "syn"
|
||||
version = "1.0.95"
|
||||
version = "1.0.98"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "fbaf6116ab8924f39d52792136fb74fd60a80194cf1b1c6ffa6453eef1c3f942"
|
||||
checksum = "c50aef8a904de4c23c788f104b7dddc7d6f79c647c7c8ce4cc8f73eb0ca773dd"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
@@ -1185,13 +1180,13 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "termimad"
|
||||
version = "0.10.3"
|
||||
version = "0.20.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "1a0cc62bee46f8532b579609b5b2ad3f9dc8e9f952dc783c7c882eed0c5ed785"
|
||||
checksum = "c8a16d7de8d4c97a4149cc3b9d3681c5dba36011c303745bb1af19636e89ba39"
|
||||
dependencies = [
|
||||
"coolor",
|
||||
"crossbeam",
|
||||
"crossterm",
|
||||
"lazy_static",
|
||||
"minimad",
|
||||
"thiserror",
|
||||
"unicode-width",
|
||||
@@ -1247,11 +1242,12 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "time"
|
||||
version = "0.1.43"
|
||||
version = "0.1.44"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ca8a50ef2360fbd1eeb0ecd46795a87a19024eb4b53c5dc916ca1fd95fe62438"
|
||||
checksum = "6db9e6914ab8b1ae1c260a4ae7a49b6c5611b40328a735b21862567685e73255"
|
||||
dependencies = [
|
||||
"libc",
|
||||
"wasi 0.10.0+wasi-snapshot-preview1",
|
||||
"winapi",
|
||||
]
|
||||
|
||||
@@ -1278,9 +1274,9 @@ checksum = "099b7128301d285f79ddd55b9a83d5e6b9e97c92e0ea0daebee7263e932de992"
|
||||
|
||||
[[package]]
|
||||
name = "unicode-ident"
|
||||
version = "1.0.0"
|
||||
version = "1.0.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d22af068fba1eb5edcb4aea19d382b2a3deb4c8f9d475c589b6ada9e0fd493ee"
|
||||
checksum = "5bd2fe26506023ed7b5e1e315add59d6f584c621d037f9368fea9cfb988f368c"
|
||||
|
||||
[[package]]
|
||||
name = "unicode-normalization"
|
||||
@@ -1329,15 +1325,21 @@ checksum = "49874b5167b65d7193b8aba1567f5c7d93d001cafc34600cee003eda787e483f"
|
||||
|
||||
[[package]]
|
||||
name = "wasi"
|
||||
version = "0.10.2+wasi-snapshot-preview1"
|
||||
version = "0.10.0+wasi-snapshot-preview1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "fd6fbd9a79829dd1ad0cc20627bf1ed606756a7f77edff7b66b7064f9cb327c6"
|
||||
checksum = "1a143597ca7c7793eff794def352d41792a93c481eb1042423ff7ff72ba2c31f"
|
||||
|
||||
[[package]]
|
||||
name = "wasi"
|
||||
version = "0.11.0+wasi-snapshot-preview1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9c8d87e72b64a3b4db28d11ce29237c246188f4f51057d65a7eab63b7987e423"
|
||||
|
||||
[[package]]
|
||||
name = "wasm-bindgen"
|
||||
version = "0.2.80"
|
||||
version = "0.2.81"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "27370197c907c55e3f1a9fbe26f44e937fe6451368324e009cba39e139dc08ad"
|
||||
checksum = "7c53b543413a17a202f4be280a7e5c62a1c69345f5de525ee64f8cfdbc954994"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"wasm-bindgen-macro",
|
||||
@@ -1345,9 +1347,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "wasm-bindgen-backend"
|
||||
version = "0.2.80"
|
||||
version = "0.2.81"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "53e04185bfa3a779273da532f5025e33398409573f348985af9a1cbf3774d3f4"
|
||||
checksum = "5491a68ab4500fa6b4d726bd67408630c3dbe9c4fe7bda16d5c82a1fd8c7340a"
|
||||
dependencies = [
|
||||
"bumpalo",
|
||||
"lazy_static",
|
||||
@@ -1360,9 +1362,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "wasm-bindgen-macro"
|
||||
version = "0.2.80"
|
||||
version = "0.2.81"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "17cae7ff784d7e83a2fe7611cfe766ecf034111b49deb850a3dc7699c08251f5"
|
||||
checksum = "c441e177922bc58f1e12c022624b6216378e5febc2f0533e41ba443d505b80aa"
|
||||
dependencies = [
|
||||
"quote",
|
||||
"wasm-bindgen-macro-support",
|
||||
@@ -1370,9 +1372,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "wasm-bindgen-macro-support"
|
||||
version = "0.2.80"
|
||||
version = "0.2.81"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "99ec0dc7a4756fffc231aab1b9f2f578d23cd391390ab27f952ae0c9b3ece20b"
|
||||
checksum = "7d94ac45fcf608c1f45ef53e748d35660f168490c10b23704c7779ab8f5c3048"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
@@ -1383,9 +1385,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "wasm-bindgen-shared"
|
||||
version = "0.2.80"
|
||||
version = "0.2.81"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d554b7f530dee5964d9a9468d95c1f8b8acae4f282807e7d27d4b03099a46744"
|
||||
checksum = "6a89911bd99e5f3659ec4acf9c4d93b0a90fe4a2a11f15328472058edc5261be"
|
||||
|
||||
[[package]]
|
||||
name = "weezl"
|
||||
@@ -1414,3 +1416,46 @@ name = "winapi-x86_64-pc-windows-gnu"
|
||||
version = "0.4.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "712e227841d057c1ee1cd2fb22fa7e5a5461ae8e48fa2ca79ec42cfc1931183f"
|
||||
|
||||
[[package]]
|
||||
name = "windows-sys"
|
||||
version = "0.36.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ea04155a16a59f9eab786fe12a4a450e75cdb175f9e0d80da1e17db09f55b8d2"
|
||||
dependencies = [
|
||||
"windows_aarch64_msvc",
|
||||
"windows_i686_gnu",
|
||||
"windows_i686_msvc",
|
||||
"windows_x86_64_gnu",
|
||||
"windows_x86_64_msvc",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "windows_aarch64_msvc"
|
||||
version = "0.36.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9bb8c3fd39ade2d67e9874ac4f3db21f0d710bee00fe7cab16949ec184eeaa47"
|
||||
|
||||
[[package]]
|
||||
name = "windows_i686_gnu"
|
||||
version = "0.36.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "180e6ccf01daf4c426b846dfc66db1fc518f074baa793aa7d9b9aaeffad6a3b6"
|
||||
|
||||
[[package]]
|
||||
name = "windows_i686_msvc"
|
||||
version = "0.36.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e2e7917148b2812d1eeafaeb22a97e4813dfa60a3f8f78ebe204bcc88f12f024"
|
||||
|
||||
[[package]]
|
||||
name = "windows_x86_64_gnu"
|
||||
version = "0.36.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "4dcd171b8776c41b97521e5da127a2d86ad280114807d0b2ab1e462bc764d9e1"
|
||||
|
||||
[[package]]
|
||||
name = "windows_x86_64_msvc"
|
||||
version = "0.36.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "c811ca4a8c853ef420abd8592ba53ddbbac90410fab6903b3e79972a631f7680"
|
||||
|
||||
+12
-3
@@ -25,10 +25,10 @@ thiserror = "1.0.31"
|
||||
|
||||
|
||||
[dev-dependencies]
|
||||
glassbench = "0.3.1"
|
||||
glassbench = "0.3.3"
|
||||
image = "0.24.2"
|
||||
resize = "0.7.2"
|
||||
rgb = "0.8.32"
|
||||
resize = "0.7.3"
|
||||
rgb = "0.8.33"
|
||||
png = "0.17.5"
|
||||
nix = { version = "0.24.1", default-features = false, features = ["sched"] }
|
||||
testing = {path= "testing" }
|
||||
@@ -74,6 +74,11 @@ name = "bench_compare_l16"
|
||||
harness = false
|
||||
|
||||
|
||||
[[bench]]
|
||||
name = "bench_compare_la16"
|
||||
harness = false
|
||||
|
||||
|
||||
[profile.dev.package.'*']
|
||||
opt-level = 3
|
||||
|
||||
@@ -85,6 +90,10 @@ opt-level = 3
|
||||
codegen-units = 1
|
||||
|
||||
|
||||
[profile.test]
|
||||
opt-level = 3
|
||||
|
||||
|
||||
[package.metadata.release]
|
||||
pre-release-replacements = [
|
||||
{file="CHANGELOG.md", search="Unreleased", replace="{{version}}"},
|
||||
|
||||
@@ -1,5 +1,9 @@
|
||||
# fast_image_resize
|
||||
|
||||
[](https://github.com/Cykooz/fast_image_resize)
|
||||
[](https://crates.io/crates/fast_image_resize)
|
||||
[](https://docs.rs/fast_image_resize)
|
||||
|
||||
Rust library for fast image resizing with using of SIMD instructions.
|
||||
|
||||
_Note: This library does not support converting image color spaces.
|
||||
@@ -23,29 +27,14 @@ Supported pixel formats and available optimisations:
|
||||
| I32 | One `i32` component per pixel | + | - | - |
|
||||
| F32 | One `f32` component per pixel | + | - | - |
|
||||
|
||||
## Benchmarks
|
||||
## Some benchmarks
|
||||
|
||||
Environment:
|
||||
[All benchmarks.](https://github.com/Cykooz/fast_image_resize/blob/main/benchmarks.md)
|
||||
|
||||
- CPU: AMD Ryzen 9 5950X
|
||||
- RAM: DDR4 3800 MHz
|
||||
- Ubuntu 22.04 (linux 5.15.0)
|
||||
- Rust 1.61.0
|
||||
- fast_image_resize = "0.9.3"
|
||||
- glassbench = "0.3.1"
|
||||
- `rustflags = ["-C", "llvm-args=-x86-branches-within-32B-boundaries"]`
|
||||
Rust libraries used to compare of resizing speed:
|
||||
|
||||
Other Rust libraries used to compare of resizing speed:
|
||||
|
||||
- image = "0.24.2" (<https://crates.io/crates/image>)
|
||||
- resize = "0.7.2" (<https://crates.io/crates/resize>)
|
||||
|
||||
Resize algorithms:
|
||||
|
||||
- Nearest
|
||||
- Convolution with Bilinear filter
|
||||
- Convolution with CatmullRom filter
|
||||
- Convolution with Lanczos3 filter
|
||||
- image (<https://crates.io/crates/image>)
|
||||
- resize (<https://crates.io/crates/resize>)
|
||||
|
||||
### Resize RGB8 image (U8x3) 4928x3279 => 852x567
|
||||
|
||||
@@ -58,11 +47,11 @@ Pipeline:
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|------------|:-------:|:--------:|:----------:|:--------:|
|
||||
| image | 18.25 | 79.66 | 138.48 | 189.00 |
|
||||
| resize | - | 50.41 | 97.00 | 142.89 |
|
||||
| fir rust | 0.26 | 37.79 | 64.21 | 93.78 |
|
||||
| fir sse4.1 | 0.26 | 26.07 | 39.62 | 53.22 |
|
||||
| fir avx2 | 0.26 | 6.95 | 8.87 | 12.68 |
|
||||
| image | 19.44 | 83.01 | 153.17 | 208.82 |
|
||||
| resize | - | 52.13 | 103.37 | 154.10 |
|
||||
| fir rust | 0.28 | 43.00 | 79.52 | 117.41 |
|
||||
| fir sse4.1 | 0.28 | 27.79 | 42.97 | 58.16 |
|
||||
| fir avx2 | 0.28 | 7.30 | 9.50 | 13.59 |
|
||||
|
||||
### Resize RGBA8 image (U8x4) 4928x3279 => 852x567
|
||||
|
||||
@@ -76,13 +65,13 @@ Pipeline:
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|------------|:-------:|:--------:|:----------:|:--------:|
|
||||
| image | 18.32 | 75.46 | 128.47 | 181.34 |
|
||||
| resize | - | 49.26 | 94.38 | 138.48 |
|
||||
| fir rust | 0.17 | 33.67 | 47.63 | 67.37 |
|
||||
| fir sse4.1 | 0.17 | 12.21 | 15.89 | 20.61 |
|
||||
| fir avx2 | 0.17 | 9.18 | 11.46 | 15.26 |
|
||||
| image | 19.73 | 82.34 | 141.74 | 198.86 |
|
||||
| resize | - | 49.91 | 100.27 | 148.99 |
|
||||
| fir rust | 0.18 | 36.84 | 52.31 | 74.99 |
|
||||
| fir sse4.1 | 0.18 | 13.21 | 17.26 | 22.42 |
|
||||
| fir avx2 | 0.18 | 9.47 | 12.03 | 16.08 |
|
||||
|
||||
### Resize grayscale image (U8) 4928x3279 => 852x567
|
||||
## Resize L8 (luma) image (U8) 4928x3279 => 852x567
|
||||
|
||||
Pipeline:
|
||||
|
||||
@@ -94,66 +83,11 @@ Pipeline:
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|------------|:-------:|:--------:|:----------:|:--------:|
|
||||
| image | 15.05 | 44.65 | 70.26 | 95.61 |
|
||||
| resize | - | 16.88 | 32.98 | 56.31 |
|
||||
| fir rust | 0.14 | 13.21 | 14.98 | 21.86 |
|
||||
| fir sse4.1 | 0.14 | 11.22 | 11.23 | 16.46 |
|
||||
| fir avx2 | 0.14 | 6.03 | 4.41 | 7.37 |
|
||||
|
||||
### Resize grayscale image with alpha channel (U8x2) 4928x3279 => 852x567
|
||||
|
||||
Pipeline:
|
||||
|
||||
`src_image => multiply by alpha => resize => divide by alpha => dst_image`
|
||||
|
||||
- Source image
|
||||
[nasa-4928x3279-rgba.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279-rgba.png)
|
||||
has converted into grayscale image with alpha channel (two bytes per pixel).
|
||||
- Numbers in table is mean duration of image resizing in milliseconds.
|
||||
- The `resize` crate does not support this pixel format.
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|------------|:-------:|:--------:|:----------:|:--------:|
|
||||
| image | 16.41 | 61.23 | 111.31 | 149.96 |
|
||||
| fir rust | 0.15 | 23.67 | 27.94 | 38.91 |
|
||||
| fir sse4.1 | 0.15 | 11.66 | 13.33 | 16.44 |
|
||||
| fir avx2 | 0.15 | 10.33 | 11.43 | 14.12 |
|
||||
|
||||
### Resize RGB16 image (U16x3) 4928x3279 => 852x567
|
||||
|
||||
Pipeline:
|
||||
|
||||
`src_image => resize => dst_image`
|
||||
|
||||
- Source image [nasa-4928x3279.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279.png)
|
||||
has converted into RGB16 image.
|
||||
- Numbers in table is mean duration of image resizing in milliseconds.
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|------------|:-------:|:--------:|:----------:|:--------:|
|
||||
| image | 17.54 | 72.24 | 126.80 | 176.03 |
|
||||
| resize | - | 52.51 | 100.40 | 147.31 |
|
||||
| fir rust | 0.31 | 39.56 | 65.65 | 92.90 |
|
||||
| fir sse4.1 | 0.31 | 22.14 | 35.69 | 50.40 |
|
||||
| fir avx2 | 0.31 | 19.12 | 28.25 | 33.86 |
|
||||
|
||||
### Resize grayscale image with 16 bits per pixel (U16) 4928x3279 => 852x567
|
||||
|
||||
Pipeline:
|
||||
|
||||
`src_image => resize => dst_image`
|
||||
|
||||
- Source image [nasa-4928x3279.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279.png)
|
||||
has converted into grayscale image with two bytes per pixel.
|
||||
- Numbers in table is mean duration of image resizing in milliseconds.
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|------------|:-------:|:--------:|:----------:|:--------:|
|
||||
| image | 15.32 | 44.77 | 70.51 | 97.25 |
|
||||
| resize | - | 15.59 | 30.51 | 52.94 |
|
||||
| fir rust | 0.16 | 16.88 | 26.23 | 35.32 |
|
||||
| fir sse4.1 | 0.16 | 7.19 | 12.13 | 17.53 |
|
||||
| fir avx2 | 0.16 | 6.56 | 8.79 | 13.39 |
|
||||
| image | 15.21 | 44.78 | 71.92 | 100.91 |
|
||||
| resize | - | 17.53 | 36.20 | 61.10 |
|
||||
| fir rust | 0.15 | 14.08 | 16.38 | 23.73 |
|
||||
| fir sse4.1 | 0.16 | 11.92 | 12.28 | 17.79 |
|
||||
| fir avx2 | 0.16 | 6.48 | 4.77 | 7.85 |
|
||||
|
||||
## Examples
|
||||
|
||||
@@ -185,10 +119,9 @@ fn main() {
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
// Create MulDiv instance
|
||||
let alpha_mul_div = fr::MulDiv::default();
|
||||
// Multiple RGB channels of source image by alpha channel
|
||||
// (not required for the Nearest algorithm)
|
||||
// (not required for the Nearest algorithm)
|
||||
let alpha_mul_div = fr::MulDiv::default();
|
||||
alpha_mul_div
|
||||
.multiply_alpha_inplace(&mut src_image.view_mut())
|
||||
.unwrap();
|
||||
|
||||
+23
-33
@@ -6,6 +6,8 @@ use fast_image_resize::MulDiv;
|
||||
use fast_image_resize::PixelType;
|
||||
use fast_image_resize::{CpuExtensions, Image};
|
||||
|
||||
mod utils;
|
||||
|
||||
// Multiplies by alpha
|
||||
|
||||
fn get_src_image(
|
||||
@@ -27,6 +29,7 @@ fn multiplies_alpha(bench: &mut Bench, pixel_type: PixelType, cpu_extensions: Cp
|
||||
let pixel: &[u8] = match pixel_type {
|
||||
PixelType::U8x4 => &[255, 128, 0, 128],
|
||||
PixelType::U8x2 => &[255, 128],
|
||||
PixelType::U16x2 => &[255, 255, 0, 128],
|
||||
_ => unreachable!(),
|
||||
};
|
||||
let src_data = get_src_image(width, height, pixel_type, pixel);
|
||||
@@ -56,6 +59,7 @@ fn divides_alpha(bench: &mut Bench, pixel_type: PixelType, cpu_extensions: CpuEx
|
||||
let pixel: &[u8] = match pixel_type {
|
||||
PixelType::U8x4 => &[128, 64, 0, 128],
|
||||
PixelType::U8x2 => &[128, 128],
|
||||
PixelType::U16x2 => &[0, 128, 0, 128],
|
||||
_ => unreachable!(),
|
||||
};
|
||||
let src_data = get_src_image(width, height, pixel_type, pixel);
|
||||
@@ -79,40 +83,26 @@ fn divides_alpha(bench: &mut Bench, pixel_type: PixelType, cpu_extensions: CpuEx
|
||||
);
|
||||
}
|
||||
|
||||
pub fn main() {
|
||||
// Pin process to #0 CPU core
|
||||
let mut cpu_set = nix::sched::CpuSet::new();
|
||||
cpu_set.set(0).unwrap();
|
||||
nix::sched::sched_setaffinity(nix::unistd::Pid::from_raw(0), &cpu_set).unwrap();
|
||||
|
||||
use glassbench::*;
|
||||
let name = env!("CARGO_CRATE_NAME");
|
||||
let cmd = Command::read();
|
||||
if cmd.include_bench(name) {
|
||||
let mut bench = create_bench(name, "Alpha", &cmd);
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
{
|
||||
multiplies_alpha(&mut bench, PixelType::U8x4, CpuExtensions::Avx2);
|
||||
multiplies_alpha(&mut bench, PixelType::U8x4, CpuExtensions::Sse4_1);
|
||||
multiplies_alpha(&mut bench, PixelType::U8x2, CpuExtensions::Avx2);
|
||||
multiplies_alpha(&mut bench, PixelType::U8x2, CpuExtensions::Sse4_1);
|
||||
fn bench_alpha(bench: &mut Bench) {
|
||||
let pixel_types = [PixelType::U8x4, PixelType::U8x2, PixelType::U16x2];
|
||||
let mut cpu_extensions = vec![CpuExtensions::None];
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
{
|
||||
cpu_extensions.push(CpuExtensions::Sse4_1);
|
||||
cpu_extensions.push(CpuExtensions::Avx2);
|
||||
}
|
||||
for pixel_type in pixel_types {
|
||||
for &extensions in cpu_extensions.iter() {
|
||||
println!("Mul {:?} {:?}", pixel_type, extensions);
|
||||
multiplies_alpha(bench, pixel_type, extensions);
|
||||
}
|
||||
multiplies_alpha(&mut bench, PixelType::U8x4, CpuExtensions::None);
|
||||
multiplies_alpha(&mut bench, PixelType::U8x2, CpuExtensions::None);
|
||||
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
{
|
||||
divides_alpha(&mut bench, PixelType::U8x4, CpuExtensions::Avx2);
|
||||
divides_alpha(&mut bench, PixelType::U8x4, CpuExtensions::Sse4_1);
|
||||
divides_alpha(&mut bench, PixelType::U8x2, CpuExtensions::Avx2);
|
||||
divides_alpha(&mut bench, PixelType::U8x2, CpuExtensions::Sse4_1);
|
||||
}
|
||||
for pixel_type in pixel_types {
|
||||
for &extensions in cpu_extensions.iter() {
|
||||
println!("Div {:?} {:?}", pixel_type, extensions);
|
||||
divides_alpha(bench, pixel_type, extensions);
|
||||
}
|
||||
divides_alpha(&mut bench, PixelType::U8x4, CpuExtensions::None);
|
||||
divides_alpha(&mut bench, PixelType::U8x2, CpuExtensions::None);
|
||||
if let Err(e) = after_bench(&mut bench, &cmd) {
|
||||
eprintln!("{:?}", e);
|
||||
}
|
||||
} else {
|
||||
println!("skipping bench {:?}", &name);
|
||||
}
|
||||
}
|
||||
|
||||
bench_main!("Bench Alpha", bench_alpha,);
|
||||
|
||||
@@ -0,0 +1,95 @@
|
||||
use std::num::NonZeroU32;
|
||||
|
||||
use glassbench::*;
|
||||
use image::imageops;
|
||||
|
||||
use fast_image_resize::pixels::U16x2;
|
||||
use fast_image_resize::{CpuExtensions, FilterType, Image, MulDiv, PixelType, ResizeAlg, Resizer};
|
||||
use testing::PixelExt;
|
||||
|
||||
mod utils;
|
||||
|
||||
pub fn bench_downscale_la16(bench: &mut Bench) {
|
||||
let src_image = U16x2::load_big_image().to_luma_alpha16();
|
||||
let new_width = NonZeroU32::new(852).unwrap();
|
||||
let new_height = NonZeroU32::new(567).unwrap();
|
||||
|
||||
let alg_names = ["Nearest", "Bilinear", "CatmullRom", "Lanczos3"];
|
||||
|
||||
// image crate
|
||||
// https://crates.io/crates/image
|
||||
for alg_name in alg_names {
|
||||
let filter = match alg_name {
|
||||
"Nearest" => imageops::Nearest,
|
||||
"Bilinear" => imageops::Triangle,
|
||||
"CatmullRom" => imageops::CatmullRom,
|
||||
"Lanczos3" => imageops::Lanczos3,
|
||||
_ => continue,
|
||||
};
|
||||
bench.task(format!("image - {}", alg_name), |task| {
|
||||
task.iter(|| {
|
||||
imageops::resize(&src_image, new_width.get(), new_height.get(), filter);
|
||||
})
|
||||
});
|
||||
}
|
||||
|
||||
// fast_image_resize crate;
|
||||
let mut cpu_ext_and_name = vec![(CpuExtensions::None, "rust")];
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
{
|
||||
cpu_ext_and_name.push((CpuExtensions::Sse4_1, "sse4.1"));
|
||||
cpu_ext_and_name.push((CpuExtensions::Avx2, "avx2"));
|
||||
}
|
||||
for (cpu_ext, ext_name) in cpu_ext_and_name {
|
||||
for alg_name in alg_names {
|
||||
let resize_alg = match alg_name {
|
||||
"Nearest" => ResizeAlg::Nearest,
|
||||
"Bilinear" => ResizeAlg::Convolution(FilterType::Bilinear),
|
||||
"CatmullRom" => ResizeAlg::Convolution(FilterType::CatmullRom),
|
||||
"Lanczos3" => ResizeAlg::Convolution(FilterType::Lanczos3),
|
||||
_ => return,
|
||||
};
|
||||
let src_image_data = U16x2::load_big_src_image();
|
||||
let src_view = src_image_data.view();
|
||||
let mut premultiplied_src_image = Image::new(
|
||||
NonZeroU32::new(src_image.width()).unwrap(),
|
||||
NonZeroU32::new(src_image.height()).unwrap(),
|
||||
src_view.pixel_type(),
|
||||
);
|
||||
let mut dst_image = Image::new(new_width, new_height, PixelType::U16x2);
|
||||
let mut dst_view = dst_image.view_mut();
|
||||
let mut mul_div = MulDiv::default();
|
||||
|
||||
let mut fast_resizer = Resizer::new(resize_alg);
|
||||
|
||||
unsafe {
|
||||
fast_resizer.reset_internal_buffers();
|
||||
fast_resizer.set_cpu_extensions(cpu_ext);
|
||||
mul_div.set_cpu_extensions(cpu_ext);
|
||||
}
|
||||
|
||||
bench.task(format!("fir {} - {}", ext_name, alg_name), |task| {
|
||||
task.iter(|| match resize_alg {
|
||||
ResizeAlg::Nearest => {
|
||||
fast_resizer
|
||||
.resize(&premultiplied_src_image.view(), &mut dst_view)
|
||||
.unwrap();
|
||||
}
|
||||
_ => {
|
||||
mul_div
|
||||
.multiply_alpha(&src_view, &mut premultiplied_src_image.view_mut())
|
||||
.unwrap();
|
||||
fast_resizer
|
||||
.resize(&premultiplied_src_image.view(), &mut dst_view)
|
||||
.unwrap();
|
||||
mul_div.divide_alpha_inplace(&mut dst_view).unwrap();
|
||||
}
|
||||
})
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
utils::print_md_table(bench);
|
||||
}
|
||||
|
||||
bench_main!("Compare resize of LA16 image", bench_downscale_la16,);
|
||||
@@ -100,6 +100,7 @@ pub fn main() {
|
||||
PixelType::U8x3,
|
||||
PixelType::U8x4,
|
||||
PixelType::U16,
|
||||
PixelType::U16x2,
|
||||
PixelType::U16x3,
|
||||
PixelType::I32,
|
||||
];
|
||||
@@ -117,6 +118,7 @@ pub fn main() {
|
||||
PixelType::U8x3 => U8x3::load_big_src_image(),
|
||||
PixelType::U8x4 => U8x4::load_big_src_image(),
|
||||
PixelType::U16 => U16::load_big_src_image(),
|
||||
PixelType::U16x2 => U16x2::load_big_src_image(),
|
||||
PixelType::U16x3 => U16x3::load_big_src_image(),
|
||||
PixelType::I32 => I32::load_big_src_image(),
|
||||
_ => unreachable!(),
|
||||
|
||||
+149
@@ -0,0 +1,149 @@
|
||||
# Benchmarks of fast_image_resize crate
|
||||
|
||||
Environment:
|
||||
|
||||
- CPU: AMD Ryzen 9 5950X
|
||||
- RAM: DDR4 3800 MHz
|
||||
- Ubuntu 22.04 (linux 5.15.0)
|
||||
- Rust 1.61.0
|
||||
- fast_image_resize = "0.9.3"
|
||||
- glassbench = "0.3.3"
|
||||
|
||||
Other Rust libraries used to compare of resizing speed:
|
||||
|
||||
- image = "0.24.2" (<https://crates.io/crates/image>)
|
||||
- resize = "0.7.3" (<https://crates.io/crates/resize>)
|
||||
|
||||
Resize algorithms:
|
||||
|
||||
- Nearest
|
||||
- Convolution with Bilinear filter
|
||||
- Convolution with CatmullRom filter
|
||||
- Convolution with Lanczos3 filter
|
||||
|
||||
## Resize RGB8 image (U8x3) 4928x3279 => 852x567
|
||||
|
||||
Pipeline:
|
||||
|
||||
`src_image => resize => dst_image`
|
||||
|
||||
- Source image [nasa-4928x3279.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279.png)
|
||||
- Numbers in table is mean duration of image resizing in milliseconds.
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|------------|:-------:|:--------:|:----------:|:--------:|
|
||||
| image | 19.44 | 83.01 | 153.17 | 208.82 |
|
||||
| resize | - | 52.13 | 103.37 | 154.10 |
|
||||
| fir rust | 0.28 | 43.00 | 79.52 | 117.41 |
|
||||
| fir sse4.1 | 0.28 | 27.79 | 42.97 | 58.16 |
|
||||
| fir avx2 | 0.28 | 7.30 | 9.50 | 13.59 |
|
||||
|
||||
## Resize RGBA8 image (U8x4) 4928x3279 => 852x567
|
||||
|
||||
Pipeline:
|
||||
|
||||
`src_image => multiply by alpha => resize => divide by alpha => dst_image`
|
||||
|
||||
- Source image
|
||||
[nasa-4928x3279-rgba.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279-rgba.png)
|
||||
- Numbers in table is mean duration of image resizing in milliseconds.
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|------------|:-------:|:--------:|:----------:|:--------:|
|
||||
| image | 19.73 | 82.34 | 141.74 | 198.86 |
|
||||
| resize | - | 49.91 | 100.27 | 148.99 |
|
||||
| fir rust | 0.18 | 36.84 | 52.31 | 74.99 |
|
||||
| fir sse4.1 | 0.18 | 13.21 | 17.26 | 22.42 |
|
||||
| fir avx2 | 0.18 | 9.47 | 12.03 | 16.08 |
|
||||
|
||||
## Resize L8 (luma) image (U8) 4928x3279 => 852x567
|
||||
|
||||
Pipeline:
|
||||
|
||||
`src_image => resize => dst_image`
|
||||
|
||||
- Source image [nasa-4928x3279.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279.png)
|
||||
has converted into grayscale image with one byte per pixel.
|
||||
- Numbers in table is mean duration of image resizing in milliseconds.
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|------------|:-------:|:--------:|:----------:|:--------:|
|
||||
| image | 15.21 | 44.78 | 71.92 | 100.91 |
|
||||
| resize | - | 17.53 | 36.20 | 61.10 |
|
||||
| fir rust | 0.15 | 14.08 | 16.38 | 23.73 |
|
||||
| fir sse4.1 | 0.16 | 11.92 | 12.28 | 17.79 |
|
||||
| fir avx2 | 0.16 | 6.48 | 4.77 | 7.85 |
|
||||
|
||||
## Resize LA8 (luma with alpha channel) image (U8x2) 4928x3279 => 852x567
|
||||
|
||||
Pipeline:
|
||||
|
||||
`src_image => multiply by alpha => resize => divide by alpha => dst_image`
|
||||
|
||||
- Source image
|
||||
[nasa-4928x3279-rgba.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279-rgba.png)
|
||||
has converted into grayscale image with alpha channel (two bytes per pixel).
|
||||
- Numbers in table is mean duration of image resizing in milliseconds.
|
||||
- The `resize` crate does not support this pixel format.
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|------------|:-------:|:--------:|:----------:|:--------:|
|
||||
| image | 17.03 | 63.71 | 119.64 | 160.06 |
|
||||
| fir rust | 0.17 | 25.20 | 31.18 | 42.83 |
|
||||
| fir sse4.1 | 0.17 | 12.88 | 14.73 | 18.16 |
|
||||
| fir avx2 | 0.17 | 11.26 | 12.40 | 15.40 |
|
||||
|
||||
## Resize RGB16 image (U16x3) 4928x3279 => 852x567
|
||||
|
||||
Pipeline:
|
||||
|
||||
`src_image => resize => dst_image`
|
||||
|
||||
- Source image [nasa-4928x3279.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279.png)
|
||||
has converted into RGB16 image.
|
||||
- Numbers in table is mean duration of image resizing in milliseconds.
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|------------|:-------:|:--------:|:----------:|:--------:|
|
||||
| image | 17.79 | 74.65 | 133.43 | 185.33 |
|
||||
| resize | - | 54.94 | 107.17 | 159.07 |
|
||||
| fir rust | 0.32 | 43.85 | 80.21 | 117.04 |
|
||||
| fir sse4.1 | 0.32 | 24.46 | 39.49 | 56.00 |
|
||||
| fir avx2 | 0.32 | 20.56 | 30.36 | 36.07 |
|
||||
|
||||
## Resize L16 image (U16) 4928x3279 => 852x567
|
||||
|
||||
Pipeline:
|
||||
|
||||
`src_image => resize => dst_image`
|
||||
|
||||
- Source image [nasa-4928x3279.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279.png)
|
||||
has converted into grayscale image with two bytes per pixel.
|
||||
- Numbers in table is mean duration of image resizing in milliseconds.
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|------------|:-------:|:--------:|:----------:|:--------:|
|
||||
| image | 15.38 | 46.00 | 74.00 | 102.80 |
|
||||
| resize | - | 15.48 | 32.07 | 57.20 |
|
||||
| fir rust | 0.17 | 19.20 | 28.51 | 37.54 |
|
||||
| fir sse4.1 | 0.17 | 7.74 | 13.16 | 19.11 |
|
||||
| fir avx2 | 0.17 | 7.06 | 9.55 | 14.82 |
|
||||
|
||||
## Resize LA16 (luma with alpha channel) image (U16x2) 4928x3279 => 852x567
|
||||
|
||||
Pipeline:
|
||||
|
||||
`src_image => multiply by alpha => resize => divide by alpha => dst_image`
|
||||
|
||||
- Source image
|
||||
[nasa-4928x3279-rgba.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279-rgba.png)
|
||||
has converted into grayscale image with alpha channel (four bytes per pixel).
|
||||
- Numbers in table is mean duration of image resizing in milliseconds.
|
||||
- The `resize` crate does not support this pixel format.
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|------------|:-------:|:--------:|:----------:|:--------:|
|
||||
| image | 17.05 | 64.88 | 117.68 | 159.22 |
|
||||
| fir rust | 0.19 | 33.85 | 52.56 | 72.33 |
|
||||
| fir sse4.1 | 0.19 | 21.80 | 33.98 | 46.49 |
|
||||
| fir avx2 | 0.19 | 15.11 | 21.76 | 28.95 |
|
||||
+42
-1
@@ -4,25 +4,51 @@ pub(crate) fn mul_div_255(a: u8, b: u8) -> u8 {
|
||||
(((tmp >> 8) + tmp) >> 8) as u8
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) fn mul_div_65535(a: u16, b: u16) -> u16 {
|
||||
let tmp = a as u32 * b as u32 + 0x8000;
|
||||
(((tmp >> 16) + tmp) >> 16) as u16
|
||||
}
|
||||
|
||||
const fn recip_alpha_array(precision: u32) -> [u32; 256] {
|
||||
let mut res = [0; 256];
|
||||
let scale = 1 << (precision + 1);
|
||||
let scaled_max = 255 * scale;
|
||||
let mut i: usize = 1;
|
||||
while i < 256 {
|
||||
res[i] = (((255 * scale / i as u32) + 1) >> 1) as u32;
|
||||
res[i] = (((scaled_max / i as u32) + 1) >> 1) as u32;
|
||||
i += 1;
|
||||
}
|
||||
res
|
||||
}
|
||||
|
||||
const fn recip_alpha16_array(precision: u64) -> [u64; 65536] {
|
||||
let mut res = [0; 65536];
|
||||
let scale = 1 << (precision + 1);
|
||||
let scaled_max = 0xffff * scale;
|
||||
let mut i: usize = 1;
|
||||
while i < 65536 {
|
||||
res[i] = (((scaled_max / i as u64) + 1) >> 1) as u64;
|
||||
i += 1;
|
||||
}
|
||||
res
|
||||
}
|
||||
|
||||
const PRECISION: u32 = 8;
|
||||
const PRECISION16: u64 = 33;
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) fn div_and_clip(v: u8, recip_alpha: u32) -> u8 {
|
||||
((v as u32 * recip_alpha) >> PRECISION).min(255) as u8
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) fn div_and_clip16(v: u16, recip_alpha: u64) -> u16 {
|
||||
((v as u64 * recip_alpha) >> PRECISION16).min(65535) as u16
|
||||
}
|
||||
|
||||
pub(crate) const RECIP_ALPHA: [u32; 256] = recip_alpha_array(PRECISION);
|
||||
pub(crate) static RECIP_ALPHA16: [u64; 65536] = recip_alpha16_array(PRECISION16);
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
@@ -66,4 +92,19 @@ mod tests {
|
||||
}
|
||||
assert_eq!(err_sum, 3468);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_recip_alpha16_array() {
|
||||
for alpha in 0..=0xffffu16 {
|
||||
let expected = if alpha == 0 {
|
||||
0
|
||||
} else {
|
||||
let scale = (1u64 << PRECISION16) as f64;
|
||||
(65535.0 * scale / alpha as f64).round() as u64
|
||||
};
|
||||
|
||||
let recip_alpha = RECIP_ALPHA16[alpha as usize];
|
||||
assert_eq!(expected, recip_alpha, "alpha {}", alpha);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -6,6 +6,7 @@ use crate::CpuExtensions;
|
||||
|
||||
mod common;
|
||||
pub(crate) mod errors;
|
||||
mod u16x2;
|
||||
mod u8x2;
|
||||
mod u8x4;
|
||||
|
||||
|
||||
@@ -0,0 +1,158 @@
|
||||
use std::arch::x86_64::*;
|
||||
|
||||
use crate::image_view::{TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::U16x2;
|
||||
|
||||
use super::sse4;
|
||||
|
||||
#[target_feature(enable = "avx2")]
|
||||
pub(crate) unsafe fn multiply_alpha(
|
||||
src_image: TypedImageView<U16x2>,
|
||||
mut dst_image: TypedImageViewMut<U16x2>,
|
||||
) {
|
||||
let src_rows = src_image.iter_rows(0);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
multiply_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "avx2")]
|
||||
pub(crate) unsafe fn multiply_alpha_inplace(mut image: TypedImageViewMut<U16x2>) {
|
||||
for dst_row in image.iter_rows_mut() {
|
||||
let src_row = std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len());
|
||||
multiply_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "avx2")]
|
||||
pub(crate) unsafe fn multiply_alpha_row(src_row: &[U16x2], dst_row: &mut [U16x2]) {
|
||||
let zero = _mm256_setzero_si256();
|
||||
let half = _mm256_set1_epi32(0x8000);
|
||||
|
||||
const MAX_A: i32 = 0xffff0000u32 as i32;
|
||||
let max_alpha = _mm256_set1_epi32(MAX_A);
|
||||
/*
|
||||
|L0 A0 | |L1 A1 | |L2 A2 | |L3 A3 |
|
||||
|0001 0203| |0405 0607| |0809 1011| |1213 1415|
|
||||
*/
|
||||
#[rustfmt::skip]
|
||||
let factor_mask = _mm256_set_epi8(
|
||||
15, 14, 15, 14, 11, 10, 11, 10, 7, 6, 7, 6, 3, 2, 3, 2,
|
||||
15, 14, 15, 14, 11, 10, 11, 10, 7, 6, 7, 6, 3, 2, 3, 2
|
||||
);
|
||||
|
||||
let src_chunks = src_row.chunks_exact(8);
|
||||
let src_remainder = src_chunks.remainder();
|
||||
let mut dst_chunks = dst_row.chunks_exact_mut(8);
|
||||
|
||||
for (src, dst) in src_chunks.zip(&mut dst_chunks) {
|
||||
let src_pixels = _mm256_loadu_si256(src.as_ptr() as *const __m256i);
|
||||
|
||||
let factor_pixels = _mm256_shuffle_epi8(src_pixels, factor_mask);
|
||||
let factor_pixels = _mm256_or_si256(factor_pixels, max_alpha);
|
||||
|
||||
let src_i32_lo = _mm256_unpacklo_epi16(src_pixels, zero);
|
||||
let factors = _mm256_unpacklo_epi16(factor_pixels, zero);
|
||||
let src_i32_lo = _mm256_add_epi32(_mm256_mullo_epi32(src_i32_lo, factors), half);
|
||||
let dst_i32_lo = _mm256_add_epi32(src_i32_lo, _mm256_srli_epi32::<16>(src_i32_lo));
|
||||
let dst_i32_lo = _mm256_srli_epi32::<16>(dst_i32_lo);
|
||||
|
||||
let src_i32_hi = _mm256_unpackhi_epi16(src_pixels, zero);
|
||||
let factors = _mm256_unpackhi_epi16(factor_pixels, zero);
|
||||
let src_i32_hi = _mm256_add_epi32(_mm256_mullo_epi32(src_i32_hi, factors), half);
|
||||
let dst_i32_hi = _mm256_add_epi32(src_i32_hi, _mm256_srli_epi32::<16>(src_i32_hi));
|
||||
let dst_i32_hi = _mm256_srli_epi32::<16>(dst_i32_hi);
|
||||
|
||||
let dst_pixels = _mm256_packus_epi32(dst_i32_lo, dst_i32_hi);
|
||||
|
||||
_mm256_storeu_si256(dst.as_mut_ptr() as *mut __m256i, dst_pixels);
|
||||
}
|
||||
|
||||
if !src_remainder.is_empty() {
|
||||
let dst_reminder = dst_chunks.into_remainder();
|
||||
sse4::multiply_alpha_row(src_remainder, dst_reminder);
|
||||
}
|
||||
}
|
||||
|
||||
// Divide
|
||||
|
||||
#[target_feature(enable = "avx2")]
|
||||
pub(crate) unsafe fn divide_alpha(
|
||||
src_image: TypedImageView<U16x2>,
|
||||
mut dst_image: TypedImageViewMut<U16x2>,
|
||||
) {
|
||||
let src_rows = src_image.iter_rows(0);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
divide_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "avx2")]
|
||||
pub(crate) unsafe fn divide_alpha_inplace(mut image: TypedImageViewMut<U16x2>) {
|
||||
for dst_row in image.iter_rows_mut() {
|
||||
let src_row = std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len());
|
||||
divide_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "avx2")]
|
||||
pub(crate) unsafe fn divide_alpha_row(src_row: &[U16x2], dst_row: &mut [U16x2]) {
|
||||
let src_chunks = src_row.chunks_exact(8);
|
||||
let src_remainder = src_chunks.remainder();
|
||||
let mut dst_chunks = dst_row.chunks_exact_mut(8);
|
||||
|
||||
for (src, dst) in src_chunks.zip(&mut dst_chunks) {
|
||||
divide_alpha_eight_pixels(src.as_ptr(), dst.as_mut_ptr());
|
||||
}
|
||||
|
||||
if !src_remainder.is_empty() {
|
||||
let dst_reminder = dst_chunks.into_remainder();
|
||||
let mut src_pixels = [U16x2([0, 0]); 8];
|
||||
src_pixels
|
||||
.iter_mut()
|
||||
.zip(src_remainder)
|
||||
.for_each(|(d, s)| *d = *s);
|
||||
|
||||
let mut dst_pixels = [U16x2([0, 0]); 8];
|
||||
divide_alpha_eight_pixels(src_pixels.as_ptr(), dst_pixels.as_mut_ptr());
|
||||
|
||||
dst_pixels
|
||||
.iter()
|
||||
.zip(dst_reminder)
|
||||
.for_each(|(s, d)| *d = *s);
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "avx2")]
|
||||
unsafe fn divide_alpha_eight_pixels(src: *const U16x2, dst: *mut U16x2) {
|
||||
let alpha_mask = _mm256_set1_epi32(0xffff0000u32 as i32);
|
||||
let luma_mask = _mm256_set1_epi32(0xffff);
|
||||
let alpha_max = _mm256_set1_ps(65535.0);
|
||||
/*
|
||||
|L0 A0 | |L1 A1 | |L2 A2 | |L3 A3 |
|
||||
|0001 0203| |0405 0607| |0809 1011| |1213 1415|
|
||||
*/
|
||||
#[rustfmt::skip]
|
||||
let alpha32_sh = _mm256_set_epi8(
|
||||
-1, -1, 15, 14, -1, -1, 11, 10, -1, -1, 7, 6, -1, -1, 3, 2,
|
||||
-1, -1, 15, 14, -1, -1, 11, 10, -1, -1, 7, 6, -1, -1, 3, 2,
|
||||
);
|
||||
|
||||
let src_pixels = _mm256_loadu_si256(src as *const __m256i);
|
||||
let alpha_f32x8 = _mm256_cvtepi32_ps(_mm256_shuffle_epi8(src_pixels, alpha32_sh));
|
||||
let luma_i32x8 = _mm256_and_si256(src_pixels, luma_mask);
|
||||
let luma_f32x8 = _mm256_cvtepi32_ps(luma_i32x8);
|
||||
let scaled_luma_f32x8 = _mm256_mul_ps(luma_f32x8, alpha_max);
|
||||
let divided_luma_f32x8 = _mm256_div_ps(scaled_luma_f32x8, alpha_f32x8);
|
||||
let divided_luma_i32x8 = _mm256_cvtps_epi32(divided_luma_f32x8);
|
||||
|
||||
let alpha = _mm256_and_si256(src_pixels, alpha_mask);
|
||||
let dst_pixels = _mm256_blendv_epi8(divided_luma_i32x8, alpha, alpha_mask);
|
||||
_mm256_storeu_si256(dst as *mut __m256i, dst_pixels);
|
||||
}
|
||||
@@ -0,0 +1,61 @@
|
||||
use crate::image_view::{TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::U16x2;
|
||||
use crate::CpuExtensions;
|
||||
|
||||
use super::AlphaMulDiv;
|
||||
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod avx2;
|
||||
mod native;
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod sse4;
|
||||
|
||||
impl AlphaMulDiv for U16x2 {
|
||||
fn multiply_alpha(
|
||||
src_image: TypedImageView<Self>,
|
||||
dst_image: TypedImageViewMut<Self>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha(src_image, dst_image) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha(src_image, dst_image) },
|
||||
_ => native::multiply_alpha(src_image, dst_image),
|
||||
}
|
||||
}
|
||||
|
||||
fn multiply_alpha_inplace(image: TypedImageViewMut<Self>, cpu_extensions: CpuExtensions) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha_inplace(image) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha_inplace(image) },
|
||||
_ => native::multiply_alpha_inplace(image),
|
||||
}
|
||||
}
|
||||
|
||||
fn divide_alpha(
|
||||
src_image: TypedImageView<Self>,
|
||||
dst_image: TypedImageViewMut<Self>,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::divide_alpha(src_image, dst_image) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha(src_image, dst_image) },
|
||||
_ => native::divide_alpha(src_image, dst_image),
|
||||
}
|
||||
}
|
||||
|
||||
fn divide_alpha_inplace(image: TypedImageViewMut<Self>, cpu_extensions: CpuExtensions) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::divide_alpha_inplace(image) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha_inplace(image) },
|
||||
_ => native::divide_alpha_inplace(image),
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,67 @@
|
||||
use crate::alpha::common::{div_and_clip16, mul_div_65535, RECIP_ALPHA16};
|
||||
use crate::image_view::{TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::U16x2;
|
||||
|
||||
pub(crate) fn multiply_alpha(
|
||||
src_image: TypedImageView<U16x2>,
|
||||
mut dst_image: TypedImageViewMut<U16x2>,
|
||||
) {
|
||||
let src_rows = src_image.iter_rows(0);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
multiply_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn multiply_alpha_inplace(mut image: TypedImageViewMut<U16x2>) {
|
||||
for dst_row in image.iter_rows_mut() {
|
||||
let src_row = unsafe { std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len()) };
|
||||
multiply_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) fn multiply_alpha_row(src_row: &[U16x2], dst_row: &mut [U16x2]) {
|
||||
for (src_pixel, dst_pixel) in src_row.iter().zip(dst_row) {
|
||||
let components: [u16; 2] = src_pixel.0;
|
||||
let alpha = components[1];
|
||||
dst_pixel.0 = [mul_div_65535(components[0], alpha), alpha];
|
||||
}
|
||||
}
|
||||
|
||||
// Divide
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn divide_alpha(
|
||||
src_image: TypedImageView<U16x2>,
|
||||
mut dst_image: TypedImageViewMut<U16x2>,
|
||||
) {
|
||||
let src_rows = src_image.iter_rows(0);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
divide_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn divide_alpha_inplace(mut image: TypedImageViewMut<U16x2>) {
|
||||
for dst_row in image.iter_rows_mut() {
|
||||
let src_row = unsafe { std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len()) };
|
||||
divide_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) fn divide_alpha_row(src_row: &[U16x2], dst_row: &mut [U16x2]) {
|
||||
src_row
|
||||
.iter()
|
||||
.zip(dst_row)
|
||||
.for_each(|(src_pixel, dst_pixel)| {
|
||||
let components: [u16; 2] = src_pixel.0;
|
||||
let alpha = components[1];
|
||||
let recip_alpha = RECIP_ALPHA16[alpha as usize];
|
||||
dst_pixel.0 = [div_and_clip16(components[0], recip_alpha), alpha];
|
||||
});
|
||||
}
|
||||
@@ -0,0 +1,150 @@
|
||||
use std::arch::x86_64::*;
|
||||
|
||||
use crate::image_view::{TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::U16x2;
|
||||
|
||||
use super::native;
|
||||
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
pub(crate) unsafe fn multiply_alpha(
|
||||
src_image: TypedImageView<U16x2>,
|
||||
mut dst_image: TypedImageViewMut<U16x2>,
|
||||
) {
|
||||
let src_rows = src_image.iter_rows(0);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
multiply_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
pub(crate) unsafe fn multiply_alpha_inplace(mut image: TypedImageViewMut<U16x2>) {
|
||||
for dst_row in image.iter_rows_mut() {
|
||||
let src_row = std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len());
|
||||
multiply_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
pub(crate) unsafe fn multiply_alpha_row(src_row: &[U16x2], dst_row: &mut [U16x2]) {
|
||||
let zero = _mm_setzero_si128();
|
||||
let half = _mm_set1_epi32(0x8000);
|
||||
|
||||
const MAX_A: i32 = 0xffff0000u32 as i32;
|
||||
let max_alpha = _mm_set1_epi32(MAX_A);
|
||||
/*
|
||||
|L0 A0 | |L1 A1 | |L2 A2 | |L3 A3 |
|
||||
|0001 0203| |0405 0607| |0809 1011| |1213 1415|
|
||||
*/
|
||||
let factor_mask = _mm_set_epi8(15, 14, 15, 14, 11, 10, 11, 10, 7, 6, 7, 6, 3, 2, 3, 2);
|
||||
|
||||
let src_chunks = src_row.chunks_exact(4);
|
||||
let src_remainder = src_chunks.remainder();
|
||||
let mut dst_chunks = dst_row.chunks_exact_mut(4);
|
||||
|
||||
for (src, dst) in src_chunks.zip(&mut dst_chunks) {
|
||||
let src_pixels = _mm_loadu_si128(src.as_ptr() as *const __m128i);
|
||||
|
||||
let factor_pixels = _mm_shuffle_epi8(src_pixels, factor_mask);
|
||||
let factor_pixels = _mm_or_si128(factor_pixels, max_alpha);
|
||||
|
||||
let src_i32_lo = _mm_unpacklo_epi16(src_pixels, zero);
|
||||
let factors = _mm_unpacklo_epi16(factor_pixels, zero);
|
||||
let src_i32_lo = _mm_add_epi32(_mm_mullo_epi32(src_i32_lo, factors), half);
|
||||
let dst_i32_lo = _mm_add_epi32(src_i32_lo, _mm_srli_epi32::<16>(src_i32_lo));
|
||||
let dst_i32_lo = _mm_srli_epi32::<16>(dst_i32_lo);
|
||||
|
||||
let src_i32_hi = _mm_unpackhi_epi16(src_pixels, zero);
|
||||
let factors = _mm_unpackhi_epi16(factor_pixels, zero);
|
||||
let src_i32_hi = _mm_add_epi32(_mm_mullo_epi32(src_i32_hi, factors), half);
|
||||
let dst_i32_hi = _mm_add_epi32(src_i32_hi, _mm_srli_epi32::<16>(src_i32_hi));
|
||||
let dst_i32_hi = _mm_srli_epi32::<16>(dst_i32_hi);
|
||||
|
||||
let dst_pixels = _mm_packus_epi32(dst_i32_lo, dst_i32_hi);
|
||||
|
||||
_mm_storeu_si128(dst.as_mut_ptr() as *mut __m128i, dst_pixels);
|
||||
}
|
||||
|
||||
if !src_remainder.is_empty() {
|
||||
let dst_reminder = dst_chunks.into_remainder();
|
||||
native::multiply_alpha_row(src_remainder, dst_reminder);
|
||||
}
|
||||
}
|
||||
|
||||
// Divide
|
||||
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
pub(crate) unsafe fn divide_alpha(
|
||||
src_image: TypedImageView<U16x2>,
|
||||
mut dst_image: TypedImageViewMut<U16x2>,
|
||||
) {
|
||||
let src_rows = src_image.iter_rows(0);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
divide_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
pub(crate) unsafe fn divide_alpha_inplace(mut image: TypedImageViewMut<U16x2>) {
|
||||
for dst_row in image.iter_rows_mut() {
|
||||
let src_row = std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len());
|
||||
divide_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
pub(crate) unsafe fn divide_alpha_row(src_row: &[U16x2], dst_row: &mut [U16x2]) {
|
||||
let src_chunks = src_row.chunks_exact(4);
|
||||
let src_remainder = src_chunks.remainder();
|
||||
let mut dst_chunks = dst_row.chunks_exact_mut(4);
|
||||
|
||||
for (src, dst) in src_chunks.zip(&mut dst_chunks) {
|
||||
divide_alpha_four_pixels(src.as_ptr(), dst.as_mut_ptr());
|
||||
}
|
||||
|
||||
if !src_remainder.is_empty() {
|
||||
let dst_reminder = dst_chunks.into_remainder();
|
||||
let mut src_pixels = [U16x2([0, 0]); 4];
|
||||
src_pixels
|
||||
.iter_mut()
|
||||
.zip(src_remainder)
|
||||
.for_each(|(d, s)| *d = *s);
|
||||
|
||||
let mut dst_pixels = [U16x2([0, 0]); 4];
|
||||
divide_alpha_four_pixels(src_pixels.as_ptr(), dst_pixels.as_mut_ptr());
|
||||
|
||||
dst_pixels
|
||||
.iter()
|
||||
.zip(dst_reminder)
|
||||
.for_each(|(s, d)| *d = *s);
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
unsafe fn divide_alpha_four_pixels(src: *const U16x2, dst: *mut U16x2) {
|
||||
let alpha_mask = _mm_set1_epi32(0xffff0000u32 as i32);
|
||||
let luma_mask = _mm_set1_epi32(0xffff);
|
||||
let alpha_max = _mm_set1_ps(65535.0);
|
||||
/*
|
||||
|L0 A0 | |L1 A1 | |L2 A2 | |L3 A3 |
|
||||
|0001 0203| |0405 0607| |0809 1011| |1213 1415|
|
||||
*/
|
||||
let alpha32_sh = _mm_set_epi8(-1, -1, 15, 14, -1, -1, 11, 10, -1, -1, 7, 6, -1, -1, 3, 2);
|
||||
|
||||
let src_pixels = _mm_loadu_si128(src as *const __m128i);
|
||||
let alpha_f32x4 = _mm_cvtepi32_ps(_mm_shuffle_epi8(src_pixels, alpha32_sh));
|
||||
let luma_i32x4 = _mm_and_si128(src_pixels, luma_mask);
|
||||
let luma_f32x4 = _mm_cvtepi32_ps(luma_i32x4);
|
||||
let scaled_luma_f32x4 = _mm_mul_ps(luma_f32x4, alpha_max);
|
||||
let divided_luma_f32x4 = _mm_div_ps(scaled_luma_f32x4, alpha_f32x4);
|
||||
let divided_luma_i32x4 = _mm_cvtps_epi32(divided_luma_f32x4);
|
||||
|
||||
let alpha = _mm_and_si128(src_pixels, alpha_mask);
|
||||
let dst_pixels = _mm_blendv_epi8(divided_luma_i32x4, alpha, alpha_mask);
|
||||
_mm_storeu_si128(dst as *mut __m128i, dst_pixels);
|
||||
}
|
||||
@@ -13,6 +13,7 @@ mod filters;
|
||||
mod i32x1;
|
||||
mod optimisations;
|
||||
mod u16x1;
|
||||
mod u16x2;
|
||||
mod u16x3;
|
||||
mod u8x1;
|
||||
mod u8x2;
|
||||
|
||||
@@ -109,17 +109,17 @@ unsafe fn horiz_convolution_four_rows(
|
||||
let mut sum = ll_sum[i];
|
||||
let source = simd_utils::loadu_si128(s_rows[i], x);
|
||||
|
||||
let l0l1_i64x4 = _mm_shuffle_epi8(source, l0l1_shuffle);
|
||||
sum = _mm_add_epi64(sum, _mm_mul_epi32(l0l1_i64x4, coeff01_i64x2));
|
||||
let l0l1_i64x2 = _mm_shuffle_epi8(source, l0l1_shuffle);
|
||||
sum = _mm_add_epi64(sum, _mm_mul_epi32(l0l1_i64x2, coeff01_i64x2));
|
||||
|
||||
let l2l3_i64x4 = _mm_shuffle_epi8(source, l2l3_shuffle);
|
||||
sum = _mm_add_epi64(sum, _mm_mul_epi32(l2l3_i64x4, coeff23_i64x2));
|
||||
let l2l3_i64x2 = _mm_shuffle_epi8(source, l2l3_shuffle);
|
||||
sum = _mm_add_epi64(sum, _mm_mul_epi32(l2l3_i64x2, coeff23_i64x2));
|
||||
|
||||
let l4l5_i64x4 = _mm_shuffle_epi8(source, l4l5_shuffle);
|
||||
sum = _mm_add_epi64(sum, _mm_mul_epi32(l4l5_i64x4, coeff45_i64x2));
|
||||
let l4l5_i64x2 = _mm_shuffle_epi8(source, l4l5_shuffle);
|
||||
sum = _mm_add_epi64(sum, _mm_mul_epi32(l4l5_i64x2, coeff45_i64x2));
|
||||
|
||||
let l6l7_i64x4 = _mm_shuffle_epi8(source, l6l7_shuffle);
|
||||
sum = _mm_add_epi64(sum, _mm_mul_epi32(l6l7_i64x4, coeff67_i64x2));
|
||||
let l6l7_i64x2 = _mm_shuffle_epi8(source, l6l7_shuffle);
|
||||
sum = _mm_add_epi64(sum, _mm_mul_epi32(l6l7_i64x2, coeff67_i64x2));
|
||||
|
||||
ll_sum[i] = sum;
|
||||
}
|
||||
@@ -137,11 +137,11 @@ unsafe fn horiz_convolution_four_rows(
|
||||
let mut sum = ll_sum[i];
|
||||
let source = simd_utils::loadl_epi64(s_rows[i], x);
|
||||
|
||||
let l0l1_i64x4 = _mm_shuffle_epi8(source, l0l1_shuffle);
|
||||
sum = _mm_add_epi64(sum, _mm_mul_epi32(l0l1_i64x4, coeff01_i64x2));
|
||||
let l0l1_i64x2 = _mm_shuffle_epi8(source, l0l1_shuffle);
|
||||
sum = _mm_add_epi64(sum, _mm_mul_epi32(l0l1_i64x2, coeff01_i64x2));
|
||||
|
||||
let l2l3_i64x4 = _mm_shuffle_epi8(source, l2l3_shuffle);
|
||||
sum = _mm_add_epi64(sum, _mm_mul_epi32(l2l3_i64x4, coeff23_i64x2));
|
||||
let l2l3_i64x2 = _mm_shuffle_epi8(source, l2l3_shuffle);
|
||||
sum = _mm_add_epi64(sum, _mm_mul_epi32(l2l3_i64x2, coeff23_i64x2));
|
||||
|
||||
ll_sum[i] = sum;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,336 @@
|
||||
use std::arch::x86_64::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::image_view::{FourRows, FourRowsMut, TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::U16x2;
|
||||
use crate::simd_utils;
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn horiz_convolution(
|
||||
src_image: TypedImageView<U16x2>,
|
||||
mut dst_image: TypedImageViewMut<U16x2>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard32::new(values);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(
|
||||
src_rows,
|
||||
dst_rows,
|
||||
&coefficients_chunks,
|
||||
&normalizer_guard,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
let mut yy = dst_height - dst_height % 4;
|
||||
while yy < dst_height {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
&coefficients_chunks,
|
||||
&normalizer_guard,
|
||||
);
|
||||
}
|
||||
yy += 1;
|
||||
}
|
||||
}
|
||||
|
||||
/// For safety, it is necessary to ensure the following conditions:
|
||||
/// - length of all rows in src_rows must be equal
|
||||
/// - length of all rows in dst_rows must be equal
|
||||
/// - coefficients_chunks.len() == dst_rows.0.len()
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
|
||||
#[target_feature(enable = "avx2")]
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: FourRows<U16x2>,
|
||||
dst_rows: FourRowsMut<U16x2>,
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer_guard: &optimisations::NormalizerGuard32,
|
||||
) {
|
||||
let (s_row0, s_row1, s_row2, s_row3) = src_rows;
|
||||
let s_rows = [s_row0, s_row1, s_row2, s_row3];
|
||||
let (d_row0, d_row1, d_row2, d_row3) = dst_rows;
|
||||
let d_rows = [d_row0, d_row1, d_row2, d_row3];
|
||||
let precision = normalizer_guard.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 4];
|
||||
|
||||
/*
|
||||
|L0 A0 | |L1 A1 | |L2 A2 | |L3 A3 |
|
||||
|0001 0203| |0405 0607| |0809 1011| |1213 1415|
|
||||
|
||||
Shuffle to extract L0 and A0 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 3, 2, -1, -1, -1, -1, -1, -1, 1, 0
|
||||
|
||||
Shuffle to extract L1 and A1 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 7, 6, -1, -1, -1, -1, -1, -1, 5, 4
|
||||
|
||||
Shuffle to extract L2 and A2 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 11, 10, -1, -1, -1, -1, -1, -1, 9, 8
|
||||
|
||||
Shuffle to extract L3 and A3 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12
|
||||
*/
|
||||
|
||||
#[rustfmt::skip]
|
||||
let p0_shuffle = _mm256_set_epi8(
|
||||
-1, -1, -1, -1, -1, -1, 3, 2, -1, -1, -1, -1, -1, -1, 1, 0,
|
||||
-1, -1, -1, -1, -1, -1, 3, 2, -1, -1, -1, -1, -1, -1, 1, 0,
|
||||
);
|
||||
#[rustfmt::skip]
|
||||
let p1_shuffle = _mm256_set_epi8(
|
||||
-1, -1, -1, -1, -1, -1, 7, 6, -1, -1, -1, -1, -1, -1, 5, 4,
|
||||
-1, -1, -1, -1, -1, -1, 7, 6, -1, -1, -1, -1, -1, -1, 5, 4,
|
||||
);
|
||||
#[rustfmt::skip]
|
||||
let p2_shuffle = _mm256_set_epi8(
|
||||
-1, -1, -1, -1, -1, -1, 11, 10, -1, -1, -1, -1, -1, -1, 9, 8,
|
||||
-1, -1, -1, -1, -1, -1, 11, 10, -1, -1, -1, -1, -1, -1, 9, 8,
|
||||
);
|
||||
#[rustfmt::skip]
|
||||
let p3_shuffle = _mm256_set_epi8(
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12,
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut ll_sum = [_mm256_set1_epi64x(half_error); 2];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
coeffs = coeffs_by_4.remainder();
|
||||
for k in coeffs_by_4 {
|
||||
let coeff0_i64x4 = _mm256_set1_epi64x(k[0] as i64);
|
||||
let coeff1_i64x4 = _mm256_set1_epi64x(k[1] as i64);
|
||||
let coeff2_i64x4 = _mm256_set1_epi64x(k[2] as i64);
|
||||
let coeff3_i64x4 = _mm256_set1_epi64x(k[3] as i64);
|
||||
|
||||
for (i, sum) in ll_sum.iter_mut().enumerate() {
|
||||
let source = _mm256_set_m128i(
|
||||
simd_utils::loadu_si128(s_rows[i * 2 + 1], x),
|
||||
simd_utils::loadu_si128(s_rows[i * 2], x),
|
||||
);
|
||||
|
||||
let pp_i64x4 = _mm256_shuffle_epi8(source, p0_shuffle);
|
||||
*sum = _mm256_add_epi64(*sum, _mm256_mul_epi32(pp_i64x4, coeff0_i64x4));
|
||||
|
||||
let pp_i64x4 = _mm256_shuffle_epi8(source, p1_shuffle);
|
||||
*sum = _mm256_add_epi64(*sum, _mm256_mul_epi32(pp_i64x4, coeff1_i64x4));
|
||||
|
||||
let pp_i64x4 = _mm256_shuffle_epi8(source, p2_shuffle);
|
||||
*sum = _mm256_add_epi64(*sum, _mm256_mul_epi32(pp_i64x4, coeff2_i64x4));
|
||||
|
||||
let pp_i64x4 = _mm256_shuffle_epi8(source, p3_shuffle);
|
||||
*sum = _mm256_add_epi64(*sum, _mm256_mul_epi32(pp_i64x4, coeff3_i64x4));
|
||||
}
|
||||
x += 4;
|
||||
}
|
||||
|
||||
let coeffs_by_2 = coeffs.chunks_exact(2);
|
||||
coeffs = coeffs_by_2.remainder();
|
||||
for k in coeffs_by_2 {
|
||||
let coeff0_i64x4 = _mm256_set1_epi64x(k[0] as i64);
|
||||
let coeff1_i64x4 = _mm256_set1_epi64x(k[1] as i64);
|
||||
|
||||
for (i, sum) in ll_sum.iter_mut().enumerate() {
|
||||
let source = _mm256_set_m128i(
|
||||
simd_utils::loadl_epi64(s_rows[i * 2 + 1], x),
|
||||
simd_utils::loadl_epi64(s_rows[i * 2], x),
|
||||
);
|
||||
|
||||
let pp_i64x4 = _mm256_shuffle_epi8(source, p0_shuffle);
|
||||
*sum = _mm256_add_epi64(*sum, _mm256_mul_epi32(pp_i64x4, coeff0_i64x4));
|
||||
|
||||
let pp_i64x4 = _mm256_shuffle_epi8(source, p1_shuffle);
|
||||
*sum = _mm256_add_epi64(*sum, _mm256_mul_epi32(pp_i64x4, coeff1_i64x4));
|
||||
}
|
||||
x += 2;
|
||||
}
|
||||
|
||||
if let Some(&k) = coeffs.get(0) {
|
||||
let coeff0_i64x4 = _mm256_set1_epi64x(k as i64);
|
||||
|
||||
for (i, sum) in ll_sum.iter_mut().enumerate() {
|
||||
let source = _mm256_set_m128i(
|
||||
simd_utils::loadl_epi32(s_rows[i * 2 + 1], x),
|
||||
simd_utils::loadl_epi32(s_rows[i * 2], x),
|
||||
);
|
||||
|
||||
let pp_i64x4 = _mm256_shuffle_epi8(source, p0_shuffle);
|
||||
*sum = _mm256_add_epi64(*sum, _mm256_mul_epi32(pp_i64x4, coeff0_i64x4));
|
||||
}
|
||||
}
|
||||
|
||||
// ll_sum.into_iter().enumerate() executes slowly than ll_sum.iter().enumerate()
|
||||
for (i, &ll) in ll_sum.iter().enumerate() {
|
||||
_mm256_storeu_si256((&mut ll_buf).as_mut_ptr() as *mut __m256i, ll);
|
||||
let dst_pixel = d_rows[i * 2].get_unchecked_mut(dst_x);
|
||||
|
||||
dst_pixel.0 = [
|
||||
normalizer_guard.clip(ll_buf[0]),
|
||||
normalizer_guard.clip(ll_buf[1]),
|
||||
];
|
||||
|
||||
let dst_pixel = d_rows[i * 2 + 1].get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = [
|
||||
normalizer_guard.clip(ll_buf[2]),
|
||||
normalizer_guard.clip(ll_buf[3]),
|
||||
];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// For safety, it is necessary to ensure the following conditions:
|
||||
/// - bounds.len() == dst_row.len()
|
||||
/// - coefficients_chunks.len() == dst_row.len()
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
|
||||
#[target_feature(enable = "avx2")]
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16x2],
|
||||
dst_row: &mut [U16x2],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer_guard: &optimisations::NormalizerGuard32,
|
||||
) {
|
||||
let precision = normalizer_guard.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 4];
|
||||
|
||||
/*
|
||||
|L0 A0 | |L1 A1 | |L2 A2 | |L3 A3 |
|
||||
|0001 0203| |0405 0607| |0809 1011| |1213 1415|
|
||||
|
||||
Shuffle to extract L0 and A0 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 3, 2, -1, -1, -1, -1, -1, -1, 1, 0
|
||||
|
||||
Shuffle to extract L1 and A1 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 7, 6, -1, -1, -1, -1, -1, -1, 5, 4
|
||||
|
||||
Shuffle to extract L2 and A2 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 11, 10, -1, -1, -1, -1, -1, -1, 9, 8
|
||||
|
||||
Shuffle to extract L3 and A3 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12
|
||||
*/
|
||||
|
||||
#[rustfmt::skip]
|
||||
let p0_shuffle = _mm256_set_epi8(
|
||||
-1, -1, -1, -1, -1, -1, 3, 2, -1, -1, -1, -1, -1, -1, 1, 0,
|
||||
-1, -1, -1, -1, -1, -1, 3, 2, -1, -1, -1, -1, -1, -1, 1, 0,
|
||||
);
|
||||
#[rustfmt::skip]
|
||||
let p1_shuffle = _mm256_set_epi8(
|
||||
-1, -1, -1, -1, -1, -1, 7, 6, -1, -1, -1, -1, -1, -1, 5, 4,
|
||||
-1, -1, -1, -1, -1, -1, 7, 6, -1, -1, -1, -1, -1, -1, 5, 4,
|
||||
);
|
||||
#[rustfmt::skip]
|
||||
let p2_shuffle = _mm256_set_epi8(
|
||||
-1, -1, -1, -1, -1, -1, 11, 10, -1, -1, -1, -1, -1, -1, 9, 8,
|
||||
-1, -1, -1, -1, -1, -1, 11, 10, -1, -1, -1, -1, -1, -1, 9, 8,
|
||||
);
|
||||
#[rustfmt::skip]
|
||||
let p3_shuffle = _mm256_set_epi8(
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12,
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut ll_sum = _mm256_set1_epi64x(0);
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
coeffs = coeffs_by_8.remainder();
|
||||
for k in coeffs_by_8 {
|
||||
let coeff04_i64x4 =
|
||||
_mm256_set_epi64x(k[4] as i64, k[4] as i64, k[0] as i64, k[0] as i64);
|
||||
let coeff15_i64x4 =
|
||||
_mm256_set_epi64x(k[5] as i64, k[5] as i64, k[1] as i64, k[1] as i64);
|
||||
let coeff26_i64x4 =
|
||||
_mm256_set_epi64x(k[6] as i64, k[6] as i64, k[2] as i64, k[2] as i64);
|
||||
let coeff37_i64x4 =
|
||||
_mm256_set_epi64x(k[7] as i64, k[7] as i64, k[3] as i64, k[3] as i64);
|
||||
|
||||
let source = simd_utils::loadu_si256(src_row, x);
|
||||
|
||||
let pp_i64x4 = _mm256_shuffle_epi8(source, p0_shuffle);
|
||||
ll_sum = _mm256_add_epi64(ll_sum, _mm256_mul_epi32(pp_i64x4, coeff04_i64x4));
|
||||
|
||||
let pp_i64x4 = _mm256_shuffle_epi8(source, p1_shuffle);
|
||||
ll_sum = _mm256_add_epi64(ll_sum, _mm256_mul_epi32(pp_i64x4, coeff15_i64x4));
|
||||
|
||||
let pp_i64x4 = _mm256_shuffle_epi8(source, p2_shuffle);
|
||||
ll_sum = _mm256_add_epi64(ll_sum, _mm256_mul_epi32(pp_i64x4, coeff26_i64x4));
|
||||
|
||||
let pp_i64x4 = _mm256_shuffle_epi8(source, p3_shuffle);
|
||||
ll_sum = _mm256_add_epi64(ll_sum, _mm256_mul_epi32(pp_i64x4, coeff37_i64x4));
|
||||
|
||||
x += 8;
|
||||
}
|
||||
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
coeffs = coeffs_by_4.remainder();
|
||||
for k in coeffs_by_4 {
|
||||
let coeff02_i64x4 =
|
||||
_mm256_set_epi64x(k[2] as i64, k[2] as i64, k[0] as i64, k[0] as i64);
|
||||
let coeff13_i64x4 =
|
||||
_mm256_set_epi64x(k[3] as i64, k[3] as i64, k[1] as i64, k[1] as i64);
|
||||
|
||||
let source = _mm256_set_m128i(
|
||||
simd_utils::loadl_epi64(src_row, x + 2),
|
||||
simd_utils::loadl_epi64(src_row, x),
|
||||
);
|
||||
|
||||
let pp_i64x4 = _mm256_shuffle_epi8(source, p0_shuffle);
|
||||
ll_sum = _mm256_add_epi64(ll_sum, _mm256_mul_epi32(pp_i64x4, coeff02_i64x4));
|
||||
|
||||
let pp_i64x4 = _mm256_shuffle_epi8(source, p1_shuffle);
|
||||
ll_sum = _mm256_add_epi64(ll_sum, _mm256_mul_epi32(pp_i64x4, coeff13_i64x4));
|
||||
|
||||
x += 4;
|
||||
}
|
||||
|
||||
let coeffs_by_2 = coeffs.chunks_exact(2);
|
||||
coeffs = coeffs_by_2.remainder();
|
||||
for k in coeffs_by_2 {
|
||||
let coeff01_i64x4 =
|
||||
_mm256_set_epi64x(k[1] as i64, k[1] as i64, k[0] as i64, k[0] as i64);
|
||||
|
||||
let source = _mm256_set_m128i(
|
||||
simd_utils::loadl_epi32(src_row, x + 1),
|
||||
simd_utils::loadl_epi32(src_row, x),
|
||||
);
|
||||
|
||||
let pp_i64x4 = _mm256_shuffle_epi8(source, p0_shuffle);
|
||||
ll_sum = _mm256_add_epi64(ll_sum, _mm256_mul_epi32(pp_i64x4, coeff01_i64x4));
|
||||
|
||||
x += 2;
|
||||
}
|
||||
|
||||
if let Some(&k) = coeffs.get(0) {
|
||||
let coeff0_i64x4 = _mm256_set_epi64x(0, 0, k as i64, k as i64);
|
||||
let source = _mm256_set_m128i(_mm_setzero_si128(), simd_utils::loadl_epi32(src_row, x));
|
||||
let p_i64x4 = _mm256_shuffle_epi8(source, p0_shuffle);
|
||||
ll_sum = _mm256_add_epi64(ll_sum, _mm256_mul_epi32(p_i64x4, coeff0_i64x4));
|
||||
}
|
||||
|
||||
_mm256_storeu_si256((&mut ll_buf).as_mut_ptr() as *mut __m256i, ll_sum);
|
||||
let dst_pixel = dst_row.get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = [
|
||||
normalizer_guard.clip(ll_buf[0] + ll_buf[2] + half_error),
|
||||
normalizer_guard.clip(ll_buf[1] + ll_buf[3] + half_error),
|
||||
]
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,38 @@
|
||||
use super::{Coefficients, Convolution};
|
||||
use crate::convolution::vertical_u16::vert_convolution_u16;
|
||||
use crate::image_view::{TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::U16x2;
|
||||
use crate::CpuExtensions;
|
||||
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod avx2;
|
||||
mod native;
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod sse4;
|
||||
|
||||
impl Convolution for U16x2 {
|
||||
fn horiz_convolution(
|
||||
src_image: TypedImageView<Self>,
|
||||
dst_image: TypedImageViewMut<Self>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::horiz_convolution(src_image, dst_image, offset, coeffs),
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_image, dst_image, offset, coeffs),
|
||||
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
|
||||
}
|
||||
}
|
||||
|
||||
fn vert_convolution(
|
||||
src_image: TypedImageView<Self>,
|
||||
dst_image: TypedImageViewMut<Self>,
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
vert_convolution_u16(src_image, dst_image, coeffs, cpu_extensions);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,36 @@
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::image_view::{TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::U16x2;
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) fn horiz_convolution(
|
||||
src_image: TypedImageView<U16x2>,
|
||||
mut dst_image: TypedImageViewMut<U16x2>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard32::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
|
||||
let initial: i64 = 1 << (precision - 1);
|
||||
|
||||
let src_rows = src_image.iter_rows(offset);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for (dst_row, src_row) in dst_rows.zip(src_rows) {
|
||||
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
|
||||
let first_x_src = coeffs_chunk.start as usize;
|
||||
let mut ss = [initial; 2];
|
||||
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
|
||||
for (&k, src_pixel) in coeffs_chunk.values.iter().zip(src_pixels) {
|
||||
for (i, s) in ss.iter_mut().enumerate() {
|
||||
*s += src_pixel.0[i] as i64 * (k as i64);
|
||||
}
|
||||
}
|
||||
for (i, s) in ss.iter().copied().enumerate() {
|
||||
dst_pixel.0[i] = normalizer_guard.clip(s);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,280 @@
|
||||
use std::arch::x86_64::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::image_view::{FourRows, FourRowsMut, TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::U16x2;
|
||||
use crate::simd_utils;
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn horiz_convolution(
|
||||
src_image: TypedImageView<U16x2>,
|
||||
mut dst_image: TypedImageViewMut<U16x2>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard32::new(values);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(
|
||||
src_rows,
|
||||
dst_rows,
|
||||
&coefficients_chunks,
|
||||
&normalizer_guard,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
let mut yy = dst_height - dst_height % 4;
|
||||
while yy < dst_height {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
&coefficients_chunks,
|
||||
&normalizer_guard,
|
||||
);
|
||||
}
|
||||
yy += 1;
|
||||
}
|
||||
}
|
||||
|
||||
/// For safety, it is necessary to ensure the following conditions:
|
||||
/// - length of all rows in src_rows must be equal
|
||||
/// - length of all rows in dst_rows must be equal
|
||||
/// - coefficients_chunks.len() == dst_rows.0.len()
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: FourRows<U16x2>,
|
||||
dst_rows: FourRowsMut<U16x2>,
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer_guard: &optimisations::NormalizerGuard32,
|
||||
) {
|
||||
let (s_row0, s_row1, s_row2, s_row3) = src_rows;
|
||||
let s_rows = [s_row0, s_row1, s_row2, s_row3];
|
||||
let (d_row0, d_row1, d_row2, d_row3) = dst_rows;
|
||||
let d_rows = [d_row0, d_row1, d_row2, d_row3];
|
||||
let precision = normalizer_guard.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 2];
|
||||
|
||||
/*
|
||||
|L0 A0 | |L1 A1 | |L2 A2 | |L3 A3 |
|
||||
|0001 0203| |0405 0607| |0809 1011| |1213 1415|
|
||||
|
||||
Shuffle to extract L0 and A0 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 3, 2, -1, -1, -1, -1, -1, -1, 1, 0
|
||||
|
||||
Shuffle to extract L1 and A1 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 7, 6, -1, -1, -1, -1, -1, -1, 5, 4
|
||||
|
||||
Shuffle to extract L2 and A2 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 11, 10, -1, -1, -1, -1, -1, -1, 9, 8
|
||||
|
||||
Shuffle to extract L3 and A3 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12
|
||||
*/
|
||||
|
||||
let p0_shuffle = _mm_set_epi8(-1, -1, -1, -1, -1, -1, 3, 2, -1, -1, -1, -1, -1, -1, 1, 0);
|
||||
let p1_shuffle = _mm_set_epi8(-1, -1, -1, -1, -1, -1, 7, 6, -1, -1, -1, -1, -1, -1, 5, 4);
|
||||
let p2_shuffle = _mm_set_epi8(-1, -1, -1, -1, -1, -1, 11, 10, -1, -1, -1, -1, -1, -1, 9, 8);
|
||||
let p3_shuffle = _mm_set_epi8(
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut ll_sum = [_mm_set1_epi64x(half_error); 4];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
coeffs = coeffs_by_4.remainder();
|
||||
|
||||
for k in coeffs_by_4 {
|
||||
let coeff0_i64x2 = _mm_set1_epi64x(k[0] as i64);
|
||||
let coeff1_i64x2 = _mm_set1_epi64x(k[1] as i64);
|
||||
let coeff2_i64x2 = _mm_set1_epi64x(k[2] as i64);
|
||||
let coeff3_i64x2 = _mm_set1_epi64x(k[3] as i64);
|
||||
|
||||
for i in 0..4 {
|
||||
let mut sum = ll_sum[i];
|
||||
let source = simd_utils::loadu_si128(s_rows[i], x);
|
||||
|
||||
let p_i64x2 = _mm_shuffle_epi8(source, p0_shuffle);
|
||||
sum = _mm_add_epi64(sum, _mm_mul_epi32(p_i64x2, coeff0_i64x2));
|
||||
|
||||
let p_i64x2 = _mm_shuffle_epi8(source, p1_shuffle);
|
||||
sum = _mm_add_epi64(sum, _mm_mul_epi32(p_i64x2, coeff1_i64x2));
|
||||
|
||||
let p_i64x2 = _mm_shuffle_epi8(source, p2_shuffle);
|
||||
sum = _mm_add_epi64(sum, _mm_mul_epi32(p_i64x2, coeff2_i64x2));
|
||||
|
||||
let p_i64x2 = _mm_shuffle_epi8(source, p3_shuffle);
|
||||
sum = _mm_add_epi64(sum, _mm_mul_epi32(p_i64x2, coeff3_i64x2));
|
||||
|
||||
ll_sum[i] = sum;
|
||||
}
|
||||
x += 4;
|
||||
}
|
||||
|
||||
let coeffs_by_2 = coeffs.chunks_exact(2);
|
||||
coeffs = coeffs_by_2.remainder();
|
||||
|
||||
for k in coeffs_by_2 {
|
||||
let coeff0_i64x2 = _mm_set1_epi64x(k[0] as i64);
|
||||
let coeff1_i64x2 = _mm_set1_epi64x(k[1] as i64);
|
||||
|
||||
for i in 0..4 {
|
||||
let mut sum = ll_sum[i];
|
||||
let source = simd_utils::loadl_epi64(s_rows[i], x);
|
||||
|
||||
let p_i64x2 = _mm_shuffle_epi8(source, p0_shuffle);
|
||||
sum = _mm_add_epi64(sum, _mm_mul_epi32(p_i64x2, coeff0_i64x2));
|
||||
|
||||
let p_i64x2 = _mm_shuffle_epi8(source, p1_shuffle);
|
||||
sum = _mm_add_epi64(sum, _mm_mul_epi32(p_i64x2, coeff1_i64x2));
|
||||
|
||||
ll_sum[i] = sum;
|
||||
}
|
||||
x += 2;
|
||||
}
|
||||
|
||||
let coeffs_by_2 = coeffs.chunks_exact(2);
|
||||
coeffs = coeffs_by_2.remainder();
|
||||
|
||||
if let Some(&k) = coeffs.get(0) {
|
||||
let coeff0_i64x2 = _mm_set1_epi64x(k as i64);
|
||||
for i in 0..4 {
|
||||
let source = simd_utils::loadl_epi32(s_rows[i], x);
|
||||
let p_i64x2 = _mm_shuffle_epi8(source, p0_shuffle);
|
||||
ll_sum[i] = _mm_add_epi64(ll_sum[i], _mm_mul_epi32(p_i64x2, coeff0_i64x2));
|
||||
}
|
||||
}
|
||||
|
||||
for i in 0..4 {
|
||||
_mm_storeu_si128((&mut ll_buf).as_mut_ptr() as *mut __m128i, ll_sum[i]);
|
||||
let dst_pixel = d_rows[i].get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = [
|
||||
normalizer_guard.clip(ll_buf[0]),
|
||||
normalizer_guard.clip(ll_buf[1]),
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// For safety, it is necessary to ensure the following conditions:
|
||||
/// - bounds.len() == dst_row.len()
|
||||
/// - coeffs.len() == dst_rows.0.len() * window_size
|
||||
/// - max(bound.start + bound.size for bound in bounds) <= src_row.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[inline]
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16x2],
|
||||
dst_row: &mut [U16x2],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer_guard: &optimisations::NormalizerGuard32,
|
||||
) {
|
||||
let precision = normalizer_guard.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 2];
|
||||
|
||||
/*
|
||||
|L0 A0 | |L1 A1 | |L2 A2 | |L3 A3 |
|
||||
|0001 0203| |0405 0607| |0809 1011| |1213 1415|
|
||||
|
||||
Shuffle to extract L0 and A0 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 3, 2, -1, -1, -1, -1, -1, -1, 1, 0
|
||||
|
||||
Shuffle to extract L1 and A1 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 7, 6, -1, -1, -1, -1, -1, -1, 5, 4
|
||||
|
||||
Shuffle to extract L2 and A2 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 11, 10, -1, -1, -1, -1, -1, -1, 9, 8
|
||||
|
||||
Shuffle to extract L3 and A3 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12
|
||||
*/
|
||||
|
||||
let p0_shuffle = _mm_set_epi8(-1, -1, -1, -1, -1, -1, 3, 2, -1, -1, -1, -1, -1, -1, 1, 0);
|
||||
let p1_shuffle = _mm_set_epi8(-1, -1, -1, -1, -1, -1, 7, 6, -1, -1, -1, -1, -1, -1, 5, 4);
|
||||
let p2_shuffle = _mm_set_epi8(-1, -1, -1, -1, -1, -1, 11, 10, -1, -1, -1, -1, -1, -1, 9, 8);
|
||||
let p3_shuffle = _mm_set_epi8(
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut ll_sum = _mm_set1_epi64x(half_error);
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
coeffs = coeffs_by_4.remainder();
|
||||
|
||||
for k in coeffs_by_4 {
|
||||
let coeff0_i64x2 = _mm_set1_epi64x(k[0] as i64);
|
||||
let coeff1_i64x2 = _mm_set1_epi64x(k[1] as i64);
|
||||
let coeff2_i64x2 = _mm_set1_epi64x(k[2] as i64);
|
||||
let coeff3_i64x2 = _mm_set1_epi64x(k[3] as i64);
|
||||
|
||||
let source = simd_utils::loadu_si128(src_row, x);
|
||||
|
||||
let p_i64x2 = _mm_shuffle_epi8(source, p0_shuffle);
|
||||
ll_sum = _mm_add_epi64(ll_sum, _mm_mul_epi32(p_i64x2, coeff0_i64x2));
|
||||
|
||||
let p_i64x2 = _mm_shuffle_epi8(source, p1_shuffle);
|
||||
ll_sum = _mm_add_epi64(ll_sum, _mm_mul_epi32(p_i64x2, coeff1_i64x2));
|
||||
|
||||
let p_i64x2 = _mm_shuffle_epi8(source, p2_shuffle);
|
||||
ll_sum = _mm_add_epi64(ll_sum, _mm_mul_epi32(p_i64x2, coeff2_i64x2));
|
||||
|
||||
let p_i64x2 = _mm_shuffle_epi8(source, p3_shuffle);
|
||||
ll_sum = _mm_add_epi64(ll_sum, _mm_mul_epi32(p_i64x2, coeff3_i64x2));
|
||||
|
||||
x += 4;
|
||||
}
|
||||
|
||||
let coeffs_by_2 = coeffs.chunks_exact(2);
|
||||
coeffs = coeffs_by_2.remainder();
|
||||
|
||||
for k in coeffs_by_2 {
|
||||
let coeff0_i64x2 = _mm_set1_epi64x(k[0] as i64);
|
||||
let coeff1_i64x2 = _mm_set1_epi64x(k[1] as i64);
|
||||
|
||||
let source = simd_utils::loadl_epi64(src_row, x);
|
||||
|
||||
let p_i64x2 = _mm_shuffle_epi8(source, p0_shuffle);
|
||||
ll_sum = _mm_add_epi64(ll_sum, _mm_mul_epi32(p_i64x2, coeff0_i64x2));
|
||||
|
||||
let p_i64x2 = _mm_shuffle_epi8(source, p1_shuffle);
|
||||
ll_sum = _mm_add_epi64(ll_sum, _mm_mul_epi32(p_i64x2, coeff1_i64x2));
|
||||
|
||||
x += 2;
|
||||
}
|
||||
|
||||
if let Some(&k) = coeffs.get(0) {
|
||||
let coeff0_i64x2 = _mm_set1_epi64x(k as i64);
|
||||
let source = simd_utils::loadl_epi32(src_row, x);
|
||||
|
||||
let p_i64x2 = _mm_shuffle_epi8(source, p0_shuffle);
|
||||
ll_sum = _mm_add_epi64(ll_sum, _mm_mul_epi32(p_i64x2, coeff0_i64x2));
|
||||
}
|
||||
|
||||
_mm_storeu_si128((&mut ll_buf).as_mut_ptr() as *mut __m128i, ll_sum);
|
||||
let dst_pixel = dst_row.get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = [
|
||||
normalizer_guard.clip(ll_buf[0]),
|
||||
normalizer_guard.clip(ll_buf[1]),
|
||||
]
|
||||
}
|
||||
}
|
||||
@@ -6,9 +6,6 @@ use crate::image_view::{FourRows, FourRowsMut, TypedImageView, TypedImageViewMut
|
||||
use crate::pixels::U16x3;
|
||||
use crate::simd_utils;
|
||||
|
||||
// This code is based on C-implementation from Pillow-SIMD package for Python
|
||||
// https://github.com/uploadcare/pillow-simd
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn horiz_convolution(
|
||||
src_image: TypedImageView<U16x3>,
|
||||
|
||||
+20
-11
@@ -1,7 +1,7 @@
|
||||
use std::num::NonZeroU32;
|
||||
|
||||
use crate::image_view::{ImageRows, ImageRowsMut, TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::{Pixel, PixelType, U16x3, U8x2, U8x3, U8x4, F32, I32, U16, U8};
|
||||
use crate::pixels::{Pixel, PixelType, U16x2, U16x3, U8x2, U8x3, U8x4, F32, I32, U16, U8};
|
||||
use crate::{ImageBufferError, ImageView, ImageViewMut};
|
||||
|
||||
#[derive(Debug)]
|
||||
@@ -23,16 +23,7 @@ impl<'a> Image<'a> {
|
||||
/// Create empty image with given dimensions and pixel type.
|
||||
pub fn new(width: NonZeroU32, height: NonZeroU32, pixel_type: PixelType) -> Self {
|
||||
let pixels_count = (width.get() * height.get()) as usize;
|
||||
let pixels = match pixel_type {
|
||||
PixelType::U8x2 => PixelsContainer::VecU8(vec![0; pixels_count * U8x2::size()]),
|
||||
PixelType::U8x3 => PixelsContainer::VecU8(vec![0; pixels_count * U8x3::size()]),
|
||||
PixelType::U16 => PixelsContainer::VecU8(vec![0; pixels_count * U16::size()]),
|
||||
PixelType::U16x3 => PixelsContainer::VecU8(vec![0; pixels_count * U16x3::size()]),
|
||||
PixelType::U8x4 => PixelsContainer::VecU8(vec![0; pixels_count * U8x4::size()]),
|
||||
PixelType::I32 => PixelsContainer::VecU8(vec![0; pixels_count * I32::size()]),
|
||||
PixelType::F32 => PixelsContainer::VecU8(vec![0; pixels_count * F32::size()]),
|
||||
PixelType::U8 => PixelsContainer::VecU8(vec![0; pixels_count]),
|
||||
};
|
||||
let pixels = PixelsContainer::VecU8(vec![0; pixels_count * pixel_type.size()]);
|
||||
Self {
|
||||
width,
|
||||
height,
|
||||
@@ -165,6 +156,15 @@ impl<'a> Image<'a> {
|
||||
.collect(),
|
||||
)
|
||||
}
|
||||
PixelType::U16x2 => {
|
||||
let pixels = unsafe { buffer.align_to::<U16x2>().1 };
|
||||
ImageRows::U16x2(
|
||||
pixels
|
||||
.chunks_exact(self.width.get() as usize)
|
||||
.take(rows_count)
|
||||
.collect(),
|
||||
)
|
||||
}
|
||||
PixelType::U16x3 => {
|
||||
let pixels = unsafe { buffer.align_to::<U16x3>().1 };
|
||||
ImageRows::U16x3(
|
||||
@@ -249,6 +249,15 @@ impl<'a> Image<'a> {
|
||||
.collect(),
|
||||
)
|
||||
}
|
||||
PixelType::U16x2 => {
|
||||
let pixels = unsafe { buffer.align_to_mut::<U16x2>().1 };
|
||||
ImageRowsMut::U16x2(
|
||||
pixels
|
||||
.chunks_exact_mut(width.get() as usize)
|
||||
.take(rows_count)
|
||||
.collect(),
|
||||
)
|
||||
}
|
||||
PixelType::U16x3 => {
|
||||
let pixels = unsafe { buffer.align_to_mut::<U16x3>().1 };
|
||||
ImageRowsMut::U16x3(
|
||||
|
||||
+52
-1
@@ -2,7 +2,7 @@ use std::num::NonZeroU32;
|
||||
use std::slice;
|
||||
|
||||
use crate::errors::{CropBoxError, ImageBufferError, ImageRowsError};
|
||||
use crate::pixels::{Pixel, PixelType, U16x3, U8x2, U8x3, U8x4, F32, I32, U16, U8};
|
||||
use crate::pixels::{Pixel, PixelType, U16x2, U16x3, U8x2, U8x3, U8x4, F32, I32, U16, U8};
|
||||
|
||||
pub(crate) type RowMut<'a, 'b, T> = &'a mut &'b mut [T];
|
||||
pub(crate) type TwoRows<'a, T> = (&'a [T], &'a [T]);
|
||||
@@ -32,6 +32,7 @@ pub enum ImageRows<'a> {
|
||||
U8x3(Vec<&'a [U8x3]>),
|
||||
U8x4(Vec<&'a [U8x4]>),
|
||||
U16(Vec<&'a [U16]>),
|
||||
U16x2(Vec<&'a [U16x2]>),
|
||||
U16x3(Vec<&'a [U16x3]>),
|
||||
I32(Vec<&'a [I32]>),
|
||||
F32(Vec<&'a [F32]>),
|
||||
@@ -49,6 +50,7 @@ impl<'a> ImageRows<'a> {
|
||||
ImageRows::U8x3(rows) => check_rows_count_and_size(width, height, rows),
|
||||
ImageRows::U8x4(rows) => check_rows_count_and_size(width, height, rows),
|
||||
ImageRows::U16(rows) => check_rows_count_and_size(width, height, rows),
|
||||
ImageRows::U16x2(rows) => check_rows_count_and_size(width, height, rows),
|
||||
ImageRows::U16x3(rows) => check_rows_count_and_size(width, height, rows),
|
||||
ImageRows::I32(rows) => check_rows_count_and_size(width, height, rows),
|
||||
ImageRows::F32(rows) => check_rows_count_and_size(width, height, rows),
|
||||
@@ -62,6 +64,7 @@ impl<'a> ImageRows<'a> {
|
||||
Self::U8x3(_) => PixelType::U8x3,
|
||||
Self::U8x4(_) => PixelType::U8x4,
|
||||
Self::U16(_) => PixelType::U16,
|
||||
Self::U16x2(_) => PixelType::U16x2,
|
||||
Self::U16x3(_) => PixelType::U16x3,
|
||||
Self::I32(_) => PixelType::I32,
|
||||
Self::F32(_) => PixelType::F32,
|
||||
@@ -83,6 +86,8 @@ image_rows_from!(U8, ImageRows::U8);
|
||||
image_rows_from!(U8x2, ImageRows::U8x2);
|
||||
image_rows_from!(U8x3, ImageRows::U8x3);
|
||||
image_rows_from!(U8x4, ImageRows::U8x4);
|
||||
image_rows_from!(U16, ImageRows::U16);
|
||||
image_rows_from!(U16x2, ImageRows::U16x2);
|
||||
image_rows_from!(U16x3, ImageRows::U16x3);
|
||||
image_rows_from!(I32, ImageRows::I32);
|
||||
image_rows_from!(F32, ImageRows::F32);
|
||||
@@ -95,6 +100,7 @@ pub enum ImageRowsMut<'a> {
|
||||
U8x3(Vec<&'a mut [U8x3]>),
|
||||
U8x4(Vec<&'a mut [U8x4]>),
|
||||
U16(Vec<&'a mut [U16]>),
|
||||
U16x2(Vec<&'a mut [U16x2]>),
|
||||
U16x3(Vec<&'a mut [U16x3]>),
|
||||
I32(Vec<&'a mut [I32]>),
|
||||
F32(Vec<&'a mut [F32]>),
|
||||
@@ -112,6 +118,7 @@ impl<'a> ImageRowsMut<'a> {
|
||||
Self::U8x3(rows) => check_rows_count_and_size(width, height, rows),
|
||||
Self::U8x4(rows) => check_rows_count_and_size(width, height, rows),
|
||||
Self::U16(rows) => check_rows_count_and_size(width, height, rows),
|
||||
Self::U16x2(rows) => check_rows_count_and_size(width, height, rows),
|
||||
Self::U16x3(rows) => check_rows_count_and_size(width, height, rows),
|
||||
Self::I32(rows) => check_rows_count_and_size(width, height, rows),
|
||||
Self::F32(rows) => check_rows_count_and_size(width, height, rows),
|
||||
@@ -125,6 +132,7 @@ impl<'a> ImageRowsMut<'a> {
|
||||
Self::U8x3(_) => PixelType::U8x3,
|
||||
Self::U8x4(_) => PixelType::U8x4,
|
||||
Self::U16(_) => PixelType::U16,
|
||||
Self::U16x2(_) => PixelType::U16x2,
|
||||
Self::U16x3(_) => PixelType::U16x3,
|
||||
Self::I32(_) => PixelType::I32,
|
||||
Self::F32(_) => PixelType::F32,
|
||||
@@ -210,6 +218,15 @@ impl<'a> ImageView<'a> {
|
||||
.collect(),
|
||||
)
|
||||
}
|
||||
PixelType::U16x2 => {
|
||||
let pixels = align_buffer_to(buffer)?;
|
||||
ImageRows::U16x2(
|
||||
pixels
|
||||
.chunks_exact(width.get() as usize)
|
||||
.take(rows_count)
|
||||
.collect(),
|
||||
)
|
||||
}
|
||||
PixelType::U16x3 => {
|
||||
let pixels = align_buffer_to(buffer)?;
|
||||
ImageRows::U16x3(
|
||||
@@ -407,6 +424,19 @@ impl<'a> ImageView<'a> {
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn u16x2_image(&self) -> Option<TypedImageView<U16x2>> {
|
||||
if let ImageRows::U16x2(ref rows) = self.rows {
|
||||
Some(TypedImageView {
|
||||
width: self.width,
|
||||
height: self.height,
|
||||
crop_box: self.crop_box,
|
||||
rows,
|
||||
})
|
||||
} else {
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn u16x3_image(&self) -> Option<TypedImageView<U16x3>> {
|
||||
if let ImageRows::U16x3(ref rows) = self.rows {
|
||||
Some(TypedImageView {
|
||||
@@ -634,6 +664,15 @@ impl<'a> ImageViewMut<'a> {
|
||||
.collect(),
|
||||
)
|
||||
}
|
||||
PixelType::U16x2 => {
|
||||
let pixels = align_buffer_to_mut(buffer)?;
|
||||
ImageRowsMut::U16x2(
|
||||
pixels
|
||||
.chunks_exact_mut(width.get() as usize)
|
||||
.take(rows_count)
|
||||
.collect(),
|
||||
)
|
||||
}
|
||||
PixelType::U16x3 => {
|
||||
let pixels = align_buffer_to_mut(buffer)?;
|
||||
ImageRowsMut::U16x3(
|
||||
@@ -741,6 +780,18 @@ impl<'a> ImageViewMut<'a> {
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn u16x2_image<'s>(&'s mut self) -> Option<TypedImageViewMut<'s, 'a, U16x2>> {
|
||||
if let ImageRowsMut::U16x2(rows) = &mut self.rows {
|
||||
Some(TypedImageViewMut {
|
||||
width: self.width,
|
||||
height: self.height,
|
||||
rows,
|
||||
})
|
||||
} else {
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn u16x3_image<'s>(&'s mut self) -> Option<TypedImageViewMut<'s, 'a, U16x3>> {
|
||||
if let ImageRowsMut::U16x3(rows) = &mut self.rows {
|
||||
Some(TypedImageViewMut {
|
||||
|
||||
+55
-1
@@ -1,6 +1,6 @@
|
||||
use crate::alpha::AlphaMulDiv;
|
||||
use crate::image_view::{TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::{U8x2, U8x4};
|
||||
use crate::pixels::{U16x2, U8x2, U8x4};
|
||||
use crate::{
|
||||
CpuExtensions, ImageView, ImageViewMut, MulDivImageError, MulDivImagesError, PixelType,
|
||||
};
|
||||
@@ -62,6 +62,11 @@ impl MulDiv {
|
||||
multiply_alpha(typed_src_image, typed_dst_image, self.cpu_extensions);
|
||||
Ok(())
|
||||
}
|
||||
PixelType::U16x2 => {
|
||||
let (typed_src_image, typed_dst_image) = assert_images_u16x2(src_image, dst_image)?;
|
||||
multiply_alpha(typed_src_image, typed_dst_image, self.cpu_extensions);
|
||||
Ok(())
|
||||
}
|
||||
_ => Err(MulDivImagesError::UnsupportedPixelType),
|
||||
}
|
||||
}
|
||||
@@ -79,6 +84,11 @@ impl MulDiv {
|
||||
multiply_alpha_inplace(typed_image, self.cpu_extensions);
|
||||
Ok(())
|
||||
}
|
||||
PixelType::U16x2 => {
|
||||
let typed_image = assert_image_u16x2(image)?;
|
||||
multiply_alpha_inplace(typed_image, self.cpu_extensions);
|
||||
Ok(())
|
||||
}
|
||||
_ => Err(MulDivImageError::UnsupportedPixelType),
|
||||
}
|
||||
}
|
||||
@@ -101,6 +111,11 @@ impl MulDiv {
|
||||
divide_alpha(typed_src_image, typed_dst_image, self.cpu_extensions);
|
||||
Ok(())
|
||||
}
|
||||
PixelType::U16x2 => {
|
||||
let (typed_src_image, typed_dst_image) = assert_images_u16x2(src_image, dst_image)?;
|
||||
divide_alpha(typed_src_image, typed_dst_image, self.cpu_extensions);
|
||||
Ok(())
|
||||
}
|
||||
_ => Err(MulDivImagesError::UnsupportedPixelType),
|
||||
}
|
||||
}
|
||||
@@ -118,6 +133,11 @@ impl MulDiv {
|
||||
divide_alpha_inplace(typed_image, self.cpu_extensions);
|
||||
Ok(())
|
||||
}
|
||||
PixelType::U16x2 => {
|
||||
let typed_image = assert_image_u16x2(image)?;
|
||||
divide_alpha_inplace(typed_image, self.cpu_extensions);
|
||||
Ok(())
|
||||
}
|
||||
_ => Err(MulDivImageError::UnsupportedPixelType),
|
||||
}
|
||||
}
|
||||
@@ -191,6 +211,40 @@ fn assert_image_u8x4<'a, 'b>(
|
||||
.ok_or(MulDivImageError::UnsupportedPixelType)
|
||||
}
|
||||
|
||||
#[inline]
|
||||
fn assert_images_u16x2<'s, 'd, 'da>(
|
||||
src_image: &'s ImageView<'s>,
|
||||
dst_image: &'d mut ImageViewMut<'da>,
|
||||
) -> Result<
|
||||
(
|
||||
TypedImageView<'s, 's, U16x2>,
|
||||
TypedImageViewMut<'d, 'da, U16x2>,
|
||||
),
|
||||
MulDivImagesError,
|
||||
> {
|
||||
let src_image_u16x2 = src_image
|
||||
.u16x2_image()
|
||||
.ok_or(MulDivImagesError::UnsupportedPixelType)?;
|
||||
let dst_image_u16x2 = dst_image
|
||||
.u16x2_image()
|
||||
.ok_or(MulDivImagesError::UnsupportedPixelType)?;
|
||||
if src_image_u16x2.width() != dst_image_u16x2.width()
|
||||
|| src_image_u16x2.height() != dst_image_u16x2.height()
|
||||
{
|
||||
return Err(MulDivImagesError::SizeIsDifferent);
|
||||
}
|
||||
Ok((src_image_u16x2, dst_image_u16x2))
|
||||
}
|
||||
|
||||
#[inline]
|
||||
fn assert_image_u16x2<'a, 'b>(
|
||||
image: &'a mut ImageViewMut<'b>,
|
||||
) -> Result<TypedImageViewMut<'a, 'b, U16x2>, MulDivImageError> {
|
||||
image
|
||||
.u16x2_image()
|
||||
.ok_or(MulDivImageError::UnsupportedPixelType)
|
||||
}
|
||||
|
||||
fn multiply_alpha<P>(
|
||||
src_image: TypedImageView<P>,
|
||||
dst_image: TypedImageViewMut<P>,
|
||||
|
||||
@@ -10,6 +10,7 @@ pub enum PixelType {
|
||||
U8x3,
|
||||
U8x4,
|
||||
U16,
|
||||
U16x2,
|
||||
U16x3,
|
||||
I32,
|
||||
F32,
|
||||
@@ -23,6 +24,7 @@ impl PixelType {
|
||||
Self::U8x2 => 2,
|
||||
Self::U8x3 => 3,
|
||||
Self::U16 => 2,
|
||||
Self::U16x2 => 4,
|
||||
Self::U16x3 => 6,
|
||||
_ => 4,
|
||||
}
|
||||
@@ -36,6 +38,7 @@ impl PixelType {
|
||||
Self::U8x3 => unsafe { buffer.align_to::<U8x3>().0.is_empty() },
|
||||
Self::U8x4 => unsafe { buffer.align_to::<U8x4>().0.is_empty() },
|
||||
Self::U16 => unsafe { buffer.align_to::<U16>().0.is_empty() },
|
||||
Self::U16x2 => unsafe { buffer.align_to::<U16x2>().0.is_empty() },
|
||||
Self::U16x3 => unsafe { buffer.align_to::<U16x3>().0.is_empty() },
|
||||
Self::I32 => unsafe { buffer.align_to::<I32>().0.is_empty() },
|
||||
Self::F32 => unsafe { buffer.align_to::<F32>().0.is_empty() },
|
||||
@@ -138,6 +141,14 @@ pixel_struct!(
|
||||
PixelType::U16,
|
||||
"One `u16` component per pixel (e.g. L16)"
|
||||
);
|
||||
pixel_struct!(
|
||||
U16x2,
|
||||
[u16; 2],
|
||||
u16,
|
||||
2,
|
||||
PixelType::U16x2,
|
||||
"Two `u16` components per pixel (e.g. LA)"
|
||||
);
|
||||
pixel_struct!(
|
||||
U16x3,
|
||||
[u16; 3],
|
||||
|
||||
@@ -111,6 +111,13 @@ impl Resizer {
|
||||
}
|
||||
}
|
||||
}
|
||||
PixelType::U16x2 => {
|
||||
if let Some(src_rows) = src_image.u16x2_image() {
|
||||
if let Some(dst_rows) = dst_image.u16x2_image() {
|
||||
self.resize_inner(src_rows, dst_rows);
|
||||
}
|
||||
}
|
||||
}
|
||||
PixelType::U16x3 => {
|
||||
if let Some(src_rows) = src_image.u16x3_image() {
|
||||
if let Some(dst_rows) = dst_image.u16x3_image() {
|
||||
|
||||
@@ -31,6 +31,7 @@ pub trait PixelExt: Pixel {
|
||||
PixelType::U8x3 => "u8x3",
|
||||
PixelType::U8x4 => "u8x4",
|
||||
PixelType::U16 => "u16",
|
||||
PixelType::U16x2 => "u16x2",
|
||||
PixelType::U16x3 => "u16x3",
|
||||
PixelType::I32 => "i32",
|
||||
PixelType::F32 => "f32",
|
||||
@@ -84,6 +85,20 @@ impl PixelExt for U8 {
|
||||
}
|
||||
|
||||
impl PixelExt for U8x2 {
|
||||
fn load_big_image() -> DynamicImage {
|
||||
ImageReader::open("./data/nasa-4928x3279-rgba.png")
|
||||
.unwrap()
|
||||
.decode()
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
fn load_small_image() -> DynamicImage {
|
||||
ImageReader::open("./data/nasa-852x567-rgba.png")
|
||||
.unwrap()
|
||||
.decode()
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
|
||||
img.to_luma_alpha8().into_raw()
|
||||
}
|
||||
@@ -132,6 +147,30 @@ impl PixelExt for U16 {
|
||||
}
|
||||
}
|
||||
|
||||
impl PixelExt for U16x2 {
|
||||
fn load_big_image() -> DynamicImage {
|
||||
ImageReader::open("./data/nasa-4928x3279-rgba.png")
|
||||
.unwrap()
|
||||
.decode()
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
fn load_small_image() -> DynamicImage {
|
||||
ImageReader::open("./data/nasa-852x567-rgba.png")
|
||||
.unwrap()
|
||||
.decode()
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
|
||||
img.to_luma_alpha16()
|
||||
.as_raw()
|
||||
.iter()
|
||||
.flat_map(|&c| c.to_le_bytes())
|
||||
.collect()
|
||||
}
|
||||
}
|
||||
|
||||
impl PixelExt for U16x3 {
|
||||
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
|
||||
img.to_rgb8()
|
||||
@@ -176,6 +215,7 @@ pub fn save_result(image: &Image, name: &str) {
|
||||
PixelType::U8x3 => ColorType::Rgb8,
|
||||
PixelType::U8x4 => ColorType::Rgba8,
|
||||
PixelType::U16 => ColorType::L16,
|
||||
PixelType::U16x2 => ColorType::La16,
|
||||
PixelType::U16x3 => ColorType::Rgb16,
|
||||
_ => panic!("Unsupported type of pixels"),
|
||||
};
|
||||
|
||||
+128
-9
@@ -1,6 +1,6 @@
|
||||
use std::num::NonZeroU32;
|
||||
|
||||
use fast_image_resize::pixels::{Pixel, U8x2, U8x4};
|
||||
use fast_image_resize::pixels::{Pixel, U16x2, U8x2, U8x4};
|
||||
use fast_image_resize::{
|
||||
CpuExtensions, Image, ImageRows, ImageRowsMut, ImageView, ImageViewMut, MulDiv, PixelType,
|
||||
};
|
||||
@@ -34,6 +34,16 @@ impl IntoImageRows for U8x4 {
|
||||
}
|
||||
}
|
||||
|
||||
impl IntoImageRows for U16x2 {
|
||||
fn into_image_rows(rows: Vec<&[Self]>) -> ImageRows {
|
||||
ImageRows::U16x2(rows)
|
||||
}
|
||||
|
||||
fn into_image_rows_mut(rows: Vec<&mut [Self]>) -> ImageRowsMut {
|
||||
ImageRowsMut::U16x2(rows)
|
||||
}
|
||||
}
|
||||
|
||||
const fn p2(l: u8, a: u8) -> U8x2 {
|
||||
U8x2(u16::from_le_bytes([l, a]))
|
||||
}
|
||||
@@ -61,10 +71,7 @@ where
|
||||
let res_pixels: Vec<P> = vec![result_pixel; src_size];
|
||||
let res_buffer = unsafe { res_pixels.align_to::<u8>().1 };
|
||||
|
||||
let rows: Vec<&[P]> = src_pixels
|
||||
.chunks_exact(width as usize)
|
||||
.map(|r| r.as_ref())
|
||||
.collect();
|
||||
let rows: Vec<&[P]> = src_pixels.chunks_exact(width as usize).collect();
|
||||
|
||||
let src_image_view = ImageView::new(
|
||||
NonZeroU32::new(width).unwrap(),
|
||||
@@ -103,10 +110,7 @@ where
|
||||
);
|
||||
|
||||
// Inplace
|
||||
let rows: Vec<&mut [P]> = src_pixels
|
||||
.chunks_exact_mut(width as usize)
|
||||
.map(|r| r.as_mut())
|
||||
.collect();
|
||||
let rows: Vec<&mut [P]> = src_pixels.chunks_exact_mut(width as usize).collect();
|
||||
let mut image_view = ImageViewMut::new(
|
||||
NonZeroU32::new(width).unwrap(),
|
||||
NonZeroU32::new(height).unwrap(),
|
||||
@@ -216,6 +220,58 @@ mod multiply_alpha_u8x2 {
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod multiply_alpha_u16x2 {
|
||||
use super::*;
|
||||
use fast_image_resize::pixels::U16x2;
|
||||
|
||||
const SRC_PIXELS: [U16x2; 9] = [
|
||||
U16x2([0xffff, 0x8000]),
|
||||
U16x2([0x8000, 0x8000]),
|
||||
U16x2([0, 0x8000]),
|
||||
U16x2([0xffff, 0xffff]),
|
||||
U16x2([0x8000, 0xffff]),
|
||||
U16x2([0, 0xffff]),
|
||||
U16x2([0xffff, 0]),
|
||||
U16x2([0x8000, 0]),
|
||||
U16x2([0, 0]),
|
||||
];
|
||||
const RES_PIXELS: [U16x2; 9] = [
|
||||
U16x2([0x8000, 0x8000]),
|
||||
U16x2([0x4000, 0x8000]),
|
||||
U16x2([0, 0x8000]),
|
||||
U16x2([0xffff, 0xffff]),
|
||||
U16x2([0x8000, 0xffff]),
|
||||
U16x2([0, 0xffff]),
|
||||
U16x2([0, 0]),
|
||||
U16x2([0, 0]),
|
||||
U16x2([0, 0]),
|
||||
];
|
||||
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
#[test]
|
||||
fn avx2_test() {
|
||||
for (s, r) in SRC_PIXELS.into_iter().zip(RES_PIXELS) {
|
||||
mul_div_alpha_test(Oper::Mul, s, r, CpuExtensions::Avx2);
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
#[test]
|
||||
fn sse4_test() {
|
||||
for (s, r) in SRC_PIXELS.into_iter().zip(RES_PIXELS) {
|
||||
mul_div_alpha_test(Oper::Mul, s, r, CpuExtensions::Sse4_1);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn native_test() {
|
||||
for (s, r) in SRC_PIXELS.into_iter().zip(RES_PIXELS) {
|
||||
mul_div_alpha_test(Oper::Mul, s, r, CpuExtensions::None);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Divides by alpha
|
||||
|
||||
#[cfg(test)]
|
||||
@@ -306,6 +362,69 @@ mod divide_alpha_u8x2 {
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod divide_alpha_u16x2 {
|
||||
use super::*;
|
||||
|
||||
const OPER: Oper = Oper::Div;
|
||||
const SRC_PIXELS: [U16x2; 9] = [
|
||||
U16x2([0x8000, 0x8000]),
|
||||
U16x2([0x4000, 0x8000]),
|
||||
U16x2([0, 0x8000]),
|
||||
U16x2([0xffff, 0xffff]),
|
||||
U16x2([0x8000, 0xffff]),
|
||||
U16x2([0, 0xffff]),
|
||||
U16x2([0xffff, 0]),
|
||||
U16x2([0x8000, 0]),
|
||||
U16x2([0, 0]),
|
||||
];
|
||||
const RES_PIXELS: [U16x2; 9] = [
|
||||
U16x2([0xffff, 0x8000]),
|
||||
U16x2([0x7fff, 0x8000]),
|
||||
U16x2([0, 0x8000]),
|
||||
U16x2([0xffff, 0xffff]),
|
||||
U16x2([0x8000, 0xffff]),
|
||||
U16x2([0, 0xffff]),
|
||||
U16x2([0, 0]),
|
||||
U16x2([0, 0]),
|
||||
U16x2([0, 0]),
|
||||
];
|
||||
const SIMD_RES_PIXELS: [U16x2; 9] = [
|
||||
U16x2([0xffff, 0x8000]),
|
||||
U16x2([0x8000, 0x8000]),
|
||||
U16x2([0, 0x8000]),
|
||||
U16x2([0xffff, 0xffff]),
|
||||
U16x2([0x8000, 0xffff]),
|
||||
U16x2([0, 0xffff]),
|
||||
U16x2([0, 0]),
|
||||
U16x2([0, 0]),
|
||||
U16x2([0, 0]),
|
||||
];
|
||||
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
#[test]
|
||||
fn avx2_test() {
|
||||
for (s, r) in SRC_PIXELS.into_iter().zip(SIMD_RES_PIXELS) {
|
||||
mul_div_alpha_test(OPER, s, r, CpuExtensions::Avx2);
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
#[test]
|
||||
fn sse4_test() {
|
||||
for (s, r) in SRC_PIXELS.into_iter().zip(SIMD_RES_PIXELS) {
|
||||
mul_div_alpha_test(OPER, s, r, CpuExtensions::Sse4_1);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn native_test() {
|
||||
for (s, r) in SRC_PIXELS.into_iter().zip(RES_PIXELS) {
|
||||
mul_div_alpha_test(OPER, s, r, CpuExtensions::None);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn multiply_alpha_real_image_test() {
|
||||
let mut pixels = vec![0u8; 256 * 256 * 4];
|
||||
|
||||
+54
-4
@@ -157,7 +157,7 @@ fn upscale_u8() {
|
||||
fn downscale_u8x2() {
|
||||
type P = U8x2;
|
||||
let buffer = downscale_test::<P>(ResizeAlg::Nearest, CpuExtensions::None);
|
||||
assert_eq!(testing::image_checksum::<2>(&buffer), [2920348, 11054250]);
|
||||
assert_eq!(testing::image_checksum::<2>(&buffer), [2920348, 6121802]);
|
||||
|
||||
let mut cpu_extensions_vec = vec![CpuExtensions::None];
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
@@ -168,7 +168,7 @@ fn downscale_u8x2() {
|
||||
for cpu_extensions in cpu_extensions_vec {
|
||||
let buffer =
|
||||
downscale_test::<P>(ResizeAlg::Convolution(FilterType::Lanczos3), cpu_extensions);
|
||||
assert_eq!(testing::image_checksum::<2>(&buffer), [2923557, 11054250]);
|
||||
assert_eq!(testing::image_checksum::<2>(&buffer), [2923557, 6122818]);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -178,7 +178,7 @@ fn upscale_u8x2() {
|
||||
let buffer = upscale_test::<P>(ResizeAlg::Nearest, CpuExtensions::None);
|
||||
assert_eq!(
|
||||
testing::image_checksum::<2>(&buffer),
|
||||
[1148754010, 4269569040]
|
||||
[1146218632, 2364895380]
|
||||
);
|
||||
|
||||
let mut cpu_extensions_vec = vec![CpuExtensions::None];
|
||||
@@ -192,7 +192,7 @@ fn upscale_u8x2() {
|
||||
upscale_test::<P>(ResizeAlg::Convolution(FilterType::Lanczos3), cpu_extensions);
|
||||
assert_eq!(
|
||||
testing::image_checksum::<2>(&buffer),
|
||||
[1148811406, 4269569040]
|
||||
[1146283728, 2364890194]
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -350,6 +350,56 @@ fn upscale_u16() {
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn downscale_u16x2() {
|
||||
type P = U16x2;
|
||||
let buffer = downscale_test::<P>(ResizeAlg::Nearest, CpuExtensions::None);
|
||||
assert_eq!(
|
||||
testing::image_u16_checksum::<2>(&buffer),
|
||||
[750529436, 1573303114]
|
||||
);
|
||||
|
||||
let mut cpu_extensions_vec = vec![CpuExtensions::None];
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
|
||||
cpu_extensions_vec.push(CpuExtensions::Avx2);
|
||||
}
|
||||
for cpu_extensions in cpu_extensions_vec {
|
||||
let buffer =
|
||||
downscale_test::<P>(ResizeAlg::Convolution(FilterType::Lanczos3), cpu_extensions);
|
||||
assert_eq!(
|
||||
testing::image_u16_checksum::<2>(&buffer),
|
||||
[751401243, 1573563971]
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn upscale_u16x2() {
|
||||
type P = U16x2;
|
||||
let buffer = upscale_test::<P>(ResizeAlg::Nearest, CpuExtensions::None);
|
||||
assert_eq!(
|
||||
testing::image_u16_checksum::<2>(&buffer),
|
||||
[294578188424, 607778112660]
|
||||
);
|
||||
|
||||
let mut cpu_extensions_vec = vec![CpuExtensions::None];
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
|
||||
cpu_extensions_vec.push(CpuExtensions::Avx2);
|
||||
}
|
||||
for cpu_extensions in cpu_extensions_vec {
|
||||
let buffer =
|
||||
upscale_test::<P>(ResizeAlg::Convolution(FilterType::Lanczos3), cpu_extensions);
|
||||
assert_eq!(
|
||||
testing::image_u16_checksum::<2>(&buffer),
|
||||
[294597368766, 607776760273]
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn downscale_u16x3() {
|
||||
type P = U16x3;
|
||||
|
||||
Reference in New Issue
Block a user