mirror of
https://github.com/Cykooz/fast_image_resize.git
synced 2026-10-07 17:01:09 +00:00
Initial commit
This commit is contained in:
@@ -0,0 +1,5 @@
|
||||
/target
|
||||
.*
|
||||
!/.gitignore
|
||||
data/result
|
||||
|
||||
@@ -0,0 +1,3 @@
|
||||
## [Unreleased] - ReleaseDate
|
||||
|
||||
- Initial version.
|
||||
Generated
+849
@@ -0,0 +1,849 @@
|
||||
# This file is automatically @generated by Cargo.
|
||||
# It is not intended for manual editing.
|
||||
version = 3
|
||||
|
||||
[[package]]
|
||||
name = "adler"
|
||||
version = "1.0.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f26201604c87b1e01bd3d98f8d5d9a8fcbb815e8cedb41ffccbeb4bf593a35fe"
|
||||
|
||||
[[package]]
|
||||
name = "adler32"
|
||||
version = "1.2.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "aae1277d39aeec15cb388266ecc24b11c80469deae6067e17a1a7aa9e5c1f234"
|
||||
|
||||
[[package]]
|
||||
name = "ahash"
|
||||
version = "0.4.7"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "739f4a8db6605981345c5654f3a85b056ce52f37a39d34da03f25bf2151ea16e"
|
||||
|
||||
[[package]]
|
||||
name = "atty"
|
||||
version = "0.2.14"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d9b39be18770d11421cdb1b9947a45dd3f37e93092cbf377614828a319d5fee8"
|
||||
dependencies = [
|
||||
"hermit-abi",
|
||||
"libc",
|
||||
"winapi",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "autocfg"
|
||||
version = "1.0.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "cdb031dd78e28731d87d56cc8ffef4a8f36ca26c38fe2de700543e627f8a464a"
|
||||
|
||||
[[package]]
|
||||
name = "bitflags"
|
||||
version = "1.2.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "cf1de2fe8c75bc145a2f577add951f8134889b4795d47466a54a5c846d691693"
|
||||
|
||||
[[package]]
|
||||
name = "bstr"
|
||||
version = "0.2.16"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "90682c8d613ad3373e66de8c6411e0ae2ab2571e879d2efbf73558cc66f21279"
|
||||
dependencies = [
|
||||
"lazy_static",
|
||||
"memchr",
|
||||
"regex-automata",
|
||||
"serde",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "bumpalo"
|
||||
version = "3.7.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9c59e7af012c713f529e7a3ee57ce9b31ddd858d4b512923602f74608b009631"
|
||||
|
||||
[[package]]
|
||||
name = "bytemuck"
|
||||
version = "1.7.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9966d2ab714d0f785dbac0a0396251a35280aeb42413281617d0209ab4898435"
|
||||
|
||||
[[package]]
|
||||
name = "byteorder"
|
||||
version = "1.4.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "14c189c53d098945499cdfa7ecc63567cf3886b3332b312a5b4585d8d3a6a610"
|
||||
|
||||
[[package]]
|
||||
name = "cast"
|
||||
version = "0.2.7"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "4c24dab4283a142afa2fdca129b80ad2c6284e073930f964c3a1293c225ee39a"
|
||||
dependencies = [
|
||||
"rustc_version",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "cfg-if"
|
||||
version = "1.0.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "baf1de4339761588bc0619e3cbc0120ee582ebb74b53b4efbf79117bd2da40fd"
|
||||
|
||||
[[package]]
|
||||
name = "clap"
|
||||
version = "2.33.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "37e58ac78573c40708d45522f0d80fa2f01cc4f9b4e2bf749807255454312002"
|
||||
dependencies = [
|
||||
"bitflags",
|
||||
"textwrap",
|
||||
"unicode-width",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "color_quant"
|
||||
version = "1.1.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3d7b894f5411737b7867f4827955924d7c254fc9f4d91a6aad6b097804b1018b"
|
||||
|
||||
[[package]]
|
||||
name = "crc32fast"
|
||||
version = "1.2.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "81156fece84ab6a9f2afdb109ce3ae577e42b1228441eded99bd77f627953b1a"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "criterion"
|
||||
version = "0.3.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ab327ed7354547cc2ef43cbe20ef68b988e70b4b593cbd66a2a61733123a3d23"
|
||||
dependencies = [
|
||||
"atty",
|
||||
"cast",
|
||||
"clap",
|
||||
"criterion-plot",
|
||||
"csv",
|
||||
"itertools 0.10.1",
|
||||
"lazy_static",
|
||||
"num-traits",
|
||||
"oorandom",
|
||||
"plotters",
|
||||
"rayon",
|
||||
"regex",
|
||||
"serde",
|
||||
"serde_cbor",
|
||||
"serde_derive",
|
||||
"serde_json",
|
||||
"tinytemplate",
|
||||
"walkdir",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "criterion-plot"
|
||||
version = "0.4.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e022feadec601fba1649cfa83586381a4ad31c6bf3a9ab7d408118b05dd9889d"
|
||||
dependencies = [
|
||||
"cast",
|
||||
"itertools 0.9.0",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "crossbeam-channel"
|
||||
version = "0.5.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "06ed27e177f16d65f0f0c22a213e17c696ace5dd64b14258b52f9417ccb52db4"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"crossbeam-utils",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "crossbeam-deque"
|
||||
version = "0.8.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "94af6efb46fef72616855b036a624cf27ba656ffc9be1b9a3c931cfc7749a9a9"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"crossbeam-epoch",
|
||||
"crossbeam-utils",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "crossbeam-epoch"
|
||||
version = "0.9.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "4ec02e091aa634e2c3ada4a392989e7c3116673ef0ac5b72232439094d73b7fd"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"crossbeam-utils",
|
||||
"lazy_static",
|
||||
"memoffset",
|
||||
"scopeguard",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "crossbeam-utils"
|
||||
version = "0.8.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d82cfc11ce7f2c3faef78d8a684447b40d503d9681acebed6cb728d45940c4db"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"lazy_static",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "csv"
|
||||
version = "1.1.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "22813a6dc45b335f9bade10bf7271dc477e81113e89eb251a0bc2a8a81c536e1"
|
||||
dependencies = [
|
||||
"bstr",
|
||||
"csv-core",
|
||||
"itoa",
|
||||
"ryu",
|
||||
"serde",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "csv-core"
|
||||
version = "0.1.10"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "2b2466559f260f48ad25fe6317b3c8dac77b5bdb5763ac7d9d6103530663bc90"
|
||||
dependencies = [
|
||||
"memchr",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "deflate"
|
||||
version = "0.8.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "73770f8e1fe7d64df17ca66ad28994a0a623ea497fa69486e14984e715c5d174"
|
||||
dependencies = [
|
||||
"adler32",
|
||||
"byteorder",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "either"
|
||||
version = "1.6.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e78d4f1cc4ae33bbfc157ed5d5a5ef3bc29227303d595861deb238fcec4e9457"
|
||||
|
||||
[[package]]
|
||||
name = "fallible_collections"
|
||||
version = "0.4.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "4ad9169582543d2cfe9961be1e9eaf4fc42f9aa3483f7c485717b8dde36466ea"
|
||||
dependencies = [
|
||||
"hashbrown",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "fast_image_resize"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"criterion",
|
||||
"image",
|
||||
"num-traits",
|
||||
"resize",
|
||||
"rgb",
|
||||
"thiserror",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "gif"
|
||||
version = "0.11.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "5a668f699973d0f573d15749b7002a9ac9e1f9c6b220e7b165601334c173d8de"
|
||||
dependencies = [
|
||||
"color_quant",
|
||||
"weezl",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "half"
|
||||
version = "1.7.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "62aca2aba2d62b4a7f5b33f3712cb1b0692779a56fb510499d5c0aa594daeaf3"
|
||||
|
||||
[[package]]
|
||||
name = "hashbrown"
|
||||
version = "0.9.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d7afe4a420e3fe79967a00898cc1f4db7c8a49a9333a29f8a4bd76a253d5cd04"
|
||||
dependencies = [
|
||||
"ahash",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "hermit-abi"
|
||||
version = "0.1.19"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "62b467343b94ba476dcb2500d242dadbb39557df889310ac77c5d99100aaac33"
|
||||
dependencies = [
|
||||
"libc",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "image"
|
||||
version = "0.23.14"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "24ffcb7e7244a9bf19d35bf2883b9c080c4ced3c07a9895572178cdb8f13f6a1"
|
||||
dependencies = [
|
||||
"bytemuck",
|
||||
"byteorder",
|
||||
"color_quant",
|
||||
"gif",
|
||||
"jpeg-decoder",
|
||||
"num-iter",
|
||||
"num-rational",
|
||||
"num-traits",
|
||||
"png",
|
||||
"scoped_threadpool",
|
||||
"tiff",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "itertools"
|
||||
version = "0.9.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "284f18f85651fe11e8a991b2adb42cb078325c996ed026d994719efcfca1d54b"
|
||||
dependencies = [
|
||||
"either",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "itertools"
|
||||
version = "0.10.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "69ddb889f9d0d08a67338271fa9b62996bc788c7796a5c18cf057420aaed5eaf"
|
||||
dependencies = [
|
||||
"either",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "itoa"
|
||||
version = "0.4.7"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "dd25036021b0de88a0aff6b850051563c6516d0bf53f8638938edbb9de732736"
|
||||
|
||||
[[package]]
|
||||
name = "jpeg-decoder"
|
||||
version = "0.1.22"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "229d53d58899083193af11e15917b5640cd40b29ff475a1fe4ef725deb02d0f2"
|
||||
dependencies = [
|
||||
"rayon",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "js-sys"
|
||||
version = "0.3.51"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "83bdfbace3a0e81a4253f73b49e960b053e396a11012cbd49b9b74d6a2b67062"
|
||||
dependencies = [
|
||||
"wasm-bindgen",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "lazy_static"
|
||||
version = "1.4.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e2abad23fbc42b3700f2f279844dc832adb2b2eb069b2df918f455c4e18cc646"
|
||||
|
||||
[[package]]
|
||||
name = "libc"
|
||||
version = "0.2.97"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "12b8adadd720df158f4d70dfe7ccc6adb0472d7c55ca83445f6a5ab3e36f8fb6"
|
||||
|
||||
[[package]]
|
||||
name = "log"
|
||||
version = "0.4.14"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "51b9bbe6c47d51fc3e1a9b945965946b4c44142ab8792c50835a980d362c2710"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "memchr"
|
||||
version = "2.4.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b16bd47d9e329435e309c58469fe0791c2d0d1ba96ec0954152a5ae2b04387dc"
|
||||
|
||||
[[package]]
|
||||
name = "memoffset"
|
||||
version = "0.6.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "59accc507f1338036a0477ef61afdae33cde60840f4dfe481319ce3ad116ddf9"
|
||||
dependencies = [
|
||||
"autocfg",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "miniz_oxide"
|
||||
version = "0.3.7"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "791daaae1ed6889560f8c4359194f56648355540573244a5448a83ba1ecc7435"
|
||||
dependencies = [
|
||||
"adler32",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "miniz_oxide"
|
||||
version = "0.4.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "a92518e98c078586bc6c934028adcca4c92a53d6a958196de835170a01d84e4b"
|
||||
dependencies = [
|
||||
"adler",
|
||||
"autocfg",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "num-integer"
|
||||
version = "0.1.44"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d2cc698a63b549a70bc047073d2949cce27cd1c7b0a4a862d08a8031bc2801db"
|
||||
dependencies = [
|
||||
"autocfg",
|
||||
"num-traits",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "num-iter"
|
||||
version = "0.1.42"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b2021c8337a54d21aca0d59a92577a029af9431cb59b909b03252b9c164fad59"
|
||||
dependencies = [
|
||||
"autocfg",
|
||||
"num-integer",
|
||||
"num-traits",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "num-rational"
|
||||
version = "0.3.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "12ac428b1cb17fce6f731001d307d351ec70a6d202fc2e60f7d4c5e42d8f4f07"
|
||||
dependencies = [
|
||||
"autocfg",
|
||||
"num-integer",
|
||||
"num-traits",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "num-traits"
|
||||
version = "0.2.14"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9a64b1ec5cda2586e284722486d802acf1f7dbdc623e2bfc57e65ca1cd099290"
|
||||
dependencies = [
|
||||
"autocfg",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "num_cpus"
|
||||
version = "1.13.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "05499f3756671c15885fee9034446956fff3f243d6077b91e5767df161f766b3"
|
||||
dependencies = [
|
||||
"hermit-abi",
|
||||
"libc",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "oorandom"
|
||||
version = "11.1.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0ab1bc2a289d34bd04a330323ac98a1b4bc82c9d9fcb1e66b63caa84da26b575"
|
||||
|
||||
[[package]]
|
||||
name = "plotters"
|
||||
version = "0.3.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "32a3fd9ec30b9749ce28cd91f255d569591cdf937fe280c312143e3c4bad6f2a"
|
||||
dependencies = [
|
||||
"num-traits",
|
||||
"plotters-backend",
|
||||
"plotters-svg",
|
||||
"wasm-bindgen",
|
||||
"web-sys",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "plotters-backend"
|
||||
version = "0.3.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d88417318da0eaf0fdcdb51a0ee6c3bed624333bff8f946733049380be67ac1c"
|
||||
|
||||
[[package]]
|
||||
name = "plotters-svg"
|
||||
version = "0.3.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "521fa9638fa597e1dc53e9412a4f9cefb01187ee1f7413076f9e6749e2885ba9"
|
||||
dependencies = [
|
||||
"plotters-backend",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "png"
|
||||
version = "0.16.8"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3c3287920cb847dee3de33d301c463fba14dda99db24214ddf93f83d3021f4c6"
|
||||
dependencies = [
|
||||
"bitflags",
|
||||
"crc32fast",
|
||||
"deflate",
|
||||
"miniz_oxide 0.3.7",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "proc-macro2"
|
||||
version = "1.0.27"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f0d8caf72986c1a598726adc988bb5984792ef84f5ee5aa50209145ee8077038"
|
||||
dependencies = [
|
||||
"unicode-xid",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "quote"
|
||||
version = "1.0.9"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "c3d0b9745dc2debf507c8422de05d7226cc1f0644216dfdfead988f9b1ab32a7"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rayon"
|
||||
version = "1.5.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "c06aca804d41dbc8ba42dfd964f0d01334eceb64314b9ecf7c5fad5188a06d90"
|
||||
dependencies = [
|
||||
"autocfg",
|
||||
"crossbeam-deque",
|
||||
"either",
|
||||
"rayon-core",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rayon-core"
|
||||
version = "1.9.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d78120e2c850279833f1dd3582f730c4ab53ed95aeaaaa862a2a5c71b1656d8e"
|
||||
dependencies = [
|
||||
"crossbeam-channel",
|
||||
"crossbeam-deque",
|
||||
"crossbeam-utils",
|
||||
"lazy_static",
|
||||
"num_cpus",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "regex"
|
||||
version = "1.5.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d07a8629359eb56f1e2fb1652bb04212c072a87ba68546a04065d525673ac461"
|
||||
dependencies = [
|
||||
"regex-syntax",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "regex-automata"
|
||||
version = "0.1.10"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "6c230d73fb8d8c1b9c0b3135c5142a8acee3a0558fb8db5cf1cb65f8d7862132"
|
||||
|
||||
[[package]]
|
||||
name = "regex-syntax"
|
||||
version = "0.6.25"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f497285884f3fcff424ffc933e56d7cbca511def0c9831a7f9b5f6153e3cc89b"
|
||||
|
||||
[[package]]
|
||||
name = "resize"
|
||||
version = "0.7.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "a2a32135a3cd4df690ee74dc0e1a77ffe2cdd4611b4e58b84df26a856d92b4ef"
|
||||
dependencies = [
|
||||
"fallible_collections",
|
||||
"rgb",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rgb"
|
||||
version = "0.8.27"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "8fddb3b23626145d1776addfc307e1a1851f60ef6ca64f376bcb889697144cf0"
|
||||
dependencies = [
|
||||
"bytemuck",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rustc_version"
|
||||
version = "0.4.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "bfa0f585226d2e68097d4f95d113b15b83a82e819ab25717ec0590d9584ef366"
|
||||
dependencies = [
|
||||
"semver",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "ryu"
|
||||
version = "1.0.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "71d301d4193d031abdd79ff7e3dd721168a9572ef3fe51a1517aba235bd8f86e"
|
||||
|
||||
[[package]]
|
||||
name = "same-file"
|
||||
version = "1.0.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "93fc1dc3aaa9bfed95e02e6eadabb4baf7e3078b0bd1b4d7b6b0b68378900502"
|
||||
dependencies = [
|
||||
"winapi-util",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "scoped_threadpool"
|
||||
version = "0.1.9"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "1d51f5df5af43ab3f1360b429fa5e0152ac5ce8c0bd6485cae490332e96846a8"
|
||||
|
||||
[[package]]
|
||||
name = "scopeguard"
|
||||
version = "1.1.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d29ab0c6d3fc0ee92fe66e2d99f700eab17a8d57d1c1d3b748380fb20baa78cd"
|
||||
|
||||
[[package]]
|
||||
name = "semver"
|
||||
version = "1.0.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "5f3aac57ee7f3272d8395c6e4f502f434f0e289fcd62876f70daa008c20dcabe"
|
||||
|
||||
[[package]]
|
||||
name = "serde"
|
||||
version = "1.0.126"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ec7505abeacaec74ae4778d9d9328fe5a5d04253220a85c4ee022239fc996d03"
|
||||
|
||||
[[package]]
|
||||
name = "serde_cbor"
|
||||
version = "0.11.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "1e18acfa2f90e8b735b2836ab8d538de304cbb6729a7360729ea5a895d15a622"
|
||||
dependencies = [
|
||||
"half",
|
||||
"serde",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "serde_derive"
|
||||
version = "1.0.126"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "963a7dbc9895aeac7ac90e74f34a5d5261828f79df35cbed41e10189d3804d43"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"syn",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "serde_json"
|
||||
version = "1.0.64"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "799e97dc9fdae36a5c8b8f2cae9ce2ee9fdce2058c57a93e6099d919fd982f79"
|
||||
dependencies = [
|
||||
"itoa",
|
||||
"ryu",
|
||||
"serde",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "syn"
|
||||
version = "1.0.73"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f71489ff30030d2ae598524f61326b902466f72a0fb1a8564c001cc63425bcc7"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"unicode-xid",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "textwrap"
|
||||
version = "0.11.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d326610f408c7a4eb6f51c37c330e496b08506c9457c9d34287ecc38809fb060"
|
||||
dependencies = [
|
||||
"unicode-width",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "thiserror"
|
||||
version = "1.0.26"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "93119e4feac1cbe6c798c34d3a53ea0026b0b1de6a120deef895137c0529bfe2"
|
||||
dependencies = [
|
||||
"thiserror-impl",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "thiserror-impl"
|
||||
version = "1.0.26"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "060d69a0afe7796bf42e9e2ff91f5ee691fb15c53d38b4b62a9a53eb23164745"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"syn",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "tiff"
|
||||
version = "0.6.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9a53f4706d65497df0c4349241deddf35f84cee19c87ed86ea8ca590f4464437"
|
||||
dependencies = [
|
||||
"jpeg-decoder",
|
||||
"miniz_oxide 0.4.4",
|
||||
"weezl",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "tinytemplate"
|
||||
version = "1.2.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "be4d6b5f19ff7664e8c98d03e2139cb510db9b0a60b55f8e8709b689d939b6bc"
|
||||
dependencies = [
|
||||
"serde",
|
||||
"serde_json",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "unicode-width"
|
||||
version = "0.1.8"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9337591893a19b88d8d87f2cec1e73fad5cdfd10e5a6f349f498ad6ea2ffb1e3"
|
||||
|
||||
[[package]]
|
||||
name = "unicode-xid"
|
||||
version = "0.2.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "8ccb82d61f80a663efe1f787a51b16b5a51e3314d6ac365b08639f52387b33f3"
|
||||
|
||||
[[package]]
|
||||
name = "walkdir"
|
||||
version = "2.3.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "808cf2735cd4b6866113f648b791c6adc5714537bc222d9347bb203386ffda56"
|
||||
dependencies = [
|
||||
"same-file",
|
||||
"winapi",
|
||||
"winapi-util",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "wasm-bindgen"
|
||||
version = "0.2.74"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d54ee1d4ed486f78874278e63e4069fc1ab9f6a18ca492076ffb90c5eb2997fd"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"wasm-bindgen-macro",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "wasm-bindgen-backend"
|
||||
version = "0.2.74"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3b33f6a0694ccfea53d94db8b2ed1c3a8a4c86dd936b13b9f0a15ec4a451b900"
|
||||
dependencies = [
|
||||
"bumpalo",
|
||||
"lazy_static",
|
||||
"log",
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"syn",
|
||||
"wasm-bindgen-shared",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "wasm-bindgen-macro"
|
||||
version = "0.2.74"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "088169ca61430fe1e58b8096c24975251700e7b1f6fd91cc9d59b04fb9b18bd4"
|
||||
dependencies = [
|
||||
"quote",
|
||||
"wasm-bindgen-macro-support",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "wasm-bindgen-macro-support"
|
||||
version = "0.2.74"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "be2241542ff3d9f241f5e2cb6dd09b37efe786df8851c54957683a49f0987a97"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"syn",
|
||||
"wasm-bindgen-backend",
|
||||
"wasm-bindgen-shared",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "wasm-bindgen-shared"
|
||||
version = "0.2.74"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d7cff876b8f18eed75a66cf49b65e7f967cb354a7aa16003fb55dbfd25b44b4f"
|
||||
|
||||
[[package]]
|
||||
name = "web-sys"
|
||||
version = "0.3.51"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e828417b379f3df7111d3a2a9e5753706cae29c41f7c4029ee9fd77f3e09e582"
|
||||
dependencies = [
|
||||
"js-sys",
|
||||
"wasm-bindgen",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "weezl"
|
||||
version = "0.1.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d8b77fdfd5a253be4ab714e4ffa3c49caf146b4de743e97510c0656cf90f1e8e"
|
||||
|
||||
[[package]]
|
||||
name = "winapi"
|
||||
version = "0.3.9"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "5c839a674fcd7a98952e593242ea400abe93992746761e38641405d28b00f419"
|
||||
dependencies = [
|
||||
"winapi-i686-pc-windows-gnu",
|
||||
"winapi-x86_64-pc-windows-gnu",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "winapi-i686-pc-windows-gnu"
|
||||
version = "0.4.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ac3b87c63620426dd9b991e5ce0329eff545bccbbb34f3be09ff6fb6ab51b7b6"
|
||||
|
||||
[[package]]
|
||||
name = "winapi-util"
|
||||
version = "0.1.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "70ec6ce85bb158151cae5e5c87f95a8e97d2c0c4b001223f33a334e3ce5de178"
|
||||
dependencies = [
|
||||
"winapi",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "winapi-x86_64-pc-windows-gnu"
|
||||
version = "0.4.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "712e227841d057c1ee1cd2fb22fa7e5a5461ae8e48fa2ca79ec42cfc1931183f"
|
||||
+60
@@ -0,0 +1,60 @@
|
||||
[package]
|
||||
name = "fast_image_resize"
|
||||
version = "0.1.0"
|
||||
authors = ["Kirill Kuzminykh <[email protected]>"]
|
||||
edition = "2018"
|
||||
license = "MIT OR Apache-2.0"
|
||||
description = "Library for fast image resizing with using of SIMD instructions"
|
||||
readme = "README.md"
|
||||
keywords = ["image", "resize"]
|
||||
repository = "https://github.com/cykooz/fast_image_resize"
|
||||
documentation = "https://docs.rs/crate/fast_image_resize"
|
||||
|
||||
# See more keys and their definitions at https://doc.rust-lang.org/cargo/reference/manifest.html
|
||||
|
||||
[dependencies]
|
||||
num-traits = "0.2.14"
|
||||
thiserror = "1.0.26"
|
||||
|
||||
|
||||
[dev-dependencies]
|
||||
criterion = "0.3.4"
|
||||
image = "0.23.14"
|
||||
resize = "0.7.0"
|
||||
rgb = "0.8.25"
|
||||
|
||||
|
||||
[[bench]]
|
||||
name = "bench_resize"
|
||||
harness = false
|
||||
|
||||
|
||||
[[bench]]
|
||||
name = "bench_alpha"
|
||||
harness = false
|
||||
|
||||
|
||||
[[bench]]
|
||||
name = "bench_compare"
|
||||
harness = false
|
||||
|
||||
|
||||
[profile.dev.package.'*']
|
||||
opt-level = 3
|
||||
|
||||
|
||||
[profile.release]
|
||||
#debug = true
|
||||
#lto = true
|
||||
opt-level = 3
|
||||
#codegen-units = 1
|
||||
|
||||
|
||||
[package.metadata.release]
|
||||
pre-release-replacements = [
|
||||
{file="CHANGELOG.md", search="Unreleased", replace="{{version}}"},
|
||||
{file="CHANGELOG.md", search="ReleaseDate", replace="{{date}}"}
|
||||
]
|
||||
|
||||
# Header of next release in CHANGELOG.md:
|
||||
# ## [Unreleased] - ReleaseDate
|
||||
+201
@@ -0,0 +1,201 @@
|
||||
Apache License
|
||||
Version 2.0, January 2004
|
||||
http://www.apache.org/licenses/
|
||||
|
||||
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
||||
|
||||
1. Definitions.
|
||||
|
||||
"License" shall mean the terms and conditions for use, reproduction,
|
||||
and distribution as defined by Sections 1 through 9 of this document.
|
||||
|
||||
"Licensor" shall mean the copyright owner or entity authorized by
|
||||
the copyright owner that is granting the License.
|
||||
|
||||
"Legal Entity" shall mean the union of the acting entity and all
|
||||
other entities that control, are controlled by, or are under common
|
||||
control with that entity. For the purposes of this definition,
|
||||
"control" means (i) the power, direct or indirect, to cause the
|
||||
direction or management of such entity, whether by contract or
|
||||
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
||||
outstanding shares, or (iii) beneficial ownership of such entity.
|
||||
|
||||
"You" (or "Your") shall mean an individual or Legal Entity
|
||||
exercising permissions granted by this License.
|
||||
|
||||
"Source" form shall mean the preferred form for making modifications,
|
||||
including but not limited to software source code, documentation
|
||||
source, and configuration files.
|
||||
|
||||
"Object" form shall mean any form resulting from mechanical
|
||||
transformation or translation of a Source form, including but
|
||||
not limited to compiled object code, generated documentation,
|
||||
and conversions to other media types.
|
||||
|
||||
"Work" shall mean the work of authorship, whether in Source or
|
||||
Object form, made available under the License, as indicated by a
|
||||
copyright notice that is included in or attached to the work
|
||||
(an example is provided in the Appendix below).
|
||||
|
||||
"Derivative Works" shall mean any work, whether in Source or Object
|
||||
form, that is based on (or derived from) the Work and for which the
|
||||
editorial revisions, annotations, elaborations, or other modifications
|
||||
represent, as a whole, an original work of authorship. For the purposes
|
||||
of this License, Derivative Works shall not include works that remain
|
||||
separable from, or merely link (or bind by name) to the interfaces of,
|
||||
the Work and Derivative Works thereof.
|
||||
|
||||
"Contribution" shall mean any work of authorship, including
|
||||
the original version of the Work and any modifications or additions
|
||||
to that Work or Derivative Works thereof, that is intentionally
|
||||
submitted to Licensor for inclusion in the Work by the copyright owner
|
||||
or by an individual or Legal Entity authorized to submit on behalf of
|
||||
the copyright owner. For the purposes of this definition, "submitted"
|
||||
means any form of electronic, verbal, or written communication sent
|
||||
to the Licensor or its representatives, including but not limited to
|
||||
communication on electronic mailing lists, source code control systems,
|
||||
and issue tracking systems that are managed by, or on behalf of, the
|
||||
Licensor for the purpose of discussing and improving the Work, but
|
||||
excluding communication that is conspicuously marked or otherwise
|
||||
designated in writing by the copyright owner as "Not a Contribution."
|
||||
|
||||
"Contributor" shall mean Licensor and any individual or Legal Entity
|
||||
on behalf of whom a Contribution has been received by Licensor and
|
||||
subsequently incorporated within the Work.
|
||||
|
||||
2. Grant of Copyright License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
copyright license to reproduce, prepare Derivative Works of,
|
||||
publicly display, publicly perform, sublicense, and distribute the
|
||||
Work and such Derivative Works in Source or Object form.
|
||||
|
||||
3. Grant of Patent License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
(except as stated in this section) patent license to make, have made,
|
||||
use, offer to sell, sell, import, and otherwise transfer the Work,
|
||||
where such license applies only to those patent claims licensable
|
||||
by such Contributor that are necessarily infringed by their
|
||||
Contribution(s) alone or by combination of their Contribution(s)
|
||||
with the Work to which such Contribution(s) was submitted. If You
|
||||
institute patent litigation against any entity (including a
|
||||
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
||||
or a Contribution incorporated within the Work constitutes direct
|
||||
or contributory patent infringement, then any patent licenses
|
||||
granted to You under this License for that Work shall terminate
|
||||
as of the date such litigation is filed.
|
||||
|
||||
4. Redistribution. You may reproduce and distribute copies of the
|
||||
Work or Derivative Works thereof in any medium, with or without
|
||||
modifications, and in Source or Object form, provided that You
|
||||
meet the following conditions:
|
||||
|
||||
(a) You must give any other recipients of the Work or
|
||||
Derivative Works a copy of this License; and
|
||||
|
||||
(b) You must cause any modified files to carry prominent notices
|
||||
stating that You changed the files; and
|
||||
|
||||
(c) You must retain, in the Source form of any Derivative Works
|
||||
that You distribute, all copyright, patent, trademark, and
|
||||
attribution notices from the Source form of the Work,
|
||||
excluding those notices that do not pertain to any part of
|
||||
the Derivative Works; and
|
||||
|
||||
(d) If the Work includes a "NOTICE" text file as part of its
|
||||
distribution, then any Derivative Works that You distribute must
|
||||
include a readable copy of the attribution notices contained
|
||||
within such NOTICE file, excluding those notices that do not
|
||||
pertain to any part of the Derivative Works, in at least one
|
||||
of the following places: within a NOTICE text file distributed
|
||||
as part of the Derivative Works; within the Source form or
|
||||
documentation, if provided along with the Derivative Works; or,
|
||||
within a display generated by the Derivative Works, if and
|
||||
wherever such third-party notices normally appear. The contents
|
||||
of the NOTICE file are for informational purposes only and
|
||||
do not modify the License. You may add Your own attribution
|
||||
notices within Derivative Works that You distribute, alongside
|
||||
or as an addendum to the NOTICE text from the Work, provided
|
||||
that such additional attribution notices cannot be construed
|
||||
as modifying the License.
|
||||
|
||||
You may add Your own copyright statement to Your modifications and
|
||||
may provide additional or different license terms and conditions
|
||||
for use, reproduction, or distribution of Your modifications, or
|
||||
for any such Derivative Works as a whole, provided Your use,
|
||||
reproduction, and distribution of the Work otherwise complies with
|
||||
the conditions stated in this License.
|
||||
|
||||
5. Submission of Contributions. Unless You explicitly state otherwise,
|
||||
any Contribution intentionally submitted for inclusion in the Work
|
||||
by You to the Licensor shall be under the terms and conditions of
|
||||
this License, without any additional terms or conditions.
|
||||
Notwithstanding the above, nothing herein shall supersede or modify
|
||||
the terms of any separate license agreement you may have executed
|
||||
with Licensor regarding such Contributions.
|
||||
|
||||
6. Trademarks. This License does not grant permission to use the trade
|
||||
names, trademarks, service marks, or product names of the Licensor,
|
||||
except as required for reasonable and customary use in describing the
|
||||
origin of the Work and reproducing the content of the NOTICE file.
|
||||
|
||||
7. Disclaimer of Warranty. Unless required by applicable law or
|
||||
agreed to in writing, Licensor provides the Work (and each
|
||||
Contributor provides its Contributions) on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
||||
implied, including, without limitation, any warranties or conditions
|
||||
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
||||
PARTICULAR PURPOSE. You are solely responsible for determining the
|
||||
appropriateness of using or redistributing the Work and assume any
|
||||
risks associated with Your exercise of permissions under this License.
|
||||
|
||||
8. Limitation of Liability. In no event and under no legal theory,
|
||||
whether in tort (including negligence), contract, or otherwise,
|
||||
unless required by applicable law (such as deliberate and grossly
|
||||
negligent acts) or agreed to in writing, shall any Contributor be
|
||||
liable to You for damages, including any direct, indirect, special,
|
||||
incidental, or consequential damages of any character arising as a
|
||||
result of this License or out of the use or inability to use the
|
||||
Work (including but not limited to damages for loss of goodwill,
|
||||
work stoppage, computer failure or malfunction, or any and all
|
||||
other commercial damages or losses), even if such Contributor
|
||||
has been advised of the possibility of such damages.
|
||||
|
||||
9. Accepting Warranty or Additional Liability. While redistributing
|
||||
the Work or Derivative Works thereof, You may choose to offer,
|
||||
and charge a fee for, acceptance of support, warranty, indemnity,
|
||||
or other liability obligations and/or rights consistent with this
|
||||
License. However, in accepting such obligations, You may act only
|
||||
on Your own behalf and on Your sole responsibility, not on behalf
|
||||
of any other Contributor, and only if You agree to indemnify,
|
||||
defend, and hold each Contributor harmless for any liability
|
||||
incurred by, or claims asserted against, such Contributor by reason
|
||||
of your accepting any such warranty or additional liability.
|
||||
|
||||
END OF TERMS AND CONDITIONS
|
||||
|
||||
APPENDIX: How to apply the Apache License to your work.
|
||||
|
||||
To apply the Apache License to your work, attach the following
|
||||
boilerplate notice, with the fields enclosed by brackets "[]"
|
||||
replaced with your own identifying information. (Don't include
|
||||
the brackets!) The text should be enclosed in the appropriate
|
||||
comment syntax for the file format. We also recommend that a
|
||||
file or class name and description of purpose be included on the
|
||||
same "printed page" as the copyright notice for easier
|
||||
identification within third-party archives.
|
||||
|
||||
Copyright (c) 2021 Kirill Kuzminykh
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
+21
@@ -0,0 +1,21 @@
|
||||
MIT License
|
||||
|
||||
Copyright (c) 2021 Kirill Kuzminykh
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
@@ -0,0 +1,49 @@
|
||||
# fast_image_resize
|
||||
|
||||
Library for fast image resizing with using of SIMD instructions.
|
||||
|
||||
[CHANGELOG](https://github.com/Cykooz/fast_image_resize/blob/master/CHANGELOG.md)
|
||||
|
||||
## Examples
|
||||
|
||||
```rust
|
||||
use std::num::NonZeroU32;
|
||||
|
||||
use fast_image_resize::{
|
||||
CropBox, FilterType, ImageData, PixelType, ResizeAlg, Resizer, SrcImageView,
|
||||
};
|
||||
|
||||
fn resize_lanczos3(src_pixels: &[u8], width: NonZeroU32, height: NonZeroU32) -> Vec<u8> {
|
||||
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
|
||||
let src_image = ImageData::new(width, height, src_pixels, PixelType::U8x4).unwrap();
|
||||
let dst_width = NonZeroU32::new(1024).unwrap();
|
||||
let dst_height = NonZeroU32::new(768).unwrap();
|
||||
let mut dst_image = ImageData::new_owned(dst_width, dst_height, src_image.pixel_type());
|
||||
|
||||
let src_view = src_image.src_view();
|
||||
let mut dst_view = dst_image.dst_view();
|
||||
resizer.resize(&src_view, &mut dst_view);
|
||||
|
||||
dst_image.get_buffer().to_owned()
|
||||
}
|
||||
|
||||
fn resize_cropped_image(mut src_view: SrcImageView) -> ImageData<Vec<u8>> {
|
||||
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
|
||||
src_view
|
||||
.set_crop_box(CropBox {
|
||||
left: 10,
|
||||
top: 10,
|
||||
width: NonZeroU32::new(100).unwrap(),
|
||||
height: NonZeroU32::new(200).unwrap(),
|
||||
})
|
||||
.unwrap();
|
||||
let dst_width = NonZeroU32::new(1024).unwrap();
|
||||
let dst_height = NonZeroU32::new(768).unwrap();
|
||||
let mut dst_image = ImageData::new_owned(dst_width, dst_height, src_view.pixel_type());
|
||||
|
||||
let mut dst_view = dst_image.dst_view();
|
||||
resizer.resize(&src_view, &mut dst_view);
|
||||
|
||||
dst_image
|
||||
}
|
||||
```
|
||||
@@ -0,0 +1,160 @@
|
||||
use criterion::{criterion_group, criterion_main, Criterion};
|
||||
|
||||
use fast_image_resize::{CpuExtensions, ImageData, MulDiv, PixelType};
|
||||
use std::num::NonZeroU32;
|
||||
|
||||
const fn p(r: u8, g: u8, b: u8, a: u8) -> u32 {
|
||||
u32::from_le_bytes([r, g, b, a])
|
||||
}
|
||||
|
||||
// Multiplies by alpha
|
||||
|
||||
fn get_src_image(width: NonZeroU32, height: NonZeroU32, pixel: u32) -> ImageData<Vec<u8>> {
|
||||
let rgba: [u8; 4] = pixel.to_le_bytes();
|
||||
let buf_size = (width.get() * height.get()) as usize * 4;
|
||||
let mut buffer = vec![0u8; buf_size];
|
||||
buffer.chunks_exact_mut(4).for_each(|c| {
|
||||
c[0] = rgba[0];
|
||||
c[1] = rgba[1];
|
||||
c[2] = rgba[2];
|
||||
c[3] = rgba[3];
|
||||
});
|
||||
ImageData::new(width, height, buffer, PixelType::U8x4).unwrap()
|
||||
}
|
||||
|
||||
fn multiplies_alpha_avx2(c: &mut Criterion) {
|
||||
let width = NonZeroU32::new(4096).unwrap();
|
||||
let height = NonZeroU32::new(2048).unwrap();
|
||||
let src_data = get_src_image(width, height, p(255, 128, 0, 128));
|
||||
let mut dst_data = ImageData::new_owned(width, height, PixelType::U8x4);
|
||||
let src_view = src_data.src_view();
|
||||
let mut dst_view = dst_data.dst_view();
|
||||
let mut alpha_mul_div: MulDiv = Default::default();
|
||||
unsafe {
|
||||
alpha_mul_div.set_cpu_extensions(CpuExtensions::Avx2);
|
||||
}
|
||||
|
||||
c.bench_function("Multiplies alpha AVX2", |b| {
|
||||
b.iter(|| {
|
||||
alpha_mul_div
|
||||
.multiply_alpha(&src_view, &mut dst_view)
|
||||
.unwrap();
|
||||
})
|
||||
});
|
||||
}
|
||||
|
||||
fn multiplies_alpha_sse2(c: &mut Criterion) {
|
||||
let width = NonZeroU32::new(4096).unwrap();
|
||||
let height = NonZeroU32::new(2048).unwrap();
|
||||
let src_data = get_src_image(width, height, p(255, 128, 0, 128));
|
||||
let mut dst_data = ImageData::new_owned(width, height, PixelType::U8x4);
|
||||
let src_view = src_data.src_view();
|
||||
let mut dst_view = dst_data.dst_view();
|
||||
let mut alpha_mul_div: MulDiv = Default::default();
|
||||
unsafe {
|
||||
alpha_mul_div.set_cpu_extensions(CpuExtensions::Sse2);
|
||||
}
|
||||
|
||||
c.bench_function("Multiplies alpha SSE2", |b| {
|
||||
b.iter(|| {
|
||||
alpha_mul_div
|
||||
.multiply_alpha(&src_view, &mut dst_view)
|
||||
.unwrap();
|
||||
})
|
||||
});
|
||||
}
|
||||
|
||||
fn multiplies_alpha_native(c: &mut Criterion) {
|
||||
let width = NonZeroU32::new(4096).unwrap();
|
||||
let height = NonZeroU32::new(2048).unwrap();
|
||||
let src_data = get_src_image(width, height, p(255, 128, 0, 128));
|
||||
let mut dst_data = ImageData::new_owned(width, height, PixelType::U8x4);
|
||||
let src_view = src_data.src_view();
|
||||
let mut dst_view = dst_data.dst_view();
|
||||
let mut alpha_mul_div: MulDiv = Default::default();
|
||||
unsafe {
|
||||
alpha_mul_div.set_cpu_extensions(CpuExtensions::None);
|
||||
}
|
||||
|
||||
c.bench_function("Multiplies alpha native", |b| {
|
||||
b.iter(|| {
|
||||
alpha_mul_div
|
||||
.multiply_alpha(&src_view, &mut dst_view)
|
||||
.unwrap();
|
||||
})
|
||||
});
|
||||
}
|
||||
|
||||
fn divides_alpha_avx2(c: &mut Criterion) {
|
||||
let width = NonZeroU32::new(4096).unwrap();
|
||||
let height = NonZeroU32::new(2048).unwrap();
|
||||
let src_data = get_src_image(width, height, p(128, 64, 0, 128));
|
||||
let mut dst_data = ImageData::new_owned(width, height, PixelType::U8x4);
|
||||
let src_view = src_data.src_view();
|
||||
let mut dst_view = dst_data.dst_view();
|
||||
let mut alpha_mul_div: MulDiv = Default::default();
|
||||
unsafe {
|
||||
alpha_mul_div.set_cpu_extensions(CpuExtensions::Avx2);
|
||||
}
|
||||
|
||||
c.bench_function("Divides alpha AVX2", |b| {
|
||||
b.iter(|| {
|
||||
alpha_mul_div
|
||||
.divide_alpha(&src_view, &mut dst_view)
|
||||
.unwrap();
|
||||
})
|
||||
});
|
||||
}
|
||||
|
||||
fn divides_alpha_sse2(c: &mut Criterion) {
|
||||
let width = NonZeroU32::new(4096).unwrap();
|
||||
let height = NonZeroU32::new(2048).unwrap();
|
||||
let src_data = get_src_image(width, height, p(128, 64, 0, 128));
|
||||
let mut dst_data = ImageData::new_owned(width, height, PixelType::U8x4);
|
||||
let src_view = src_data.src_view();
|
||||
let mut dst_view = dst_data.dst_view();
|
||||
let mut alpha_mul_div: MulDiv = Default::default();
|
||||
unsafe {
|
||||
alpha_mul_div.set_cpu_extensions(CpuExtensions::Sse2);
|
||||
}
|
||||
|
||||
c.bench_function("Divides alpha SSE2", |b| {
|
||||
b.iter(|| {
|
||||
alpha_mul_div
|
||||
.divide_alpha(&src_view, &mut dst_view)
|
||||
.unwrap();
|
||||
})
|
||||
});
|
||||
}
|
||||
|
||||
fn divides_alpha_native(c: &mut Criterion) {
|
||||
let width = NonZeroU32::new(4096).unwrap();
|
||||
let height = NonZeroU32::new(2048).unwrap();
|
||||
let src_data = get_src_image(width, height, p(128, 64, 0, 128));
|
||||
let mut dst_data = ImageData::new_owned(width, height, PixelType::U8x4);
|
||||
let src_view = src_data.src_view();
|
||||
let mut dst_view = dst_data.dst_view();
|
||||
let mut alpha_mul_div: MulDiv = Default::default();
|
||||
unsafe {
|
||||
alpha_mul_div.set_cpu_extensions(CpuExtensions::None);
|
||||
}
|
||||
|
||||
c.bench_function("Divides alpha native", |b| {
|
||||
b.iter(|| {
|
||||
alpha_mul_div
|
||||
.divide_alpha(&src_view, &mut dst_view)
|
||||
.unwrap();
|
||||
})
|
||||
});
|
||||
}
|
||||
|
||||
criterion_group!(
|
||||
benches,
|
||||
multiplies_alpha_avx2,
|
||||
multiplies_alpha_sse2,
|
||||
multiplies_alpha_native,
|
||||
divides_alpha_avx2,
|
||||
divides_alpha_sse2,
|
||||
divides_alpha_native,
|
||||
);
|
||||
criterion_main!(benches);
|
||||
@@ -0,0 +1,153 @@
|
||||
use std::num::NonZeroU32;
|
||||
|
||||
use criterion::{criterion_group, criterion_main, BenchmarkId, Criterion};
|
||||
use image::imageops;
|
||||
use resize::px::RGBA;
|
||||
use resize::Pixel::RGBA8;
|
||||
use rgb::FromSlice;
|
||||
|
||||
use fast_image_resize::ImageData;
|
||||
use fast_image_resize::{CpuExtensions, FilterType, PixelType, ResizeAlg, Resizer};
|
||||
mod utils;
|
||||
|
||||
pub fn bench_downscale(c: &mut Criterion) {
|
||||
let src_image = utils::get_big_rgb_image();
|
||||
let new_width = NonZeroU32::new(852).unwrap();
|
||||
let new_height = NonZeroU32::new(567).unwrap();
|
||||
|
||||
let mut group = c.benchmark_group("Downscale (4928x3279 => 852x567)");
|
||||
|
||||
for alg_num in 0..4 {
|
||||
// image crate
|
||||
// https://crates.io/crates/image
|
||||
group.bench_with_input(
|
||||
BenchmarkId::new("image", alg_num),
|
||||
&alg_num,
|
||||
|b, alg_num| {
|
||||
b.iter(|| {
|
||||
let filter = match alg_num {
|
||||
0 => imageops::Nearest,
|
||||
1 => imageops::Triangle,
|
||||
2 => imageops::CatmullRom,
|
||||
3 => imageops::Lanczos3,
|
||||
_ => return,
|
||||
};
|
||||
imageops::resize(&src_image, new_width.get(), new_height.get(), filter);
|
||||
})
|
||||
},
|
||||
);
|
||||
|
||||
// resize crate
|
||||
// https://crates.io/crates/resize
|
||||
let resize_src_image = src_image.as_raw().as_rgba();
|
||||
let mut dst = vec![RGBA::new(0, 0, 0, 0); (new_width.get() * new_height.get()) as usize];
|
||||
|
||||
group.bench_with_input(
|
||||
BenchmarkId::new("resize", alg_num),
|
||||
&alg_num,
|
||||
|b, alg_num| {
|
||||
let filter = match alg_num {
|
||||
0 => resize::Type::Point,
|
||||
1 => resize::Type::Triangle,
|
||||
2 => resize::Type::Catrom,
|
||||
3 => resize::Type::Lanczos3,
|
||||
_ => return,
|
||||
};
|
||||
let mut resizer = resize::new(
|
||||
src_image.width() as usize,
|
||||
src_image.height() as usize,
|
||||
new_width.get() as usize,
|
||||
new_height.get() as usize,
|
||||
RGBA8,
|
||||
filter,
|
||||
)
|
||||
.unwrap();
|
||||
b.iter(|| {
|
||||
resizer.resize(resize_src_image, &mut dst).unwrap();
|
||||
})
|
||||
},
|
||||
);
|
||||
|
||||
// fast_image_resize crate
|
||||
let src_image_data = ImageData::new(
|
||||
NonZeroU32::new(src_image.width()).unwrap(),
|
||||
NonZeroU32::new(src_image.height()).unwrap(),
|
||||
src_image.as_raw(),
|
||||
PixelType::U8x4,
|
||||
)
|
||||
.unwrap();
|
||||
let src_view = src_image_data.src_view();
|
||||
let mut dst_image = ImageData::new_owned(new_width, new_height, PixelType::U8x4);
|
||||
let mut dst_view = dst_image.dst_view();
|
||||
let mut fast_resizer = Resizer::new(ResizeAlg::Nearest);
|
||||
unsafe {
|
||||
fast_resizer.set_cpu_extensions(CpuExtensions::None);
|
||||
}
|
||||
group.bench_with_input(
|
||||
BenchmarkId::new("fast_wo_simd", alg_num),
|
||||
&alg_num,
|
||||
|b, alg_num| {
|
||||
let resize_alg = match alg_num {
|
||||
0 => ResizeAlg::Nearest,
|
||||
1 => ResizeAlg::Convolution(FilterType::Bilinear),
|
||||
2 => ResizeAlg::Convolution(FilterType::CatmulRom),
|
||||
3 => ResizeAlg::Convolution(FilterType::Lanczos3),
|
||||
_ => return,
|
||||
};
|
||||
fast_resizer.algorithm = resize_alg;
|
||||
b.iter(|| {
|
||||
fast_resizer.resize(&src_view, &mut dst_view);
|
||||
})
|
||||
},
|
||||
);
|
||||
|
||||
let mut fast_resizer = Resizer::new(ResizeAlg::Nearest);
|
||||
unsafe {
|
||||
fast_resizer.set_cpu_extensions(CpuExtensions::Sse4_1);
|
||||
}
|
||||
group.bench_with_input(
|
||||
BenchmarkId::new("fast_sse4", alg_num),
|
||||
&alg_num,
|
||||
|b, alg_num| {
|
||||
let resize_alg = match alg_num {
|
||||
0 => ResizeAlg::Nearest,
|
||||
1 => ResizeAlg::Convolution(FilterType::Bilinear),
|
||||
2 => ResizeAlg::Convolution(FilterType::CatmulRom),
|
||||
3 => ResizeAlg::Convolution(FilterType::Lanczos3),
|
||||
_ => return,
|
||||
};
|
||||
fast_resizer.algorithm = resize_alg;
|
||||
b.iter(|| {
|
||||
fast_resizer.resize(&src_view, &mut dst_view);
|
||||
})
|
||||
},
|
||||
);
|
||||
|
||||
let mut fast_resizer = Resizer::new(ResizeAlg::Nearest);
|
||||
unsafe {
|
||||
fast_resizer.set_cpu_extensions(CpuExtensions::Avx2);
|
||||
}
|
||||
group.bench_with_input(
|
||||
BenchmarkId::new("fast_avx2", alg_num),
|
||||
&alg_num,
|
||||
|b, alg_num| {
|
||||
let resize_alg = match alg_num {
|
||||
0 => ResizeAlg::Nearest,
|
||||
1 => ResizeAlg::Convolution(FilterType::Bilinear),
|
||||
2 => ResizeAlg::Convolution(FilterType::CatmulRom),
|
||||
3 => ResizeAlg::Convolution(FilterType::Lanczos3),
|
||||
_ => return,
|
||||
};
|
||||
fast_resizer.algorithm = resize_alg;
|
||||
b.iter(|| {
|
||||
fast_resizer.resize(&src_view, &mut dst_view);
|
||||
})
|
||||
},
|
||||
);
|
||||
}
|
||||
|
||||
group.finish();
|
||||
}
|
||||
|
||||
criterion_group!(benches, bench_downscale,);
|
||||
criterion_main!(benches);
|
||||
@@ -0,0 +1,174 @@
|
||||
use std::num::NonZeroU32;
|
||||
|
||||
use criterion::{criterion_group, criterion_main, Criterion};
|
||||
|
||||
use fast_image_resize::ImageData;
|
||||
use fast_image_resize::{CpuExtensions, FilterType, PixelType, ResizeAlg, Resizer};
|
||||
|
||||
mod utils;
|
||||
|
||||
const NEW_WIDTH: u32 = 852;
|
||||
const NEW_HEIGHT: u32 = 567;
|
||||
|
||||
const NEW_BIG_WIDTH: u32 = 4928;
|
||||
const NEW_BIG_HEIGHT: u32 = 3279;
|
||||
|
||||
fn get_big_source_image() -> ImageData<Vec<u8>> {
|
||||
let img = utils::get_big_rgb_image();
|
||||
let width = img.width();
|
||||
let height = img.height();
|
||||
let buf = img.as_raw().clone();
|
||||
ImageData::new(
|
||||
NonZeroU32::new(width).unwrap(),
|
||||
NonZeroU32::new(height).unwrap(),
|
||||
buf,
|
||||
PixelType::U8x4,
|
||||
)
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
fn get_small_source_image() -> ImageData<Vec<u8>> {
|
||||
let img = utils::get_small_rgb_image();
|
||||
let width = img.width();
|
||||
let height = img.height();
|
||||
let buf = img.as_raw().clone();
|
||||
ImageData::new(
|
||||
NonZeroU32::new(width).unwrap(),
|
||||
NonZeroU32::new(height).unwrap(),
|
||||
buf,
|
||||
PixelType::U8x4,
|
||||
)
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
fn downscale_nearest_wo_simd_bench(c: &mut Criterion) {
|
||||
let image = get_big_source_image();
|
||||
let mut res_image = ImageData::new_owned(
|
||||
NonZeroU32::new(NEW_WIDTH).unwrap(),
|
||||
NonZeroU32::new(NEW_HEIGHT).unwrap(),
|
||||
image.pixel_type(),
|
||||
);
|
||||
let src_image = image.src_view();
|
||||
let mut dst_image = res_image.dst_view();
|
||||
let mut resizer = Resizer::new(ResizeAlg::Nearest);
|
||||
unsafe {
|
||||
resizer.set_cpu_extensions(CpuExtensions::None);
|
||||
}
|
||||
c.bench_function("Downscale nearest wo SIMD", |b| {
|
||||
b.iter(|| {
|
||||
resizer.resize(&src_image, &mut dst_image);
|
||||
})
|
||||
});
|
||||
}
|
||||
|
||||
fn downscale_lanczos3_wo_simd_bench(c: &mut Criterion) {
|
||||
let image = get_big_source_image();
|
||||
let mut res_image = ImageData::new_owned(
|
||||
NonZeroU32::new(NEW_WIDTH).unwrap(),
|
||||
NonZeroU32::new(NEW_HEIGHT).unwrap(),
|
||||
image.pixel_type(),
|
||||
);
|
||||
let src_image = image.src_view();
|
||||
let mut dst_image = res_image.dst_view();
|
||||
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
|
||||
unsafe {
|
||||
resizer.set_cpu_extensions(CpuExtensions::None);
|
||||
}
|
||||
c.bench_function("Downscale lanczos3 wo SIMD", |b| {
|
||||
b.iter(|| {
|
||||
resizer.resize(&src_image, &mut dst_image);
|
||||
})
|
||||
});
|
||||
}
|
||||
|
||||
fn sse4_lanczos3_bench(c: &mut Criterion) {
|
||||
let image = get_big_source_image();
|
||||
let mut res_image = ImageData::new_owned(
|
||||
NonZeroU32::new(NEW_WIDTH).unwrap(),
|
||||
NonZeroU32::new(NEW_HEIGHT).unwrap(),
|
||||
image.pixel_type(),
|
||||
);
|
||||
let src_image = image.src_view();
|
||||
let mut dst_image = res_image.dst_view();
|
||||
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
|
||||
unsafe {
|
||||
resizer.set_cpu_extensions(CpuExtensions::Sse4_1);
|
||||
}
|
||||
c.bench_function("sse4 lanczos3", |b| {
|
||||
b.iter(|| {
|
||||
resizer.resize(&src_image, &mut dst_image);
|
||||
})
|
||||
});
|
||||
}
|
||||
|
||||
fn avx2_lanczos3_bench(c: &mut Criterion) {
|
||||
let image = get_big_source_image();
|
||||
let mut res_image = ImageData::new_owned(
|
||||
NonZeroU32::new(NEW_WIDTH).unwrap(),
|
||||
NonZeroU32::new(NEW_HEIGHT).unwrap(),
|
||||
image.pixel_type(),
|
||||
);
|
||||
let src_image = image.src_view();
|
||||
let mut dst_image = res_image.dst_view();
|
||||
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
|
||||
unsafe {
|
||||
resizer.set_cpu_extensions(CpuExtensions::Avx2);
|
||||
}
|
||||
c.bench_function("avx2 lanczos3", |b| {
|
||||
b.iter(|| {
|
||||
resizer.resize(&src_image, &mut dst_image);
|
||||
})
|
||||
});
|
||||
}
|
||||
|
||||
fn avx2_supersampling_lanczos3_bench(c: &mut Criterion) {
|
||||
let image = get_big_source_image();
|
||||
let mut res_image = ImageData::new_owned(
|
||||
NonZeroU32::new(NEW_WIDTH).unwrap(),
|
||||
NonZeroU32::new(NEW_HEIGHT).unwrap(),
|
||||
image.pixel_type(),
|
||||
);
|
||||
let src_image = image.src_view();
|
||||
let mut dst_image = res_image.dst_view();
|
||||
let mut resizer = Resizer::new(ResizeAlg::SuperSampling(FilterType::Lanczos3, 2));
|
||||
unsafe {
|
||||
resizer.set_cpu_extensions(CpuExtensions::Avx2);
|
||||
}
|
||||
c.bench_function("avx2 supersampling lanczos3", |b| {
|
||||
b.iter(|| {
|
||||
resizer.resize(&src_image, &mut dst_image);
|
||||
})
|
||||
});
|
||||
}
|
||||
|
||||
fn avx2_lanczos3_upscale_bench(c: &mut Criterion) {
|
||||
let image = get_small_source_image();
|
||||
let mut res_image = ImageData::new_owned(
|
||||
NonZeroU32::new(NEW_BIG_WIDTH).unwrap(),
|
||||
NonZeroU32::new(NEW_BIG_HEIGHT).unwrap(),
|
||||
image.pixel_type(),
|
||||
);
|
||||
let src_image = image.src_view();
|
||||
let mut dst_image = res_image.dst_view();
|
||||
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
|
||||
unsafe {
|
||||
resizer.set_cpu_extensions(CpuExtensions::Avx2);
|
||||
}
|
||||
c.bench_function("avx2 lanczos3 upscale", |b| {
|
||||
b.iter(|| {
|
||||
resizer.resize(&src_image, &mut dst_image);
|
||||
})
|
||||
});
|
||||
}
|
||||
|
||||
criterion_group!(
|
||||
benches,
|
||||
// downscale_nearest_wo_simd_bench,
|
||||
// downscale_lanczos3_wo_simd_bench,
|
||||
// resize_lanczos3_bench,
|
||||
// sse4_lanczos3_bench,
|
||||
avx2_lanczos3_bench,
|
||||
// avx2_supersampling_lanczos3_bench,
|
||||
// avx2_lanczos3_upscale_bench,
|
||||
);
|
||||
criterion_main!(benches);
|
||||
@@ -0,0 +1,22 @@
|
||||
use std::env;
|
||||
|
||||
use image::io::Reader;
|
||||
use image::RgbaImage;
|
||||
|
||||
pub fn get_big_rgb_image() -> RgbaImage {
|
||||
let cur_dir = env::current_dir().unwrap();
|
||||
let img = Reader::open(cur_dir.join("data/nasa-4928x3279.png"))
|
||||
.unwrap()
|
||||
.decode()
|
||||
.unwrap();
|
||||
img.to_rgba8()
|
||||
}
|
||||
|
||||
pub fn get_small_rgb_image() -> RgbaImage {
|
||||
let cur_dir = env::current_dir().unwrap();
|
||||
let img = Reader::open(cur_dir.join("data/nasa-852x567.png"))
|
||||
.unwrap()
|
||||
.decode()
|
||||
.unwrap();
|
||||
img.to_rgba8()
|
||||
}
|
||||
Binary file not shown.
|
After Width: | Height: | Size: 15 MiB |
Binary file not shown.
|
After Width: | Height: | Size: 13 MiB |
Binary file not shown.
|
After Width: | Height: | Size: 688 KiB |
@@ -0,0 +1,172 @@
|
||||
use std::arch::x86_64::*;
|
||||
|
||||
use crate::simd_utils;
|
||||
use crate::{DstImageView, SrcImageView};
|
||||
|
||||
pub(crate) fn divide_alpha_avx2(src_image: &SrcImageView, dst_image: &mut DstImageView) {
|
||||
let width = src_image.width().get();
|
||||
let src_rows = src_image.iter_rows(0, src_image.height().get());
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
divide_alpha_row_avx2(src_row, dst_row, width as usize);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn divide_alpha_inplace_avx2(image: &mut DstImageView) {
|
||||
let width = image.width().get() as usize;
|
||||
for dst_row in image.iter_rows_mut() {
|
||||
unsafe {
|
||||
let src_row = std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len());
|
||||
divide_alpha_row_avx2(src_row, dst_row, width);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn divide_alpha_sse2(src_image: &SrcImageView, dst_image: &mut DstImageView) {
|
||||
let width = src_image.width().get() as usize;
|
||||
let src_rows = src_image.iter_rows(0, src_image.height().get());
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
divide_alpha_row_sse2(src_row, dst_row, width);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn divide_alpha_inplace_sse2(image: &mut DstImageView) {
|
||||
let width = image.width().get() as usize;
|
||||
for dst_row in image.iter_rows_mut() {
|
||||
unsafe {
|
||||
let src_row = std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len());
|
||||
divide_alpha_row_sse2(src_row, dst_row, width);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn divide_alpha_native(src_image: &SrcImageView, dst_image: &mut DstImageView) {
|
||||
let src_rows = src_image.iter_rows(0, src_image.height().get());
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
divide_alpha_row_native(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn divide_alpha_inplace_native(image: &mut DstImageView) {
|
||||
for dst_row in image.iter_rows_mut() {
|
||||
let src_row = unsafe { std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len()) };
|
||||
divide_alpha_row_native(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "avx2")]
|
||||
unsafe fn divide_alpha_row_avx2(src_row: &[u32], dst_row: &mut [u32], width: usize) {
|
||||
let mut x: usize = 0;
|
||||
let zero = _mm256_setzero_si256();
|
||||
#[rustfmt::skip]
|
||||
let alpha_mask = _mm256_set1_epi32(0xff000000u32 as i32);
|
||||
#[rustfmt::skip]
|
||||
let shuffle1 = _mm256_set_epi8(
|
||||
5, 4, 5, 4, 5, 4, 5, 4, 1, 0, 1, 0, 1, 0, 1, 0,
|
||||
5, 4, 5, 4, 5, 4, 5, 4, 1, 0, 1, 0, 1, 0, 1, 0,
|
||||
);
|
||||
#[rustfmt::skip]
|
||||
let shuffle2 = _mm256_set_epi8(
|
||||
13, 12, 13, 12, 13, 12, 13, 12, 9, 8, 9, 8, 9, 8, 9, 8,
|
||||
13, 12, 13, 12, 13, 12, 13, 12, 9, 8, 9, 8, 9, 8, 9, 8,
|
||||
);
|
||||
let alpha_scale = _mm256_set1_ps(255.0 * 256.0);
|
||||
|
||||
while x < width.saturating_sub(7) {
|
||||
let mut source = simd_utils::loadu_si256(src_row, x);
|
||||
|
||||
let alpha_f32 = _mm256_cvtepi32_ps(_mm256_srli_epi32(source, 24));
|
||||
let scaled_alpha_f32 = _mm256_mul_ps(alpha_scale, _mm256_rcp_ps(alpha_f32));
|
||||
let scaled_alpha_i32 = _mm256_cvtps_epi32(scaled_alpha_f32);
|
||||
let mma0 = _mm256_shuffle_epi8(scaled_alpha_i32, shuffle1);
|
||||
let mma1 = _mm256_shuffle_epi8(scaled_alpha_i32, shuffle2);
|
||||
|
||||
let mut pix0 = _mm256_unpacklo_epi8(zero, source);
|
||||
let mut pix1 = _mm256_unpackhi_epi8(zero, source);
|
||||
|
||||
pix0 = _mm256_mulhi_epu16(pix0, mma0);
|
||||
pix1 = _mm256_mulhi_epu16(pix1, mma1);
|
||||
|
||||
let alpha = _mm256_and_si256(source, alpha_mask);
|
||||
source = _mm256_packus_epi16(pix0, pix1);
|
||||
source = _mm256_blendv_epi8(source, alpha, alpha_mask);
|
||||
|
||||
let dst_ptr = dst_row.get_unchecked_mut(x..).as_mut_ptr() as *mut __m256i;
|
||||
_mm256_storeu_si256(dst_ptr, source);
|
||||
|
||||
x += 8;
|
||||
}
|
||||
|
||||
let src_tail = &src_row[x..];
|
||||
let dst_tail = &mut dst_row[x..];
|
||||
divide_alpha_row_native(src_tail, dst_tail);
|
||||
}
|
||||
|
||||
#[target_feature(enable = "sse2")]
|
||||
unsafe fn divide_alpha_row_sse2(src_row: &[u32], dst_row: &mut [u32], width: usize) {
|
||||
let zero = _mm_setzero_si128();
|
||||
let alpha_mask = _mm_set1_epi32(0xff000000u32 as i32);
|
||||
let shuffle0 = _mm_set_epi8(5, 4, 5, 4, 5, 4, 5, 4, 1, 0, 1, 0, 1, 0, 1, 0);
|
||||
let shuffle1 = _mm_set_epi8(13, 12, 13, 12, 13, 12, 13, 12, 9, 8, 9, 8, 9, 8, 9, 8);
|
||||
let alpha_scale = _mm_set1_ps(255.0 * 256.0);
|
||||
|
||||
let mut x: usize = 0;
|
||||
while x < width.saturating_sub(3) {
|
||||
let mut source = simd_utils::loadu_si128(src_row, x);
|
||||
|
||||
let alpha = _mm_and_si128(source, alpha_mask);
|
||||
let alpha_f32 = _mm_cvtepi32_ps(_mm_srli_epi32(source, 24));
|
||||
let scaled_recip_alpha_f32 = _mm_mul_ps(alpha_scale, _mm_rcp_ps(alpha_f32));
|
||||
let scaled_recip_alpha_i32 = _mm_cvtps_epi32(scaled_recip_alpha_f32);
|
||||
let mma0 = _mm_shuffle_epi8(scaled_recip_alpha_i32, shuffle0);
|
||||
let mma1 = _mm_shuffle_epi8(scaled_recip_alpha_i32, shuffle1);
|
||||
|
||||
let mut pix0 = _mm_unpacklo_epi8(zero, source);
|
||||
let mut pix1 = _mm_unpackhi_epi8(zero, source);
|
||||
|
||||
pix0 = _mm_mulhi_epu16(pix0, mma0);
|
||||
pix1 = _mm_mulhi_epu16(pix1, mma1);
|
||||
|
||||
source = _mm_packus_epi16(pix0, pix1);
|
||||
source = _mm_blendv_epi8(source, alpha, alpha_mask);
|
||||
let dst_ptr = dst_row.get_unchecked_mut(x..).as_mut_ptr() as *mut __m128i;
|
||||
_mm_storeu_si128(dst_ptr, source);
|
||||
|
||||
x += 4;
|
||||
}
|
||||
|
||||
let src_tail = &src_row[x..];
|
||||
let dst_tail = &mut dst_row[x..];
|
||||
divide_alpha_row_native(src_tail, dst_tail);
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
fn div_and_clip(v: u8, rev_alpha: f32) -> u8 {
|
||||
let res = v as f32 * rev_alpha;
|
||||
res.min(255.) as u8
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
fn divide_alpha_row_native(src_row: &[u32], dst_row: &mut [u32]) {
|
||||
for (src_pixel, dst_pixel) in src_row.iter().zip(dst_row) {
|
||||
let components: [u8; 4] = src_pixel.to_le_bytes();
|
||||
let alpha = components[3];
|
||||
let recip_alpha = if alpha == 0 { 0. } else { 255. / alpha as f32 };
|
||||
let res = [
|
||||
div_and_clip(components[0], recip_alpha),
|
||||
div_and_clip(components[1], recip_alpha),
|
||||
div_and_clip(components[2], recip_alpha),
|
||||
alpha,
|
||||
];
|
||||
*dst_pixel = u32::from_le_bytes(res);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,128 @@
|
||||
use thiserror::Error;
|
||||
|
||||
use crate::{CpuExtensions, PixelType};
|
||||
use crate::{DstImageView, SrcImageView};
|
||||
|
||||
mod div;
|
||||
mod mul;
|
||||
|
||||
#[derive(Error, Debug, Clone, Copy)]
|
||||
#[non_exhaustive]
|
||||
pub enum MulDivImagesError {
|
||||
#[error("Size of source image does not match to destination image")]
|
||||
SizeIsDifferent,
|
||||
#[error("Pixel type of source image does not match to destination image")]
|
||||
PixelTypeIsDifferent,
|
||||
#[error("Pixel type of image is not supported")]
|
||||
UnsupportedPixelType,
|
||||
}
|
||||
|
||||
#[derive(Error, Debug, Clone, Copy)]
|
||||
#[non_exhaustive]
|
||||
pub enum MulDivImageError {
|
||||
#[error("Pixel type of image is not supported")]
|
||||
UnsupportedPixelType,
|
||||
}
|
||||
|
||||
/// Methods of this structure used to multiplies or divides RGB-channels
|
||||
/// by alpha-channel.
|
||||
#[derive(Default, Debug, Clone)]
|
||||
pub struct MulDiv {
|
||||
cpu_extensions: CpuExtensions,
|
||||
}
|
||||
|
||||
impl MulDiv {
|
||||
#[inline(always)]
|
||||
pub fn cpu_extensions(&self) -> CpuExtensions {
|
||||
self.cpu_extensions
|
||||
}
|
||||
|
||||
/// # Safety
|
||||
/// This is unsafe because this method allows you to set a CPU-extensions
|
||||
/// that is not actually supported by your CPU.
|
||||
pub unsafe fn set_cpu_extensions(&mut self, extensions: CpuExtensions) {
|
||||
self.cpu_extensions = extensions;
|
||||
}
|
||||
|
||||
/// Multiplies RGB-channels of source image by alpha-channel and store
|
||||
/// result into destination image.
|
||||
pub fn multiply_alpha(
|
||||
&self,
|
||||
src_image: &SrcImageView,
|
||||
dst_image: &mut DstImageView,
|
||||
) -> Result<(), MulDivImagesError> {
|
||||
self.assert_images(src_image, dst_image)?;
|
||||
match self.cpu_extensions {
|
||||
CpuExtensions::Avx2 => mul::multiply_alpha_avx2(src_image, dst_image),
|
||||
// CpuExtensions::Sse2 => mul::multiply_alpha_sse2(src_image, dst_image),
|
||||
_ => mul::multiply_alpha_native(src_image, dst_image),
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Multiplies RGB-channels of image by alpha-channel inplace.
|
||||
pub fn multiply_alpha_inplace(&self, image: &mut DstImageView) -> Result<(), MulDivImageError> {
|
||||
self.assert_image(image)?;
|
||||
match self.cpu_extensions {
|
||||
CpuExtensions::Avx2 => mul::multiply_alpha_inplace_avx2(image),
|
||||
// CpuExtensions::Sse2 => mul::multiply_alpha_sse2(src_image, dst_image),
|
||||
_ => mul::multiply_alpha_inplace_native(image),
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Divides RGB-channels of source image by alpha-channel and store
|
||||
/// result into destination image.
|
||||
pub fn divide_alpha(
|
||||
&self,
|
||||
src_image: &SrcImageView,
|
||||
dst_image: &mut DstImageView,
|
||||
) -> Result<(), MulDivImagesError> {
|
||||
self.assert_images(src_image, dst_image)?;
|
||||
match self.cpu_extensions {
|
||||
CpuExtensions::Avx2 => div::divide_alpha_avx2(src_image, dst_image),
|
||||
CpuExtensions::Sse4_1 | CpuExtensions::Sse2 => {
|
||||
div::divide_alpha_sse2(src_image, dst_image)
|
||||
}
|
||||
_ => div::divide_alpha_native(src_image, dst_image),
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Divides RGB-channels of image by alpha-channel inplace.
|
||||
pub fn divide_alpha_inplace(&self, image: &mut DstImageView) -> Result<(), MulDivImageError> {
|
||||
self.assert_image(image)?;
|
||||
match self.cpu_extensions {
|
||||
CpuExtensions::Avx2 => div::divide_alpha_inplace_avx2(image),
|
||||
CpuExtensions::Sse4_1 | CpuExtensions::Sse2 => div::divide_alpha_inplace_sse2(image),
|
||||
_ => div::divide_alpha_inplace_native(image),
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[inline]
|
||||
fn assert_images(
|
||||
&self,
|
||||
src_image: &SrcImageView,
|
||||
dst_image: &DstImageView,
|
||||
) -> Result<(), MulDivImagesError> {
|
||||
if src_image.width() != dst_image.width() || src_image.height() != dst_image.height() {
|
||||
return Err(MulDivImagesError::SizeIsDifferent);
|
||||
}
|
||||
if src_image.pixel_type() != PixelType::U8x4 {
|
||||
return Err(MulDivImagesError::UnsupportedPixelType);
|
||||
}
|
||||
if src_image.pixel_type() != dst_image.pixel_type() {
|
||||
return Err(MulDivImagesError::PixelTypeIsDifferent);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[inline]
|
||||
fn assert_image(&self, image: &DstImageView) -> Result<(), MulDivImageError> {
|
||||
if image.pixel_type() != PixelType::U8x4 {
|
||||
return Err(MulDivImageError::UnsupportedPixelType);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,161 @@
|
||||
use std::arch::x86_64::*;
|
||||
|
||||
use crate::simd_utils;
|
||||
use crate::{DstImageView, SrcImageView};
|
||||
|
||||
pub(crate) fn multiply_alpha_avx2(src_image: &SrcImageView, dst_image: &mut DstImageView) {
|
||||
let width = src_image.width().get() as usize;
|
||||
let src_rows = src_image.iter_rows(0, src_image.height().get());
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
multiply_alpha_row_avx2(src_row, dst_row, width);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn multiply_alpha_inplace_avx2(image: &mut DstImageView) {
|
||||
let width = image.width().get() as usize;
|
||||
for dst_row in image.iter_rows_mut() {
|
||||
unsafe {
|
||||
let src_row = std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len());
|
||||
multiply_alpha_row_avx2(src_row, dst_row, width);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[allow(dead_code)]
|
||||
pub(crate) fn multiply_alpha_sse2(src_image: &SrcImageView, dst_image: &mut DstImageView) {
|
||||
let width = src_image.width().get() as usize;
|
||||
let src_rows = src_image.iter_rows(0, src_image.height().get());
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
multiply_alpha_row_sse2(src_row, dst_row, width);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn multiply_alpha_native(src_image: &SrcImageView, dst_image: &mut DstImageView) {
|
||||
let src_rows = src_image.iter_rows(0, src_image.height().get());
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
multiply_alpha_row_native(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn multiply_alpha_inplace_native(image: &mut DstImageView) {
|
||||
for dst_row in image.iter_rows_mut() {
|
||||
let src_row = unsafe { std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len()) };
|
||||
multiply_alpha_row_native(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
/// https://github.com/Wizermil/premultiply_alpha/blob/master/premultiply_alpha/premultiply_alpha.hpp#L232
|
||||
#[target_feature(enable = "avx2")]
|
||||
unsafe fn multiply_alpha_row_avx2(src_row: &[u32], dst_row: &mut [u32], width: usize) {
|
||||
let mask_alpha_color_odd_255 = _mm256_set1_epi32(0xff000000u32 as i32);
|
||||
let div_255 = _mm256_set1_epi16(0x8081u16 as i16);
|
||||
#[rustfmt::skip]
|
||||
let mask_shuffle_alpha = _mm256_set_epi8(
|
||||
15, -1, 15, -1, 11, -1, 11, -1, 7, -1, 7, -1, 3, -1, 3, -1,
|
||||
15, -1, 15, -1, 11, -1, 11, -1, 7, -1, 7, -1, 3, -1, 3, -1,
|
||||
);
|
||||
#[rustfmt::skip]
|
||||
let mask_shuffle_color_odd = _mm256_set_epi8(
|
||||
-1, -1, 13, -1, -1, -1, 9, -1, -1, -1, 5, -1, -1, -1, 1, -1,
|
||||
-1, -1, 13, -1, -1, -1, 9, -1, -1, -1, 5, -1, -1, -1, 1, -1,
|
||||
);
|
||||
|
||||
let mut x: usize = 0;
|
||||
while x < width.saturating_sub(7) {
|
||||
let mut color = simd_utils::loadu_si256(src_row, x);
|
||||
|
||||
let alpha = _mm256_shuffle_epi8(color, mask_shuffle_alpha);
|
||||
let mut color_even = _mm256_slli_epi16(color, 8);
|
||||
let mut color_odd = _mm256_shuffle_epi8(color, mask_shuffle_color_odd);
|
||||
color_odd = _mm256_or_si256(color_odd, mask_alpha_color_odd_255);
|
||||
|
||||
color_odd = _mm256_mulhi_epu16(color_odd, alpha);
|
||||
color_even = _mm256_mulhi_epu16(color_even, alpha);
|
||||
|
||||
color_odd = _mm256_srli_epi16(_mm256_mulhi_epu16(color_odd, div_255), 7);
|
||||
color_even = _mm256_srli_epi16(_mm256_mulhi_epu16(color_even, div_255), 7);
|
||||
|
||||
color = _mm256_or_si256(color_even, _mm256_slli_epi16(color_odd, 8));
|
||||
|
||||
let dst_ptr = dst_row.get_unchecked_mut(x..).as_mut_ptr() as *mut __m256i;
|
||||
_mm256_storeu_si256(dst_ptr, color);
|
||||
|
||||
x += 8;
|
||||
}
|
||||
|
||||
let src_tail = &src_row[x..];
|
||||
let dst_tail = &mut dst_row[x..];
|
||||
multiply_alpha_row_native(src_tail, dst_tail);
|
||||
}
|
||||
|
||||
/// https://github.com/Wizermil/premultiply_alpha/blob/master/premultiply_alpha/premultiply_alpha.hpp#L108
|
||||
/// This implementation is twice slowly than native version.
|
||||
#[allow(dead_code)]
|
||||
#[target_feature(enable = "sse2")]
|
||||
unsafe fn multiply_alpha_row_sse2(src_row: &[u32], dst_row: &mut [u32], width: usize) {
|
||||
let mask_alpha_color_odd_255 = _mm_set1_epi32(0xff000000u32 as i32);
|
||||
let div_255 = _mm_set1_epi16(0x8081u16 as i16);
|
||||
|
||||
let mask_shuffle_alpha =
|
||||
_mm_set_epi8(15, -1, 15, -1, 11, -1, 11, -1, 7, -1, 7, -1, 3, -1, 3, -1);
|
||||
let mask_shuffle_color_odd =
|
||||
_mm_set_epi8(-1, -1, 13, -1, -1, -1, 9, -1, -1, -1, 5, -1, -1, -1, 1, -1);
|
||||
|
||||
let mut x: usize = 0;
|
||||
while x < width.saturating_sub(3) {
|
||||
let mut color = simd_utils::loadu_si128(src_row, x);
|
||||
let alpha = _mm_shuffle_epi8(color, mask_shuffle_alpha);
|
||||
|
||||
let mut color_even = _mm_slli_epi16(color, 8);
|
||||
let mut color_odd = _mm_shuffle_epi8(color, mask_shuffle_color_odd);
|
||||
color_odd = _mm_or_si128(color_odd, mask_alpha_color_odd_255);
|
||||
|
||||
color_odd = _mm_mulhi_epu16(color_odd, alpha);
|
||||
color_even = _mm_mulhi_epu16(color_even, alpha);
|
||||
|
||||
color_odd = _mm_srli_epi16(_mm_mulhi_epu16(color_odd, div_255), 7);
|
||||
color_even = _mm_srli_epi16(_mm_mulhi_epu16(color_even, div_255), 7);
|
||||
|
||||
color = _mm_or_si128(color_even, _mm_slli_epi16(color_odd, 8));
|
||||
|
||||
let dst_ptr = dst_row.get_unchecked_mut(x..).as_mut_ptr() as *mut __m128i;
|
||||
_mm_storeu_si128(dst_ptr, color);
|
||||
|
||||
x += 4;
|
||||
}
|
||||
|
||||
let src_tail = &src_row[x..];
|
||||
let dst_tail = &mut dst_row[x..];
|
||||
multiply_alpha_row_native(src_tail, dst_tail);
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
fn multiply_alpha_row_native(src_row: &[u32], dst_row: &mut [u32]) {
|
||||
for (src_pixel, dst_pixel) in src_row.iter().zip(dst_row) {
|
||||
let components: [u8; 4] = src_pixel.to_le_bytes();
|
||||
let alpha = components[3];
|
||||
let res: [u8; 4] = [
|
||||
mul_div_255(components[0], alpha),
|
||||
mul_div_255(components[1], alpha),
|
||||
mul_div_255(components[2], alpha),
|
||||
alpha,
|
||||
];
|
||||
*dst_pixel = u32::from_le_bytes(res);
|
||||
}
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
fn mul_div_255(a: u8, b: u8) -> u8 {
|
||||
let tmp = a as u32 * b as u32 + 128;
|
||||
(((tmp >> 8) + tmp) >> 8) as u8
|
||||
}
|
||||
@@ -0,0 +1,527 @@
|
||||
use std::arch::x86_64::*;
|
||||
use std::intrinsics::transmute;
|
||||
|
||||
use crate::convolution::{Bound, Coefficients, Convolution};
|
||||
use crate::image_view::{DstImageView, FourRows, FourRowsMut, SrcImageView};
|
||||
use crate::{optimisations, simd_utils};
|
||||
|
||||
pub struct Avx2;
|
||||
|
||||
// This code is based on C-implementation from Pillow-SIMD package for Python
|
||||
// https://github.com/uploadcare/pillow-simd
|
||||
impl Avx2 {
|
||||
/// For safety, it is necessary to ensure the following conditions:
|
||||
/// - length of all rows in src_rows must be equal
|
||||
/// - length of all rows in dst_rows must be equal
|
||||
/// - bounds.len() == dst_rows.0.len()
|
||||
/// - coeffs.len() == dst_rows.0.len() * window_size
|
||||
/// - max(bound.size for bound in bounds) <= window_size
|
||||
/// - max(bound.start + bound.size for bound in bounds) <= src_row.0.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[inline]
|
||||
#[target_feature(enable = "avx2")]
|
||||
unsafe fn horiz_convolution_8u4x(
|
||||
&self,
|
||||
src_rows: FourRows,
|
||||
dst_rows: FourRowsMut,
|
||||
coeffs: &[i16],
|
||||
window_size: usize,
|
||||
bounds: &[Bound],
|
||||
precision: u8,
|
||||
) {
|
||||
let (s_row0, s_row1, s_row2, s_row3) = src_rows;
|
||||
let (d_row0, d_row1, d_row2, d_row3) = dst_rows;
|
||||
let zero = _mm256_setzero_si256();
|
||||
let initial = _mm256_set1_epi32(1 << (precision - 1));
|
||||
|
||||
#[rustfmt::skip]
|
||||
let sh1 = _mm256_set_epi8(
|
||||
-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0,
|
||||
-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0,
|
||||
);
|
||||
#[rustfmt::skip]
|
||||
let sh2 = _mm256_set_epi8(
|
||||
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
|
||||
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
|
||||
);
|
||||
|
||||
let coeffs_chunks = coeffs.chunks(window_size);
|
||||
for (dst_x, (&bound, k)) in bounds.iter().zip(coeffs_chunks).enumerate() {
|
||||
let x_start = bound.start as usize;
|
||||
let x_size = bound.size as usize;
|
||||
let mut x: usize = 0;
|
||||
|
||||
let mut sss0 = initial;
|
||||
let mut sss1 = initial;
|
||||
|
||||
while x < x_size.saturating_sub(3) {
|
||||
let mmk0 = simd_utils::ptr_i16_to_256set1_epi32(k, x);
|
||||
let mmk1 = simd_utils::ptr_i16_to_256set1_epi32(k, x + 2);
|
||||
|
||||
let mut source = _mm256_inserti128_si256(
|
||||
_mm256_castsi128_si256(simd_utils::loadu_si128(s_row0, x + x_start)),
|
||||
simd_utils::loadu_si128(s_row1, x + x_start),
|
||||
1,
|
||||
);
|
||||
let mut pix = _mm256_shuffle_epi8(source, sh1);
|
||||
sss0 = _mm256_add_epi32(sss0, _mm256_madd_epi16(pix, mmk0));
|
||||
pix = _mm256_shuffle_epi8(source, sh2);
|
||||
sss0 = _mm256_add_epi32(sss0, _mm256_madd_epi16(pix, mmk1));
|
||||
|
||||
source = _mm256_inserti128_si256(
|
||||
_mm256_castsi128_si256(simd_utils::loadu_si128(s_row2, x + x_start)),
|
||||
simd_utils::loadu_si128(s_row3, x + x_start),
|
||||
1,
|
||||
);
|
||||
pix = _mm256_shuffle_epi8(source, sh1);
|
||||
sss1 = _mm256_add_epi32(sss1, _mm256_madd_epi16(pix, mmk0));
|
||||
pix = _mm256_shuffle_epi8(source, sh2);
|
||||
sss1 = _mm256_add_epi32(sss1, _mm256_madd_epi16(pix, mmk1));
|
||||
|
||||
x += 4;
|
||||
}
|
||||
|
||||
while x < x_size.saturating_sub(1) {
|
||||
let mmk = simd_utils::ptr_i16_to_256set1_epi32(k, x);
|
||||
|
||||
let mut pix = _mm256_inserti128_si256(
|
||||
_mm256_castsi128_si256(simd_utils::loadl_epi64(s_row0, x + x_start)),
|
||||
simd_utils::loadl_epi64(s_row1, x + x_start),
|
||||
1,
|
||||
);
|
||||
pix = _mm256_shuffle_epi8(pix, sh1);
|
||||
sss0 = _mm256_add_epi32(sss0, _mm256_madd_epi16(pix, mmk));
|
||||
|
||||
pix = _mm256_inserti128_si256(
|
||||
_mm256_castsi128_si256(simd_utils::loadl_epi64(s_row2, x + x_start)),
|
||||
simd_utils::loadl_epi64(s_row3, x + x_start),
|
||||
1,
|
||||
);
|
||||
pix = _mm256_shuffle_epi8(pix, sh1);
|
||||
sss1 = _mm256_add_epi32(sss1, _mm256_madd_epi16(pix, mmk));
|
||||
|
||||
x += 2;
|
||||
}
|
||||
|
||||
while x < x_size {
|
||||
// [16] xx k0 xx k0 xx k0 xx k0 xx k0 xx k0 xx k0 xx k0
|
||||
let mmk = _mm256_set1_epi32(*k.get_unchecked(x) as i32);
|
||||
|
||||
// [16] xx a0 xx b0 xx g0 xx r0 xx a0 xx b0 xx g0 xx r0
|
||||
let mut pix = _mm256_inserti128_si256(
|
||||
_mm256_castsi128_si256(simd_utils::mm_cvtepu8_epi32(s_row0, x + x_start)),
|
||||
simd_utils::mm_cvtepu8_epi32(s_row1, x + x_start),
|
||||
1,
|
||||
);
|
||||
sss0 = _mm256_add_epi32(sss0, _mm256_madd_epi16(pix, mmk));
|
||||
|
||||
pix = _mm256_inserti128_si256(
|
||||
_mm256_castsi128_si256(simd_utils::mm_cvtepu8_epi32(s_row2, x + x_start)),
|
||||
simd_utils::mm_cvtepu8_epi32(s_row3, x + x_start),
|
||||
1,
|
||||
);
|
||||
sss1 = _mm256_add_epi32(sss1, _mm256_madd_epi16(pix, mmk));
|
||||
x += 1;
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss0 = _mm256_srai_epi32(sss0, $imm8);
|
||||
sss1 = _mm256_srai_epi32(sss1, $imm8);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
|
||||
sss0 = _mm256_packs_epi32(sss0, zero);
|
||||
sss1 = _mm256_packs_epi32(sss1, zero);
|
||||
sss0 = _mm256_packus_epi16(sss0, zero);
|
||||
sss1 = _mm256_packus_epi16(sss1, zero);
|
||||
*d_row0.get_unchecked_mut(dst_x) =
|
||||
transmute(_mm_cvtsi128_si32(_mm256_extracti128_si256(sss0, 0)));
|
||||
*d_row1.get_unchecked_mut(dst_x) =
|
||||
transmute(_mm_cvtsi128_si32(_mm256_extracti128_si256(sss0, 1)));
|
||||
*d_row2.get_unchecked_mut(dst_x) =
|
||||
transmute(_mm_cvtsi128_si32(_mm256_extracti128_si256(sss1, 0)));
|
||||
*d_row3.get_unchecked_mut(dst_x) =
|
||||
transmute(_mm_cvtsi128_si32(_mm256_extracti128_si256(sss1, 1)));
|
||||
}
|
||||
}
|
||||
|
||||
/// For safety, it is necessary to ensure the following conditions:
|
||||
/// - bounds.len() == dst_row.len()
|
||||
/// - coeffs.len() == dst_rows.0.len() * window_size
|
||||
/// - max(bound.start + bound.size for bound in bounds) <= src_row.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[inline]
|
||||
#[target_feature(enable = "avx2")]
|
||||
unsafe fn horiz_convolution_8u(
|
||||
&self,
|
||||
src_row: &[u32],
|
||||
dst_row: &mut [u32],
|
||||
coeffs: &[i16],
|
||||
window_size: usize,
|
||||
bounds: &[Bound],
|
||||
precision: u8,
|
||||
) {
|
||||
#[rustfmt::skip]
|
||||
let sh1 = _mm256_set_epi8(
|
||||
-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0,
|
||||
-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0,
|
||||
);
|
||||
#[rustfmt::skip]
|
||||
let sh2 = _mm256_set_epi8(
|
||||
11, 10, 9, 8, 11, 10, 9, 8, 11, 10, 9, 8, 11, 10, 9, 8,
|
||||
3, 2, 1, 0, 3, 2, 1, 0, 3, 2, 1, 0, 3, 2, 1, 0,
|
||||
);
|
||||
#[rustfmt::skip]
|
||||
let sh3 = _mm256_set_epi8(
|
||||
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
|
||||
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
|
||||
);
|
||||
#[rustfmt::skip]
|
||||
let sh4 = _mm256_set_epi8(
|
||||
15, 14, 13, 12, 15, 14, 13, 12, 15, 14, 13, 12, 15, 14, 13, 12,
|
||||
7, 6, 5, 4, 7, 6, 5, 4, 7, 6, 5, 4, 7, 6, 5, 4,
|
||||
);
|
||||
#[rustfmt::skip]
|
||||
let sh5 = _mm256_set_epi8(
|
||||
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
|
||||
-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0,
|
||||
);
|
||||
#[rustfmt::skip]
|
||||
let sh6 = _mm256_set_epi8(
|
||||
7, 6, 5, 4, 7, 6, 5, 4, 7, 6, 5, 4, 7, 6, 5, 4,
|
||||
3, 2, 1, 0, 3, 2, 1, 0, 3, 2, 1, 0, 3, 2, 1, 0,
|
||||
);
|
||||
let sh7 = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
|
||||
|
||||
let coeffs_chunks = coeffs.chunks(window_size);
|
||||
for (xx, (&bound, k)) in bounds.iter().zip(coeffs_chunks).enumerate() {
|
||||
let x_start = bound.start as usize;
|
||||
let x_size = bound.size as usize;
|
||||
let mut x: usize = 0;
|
||||
|
||||
let mut sss: __m128i = if x_size < 8 {
|
||||
_mm_set1_epi32(1 << (precision - 1))
|
||||
} else {
|
||||
// Lower part will be added to higher, use only half of the error
|
||||
let mut sss256 = _mm256_set1_epi32(1 << (precision - 2));
|
||||
|
||||
while x < x_size.saturating_sub(7) {
|
||||
let tmp = simd_utils::loadu_si128(k, x);
|
||||
let ksource = _mm256_insertf128_si256(_mm256_castsi128_si256(tmp), tmp, 1);
|
||||
|
||||
let source = simd_utils::loadu_si256(src_row, x + x_start);
|
||||
|
||||
let mut pix = _mm256_shuffle_epi8(source, sh1);
|
||||
let mut mmk = _mm256_shuffle_epi8(ksource, sh2);
|
||||
sss256 = _mm256_add_epi32(sss256, _mm256_madd_epi16(pix, mmk));
|
||||
|
||||
pix = _mm256_shuffle_epi8(source, sh3);
|
||||
mmk = _mm256_shuffle_epi8(ksource, sh4);
|
||||
sss256 = _mm256_add_epi32(sss256, _mm256_madd_epi16(pix, mmk));
|
||||
|
||||
x += 8;
|
||||
}
|
||||
|
||||
while x < x_size.saturating_sub(3) {
|
||||
let tmp = simd_utils::loadl_epi64(k, x);
|
||||
let ksource = _mm256_insertf128_si256(_mm256_castsi128_si256(tmp), tmp, 1);
|
||||
|
||||
let tmp = simd_utils::loadu_si128(src_row, x + x_start);
|
||||
let source = _mm256_insertf128_si256(_mm256_castsi128_si256(tmp), tmp, 1);
|
||||
|
||||
let pix = _mm256_shuffle_epi8(source, sh5);
|
||||
let mmk = _mm256_shuffle_epi8(ksource, sh6);
|
||||
sss256 = _mm256_add_epi32(sss256, _mm256_madd_epi16(pix, mmk));
|
||||
|
||||
x += 4;
|
||||
}
|
||||
|
||||
_mm_add_epi32(
|
||||
_mm256_extracti128_si256(sss256, 0),
|
||||
_mm256_extracti128_si256(sss256, 1),
|
||||
)
|
||||
};
|
||||
|
||||
while x < x_size.saturating_sub(1) {
|
||||
let mmk = simd_utils::ptr_i16_to_set1_epi32(k, x);
|
||||
let source = simd_utils::loadl_epi64(src_row, x + x_start);
|
||||
let pix = _mm_shuffle_epi8(source, sh7);
|
||||
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
x += 2
|
||||
}
|
||||
|
||||
while x < x_size {
|
||||
let pix = simd_utils::mm_cvtepu8_epi32(src_row, x + x_start);
|
||||
let mmk = _mm_set1_epi32(*k.get_unchecked(x) as i32);
|
||||
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
x += 1;
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss = _mm_srai_epi32(sss, $imm8);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
|
||||
sss = _mm_packs_epi32(sss, sss);
|
||||
*dst_row.get_unchecked_mut(xx) =
|
||||
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss, sss)));
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "avx2")]
|
||||
pub unsafe fn vert_convolution_8u(
|
||||
&self,
|
||||
src_img: &SrcImageView,
|
||||
dst_row: &mut [u32],
|
||||
coeffs: &[i16],
|
||||
bound: Bound,
|
||||
precision: u8,
|
||||
) {
|
||||
let src_width = src_img.width().get() as usize;
|
||||
let y_start = bound.start;
|
||||
let y_size = bound.size;
|
||||
|
||||
let initial = _mm_set1_epi32(1 << (precision - 1));
|
||||
let initial_256 = _mm256_set1_epi32(1 << (precision - 1));
|
||||
|
||||
let mut x: usize = 0;
|
||||
while x < src_width.saturating_sub(7) {
|
||||
let mut sss0 = initial_256;
|
||||
let mut sss1 = initial_256;
|
||||
let mut sss2 = initial_256;
|
||||
let mut sss3 = initial_256;
|
||||
|
||||
let mut y: u32 = 0;
|
||||
|
||||
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
|
||||
// Load two coefficients at once
|
||||
let mmk = simd_utils::ptr_i16_to_256set1_epi32(coeffs, y as usize);
|
||||
|
||||
let source1 = simd_utils::loadu_si256(s_row1, x); // top line
|
||||
let source2 = simd_utils::loadu_si256(s_row2, x); // bottom line
|
||||
|
||||
let mut source = _mm256_unpacklo_epi8(source1, source2);
|
||||
let mut pix = _mm256_unpacklo_epi8(source, _mm256_setzero_si256());
|
||||
sss0 = _mm256_add_epi32(sss0, _mm256_madd_epi16(pix, mmk));
|
||||
pix = _mm256_unpackhi_epi8(source, _mm256_setzero_si256());
|
||||
sss1 = _mm256_add_epi32(sss1, _mm256_madd_epi16(pix, mmk));
|
||||
|
||||
source = _mm256_unpackhi_epi8(source1, source2);
|
||||
pix = _mm256_unpacklo_epi8(source, _mm256_setzero_si256());
|
||||
sss2 = _mm256_add_epi32(sss2, _mm256_madd_epi16(pix, mmk));
|
||||
pix = _mm256_unpackhi_epi8(source, _mm256_setzero_si256());
|
||||
sss3 = _mm256_add_epi32(sss3, _mm256_madd_epi16(pix, mmk));
|
||||
|
||||
y += 2;
|
||||
}
|
||||
|
||||
for s_row in src_img.iter_rows(y_start + y, y_start + y_size) {
|
||||
let mmk = _mm256_set1_epi32(coeffs[y as usize] as i32);
|
||||
|
||||
let source1 = simd_utils::loadu_si256(s_row, x); // top line
|
||||
let source2 = _mm256_setzero_si256(); // bottom line is empty
|
||||
|
||||
let mut source = _mm256_unpacklo_epi8(source1, source2);
|
||||
let mut pix = _mm256_unpacklo_epi8(source, _mm256_setzero_si256());
|
||||
sss0 = _mm256_add_epi32(sss0, _mm256_madd_epi16(pix, mmk));
|
||||
pix = _mm256_unpackhi_epi8(source, _mm256_setzero_si256());
|
||||
sss1 = _mm256_add_epi32(sss1, _mm256_madd_epi16(pix, mmk));
|
||||
|
||||
source = _mm256_unpackhi_epi8(source1, _mm256_setzero_si256());
|
||||
pix = _mm256_unpacklo_epi8(source, _mm256_setzero_si256());
|
||||
sss2 = _mm256_add_epi32(sss2, _mm256_madd_epi16(pix, mmk));
|
||||
pix = _mm256_unpackhi_epi8(source, _mm256_setzero_si256());
|
||||
sss3 = _mm256_add_epi32(sss3, _mm256_madd_epi16(pix, mmk));
|
||||
|
||||
y += 1;
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss0 = _mm256_srai_epi32(sss0, $imm8);
|
||||
sss1 = _mm256_srai_epi32(sss1, $imm8);
|
||||
sss2 = _mm256_srai_epi32(sss2, $imm8);
|
||||
sss3 = _mm256_srai_epi32(sss3, $imm8);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
|
||||
sss0 = _mm256_packs_epi32(sss0, sss1);
|
||||
sss2 = _mm256_packs_epi32(sss2, sss3);
|
||||
sss0 = _mm256_packus_epi16(sss0, sss2);
|
||||
let dst_ptr = dst_row.get_unchecked_mut(x..).as_mut_ptr() as *mut __m256i;
|
||||
_mm256_storeu_si256(dst_ptr, sss0);
|
||||
|
||||
x += 8;
|
||||
}
|
||||
|
||||
while x < src_width.saturating_sub(1) {
|
||||
let mut sss0 = initial; // left row
|
||||
let mut sss1 = initial; // right row
|
||||
let mut y: u32 = 0;
|
||||
|
||||
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
|
||||
// Load two coefficients at once
|
||||
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
|
||||
|
||||
let source1 = simd_utils::loadl_epi64(s_row1, x); // top line
|
||||
let source2 = simd_utils::loadl_epi64(s_row2, x); // bottom line
|
||||
|
||||
let source = _mm_unpacklo_epi8(source1, source2);
|
||||
let mut pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
|
||||
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
|
||||
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
|
||||
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
y += 2;
|
||||
}
|
||||
|
||||
for s_row in src_img.iter_rows(y_start + y, y_start + y_size) {
|
||||
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
|
||||
|
||||
let source1 = simd_utils::loadl_epi64(s_row, x); // top line
|
||||
let source2 = _mm_setzero_si128(); // bottom line is empty
|
||||
|
||||
let source = _mm_unpacklo_epi8(source1, source2);
|
||||
let mut pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
|
||||
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
|
||||
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
|
||||
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
y += 1;
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss0 = _mm_srai_epi32(sss0, $imm8);
|
||||
sss1 = _mm_srai_epi32(sss1, $imm8);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
|
||||
sss0 = _mm_packs_epi32(sss0, sss1);
|
||||
sss0 = _mm_packus_epi16(sss0, sss0);
|
||||
let dst_ptr = dst_row.get_unchecked_mut(x..).as_mut_ptr() as *mut __m128i;
|
||||
_mm_storel_epi64(dst_ptr, sss0);
|
||||
|
||||
x += 2;
|
||||
}
|
||||
|
||||
while x < src_width {
|
||||
let mut sss = initial;
|
||||
let mut y: u32 = 0;
|
||||
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
|
||||
// Load two coefficients at once
|
||||
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
|
||||
|
||||
let source1 = simd_utils::mm_cvtsi32_si128(s_row1, x); // top line
|
||||
let source2 = simd_utils::mm_cvtsi32_si128(s_row2, x); // bottom line
|
||||
|
||||
let source = _mm_unpacklo_epi8(source1, source2);
|
||||
let pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
|
||||
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
y += 2;
|
||||
}
|
||||
|
||||
for s_row in src_img.iter_rows(y_start + y, y_start + y_size) {
|
||||
let pix = simd_utils::mm_cvtepu8_epi32(s_row, x);
|
||||
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
|
||||
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
y += 1;
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss = _mm_srai_epi32(sss, $imm8);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
|
||||
sss = _mm_packs_epi32(sss, sss);
|
||||
*dst_row.get_unchecked_mut(x) =
|
||||
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss, sss)));
|
||||
|
||||
x += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl Convolution for Avx2 {
|
||||
#[inline]
|
||||
fn horiz_convolution(
|
||||
&self,
|
||||
src_image: &SrcImageView,
|
||||
dst_image: &mut DstImageView,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let mut normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coeffs_i16 = normalizer_guard.normalized();
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
self.horiz_convolution_8u4x(
|
||||
src_rows,
|
||||
dst_rows,
|
||||
coeffs_i16,
|
||||
window_size,
|
||||
&bounds_per_pixel,
|
||||
precision,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
let mut yy = dst_height - dst_height % 4;
|
||||
while yy < dst_height {
|
||||
unsafe {
|
||||
self.horiz_convolution_8u(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
coeffs_i16,
|
||||
window_size,
|
||||
&bounds_per_pixel,
|
||||
precision,
|
||||
);
|
||||
}
|
||||
yy += 1;
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
fn vert_convolution(
|
||||
&self,
|
||||
src_image: &SrcImageView,
|
||||
dst_image: &mut DstImageView,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let mut normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coeffs_i16 = normalizer_guard.normalized();
|
||||
let coeffs_chunks = coeffs_i16.chunks(window_size);
|
||||
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for ((&bound, k), dst_row) in bounds.iter().zip(coeffs_chunks).zip(dst_rows) {
|
||||
unsafe {
|
||||
self.vert_convolution_8u(src_image, dst_row, k, bound, precision);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,133 @@
|
||||
use std::f64::consts::PI;
|
||||
|
||||
pub type FilterFn<'a> = &'a dyn Fn(f64) -> f64;
|
||||
|
||||
#[derive(Clone, Copy, Debug, PartialEq)]
|
||||
#[non_exhaustive]
|
||||
pub enum FilterType {
|
||||
/// Each pixel of source image contributes to one pixel of the
|
||||
/// destination image with identical weights. For upscaling is equivalent
|
||||
/// of `Nearest` resize algorithm.
|
||||
Box,
|
||||
/// Bilinear filter calculate the output pixel value using linear
|
||||
/// interpolation on all pixels that may contribute to the output value.
|
||||
Bilinear,
|
||||
/// Hamming filter has the same performance as `Bilinear` filter while
|
||||
/// providing the image downscaling quality comparable to bicubic
|
||||
/// (`CatmulRom` or `Mitchell`). Produces a sharper image than `Bilinear`,
|
||||
/// doesn't have dislocations on local level like with `Box`.
|
||||
/// The filter don’t show good quality for the image upscaling.
|
||||
Hamming,
|
||||
/// Catmull-Rom bicubic filter calculate the output pixel value using
|
||||
/// cubic interpolation on all pixels that may contribute to the output
|
||||
/// value.
|
||||
CatmulRom,
|
||||
/// Mitchell–Netravali bicubic filter calculate the output pixel value
|
||||
/// using cubic interpolation on all pixels that may contribute to the
|
||||
/// output value.
|
||||
Mitchell,
|
||||
/// Lanczos3 filter calculate the output pixel value using a high-quality
|
||||
/// Lanczos filter (a truncated sinc) on all pixels that may contribute
|
||||
/// to the output value.
|
||||
Lanczos3,
|
||||
}
|
||||
|
||||
impl Default for FilterType {
|
||||
fn default() -> Self {
|
||||
FilterType::Lanczos3
|
||||
}
|
||||
}
|
||||
|
||||
/// Returns reference to filter function and value of `filter_support`.
|
||||
#[inline]
|
||||
pub fn get_filter_func(filter_type: FilterType) -> (FilterFn<'static>, f64) {
|
||||
match filter_type {
|
||||
FilterType::Box => (&box_filter, 0.5),
|
||||
FilterType::Bilinear => (&bilinear_filter, 1.0),
|
||||
FilterType::Hamming => (&hamming_filter, 1.0),
|
||||
FilterType::CatmulRom => (&catmul_filter, 2.0),
|
||||
FilterType::Mitchell => (&mitchell_filter, 2.0),
|
||||
FilterType::Lanczos3 => (&lanczos_filter, 3.0),
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
fn box_filter(x: f64) -> f64 {
|
||||
if x > -0.5 && x <= 0.5 {
|
||||
1.0
|
||||
} else {
|
||||
0.0
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
fn bilinear_filter(mut x: f64) -> f64 {
|
||||
x = x.abs();
|
||||
if x < 1.0 {
|
||||
1.0 - x
|
||||
} else {
|
||||
0.0
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
fn hamming_filter(mut x: f64) -> f64 {
|
||||
x = x.abs();
|
||||
if x == 0.0 {
|
||||
1.0
|
||||
} else if x >= 1.0 {
|
||||
0.0
|
||||
} else {
|
||||
x *= PI;
|
||||
(0.54 + 0.46 * x.cos()) * x.sin() / x
|
||||
}
|
||||
}
|
||||
|
||||
/// Catmull-Rom (bicubic) filter
|
||||
/// https://en.wikipedia.org/wiki/Bicubic_interpolation#Bicubic_convolution_algorithm
|
||||
#[inline]
|
||||
fn catmul_filter(mut x: f64) -> f64 {
|
||||
const A: f64 = -0.5;
|
||||
x = x.abs();
|
||||
if x < 1.0 {
|
||||
((A + 2.) * x - (A + 3.)) * x * x + 1.
|
||||
} else if x < 2.0 {
|
||||
(((x - 5.) * x + 8.) * x - 4.) * A
|
||||
} else {
|
||||
0.0
|
||||
}
|
||||
}
|
||||
|
||||
/// Mitchell–Netravali filter (B = C = 1/3)
|
||||
/// https://en.wikipedia.org/wiki/Mitchell%E2%80%93Netravali_filters
|
||||
#[inline]
|
||||
fn mitchell_filter(mut x: f64) -> f64 {
|
||||
x = x.abs();
|
||||
if x < 1.0 {
|
||||
(7. * x / 6. - 2.) * x * x + 16. / 18.
|
||||
} else if x < 2.0 {
|
||||
((2. - 7. * x / 18.) * x - 10. / 3.) * x + 16. / 9.
|
||||
} else {
|
||||
0.0
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
fn sinc_filter(mut x: f64) -> f64 {
|
||||
if x == 0.0 {
|
||||
1.0
|
||||
} else {
|
||||
x *= PI;
|
||||
x.sin() / x
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
fn lanczos_filter(x: f64) -> f64 {
|
||||
// truncated sinc
|
||||
if (-3.0..3.0).contains(&x) {
|
||||
sinc_filter(x) * sinc_filter(x / 3.)
|
||||
} else {
|
||||
0.0
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,40 @@
|
||||
macro_rules! constify_imm8 {
|
||||
($imm8:expr, $expand:ident) => {
|
||||
#[allow(overflowing_literals)]
|
||||
match ($imm8) & 0b0011_1111 {
|
||||
0 => $expand!(0),
|
||||
1 => $expand!(1),
|
||||
2 => $expand!(2),
|
||||
3 => $expand!(3),
|
||||
4 => $expand!(4),
|
||||
5 => $expand!(5),
|
||||
6 => $expand!(6),
|
||||
7 => $expand!(7),
|
||||
8 => $expand!(8),
|
||||
9 => $expand!(9),
|
||||
10 => $expand!(10),
|
||||
12 => $expand!(12),
|
||||
13 => $expand!(13),
|
||||
14 => $expand!(14),
|
||||
15 => $expand!(15),
|
||||
16 => $expand!(16),
|
||||
17 => $expand!(17),
|
||||
18 => $expand!(18),
|
||||
19 => $expand!(19),
|
||||
20 => $expand!(20),
|
||||
21 => $expand!(21),
|
||||
22 => $expand!(22),
|
||||
23 => $expand!(23),
|
||||
24 => $expand!(24),
|
||||
25 => $expand!(25),
|
||||
26 => $expand!(26),
|
||||
27 => $expand!(27),
|
||||
28 => $expand!(28),
|
||||
29 => $expand!(29),
|
||||
30 => $expand!(30),
|
||||
31 => $expand!(31),
|
||||
32 => $expand!(32),
|
||||
_ => unreachable!(),
|
||||
}
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,114 @@
|
||||
use std::num::NonZeroU32;
|
||||
|
||||
pub use avx2::Avx2;
|
||||
pub use filters::{get_filter_func, FilterType};
|
||||
pub use native::{NativeF32, NativeI32, NativeU8x4};
|
||||
pub use sse4::Sse4;
|
||||
|
||||
use crate::image_view::{DstImageView, SrcImageView};
|
||||
|
||||
#[macro_use]
|
||||
mod macros;
|
||||
|
||||
mod avx2;
|
||||
mod filters;
|
||||
mod native;
|
||||
mod sse4;
|
||||
|
||||
pub trait Convolution {
|
||||
fn horiz_convolution(
|
||||
&self,
|
||||
src_image: &SrcImageView,
|
||||
dst_image: &mut DstImageView,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
);
|
||||
|
||||
fn vert_convolution(
|
||||
&self,
|
||||
src_image: &SrcImageView,
|
||||
dst_image: &mut DstImageView,
|
||||
coeffs: Coefficients,
|
||||
);
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
pub struct Bound {
|
||||
pub start: u32,
|
||||
pub size: u32,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct Coefficients {
|
||||
pub values: Vec<f64>,
|
||||
pub window_size: usize,
|
||||
pub bounds: Vec<Bound>,
|
||||
}
|
||||
|
||||
pub fn precompute_coefficients(
|
||||
in_size: NonZeroU32,
|
||||
in0: f64, // Left border for cropping
|
||||
in1: f64, // Right border for cropping
|
||||
out_size: NonZeroU32,
|
||||
filter: &dyn Fn(f64) -> f64,
|
||||
filter_support: f64,
|
||||
) -> Coefficients {
|
||||
let in_size = in_size.get();
|
||||
let out_size = out_size.get();
|
||||
|
||||
let scale = (in1 - in0) / out_size as f64;
|
||||
let filter_scale = scale.max(1.0);
|
||||
|
||||
// Determine filter radius size (length of resampling filter)
|
||||
let filter_radius = filter_support * filter_scale;
|
||||
// Maximum number of coeffs per out pixel
|
||||
let window_size = filter_radius.ceil() as usize * 2 + 1;
|
||||
// Optimization: replace division by filter_scale
|
||||
// with multiplication by inv_filter_scale.
|
||||
let inv_filter_scale = 1.0 / filter_scale;
|
||||
|
||||
let count_of_coeffs = window_size * out_size as usize;
|
||||
let mut coeffs: Vec<f64> = Vec::with_capacity(count_of_coeffs);
|
||||
let mut bounds: Vec<Bound> = Vec::with_capacity(out_size as usize);
|
||||
|
||||
for out_x in 0..out_size {
|
||||
// Find the point in the input image corresponding to the centre
|
||||
// of the current pixel in the output image.
|
||||
let in_center = in0 + (out_x as f64 + 0.5) * scale;
|
||||
|
||||
// x_min and x_max are slice bounds for the input pixels relevant
|
||||
// to the output pixel we are calculating. Pixel x is relevant
|
||||
// if and only if (x >= x_min) && (x < x_max).
|
||||
// Invariant: 0 <= x_min < x_max <= width
|
||||
let x_min = (in_center - filter_radius).floor().max(0.) as u32;
|
||||
let x_max = (in_center + filter_radius).ceil().min(in_size as f64) as u32;
|
||||
|
||||
let cur_index = coeffs.len();
|
||||
let mut ww: f64 = 0.0;
|
||||
|
||||
// Optimisation for follow for-cycle:
|
||||
// (x + 0.5) - in_center => x - (in_center - 0.5) => x - center
|
||||
let center = in_center - 0.5;
|
||||
|
||||
for x in x_min..x_max {
|
||||
let w: f64 = filter((x as f64 - center) * inv_filter_scale);
|
||||
coeffs.push(w);
|
||||
ww += w;
|
||||
}
|
||||
if ww != 0.0 {
|
||||
coeffs[cur_index..].iter_mut().for_each(|w| *w /= ww);
|
||||
}
|
||||
// Remaining values should stay empty if they are used despite of x_max.
|
||||
coeffs.resize(cur_index + window_size, 0.);
|
||||
bounds.push(Bound {
|
||||
start: x_min,
|
||||
size: x_max - x_min,
|
||||
});
|
||||
}
|
||||
|
||||
Coefficients {
|
||||
values: coeffs,
|
||||
window_size,
|
||||
bounds,
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,244 @@
|
||||
use std::slice;
|
||||
|
||||
use crate::convolution::{Coefficients, Convolution};
|
||||
use crate::image_view::{DstImageView, SrcImageView};
|
||||
use crate::optimisations;
|
||||
|
||||
pub struct NativeU8x4;
|
||||
|
||||
impl Convolution for NativeU8x4 {
|
||||
fn horiz_convolution(
|
||||
&self,
|
||||
src_image: &SrcImageView,
|
||||
dst_image: &mut DstImageView,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let mut normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coeffs_i16 = normalizer_guard.normalized();
|
||||
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for (y_dst, dst_row) in dst_rows.enumerate() {
|
||||
let y_src = y_dst as u32 + offset;
|
||||
|
||||
for (x_dst, (&bound, dst_pixel)) in bounds.iter().zip(dst_row.iter_mut()).enumerate() {
|
||||
let first_x_src = bound.start;
|
||||
let start_index = window_size * x_dst;
|
||||
let end_index = start_index + bound.size as usize;
|
||||
let ks = &coeffs_i16[start_index..end_index];
|
||||
|
||||
let mut ss0 = 1 << (precision - 1);
|
||||
let mut ss1 = ss0;
|
||||
let mut ss2 = ss0;
|
||||
let mut ss3 = ss0;
|
||||
let src_pixels = src_image.iter_horiz(first_x_src, y_src);
|
||||
for (&k, &src_pixel) in ks.iter().zip(src_pixels) {
|
||||
let components: [u8; 4] = src_pixel.to_le_bytes();
|
||||
ss0 += components[0] as i32 * (k as i32);
|
||||
ss1 += components[1] as i32 * (k as i32);
|
||||
ss2 += components[2] as i32 * (k as i32);
|
||||
ss3 += components[3] as i32 * (k as i32);
|
||||
}
|
||||
let t: [u8; 4] = unsafe {
|
||||
[
|
||||
optimisations::clip8(ss0, precision),
|
||||
optimisations::clip8(ss1, precision),
|
||||
optimisations::clip8(ss2, precision),
|
||||
optimisations::clip8(ss3, precision),
|
||||
]
|
||||
};
|
||||
*dst_pixel = u32::from_le_bytes(t);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn vert_convolution(
|
||||
&self,
|
||||
src_image: &SrcImageView,
|
||||
dst_image: &mut DstImageView,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let mut normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coeffs_i16 = normalizer_guard.normalized();
|
||||
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for (y_dst, (&bound, dst_row)) in bounds.iter().zip(dst_rows).enumerate() {
|
||||
let first_y_src = bound.start;
|
||||
let start_index = window_size * y_dst;
|
||||
let end_index = start_index + bound.size as usize;
|
||||
let ks = &coeffs_i16[start_index..end_index];
|
||||
|
||||
for (x_src, out_pixel) in dst_row.iter_mut().enumerate() {
|
||||
let mut ss0 = 1 << (precision - 1);
|
||||
let mut ss1 = ss0;
|
||||
let mut ss2 = ss0;
|
||||
let mut ss3 = ss0;
|
||||
for (dy, &k) in ks.iter().enumerate() {
|
||||
let pixel = src_image.get_pixel_u32(x_src as u32, first_y_src + dy as u32);
|
||||
let components: [u8; 4] = pixel.to_le_bytes();
|
||||
ss0 += components[0] as i32 * (k as i32);
|
||||
ss1 += components[1] as i32 * (k as i32);
|
||||
ss2 += components[2] as i32 * (k as i32);
|
||||
ss3 += components[3] as i32 * (k as i32);
|
||||
}
|
||||
let t: [u8; 4] = unsafe {
|
||||
[
|
||||
optimisations::clip8(ss0, precision),
|
||||
optimisations::clip8(ss1, precision),
|
||||
optimisations::clip8(ss2, precision),
|
||||
optimisations::clip8(ss3, precision),
|
||||
]
|
||||
};
|
||||
*out_pixel = u32::from_le_bytes(t);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub struct NativeI32;
|
||||
|
||||
impl Convolution for NativeI32 {
|
||||
fn horiz_convolution(
|
||||
&self,
|
||||
src_image: &SrcImageView,
|
||||
dst_image: &mut DstImageView,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
for y_dst in 0..dst_image.height().get() {
|
||||
let y_src = y_dst + offset;
|
||||
if let Some(out_row) = dst_image.get_row_mut(y_dst) {
|
||||
let out_row_i32 = unsafe {
|
||||
let len = out_row.len();
|
||||
let ptr = out_row.as_mut_ptr();
|
||||
slice::from_raw_parts_mut(ptr as *mut i32, len)
|
||||
};
|
||||
for (x_dst, (&bound, out_pixel)) in bounds.iter().zip(out_row_i32).enumerate() {
|
||||
let first_x_src = bound.start;
|
||||
let start_index = window_size * x_dst;
|
||||
let end_index = start_index + bound.size as usize;
|
||||
|
||||
let ks = &values[start_index..end_index];
|
||||
|
||||
let mut ss = 0.;
|
||||
let pixels = src_image.iter_horiz_i32(first_x_src, y_src);
|
||||
for (&k, &pixel) in ks.iter().zip(pixels) {
|
||||
ss += pixel as f64 * k;
|
||||
}
|
||||
*out_pixel = ss.round() as i32;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn vert_convolution(
|
||||
&self,
|
||||
image: &SrcImageView,
|
||||
out_image: &mut DstImageView,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
for (y_dst, &bound) in bounds.iter().enumerate() {
|
||||
let first_y_src = bound.start;
|
||||
let start_index = window_size * y_dst;
|
||||
let end_index = start_index + bound.size as usize;
|
||||
let ks = &values[start_index..end_index];
|
||||
|
||||
if let Some(out_row) = out_image.get_row_mut(y_dst as u32) {
|
||||
let out_row_i32 = unsafe {
|
||||
let len = out_row.len();
|
||||
let ptr = out_row.as_mut_ptr();
|
||||
slice::from_raw_parts_mut(ptr as *mut i32, len)
|
||||
};
|
||||
for (x_src, out_pixel) in out_row_i32.iter_mut().enumerate() {
|
||||
let mut ss = 0.;
|
||||
for (dy, &k) in ks.iter().enumerate() {
|
||||
let pixel = image.get_pixel_i32(x_src as u32, first_y_src + dy as u32);
|
||||
ss += pixel as f64 * k;
|
||||
}
|
||||
*out_pixel = ss.round() as i32;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub struct NativeF32;
|
||||
|
||||
impl Convolution for NativeF32 {
|
||||
fn horiz_convolution(
|
||||
&self,
|
||||
src_image: &SrcImageView,
|
||||
dst_image: &mut DstImageView,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
for y_dst in 0..dst_image.height().get() {
|
||||
let y_src = y_dst + offset;
|
||||
if let Some(out_row) = dst_image.get_row_mut(y_dst) {
|
||||
let out_row_f32 = unsafe {
|
||||
let len = out_row.len();
|
||||
let ptr = out_row.as_mut_ptr();
|
||||
slice::from_raw_parts_mut(ptr as *mut f32, len)
|
||||
};
|
||||
for (x_dst, (&bound, out_pixel)) in bounds.iter().zip(out_row_f32).enumerate() {
|
||||
let first_x_src = bound.start;
|
||||
let start_index = window_size * x_dst;
|
||||
let end_index = start_index + bound.size as usize;
|
||||
|
||||
let ks = &values[start_index..end_index];
|
||||
|
||||
let mut ss = 0.;
|
||||
let pixels = src_image.iter_horiz_f32(first_x_src, y_src);
|
||||
for (&k, &pixel) in ks.iter().zip(pixels) {
|
||||
ss += pixel as f64 * k;
|
||||
}
|
||||
*out_pixel = ss as f32;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn vert_convolution(
|
||||
&self,
|
||||
src_image: &SrcImageView,
|
||||
dst_image: &mut DstImageView,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
for (y_dst, &bound) in bounds.iter().enumerate() {
|
||||
let first_y_src = bound.start;
|
||||
let start_index = window_size * y_dst;
|
||||
let end_index = start_index + bound.size as usize;
|
||||
let ks = &values[start_index..end_index];
|
||||
|
||||
if let Some(out_row) = dst_image.get_row_mut(y_dst as u32) {
|
||||
let out_row_f32 = unsafe {
|
||||
let len = out_row.len();
|
||||
let ptr = out_row.as_mut_ptr();
|
||||
slice::from_raw_parts_mut(ptr as *mut f32, len)
|
||||
};
|
||||
for (x_src, out_pixel) in out_row_f32.iter_mut().enumerate() {
|
||||
let mut ss = 0.;
|
||||
for (dy, &k) in ks.iter().enumerate() {
|
||||
let pixel = src_image.get_pixel_f32(x_src as u32, first_y_src + dy as u32);
|
||||
ss += pixel as f64 * k;
|
||||
}
|
||||
*out_pixel = ss as f32;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,545 @@
|
||||
use std::arch::x86_64::*;
|
||||
use std::intrinsics::transmute;
|
||||
|
||||
use crate::convolution::{Bound, Coefficients, Convolution};
|
||||
use crate::image_view::{DstImageView, FourRows, FourRowsMut, SrcImageView};
|
||||
use crate::{optimisations, simd_utils};
|
||||
|
||||
pub struct Sse4;
|
||||
|
||||
// This code is based on C-implementation from Pillow-SIMD package for Python
|
||||
// https://github.com/uploadcare/pillow-simd
|
||||
impl Sse4 {
|
||||
/// For safety, it is necessary to ensure the following conditions:
|
||||
/// - length of all rows in src_rows must be equal
|
||||
/// - length of all rows in dst_rows must be equal
|
||||
/// - bounds.len() == dst_rows.0.len()
|
||||
/// - coeffs.len() == dst_rows.0.len() * window_size
|
||||
/// - max(bound.size for bound in bounds) <= window_size
|
||||
/// - max(bound.start + bound.size for bound in bounds) <= src_row.0.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
unsafe fn horiz_convolution_8u4x(
|
||||
&self,
|
||||
src_rows: FourRows,
|
||||
dst_rows: FourRowsMut,
|
||||
coeffs: &[i16],
|
||||
window_size: usize,
|
||||
bounds: &[Bound],
|
||||
precision: u8,
|
||||
) {
|
||||
let (s_row0, s_row1, s_row2, s_row3) = src_rows;
|
||||
let (d_row0, d_row1, d_row2, d_row3) = dst_rows;
|
||||
let initial = _mm_set1_epi32(1 << (precision - 1));
|
||||
let mask_lo = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
|
||||
let mask_hi = _mm_set_epi8(-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8);
|
||||
let mask = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
|
||||
|
||||
let coeffs_chunks = coeffs.chunks(window_size);
|
||||
for (xx, (&bound, k)) in bounds.iter().zip(coeffs_chunks).enumerate() {
|
||||
let x_start = bound.start as usize;
|
||||
let x_size = bound.size as usize;
|
||||
let mut x: usize = 0;
|
||||
|
||||
let mut sss0 = initial;
|
||||
let mut sss1 = initial;
|
||||
let mut sss2 = initial;
|
||||
let mut sss3 = initial;
|
||||
|
||||
while x < x_size.saturating_sub(3) {
|
||||
let mmk_lo = simd_utils::ptr_i16_to_set1_epi32(k, x);
|
||||
let mmk_hi = simd_utils::ptr_i16_to_set1_epi32(k, x + 2);
|
||||
|
||||
// [8] a3 b3 g3 r3 a2 b2 g2 r2 a1 b1 g1 r1 a0 b0 g0 r0
|
||||
let mut source = simd_utils::loadu_si128(s_row0, x + x_start);
|
||||
// [16] a1 a0 b1 b0 g1 g0 r1 r0
|
||||
let mut pix = _mm_shuffle_epi8(source, mask_lo);
|
||||
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk_lo));
|
||||
// [16] a3 a2 b3 b2 g3 g2 r3 r2
|
||||
pix = _mm_shuffle_epi8(source, mask_hi);
|
||||
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk_hi));
|
||||
|
||||
source = simd_utils::loadu_si128(s_row1, x + x_start);
|
||||
pix = _mm_shuffle_epi8(source, mask_lo);
|
||||
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk_lo));
|
||||
pix = _mm_shuffle_epi8(source, mask_hi);
|
||||
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk_hi));
|
||||
|
||||
source = simd_utils::loadu_si128(s_row2, x + x_start);
|
||||
pix = _mm_shuffle_epi8(source, mask_lo);
|
||||
sss2 = _mm_add_epi32(sss2, _mm_madd_epi16(pix, mmk_lo));
|
||||
pix = _mm_shuffle_epi8(source, mask_hi);
|
||||
sss2 = _mm_add_epi32(sss2, _mm_madd_epi16(pix, mmk_hi));
|
||||
|
||||
source = simd_utils::loadu_si128(s_row3, x + x_start);
|
||||
pix = _mm_shuffle_epi8(source, mask_lo);
|
||||
sss3 = _mm_add_epi32(sss3, _mm_madd_epi16(pix, mmk_lo));
|
||||
pix = _mm_shuffle_epi8(source, mask_hi);
|
||||
sss3 = _mm_add_epi32(sss3, _mm_madd_epi16(pix, mmk_hi));
|
||||
x += 4;
|
||||
}
|
||||
|
||||
while x < x_size.saturating_sub(1) {
|
||||
// [16] k1 k0 k1 k0 k1 k0 k1 k0
|
||||
let mmk = simd_utils::ptr_i16_to_set1_epi32(k, x);
|
||||
|
||||
// [8] x x x x x x x x a1 b1 g1 r1 a0 b0 g0 r0
|
||||
let mut pix = simd_utils::loadl_epi64(s_row0, x + x_start);
|
||||
// [16] a1 a0 b1 b0 g1 g0 r1 r0
|
||||
pix = _mm_shuffle_epi8(pix, mask);
|
||||
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
pix = simd_utils::loadl_epi64(s_row1, x + x_start);
|
||||
pix = _mm_shuffle_epi8(pix, mask);
|
||||
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
pix = simd_utils::loadl_epi64(s_row2, x + x_start);
|
||||
pix = _mm_shuffle_epi8(pix, mask);
|
||||
sss2 = _mm_add_epi32(sss2, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
pix = simd_utils::loadl_epi64(s_row3, x + x_start);
|
||||
pix = _mm_shuffle_epi8(pix, mask);
|
||||
sss3 = _mm_add_epi32(sss3, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
x += 2;
|
||||
}
|
||||
|
||||
while x < x_size {
|
||||
// [16] xx k0 xx k0 xx k0 xx k0
|
||||
let mmk = _mm_set1_epi32(*k.get_unchecked(x) as i32);
|
||||
// [16] xx a0 xx b0 xx g0 xx r0
|
||||
let mut pix = simd_utils::mm_cvtepu8_epi32(s_row0, x);
|
||||
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
pix = simd_utils::mm_cvtepu8_epi32(s_row1, x);
|
||||
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
pix = simd_utils::mm_cvtepu8_epi32(s_row2, x);
|
||||
sss2 = _mm_add_epi32(sss2, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
pix = simd_utils::mm_cvtepu8_epi32(s_row3, x);
|
||||
sss3 = _mm_add_epi32(sss3, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
x += 1;
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss0 = _mm_srai_epi32(sss0, $imm8);
|
||||
sss1 = _mm_srai_epi32(sss1, $imm8);
|
||||
sss2 = _mm_srai_epi32(sss2, $imm8);
|
||||
sss3 = _mm_srai_epi32(sss3, $imm8);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
|
||||
sss0 = _mm_packs_epi32(sss0, sss0);
|
||||
sss1 = _mm_packs_epi32(sss1, sss1);
|
||||
sss2 = _mm_packs_epi32(sss2, sss2);
|
||||
sss3 = _mm_packs_epi32(sss3, sss3);
|
||||
*d_row0.get_unchecked_mut(xx) =
|
||||
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss0, sss0)));
|
||||
*d_row1.get_unchecked_mut(xx) =
|
||||
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss1, sss1)));
|
||||
*d_row2.get_unchecked_mut(xx) =
|
||||
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss2, sss2)));
|
||||
*d_row3.get_unchecked_mut(xx) =
|
||||
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss3, sss3)));
|
||||
}
|
||||
}
|
||||
|
||||
/// For safety, it is necessary to ensure the following conditions:
|
||||
/// - bounds.len() == dst_row.len()
|
||||
/// - coeffs.len() == dst_rows.0.len() * window_size
|
||||
/// - max(bound.start + bound.size for bound in bounds) <= src_row.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
unsafe fn horiz_convolution_8u(
|
||||
&self,
|
||||
src_row: &[u32],
|
||||
dst_row: &mut [u32],
|
||||
coeffs: &[i16],
|
||||
window_size: usize,
|
||||
bounds: &[Bound],
|
||||
precision: u8,
|
||||
) {
|
||||
let coeffs_chunks = coeffs.chunks(window_size);
|
||||
let initial = _mm_set1_epi32(1 << (precision - 1));
|
||||
let sh1 = _mm_set_epi8(-1, 11, -1, 3, -1, 10, -1, 2, -1, 9, -1, 1, -1, 8, -1, 0);
|
||||
let sh2 = _mm_set_epi8(5, 4, 1, 0, 5, 4, 1, 0, 5, 4, 1, 0, 5, 4, 1, 0);
|
||||
let sh3 = _mm_set_epi8(-1, 15, -1, 7, -1, 14, -1, 6, -1, 13, -1, 5, -1, 12, -1, 4);
|
||||
let sh4 = _mm_set_epi8(7, 6, 3, 2, 7, 6, 3, 2, 7, 6, 3, 2, 7, 6, 3, 2);
|
||||
let sh5 = _mm_set_epi8(13, 12, 9, 8, 13, 12, 9, 8, 13, 12, 9, 8, 13, 12, 9, 8);
|
||||
let sh6 = _mm_set_epi8(
|
||||
15, 14, 11, 10, 15, 14, 11, 10, 15, 14, 11, 10, 15, 14, 11, 10,
|
||||
);
|
||||
let sh7 = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
|
||||
|
||||
for (xx, (&bound, k)) in bounds.iter().zip(coeffs_chunks).enumerate() {
|
||||
let x_start = bound.start as usize;
|
||||
let x_size = bound.size as usize;
|
||||
let mut x: usize = 0;
|
||||
let mut sss = initial;
|
||||
|
||||
while x < x_size.saturating_sub(7) {
|
||||
let ksource = simd_utils::loadu_si128(k, x);
|
||||
|
||||
let mut source = simd_utils::loadu_si128(src_row, x + x_start);
|
||||
|
||||
let mut pix = _mm_shuffle_epi8(source, sh1);
|
||||
let mut mmk = _mm_shuffle_epi8(ksource, sh2);
|
||||
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
pix = _mm_shuffle_epi8(source, sh3);
|
||||
mmk = _mm_shuffle_epi8(ksource, sh4);
|
||||
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
source = simd_utils::loadu_si128(src_row, x + 4 + x_start);
|
||||
|
||||
pix = _mm_shuffle_epi8(source, sh1);
|
||||
mmk = _mm_shuffle_epi8(ksource, sh5);
|
||||
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
pix = _mm_shuffle_epi8(source, sh3);
|
||||
mmk = _mm_shuffle_epi8(ksource, sh6);
|
||||
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
x += 8;
|
||||
}
|
||||
|
||||
while x < x_size.saturating_sub(3) {
|
||||
let source = simd_utils::loadu_si128(src_row, x + x_start);
|
||||
let ksource = simd_utils::loadl_epi64(k, x);
|
||||
|
||||
let mut pix = _mm_shuffle_epi8(source, sh1);
|
||||
let mut mmk = _mm_shuffle_epi8(ksource, sh2);
|
||||
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
pix = _mm_shuffle_epi8(source, sh3);
|
||||
mmk = _mm_shuffle_epi8(ksource, sh4);
|
||||
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
x += 4;
|
||||
}
|
||||
|
||||
while x < x_size.saturating_sub(1) {
|
||||
let mmk = simd_utils::ptr_i16_to_set1_epi32(k, x);
|
||||
let source = simd_utils::loadl_epi64(src_row, x + x_start);
|
||||
let pix = _mm_shuffle_epi8(source, sh7);
|
||||
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
x += 2
|
||||
}
|
||||
|
||||
while x < x_size {
|
||||
let pix = simd_utils::mm_cvtepu8_epi32(src_row, x + x_start);
|
||||
let mmk = _mm_set1_epi32(*k.get_unchecked(x) as i32);
|
||||
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
x += 1;
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss = _mm_srai_epi32(sss, $imm8);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
|
||||
sss = _mm_packs_epi32(sss, sss);
|
||||
*dst_row.get_unchecked_mut(xx) =
|
||||
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss, sss)));
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
pub unsafe fn vert_convolution_8u(
|
||||
&self,
|
||||
src_img: &SrcImageView,
|
||||
dst_row: &mut [u32],
|
||||
coeffs: &[i16],
|
||||
bound: Bound,
|
||||
precision: u8,
|
||||
) {
|
||||
let mut xx: usize = 0;
|
||||
let src_width = src_img.width().get() as usize;
|
||||
let y_start = bound.start;
|
||||
let y_size = bound.size;
|
||||
|
||||
let initial = _mm_set1_epi32(1 << (precision - 1));
|
||||
|
||||
while xx < src_width.saturating_sub(7) {
|
||||
let mut sss0 = initial;
|
||||
let mut sss1 = initial;
|
||||
let mut sss2 = initial;
|
||||
let mut sss3 = initial;
|
||||
let mut sss4 = initial;
|
||||
let mut sss5 = initial;
|
||||
let mut sss6 = initial;
|
||||
let mut sss7 = initial;
|
||||
|
||||
let mut y: u32 = 0;
|
||||
|
||||
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
|
||||
// Load two coefficients at once
|
||||
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
|
||||
|
||||
let mut source1 = simd_utils::loadu_si128(s_row1, xx); // top line
|
||||
let mut source2 = simd_utils::loadu_si128(s_row2, xx); // bottom line
|
||||
|
||||
let mut source = _mm_unpacklo_epi8(source1, source2);
|
||||
let mut pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
|
||||
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
|
||||
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
|
||||
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
source = _mm_unpackhi_epi8(source1, source2);
|
||||
pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
|
||||
sss2 = _mm_add_epi32(sss2, _mm_madd_epi16(pix, mmk));
|
||||
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
|
||||
sss3 = _mm_add_epi32(sss3, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
source1 = simd_utils::loadu_si128(s_row1, xx + 4); // top line
|
||||
source2 = simd_utils::loadu_si128(s_row2, xx + 4); // bottom line
|
||||
|
||||
source = _mm_unpacklo_epi8(source1, source2);
|
||||
pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
|
||||
sss4 = _mm_add_epi32(sss4, _mm_madd_epi16(pix, mmk));
|
||||
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
|
||||
sss5 = _mm_add_epi32(sss5, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
source = _mm_unpackhi_epi8(source1, source2);
|
||||
pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
|
||||
sss6 = _mm_add_epi32(sss6, _mm_madd_epi16(pix, mmk));
|
||||
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
|
||||
sss7 = _mm_add_epi32(sss7, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
y += 2;
|
||||
}
|
||||
|
||||
for s_row in src_img.iter_rows(y_start + y, y_start + y_size) {
|
||||
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
|
||||
|
||||
let mut source1 = simd_utils::loadu_si128(s_row, xx); // top line
|
||||
|
||||
let mut source = _mm_unpacklo_epi8(source1, _mm_setzero_si128());
|
||||
let mut pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
|
||||
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
|
||||
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
|
||||
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
source = _mm_unpackhi_epi8(source1, _mm_setzero_si128());
|
||||
pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
|
||||
sss2 = _mm_add_epi32(sss2, _mm_madd_epi16(pix, mmk));
|
||||
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
|
||||
sss3 = _mm_add_epi32(sss3, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
source1 = simd_utils::loadu_si128(s_row, xx + 4); // top line
|
||||
|
||||
source = _mm_unpacklo_epi8(source1, _mm_setzero_si128());
|
||||
pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
|
||||
sss4 = _mm_add_epi32(sss4, _mm_madd_epi16(pix, mmk));
|
||||
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
|
||||
sss5 = _mm_add_epi32(sss5, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
source = _mm_unpackhi_epi8(source1, _mm_setzero_si128());
|
||||
pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
|
||||
sss6 = _mm_add_epi32(sss6, _mm_madd_epi16(pix, mmk));
|
||||
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
|
||||
sss7 = _mm_add_epi32(sss7, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
y += 1;
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss0 = _mm_srai_epi32(sss0, $imm8);
|
||||
sss1 = _mm_srai_epi32(sss1, $imm8);
|
||||
sss2 = _mm_srai_epi32(sss2, $imm8);
|
||||
sss3 = _mm_srai_epi32(sss3, $imm8);
|
||||
sss4 = _mm_srai_epi32(sss4, $imm8);
|
||||
sss5 = _mm_srai_epi32(sss5, $imm8);
|
||||
sss6 = _mm_srai_epi32(sss6, $imm8);
|
||||
sss7 = _mm_srai_epi32(sss7, $imm8);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
|
||||
sss0 = _mm_packs_epi32(sss0, sss1);
|
||||
sss2 = _mm_packs_epi32(sss2, sss3);
|
||||
sss0 = _mm_packus_epi16(sss0, sss2);
|
||||
let dst_ptr = dst_row.get_unchecked_mut(xx..).as_mut_ptr() as *mut __m128i;
|
||||
_mm_storeu_si128(dst_ptr, sss0);
|
||||
sss4 = _mm_packs_epi32(sss4, sss5);
|
||||
sss6 = _mm_packs_epi32(sss6, sss7);
|
||||
sss4 = _mm_packus_epi16(sss4, sss6);
|
||||
let dst_ptr = dst_row.get_unchecked_mut(xx + 4..).as_mut_ptr() as *mut __m128i;
|
||||
_mm_storeu_si128(dst_ptr, sss4);
|
||||
|
||||
xx += 8;
|
||||
}
|
||||
|
||||
while xx < src_width.saturating_sub(1) {
|
||||
let mut sss0 = initial; // left row
|
||||
let mut sss1 = initial; // right row
|
||||
let mut y: u32 = 0;
|
||||
|
||||
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
|
||||
// Load two coefficients at once
|
||||
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
|
||||
|
||||
let source1 = simd_utils::loadl_epi64(s_row1, xx); // top line
|
||||
let source2 = simd_utils::loadl_epi64(s_row2, xx); // bottom line
|
||||
|
||||
let source = _mm_unpacklo_epi8(source1, source2);
|
||||
let mut pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
|
||||
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
|
||||
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
|
||||
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
y += 2;
|
||||
}
|
||||
|
||||
for s_row1 in src_img.iter_rows(y_start + y, y_start + y_size) {
|
||||
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
|
||||
|
||||
let source1 = simd_utils::loadl_epi64(s_row1, xx); // top line
|
||||
|
||||
let source = _mm_unpacklo_epi8(source1, _mm_setzero_si128());
|
||||
let mut pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
|
||||
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
|
||||
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
|
||||
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
y += 1;
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss0 = _mm_srai_epi32(sss0, $imm8);
|
||||
sss1 = _mm_srai_epi32(sss1, $imm8);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
|
||||
sss0 = _mm_packs_epi32(sss0, sss1);
|
||||
sss0 = _mm_packus_epi16(sss0, sss0);
|
||||
let dst_ptr = dst_row.get_unchecked_mut(xx..).as_mut_ptr() as *mut __m128i;
|
||||
_mm_storel_epi64(dst_ptr, sss0);
|
||||
|
||||
//
|
||||
xx += 2;
|
||||
}
|
||||
|
||||
while xx < src_width {
|
||||
let mut sss = initial;
|
||||
let mut y: u32 = 0;
|
||||
|
||||
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
|
||||
// Load two coefficients at once
|
||||
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
|
||||
|
||||
let source1 = simd_utils::mm_cvtsi32_si128(s_row1, xx); // top line
|
||||
let source2 = simd_utils::mm_cvtsi32_si128(s_row2, xx); // bottom line
|
||||
|
||||
let source = _mm_unpacklo_epi8(source1, source2);
|
||||
let pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
|
||||
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
y += 2;
|
||||
}
|
||||
|
||||
for s_row in src_img.iter_rows(y_start + y, y_start + y_size) {
|
||||
let pix = simd_utils::mm_cvtepu8_epi32(s_row, xx);
|
||||
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
|
||||
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
y += 1;
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss = _mm_srai_epi32(sss, $imm8);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
|
||||
sss = _mm_packs_epi32(sss, sss);
|
||||
*dst_row.get_unchecked_mut(xx) =
|
||||
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss, sss)));
|
||||
|
||||
xx += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl Convolution for Sse4 {
|
||||
#[inline]
|
||||
fn horiz_convolution(
|
||||
&self,
|
||||
src_image: &SrcImageView,
|
||||
dst_image: &mut DstImageView,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let mut normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coeffs_i16 = normalizer_guard.normalized();
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
self.horiz_convolution_8u4x(
|
||||
src_rows,
|
||||
dst_rows,
|
||||
coeffs_i16,
|
||||
window_size,
|
||||
&bounds_per_pixel,
|
||||
precision,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
let mut yy = dst_height - dst_height % 4;
|
||||
while yy < dst_height {
|
||||
unsafe {
|
||||
self.horiz_convolution_8u(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
coeffs_i16,
|
||||
window_size,
|
||||
&bounds_per_pixel,
|
||||
precision,
|
||||
);
|
||||
}
|
||||
yy += 1;
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
fn vert_convolution(
|
||||
&self,
|
||||
src_image: &SrcImageView,
|
||||
dst_image: &mut DstImageView,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let mut normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coeffs_i16 = normalizer_guard.normalized();
|
||||
let coeffs_chunks = coeffs_i16.chunks(window_size);
|
||||
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for ((&bound, k), dst_row) in bounds.iter().zip(coeffs_chunks).zip(dst_rows) {
|
||||
unsafe {
|
||||
self.vert_convolution_8u(src_image, dst_row, k, bound, precision);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,15 @@
|
||||
use thiserror::Error;
|
||||
|
||||
#[derive(Error, Debug, Clone, Copy)]
|
||||
pub enum ImageError {
|
||||
#[error("Buffer size don't corresponds to image dimensions")]
|
||||
InvalidBufferSize,
|
||||
}
|
||||
|
||||
#[derive(Error, Debug, Clone, Copy)]
|
||||
pub enum CropBoxError {
|
||||
#[error("Position of the crop box is out of the image boundaries")]
|
||||
PositionIsOutOfImageBoundaries,
|
||||
#[error("Size of the crop box is out of the image boundaries")]
|
||||
SizeIsOutOfImageBoundaries,
|
||||
}
|
||||
@@ -0,0 +1,80 @@
|
||||
use std::num::NonZeroU32;
|
||||
|
||||
use crate::{DstImageView, ImageError, PixelType, SrcImageView};
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct ImageData<T: AsRef<[u8]>> {
|
||||
width: NonZeroU32,
|
||||
height: NonZeroU32,
|
||||
pixels: T,
|
||||
pixel_type: PixelType,
|
||||
}
|
||||
|
||||
impl<T: AsRef<[u8]>> ImageData<T> {
|
||||
pub fn new(
|
||||
width: NonZeroU32,
|
||||
height: NonZeroU32,
|
||||
pixels: T,
|
||||
pixel_type: PixelType,
|
||||
) -> Result<Self, ImageError> {
|
||||
let size = (width.get() * height.get()) as usize * 4;
|
||||
if pixels.as_ref().len() != size {
|
||||
return Err(ImageError::InvalidBufferSize);
|
||||
}
|
||||
Ok(Self {
|
||||
width,
|
||||
height,
|
||||
pixels,
|
||||
pixel_type,
|
||||
})
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub fn pixel_type(&self) -> PixelType {
|
||||
self.pixel_type
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub fn width(&self) -> NonZeroU32 {
|
||||
self.width
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub fn height(&self) -> NonZeroU32 {
|
||||
self.height
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub fn get_buffer(&self) -> &[u8] {
|
||||
self.pixels.as_ref()
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub fn src_view(&self) -> SrcImageView {
|
||||
let pixels = unsafe { self.pixels.as_ref().align_to::<u32>().1 };
|
||||
let rows = pixels.chunks(self.width.get() as usize).collect();
|
||||
SrcImageView::new(self.width, self.height, rows, self.pixel_type).unwrap()
|
||||
}
|
||||
}
|
||||
|
||||
impl<T: AsRef<[u8]> + AsMut<[u8]>> ImageData<T> {
|
||||
#[inline(always)]
|
||||
pub fn dst_view(&mut self) -> DstImageView {
|
||||
let pixels = unsafe { self.pixels.as_mut().align_to_mut::<u32>().1 };
|
||||
let rows = pixels.chunks_mut(self.width.get() as usize).collect();
|
||||
DstImageView::new(self.width, self.height, rows, self.pixel_type).unwrap()
|
||||
}
|
||||
}
|
||||
|
||||
impl ImageData<Vec<u8>> {
|
||||
pub fn new_owned(width: NonZeroU32, height: NonZeroU32, pixel_type: PixelType) -> Self {
|
||||
let size = (width.get() * height.get()) as usize * 4;
|
||||
let pixels = vec![0; size];
|
||||
Self {
|
||||
width,
|
||||
height,
|
||||
pixels,
|
||||
pixel_type,
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,309 @@
|
||||
use std::mem::transmute;
|
||||
use std::num::NonZeroU32;
|
||||
use std::slice;
|
||||
|
||||
use crate::errors::{CropBoxError, ImageError};
|
||||
|
||||
pub type TwoRows<'a> = (&'a [u32], &'a [u32]);
|
||||
pub type FourRows<'a> = (&'a [u32], &'a [u32], &'a [u32], &'a [u32]);
|
||||
pub type RowMut<'a, 'b> = &'b mut &'a mut [u32];
|
||||
pub type FourRowsMut<'a, 'b> = (
|
||||
&'b mut &'a mut [u32],
|
||||
&'b mut &'a mut [u32],
|
||||
&'b mut &'a mut [u32],
|
||||
&'b mut &'a mut [u32],
|
||||
);
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq)]
|
||||
pub enum PixelType {
|
||||
U8x4,
|
||||
I32,
|
||||
F32,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
pub struct CropBox {
|
||||
pub left: u32,
|
||||
pub top: u32,
|
||||
pub width: NonZeroU32,
|
||||
pub height: NonZeroU32,
|
||||
}
|
||||
|
||||
/// An immutable view of image data used by resizer as source image.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct SrcImageView<'a> {
|
||||
width: NonZeroU32,
|
||||
height: NonZeroU32,
|
||||
crop_box: CropBox,
|
||||
rows: Vec<&'a [u32]>,
|
||||
pixel_type: PixelType,
|
||||
}
|
||||
|
||||
/// An mutable view of image data used by resizer as destination image.
|
||||
#[derive(Debug)]
|
||||
pub struct DstImageView<'a> {
|
||||
width: NonZeroU32,
|
||||
height: NonZeroU32,
|
||||
rows: Vec<&'a mut [u32]>,
|
||||
pixel_type: PixelType,
|
||||
}
|
||||
|
||||
impl<'a> SrcImageView<'a> {
|
||||
#[inline(always)]
|
||||
pub fn new(
|
||||
width: NonZeroU32,
|
||||
height: NonZeroU32,
|
||||
rows: Vec<&'a [u32]>,
|
||||
pixel_type: PixelType,
|
||||
) -> Result<Self, ImageError> {
|
||||
if rows.len() != height.get() as usize {
|
||||
return Err(ImageError::InvalidBufferSize);
|
||||
}
|
||||
let row_size = width.get() as usize;
|
||||
if rows.iter().any(|row| row.len() != row_size) {
|
||||
return Err(ImageError::InvalidBufferSize);
|
||||
}
|
||||
Ok(Self {
|
||||
width,
|
||||
height,
|
||||
crop_box: CropBox {
|
||||
left: 0,
|
||||
top: 0,
|
||||
width,
|
||||
height,
|
||||
},
|
||||
rows,
|
||||
pixel_type,
|
||||
})
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub fn pixel_type(&self) -> PixelType {
|
||||
self.pixel_type
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub fn width(&self) -> NonZeroU32 {
|
||||
self.width
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub fn height(&self) -> NonZeroU32 {
|
||||
self.height
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub fn crop_box(&self) -> CropBox {
|
||||
self.crop_box
|
||||
}
|
||||
|
||||
pub fn set_crop_box(&mut self, crop_box: CropBox) -> Result<(), CropBoxError> {
|
||||
if crop_box.left >= self.width.get() || crop_box.top >= self.height.get() {
|
||||
return Err(CropBoxError::PositionIsOutOfImageBoundaries);
|
||||
}
|
||||
let right = crop_box.left + crop_box.width.get();
|
||||
let bottom = crop_box.top + crop_box.height.get();
|
||||
if right > self.width.get() || bottom > self.height.get() {
|
||||
return Err(CropBoxError::SizeIsOutOfImageBoundaries);
|
||||
}
|
||||
self.crop_box = crop_box;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub fn get_buffer(&self) -> Vec<u8> {
|
||||
let row_size = self.width.get() as usize;
|
||||
self.rows
|
||||
.iter()
|
||||
.map(|row| unsafe { row[0..row_size].align_to::<u8>().1 })
|
||||
.flatten()
|
||||
.copied()
|
||||
.collect()
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn get_pixel_u32(&self, x: u32, y: u32) -> u32 {
|
||||
self.rows[y as usize][x as usize]
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) fn get_pixel_i32(&self, x: u32, y: u32) -> i32 {
|
||||
unsafe { transmute(self.get_pixel_u32(x, y)) }
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) fn get_pixel_f32(&self, x: u32, y: u32) -> f32 {
|
||||
f32::from_bits(self.get_pixel_u32(x, y))
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) fn iter_4_rows(
|
||||
&'a self,
|
||||
start_y: u32,
|
||||
max_y: u32,
|
||||
) -> impl Iterator<Item = FourRows<'a>> {
|
||||
let start_y = start_y as usize;
|
||||
let max_y = max_y.min(self.height.get()) as usize;
|
||||
let rows = self.rows.get(start_y..max_y).unwrap_or_else(|| &[]);
|
||||
rows.chunks_exact(4).map(|rows| match *rows {
|
||||
[r0, r1, r2, r3] => (r0, r1, r2, r3),
|
||||
_ => unreachable!(),
|
||||
})
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) fn iter_2_rows(
|
||||
&'a self,
|
||||
start_y: u32,
|
||||
max_y: u32,
|
||||
) -> impl Iterator<Item = TwoRows<'a>> {
|
||||
let start_y = start_y as usize;
|
||||
let max_y = max_y.min(self.height.get()) as usize;
|
||||
let rows = self.rows.get(start_y..max_y).unwrap_or_else(|| &[]);
|
||||
rows.chunks_exact(2).map(|rows| match *rows {
|
||||
[r0, r1] => (r0, r1),
|
||||
_ => unreachable!(),
|
||||
})
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) fn iter_rows(&'a self, start_y: u32, max_y: u32) -> impl Iterator<Item = &'a [u32]> {
|
||||
let start_y = start_y as usize;
|
||||
let max_y = max_y.min(self.height.get()) as usize;
|
||||
let rows = self.rows.get(start_y..max_y).unwrap_or_else(|| &[]);
|
||||
rows.iter().copied()
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) fn iter_horiz(&self, x: u32, y: u32) -> &[u32] {
|
||||
if let Some(&row) = self.rows.get(y as usize) {
|
||||
let start_pos = x as usize;
|
||||
if let Some(res) = row.get(start_pos..) {
|
||||
return res;
|
||||
}
|
||||
}
|
||||
&[]
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn iter_horiz_i32(&self, x: u32, y: u32) -> &[i32] {
|
||||
let row = self.iter_horiz(x, y);
|
||||
let ptr = row.as_ptr();
|
||||
unsafe { slice::from_raw_parts(ptr as *const i32, row.len()) }
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn iter_horiz_f32(&self, x: u32, y: u32) -> &[f32] {
|
||||
let row = self.iter_horiz(x, y);
|
||||
let ptr = row.as_ptr();
|
||||
unsafe { slice::from_raw_parts(ptr as *const f32, row.len()) }
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) fn get_row(&self, y: u32) -> Option<&[u32]> {
|
||||
self.rows.get(y as usize).copied()
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) fn iter_rows_with_step(
|
||||
&self,
|
||||
mut y: f64,
|
||||
step: f64,
|
||||
max_count: usize,
|
||||
) -> impl Iterator<Item = &[u32]> {
|
||||
let steps = (self.height.get() as f64 - y) / step;
|
||||
let steps = (steps.max(0.).ceil() as usize).min(max_count);
|
||||
(0..steps).map(move |_| {
|
||||
// Safety of value of y guaranteed by calculation of steps count
|
||||
let row = unsafe { *self.rows.get_unchecked(y as usize) };
|
||||
y += step;
|
||||
row
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
impl<'a> DstImageView<'a> {
|
||||
#[inline(always)]
|
||||
pub fn new(
|
||||
width: NonZeroU32,
|
||||
height: NonZeroU32,
|
||||
rows: Vec<&'a mut [u32]>,
|
||||
pixel_type: PixelType,
|
||||
) -> Result<Self, ImageError> {
|
||||
if rows.len() != height.get() as usize {
|
||||
return Err(ImageError::InvalidBufferSize);
|
||||
}
|
||||
let row_size = width.get() as usize;
|
||||
if rows.iter().any(|row| row.len() != row_size) {
|
||||
return Err(ImageError::InvalidBufferSize);
|
||||
}
|
||||
Ok(Self {
|
||||
width,
|
||||
height,
|
||||
rows,
|
||||
pixel_type,
|
||||
})
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub fn pixel_type(&self) -> PixelType {
|
||||
self.pixel_type
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub fn width(&self) -> NonZeroU32 {
|
||||
self.width
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub fn height(&self) -> NonZeroU32 {
|
||||
self.height
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) fn iter_rows_mut(&mut self) -> slice::IterMut<&'a mut [u32]> {
|
||||
self.rows.iter_mut()
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) fn iter_4_rows_mut(&mut self) -> impl Iterator<Item = FourRowsMut<'a, '_>> {
|
||||
self.rows.chunks_exact_mut(4).map(|rows| match rows {
|
||||
[a, b, c, d] => (a, b, c, d),
|
||||
_ => unreachable!(),
|
||||
})
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) fn get_row_mut(&mut self, y: u32) -> Option<RowMut<'a, '_>> {
|
||||
self.rows.get_mut(y as usize)
|
||||
}
|
||||
}
|
||||
|
||||
// pub struct FourRowsIterator<'a> {
|
||||
// chunks: slice::ChunksExact<'a, &'a [u32]>,
|
||||
// }
|
||||
//
|
||||
// impl<'a> FourRowsIterator<'a> {
|
||||
// #[inline(always)]
|
||||
// fn new(image: &'a SrcImageView<'a>, start_y: u32, max_y: u32) -> Self {
|
||||
// let start_y = start_y as usize;
|
||||
// let max_y = max_y as usize;
|
||||
// let max_y = max_y.min(image.height.get() as usize);
|
||||
// let rows = unsafe { image.rows.get_unchecked(start_y..max_y) };
|
||||
// Self {
|
||||
// chunks: rows.chunks_exact(4),
|
||||
// }
|
||||
// }
|
||||
// }
|
||||
//
|
||||
// impl<'a> Iterator for FourRowsIterator<'a> {
|
||||
// type Item = FourRows<'a>;
|
||||
//
|
||||
// #[inline(always)]
|
||||
// fn next(&mut self) -> Option<Self::Item> {
|
||||
// match self.chunks.next() {
|
||||
// Some(&[r0, r1, r2, r3]) => Some((r0, r1, r2, r3)),
|
||||
// _ => None,
|
||||
// }
|
||||
// }
|
||||
// }
|
||||
+15
@@ -0,0 +1,15 @@
|
||||
pub use alpha::{MulDiv, MulDivImageError, MulDivImagesError};
|
||||
pub use convolution::FilterType;
|
||||
pub use errors::{CropBoxError, ImageError};
|
||||
pub use image_data::ImageData;
|
||||
pub use image_view::{CropBox, DstImageView, PixelType, SrcImageView};
|
||||
pub use resizer::{CpuExtensions, ResizeAlg, Resizer};
|
||||
|
||||
mod alpha;
|
||||
mod convolution;
|
||||
mod errors;
|
||||
mod image_data;
|
||||
mod image_view;
|
||||
mod optimisations;
|
||||
mod resizer;
|
||||
mod simd_utils;
|
||||
@@ -0,0 +1,130 @@
|
||||
use std::slice;
|
||||
|
||||
/* Handles values form -640 to 639. */
|
||||
const CLIP8_LOOKUPS: [u8; 1280] = [
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25,
|
||||
26, 27, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 47, 48, 49,
|
||||
50, 51, 52, 53, 54, 55, 56, 57, 58, 59, 60, 61, 62, 63, 64, 65, 66, 67, 68, 69, 70, 71, 72, 73,
|
||||
74, 75, 76, 77, 78, 79, 80, 81, 82, 83, 84, 85, 86, 87, 88, 89, 90, 91, 92, 93, 94, 95, 96, 97,
|
||||
98, 99, 100, 101, 102, 103, 104, 105, 106, 107, 108, 109, 110, 111, 112, 113, 114, 115, 116,
|
||||
117, 118, 119, 120, 121, 122, 123, 124, 125, 126, 127, 128, 129, 130, 131, 132, 133, 134, 135,
|
||||
136, 137, 138, 139, 140, 141, 142, 143, 144, 145, 146, 147, 148, 149, 150, 151, 152, 153, 154,
|
||||
155, 156, 157, 158, 159, 160, 161, 162, 163, 164, 165, 166, 167, 168, 169, 170, 171, 172, 173,
|
||||
174, 175, 176, 177, 178, 179, 180, 181, 182, 183, 184, 185, 186, 187, 188, 189, 190, 191, 192,
|
||||
193, 194, 195, 196, 197, 198, 199, 200, 201, 202, 203, 204, 205, 206, 207, 208, 209, 210, 211,
|
||||
212, 213, 214, 215, 216, 217, 218, 219, 220, 221, 222, 223, 224, 225, 226, 227, 228, 229, 230,
|
||||
231, 232, 233, 234, 235, 236, 237, 238, 239, 240, 241, 242, 243, 244, 245, 246, 247, 248, 249,
|
||||
250, 251, 252, 253, 254, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
];
|
||||
|
||||
// 8 bits for result. Filter can have negative areas.
|
||||
// In one cases the sum of the coefficients will be negative,
|
||||
// in the other it will be more than 1.0. That is why we need
|
||||
// two extra bits for overflow and i32 type.
|
||||
const PRECISION_BITS: u8 = 32 - 8 - 2;
|
||||
/* We use signed INT16 type to store coefficients. */
|
||||
const MAX_COEFS_PRECISION: u8 = 16 - 1;
|
||||
|
||||
/// The function must be used with the ``v`` and ``precision`` values
|
||||
/// such that the expression
|
||||
/// ```
|
||||
/// v >> precision
|
||||
/// ```
|
||||
/// produces a result in the range ``[-512, 511]``.
|
||||
#[inline(always)]
|
||||
pub unsafe fn clip8(v: i32, precision: u8) -> u8 {
|
||||
let index = (640 + (v >> precision)) as usize;
|
||||
// index must be in range [(640-512)..(640+511)]
|
||||
*CLIP8_LOOKUPS.get_unchecked(index)
|
||||
}
|
||||
|
||||
/// Converts `Vec<f64>` into `&[i16]` without additional memory allocations.
|
||||
/// The memory buffer from `Vec<f64>` uses as `[i16]` .
|
||||
pub struct NormalizerGuard {
|
||||
values: Vec<f64>,
|
||||
precision: u8,
|
||||
}
|
||||
|
||||
impl NormalizerGuard {
|
||||
#[inline]
|
||||
pub fn new(mut values: Vec<f64>) -> Self {
|
||||
let max_weight = values
|
||||
.iter()
|
||||
.max_by(|&x, &y| x.partial_cmp(&y).unwrap())
|
||||
.unwrap_or(&0.0)
|
||||
.to_owned();
|
||||
|
||||
let mut precision = 0u8;
|
||||
for cur_precision in 0..PRECISION_BITS {
|
||||
precision = cur_precision;
|
||||
let next_value: i32 = (max_weight * (1 << (precision + 1)) as f64).round() as i32;
|
||||
// The next value will be outside of the range, so just stop
|
||||
if next_value >= (1 << MAX_COEFS_PRECISION) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
let len = values.len();
|
||||
let ptr = values.as_mut_ptr();
|
||||
// Size of `[i16]` always will be not greater than `[f64]` with same number of items
|
||||
let values_i16 = unsafe { slice::from_raw_parts_mut(ptr as *mut i16, len) };
|
||||
|
||||
let scale = (1 << precision) as f64;
|
||||
for (&src, dst) in values.iter().zip(values_i16.iter_mut()) {
|
||||
*dst = (src * scale).round() as i16;
|
||||
}
|
||||
Self { values, precision }
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn normalized(&mut self) -> &[i16] {
|
||||
let len = self.values.len();
|
||||
let ptr = self.values.as_mut_ptr();
|
||||
unsafe { slice::from_raw_parts_mut(ptr as *mut i16, len) }
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn precision(&self) -> u8 {
|
||||
self.precision
|
||||
}
|
||||
}
|
||||
+292
@@ -0,0 +1,292 @@
|
||||
use std::num::NonZeroU32;
|
||||
|
||||
use crate::convolution::{self, Convolution, FilterType};
|
||||
use crate::image_data::ImageData;
|
||||
use crate::image_view::{DstImageView, PixelType, SrcImageView};
|
||||
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
pub enum CpuExtensions {
|
||||
None,
|
||||
Sse2,
|
||||
Sse4_1,
|
||||
Avx2,
|
||||
}
|
||||
|
||||
impl Default for CpuExtensions {
|
||||
fn default() -> Self {
|
||||
if is_x86_feature_detected!("avx2") {
|
||||
Self::Avx2
|
||||
} else if is_x86_feature_detected!("sse4.1") {
|
||||
Self::Sse4_1
|
||||
} else if is_x86_feature_detected!("sse2") {
|
||||
Self::Sse2
|
||||
} else {
|
||||
Self::None
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl CpuExtensions {
|
||||
#[inline]
|
||||
fn get_resampler(&self, pixel_type: PixelType) -> &dyn Convolution {
|
||||
match pixel_type {
|
||||
PixelType::U8x4 => match self {
|
||||
Self::Sse4_1 => &convolution::Sse4,
|
||||
Self::Avx2 => &convolution::Avx2,
|
||||
_ => &convolution::NativeU8x4,
|
||||
},
|
||||
PixelType::I32 => &convolution::NativeI32,
|
||||
PixelType::F32 => &convolution::NativeF32,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
#[non_exhaustive]
|
||||
pub enum ResizeAlg {
|
||||
Nearest,
|
||||
Convolution(FilterType),
|
||||
SuperSampling(FilterType, u8),
|
||||
}
|
||||
|
||||
impl Default for ResizeAlg {
|
||||
fn default() -> Self {
|
||||
Self::Convolution(FilterType::Lanczos3)
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Default, Debug, Clone)]
|
||||
pub struct Resizer {
|
||||
pub algorithm: ResizeAlg,
|
||||
cpu_extensions: CpuExtensions,
|
||||
convolution_buffer: Vec<u8>,
|
||||
super_sampling_buffer: Vec<u8>,
|
||||
}
|
||||
|
||||
impl Resizer {
|
||||
pub fn new(algorithm: ResizeAlg) -> Self {
|
||||
Self {
|
||||
algorithm,
|
||||
..Default::default()
|
||||
}
|
||||
}
|
||||
|
||||
pub fn resize(&mut self, src_image: &SrcImageView, dst_image: &mut DstImageView) {
|
||||
match self.algorithm {
|
||||
ResizeAlg::Nearest => resample_nearest(&src_image, dst_image),
|
||||
ResizeAlg::Convolution(filter_type) => {
|
||||
let convolution_buffer = &mut self.convolution_buffer;
|
||||
resample_convolution(
|
||||
src_image,
|
||||
dst_image,
|
||||
filter_type,
|
||||
self.cpu_extensions,
|
||||
convolution_buffer,
|
||||
)
|
||||
}
|
||||
ResizeAlg::SuperSampling(filter_type, multiplicity) => {
|
||||
let convolution_buffer = &mut self.convolution_buffer;
|
||||
let super_sampling_buffer = &mut self.super_sampling_buffer;
|
||||
resample_super_sampling(
|
||||
src_image,
|
||||
dst_image,
|
||||
filter_type,
|
||||
multiplicity,
|
||||
self.cpu_extensions,
|
||||
super_sampling_buffer,
|
||||
convolution_buffer,
|
||||
)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Returns the size of internal buffers used to store the results of
|
||||
/// intermediate resizing steps.
|
||||
pub fn size_of_internal_buffers(&self) -> usize {
|
||||
self.convolution_buffer.len() + self.super_sampling_buffer.len()
|
||||
}
|
||||
|
||||
/// Deallocates the internal buffers used to store the results of
|
||||
/// intermediate resizing steps.
|
||||
pub fn reset_internal_buffers(&mut self) {
|
||||
if self.convolution_buffer.capacity() > 0 {
|
||||
self.convolution_buffer = Vec::new();
|
||||
}
|
||||
if self.super_sampling_buffer.capacity() > 0 {
|
||||
self.super_sampling_buffer = Vec::new();
|
||||
}
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub fn cpu_extensions(&self) -> CpuExtensions {
|
||||
self.cpu_extensions
|
||||
}
|
||||
|
||||
/// # Safety
|
||||
/// This is unsafe because this method allows you to set a CPU-extensions
|
||||
/// that is not actually supported by your CPU.
|
||||
pub unsafe fn set_cpu_extensions(&mut self, extensions: CpuExtensions) {
|
||||
self.cpu_extensions = extensions;
|
||||
}
|
||||
}
|
||||
|
||||
fn get_temp_image_from_buffer(
|
||||
buffer: &mut Vec<u8>,
|
||||
width: NonZeroU32,
|
||||
height: NonZeroU32,
|
||||
pixel_type: PixelType,
|
||||
) -> ImageData<&mut [u8]> {
|
||||
let buf_size = (width.get() * height.get()) as usize * 4;
|
||||
if buffer.len() < buf_size {
|
||||
buffer.resize(buf_size, 0);
|
||||
}
|
||||
let pixels = &mut buffer[0..buf_size];
|
||||
ImageData::new(width, height, pixels, pixel_type).unwrap()
|
||||
}
|
||||
|
||||
fn resample_nearest(src_image: &SrcImageView, dst_image: &mut DstImageView) {
|
||||
let crop_box = src_image.crop_box();
|
||||
let dst_width = dst_image.width().get();
|
||||
let x_scale = crop_box.width.get() as f64 / dst_width as f64;
|
||||
let y_scale = crop_box.height.get() as f64 / dst_image.height().get() as f64;
|
||||
|
||||
// Pretabulate horizontal pixel positions
|
||||
let x_in_start = crop_box.left as f64 + x_scale * 0.5;
|
||||
let max_src_x = src_image.width().get() as usize;
|
||||
let x_in_tab: Vec<usize> = (0..dst_width)
|
||||
.map(|x| ((x_in_start + x_scale * x as f64) as usize).min(max_src_x))
|
||||
.collect();
|
||||
|
||||
let y_in_start = crop_box.top as f64 + y_scale * 0.5;
|
||||
|
||||
let src_rows =
|
||||
src_image.iter_rows_with_step(y_in_start, y_scale, dst_image.height().get() as usize);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for (out_row, in_row) in dst_rows.zip(src_rows) {
|
||||
for (&x_in, out_pixel) in x_in_tab.iter().zip(out_row.iter_mut()) {
|
||||
// Safety of value of x_in guaranteed by algorithm of creating of x_in_tab
|
||||
*out_pixel = unsafe { *in_row.get_unchecked(x_in) };
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn resample_convolution(
|
||||
src_image: &SrcImageView,
|
||||
dst_image: &mut DstImageView,
|
||||
filter_type: FilterType,
|
||||
cpu_extensions: CpuExtensions,
|
||||
temp_buffer: &mut Vec<u8>,
|
||||
) {
|
||||
let crop_box = src_image.crop_box();
|
||||
let dst_width = dst_image.width();
|
||||
let dst_height = dst_image.height();
|
||||
let (filter_fn, filter_support) = convolution::get_filter_func(filter_type);
|
||||
let resampler = cpu_extensions.get_resampler(src_image.pixel_type());
|
||||
|
||||
let need_horizontal = dst_width != src_image.width() || crop_box.width != src_image.width();
|
||||
let need_vertical = dst_height != src_image.height() || crop_box.height != src_image.height();
|
||||
|
||||
let mut vert_coeffs = convolution::precompute_coefficients(
|
||||
src_image.height(),
|
||||
crop_box.top as f64,
|
||||
crop_box.top as f64 + crop_box.height.get() as f64,
|
||||
dst_height,
|
||||
filter_fn,
|
||||
filter_support,
|
||||
);
|
||||
|
||||
if need_horizontal {
|
||||
let horiz_coeffs = convolution::precompute_coefficients(
|
||||
src_image.width(),
|
||||
crop_box.left as f64,
|
||||
crop_box.left as f64 + crop_box.width.get() as f64,
|
||||
dst_width,
|
||||
filter_fn,
|
||||
filter_support,
|
||||
);
|
||||
|
||||
// First used row in the source image
|
||||
let y_first = vert_coeffs.bounds[0].start;
|
||||
|
||||
if need_vertical {
|
||||
// Shift bounds for vertical pass
|
||||
vert_coeffs
|
||||
.bounds
|
||||
.iter_mut()
|
||||
.for_each(|b| b.start -= y_first);
|
||||
|
||||
// Last used row in the source image
|
||||
let last_y_bound = vert_coeffs.bounds.last().unwrap();
|
||||
let y_last = last_y_bound.start + last_y_bound.size;
|
||||
|
||||
let temp_height = NonZeroU32::new(y_last - y_first).unwrap();
|
||||
let mut temp_image = get_temp_image_from_buffer(
|
||||
temp_buffer,
|
||||
dst_width,
|
||||
temp_height,
|
||||
src_image.pixel_type(),
|
||||
);
|
||||
resampler.horiz_convolution(
|
||||
src_image,
|
||||
&mut temp_image.dst_view(),
|
||||
y_first,
|
||||
horiz_coeffs,
|
||||
);
|
||||
resampler.vert_convolution(&temp_image.src_view(), dst_image, vert_coeffs);
|
||||
} else {
|
||||
resampler.horiz_convolution(src_image, dst_image, y_first, horiz_coeffs);
|
||||
}
|
||||
} else if need_vertical {
|
||||
resampler.vert_convolution(src_image, dst_image, vert_coeffs);
|
||||
}
|
||||
}
|
||||
|
||||
fn resample_super_sampling(
|
||||
src_image: &SrcImageView,
|
||||
dst_image: &mut DstImageView,
|
||||
filter_type: FilterType,
|
||||
multiplicity: u8,
|
||||
cpu_extensions: CpuExtensions,
|
||||
temp_buffer: &mut Vec<u8>,
|
||||
convolution_temp_buffer: &mut Vec<u8>,
|
||||
) {
|
||||
let crop_box = src_image.crop_box();
|
||||
let dst_width = dst_image.width().get();
|
||||
let dst_height = dst_image.height().get();
|
||||
let width_scale = crop_box.width.get() as f32 / dst_width as f32;
|
||||
let height_scale = crop_box.height.get() as f32 / dst_height as f32;
|
||||
// It makes sense to resize the image in two steps only if the image
|
||||
// size is greater than the required size by multiplicity times.
|
||||
let factor = width_scale.min(height_scale) / multiplicity as f32;
|
||||
if factor > 1.2 {
|
||||
// First step is resizing the source image by fastest algorithm.
|
||||
// The temporary image will be about ``multiplicity`` times larger
|
||||
// than required.
|
||||
let tmp_width =
|
||||
NonZeroU32::new((crop_box.width.get() as f32 / factor).round() as u32).unwrap();
|
||||
let tmp_height =
|
||||
NonZeroU32::new((crop_box.height.get() as f32 / factor).round() as u32).unwrap();
|
||||
|
||||
let mut tmp_img =
|
||||
get_temp_image_from_buffer(temp_buffer, tmp_width, tmp_height, src_image.pixel_type());
|
||||
resample_nearest(src_image, &mut tmp_img.dst_view());
|
||||
// Second step is resizing the temporary image with a convolution.
|
||||
resample_convolution(
|
||||
&tmp_img.src_view(),
|
||||
dst_image,
|
||||
filter_type,
|
||||
cpu_extensions,
|
||||
convolution_temp_buffer,
|
||||
);
|
||||
} else {
|
||||
// There is no point in doing the resizing in two steps.
|
||||
// We immediately resize the original image with a convolution.
|
||||
resample_convolution(
|
||||
src_image,
|
||||
dst_image,
|
||||
filter_type,
|
||||
cpu_extensions,
|
||||
convolution_temp_buffer,
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,39 @@
|
||||
use std::arch::x86_64::*;
|
||||
use std::intrinsics::transmute;
|
||||
|
||||
#[inline(always)]
|
||||
pub unsafe fn loadu_si128<T>(buf: &[T], index: usize) -> __m128i {
|
||||
_mm_loadu_si128(buf.get_unchecked(index..).as_ptr() as *const __m128i)
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub unsafe fn loadu_si256<T>(buf: &[T], index: usize) -> __m256i {
|
||||
_mm256_loadu_si256(buf.get_unchecked(index..).as_ptr() as *const __m256i)
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub unsafe fn loadl_epi64<T>(buf: &[T], index: usize) -> __m128i {
|
||||
_mm_loadl_epi64(buf.get_unchecked(index..).as_ptr() as *const __m128i)
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub unsafe fn mm_cvtepu8_epi32(buf: &[u32], index: usize) -> __m128i {
|
||||
let v: i32 = transmute(*buf.get_unchecked(index));
|
||||
_mm_cvtepu8_epi32(_mm_cvtsi32_si128(v))
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub unsafe fn mm_cvtsi32_si128(buf: &[u32], index: usize) -> __m128i {
|
||||
let v: i32 = transmute(*buf.get_unchecked(index));
|
||||
_mm_cvtsi32_si128(v)
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub unsafe fn ptr_i16_to_set1_epi32(buf: &[i16], index: usize) -> __m128i {
|
||||
_mm_set1_epi32(*(buf.get_unchecked(index..).as_ptr() as *const i32))
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub unsafe fn ptr_i16_to_256set1_epi32(buf: &[i16], index: usize) -> __m256i {
|
||||
_mm256_set1_epi32(*(buf.get_unchecked(index..).as_ptr() as *const i32))
|
||||
}
|
||||
@@ -0,0 +1,179 @@
|
||||
use fast_image_resize::{CpuExtensions, DstImageView, ImageData, MulDiv, PixelType, SrcImageView};
|
||||
use std::convert::TryInto;
|
||||
use std::num::NonZeroU32;
|
||||
|
||||
const fn p(r: u8, g: u8, b: u8, a: u8) -> u32 {
|
||||
u32::from_le_bytes([r, g, b, a])
|
||||
}
|
||||
|
||||
// Multiplies by alpha
|
||||
|
||||
fn multiply_alpha_test(cpu_extensions: CpuExtensions) {
|
||||
let width: u32 = 8 + 8 + 7;
|
||||
let height: u32 = 3;
|
||||
|
||||
let src_pixels = [p(255, 128, 0, 128), p(255, 128, 0, 255), p(255, 128, 0, 0)];
|
||||
let res_pixels = [p(128, 64, 0, 128), p(255, 128, 0, 255), p(0, 0, 0, 0)];
|
||||
|
||||
let mut src_rows: [Vec<u32>; 3] = [
|
||||
vec![src_pixels[0]; width as usize],
|
||||
vec![src_pixels[1]; width as usize],
|
||||
vec![src_pixels[2]; width as usize],
|
||||
];
|
||||
|
||||
let rows: Vec<&[u32]> = src_rows.iter().map(|r| r.as_ref()).collect();
|
||||
let src_image_view = SrcImageView::new(
|
||||
NonZeroU32::new(width).unwrap(),
|
||||
NonZeroU32::new(height).unwrap(),
|
||||
rows,
|
||||
PixelType::U8x4,
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
let mut dst_image = ImageData::new_owned(
|
||||
NonZeroU32::new(width).unwrap(),
|
||||
NonZeroU32::new(height).unwrap(),
|
||||
PixelType::U8x4,
|
||||
);
|
||||
let mut dst_image_view = dst_image.dst_view();
|
||||
|
||||
let mut alpha_mul_div: MulDiv = Default::default();
|
||||
unsafe {
|
||||
alpha_mul_div.set_cpu_extensions(cpu_extensions);
|
||||
}
|
||||
|
||||
alpha_mul_div
|
||||
.multiply_alpha(&src_image_view, &mut dst_image_view)
|
||||
.unwrap();
|
||||
|
||||
let dst_pixels: Vec<u32> = dst_image
|
||||
.get_buffer()
|
||||
.chunks_exact(4)
|
||||
.map(|c| u32::from_le_bytes(c.try_into().unwrap()))
|
||||
.collect();
|
||||
let dst_rows = dst_pixels.chunks_exact(width as usize);
|
||||
for (row, &valid_pixel) in dst_rows.zip(res_pixels.iter()) {
|
||||
for &pixel in row.iter() {
|
||||
assert_eq!(pixel.to_le_bytes(), valid_pixel.to_le_bytes());
|
||||
}
|
||||
}
|
||||
|
||||
// Inplace
|
||||
let rows: Vec<&mut [u32]> = src_rows.iter_mut().map(|r| r.as_mut()).collect();
|
||||
let mut image_view = DstImageView::new(
|
||||
NonZeroU32::new(width).unwrap(),
|
||||
NonZeroU32::new(height).unwrap(),
|
||||
rows,
|
||||
PixelType::U8x4,
|
||||
)
|
||||
.unwrap();
|
||||
alpha_mul_div
|
||||
.multiply_alpha_inplace(&mut image_view)
|
||||
.unwrap();
|
||||
|
||||
for (row, &valid_pixel) in src_rows.iter().zip(res_pixels.iter()) {
|
||||
for &pixel in row.iter() {
|
||||
assert_eq!(pixel.to_le_bytes(), valid_pixel.to_le_bytes());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn multiply_alpha_avx2_test() {
|
||||
multiply_alpha_test(CpuExtensions::Avx2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn multiply_alpha_sse2_test() {
|
||||
multiply_alpha_test(CpuExtensions::Sse2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn multiply_alpha_native_test() {
|
||||
multiply_alpha_test(CpuExtensions::None);
|
||||
}
|
||||
|
||||
// Divides by alpha
|
||||
|
||||
fn divide_alpha_test(cpu_extensions: CpuExtensions) {
|
||||
let width: u32 = 8 + 8 + 7;
|
||||
let height: u32 = 3;
|
||||
|
||||
let src_pixels = [p(128, 64, 0, 128), p(255, 128, 0, 255), p(255, 128, 0, 0)];
|
||||
let res_pixels = [p(255, 127, 0, 128), p(255, 128, 0, 255), p(0, 0, 0, 0)];
|
||||
|
||||
let mut src_rows: [Vec<u32>; 3] = [
|
||||
vec![src_pixels[0]; width as usize],
|
||||
vec![src_pixels[1]; width as usize],
|
||||
vec![src_pixels[2]; width as usize],
|
||||
];
|
||||
|
||||
let rows: Vec<&[u32]> = src_rows.iter().map(|r| r.as_ref()).collect();
|
||||
let src_image_view = SrcImageView::new(
|
||||
NonZeroU32::new(width).unwrap(),
|
||||
NonZeroU32::new(height).unwrap(),
|
||||
rows,
|
||||
PixelType::U8x4,
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
let mut dst_image = ImageData::new_owned(
|
||||
NonZeroU32::new(width).unwrap(),
|
||||
NonZeroU32::new(height).unwrap(),
|
||||
PixelType::U8x4,
|
||||
);
|
||||
let mut dst_image_view = dst_image.dst_view();
|
||||
|
||||
let mut alpha_mul_div: MulDiv = Default::default();
|
||||
unsafe {
|
||||
alpha_mul_div.set_cpu_extensions(cpu_extensions);
|
||||
}
|
||||
|
||||
alpha_mul_div
|
||||
.divide_alpha(&src_image_view, &mut dst_image_view)
|
||||
.unwrap();
|
||||
|
||||
let dst_pixels: Vec<u32> = dst_image
|
||||
.get_buffer()
|
||||
.chunks_exact(4)
|
||||
.map(|c| u32::from_le_bytes(c.try_into().unwrap()))
|
||||
.collect();
|
||||
let dst_rows = dst_pixels.chunks_exact(width as usize);
|
||||
for (row, &valid_pixel) in dst_rows.zip(res_pixels.iter()) {
|
||||
for &pixel in row.iter() {
|
||||
assert_eq!(pixel.to_le_bytes(), valid_pixel.to_le_bytes());
|
||||
}
|
||||
}
|
||||
|
||||
// Inplace
|
||||
let rows: Vec<&mut [u32]> = src_rows.iter_mut().map(|r| r.as_mut()).collect();
|
||||
let mut image_view = DstImageView::new(
|
||||
NonZeroU32::new(width).unwrap(),
|
||||
NonZeroU32::new(height).unwrap(),
|
||||
rows,
|
||||
PixelType::U8x4,
|
||||
)
|
||||
.unwrap();
|
||||
alpha_mul_div.divide_alpha_inplace(&mut image_view).unwrap();
|
||||
|
||||
for (row, &valid_pixel) in src_rows.iter().zip(res_pixels.iter()) {
|
||||
for &pixel in row.iter() {
|
||||
assert_eq!(pixel.to_le_bytes(), valid_pixel.to_le_bytes());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn divide_alpha_avx2_test() {
|
||||
divide_alpha_test(CpuExtensions::Avx2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn divide_alpha_sse2_test() {
|
||||
divide_alpha_test(CpuExtensions::Sse2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn divide_alpha_native_test() {
|
||||
divide_alpha_test(CpuExtensions::None);
|
||||
}
|
||||
@@ -0,0 +1,39 @@
|
||||
use std::num::NonZeroU32;
|
||||
|
||||
use fast_image_resize::{
|
||||
CropBox, FilterType, ImageData, PixelType, ResizeAlg, Resizer, SrcImageView,
|
||||
};
|
||||
|
||||
fn resize_lanczos3(src_pixels: &[u8], width: NonZeroU32, height: NonZeroU32) -> Vec<u8> {
|
||||
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
|
||||
let src_image = ImageData::new(width, height, src_pixels, PixelType::U8x4).unwrap();
|
||||
let dst_width = NonZeroU32::new(1024).unwrap();
|
||||
let dst_height = NonZeroU32::new(768).unwrap();
|
||||
let mut dst_image = ImageData::new_owned(dst_width, dst_height, src_image.pixel_type());
|
||||
|
||||
let src_view = src_image.src_view();
|
||||
let mut dst_view = dst_image.dst_view();
|
||||
resizer.resize(&src_view, &mut dst_view);
|
||||
|
||||
dst_image.get_buffer().to_owned()
|
||||
}
|
||||
|
||||
fn resize_cropped_image(mut src_view: SrcImageView) -> ImageData<Vec<u8>> {
|
||||
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
|
||||
src_view
|
||||
.set_crop_box(CropBox {
|
||||
left: 10,
|
||||
top: 10,
|
||||
width: NonZeroU32::new(100).unwrap(),
|
||||
height: NonZeroU32::new(200).unwrap(),
|
||||
})
|
||||
.unwrap();
|
||||
let dst_width = NonZeroU32::new(1024).unwrap();
|
||||
let dst_height = NonZeroU32::new(768).unwrap();
|
||||
let mut dst_image = ImageData::new_owned(dst_width, dst_height, src_view.pixel_type());
|
||||
|
||||
let mut dst_view = dst_image.dst_view();
|
||||
resizer.resize(&src_view, &mut dst_view);
|
||||
|
||||
dst_image
|
||||
}
|
||||
@@ -0,0 +1,179 @@
|
||||
use std::fs::File;
|
||||
use std::num::NonZeroU32;
|
||||
|
||||
use image::codecs::png::PngEncoder;
|
||||
use image::io::Reader as ImageReader;
|
||||
use image::{ColorType, GenericImageView};
|
||||
|
||||
use fast_image_resize::ImageData;
|
||||
use fast_image_resize::{CpuExtensions, FilterType, PixelType, ResizeAlg, Resizer, SrcImageView};
|
||||
|
||||
fn get_source_image() -> ImageData<Vec<u8>> {
|
||||
let img = ImageReader::open("./data/nasa-4928x3279.png")
|
||||
.unwrap()
|
||||
.decode()
|
||||
.unwrap();
|
||||
let width = img.width();
|
||||
let height = img.height();
|
||||
let rgb = img.to_rgba8();
|
||||
let buf = rgb.as_raw().clone();
|
||||
ImageData::new(
|
||||
NonZeroU32::new(width).unwrap(),
|
||||
NonZeroU32::new(height).unwrap(),
|
||||
buf,
|
||||
PixelType::U8x4,
|
||||
)
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
fn get_small_source_image() -> ImageData<Vec<u8>> {
|
||||
let img = ImageReader::open("./data/nasa-852x567.png")
|
||||
.unwrap()
|
||||
.decode()
|
||||
.unwrap();
|
||||
let width = img.width();
|
||||
let height = img.height();
|
||||
let rgb = img.to_rgba8();
|
||||
let buf = rgb.as_raw().clone();
|
||||
ImageData::new(
|
||||
NonZeroU32::new(width).unwrap(),
|
||||
NonZeroU32::new(height).unwrap(),
|
||||
buf,
|
||||
PixelType::U8x4,
|
||||
)
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
fn get_new_height(src_image: &SrcImageView, new_width: u32) -> u32 {
|
||||
let scale = new_width as f32 / src_image.width().get() as f32;
|
||||
(src_image.height().get() as f32 * scale).round() as u32
|
||||
}
|
||||
|
||||
const NEW_WIDTH: u32 = 255;
|
||||
const NEW_BIG_WIDTH: u32 = 5016;
|
||||
|
||||
fn save_result(image: &SrcImageView, name: &str) {
|
||||
std::fs::create_dir_all("./data/result").unwrap();
|
||||
let mut file = File::create(format!("./data/result/{}.png", name)).unwrap();
|
||||
let encoder = PngEncoder::new(&mut file);
|
||||
encoder
|
||||
.encode(
|
||||
&image.get_buffer(),
|
||||
image.width().get(),
|
||||
image.height().get(),
|
||||
ColorType::Rgba8,
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn resample_wo_simd_lanczos3_test() {
|
||||
let image = get_source_image();
|
||||
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
|
||||
unsafe {
|
||||
resizer.set_cpu_extensions(CpuExtensions::None);
|
||||
}
|
||||
let new_height = get_new_height(&image.src_view(), NEW_WIDTH);
|
||||
let mut result = ImageData::new_owned(
|
||||
NonZeroU32::new(NEW_WIDTH).unwrap(),
|
||||
NonZeroU32::new(new_height).unwrap(),
|
||||
image.pixel_type(),
|
||||
);
|
||||
resizer.resize(&image.src_view(), &mut result.dst_view());
|
||||
save_result(&result.src_view(), "lanczos3_wo_simd");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn resample_sse4_lanczos3_test() {
|
||||
let image = get_source_image();
|
||||
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
|
||||
unsafe {
|
||||
resizer.set_cpu_extensions(CpuExtensions::Sse4_1);
|
||||
}
|
||||
let new_height = get_new_height(&image.src_view(), NEW_WIDTH);
|
||||
let mut result = ImageData::new_owned(
|
||||
NonZeroU32::new(NEW_WIDTH).unwrap(),
|
||||
NonZeroU32::new(new_height).unwrap(),
|
||||
image.pixel_type(),
|
||||
);
|
||||
resizer.resize(&image.src_view(), &mut result.dst_view());
|
||||
save_result(&result.src_view(), "lanczos3_sse4");
|
||||
}
|
||||
|
||||
fn resize_lanczos3(src_pixels: &[u8], width: NonZeroU32, height: NonZeroU32) -> Vec<u8> {
|
||||
let src_image = ImageData::new(width, height, src_pixels, PixelType::U8x4).unwrap();
|
||||
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
|
||||
let dst_width = NonZeroU32::new(1024).unwrap();
|
||||
let dst_height = NonZeroU32::new(768).unwrap();
|
||||
let mut dst_image = ImageData::new_owned(dst_width, dst_height, src_image.pixel_type());
|
||||
resizer.resize(&src_image.src_view(), &mut dst_image.dst_view());
|
||||
dst_image.get_buffer().to_owned()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn resample_avx2_lanczos3_test() {
|
||||
let image = get_source_image();
|
||||
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
|
||||
unsafe {
|
||||
resizer.set_cpu_extensions(CpuExtensions::Avx2);
|
||||
}
|
||||
let new_height = get_new_height(&image.src_view(), NEW_WIDTH);
|
||||
let mut result = ImageData::new_owned(
|
||||
NonZeroU32::new(NEW_WIDTH).unwrap(),
|
||||
NonZeroU32::new(new_height).unwrap(),
|
||||
image.pixel_type(),
|
||||
);
|
||||
resizer.resize(&image.src_view(), &mut result.dst_view());
|
||||
save_result(&result.src_view(), "lanczos3_avx2");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn resample_avx2_lanczos3_upscale_test() {
|
||||
let image = get_small_source_image();
|
||||
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
|
||||
unsafe {
|
||||
resizer.set_cpu_extensions(CpuExtensions::Avx2);
|
||||
}
|
||||
let new_height = get_new_height(&image.src_view(), NEW_BIG_WIDTH);
|
||||
let mut result = ImageData::new_owned(
|
||||
NonZeroU32::new(NEW_BIG_WIDTH).unwrap(),
|
||||
NonZeroU32::new(new_height).unwrap(),
|
||||
image.pixel_type(),
|
||||
);
|
||||
resizer.resize(&image.src_view(), &mut result.dst_view());
|
||||
save_result(&result.src_view(), "lanczos3_avx2_upscale");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn resample_nearest_test() {
|
||||
let image = get_source_image();
|
||||
let mut resizer = Resizer::new(ResizeAlg::Nearest);
|
||||
unsafe {
|
||||
resizer.set_cpu_extensions(CpuExtensions::None);
|
||||
}
|
||||
let new_height = get_new_height(&image.src_view(), NEW_WIDTH);
|
||||
let mut result = ImageData::new_owned(
|
||||
NonZeroU32::new(NEW_WIDTH).unwrap(),
|
||||
NonZeroU32::new(new_height).unwrap(),
|
||||
image.pixel_type(),
|
||||
);
|
||||
resizer.resize(&image.src_view(), &mut result.dst_view());
|
||||
save_result(&result.src_view(), "nearest_wo_simd");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn resample_super_sampling_test() {
|
||||
let image = get_source_image();
|
||||
let mut resizer = Resizer::new(ResizeAlg::SuperSampling(FilterType::Lanczos3, 2));
|
||||
unsafe {
|
||||
resizer.set_cpu_extensions(CpuExtensions::Avx2);
|
||||
}
|
||||
let new_height = get_new_height(&image.src_view(), NEW_WIDTH);
|
||||
let mut result = ImageData::new_owned(
|
||||
NonZeroU32::new(NEW_WIDTH).unwrap(),
|
||||
NonZeroU32::new(new_height).unwrap(),
|
||||
image.pixel_type(),
|
||||
);
|
||||
resizer.resize(&image.src_view(), &mut result.dst_view());
|
||||
save_result(&result.src_view(), "super_sampling_avx2");
|
||||
}
|
||||
Reference in New Issue
Block a user