Initial commit

This commit is contained in:
Kirill Kuzminykh
2021-07-03 14:29:58 +03:00
commit d8033a005b
33 changed files with 5038 additions and 0 deletions
+5
View File
@@ -0,0 +1,5 @@
/target
.*
!/.gitignore
data/result
+3
View File
@@ -0,0 +1,3 @@
## [Unreleased] - ReleaseDate
- Initial version.
Generated
+849
View File
@@ -0,0 +1,849 @@
# This file is automatically @generated by Cargo.
# It is not intended for manual editing.
version = 3
[[package]]
name = "adler"
version = "1.0.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f26201604c87b1e01bd3d98f8d5d9a8fcbb815e8cedb41ffccbeb4bf593a35fe"
[[package]]
name = "adler32"
version = "1.2.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "aae1277d39aeec15cb388266ecc24b11c80469deae6067e17a1a7aa9e5c1f234"
[[package]]
name = "ahash"
version = "0.4.7"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "739f4a8db6605981345c5654f3a85b056ce52f37a39d34da03f25bf2151ea16e"
[[package]]
name = "atty"
version = "0.2.14"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d9b39be18770d11421cdb1b9947a45dd3f37e93092cbf377614828a319d5fee8"
dependencies = [
"hermit-abi",
"libc",
"winapi",
]
[[package]]
name = "autocfg"
version = "1.0.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "cdb031dd78e28731d87d56cc8ffef4a8f36ca26c38fe2de700543e627f8a464a"
[[package]]
name = "bitflags"
version = "1.2.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "cf1de2fe8c75bc145a2f577add951f8134889b4795d47466a54a5c846d691693"
[[package]]
name = "bstr"
version = "0.2.16"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "90682c8d613ad3373e66de8c6411e0ae2ab2571e879d2efbf73558cc66f21279"
dependencies = [
"lazy_static",
"memchr",
"regex-automata",
"serde",
]
[[package]]
name = "bumpalo"
version = "3.7.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9c59e7af012c713f529e7a3ee57ce9b31ddd858d4b512923602f74608b009631"
[[package]]
name = "bytemuck"
version = "1.7.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9966d2ab714d0f785dbac0a0396251a35280aeb42413281617d0209ab4898435"
[[package]]
name = "byteorder"
version = "1.4.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "14c189c53d098945499cdfa7ecc63567cf3886b3332b312a5b4585d8d3a6a610"
[[package]]
name = "cast"
version = "0.2.7"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "4c24dab4283a142afa2fdca129b80ad2c6284e073930f964c3a1293c225ee39a"
dependencies = [
"rustc_version",
]
[[package]]
name = "cfg-if"
version = "1.0.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "baf1de4339761588bc0619e3cbc0120ee582ebb74b53b4efbf79117bd2da40fd"
[[package]]
name = "clap"
version = "2.33.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "37e58ac78573c40708d45522f0d80fa2f01cc4f9b4e2bf749807255454312002"
dependencies = [
"bitflags",
"textwrap",
"unicode-width",
]
[[package]]
name = "color_quant"
version = "1.1.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "3d7b894f5411737b7867f4827955924d7c254fc9f4d91a6aad6b097804b1018b"
[[package]]
name = "crc32fast"
version = "1.2.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "81156fece84ab6a9f2afdb109ce3ae577e42b1228441eded99bd77f627953b1a"
dependencies = [
"cfg-if",
]
[[package]]
name = "criterion"
version = "0.3.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ab327ed7354547cc2ef43cbe20ef68b988e70b4b593cbd66a2a61733123a3d23"
dependencies = [
"atty",
"cast",
"clap",
"criterion-plot",
"csv",
"itertools 0.10.1",
"lazy_static",
"num-traits",
"oorandom",
"plotters",
"rayon",
"regex",
"serde",
"serde_cbor",
"serde_derive",
"serde_json",
"tinytemplate",
"walkdir",
]
[[package]]
name = "criterion-plot"
version = "0.4.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e022feadec601fba1649cfa83586381a4ad31c6bf3a9ab7d408118b05dd9889d"
dependencies = [
"cast",
"itertools 0.9.0",
]
[[package]]
name = "crossbeam-channel"
version = "0.5.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "06ed27e177f16d65f0f0c22a213e17c696ace5dd64b14258b52f9417ccb52db4"
dependencies = [
"cfg-if",
"crossbeam-utils",
]
[[package]]
name = "crossbeam-deque"
version = "0.8.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "94af6efb46fef72616855b036a624cf27ba656ffc9be1b9a3c931cfc7749a9a9"
dependencies = [
"cfg-if",
"crossbeam-epoch",
"crossbeam-utils",
]
[[package]]
name = "crossbeam-epoch"
version = "0.9.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "4ec02e091aa634e2c3ada4a392989e7c3116673ef0ac5b72232439094d73b7fd"
dependencies = [
"cfg-if",
"crossbeam-utils",
"lazy_static",
"memoffset",
"scopeguard",
]
[[package]]
name = "crossbeam-utils"
version = "0.8.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d82cfc11ce7f2c3faef78d8a684447b40d503d9681acebed6cb728d45940c4db"
dependencies = [
"cfg-if",
"lazy_static",
]
[[package]]
name = "csv"
version = "1.1.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "22813a6dc45b335f9bade10bf7271dc477e81113e89eb251a0bc2a8a81c536e1"
dependencies = [
"bstr",
"csv-core",
"itoa",
"ryu",
"serde",
]
[[package]]
name = "csv-core"
version = "0.1.10"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "2b2466559f260f48ad25fe6317b3c8dac77b5bdb5763ac7d9d6103530663bc90"
dependencies = [
"memchr",
]
[[package]]
name = "deflate"
version = "0.8.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "73770f8e1fe7d64df17ca66ad28994a0a623ea497fa69486e14984e715c5d174"
dependencies = [
"adler32",
"byteorder",
]
[[package]]
name = "either"
version = "1.6.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e78d4f1cc4ae33bbfc157ed5d5a5ef3bc29227303d595861deb238fcec4e9457"
[[package]]
name = "fallible_collections"
version = "0.4.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "4ad9169582543d2cfe9961be1e9eaf4fc42f9aa3483f7c485717b8dde36466ea"
dependencies = [
"hashbrown",
]
[[package]]
name = "fast_image_resize"
version = "0.1.0"
dependencies = [
"criterion",
"image",
"num-traits",
"resize",
"rgb",
"thiserror",
]
[[package]]
name = "gif"
version = "0.11.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "5a668f699973d0f573d15749b7002a9ac9e1f9c6b220e7b165601334c173d8de"
dependencies = [
"color_quant",
"weezl",
]
[[package]]
name = "half"
version = "1.7.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "62aca2aba2d62b4a7f5b33f3712cb1b0692779a56fb510499d5c0aa594daeaf3"
[[package]]
name = "hashbrown"
version = "0.9.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d7afe4a420e3fe79967a00898cc1f4db7c8a49a9333a29f8a4bd76a253d5cd04"
dependencies = [
"ahash",
]
[[package]]
name = "hermit-abi"
version = "0.1.19"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "62b467343b94ba476dcb2500d242dadbb39557df889310ac77c5d99100aaac33"
dependencies = [
"libc",
]
[[package]]
name = "image"
version = "0.23.14"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "24ffcb7e7244a9bf19d35bf2883b9c080c4ced3c07a9895572178cdb8f13f6a1"
dependencies = [
"bytemuck",
"byteorder",
"color_quant",
"gif",
"jpeg-decoder",
"num-iter",
"num-rational",
"num-traits",
"png",
"scoped_threadpool",
"tiff",
]
[[package]]
name = "itertools"
version = "0.9.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "284f18f85651fe11e8a991b2adb42cb078325c996ed026d994719efcfca1d54b"
dependencies = [
"either",
]
[[package]]
name = "itertools"
version = "0.10.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "69ddb889f9d0d08a67338271fa9b62996bc788c7796a5c18cf057420aaed5eaf"
dependencies = [
"either",
]
[[package]]
name = "itoa"
version = "0.4.7"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "dd25036021b0de88a0aff6b850051563c6516d0bf53f8638938edbb9de732736"
[[package]]
name = "jpeg-decoder"
version = "0.1.22"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "229d53d58899083193af11e15917b5640cd40b29ff475a1fe4ef725deb02d0f2"
dependencies = [
"rayon",
]
[[package]]
name = "js-sys"
version = "0.3.51"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "83bdfbace3a0e81a4253f73b49e960b053e396a11012cbd49b9b74d6a2b67062"
dependencies = [
"wasm-bindgen",
]
[[package]]
name = "lazy_static"
version = "1.4.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e2abad23fbc42b3700f2f279844dc832adb2b2eb069b2df918f455c4e18cc646"
[[package]]
name = "libc"
version = "0.2.97"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "12b8adadd720df158f4d70dfe7ccc6adb0472d7c55ca83445f6a5ab3e36f8fb6"
[[package]]
name = "log"
version = "0.4.14"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "51b9bbe6c47d51fc3e1a9b945965946b4c44142ab8792c50835a980d362c2710"
dependencies = [
"cfg-if",
]
[[package]]
name = "memchr"
version = "2.4.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b16bd47d9e329435e309c58469fe0791c2d0d1ba96ec0954152a5ae2b04387dc"
[[package]]
name = "memoffset"
version = "0.6.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "59accc507f1338036a0477ef61afdae33cde60840f4dfe481319ce3ad116ddf9"
dependencies = [
"autocfg",
]
[[package]]
name = "miniz_oxide"
version = "0.3.7"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "791daaae1ed6889560f8c4359194f56648355540573244a5448a83ba1ecc7435"
dependencies = [
"adler32",
]
[[package]]
name = "miniz_oxide"
version = "0.4.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "a92518e98c078586bc6c934028adcca4c92a53d6a958196de835170a01d84e4b"
dependencies = [
"adler",
"autocfg",
]
[[package]]
name = "num-integer"
version = "0.1.44"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d2cc698a63b549a70bc047073d2949cce27cd1c7b0a4a862d08a8031bc2801db"
dependencies = [
"autocfg",
"num-traits",
]
[[package]]
name = "num-iter"
version = "0.1.42"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b2021c8337a54d21aca0d59a92577a029af9431cb59b909b03252b9c164fad59"
dependencies = [
"autocfg",
"num-integer",
"num-traits",
]
[[package]]
name = "num-rational"
version = "0.3.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "12ac428b1cb17fce6f731001d307d351ec70a6d202fc2e60f7d4c5e42d8f4f07"
dependencies = [
"autocfg",
"num-integer",
"num-traits",
]
[[package]]
name = "num-traits"
version = "0.2.14"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9a64b1ec5cda2586e284722486d802acf1f7dbdc623e2bfc57e65ca1cd099290"
dependencies = [
"autocfg",
]
[[package]]
name = "num_cpus"
version = "1.13.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "05499f3756671c15885fee9034446956fff3f243d6077b91e5767df161f766b3"
dependencies = [
"hermit-abi",
"libc",
]
[[package]]
name = "oorandom"
version = "11.1.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "0ab1bc2a289d34bd04a330323ac98a1b4bc82c9d9fcb1e66b63caa84da26b575"
[[package]]
name = "plotters"
version = "0.3.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "32a3fd9ec30b9749ce28cd91f255d569591cdf937fe280c312143e3c4bad6f2a"
dependencies = [
"num-traits",
"plotters-backend",
"plotters-svg",
"wasm-bindgen",
"web-sys",
]
[[package]]
name = "plotters-backend"
version = "0.3.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d88417318da0eaf0fdcdb51a0ee6c3bed624333bff8f946733049380be67ac1c"
[[package]]
name = "plotters-svg"
version = "0.3.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "521fa9638fa597e1dc53e9412a4f9cefb01187ee1f7413076f9e6749e2885ba9"
dependencies = [
"plotters-backend",
]
[[package]]
name = "png"
version = "0.16.8"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "3c3287920cb847dee3de33d301c463fba14dda99db24214ddf93f83d3021f4c6"
dependencies = [
"bitflags",
"crc32fast",
"deflate",
"miniz_oxide 0.3.7",
]
[[package]]
name = "proc-macro2"
version = "1.0.27"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f0d8caf72986c1a598726adc988bb5984792ef84f5ee5aa50209145ee8077038"
dependencies = [
"unicode-xid",
]
[[package]]
name = "quote"
version = "1.0.9"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c3d0b9745dc2debf507c8422de05d7226cc1f0644216dfdfead988f9b1ab32a7"
dependencies = [
"proc-macro2",
]
[[package]]
name = "rayon"
version = "1.5.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c06aca804d41dbc8ba42dfd964f0d01334eceb64314b9ecf7c5fad5188a06d90"
dependencies = [
"autocfg",
"crossbeam-deque",
"either",
"rayon-core",
]
[[package]]
name = "rayon-core"
version = "1.9.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d78120e2c850279833f1dd3582f730c4ab53ed95aeaaaa862a2a5c71b1656d8e"
dependencies = [
"crossbeam-channel",
"crossbeam-deque",
"crossbeam-utils",
"lazy_static",
"num_cpus",
]
[[package]]
name = "regex"
version = "1.5.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d07a8629359eb56f1e2fb1652bb04212c072a87ba68546a04065d525673ac461"
dependencies = [
"regex-syntax",
]
[[package]]
name = "regex-automata"
version = "0.1.10"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "6c230d73fb8d8c1b9c0b3135c5142a8acee3a0558fb8db5cf1cb65f8d7862132"
[[package]]
name = "regex-syntax"
version = "0.6.25"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f497285884f3fcff424ffc933e56d7cbca511def0c9831a7f9b5f6153e3cc89b"
[[package]]
name = "resize"
version = "0.7.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "a2a32135a3cd4df690ee74dc0e1a77ffe2cdd4611b4e58b84df26a856d92b4ef"
dependencies = [
"fallible_collections",
"rgb",
]
[[package]]
name = "rgb"
version = "0.8.27"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8fddb3b23626145d1776addfc307e1a1851f60ef6ca64f376bcb889697144cf0"
dependencies = [
"bytemuck",
]
[[package]]
name = "rustc_version"
version = "0.4.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "bfa0f585226d2e68097d4f95d113b15b83a82e819ab25717ec0590d9584ef366"
dependencies = [
"semver",
]
[[package]]
name = "ryu"
version = "1.0.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "71d301d4193d031abdd79ff7e3dd721168a9572ef3fe51a1517aba235bd8f86e"
[[package]]
name = "same-file"
version = "1.0.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "93fc1dc3aaa9bfed95e02e6eadabb4baf7e3078b0bd1b4d7b6b0b68378900502"
dependencies = [
"winapi-util",
]
[[package]]
name = "scoped_threadpool"
version = "0.1.9"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1d51f5df5af43ab3f1360b429fa5e0152ac5ce8c0bd6485cae490332e96846a8"
[[package]]
name = "scopeguard"
version = "1.1.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d29ab0c6d3fc0ee92fe66e2d99f700eab17a8d57d1c1d3b748380fb20baa78cd"
[[package]]
name = "semver"
version = "1.0.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "5f3aac57ee7f3272d8395c6e4f502f434f0e289fcd62876f70daa008c20dcabe"
[[package]]
name = "serde"
version = "1.0.126"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ec7505abeacaec74ae4778d9d9328fe5a5d04253220a85c4ee022239fc996d03"
[[package]]
name = "serde_cbor"
version = "0.11.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1e18acfa2f90e8b735b2836ab8d538de304cbb6729a7360729ea5a895d15a622"
dependencies = [
"half",
"serde",
]
[[package]]
name = "serde_derive"
version = "1.0.126"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "963a7dbc9895aeac7ac90e74f34a5d5261828f79df35cbed41e10189d3804d43"
dependencies = [
"proc-macro2",
"quote",
"syn",
]
[[package]]
name = "serde_json"
version = "1.0.64"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "799e97dc9fdae36a5c8b8f2cae9ce2ee9fdce2058c57a93e6099d919fd982f79"
dependencies = [
"itoa",
"ryu",
"serde",
]
[[package]]
name = "syn"
version = "1.0.73"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f71489ff30030d2ae598524f61326b902466f72a0fb1a8564c001cc63425bcc7"
dependencies = [
"proc-macro2",
"quote",
"unicode-xid",
]
[[package]]
name = "textwrap"
version = "0.11.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d326610f408c7a4eb6f51c37c330e496b08506c9457c9d34287ecc38809fb060"
dependencies = [
"unicode-width",
]
[[package]]
name = "thiserror"
version = "1.0.26"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "93119e4feac1cbe6c798c34d3a53ea0026b0b1de6a120deef895137c0529bfe2"
dependencies = [
"thiserror-impl",
]
[[package]]
name = "thiserror-impl"
version = "1.0.26"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "060d69a0afe7796bf42e9e2ff91f5ee691fb15c53d38b4b62a9a53eb23164745"
dependencies = [
"proc-macro2",
"quote",
"syn",
]
[[package]]
name = "tiff"
version = "0.6.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9a53f4706d65497df0c4349241deddf35f84cee19c87ed86ea8ca590f4464437"
dependencies = [
"jpeg-decoder",
"miniz_oxide 0.4.4",
"weezl",
]
[[package]]
name = "tinytemplate"
version = "1.2.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "be4d6b5f19ff7664e8c98d03e2139cb510db9b0a60b55f8e8709b689d939b6bc"
dependencies = [
"serde",
"serde_json",
]
[[package]]
name = "unicode-width"
version = "0.1.8"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9337591893a19b88d8d87f2cec1e73fad5cdfd10e5a6f349f498ad6ea2ffb1e3"
[[package]]
name = "unicode-xid"
version = "0.2.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8ccb82d61f80a663efe1f787a51b16b5a51e3314d6ac365b08639f52387b33f3"
[[package]]
name = "walkdir"
version = "2.3.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "808cf2735cd4b6866113f648b791c6adc5714537bc222d9347bb203386ffda56"
dependencies = [
"same-file",
"winapi",
"winapi-util",
]
[[package]]
name = "wasm-bindgen"
version = "0.2.74"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d54ee1d4ed486f78874278e63e4069fc1ab9f6a18ca492076ffb90c5eb2997fd"
dependencies = [
"cfg-if",
"wasm-bindgen-macro",
]
[[package]]
name = "wasm-bindgen-backend"
version = "0.2.74"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "3b33f6a0694ccfea53d94db8b2ed1c3a8a4c86dd936b13b9f0a15ec4a451b900"
dependencies = [
"bumpalo",
"lazy_static",
"log",
"proc-macro2",
"quote",
"syn",
"wasm-bindgen-shared",
]
[[package]]
name = "wasm-bindgen-macro"
version = "0.2.74"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "088169ca61430fe1e58b8096c24975251700e7b1f6fd91cc9d59b04fb9b18bd4"
dependencies = [
"quote",
"wasm-bindgen-macro-support",
]
[[package]]
name = "wasm-bindgen-macro-support"
version = "0.2.74"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "be2241542ff3d9f241f5e2cb6dd09b37efe786df8851c54957683a49f0987a97"
dependencies = [
"proc-macro2",
"quote",
"syn",
"wasm-bindgen-backend",
"wasm-bindgen-shared",
]
[[package]]
name = "wasm-bindgen-shared"
version = "0.2.74"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d7cff876b8f18eed75a66cf49b65e7f967cb354a7aa16003fb55dbfd25b44b4f"
[[package]]
name = "web-sys"
version = "0.3.51"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e828417b379f3df7111d3a2a9e5753706cae29c41f7c4029ee9fd77f3e09e582"
dependencies = [
"js-sys",
"wasm-bindgen",
]
[[package]]
name = "weezl"
version = "0.1.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d8b77fdfd5a253be4ab714e4ffa3c49caf146b4de743e97510c0656cf90f1e8e"
[[package]]
name = "winapi"
version = "0.3.9"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "5c839a674fcd7a98952e593242ea400abe93992746761e38641405d28b00f419"
dependencies = [
"winapi-i686-pc-windows-gnu",
"winapi-x86_64-pc-windows-gnu",
]
[[package]]
name = "winapi-i686-pc-windows-gnu"
version = "0.4.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ac3b87c63620426dd9b991e5ce0329eff545bccbbb34f3be09ff6fb6ab51b7b6"
[[package]]
name = "winapi-util"
version = "0.1.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "70ec6ce85bb158151cae5e5c87f95a8e97d2c0c4b001223f33a334e3ce5de178"
dependencies = [
"winapi",
]
[[package]]
name = "winapi-x86_64-pc-windows-gnu"
version = "0.4.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "712e227841d057c1ee1cd2fb22fa7e5a5461ae8e48fa2ca79ec42cfc1931183f"
+60
View File
@@ -0,0 +1,60 @@
[package]
name = "fast_image_resize"
version = "0.1.0"
authors = ["Kirill Kuzminykh <[email protected]>"]
edition = "2018"
license = "MIT OR Apache-2.0"
description = "Library for fast image resizing with using of SIMD instructions"
readme = "README.md"
keywords = ["image", "resize"]
repository = "https://github.com/cykooz/fast_image_resize"
documentation = "https://docs.rs/crate/fast_image_resize"
# See more keys and their definitions at https://doc.rust-lang.org/cargo/reference/manifest.html
[dependencies]
num-traits = "0.2.14"
thiserror = "1.0.26"
[dev-dependencies]
criterion = "0.3.4"
image = "0.23.14"
resize = "0.7.0"
rgb = "0.8.25"
[[bench]]
name = "bench_resize"
harness = false
[[bench]]
name = "bench_alpha"
harness = false
[[bench]]
name = "bench_compare"
harness = false
[profile.dev.package.'*']
opt-level = 3
[profile.release]
#debug = true
#lto = true
opt-level = 3
#codegen-units = 1
[package.metadata.release]
pre-release-replacements = [
{file="CHANGELOG.md", search="Unreleased", replace="{{version}}"},
{file="CHANGELOG.md", search="ReleaseDate", replace="{{date}}"}
]
# Header of next release in CHANGELOG.md:
# ## [Unreleased] - ReleaseDate
+201
View File
@@ -0,0 +1,201 @@
Apache License
Version 2.0, January 2004
http://www.apache.org/licenses/
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
1. Definitions.
"License" shall mean the terms and conditions for use, reproduction,
and distribution as defined by Sections 1 through 9 of this document.
"Licensor" shall mean the copyright owner or entity authorized by
the copyright owner that is granting the License.
"Legal Entity" shall mean the union of the acting entity and all
other entities that control, are controlled by, or are under common
control with that entity. For the purposes of this definition,
"control" means (i) the power, direct or indirect, to cause the
direction or management of such entity, whether by contract or
otherwise, or (ii) ownership of fifty percent (50%) or more of the
outstanding shares, or (iii) beneficial ownership of such entity.
"You" (or "Your") shall mean an individual or Legal Entity
exercising permissions granted by this License.
"Source" form shall mean the preferred form for making modifications,
including but not limited to software source code, documentation
source, and configuration files.
"Object" form shall mean any form resulting from mechanical
transformation or translation of a Source form, including but
not limited to compiled object code, generated documentation,
and conversions to other media types.
"Work" shall mean the work of authorship, whether in Source or
Object form, made available under the License, as indicated by a
copyright notice that is included in or attached to the work
(an example is provided in the Appendix below).
"Derivative Works" shall mean any work, whether in Source or Object
form, that is based on (or derived from) the Work and for which the
editorial revisions, annotations, elaborations, or other modifications
represent, as a whole, an original work of authorship. For the purposes
of this License, Derivative Works shall not include works that remain
separable from, or merely link (or bind by name) to the interfaces of,
the Work and Derivative Works thereof.
"Contribution" shall mean any work of authorship, including
the original version of the Work and any modifications or additions
to that Work or Derivative Works thereof, that is intentionally
submitted to Licensor for inclusion in the Work by the copyright owner
or by an individual or Legal Entity authorized to submit on behalf of
the copyright owner. For the purposes of this definition, "submitted"
means any form of electronic, verbal, or written communication sent
to the Licensor or its representatives, including but not limited to
communication on electronic mailing lists, source code control systems,
and issue tracking systems that are managed by, or on behalf of, the
Licensor for the purpose of discussing and improving the Work, but
excluding communication that is conspicuously marked or otherwise
designated in writing by the copyright owner as "Not a Contribution."
"Contributor" shall mean Licensor and any individual or Legal Entity
on behalf of whom a Contribution has been received by Licensor and
subsequently incorporated within the Work.
2. Grant of Copyright License. Subject to the terms and conditions of
this License, each Contributor hereby grants to You a perpetual,
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
copyright license to reproduce, prepare Derivative Works of,
publicly display, publicly perform, sublicense, and distribute the
Work and such Derivative Works in Source or Object form.
3. Grant of Patent License. Subject to the terms and conditions of
this License, each Contributor hereby grants to You a perpetual,
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
(except as stated in this section) patent license to make, have made,
use, offer to sell, sell, import, and otherwise transfer the Work,
where such license applies only to those patent claims licensable
by such Contributor that are necessarily infringed by their
Contribution(s) alone or by combination of their Contribution(s)
with the Work to which such Contribution(s) was submitted. If You
institute patent litigation against any entity (including a
cross-claim or counterclaim in a lawsuit) alleging that the Work
or a Contribution incorporated within the Work constitutes direct
or contributory patent infringement, then any patent licenses
granted to You under this License for that Work shall terminate
as of the date such litigation is filed.
4. Redistribution. You may reproduce and distribute copies of the
Work or Derivative Works thereof in any medium, with or without
modifications, and in Source or Object form, provided that You
meet the following conditions:
(a) You must give any other recipients of the Work or
Derivative Works a copy of this License; and
(b) You must cause any modified files to carry prominent notices
stating that You changed the files; and
(c) You must retain, in the Source form of any Derivative Works
that You distribute, all copyright, patent, trademark, and
attribution notices from the Source form of the Work,
excluding those notices that do not pertain to any part of
the Derivative Works; and
(d) If the Work includes a "NOTICE" text file as part of its
distribution, then any Derivative Works that You distribute must
include a readable copy of the attribution notices contained
within such NOTICE file, excluding those notices that do not
pertain to any part of the Derivative Works, in at least one
of the following places: within a NOTICE text file distributed
as part of the Derivative Works; within the Source form or
documentation, if provided along with the Derivative Works; or,
within a display generated by the Derivative Works, if and
wherever such third-party notices normally appear. The contents
of the NOTICE file are for informational purposes only and
do not modify the License. You may add Your own attribution
notices within Derivative Works that You distribute, alongside
or as an addendum to the NOTICE text from the Work, provided
that such additional attribution notices cannot be construed
as modifying the License.
You may add Your own copyright statement to Your modifications and
may provide additional or different license terms and conditions
for use, reproduction, or distribution of Your modifications, or
for any such Derivative Works as a whole, provided Your use,
reproduction, and distribution of the Work otherwise complies with
the conditions stated in this License.
5. Submission of Contributions. Unless You explicitly state otherwise,
any Contribution intentionally submitted for inclusion in the Work
by You to the Licensor shall be under the terms and conditions of
this License, without any additional terms or conditions.
Notwithstanding the above, nothing herein shall supersede or modify
the terms of any separate license agreement you may have executed
with Licensor regarding such Contributions.
6. Trademarks. This License does not grant permission to use the trade
names, trademarks, service marks, or product names of the Licensor,
except as required for reasonable and customary use in describing the
origin of the Work and reproducing the content of the NOTICE file.
7. Disclaimer of Warranty. Unless required by applicable law or
agreed to in writing, Licensor provides the Work (and each
Contributor provides its Contributions) on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
implied, including, without limitation, any warranties or conditions
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
PARTICULAR PURPOSE. You are solely responsible for determining the
appropriateness of using or redistributing the Work and assume any
risks associated with Your exercise of permissions under this License.
8. Limitation of Liability. In no event and under no legal theory,
whether in tort (including negligence), contract, or otherwise,
unless required by applicable law (such as deliberate and grossly
negligent acts) or agreed to in writing, shall any Contributor be
liable to You for damages, including any direct, indirect, special,
incidental, or consequential damages of any character arising as a
result of this License or out of the use or inability to use the
Work (including but not limited to damages for loss of goodwill,
work stoppage, computer failure or malfunction, or any and all
other commercial damages or losses), even if such Contributor
has been advised of the possibility of such damages.
9. Accepting Warranty or Additional Liability. While redistributing
the Work or Derivative Works thereof, You may choose to offer,
and charge a fee for, acceptance of support, warranty, indemnity,
or other liability obligations and/or rights consistent with this
License. However, in accepting such obligations, You may act only
on Your own behalf and on Your sole responsibility, not on behalf
of any other Contributor, and only if You agree to indemnify,
defend, and hold each Contributor harmless for any liability
incurred by, or claims asserted against, such Contributor by reason
of your accepting any such warranty or additional liability.
END OF TERMS AND CONDITIONS
APPENDIX: How to apply the Apache License to your work.
To apply the Apache License to your work, attach the following
boilerplate notice, with the fields enclosed by brackets "[]"
replaced with your own identifying information. (Don't include
the brackets!) The text should be enclosed in the appropriate
comment syntax for the file format. We also recommend that a
file or class name and description of purpose be included on the
same "printed page" as the copyright notice for easier
identification within third-party archives.
Copyright (c) 2021 Kirill Kuzminykh
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
+21
View File
@@ -0,0 +1,21 @@
MIT License
Copyright (c) 2021 Kirill Kuzminykh
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
SOFTWARE.
+49
View File
@@ -0,0 +1,49 @@
# fast_image_resize
Library for fast image resizing with using of SIMD instructions.
[CHANGELOG](https://github.com/Cykooz/fast_image_resize/blob/master/CHANGELOG.md)
## Examples
```rust
use std::num::NonZeroU32;
use fast_image_resize::{
CropBox, FilterType, ImageData, PixelType, ResizeAlg, Resizer, SrcImageView,
};
fn resize_lanczos3(src_pixels: &[u8], width: NonZeroU32, height: NonZeroU32) -> Vec<u8> {
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
let src_image = ImageData::new(width, height, src_pixels, PixelType::U8x4).unwrap();
let dst_width = NonZeroU32::new(1024).unwrap();
let dst_height = NonZeroU32::new(768).unwrap();
let mut dst_image = ImageData::new_owned(dst_width, dst_height, src_image.pixel_type());
let src_view = src_image.src_view();
let mut dst_view = dst_image.dst_view();
resizer.resize(&src_view, &mut dst_view);
dst_image.get_buffer().to_owned()
}
fn resize_cropped_image(mut src_view: SrcImageView) -> ImageData<Vec<u8>> {
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
src_view
.set_crop_box(CropBox {
left: 10,
top: 10,
width: NonZeroU32::new(100).unwrap(),
height: NonZeroU32::new(200).unwrap(),
})
.unwrap();
let dst_width = NonZeroU32::new(1024).unwrap();
let dst_height = NonZeroU32::new(768).unwrap();
let mut dst_image = ImageData::new_owned(dst_width, dst_height, src_view.pixel_type());
let mut dst_view = dst_image.dst_view();
resizer.resize(&src_view, &mut dst_view);
dst_image
}
```
+160
View File
@@ -0,0 +1,160 @@
use criterion::{criterion_group, criterion_main, Criterion};
use fast_image_resize::{CpuExtensions, ImageData, MulDiv, PixelType};
use std::num::NonZeroU32;
const fn p(r: u8, g: u8, b: u8, a: u8) -> u32 {
u32::from_le_bytes([r, g, b, a])
}
// Multiplies by alpha
fn get_src_image(width: NonZeroU32, height: NonZeroU32, pixel: u32) -> ImageData<Vec<u8>> {
let rgba: [u8; 4] = pixel.to_le_bytes();
let buf_size = (width.get() * height.get()) as usize * 4;
let mut buffer = vec![0u8; buf_size];
buffer.chunks_exact_mut(4).for_each(|c| {
c[0] = rgba[0];
c[1] = rgba[1];
c[2] = rgba[2];
c[3] = rgba[3];
});
ImageData::new(width, height, buffer, PixelType::U8x4).unwrap()
}
fn multiplies_alpha_avx2(c: &mut Criterion) {
let width = NonZeroU32::new(4096).unwrap();
let height = NonZeroU32::new(2048).unwrap();
let src_data = get_src_image(width, height, p(255, 128, 0, 128));
let mut dst_data = ImageData::new_owned(width, height, PixelType::U8x4);
let src_view = src_data.src_view();
let mut dst_view = dst_data.dst_view();
let mut alpha_mul_div: MulDiv = Default::default();
unsafe {
alpha_mul_div.set_cpu_extensions(CpuExtensions::Avx2);
}
c.bench_function("Multiplies alpha AVX2", |b| {
b.iter(|| {
alpha_mul_div
.multiply_alpha(&src_view, &mut dst_view)
.unwrap();
})
});
}
fn multiplies_alpha_sse2(c: &mut Criterion) {
let width = NonZeroU32::new(4096).unwrap();
let height = NonZeroU32::new(2048).unwrap();
let src_data = get_src_image(width, height, p(255, 128, 0, 128));
let mut dst_data = ImageData::new_owned(width, height, PixelType::U8x4);
let src_view = src_data.src_view();
let mut dst_view = dst_data.dst_view();
let mut alpha_mul_div: MulDiv = Default::default();
unsafe {
alpha_mul_div.set_cpu_extensions(CpuExtensions::Sse2);
}
c.bench_function("Multiplies alpha SSE2", |b| {
b.iter(|| {
alpha_mul_div
.multiply_alpha(&src_view, &mut dst_view)
.unwrap();
})
});
}
fn multiplies_alpha_native(c: &mut Criterion) {
let width = NonZeroU32::new(4096).unwrap();
let height = NonZeroU32::new(2048).unwrap();
let src_data = get_src_image(width, height, p(255, 128, 0, 128));
let mut dst_data = ImageData::new_owned(width, height, PixelType::U8x4);
let src_view = src_data.src_view();
let mut dst_view = dst_data.dst_view();
let mut alpha_mul_div: MulDiv = Default::default();
unsafe {
alpha_mul_div.set_cpu_extensions(CpuExtensions::None);
}
c.bench_function("Multiplies alpha native", |b| {
b.iter(|| {
alpha_mul_div
.multiply_alpha(&src_view, &mut dst_view)
.unwrap();
})
});
}
fn divides_alpha_avx2(c: &mut Criterion) {
let width = NonZeroU32::new(4096).unwrap();
let height = NonZeroU32::new(2048).unwrap();
let src_data = get_src_image(width, height, p(128, 64, 0, 128));
let mut dst_data = ImageData::new_owned(width, height, PixelType::U8x4);
let src_view = src_data.src_view();
let mut dst_view = dst_data.dst_view();
let mut alpha_mul_div: MulDiv = Default::default();
unsafe {
alpha_mul_div.set_cpu_extensions(CpuExtensions::Avx2);
}
c.bench_function("Divides alpha AVX2", |b| {
b.iter(|| {
alpha_mul_div
.divide_alpha(&src_view, &mut dst_view)
.unwrap();
})
});
}
fn divides_alpha_sse2(c: &mut Criterion) {
let width = NonZeroU32::new(4096).unwrap();
let height = NonZeroU32::new(2048).unwrap();
let src_data = get_src_image(width, height, p(128, 64, 0, 128));
let mut dst_data = ImageData::new_owned(width, height, PixelType::U8x4);
let src_view = src_data.src_view();
let mut dst_view = dst_data.dst_view();
let mut alpha_mul_div: MulDiv = Default::default();
unsafe {
alpha_mul_div.set_cpu_extensions(CpuExtensions::Sse2);
}
c.bench_function("Divides alpha SSE2", |b| {
b.iter(|| {
alpha_mul_div
.divide_alpha(&src_view, &mut dst_view)
.unwrap();
})
});
}
fn divides_alpha_native(c: &mut Criterion) {
let width = NonZeroU32::new(4096).unwrap();
let height = NonZeroU32::new(2048).unwrap();
let src_data = get_src_image(width, height, p(128, 64, 0, 128));
let mut dst_data = ImageData::new_owned(width, height, PixelType::U8x4);
let src_view = src_data.src_view();
let mut dst_view = dst_data.dst_view();
let mut alpha_mul_div: MulDiv = Default::default();
unsafe {
alpha_mul_div.set_cpu_extensions(CpuExtensions::None);
}
c.bench_function("Divides alpha native", |b| {
b.iter(|| {
alpha_mul_div
.divide_alpha(&src_view, &mut dst_view)
.unwrap();
})
});
}
criterion_group!(
benches,
multiplies_alpha_avx2,
multiplies_alpha_sse2,
multiplies_alpha_native,
divides_alpha_avx2,
divides_alpha_sse2,
divides_alpha_native,
);
criterion_main!(benches);
+153
View File
@@ -0,0 +1,153 @@
use std::num::NonZeroU32;
use criterion::{criterion_group, criterion_main, BenchmarkId, Criterion};
use image::imageops;
use resize::px::RGBA;
use resize::Pixel::RGBA8;
use rgb::FromSlice;
use fast_image_resize::ImageData;
use fast_image_resize::{CpuExtensions, FilterType, PixelType, ResizeAlg, Resizer};
mod utils;
pub fn bench_downscale(c: &mut Criterion) {
let src_image = utils::get_big_rgb_image();
let new_width = NonZeroU32::new(852).unwrap();
let new_height = NonZeroU32::new(567).unwrap();
let mut group = c.benchmark_group("Downscale (4928x3279 => 852x567)");
for alg_num in 0..4 {
// image crate
// https://crates.io/crates/image
group.bench_with_input(
BenchmarkId::new("image", alg_num),
&alg_num,
|b, alg_num| {
b.iter(|| {
let filter = match alg_num {
0 => imageops::Nearest,
1 => imageops::Triangle,
2 => imageops::CatmullRom,
3 => imageops::Lanczos3,
_ => return,
};
imageops::resize(&src_image, new_width.get(), new_height.get(), filter);
})
},
);
// resize crate
// https://crates.io/crates/resize
let resize_src_image = src_image.as_raw().as_rgba();
let mut dst = vec![RGBA::new(0, 0, 0, 0); (new_width.get() * new_height.get()) as usize];
group.bench_with_input(
BenchmarkId::new("resize", alg_num),
&alg_num,
|b, alg_num| {
let filter = match alg_num {
0 => resize::Type::Point,
1 => resize::Type::Triangle,
2 => resize::Type::Catrom,
3 => resize::Type::Lanczos3,
_ => return,
};
let mut resizer = resize::new(
src_image.width() as usize,
src_image.height() as usize,
new_width.get() as usize,
new_height.get() as usize,
RGBA8,
filter,
)
.unwrap();
b.iter(|| {
resizer.resize(resize_src_image, &mut dst).unwrap();
})
},
);
// fast_image_resize crate
let src_image_data = ImageData::new(
NonZeroU32::new(src_image.width()).unwrap(),
NonZeroU32::new(src_image.height()).unwrap(),
src_image.as_raw(),
PixelType::U8x4,
)
.unwrap();
let src_view = src_image_data.src_view();
let mut dst_image = ImageData::new_owned(new_width, new_height, PixelType::U8x4);
let mut dst_view = dst_image.dst_view();
let mut fast_resizer = Resizer::new(ResizeAlg::Nearest);
unsafe {
fast_resizer.set_cpu_extensions(CpuExtensions::None);
}
group.bench_with_input(
BenchmarkId::new("fast_wo_simd", alg_num),
&alg_num,
|b, alg_num| {
let resize_alg = match alg_num {
0 => ResizeAlg::Nearest,
1 => ResizeAlg::Convolution(FilterType::Bilinear),
2 => ResizeAlg::Convolution(FilterType::CatmulRom),
3 => ResizeAlg::Convolution(FilterType::Lanczos3),
_ => return,
};
fast_resizer.algorithm = resize_alg;
b.iter(|| {
fast_resizer.resize(&src_view, &mut dst_view);
})
},
);
let mut fast_resizer = Resizer::new(ResizeAlg::Nearest);
unsafe {
fast_resizer.set_cpu_extensions(CpuExtensions::Sse4_1);
}
group.bench_with_input(
BenchmarkId::new("fast_sse4", alg_num),
&alg_num,
|b, alg_num| {
let resize_alg = match alg_num {
0 => ResizeAlg::Nearest,
1 => ResizeAlg::Convolution(FilterType::Bilinear),
2 => ResizeAlg::Convolution(FilterType::CatmulRom),
3 => ResizeAlg::Convolution(FilterType::Lanczos3),
_ => return,
};
fast_resizer.algorithm = resize_alg;
b.iter(|| {
fast_resizer.resize(&src_view, &mut dst_view);
})
},
);
let mut fast_resizer = Resizer::new(ResizeAlg::Nearest);
unsafe {
fast_resizer.set_cpu_extensions(CpuExtensions::Avx2);
}
group.bench_with_input(
BenchmarkId::new("fast_avx2", alg_num),
&alg_num,
|b, alg_num| {
let resize_alg = match alg_num {
0 => ResizeAlg::Nearest,
1 => ResizeAlg::Convolution(FilterType::Bilinear),
2 => ResizeAlg::Convolution(FilterType::CatmulRom),
3 => ResizeAlg::Convolution(FilterType::Lanczos3),
_ => return,
};
fast_resizer.algorithm = resize_alg;
b.iter(|| {
fast_resizer.resize(&src_view, &mut dst_view);
})
},
);
}
group.finish();
}
criterion_group!(benches, bench_downscale,);
criterion_main!(benches);
+174
View File
@@ -0,0 +1,174 @@
use std::num::NonZeroU32;
use criterion::{criterion_group, criterion_main, Criterion};
use fast_image_resize::ImageData;
use fast_image_resize::{CpuExtensions, FilterType, PixelType, ResizeAlg, Resizer};
mod utils;
const NEW_WIDTH: u32 = 852;
const NEW_HEIGHT: u32 = 567;
const NEW_BIG_WIDTH: u32 = 4928;
const NEW_BIG_HEIGHT: u32 = 3279;
fn get_big_source_image() -> ImageData<Vec<u8>> {
let img = utils::get_big_rgb_image();
let width = img.width();
let height = img.height();
let buf = img.as_raw().clone();
ImageData::new(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
buf,
PixelType::U8x4,
)
.unwrap()
}
fn get_small_source_image() -> ImageData<Vec<u8>> {
let img = utils::get_small_rgb_image();
let width = img.width();
let height = img.height();
let buf = img.as_raw().clone();
ImageData::new(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
buf,
PixelType::U8x4,
)
.unwrap()
}
fn downscale_nearest_wo_simd_bench(c: &mut Criterion) {
let image = get_big_source_image();
let mut res_image = ImageData::new_owned(
NonZeroU32::new(NEW_WIDTH).unwrap(),
NonZeroU32::new(NEW_HEIGHT).unwrap(),
image.pixel_type(),
);
let src_image = image.src_view();
let mut dst_image = res_image.dst_view();
let mut resizer = Resizer::new(ResizeAlg::Nearest);
unsafe {
resizer.set_cpu_extensions(CpuExtensions::None);
}
c.bench_function("Downscale nearest wo SIMD", |b| {
b.iter(|| {
resizer.resize(&src_image, &mut dst_image);
})
});
}
fn downscale_lanczos3_wo_simd_bench(c: &mut Criterion) {
let image = get_big_source_image();
let mut res_image = ImageData::new_owned(
NonZeroU32::new(NEW_WIDTH).unwrap(),
NonZeroU32::new(NEW_HEIGHT).unwrap(),
image.pixel_type(),
);
let src_image = image.src_view();
let mut dst_image = res_image.dst_view();
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
unsafe {
resizer.set_cpu_extensions(CpuExtensions::None);
}
c.bench_function("Downscale lanczos3 wo SIMD", |b| {
b.iter(|| {
resizer.resize(&src_image, &mut dst_image);
})
});
}
fn sse4_lanczos3_bench(c: &mut Criterion) {
let image = get_big_source_image();
let mut res_image = ImageData::new_owned(
NonZeroU32::new(NEW_WIDTH).unwrap(),
NonZeroU32::new(NEW_HEIGHT).unwrap(),
image.pixel_type(),
);
let src_image = image.src_view();
let mut dst_image = res_image.dst_view();
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
unsafe {
resizer.set_cpu_extensions(CpuExtensions::Sse4_1);
}
c.bench_function("sse4 lanczos3", |b| {
b.iter(|| {
resizer.resize(&src_image, &mut dst_image);
})
});
}
fn avx2_lanczos3_bench(c: &mut Criterion) {
let image = get_big_source_image();
let mut res_image = ImageData::new_owned(
NonZeroU32::new(NEW_WIDTH).unwrap(),
NonZeroU32::new(NEW_HEIGHT).unwrap(),
image.pixel_type(),
);
let src_image = image.src_view();
let mut dst_image = res_image.dst_view();
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
unsafe {
resizer.set_cpu_extensions(CpuExtensions::Avx2);
}
c.bench_function("avx2 lanczos3", |b| {
b.iter(|| {
resizer.resize(&src_image, &mut dst_image);
})
});
}
fn avx2_supersampling_lanczos3_bench(c: &mut Criterion) {
let image = get_big_source_image();
let mut res_image = ImageData::new_owned(
NonZeroU32::new(NEW_WIDTH).unwrap(),
NonZeroU32::new(NEW_HEIGHT).unwrap(),
image.pixel_type(),
);
let src_image = image.src_view();
let mut dst_image = res_image.dst_view();
let mut resizer = Resizer::new(ResizeAlg::SuperSampling(FilterType::Lanczos3, 2));
unsafe {
resizer.set_cpu_extensions(CpuExtensions::Avx2);
}
c.bench_function("avx2 supersampling lanczos3", |b| {
b.iter(|| {
resizer.resize(&src_image, &mut dst_image);
})
});
}
fn avx2_lanczos3_upscale_bench(c: &mut Criterion) {
let image = get_small_source_image();
let mut res_image = ImageData::new_owned(
NonZeroU32::new(NEW_BIG_WIDTH).unwrap(),
NonZeroU32::new(NEW_BIG_HEIGHT).unwrap(),
image.pixel_type(),
);
let src_image = image.src_view();
let mut dst_image = res_image.dst_view();
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
unsafe {
resizer.set_cpu_extensions(CpuExtensions::Avx2);
}
c.bench_function("avx2 lanczos3 upscale", |b| {
b.iter(|| {
resizer.resize(&src_image, &mut dst_image);
})
});
}
criterion_group!(
benches,
// downscale_nearest_wo_simd_bench,
// downscale_lanczos3_wo_simd_bench,
// resize_lanczos3_bench,
// sse4_lanczos3_bench,
avx2_lanczos3_bench,
// avx2_supersampling_lanczos3_bench,
// avx2_lanczos3_upscale_bench,
);
criterion_main!(benches);
+22
View File
@@ -0,0 +1,22 @@
use std::env;
use image::io::Reader;
use image::RgbaImage;
pub fn get_big_rgb_image() -> RgbaImage {
let cur_dir = env::current_dir().unwrap();
let img = Reader::open(cur_dir.join("data/nasa-4928x3279.png"))
.unwrap()
.decode()
.unwrap();
img.to_rgba8()
}
pub fn get_small_rgb_image() -> RgbaImage {
let cur_dir = env::current_dir().unwrap();
let img = Reader::open(cur_dir.join("data/nasa-852x567.png"))
.unwrap()
.decode()
.unwrap();
img.to_rgba8()
}
Binary file not shown.

After

Width:  |  Height:  |  Size: 15 MiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 13 MiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 688 KiB

+172
View File
@@ -0,0 +1,172 @@
use std::arch::x86_64::*;
use crate::simd_utils;
use crate::{DstImageView, SrcImageView};
pub(crate) fn divide_alpha_avx2(src_image: &SrcImageView, dst_image: &mut DstImageView) {
let width = src_image.width().get();
let src_rows = src_image.iter_rows(0, src_image.height().get());
let dst_rows = dst_image.iter_rows_mut();
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
divide_alpha_row_avx2(src_row, dst_row, width as usize);
}
}
}
pub(crate) fn divide_alpha_inplace_avx2(image: &mut DstImageView) {
let width = image.width().get() as usize;
for dst_row in image.iter_rows_mut() {
unsafe {
let src_row = std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len());
divide_alpha_row_avx2(src_row, dst_row, width);
}
}
}
pub(crate) fn divide_alpha_sse2(src_image: &SrcImageView, dst_image: &mut DstImageView) {
let width = src_image.width().get() as usize;
let src_rows = src_image.iter_rows(0, src_image.height().get());
let dst_rows = dst_image.iter_rows_mut();
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
divide_alpha_row_sse2(src_row, dst_row, width);
}
}
}
pub(crate) fn divide_alpha_inplace_sse2(image: &mut DstImageView) {
let width = image.width().get() as usize;
for dst_row in image.iter_rows_mut() {
unsafe {
let src_row = std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len());
divide_alpha_row_sse2(src_row, dst_row, width);
}
}
}
pub(crate) fn divide_alpha_native(src_image: &SrcImageView, dst_image: &mut DstImageView) {
let src_rows = src_image.iter_rows(0, src_image.height().get());
let dst_rows = dst_image.iter_rows_mut();
for (src_row, dst_row) in src_rows.zip(dst_rows) {
divide_alpha_row_native(src_row, dst_row);
}
}
pub(crate) fn divide_alpha_inplace_native(image: &mut DstImageView) {
for dst_row in image.iter_rows_mut() {
let src_row = unsafe { std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len()) };
divide_alpha_row_native(src_row, dst_row);
}
}
#[target_feature(enable = "avx2")]
unsafe fn divide_alpha_row_avx2(src_row: &[u32], dst_row: &mut [u32], width: usize) {
let mut x: usize = 0;
let zero = _mm256_setzero_si256();
#[rustfmt::skip]
let alpha_mask = _mm256_set1_epi32(0xff000000u32 as i32);
#[rustfmt::skip]
let shuffle1 = _mm256_set_epi8(
5, 4, 5, 4, 5, 4, 5, 4, 1, 0, 1, 0, 1, 0, 1, 0,
5, 4, 5, 4, 5, 4, 5, 4, 1, 0, 1, 0, 1, 0, 1, 0,
);
#[rustfmt::skip]
let shuffle2 = _mm256_set_epi8(
13, 12, 13, 12, 13, 12, 13, 12, 9, 8, 9, 8, 9, 8, 9, 8,
13, 12, 13, 12, 13, 12, 13, 12, 9, 8, 9, 8, 9, 8, 9, 8,
);
let alpha_scale = _mm256_set1_ps(255.0 * 256.0);
while x < width.saturating_sub(7) {
let mut source = simd_utils::loadu_si256(src_row, x);
let alpha_f32 = _mm256_cvtepi32_ps(_mm256_srli_epi32(source, 24));
let scaled_alpha_f32 = _mm256_mul_ps(alpha_scale, _mm256_rcp_ps(alpha_f32));
let scaled_alpha_i32 = _mm256_cvtps_epi32(scaled_alpha_f32);
let mma0 = _mm256_shuffle_epi8(scaled_alpha_i32, shuffle1);
let mma1 = _mm256_shuffle_epi8(scaled_alpha_i32, shuffle2);
let mut pix0 = _mm256_unpacklo_epi8(zero, source);
let mut pix1 = _mm256_unpackhi_epi8(zero, source);
pix0 = _mm256_mulhi_epu16(pix0, mma0);
pix1 = _mm256_mulhi_epu16(pix1, mma1);
let alpha = _mm256_and_si256(source, alpha_mask);
source = _mm256_packus_epi16(pix0, pix1);
source = _mm256_blendv_epi8(source, alpha, alpha_mask);
let dst_ptr = dst_row.get_unchecked_mut(x..).as_mut_ptr() as *mut __m256i;
_mm256_storeu_si256(dst_ptr, source);
x += 8;
}
let src_tail = &src_row[x..];
let dst_tail = &mut dst_row[x..];
divide_alpha_row_native(src_tail, dst_tail);
}
#[target_feature(enable = "sse2")]
unsafe fn divide_alpha_row_sse2(src_row: &[u32], dst_row: &mut [u32], width: usize) {
let zero = _mm_setzero_si128();
let alpha_mask = _mm_set1_epi32(0xff000000u32 as i32);
let shuffle0 = _mm_set_epi8(5, 4, 5, 4, 5, 4, 5, 4, 1, 0, 1, 0, 1, 0, 1, 0);
let shuffle1 = _mm_set_epi8(13, 12, 13, 12, 13, 12, 13, 12, 9, 8, 9, 8, 9, 8, 9, 8);
let alpha_scale = _mm_set1_ps(255.0 * 256.0);
let mut x: usize = 0;
while x < width.saturating_sub(3) {
let mut source = simd_utils::loadu_si128(src_row, x);
let alpha = _mm_and_si128(source, alpha_mask);
let alpha_f32 = _mm_cvtepi32_ps(_mm_srli_epi32(source, 24));
let scaled_recip_alpha_f32 = _mm_mul_ps(alpha_scale, _mm_rcp_ps(alpha_f32));
let scaled_recip_alpha_i32 = _mm_cvtps_epi32(scaled_recip_alpha_f32);
let mma0 = _mm_shuffle_epi8(scaled_recip_alpha_i32, shuffle0);
let mma1 = _mm_shuffle_epi8(scaled_recip_alpha_i32, shuffle1);
let mut pix0 = _mm_unpacklo_epi8(zero, source);
let mut pix1 = _mm_unpackhi_epi8(zero, source);
pix0 = _mm_mulhi_epu16(pix0, mma0);
pix1 = _mm_mulhi_epu16(pix1, mma1);
source = _mm_packus_epi16(pix0, pix1);
source = _mm_blendv_epi8(source, alpha, alpha_mask);
let dst_ptr = dst_row.get_unchecked_mut(x..).as_mut_ptr() as *mut __m128i;
_mm_storeu_si128(dst_ptr, source);
x += 4;
}
let src_tail = &src_row[x..];
let dst_tail = &mut dst_row[x..];
divide_alpha_row_native(src_tail, dst_tail);
}
#[inline(always)]
fn div_and_clip(v: u8, rev_alpha: f32) -> u8 {
let res = v as f32 * rev_alpha;
res.min(255.) as u8
}
#[inline(always)]
fn divide_alpha_row_native(src_row: &[u32], dst_row: &mut [u32]) {
for (src_pixel, dst_pixel) in src_row.iter().zip(dst_row) {
let components: [u8; 4] = src_pixel.to_le_bytes();
let alpha = components[3];
let recip_alpha = if alpha == 0 { 0. } else { 255. / alpha as f32 };
let res = [
div_and_clip(components[0], recip_alpha),
div_and_clip(components[1], recip_alpha),
div_and_clip(components[2], recip_alpha),
alpha,
];
*dst_pixel = u32::from_le_bytes(res);
}
}
+128
View File
@@ -0,0 +1,128 @@
use thiserror::Error;
use crate::{CpuExtensions, PixelType};
use crate::{DstImageView, SrcImageView};
mod div;
mod mul;
#[derive(Error, Debug, Clone, Copy)]
#[non_exhaustive]
pub enum MulDivImagesError {
#[error("Size of source image does not match to destination image")]
SizeIsDifferent,
#[error("Pixel type of source image does not match to destination image")]
PixelTypeIsDifferent,
#[error("Pixel type of image is not supported")]
UnsupportedPixelType,
}
#[derive(Error, Debug, Clone, Copy)]
#[non_exhaustive]
pub enum MulDivImageError {
#[error("Pixel type of image is not supported")]
UnsupportedPixelType,
}
/// Methods of this structure used to multiplies or divides RGB-channels
/// by alpha-channel.
#[derive(Default, Debug, Clone)]
pub struct MulDiv {
cpu_extensions: CpuExtensions,
}
impl MulDiv {
#[inline(always)]
pub fn cpu_extensions(&self) -> CpuExtensions {
self.cpu_extensions
}
/// # Safety
/// This is unsafe because this method allows you to set a CPU-extensions
/// that is not actually supported by your CPU.
pub unsafe fn set_cpu_extensions(&mut self, extensions: CpuExtensions) {
self.cpu_extensions = extensions;
}
/// Multiplies RGB-channels of source image by alpha-channel and store
/// result into destination image.
pub fn multiply_alpha(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
) -> Result<(), MulDivImagesError> {
self.assert_images(src_image, dst_image)?;
match self.cpu_extensions {
CpuExtensions::Avx2 => mul::multiply_alpha_avx2(src_image, dst_image),
// CpuExtensions::Sse2 => mul::multiply_alpha_sse2(src_image, dst_image),
_ => mul::multiply_alpha_native(src_image, dst_image),
}
Ok(())
}
/// Multiplies RGB-channels of image by alpha-channel inplace.
pub fn multiply_alpha_inplace(&self, image: &mut DstImageView) -> Result<(), MulDivImageError> {
self.assert_image(image)?;
match self.cpu_extensions {
CpuExtensions::Avx2 => mul::multiply_alpha_inplace_avx2(image),
// CpuExtensions::Sse2 => mul::multiply_alpha_sse2(src_image, dst_image),
_ => mul::multiply_alpha_inplace_native(image),
}
Ok(())
}
/// Divides RGB-channels of source image by alpha-channel and store
/// result into destination image.
pub fn divide_alpha(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
) -> Result<(), MulDivImagesError> {
self.assert_images(src_image, dst_image)?;
match self.cpu_extensions {
CpuExtensions::Avx2 => div::divide_alpha_avx2(src_image, dst_image),
CpuExtensions::Sse4_1 | CpuExtensions::Sse2 => {
div::divide_alpha_sse2(src_image, dst_image)
}
_ => div::divide_alpha_native(src_image, dst_image),
}
Ok(())
}
/// Divides RGB-channels of image by alpha-channel inplace.
pub fn divide_alpha_inplace(&self, image: &mut DstImageView) -> Result<(), MulDivImageError> {
self.assert_image(image)?;
match self.cpu_extensions {
CpuExtensions::Avx2 => div::divide_alpha_inplace_avx2(image),
CpuExtensions::Sse4_1 | CpuExtensions::Sse2 => div::divide_alpha_inplace_sse2(image),
_ => div::divide_alpha_inplace_native(image),
}
Ok(())
}
#[inline]
fn assert_images(
&self,
src_image: &SrcImageView,
dst_image: &DstImageView,
) -> Result<(), MulDivImagesError> {
if src_image.width() != dst_image.width() || src_image.height() != dst_image.height() {
return Err(MulDivImagesError::SizeIsDifferent);
}
if src_image.pixel_type() != PixelType::U8x4 {
return Err(MulDivImagesError::UnsupportedPixelType);
}
if src_image.pixel_type() != dst_image.pixel_type() {
return Err(MulDivImagesError::PixelTypeIsDifferent);
}
Ok(())
}
#[inline]
fn assert_image(&self, image: &DstImageView) -> Result<(), MulDivImageError> {
if image.pixel_type() != PixelType::U8x4 {
return Err(MulDivImageError::UnsupportedPixelType);
}
Ok(())
}
}
+161
View File
@@ -0,0 +1,161 @@
use std::arch::x86_64::*;
use crate::simd_utils;
use crate::{DstImageView, SrcImageView};
pub(crate) fn multiply_alpha_avx2(src_image: &SrcImageView, dst_image: &mut DstImageView) {
let width = src_image.width().get() as usize;
let src_rows = src_image.iter_rows(0, src_image.height().get());
let dst_rows = dst_image.iter_rows_mut();
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
multiply_alpha_row_avx2(src_row, dst_row, width);
}
}
}
pub(crate) fn multiply_alpha_inplace_avx2(image: &mut DstImageView) {
let width = image.width().get() as usize;
for dst_row in image.iter_rows_mut() {
unsafe {
let src_row = std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len());
multiply_alpha_row_avx2(src_row, dst_row, width);
}
}
}
#[allow(dead_code)]
pub(crate) fn multiply_alpha_sse2(src_image: &SrcImageView, dst_image: &mut DstImageView) {
let width = src_image.width().get() as usize;
let src_rows = src_image.iter_rows(0, src_image.height().get());
let dst_rows = dst_image.iter_rows_mut();
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
multiply_alpha_row_sse2(src_row, dst_row, width);
}
}
}
pub(crate) fn multiply_alpha_native(src_image: &SrcImageView, dst_image: &mut DstImageView) {
let src_rows = src_image.iter_rows(0, src_image.height().get());
let dst_rows = dst_image.iter_rows_mut();
for (src_row, dst_row) in src_rows.zip(dst_rows) {
multiply_alpha_row_native(src_row, dst_row);
}
}
pub(crate) fn multiply_alpha_inplace_native(image: &mut DstImageView) {
for dst_row in image.iter_rows_mut() {
let src_row = unsafe { std::slice::from_raw_parts(dst_row.as_ptr(), dst_row.len()) };
multiply_alpha_row_native(src_row, dst_row);
}
}
/// https://github.com/Wizermil/premultiply_alpha/blob/master/premultiply_alpha/premultiply_alpha.hpp#L232
#[target_feature(enable = "avx2")]
unsafe fn multiply_alpha_row_avx2(src_row: &[u32], dst_row: &mut [u32], width: usize) {
let mask_alpha_color_odd_255 = _mm256_set1_epi32(0xff000000u32 as i32);
let div_255 = _mm256_set1_epi16(0x8081u16 as i16);
#[rustfmt::skip]
let mask_shuffle_alpha = _mm256_set_epi8(
15, -1, 15, -1, 11, -1, 11, -1, 7, -1, 7, -1, 3, -1, 3, -1,
15, -1, 15, -1, 11, -1, 11, -1, 7, -1, 7, -1, 3, -1, 3, -1,
);
#[rustfmt::skip]
let mask_shuffle_color_odd = _mm256_set_epi8(
-1, -1, 13, -1, -1, -1, 9, -1, -1, -1, 5, -1, -1, -1, 1, -1,
-1, -1, 13, -1, -1, -1, 9, -1, -1, -1, 5, -1, -1, -1, 1, -1,
);
let mut x: usize = 0;
while x < width.saturating_sub(7) {
let mut color = simd_utils::loadu_si256(src_row, x);
let alpha = _mm256_shuffle_epi8(color, mask_shuffle_alpha);
let mut color_even = _mm256_slli_epi16(color, 8);
let mut color_odd = _mm256_shuffle_epi8(color, mask_shuffle_color_odd);
color_odd = _mm256_or_si256(color_odd, mask_alpha_color_odd_255);
color_odd = _mm256_mulhi_epu16(color_odd, alpha);
color_even = _mm256_mulhi_epu16(color_even, alpha);
color_odd = _mm256_srli_epi16(_mm256_mulhi_epu16(color_odd, div_255), 7);
color_even = _mm256_srli_epi16(_mm256_mulhi_epu16(color_even, div_255), 7);
color = _mm256_or_si256(color_even, _mm256_slli_epi16(color_odd, 8));
let dst_ptr = dst_row.get_unchecked_mut(x..).as_mut_ptr() as *mut __m256i;
_mm256_storeu_si256(dst_ptr, color);
x += 8;
}
let src_tail = &src_row[x..];
let dst_tail = &mut dst_row[x..];
multiply_alpha_row_native(src_tail, dst_tail);
}
/// https://github.com/Wizermil/premultiply_alpha/blob/master/premultiply_alpha/premultiply_alpha.hpp#L108
/// This implementation is twice slowly than native version.
#[allow(dead_code)]
#[target_feature(enable = "sse2")]
unsafe fn multiply_alpha_row_sse2(src_row: &[u32], dst_row: &mut [u32], width: usize) {
let mask_alpha_color_odd_255 = _mm_set1_epi32(0xff000000u32 as i32);
let div_255 = _mm_set1_epi16(0x8081u16 as i16);
let mask_shuffle_alpha =
_mm_set_epi8(15, -1, 15, -1, 11, -1, 11, -1, 7, -1, 7, -1, 3, -1, 3, -1);
let mask_shuffle_color_odd =
_mm_set_epi8(-1, -1, 13, -1, -1, -1, 9, -1, -1, -1, 5, -1, -1, -1, 1, -1);
let mut x: usize = 0;
while x < width.saturating_sub(3) {
let mut color = simd_utils::loadu_si128(src_row, x);
let alpha = _mm_shuffle_epi8(color, mask_shuffle_alpha);
let mut color_even = _mm_slli_epi16(color, 8);
let mut color_odd = _mm_shuffle_epi8(color, mask_shuffle_color_odd);
color_odd = _mm_or_si128(color_odd, mask_alpha_color_odd_255);
color_odd = _mm_mulhi_epu16(color_odd, alpha);
color_even = _mm_mulhi_epu16(color_even, alpha);
color_odd = _mm_srli_epi16(_mm_mulhi_epu16(color_odd, div_255), 7);
color_even = _mm_srli_epi16(_mm_mulhi_epu16(color_even, div_255), 7);
color = _mm_or_si128(color_even, _mm_slli_epi16(color_odd, 8));
let dst_ptr = dst_row.get_unchecked_mut(x..).as_mut_ptr() as *mut __m128i;
_mm_storeu_si128(dst_ptr, color);
x += 4;
}
let src_tail = &src_row[x..];
let dst_tail = &mut dst_row[x..];
multiply_alpha_row_native(src_tail, dst_tail);
}
#[inline(always)]
fn multiply_alpha_row_native(src_row: &[u32], dst_row: &mut [u32]) {
for (src_pixel, dst_pixel) in src_row.iter().zip(dst_row) {
let components: [u8; 4] = src_pixel.to_le_bytes();
let alpha = components[3];
let res: [u8; 4] = [
mul_div_255(components[0], alpha),
mul_div_255(components[1], alpha),
mul_div_255(components[2], alpha),
alpha,
];
*dst_pixel = u32::from_le_bytes(res);
}
}
#[inline(always)]
fn mul_div_255(a: u8, b: u8) -> u8 {
let tmp = a as u32 * b as u32 + 128;
(((tmp >> 8) + tmp) >> 8) as u8
}
+527
View File
@@ -0,0 +1,527 @@
use std::arch::x86_64::*;
use std::intrinsics::transmute;
use crate::convolution::{Bound, Coefficients, Convolution};
use crate::image_view::{DstImageView, FourRows, FourRowsMut, SrcImageView};
use crate::{optimisations, simd_utils};
pub struct Avx2;
// This code is based on C-implementation from Pillow-SIMD package for Python
// https://github.com/uploadcare/pillow-simd
impl Avx2 {
/// For safety, it is necessary to ensure the following conditions:
/// - length of all rows in src_rows must be equal
/// - length of all rows in dst_rows must be equal
/// - bounds.len() == dst_rows.0.len()
/// - coeffs.len() == dst_rows.0.len() * window_size
/// - max(bound.size for bound in bounds) <= window_size
/// - max(bound.start + bound.size for bound in bounds) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
#[inline]
#[target_feature(enable = "avx2")]
unsafe fn horiz_convolution_8u4x(
&self,
src_rows: FourRows,
dst_rows: FourRowsMut,
coeffs: &[i16],
window_size: usize,
bounds: &[Bound],
precision: u8,
) {
let (s_row0, s_row1, s_row2, s_row3) = src_rows;
let (d_row0, d_row1, d_row2, d_row3) = dst_rows;
let zero = _mm256_setzero_si256();
let initial = _mm256_set1_epi32(1 << (precision - 1));
#[rustfmt::skip]
let sh1 = _mm256_set_epi8(
-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0,
-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0,
);
#[rustfmt::skip]
let sh2 = _mm256_set_epi8(
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
);
let coeffs_chunks = coeffs.chunks(window_size);
for (dst_x, (&bound, k)) in bounds.iter().zip(coeffs_chunks).enumerate() {
let x_start = bound.start as usize;
let x_size = bound.size as usize;
let mut x: usize = 0;
let mut sss0 = initial;
let mut sss1 = initial;
while x < x_size.saturating_sub(3) {
let mmk0 = simd_utils::ptr_i16_to_256set1_epi32(k, x);
let mmk1 = simd_utils::ptr_i16_to_256set1_epi32(k, x + 2);
let mut source = _mm256_inserti128_si256(
_mm256_castsi128_si256(simd_utils::loadu_si128(s_row0, x + x_start)),
simd_utils::loadu_si128(s_row1, x + x_start),
1,
);
let mut pix = _mm256_shuffle_epi8(source, sh1);
sss0 = _mm256_add_epi32(sss0, _mm256_madd_epi16(pix, mmk0));
pix = _mm256_shuffle_epi8(source, sh2);
sss0 = _mm256_add_epi32(sss0, _mm256_madd_epi16(pix, mmk1));
source = _mm256_inserti128_si256(
_mm256_castsi128_si256(simd_utils::loadu_si128(s_row2, x + x_start)),
simd_utils::loadu_si128(s_row3, x + x_start),
1,
);
pix = _mm256_shuffle_epi8(source, sh1);
sss1 = _mm256_add_epi32(sss1, _mm256_madd_epi16(pix, mmk0));
pix = _mm256_shuffle_epi8(source, sh2);
sss1 = _mm256_add_epi32(sss1, _mm256_madd_epi16(pix, mmk1));
x += 4;
}
while x < x_size.saturating_sub(1) {
let mmk = simd_utils::ptr_i16_to_256set1_epi32(k, x);
let mut pix = _mm256_inserti128_si256(
_mm256_castsi128_si256(simd_utils::loadl_epi64(s_row0, x + x_start)),
simd_utils::loadl_epi64(s_row1, x + x_start),
1,
);
pix = _mm256_shuffle_epi8(pix, sh1);
sss0 = _mm256_add_epi32(sss0, _mm256_madd_epi16(pix, mmk));
pix = _mm256_inserti128_si256(
_mm256_castsi128_si256(simd_utils::loadl_epi64(s_row2, x + x_start)),
simd_utils::loadl_epi64(s_row3, x + x_start),
1,
);
pix = _mm256_shuffle_epi8(pix, sh1);
sss1 = _mm256_add_epi32(sss1, _mm256_madd_epi16(pix, mmk));
x += 2;
}
while x < x_size {
// [16] xx k0 xx k0 xx k0 xx k0 xx k0 xx k0 xx k0 xx k0
let mmk = _mm256_set1_epi32(*k.get_unchecked(x) as i32);
// [16] xx a0 xx b0 xx g0 xx r0 xx a0 xx b0 xx g0 xx r0
let mut pix = _mm256_inserti128_si256(
_mm256_castsi128_si256(simd_utils::mm_cvtepu8_epi32(s_row0, x + x_start)),
simd_utils::mm_cvtepu8_epi32(s_row1, x + x_start),
1,
);
sss0 = _mm256_add_epi32(sss0, _mm256_madd_epi16(pix, mmk));
pix = _mm256_inserti128_si256(
_mm256_castsi128_si256(simd_utils::mm_cvtepu8_epi32(s_row2, x + x_start)),
simd_utils::mm_cvtepu8_epi32(s_row3, x + x_start),
1,
);
sss1 = _mm256_add_epi32(sss1, _mm256_madd_epi16(pix, mmk));
x += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss0 = _mm256_srai_epi32(sss0, $imm8);
sss1 = _mm256_srai_epi32(sss1, $imm8);
}};
}
constify_imm8!(precision, call);
sss0 = _mm256_packs_epi32(sss0, zero);
sss1 = _mm256_packs_epi32(sss1, zero);
sss0 = _mm256_packus_epi16(sss0, zero);
sss1 = _mm256_packus_epi16(sss1, zero);
*d_row0.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm256_extracti128_si256(sss0, 0)));
*d_row1.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm256_extracti128_si256(sss0, 1)));
*d_row2.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm256_extracti128_si256(sss1, 0)));
*d_row3.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm256_extracti128_si256(sss1, 1)));
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - bounds.len() == dst_row.len()
/// - coeffs.len() == dst_rows.0.len() * window_size
/// - max(bound.start + bound.size for bound in bounds) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
#[inline]
#[target_feature(enable = "avx2")]
unsafe fn horiz_convolution_8u(
&self,
src_row: &[u32],
dst_row: &mut [u32],
coeffs: &[i16],
window_size: usize,
bounds: &[Bound],
precision: u8,
) {
#[rustfmt::skip]
let sh1 = _mm256_set_epi8(
-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0,
-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0,
);
#[rustfmt::skip]
let sh2 = _mm256_set_epi8(
11, 10, 9, 8, 11, 10, 9, 8, 11, 10, 9, 8, 11, 10, 9, 8,
3, 2, 1, 0, 3, 2, 1, 0, 3, 2, 1, 0, 3, 2, 1, 0,
);
#[rustfmt::skip]
let sh3 = _mm256_set_epi8(
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
);
#[rustfmt::skip]
let sh4 = _mm256_set_epi8(
15, 14, 13, 12, 15, 14, 13, 12, 15, 14, 13, 12, 15, 14, 13, 12,
7, 6, 5, 4, 7, 6, 5, 4, 7, 6, 5, 4, 7, 6, 5, 4,
);
#[rustfmt::skip]
let sh5 = _mm256_set_epi8(
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0,
);
#[rustfmt::skip]
let sh6 = _mm256_set_epi8(
7, 6, 5, 4, 7, 6, 5, 4, 7, 6, 5, 4, 7, 6, 5, 4,
3, 2, 1, 0, 3, 2, 1, 0, 3, 2, 1, 0, 3, 2, 1, 0,
);
let sh7 = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
let coeffs_chunks = coeffs.chunks(window_size);
for (xx, (&bound, k)) in bounds.iter().zip(coeffs_chunks).enumerate() {
let x_start = bound.start as usize;
let x_size = bound.size as usize;
let mut x: usize = 0;
let mut sss: __m128i = if x_size < 8 {
_mm_set1_epi32(1 << (precision - 1))
} else {
// Lower part will be added to higher, use only half of the error
let mut sss256 = _mm256_set1_epi32(1 << (precision - 2));
while x < x_size.saturating_sub(7) {
let tmp = simd_utils::loadu_si128(k, x);
let ksource = _mm256_insertf128_si256(_mm256_castsi128_si256(tmp), tmp, 1);
let source = simd_utils::loadu_si256(src_row, x + x_start);
let mut pix = _mm256_shuffle_epi8(source, sh1);
let mut mmk = _mm256_shuffle_epi8(ksource, sh2);
sss256 = _mm256_add_epi32(sss256, _mm256_madd_epi16(pix, mmk));
pix = _mm256_shuffle_epi8(source, sh3);
mmk = _mm256_shuffle_epi8(ksource, sh4);
sss256 = _mm256_add_epi32(sss256, _mm256_madd_epi16(pix, mmk));
x += 8;
}
while x < x_size.saturating_sub(3) {
let tmp = simd_utils::loadl_epi64(k, x);
let ksource = _mm256_insertf128_si256(_mm256_castsi128_si256(tmp), tmp, 1);
let tmp = simd_utils::loadu_si128(src_row, x + x_start);
let source = _mm256_insertf128_si256(_mm256_castsi128_si256(tmp), tmp, 1);
let pix = _mm256_shuffle_epi8(source, sh5);
let mmk = _mm256_shuffle_epi8(ksource, sh6);
sss256 = _mm256_add_epi32(sss256, _mm256_madd_epi16(pix, mmk));
x += 4;
}
_mm_add_epi32(
_mm256_extracti128_si256(sss256, 0),
_mm256_extracti128_si256(sss256, 1),
)
};
while x < x_size.saturating_sub(1) {
let mmk = simd_utils::ptr_i16_to_set1_epi32(k, x);
let source = simd_utils::loadl_epi64(src_row, x + x_start);
let pix = _mm_shuffle_epi8(source, sh7);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
x += 2
}
while x < x_size {
let pix = simd_utils::mm_cvtepu8_epi32(src_row, x + x_start);
let mmk = _mm_set1_epi32(*k.get_unchecked(x) as i32);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
x += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss = _mm_srai_epi32(sss, $imm8);
}};
}
constify_imm8!(precision, call);
sss = _mm_packs_epi32(sss, sss);
*dst_row.get_unchecked_mut(xx) =
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss, sss)));
}
}
#[inline]
#[target_feature(enable = "avx2")]
pub unsafe fn vert_convolution_8u(
&self,
src_img: &SrcImageView,
dst_row: &mut [u32],
coeffs: &[i16],
bound: Bound,
precision: u8,
) {
let src_width = src_img.width().get() as usize;
let y_start = bound.start;
let y_size = bound.size;
let initial = _mm_set1_epi32(1 << (precision - 1));
let initial_256 = _mm256_set1_epi32(1 << (precision - 1));
let mut x: usize = 0;
while x < src_width.saturating_sub(7) {
let mut sss0 = initial_256;
let mut sss1 = initial_256;
let mut sss2 = initial_256;
let mut sss3 = initial_256;
let mut y: u32 = 0;
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
// Load two coefficients at once
let mmk = simd_utils::ptr_i16_to_256set1_epi32(coeffs, y as usize);
let source1 = simd_utils::loadu_si256(s_row1, x); // top line
let source2 = simd_utils::loadu_si256(s_row2, x); // bottom line
let mut source = _mm256_unpacklo_epi8(source1, source2);
let mut pix = _mm256_unpacklo_epi8(source, _mm256_setzero_si256());
sss0 = _mm256_add_epi32(sss0, _mm256_madd_epi16(pix, mmk));
pix = _mm256_unpackhi_epi8(source, _mm256_setzero_si256());
sss1 = _mm256_add_epi32(sss1, _mm256_madd_epi16(pix, mmk));
source = _mm256_unpackhi_epi8(source1, source2);
pix = _mm256_unpacklo_epi8(source, _mm256_setzero_si256());
sss2 = _mm256_add_epi32(sss2, _mm256_madd_epi16(pix, mmk));
pix = _mm256_unpackhi_epi8(source, _mm256_setzero_si256());
sss3 = _mm256_add_epi32(sss3, _mm256_madd_epi16(pix, mmk));
y += 2;
}
for s_row in src_img.iter_rows(y_start + y, y_start + y_size) {
let mmk = _mm256_set1_epi32(coeffs[y as usize] as i32);
let source1 = simd_utils::loadu_si256(s_row, x); // top line
let source2 = _mm256_setzero_si256(); // bottom line is empty
let mut source = _mm256_unpacklo_epi8(source1, source2);
let mut pix = _mm256_unpacklo_epi8(source, _mm256_setzero_si256());
sss0 = _mm256_add_epi32(sss0, _mm256_madd_epi16(pix, mmk));
pix = _mm256_unpackhi_epi8(source, _mm256_setzero_si256());
sss1 = _mm256_add_epi32(sss1, _mm256_madd_epi16(pix, mmk));
source = _mm256_unpackhi_epi8(source1, _mm256_setzero_si256());
pix = _mm256_unpacklo_epi8(source, _mm256_setzero_si256());
sss2 = _mm256_add_epi32(sss2, _mm256_madd_epi16(pix, mmk));
pix = _mm256_unpackhi_epi8(source, _mm256_setzero_si256());
sss3 = _mm256_add_epi32(sss3, _mm256_madd_epi16(pix, mmk));
y += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss0 = _mm256_srai_epi32(sss0, $imm8);
sss1 = _mm256_srai_epi32(sss1, $imm8);
sss2 = _mm256_srai_epi32(sss2, $imm8);
sss3 = _mm256_srai_epi32(sss3, $imm8);
}};
}
constify_imm8!(precision, call);
sss0 = _mm256_packs_epi32(sss0, sss1);
sss2 = _mm256_packs_epi32(sss2, sss3);
sss0 = _mm256_packus_epi16(sss0, sss2);
let dst_ptr = dst_row.get_unchecked_mut(x..).as_mut_ptr() as *mut __m256i;
_mm256_storeu_si256(dst_ptr, sss0);
x += 8;
}
while x < src_width.saturating_sub(1) {
let mut sss0 = initial; // left row
let mut sss1 = initial; // right row
let mut y: u32 = 0;
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
// Load two coefficients at once
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
let source1 = simd_utils::loadl_epi64(s_row1, x); // top line
let source2 = simd_utils::loadl_epi64(s_row2, x); // bottom line
let source = _mm_unpacklo_epi8(source1, source2);
let mut pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
y += 2;
}
for s_row in src_img.iter_rows(y_start + y, y_start + y_size) {
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
let source1 = simd_utils::loadl_epi64(s_row, x); // top line
let source2 = _mm_setzero_si128(); // bottom line is empty
let source = _mm_unpacklo_epi8(source1, source2);
let mut pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
y += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss0 = _mm_srai_epi32(sss0, $imm8);
sss1 = _mm_srai_epi32(sss1, $imm8);
}};
}
constify_imm8!(precision, call);
sss0 = _mm_packs_epi32(sss0, sss1);
sss0 = _mm_packus_epi16(sss0, sss0);
let dst_ptr = dst_row.get_unchecked_mut(x..).as_mut_ptr() as *mut __m128i;
_mm_storel_epi64(dst_ptr, sss0);
x += 2;
}
while x < src_width {
let mut sss = initial;
let mut y: u32 = 0;
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
// Load two coefficients at once
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
let source1 = simd_utils::mm_cvtsi32_si128(s_row1, x); // top line
let source2 = simd_utils::mm_cvtsi32_si128(s_row2, x); // bottom line
let source = _mm_unpacklo_epi8(source1, source2);
let pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
y += 2;
}
for s_row in src_img.iter_rows(y_start + y, y_start + y_size) {
let pix = simd_utils::mm_cvtepu8_epi32(s_row, x);
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
y += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss = _mm_srai_epi32(sss, $imm8);
}};
}
constify_imm8!(precision, call);
sss = _mm_packs_epi32(sss, sss);
*dst_row.get_unchecked_mut(x) =
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss, sss)));
x += 1;
}
}
}
impl Convolution for Avx2 {
#[inline]
fn horiz_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
offset: u32,
coeffs: Coefficients,
) {
let (values, window_size, bounds_per_pixel) =
(coeffs.values, coeffs.window_size, coeffs.bounds);
let mut normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coeffs_i16 = normalizer_guard.normalized();
let dst_height = dst_image.height().get();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_image.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
self.horiz_convolution_8u4x(
src_rows,
dst_rows,
coeffs_i16,
window_size,
&bounds_per_pixel,
precision,
);
}
}
let mut yy = dst_height - dst_height % 4;
while yy < dst_height {
unsafe {
self.horiz_convolution_8u(
src_image.get_row(yy + offset).unwrap(),
dst_image.get_row_mut(yy).unwrap(),
coeffs_i16,
window_size,
&bounds_per_pixel,
precision,
);
}
yy += 1;
}
}
#[inline]
fn vert_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let mut normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coeffs_i16 = normalizer_guard.normalized();
let coeffs_chunks = coeffs_i16.chunks(window_size);
let dst_rows = dst_image.iter_rows_mut();
for ((&bound, k), dst_row) in bounds.iter().zip(coeffs_chunks).zip(dst_rows) {
unsafe {
self.vert_convolution_8u(src_image, dst_row, k, bound, precision);
}
}
}
}
+133
View File
@@ -0,0 +1,133 @@
use std::f64::consts::PI;
pub type FilterFn<'a> = &'a dyn Fn(f64) -> f64;
#[derive(Clone, Copy, Debug, PartialEq)]
#[non_exhaustive]
pub enum FilterType {
/// Each pixel of source image contributes to one pixel of the
/// destination image with identical weights. For upscaling is equivalent
/// of `Nearest` resize algorithm.
Box,
/// Bilinear filter calculate the output pixel value using linear
/// interpolation on all pixels that may contribute to the output value.
Bilinear,
/// Hamming filter has the same performance as `Bilinear` filter while
/// providing the image downscaling quality comparable to bicubic
/// (`CatmulRom` or `Mitchell`). Produces a sharper image than `Bilinear`,
/// doesn't have dislocations on local level like with `Box`.
/// The filter don’t show good quality for the image upscaling.
Hamming,
/// Catmull-Rom bicubic filter calculate the output pixel value using
/// cubic interpolation on all pixels that may contribute to the output
/// value.
CatmulRom,
/// Mitchell–Netravali bicubic filter calculate the output pixel value
/// using cubic interpolation on all pixels that may contribute to the
/// output value.
Mitchell,
/// Lanczos3 filter calculate the output pixel value using a high-quality
/// Lanczos filter (a truncated sinc) on all pixels that may contribute
/// to the output value.
Lanczos3,
}
impl Default for FilterType {
fn default() -> Self {
FilterType::Lanczos3
}
}
/// Returns reference to filter function and value of `filter_support`.
#[inline]
pub fn get_filter_func(filter_type: FilterType) -> (FilterFn<'static>, f64) {
match filter_type {
FilterType::Box => (&box_filter, 0.5),
FilterType::Bilinear => (&bilinear_filter, 1.0),
FilterType::Hamming => (&hamming_filter, 1.0),
FilterType::CatmulRom => (&catmul_filter, 2.0),
FilterType::Mitchell => (&mitchell_filter, 2.0),
FilterType::Lanczos3 => (&lanczos_filter, 3.0),
}
}
#[inline]
fn box_filter(x: f64) -> f64 {
if x > -0.5 && x <= 0.5 {
1.0
} else {
0.0
}
}
#[inline]
fn bilinear_filter(mut x: f64) -> f64 {
x = x.abs();
if x < 1.0 {
1.0 - x
} else {
0.0
}
}
#[inline]
fn hamming_filter(mut x: f64) -> f64 {
x = x.abs();
if x == 0.0 {
1.0
} else if x >= 1.0 {
0.0
} else {
x *= PI;
(0.54 + 0.46 * x.cos()) * x.sin() / x
}
}
/// Catmull-Rom (bicubic) filter
/// https://en.wikipedia.org/wiki/Bicubic_interpolation#Bicubic_convolution_algorithm
#[inline]
fn catmul_filter(mut x: f64) -> f64 {
const A: f64 = -0.5;
x = x.abs();
if x < 1.0 {
((A + 2.) * x - (A + 3.)) * x * x + 1.
} else if x < 2.0 {
(((x - 5.) * x + 8.) * x - 4.) * A
} else {
0.0
}
}
/// Mitchell–Netravali filter (B = C = 1/3)
/// https://en.wikipedia.org/wiki/Mitchell%E2%80%93Netravali_filters
#[inline]
fn mitchell_filter(mut x: f64) -> f64 {
x = x.abs();
if x < 1.0 {
(7. * x / 6. - 2.) * x * x + 16. / 18.
} else if x < 2.0 {
((2. - 7. * x / 18.) * x - 10. / 3.) * x + 16. / 9.
} else {
0.0
}
}
#[inline]
fn sinc_filter(mut x: f64) -> f64 {
if x == 0.0 {
1.0
} else {
x *= PI;
x.sin() / x
}
}
#[inline]
fn lanczos_filter(x: f64) -> f64 {
// truncated sinc
if (-3.0..3.0).contains(&x) {
sinc_filter(x) * sinc_filter(x / 3.)
} else {
0.0
}
}
+40
View File
@@ -0,0 +1,40 @@
macro_rules! constify_imm8 {
($imm8:expr, $expand:ident) => {
#[allow(overflowing_literals)]
match ($imm8) & 0b0011_1111 {
0 => $expand!(0),
1 => $expand!(1),
2 => $expand!(2),
3 => $expand!(3),
4 => $expand!(4),
5 => $expand!(5),
6 => $expand!(6),
7 => $expand!(7),
8 => $expand!(8),
9 => $expand!(9),
10 => $expand!(10),
12 => $expand!(12),
13 => $expand!(13),
14 => $expand!(14),
15 => $expand!(15),
16 => $expand!(16),
17 => $expand!(17),
18 => $expand!(18),
19 => $expand!(19),
20 => $expand!(20),
21 => $expand!(21),
22 => $expand!(22),
23 => $expand!(23),
24 => $expand!(24),
25 => $expand!(25),
26 => $expand!(26),
27 => $expand!(27),
28 => $expand!(28),
29 => $expand!(29),
30 => $expand!(30),
31 => $expand!(31),
32 => $expand!(32),
_ => unreachable!(),
}
};
}
+114
View File
@@ -0,0 +1,114 @@
use std::num::NonZeroU32;
pub use avx2::Avx2;
pub use filters::{get_filter_func, FilterType};
pub use native::{NativeF32, NativeI32, NativeU8x4};
pub use sse4::Sse4;
use crate::image_view::{DstImageView, SrcImageView};
#[macro_use]
mod macros;
mod avx2;
mod filters;
mod native;
mod sse4;
pub trait Convolution {
fn horiz_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
offset: u32,
coeffs: Coefficients,
);
fn vert_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
coeffs: Coefficients,
);
}
#[derive(Debug, Clone, Copy)]
pub struct Bound {
pub start: u32,
pub size: u32,
}
#[derive(Debug, Clone)]
pub struct Coefficients {
pub values: Vec<f64>,
pub window_size: usize,
pub bounds: Vec<Bound>,
}
pub fn precompute_coefficients(
in_size: NonZeroU32,
in0: f64, // Left border for cropping
in1: f64, // Right border for cropping
out_size: NonZeroU32,
filter: &dyn Fn(f64) -> f64,
filter_support: f64,
) -> Coefficients {
let in_size = in_size.get();
let out_size = out_size.get();
let scale = (in1 - in0) / out_size as f64;
let filter_scale = scale.max(1.0);
// Determine filter radius size (length of resampling filter)
let filter_radius = filter_support * filter_scale;
// Maximum number of coeffs per out pixel
let window_size = filter_radius.ceil() as usize * 2 + 1;
// Optimization: replace division by filter_scale
// with multiplication by inv_filter_scale.
let inv_filter_scale = 1.0 / filter_scale;
let count_of_coeffs = window_size * out_size as usize;
let mut coeffs: Vec<f64> = Vec::with_capacity(count_of_coeffs);
let mut bounds: Vec<Bound> = Vec::with_capacity(out_size as usize);
for out_x in 0..out_size {
// Find the point in the input image corresponding to the centre
// of the current pixel in the output image.
let in_center = in0 + (out_x as f64 + 0.5) * scale;
// x_min and x_max are slice bounds for the input pixels relevant
// to the output pixel we are calculating. Pixel x is relevant
// if and only if (x >= x_min) && (x < x_max).
// Invariant: 0 <= x_min < x_max <= width
let x_min = (in_center - filter_radius).floor().max(0.) as u32;
let x_max = (in_center + filter_radius).ceil().min(in_size as f64) as u32;
let cur_index = coeffs.len();
let mut ww: f64 = 0.0;
// Optimisation for follow for-cycle:
// (x + 0.5) - in_center => x - (in_center - 0.5) => x - center
let center = in_center - 0.5;
for x in x_min..x_max {
let w: f64 = filter((x as f64 - center) * inv_filter_scale);
coeffs.push(w);
ww += w;
}
if ww != 0.0 {
coeffs[cur_index..].iter_mut().for_each(|w| *w /= ww);
}
// Remaining values should stay empty if they are used despite of x_max.
coeffs.resize(cur_index + window_size, 0.);
bounds.push(Bound {
start: x_min,
size: x_max - x_min,
});
}
Coefficients {
values: coeffs,
window_size,
bounds,
}
}
+244
View File
@@ -0,0 +1,244 @@
use std::slice;
use crate::convolution::{Coefficients, Convolution};
use crate::image_view::{DstImageView, SrcImageView};
use crate::optimisations;
pub struct NativeU8x4;
impl Convolution for NativeU8x4 {
fn horiz_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
offset: u32,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let mut normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coeffs_i16 = normalizer_guard.normalized();
let dst_rows = dst_image.iter_rows_mut();
for (y_dst, dst_row) in dst_rows.enumerate() {
let y_src = y_dst as u32 + offset;
for (x_dst, (&bound, dst_pixel)) in bounds.iter().zip(dst_row.iter_mut()).enumerate() {
let first_x_src = bound.start;
let start_index = window_size * x_dst;
let end_index = start_index + bound.size as usize;
let ks = &coeffs_i16[start_index..end_index];
let mut ss0 = 1 << (precision - 1);
let mut ss1 = ss0;
let mut ss2 = ss0;
let mut ss3 = ss0;
let src_pixels = src_image.iter_horiz(first_x_src, y_src);
for (&k, &src_pixel) in ks.iter().zip(src_pixels) {
let components: [u8; 4] = src_pixel.to_le_bytes();
ss0 += components[0] as i32 * (k as i32);
ss1 += components[1] as i32 * (k as i32);
ss2 += components[2] as i32 * (k as i32);
ss3 += components[3] as i32 * (k as i32);
}
let t: [u8; 4] = unsafe {
[
optimisations::clip8(ss0, precision),
optimisations::clip8(ss1, precision),
optimisations::clip8(ss2, precision),
optimisations::clip8(ss3, precision),
]
};
*dst_pixel = u32::from_le_bytes(t);
}
}
}
fn vert_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let mut normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coeffs_i16 = normalizer_guard.normalized();
let dst_rows = dst_image.iter_rows_mut();
for (y_dst, (&bound, dst_row)) in bounds.iter().zip(dst_rows).enumerate() {
let first_y_src = bound.start;
let start_index = window_size * y_dst;
let end_index = start_index + bound.size as usize;
let ks = &coeffs_i16[start_index..end_index];
for (x_src, out_pixel) in dst_row.iter_mut().enumerate() {
let mut ss0 = 1 << (precision - 1);
let mut ss1 = ss0;
let mut ss2 = ss0;
let mut ss3 = ss0;
for (dy, &k) in ks.iter().enumerate() {
let pixel = src_image.get_pixel_u32(x_src as u32, first_y_src + dy as u32);
let components: [u8; 4] = pixel.to_le_bytes();
ss0 += components[0] as i32 * (k as i32);
ss1 += components[1] as i32 * (k as i32);
ss2 += components[2] as i32 * (k as i32);
ss3 += components[3] as i32 * (k as i32);
}
let t: [u8; 4] = unsafe {
[
optimisations::clip8(ss0, precision),
optimisations::clip8(ss1, precision),
optimisations::clip8(ss2, precision),
optimisations::clip8(ss3, precision),
]
};
*out_pixel = u32::from_le_bytes(t);
}
}
}
}
pub struct NativeI32;
impl Convolution for NativeI32 {
fn horiz_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
offset: u32,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
for y_dst in 0..dst_image.height().get() {
let y_src = y_dst + offset;
if let Some(out_row) = dst_image.get_row_mut(y_dst) {
let out_row_i32 = unsafe {
let len = out_row.len();
let ptr = out_row.as_mut_ptr();
slice::from_raw_parts_mut(ptr as *mut i32, len)
};
for (x_dst, (&bound, out_pixel)) in bounds.iter().zip(out_row_i32).enumerate() {
let first_x_src = bound.start;
let start_index = window_size * x_dst;
let end_index = start_index + bound.size as usize;
let ks = &values[start_index..end_index];
let mut ss = 0.;
let pixels = src_image.iter_horiz_i32(first_x_src, y_src);
for (&k, &pixel) in ks.iter().zip(pixels) {
ss += pixel as f64 * k;
}
*out_pixel = ss.round() as i32;
}
}
}
}
fn vert_convolution(
&self,
image: &SrcImageView,
out_image: &mut DstImageView,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
for (y_dst, &bound) in bounds.iter().enumerate() {
let first_y_src = bound.start;
let start_index = window_size * y_dst;
let end_index = start_index + bound.size as usize;
let ks = &values[start_index..end_index];
if let Some(out_row) = out_image.get_row_mut(y_dst as u32) {
let out_row_i32 = unsafe {
let len = out_row.len();
let ptr = out_row.as_mut_ptr();
slice::from_raw_parts_mut(ptr as *mut i32, len)
};
for (x_src, out_pixel) in out_row_i32.iter_mut().enumerate() {
let mut ss = 0.;
for (dy, &k) in ks.iter().enumerate() {
let pixel = image.get_pixel_i32(x_src as u32, first_y_src + dy as u32);
ss += pixel as f64 * k;
}
*out_pixel = ss.round() as i32;
}
}
}
}
}
pub struct NativeF32;
impl Convolution for NativeF32 {
fn horiz_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
offset: u32,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
for y_dst in 0..dst_image.height().get() {
let y_src = y_dst + offset;
if let Some(out_row) = dst_image.get_row_mut(y_dst) {
let out_row_f32 = unsafe {
let len = out_row.len();
let ptr = out_row.as_mut_ptr();
slice::from_raw_parts_mut(ptr as *mut f32, len)
};
for (x_dst, (&bound, out_pixel)) in bounds.iter().zip(out_row_f32).enumerate() {
let first_x_src = bound.start;
let start_index = window_size * x_dst;
let end_index = start_index + bound.size as usize;
let ks = &values[start_index..end_index];
let mut ss = 0.;
let pixels = src_image.iter_horiz_f32(first_x_src, y_src);
for (&k, &pixel) in ks.iter().zip(pixels) {
ss += pixel as f64 * k;
}
*out_pixel = ss as f32;
}
}
}
}
fn vert_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
for (y_dst, &bound) in bounds.iter().enumerate() {
let first_y_src = bound.start;
let start_index = window_size * y_dst;
let end_index = start_index + bound.size as usize;
let ks = &values[start_index..end_index];
if let Some(out_row) = dst_image.get_row_mut(y_dst as u32) {
let out_row_f32 = unsafe {
let len = out_row.len();
let ptr = out_row.as_mut_ptr();
slice::from_raw_parts_mut(ptr as *mut f32, len)
};
for (x_src, out_pixel) in out_row_f32.iter_mut().enumerate() {
let mut ss = 0.;
for (dy, &k) in ks.iter().enumerate() {
let pixel = src_image.get_pixel_f32(x_src as u32, first_y_src + dy as u32);
ss += pixel as f64 * k;
}
*out_pixel = ss as f32;
}
}
}
}
}
+545
View File
@@ -0,0 +1,545 @@
use std::arch::x86_64::*;
use std::intrinsics::transmute;
use crate::convolution::{Bound, Coefficients, Convolution};
use crate::image_view::{DstImageView, FourRows, FourRowsMut, SrcImageView};
use crate::{optimisations, simd_utils};
pub struct Sse4;
// This code is based on C-implementation from Pillow-SIMD package for Python
// https://github.com/uploadcare/pillow-simd
impl Sse4 {
/// For safety, it is necessary to ensure the following conditions:
/// - length of all rows in src_rows must be equal
/// - length of all rows in dst_rows must be equal
/// - bounds.len() == dst_rows.0.len()
/// - coeffs.len() == dst_rows.0.len() * window_size
/// - max(bound.size for bound in bounds) <= window_size
/// - max(bound.start + bound.size for bound in bounds) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "sse4.1")]
unsafe fn horiz_convolution_8u4x(
&self,
src_rows: FourRows,
dst_rows: FourRowsMut,
coeffs: &[i16],
window_size: usize,
bounds: &[Bound],
precision: u8,
) {
let (s_row0, s_row1, s_row2, s_row3) = src_rows;
let (d_row0, d_row1, d_row2, d_row3) = dst_rows;
let initial = _mm_set1_epi32(1 << (precision - 1));
let mask_lo = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
let mask_hi = _mm_set_epi8(-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8);
let mask = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
let coeffs_chunks = coeffs.chunks(window_size);
for (xx, (&bound, k)) in bounds.iter().zip(coeffs_chunks).enumerate() {
let x_start = bound.start as usize;
let x_size = bound.size as usize;
let mut x: usize = 0;
let mut sss0 = initial;
let mut sss1 = initial;
let mut sss2 = initial;
let mut sss3 = initial;
while x < x_size.saturating_sub(3) {
let mmk_lo = simd_utils::ptr_i16_to_set1_epi32(k, x);
let mmk_hi = simd_utils::ptr_i16_to_set1_epi32(k, x + 2);
// [8] a3 b3 g3 r3 a2 b2 g2 r2 a1 b1 g1 r1 a0 b0 g0 r0
let mut source = simd_utils::loadu_si128(s_row0, x + x_start);
// [16] a1 a0 b1 b0 g1 g0 r1 r0
let mut pix = _mm_shuffle_epi8(source, mask_lo);
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk_lo));
// [16] a3 a2 b3 b2 g3 g2 r3 r2
pix = _mm_shuffle_epi8(source, mask_hi);
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk_hi));
source = simd_utils::loadu_si128(s_row1, x + x_start);
pix = _mm_shuffle_epi8(source, mask_lo);
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk_lo));
pix = _mm_shuffle_epi8(source, mask_hi);
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk_hi));
source = simd_utils::loadu_si128(s_row2, x + x_start);
pix = _mm_shuffle_epi8(source, mask_lo);
sss2 = _mm_add_epi32(sss2, _mm_madd_epi16(pix, mmk_lo));
pix = _mm_shuffle_epi8(source, mask_hi);
sss2 = _mm_add_epi32(sss2, _mm_madd_epi16(pix, mmk_hi));
source = simd_utils::loadu_si128(s_row3, x + x_start);
pix = _mm_shuffle_epi8(source, mask_lo);
sss3 = _mm_add_epi32(sss3, _mm_madd_epi16(pix, mmk_lo));
pix = _mm_shuffle_epi8(source, mask_hi);
sss3 = _mm_add_epi32(sss3, _mm_madd_epi16(pix, mmk_hi));
x += 4;
}
while x < x_size.saturating_sub(1) {
// [16] k1 k0 k1 k0 k1 k0 k1 k0
let mmk = simd_utils::ptr_i16_to_set1_epi32(k, x);
// [8] x x x x x x x x a1 b1 g1 r1 a0 b0 g0 r0
let mut pix = simd_utils::loadl_epi64(s_row0, x + x_start);
// [16] a1 a0 b1 b0 g1 g0 r1 r0
pix = _mm_shuffle_epi8(pix, mask);
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
pix = simd_utils::loadl_epi64(s_row1, x + x_start);
pix = _mm_shuffle_epi8(pix, mask);
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
pix = simd_utils::loadl_epi64(s_row2, x + x_start);
pix = _mm_shuffle_epi8(pix, mask);
sss2 = _mm_add_epi32(sss2, _mm_madd_epi16(pix, mmk));
pix = simd_utils::loadl_epi64(s_row3, x + x_start);
pix = _mm_shuffle_epi8(pix, mask);
sss3 = _mm_add_epi32(sss3, _mm_madd_epi16(pix, mmk));
x += 2;
}
while x < x_size {
// [16] xx k0 xx k0 xx k0 xx k0
let mmk = _mm_set1_epi32(*k.get_unchecked(x) as i32);
// [16] xx a0 xx b0 xx g0 xx r0
let mut pix = simd_utils::mm_cvtepu8_epi32(s_row0, x);
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
pix = simd_utils::mm_cvtepu8_epi32(s_row1, x);
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
pix = simd_utils::mm_cvtepu8_epi32(s_row2, x);
sss2 = _mm_add_epi32(sss2, _mm_madd_epi16(pix, mmk));
pix = simd_utils::mm_cvtepu8_epi32(s_row3, x);
sss3 = _mm_add_epi32(sss3, _mm_madd_epi16(pix, mmk));
x += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss0 = _mm_srai_epi32(sss0, $imm8);
sss1 = _mm_srai_epi32(sss1, $imm8);
sss2 = _mm_srai_epi32(sss2, $imm8);
sss3 = _mm_srai_epi32(sss3, $imm8);
}};
}
constify_imm8!(precision, call);
sss0 = _mm_packs_epi32(sss0, sss0);
sss1 = _mm_packs_epi32(sss1, sss1);
sss2 = _mm_packs_epi32(sss2, sss2);
sss3 = _mm_packs_epi32(sss3, sss3);
*d_row0.get_unchecked_mut(xx) =
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss0, sss0)));
*d_row1.get_unchecked_mut(xx) =
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss1, sss1)));
*d_row2.get_unchecked_mut(xx) =
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss2, sss2)));
*d_row3.get_unchecked_mut(xx) =
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss3, sss3)));
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - bounds.len() == dst_row.len()
/// - coeffs.len() == dst_rows.0.len() * window_size
/// - max(bound.start + bound.size for bound in bounds) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "sse4.1")]
unsafe fn horiz_convolution_8u(
&self,
src_row: &[u32],
dst_row: &mut [u32],
coeffs: &[i16],
window_size: usize,
bounds: &[Bound],
precision: u8,
) {
let coeffs_chunks = coeffs.chunks(window_size);
let initial = _mm_set1_epi32(1 << (precision - 1));
let sh1 = _mm_set_epi8(-1, 11, -1, 3, -1, 10, -1, 2, -1, 9, -1, 1, -1, 8, -1, 0);
let sh2 = _mm_set_epi8(5, 4, 1, 0, 5, 4, 1, 0, 5, 4, 1, 0, 5, 4, 1, 0);
let sh3 = _mm_set_epi8(-1, 15, -1, 7, -1, 14, -1, 6, -1, 13, -1, 5, -1, 12, -1, 4);
let sh4 = _mm_set_epi8(7, 6, 3, 2, 7, 6, 3, 2, 7, 6, 3, 2, 7, 6, 3, 2);
let sh5 = _mm_set_epi8(13, 12, 9, 8, 13, 12, 9, 8, 13, 12, 9, 8, 13, 12, 9, 8);
let sh6 = _mm_set_epi8(
15, 14, 11, 10, 15, 14, 11, 10, 15, 14, 11, 10, 15, 14, 11, 10,
);
let sh7 = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
for (xx, (&bound, k)) in bounds.iter().zip(coeffs_chunks).enumerate() {
let x_start = bound.start as usize;
let x_size = bound.size as usize;
let mut x: usize = 0;
let mut sss = initial;
while x < x_size.saturating_sub(7) {
let ksource = simd_utils::loadu_si128(k, x);
let mut source = simd_utils::loadu_si128(src_row, x + x_start);
let mut pix = _mm_shuffle_epi8(source, sh1);
let mut mmk = _mm_shuffle_epi8(ksource, sh2);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
pix = _mm_shuffle_epi8(source, sh3);
mmk = _mm_shuffle_epi8(ksource, sh4);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
source = simd_utils::loadu_si128(src_row, x + 4 + x_start);
pix = _mm_shuffle_epi8(source, sh1);
mmk = _mm_shuffle_epi8(ksource, sh5);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
pix = _mm_shuffle_epi8(source, sh3);
mmk = _mm_shuffle_epi8(ksource, sh6);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
x += 8;
}
while x < x_size.saturating_sub(3) {
let source = simd_utils::loadu_si128(src_row, x + x_start);
let ksource = simd_utils::loadl_epi64(k, x);
let mut pix = _mm_shuffle_epi8(source, sh1);
let mut mmk = _mm_shuffle_epi8(ksource, sh2);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
pix = _mm_shuffle_epi8(source, sh3);
mmk = _mm_shuffle_epi8(ksource, sh4);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
x += 4;
}
while x < x_size.saturating_sub(1) {
let mmk = simd_utils::ptr_i16_to_set1_epi32(k, x);
let source = simd_utils::loadl_epi64(src_row, x + x_start);
let pix = _mm_shuffle_epi8(source, sh7);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
x += 2
}
while x < x_size {
let pix = simd_utils::mm_cvtepu8_epi32(src_row, x + x_start);
let mmk = _mm_set1_epi32(*k.get_unchecked(x) as i32);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
x += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss = _mm_srai_epi32(sss, $imm8);
}};
}
constify_imm8!(precision, call);
sss = _mm_packs_epi32(sss, sss);
*dst_row.get_unchecked_mut(xx) =
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss, sss)));
}
}
#[target_feature(enable = "sse4.1")]
pub unsafe fn vert_convolution_8u(
&self,
src_img: &SrcImageView,
dst_row: &mut [u32],
coeffs: &[i16],
bound: Bound,
precision: u8,
) {
let mut xx: usize = 0;
let src_width = src_img.width().get() as usize;
let y_start = bound.start;
let y_size = bound.size;
let initial = _mm_set1_epi32(1 << (precision - 1));
while xx < src_width.saturating_sub(7) {
let mut sss0 = initial;
let mut sss1 = initial;
let mut sss2 = initial;
let mut sss3 = initial;
let mut sss4 = initial;
let mut sss5 = initial;
let mut sss6 = initial;
let mut sss7 = initial;
let mut y: u32 = 0;
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
// Load two coefficients at once
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
let mut source1 = simd_utils::loadu_si128(s_row1, xx); // top line
let mut source2 = simd_utils::loadu_si128(s_row2, xx); // bottom line
let mut source = _mm_unpacklo_epi8(source1, source2);
let mut pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
source = _mm_unpackhi_epi8(source1, source2);
pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss2 = _mm_add_epi32(sss2, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss3 = _mm_add_epi32(sss3, _mm_madd_epi16(pix, mmk));
source1 = simd_utils::loadu_si128(s_row1, xx + 4); // top line
source2 = simd_utils::loadu_si128(s_row2, xx + 4); // bottom line
source = _mm_unpacklo_epi8(source1, source2);
pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss4 = _mm_add_epi32(sss4, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss5 = _mm_add_epi32(sss5, _mm_madd_epi16(pix, mmk));
source = _mm_unpackhi_epi8(source1, source2);
pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss6 = _mm_add_epi32(sss6, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss7 = _mm_add_epi32(sss7, _mm_madd_epi16(pix, mmk));
y += 2;
}
for s_row in src_img.iter_rows(y_start + y, y_start + y_size) {
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
let mut source1 = simd_utils::loadu_si128(s_row, xx); // top line
let mut source = _mm_unpacklo_epi8(source1, _mm_setzero_si128());
let mut pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
source = _mm_unpackhi_epi8(source1, _mm_setzero_si128());
pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss2 = _mm_add_epi32(sss2, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss3 = _mm_add_epi32(sss3, _mm_madd_epi16(pix, mmk));
source1 = simd_utils::loadu_si128(s_row, xx + 4); // top line
source = _mm_unpacklo_epi8(source1, _mm_setzero_si128());
pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss4 = _mm_add_epi32(sss4, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss5 = _mm_add_epi32(sss5, _mm_madd_epi16(pix, mmk));
source = _mm_unpackhi_epi8(source1, _mm_setzero_si128());
pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss6 = _mm_add_epi32(sss6, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss7 = _mm_add_epi32(sss7, _mm_madd_epi16(pix, mmk));
y += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss0 = _mm_srai_epi32(sss0, $imm8);
sss1 = _mm_srai_epi32(sss1, $imm8);
sss2 = _mm_srai_epi32(sss2, $imm8);
sss3 = _mm_srai_epi32(sss3, $imm8);
sss4 = _mm_srai_epi32(sss4, $imm8);
sss5 = _mm_srai_epi32(sss5, $imm8);
sss6 = _mm_srai_epi32(sss6, $imm8);
sss7 = _mm_srai_epi32(sss7, $imm8);
}};
}
constify_imm8!(precision, call);
sss0 = _mm_packs_epi32(sss0, sss1);
sss2 = _mm_packs_epi32(sss2, sss3);
sss0 = _mm_packus_epi16(sss0, sss2);
let dst_ptr = dst_row.get_unchecked_mut(xx..).as_mut_ptr() as *mut __m128i;
_mm_storeu_si128(dst_ptr, sss0);
sss4 = _mm_packs_epi32(sss4, sss5);
sss6 = _mm_packs_epi32(sss6, sss7);
sss4 = _mm_packus_epi16(sss4, sss6);
let dst_ptr = dst_row.get_unchecked_mut(xx + 4..).as_mut_ptr() as *mut __m128i;
_mm_storeu_si128(dst_ptr, sss4);
xx += 8;
}
while xx < src_width.saturating_sub(1) {
let mut sss0 = initial; // left row
let mut sss1 = initial; // right row
let mut y: u32 = 0;
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
// Load two coefficients at once
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
let source1 = simd_utils::loadl_epi64(s_row1, xx); // top line
let source2 = simd_utils::loadl_epi64(s_row2, xx); // bottom line
let source = _mm_unpacklo_epi8(source1, source2);
let mut pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
y += 2;
}
for s_row1 in src_img.iter_rows(y_start + y, y_start + y_size) {
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
let source1 = simd_utils::loadl_epi64(s_row1, xx); // top line
let source = _mm_unpacklo_epi8(source1, _mm_setzero_si128());
let mut pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
y += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss0 = _mm_srai_epi32(sss0, $imm8);
sss1 = _mm_srai_epi32(sss1, $imm8);
}};
}
constify_imm8!(precision, call);
sss0 = _mm_packs_epi32(sss0, sss1);
sss0 = _mm_packus_epi16(sss0, sss0);
let dst_ptr = dst_row.get_unchecked_mut(xx..).as_mut_ptr() as *mut __m128i;
_mm_storel_epi64(dst_ptr, sss0);
//
xx += 2;
}
while xx < src_width {
let mut sss = initial;
let mut y: u32 = 0;
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
// Load two coefficients at once
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
let source1 = simd_utils::mm_cvtsi32_si128(s_row1, xx); // top line
let source2 = simd_utils::mm_cvtsi32_si128(s_row2, xx); // bottom line
let source = _mm_unpacklo_epi8(source1, source2);
let pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
y += 2;
}
for s_row in src_img.iter_rows(y_start + y, y_start + y_size) {
let pix = simd_utils::mm_cvtepu8_epi32(s_row, xx);
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
y += 1;
}
macro_rules! call {
($imm8:expr) => {{
sss = _mm_srai_epi32(sss, $imm8);
}};
}
constify_imm8!(precision, call);
sss = _mm_packs_epi32(sss, sss);
*dst_row.get_unchecked_mut(xx) =
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss, sss)));
xx += 1;
}
}
}
impl Convolution for Sse4 {
#[inline]
fn horiz_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
offset: u32,
coeffs: Coefficients,
) {
let (values, window_size, bounds_per_pixel) =
(coeffs.values, coeffs.window_size, coeffs.bounds);
let mut normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coeffs_i16 = normalizer_guard.normalized();
let dst_height = dst_image.height().get();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_image.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
self.horiz_convolution_8u4x(
src_rows,
dst_rows,
coeffs_i16,
window_size,
&bounds_per_pixel,
precision,
);
}
}
let mut yy = dst_height - dst_height % 4;
while yy < dst_height {
unsafe {
self.horiz_convolution_8u(
src_image.get_row(yy + offset).unwrap(),
dst_image.get_row_mut(yy).unwrap(),
coeffs_i16,
window_size,
&bounds_per_pixel,
precision,
);
}
yy += 1;
}
}
#[inline]
fn vert_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let mut normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coeffs_i16 = normalizer_guard.normalized();
let coeffs_chunks = coeffs_i16.chunks(window_size);
let dst_rows = dst_image.iter_rows_mut();
for ((&bound, k), dst_row) in bounds.iter().zip(coeffs_chunks).zip(dst_rows) {
unsafe {
self.vert_convolution_8u(src_image, dst_row, k, bound, precision);
}
}
}
}
+15
View File
@@ -0,0 +1,15 @@
use thiserror::Error;
#[derive(Error, Debug, Clone, Copy)]
pub enum ImageError {
#[error("Buffer size don't corresponds to image dimensions")]
InvalidBufferSize,
}
#[derive(Error, Debug, Clone, Copy)]
pub enum CropBoxError {
#[error("Position of the crop box is out of the image boundaries")]
PositionIsOutOfImageBoundaries,
#[error("Size of the crop box is out of the image boundaries")]
SizeIsOutOfImageBoundaries,
}
+80
View File
@@ -0,0 +1,80 @@
use std::num::NonZeroU32;
use crate::{DstImageView, ImageError, PixelType, SrcImageView};
#[derive(Debug, Clone)]
pub struct ImageData<T: AsRef<[u8]>> {
width: NonZeroU32,
height: NonZeroU32,
pixels: T,
pixel_type: PixelType,
}
impl<T: AsRef<[u8]>> ImageData<T> {
pub fn new(
width: NonZeroU32,
height: NonZeroU32,
pixels: T,
pixel_type: PixelType,
) -> Result<Self, ImageError> {
let size = (width.get() * height.get()) as usize * 4;
if pixels.as_ref().len() != size {
return Err(ImageError::InvalidBufferSize);
}
Ok(Self {
width,
height,
pixels,
pixel_type,
})
}
#[inline(always)]
pub fn pixel_type(&self) -> PixelType {
self.pixel_type
}
#[inline(always)]
pub fn width(&self) -> NonZeroU32 {
self.width
}
#[inline(always)]
pub fn height(&self) -> NonZeroU32 {
self.height
}
#[inline(always)]
pub fn get_buffer(&self) -> &[u8] {
self.pixels.as_ref()
}
#[inline(always)]
pub fn src_view(&self) -> SrcImageView {
let pixels = unsafe { self.pixels.as_ref().align_to::<u32>().1 };
let rows = pixels.chunks(self.width.get() as usize).collect();
SrcImageView::new(self.width, self.height, rows, self.pixel_type).unwrap()
}
}
impl<T: AsRef<[u8]> + AsMut<[u8]>> ImageData<T> {
#[inline(always)]
pub fn dst_view(&mut self) -> DstImageView {
let pixels = unsafe { self.pixels.as_mut().align_to_mut::<u32>().1 };
let rows = pixels.chunks_mut(self.width.get() as usize).collect();
DstImageView::new(self.width, self.height, rows, self.pixel_type).unwrap()
}
}
impl ImageData<Vec<u8>> {
pub fn new_owned(width: NonZeroU32, height: NonZeroU32, pixel_type: PixelType) -> Self {
let size = (width.get() * height.get()) as usize * 4;
let pixels = vec![0; size];
Self {
width,
height,
pixels,
pixel_type,
}
}
}
+309
View File
@@ -0,0 +1,309 @@
use std::mem::transmute;
use std::num::NonZeroU32;
use std::slice;
use crate::errors::{CropBoxError, ImageError};
pub type TwoRows<'a> = (&'a [u32], &'a [u32]);
pub type FourRows<'a> = (&'a [u32], &'a [u32], &'a [u32], &'a [u32]);
pub type RowMut<'a, 'b> = &'b mut &'a mut [u32];
pub type FourRowsMut<'a, 'b> = (
&'b mut &'a mut [u32],
&'b mut &'a mut [u32],
&'b mut &'a mut [u32],
&'b mut &'a mut [u32],
);
#[derive(Debug, Clone, Copy, PartialEq)]
pub enum PixelType {
U8x4,
I32,
F32,
}
#[derive(Debug, Clone, Copy)]
pub struct CropBox {
pub left: u32,
pub top: u32,
pub width: NonZeroU32,
pub height: NonZeroU32,
}
/// An immutable view of image data used by resizer as source image.
#[derive(Debug, Clone)]
pub struct SrcImageView<'a> {
width: NonZeroU32,
height: NonZeroU32,
crop_box: CropBox,
rows: Vec<&'a [u32]>,
pixel_type: PixelType,
}
/// An mutable view of image data used by resizer as destination image.
#[derive(Debug)]
pub struct DstImageView<'a> {
width: NonZeroU32,
height: NonZeroU32,
rows: Vec<&'a mut [u32]>,
pixel_type: PixelType,
}
impl<'a> SrcImageView<'a> {
#[inline(always)]
pub fn new(
width: NonZeroU32,
height: NonZeroU32,
rows: Vec<&'a [u32]>,
pixel_type: PixelType,
) -> Result<Self, ImageError> {
if rows.len() != height.get() as usize {
return Err(ImageError::InvalidBufferSize);
}
let row_size = width.get() as usize;
if rows.iter().any(|row| row.len() != row_size) {
return Err(ImageError::InvalidBufferSize);
}
Ok(Self {
width,
height,
crop_box: CropBox {
left: 0,
top: 0,
width,
height,
},
rows,
pixel_type,
})
}
#[inline(always)]
pub fn pixel_type(&self) -> PixelType {
self.pixel_type
}
#[inline(always)]
pub fn width(&self) -> NonZeroU32 {
self.width
}
#[inline(always)]
pub fn height(&self) -> NonZeroU32 {
self.height
}
#[inline(always)]
pub fn crop_box(&self) -> CropBox {
self.crop_box
}
pub fn set_crop_box(&mut self, crop_box: CropBox) -> Result<(), CropBoxError> {
if crop_box.left >= self.width.get() || crop_box.top >= self.height.get() {
return Err(CropBoxError::PositionIsOutOfImageBoundaries);
}
let right = crop_box.left + crop_box.width.get();
let bottom = crop_box.top + crop_box.height.get();
if right > self.width.get() || bottom > self.height.get() {
return Err(CropBoxError::SizeIsOutOfImageBoundaries);
}
self.crop_box = crop_box;
Ok(())
}
#[inline(always)]
pub fn get_buffer(&self) -> Vec<u8> {
let row_size = self.width.get() as usize;
self.rows
.iter()
.map(|row| unsafe { row[0..row_size].align_to::<u8>().1 })
.flatten()
.copied()
.collect()
}
#[inline]
pub(crate) fn get_pixel_u32(&self, x: u32, y: u32) -> u32 {
self.rows[y as usize][x as usize]
}
#[inline(always)]
pub(crate) fn get_pixel_i32(&self, x: u32, y: u32) -> i32 {
unsafe { transmute(self.get_pixel_u32(x, y)) }
}
#[inline(always)]
pub(crate) fn get_pixel_f32(&self, x: u32, y: u32) -> f32 {
f32::from_bits(self.get_pixel_u32(x, y))
}
#[inline(always)]
pub(crate) fn iter_4_rows(
&'a self,
start_y: u32,
max_y: u32,
) -> impl Iterator<Item = FourRows<'a>> {
let start_y = start_y as usize;
let max_y = max_y.min(self.height.get()) as usize;
let rows = self.rows.get(start_y..max_y).unwrap_or_else(|| &[]);
rows.chunks_exact(4).map(|rows| match *rows {
[r0, r1, r2, r3] => (r0, r1, r2, r3),
_ => unreachable!(),
})
}
#[inline(always)]
pub(crate) fn iter_2_rows(
&'a self,
start_y: u32,
max_y: u32,
) -> impl Iterator<Item = TwoRows<'a>> {
let start_y = start_y as usize;
let max_y = max_y.min(self.height.get()) as usize;
let rows = self.rows.get(start_y..max_y).unwrap_or_else(|| &[]);
rows.chunks_exact(2).map(|rows| match *rows {
[r0, r1] => (r0, r1),
_ => unreachable!(),
})
}
#[inline(always)]
pub(crate) fn iter_rows(&'a self, start_y: u32, max_y: u32) -> impl Iterator<Item = &'a [u32]> {
let start_y = start_y as usize;
let max_y = max_y.min(self.height.get()) as usize;
let rows = self.rows.get(start_y..max_y).unwrap_or_else(|| &[]);
rows.iter().copied()
}
#[inline(always)]
pub(crate) fn iter_horiz(&self, x: u32, y: u32) -> &[u32] {
if let Some(&row) = self.rows.get(y as usize) {
let start_pos = x as usize;
if let Some(res) = row.get(start_pos..) {
return res;
}
}
&[]
}
#[inline]
pub(crate) fn iter_horiz_i32(&self, x: u32, y: u32) -> &[i32] {
let row = self.iter_horiz(x, y);
let ptr = row.as_ptr();
unsafe { slice::from_raw_parts(ptr as *const i32, row.len()) }
}
#[inline]
pub(crate) fn iter_horiz_f32(&self, x: u32, y: u32) -> &[f32] {
let row = self.iter_horiz(x, y);
let ptr = row.as_ptr();
unsafe { slice::from_raw_parts(ptr as *const f32, row.len()) }
}
#[inline(always)]
pub(crate) fn get_row(&self, y: u32) -> Option<&[u32]> {
self.rows.get(y as usize).copied()
}
#[inline(always)]
pub(crate) fn iter_rows_with_step(
&self,
mut y: f64,
step: f64,
max_count: usize,
) -> impl Iterator<Item = &[u32]> {
let steps = (self.height.get() as f64 - y) / step;
let steps = (steps.max(0.).ceil() as usize).min(max_count);
(0..steps).map(move |_| {
// Safety of value of y guaranteed by calculation of steps count
let row = unsafe { *self.rows.get_unchecked(y as usize) };
y += step;
row
})
}
}
impl<'a> DstImageView<'a> {
#[inline(always)]
pub fn new(
width: NonZeroU32,
height: NonZeroU32,
rows: Vec<&'a mut [u32]>,
pixel_type: PixelType,
) -> Result<Self, ImageError> {
if rows.len() != height.get() as usize {
return Err(ImageError::InvalidBufferSize);
}
let row_size = width.get() as usize;
if rows.iter().any(|row| row.len() != row_size) {
return Err(ImageError::InvalidBufferSize);
}
Ok(Self {
width,
height,
rows,
pixel_type,
})
}
#[inline(always)]
pub fn pixel_type(&self) -> PixelType {
self.pixel_type
}
#[inline(always)]
pub fn width(&self) -> NonZeroU32 {
self.width
}
#[inline(always)]
pub fn height(&self) -> NonZeroU32 {
self.height
}
#[inline(always)]
pub(crate) fn iter_rows_mut(&mut self) -> slice::IterMut<&'a mut [u32]> {
self.rows.iter_mut()
}
#[inline(always)]
pub(crate) fn iter_4_rows_mut(&mut self) -> impl Iterator<Item = FourRowsMut<'a, '_>> {
self.rows.chunks_exact_mut(4).map(|rows| match rows {
[a, b, c, d] => (a, b, c, d),
_ => unreachable!(),
})
}
#[inline(always)]
pub(crate) fn get_row_mut(&mut self, y: u32) -> Option<RowMut<'a, '_>> {
self.rows.get_mut(y as usize)
}
}
// pub struct FourRowsIterator<'a> {
// chunks: slice::ChunksExact<'a, &'a [u32]>,
// }
//
// impl<'a> FourRowsIterator<'a> {
// #[inline(always)]
// fn new(image: &'a SrcImageView<'a>, start_y: u32, max_y: u32) -> Self {
// let start_y = start_y as usize;
// let max_y = max_y as usize;
// let max_y = max_y.min(image.height.get() as usize);
// let rows = unsafe { image.rows.get_unchecked(start_y..max_y) };
// Self {
// chunks: rows.chunks_exact(4),
// }
// }
// }
//
// impl<'a> Iterator for FourRowsIterator<'a> {
// type Item = FourRows<'a>;
//
// #[inline(always)]
// fn next(&mut self) -> Option<Self::Item> {
// match self.chunks.next() {
// Some(&[r0, r1, r2, r3]) => Some((r0, r1, r2, r3)),
// _ => None,
// }
// }
// }
+15
View File
@@ -0,0 +1,15 @@
pub use alpha::{MulDiv, MulDivImageError, MulDivImagesError};
pub use convolution::FilterType;
pub use errors::{CropBoxError, ImageError};
pub use image_data::ImageData;
pub use image_view::{CropBox, DstImageView, PixelType, SrcImageView};
pub use resizer::{CpuExtensions, ResizeAlg, Resizer};
mod alpha;
mod convolution;
mod errors;
mod image_data;
mod image_view;
mod optimisations;
mod resizer;
mod simd_utils;
+130
View File
@@ -0,0 +1,130 @@
use std::slice;
/* Handles values form -640 to 639. */
const CLIP8_LOOKUPS: [u8; 1280] = [
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25,
26, 27, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 47, 48, 49,
50, 51, 52, 53, 54, 55, 56, 57, 58, 59, 60, 61, 62, 63, 64, 65, 66, 67, 68, 69, 70, 71, 72, 73,
74, 75, 76, 77, 78, 79, 80, 81, 82, 83, 84, 85, 86, 87, 88, 89, 90, 91, 92, 93, 94, 95, 96, 97,
98, 99, 100, 101, 102, 103, 104, 105, 106, 107, 108, 109, 110, 111, 112, 113, 114, 115, 116,
117, 118, 119, 120, 121, 122, 123, 124, 125, 126, 127, 128, 129, 130, 131, 132, 133, 134, 135,
136, 137, 138, 139, 140, 141, 142, 143, 144, 145, 146, 147, 148, 149, 150, 151, 152, 153, 154,
155, 156, 157, 158, 159, 160, 161, 162, 163, 164, 165, 166, 167, 168, 169, 170, 171, 172, 173,
174, 175, 176, 177, 178, 179, 180, 181, 182, 183, 184, 185, 186, 187, 188, 189, 190, 191, 192,
193, 194, 195, 196, 197, 198, 199, 200, 201, 202, 203, 204, 205, 206, 207, 208, 209, 210, 211,
212, 213, 214, 215, 216, 217, 218, 219, 220, 221, 222, 223, 224, 225, 226, 227, 228, 229, 230,
231, 232, 233, 234, 235, 236, 237, 238, 239, 240, 241, 242, 243, 244, 245, 246, 247, 248, 249,
250, 251, 252, 253, 254, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
];
// 8 bits for result. Filter can have negative areas.
// In one cases the sum of the coefficients will be negative,
// in the other it will be more than 1.0. That is why we need
// two extra bits for overflow and i32 type.
const PRECISION_BITS: u8 = 32 - 8 - 2;
/* We use signed INT16 type to store coefficients. */
const MAX_COEFS_PRECISION: u8 = 16 - 1;
/// The function must be used with the ``v`` and ``precision`` values
/// such that the expression
/// ```
/// v >> precision
/// ```
/// produces a result in the range ``[-512, 511]``.
#[inline(always)]
pub unsafe fn clip8(v: i32, precision: u8) -> u8 {
let index = (640 + (v >> precision)) as usize;
// index must be in range [(640-512)..(640+511)]
*CLIP8_LOOKUPS.get_unchecked(index)
}
/// Converts `Vec<f64>` into `&[i16]` without additional memory allocations.
/// The memory buffer from `Vec<f64>` uses as `[i16]` .
pub struct NormalizerGuard {
values: Vec<f64>,
precision: u8,
}
impl NormalizerGuard {
#[inline]
pub fn new(mut values: Vec<f64>) -> Self {
let max_weight = values
.iter()
.max_by(|&x, &y| x.partial_cmp(&y).unwrap())
.unwrap_or(&0.0)
.to_owned();
let mut precision = 0u8;
for cur_precision in 0..PRECISION_BITS {
precision = cur_precision;
let next_value: i32 = (max_weight * (1 << (precision + 1)) as f64).round() as i32;
// The next value will be outside of the range, so just stop
if next_value >= (1 << MAX_COEFS_PRECISION) {
break;
}
}
let len = values.len();
let ptr = values.as_mut_ptr();
// Size of `[i16]` always will be not greater than `[f64]` with same number of items
let values_i16 = unsafe { slice::from_raw_parts_mut(ptr as *mut i16, len) };
let scale = (1 << precision) as f64;
for (&src, dst) in values.iter().zip(values_i16.iter_mut()) {
*dst = (src * scale).round() as i16;
}
Self { values, precision }
}
#[inline]
pub fn normalized(&mut self) -> &[i16] {
let len = self.values.len();
let ptr = self.values.as_mut_ptr();
unsafe { slice::from_raw_parts_mut(ptr as *mut i16, len) }
}
#[inline]
pub fn precision(&self) -> u8 {
self.precision
}
}
+292
View File
@@ -0,0 +1,292 @@
use std::num::NonZeroU32;
use crate::convolution::{self, Convolution, FilterType};
use crate::image_data::ImageData;
use crate::image_view::{DstImageView, PixelType, SrcImageView};
#[derive(Debug, Clone, Copy)]
pub enum CpuExtensions {
None,
Sse2,
Sse4_1,
Avx2,
}
impl Default for CpuExtensions {
fn default() -> Self {
if is_x86_feature_detected!("avx2") {
Self::Avx2
} else if is_x86_feature_detected!("sse4.1") {
Self::Sse4_1
} else if is_x86_feature_detected!("sse2") {
Self::Sse2
} else {
Self::None
}
}
}
impl CpuExtensions {
#[inline]
fn get_resampler(&self, pixel_type: PixelType) -> &dyn Convolution {
match pixel_type {
PixelType::U8x4 => match self {
Self::Sse4_1 => &convolution::Sse4,
Self::Avx2 => &convolution::Avx2,
_ => &convolution::NativeU8x4,
},
PixelType::I32 => &convolution::NativeI32,
PixelType::F32 => &convolution::NativeF32,
}
}
}
#[derive(Debug, Clone, Copy)]
#[non_exhaustive]
pub enum ResizeAlg {
Nearest,
Convolution(FilterType),
SuperSampling(FilterType, u8),
}
impl Default for ResizeAlg {
fn default() -> Self {
Self::Convolution(FilterType::Lanczos3)
}
}
#[derive(Default, Debug, Clone)]
pub struct Resizer {
pub algorithm: ResizeAlg,
cpu_extensions: CpuExtensions,
convolution_buffer: Vec<u8>,
super_sampling_buffer: Vec<u8>,
}
impl Resizer {
pub fn new(algorithm: ResizeAlg) -> Self {
Self {
algorithm,
..Default::default()
}
}
pub fn resize(&mut self, src_image: &SrcImageView, dst_image: &mut DstImageView) {
match self.algorithm {
ResizeAlg::Nearest => resample_nearest(&src_image, dst_image),
ResizeAlg::Convolution(filter_type) => {
let convolution_buffer = &mut self.convolution_buffer;
resample_convolution(
src_image,
dst_image,
filter_type,
self.cpu_extensions,
convolution_buffer,
)
}
ResizeAlg::SuperSampling(filter_type, multiplicity) => {
let convolution_buffer = &mut self.convolution_buffer;
let super_sampling_buffer = &mut self.super_sampling_buffer;
resample_super_sampling(
src_image,
dst_image,
filter_type,
multiplicity,
self.cpu_extensions,
super_sampling_buffer,
convolution_buffer,
)
}
}
}
/// Returns the size of internal buffers used to store the results of
/// intermediate resizing steps.
pub fn size_of_internal_buffers(&self) -> usize {
self.convolution_buffer.len() + self.super_sampling_buffer.len()
}
/// Deallocates the internal buffers used to store the results of
/// intermediate resizing steps.
pub fn reset_internal_buffers(&mut self) {
if self.convolution_buffer.capacity() > 0 {
self.convolution_buffer = Vec::new();
}
if self.super_sampling_buffer.capacity() > 0 {
self.super_sampling_buffer = Vec::new();
}
}
#[inline(always)]
pub fn cpu_extensions(&self) -> CpuExtensions {
self.cpu_extensions
}
/// # Safety
/// This is unsafe because this method allows you to set a CPU-extensions
/// that is not actually supported by your CPU.
pub unsafe fn set_cpu_extensions(&mut self, extensions: CpuExtensions) {
self.cpu_extensions = extensions;
}
}
fn get_temp_image_from_buffer(
buffer: &mut Vec<u8>,
width: NonZeroU32,
height: NonZeroU32,
pixel_type: PixelType,
) -> ImageData<&mut [u8]> {
let buf_size = (width.get() * height.get()) as usize * 4;
if buffer.len() < buf_size {
buffer.resize(buf_size, 0);
}
let pixels = &mut buffer[0..buf_size];
ImageData::new(width, height, pixels, pixel_type).unwrap()
}
fn resample_nearest(src_image: &SrcImageView, dst_image: &mut DstImageView) {
let crop_box = src_image.crop_box();
let dst_width = dst_image.width().get();
let x_scale = crop_box.width.get() as f64 / dst_width as f64;
let y_scale = crop_box.height.get() as f64 / dst_image.height().get() as f64;
// Pretabulate horizontal pixel positions
let x_in_start = crop_box.left as f64 + x_scale * 0.5;
let max_src_x = src_image.width().get() as usize;
let x_in_tab: Vec<usize> = (0..dst_width)
.map(|x| ((x_in_start + x_scale * x as f64) as usize).min(max_src_x))
.collect();
let y_in_start = crop_box.top as f64 + y_scale * 0.5;
let src_rows =
src_image.iter_rows_with_step(y_in_start, y_scale, dst_image.height().get() as usize);
let dst_rows = dst_image.iter_rows_mut();
for (out_row, in_row) in dst_rows.zip(src_rows) {
for (&x_in, out_pixel) in x_in_tab.iter().zip(out_row.iter_mut()) {
// Safety of value of x_in guaranteed by algorithm of creating of x_in_tab
*out_pixel = unsafe { *in_row.get_unchecked(x_in) };
}
}
}
fn resample_convolution(
src_image: &SrcImageView,
dst_image: &mut DstImageView,
filter_type: FilterType,
cpu_extensions: CpuExtensions,
temp_buffer: &mut Vec<u8>,
) {
let crop_box = src_image.crop_box();
let dst_width = dst_image.width();
let dst_height = dst_image.height();
let (filter_fn, filter_support) = convolution::get_filter_func(filter_type);
let resampler = cpu_extensions.get_resampler(src_image.pixel_type());
let need_horizontal = dst_width != src_image.width() || crop_box.width != src_image.width();
let need_vertical = dst_height != src_image.height() || crop_box.height != src_image.height();
let mut vert_coeffs = convolution::precompute_coefficients(
src_image.height(),
crop_box.top as f64,
crop_box.top as f64 + crop_box.height.get() as f64,
dst_height,
filter_fn,
filter_support,
);
if need_horizontal {
let horiz_coeffs = convolution::precompute_coefficients(
src_image.width(),
crop_box.left as f64,
crop_box.left as f64 + crop_box.width.get() as f64,
dst_width,
filter_fn,
filter_support,
);
// First used row in the source image
let y_first = vert_coeffs.bounds[0].start;
if need_vertical {
// Shift bounds for vertical pass
vert_coeffs
.bounds
.iter_mut()
.for_each(|b| b.start -= y_first);
// Last used row in the source image
let last_y_bound = vert_coeffs.bounds.last().unwrap();
let y_last = last_y_bound.start + last_y_bound.size;
let temp_height = NonZeroU32::new(y_last - y_first).unwrap();
let mut temp_image = get_temp_image_from_buffer(
temp_buffer,
dst_width,
temp_height,
src_image.pixel_type(),
);
resampler.horiz_convolution(
src_image,
&mut temp_image.dst_view(),
y_first,
horiz_coeffs,
);
resampler.vert_convolution(&temp_image.src_view(), dst_image, vert_coeffs);
} else {
resampler.horiz_convolution(src_image, dst_image, y_first, horiz_coeffs);
}
} else if need_vertical {
resampler.vert_convolution(src_image, dst_image, vert_coeffs);
}
}
fn resample_super_sampling(
src_image: &SrcImageView,
dst_image: &mut DstImageView,
filter_type: FilterType,
multiplicity: u8,
cpu_extensions: CpuExtensions,
temp_buffer: &mut Vec<u8>,
convolution_temp_buffer: &mut Vec<u8>,
) {
let crop_box = src_image.crop_box();
let dst_width = dst_image.width().get();
let dst_height = dst_image.height().get();
let width_scale = crop_box.width.get() as f32 / dst_width as f32;
let height_scale = crop_box.height.get() as f32 / dst_height as f32;
// It makes sense to resize the image in two steps only if the image
// size is greater than the required size by multiplicity times.
let factor = width_scale.min(height_scale) / multiplicity as f32;
if factor > 1.2 {
// First step is resizing the source image by fastest algorithm.
// The temporary image will be about ``multiplicity`` times larger
// than required.
let tmp_width =
NonZeroU32::new((crop_box.width.get() as f32 / factor).round() as u32).unwrap();
let tmp_height =
NonZeroU32::new((crop_box.height.get() as f32 / factor).round() as u32).unwrap();
let mut tmp_img =
get_temp_image_from_buffer(temp_buffer, tmp_width, tmp_height, src_image.pixel_type());
resample_nearest(src_image, &mut tmp_img.dst_view());
// Second step is resizing the temporary image with a convolution.
resample_convolution(
&tmp_img.src_view(),
dst_image,
filter_type,
cpu_extensions,
convolution_temp_buffer,
);
} else {
// There is no point in doing the resizing in two steps.
// We immediately resize the original image with a convolution.
resample_convolution(
src_image,
dst_image,
filter_type,
cpu_extensions,
convolution_temp_buffer,
);
}
}
+39
View File
@@ -0,0 +1,39 @@
use std::arch::x86_64::*;
use std::intrinsics::transmute;
#[inline(always)]
pub unsafe fn loadu_si128<T>(buf: &[T], index: usize) -> __m128i {
_mm_loadu_si128(buf.get_unchecked(index..).as_ptr() as *const __m128i)
}
#[inline(always)]
pub unsafe fn loadu_si256<T>(buf: &[T], index: usize) -> __m256i {
_mm256_loadu_si256(buf.get_unchecked(index..).as_ptr() as *const __m256i)
}
#[inline(always)]
pub unsafe fn loadl_epi64<T>(buf: &[T], index: usize) -> __m128i {
_mm_loadl_epi64(buf.get_unchecked(index..).as_ptr() as *const __m128i)
}
#[inline(always)]
pub unsafe fn mm_cvtepu8_epi32(buf: &[u32], index: usize) -> __m128i {
let v: i32 = transmute(*buf.get_unchecked(index));
_mm_cvtepu8_epi32(_mm_cvtsi32_si128(v))
}
#[inline(always)]
pub unsafe fn mm_cvtsi32_si128(buf: &[u32], index: usize) -> __m128i {
let v: i32 = transmute(*buf.get_unchecked(index));
_mm_cvtsi32_si128(v)
}
#[inline(always)]
pub unsafe fn ptr_i16_to_set1_epi32(buf: &[i16], index: usize) -> __m128i {
_mm_set1_epi32(*(buf.get_unchecked(index..).as_ptr() as *const i32))
}
#[inline(always)]
pub unsafe fn ptr_i16_to_256set1_epi32(buf: &[i16], index: usize) -> __m256i {
_mm256_set1_epi32(*(buf.get_unchecked(index..).as_ptr() as *const i32))
}
+179
View File
@@ -0,0 +1,179 @@
use fast_image_resize::{CpuExtensions, DstImageView, ImageData, MulDiv, PixelType, SrcImageView};
use std::convert::TryInto;
use std::num::NonZeroU32;
const fn p(r: u8, g: u8, b: u8, a: u8) -> u32 {
u32::from_le_bytes([r, g, b, a])
}
// Multiplies by alpha
fn multiply_alpha_test(cpu_extensions: CpuExtensions) {
let width: u32 = 8 + 8 + 7;
let height: u32 = 3;
let src_pixels = [p(255, 128, 0, 128), p(255, 128, 0, 255), p(255, 128, 0, 0)];
let res_pixels = [p(128, 64, 0, 128), p(255, 128, 0, 255), p(0, 0, 0, 0)];
let mut src_rows: [Vec<u32>; 3] = [
vec![src_pixels[0]; width as usize],
vec![src_pixels[1]; width as usize],
vec![src_pixels[2]; width as usize],
];
let rows: Vec<&[u32]> = src_rows.iter().map(|r| r.as_ref()).collect();
let src_image_view = SrcImageView::new(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
rows,
PixelType::U8x4,
)
.unwrap();
let mut dst_image = ImageData::new_owned(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
PixelType::U8x4,
);
let mut dst_image_view = dst_image.dst_view();
let mut alpha_mul_div: MulDiv = Default::default();
unsafe {
alpha_mul_div.set_cpu_extensions(cpu_extensions);
}
alpha_mul_div
.multiply_alpha(&src_image_view, &mut dst_image_view)
.unwrap();
let dst_pixels: Vec<u32> = dst_image
.get_buffer()
.chunks_exact(4)
.map(|c| u32::from_le_bytes(c.try_into().unwrap()))
.collect();
let dst_rows = dst_pixels.chunks_exact(width as usize);
for (row, &valid_pixel) in dst_rows.zip(res_pixels.iter()) {
for &pixel in row.iter() {
assert_eq!(pixel.to_le_bytes(), valid_pixel.to_le_bytes());
}
}
// Inplace
let rows: Vec<&mut [u32]> = src_rows.iter_mut().map(|r| r.as_mut()).collect();
let mut image_view = DstImageView::new(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
rows,
PixelType::U8x4,
)
.unwrap();
alpha_mul_div
.multiply_alpha_inplace(&mut image_view)
.unwrap();
for (row, &valid_pixel) in src_rows.iter().zip(res_pixels.iter()) {
for &pixel in row.iter() {
assert_eq!(pixel.to_le_bytes(), valid_pixel.to_le_bytes());
}
}
}
#[test]
fn multiply_alpha_avx2_test() {
multiply_alpha_test(CpuExtensions::Avx2);
}
#[test]
fn multiply_alpha_sse2_test() {
multiply_alpha_test(CpuExtensions::Sse2);
}
#[test]
fn multiply_alpha_native_test() {
multiply_alpha_test(CpuExtensions::None);
}
// Divides by alpha
fn divide_alpha_test(cpu_extensions: CpuExtensions) {
let width: u32 = 8 + 8 + 7;
let height: u32 = 3;
let src_pixels = [p(128, 64, 0, 128), p(255, 128, 0, 255), p(255, 128, 0, 0)];
let res_pixels = [p(255, 127, 0, 128), p(255, 128, 0, 255), p(0, 0, 0, 0)];
let mut src_rows: [Vec<u32>; 3] = [
vec![src_pixels[0]; width as usize],
vec![src_pixels[1]; width as usize],
vec![src_pixels[2]; width as usize],
];
let rows: Vec<&[u32]> = src_rows.iter().map(|r| r.as_ref()).collect();
let src_image_view = SrcImageView::new(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
rows,
PixelType::U8x4,
)
.unwrap();
let mut dst_image = ImageData::new_owned(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
PixelType::U8x4,
);
let mut dst_image_view = dst_image.dst_view();
let mut alpha_mul_div: MulDiv = Default::default();
unsafe {
alpha_mul_div.set_cpu_extensions(cpu_extensions);
}
alpha_mul_div
.divide_alpha(&src_image_view, &mut dst_image_view)
.unwrap();
let dst_pixels: Vec<u32> = dst_image
.get_buffer()
.chunks_exact(4)
.map(|c| u32::from_le_bytes(c.try_into().unwrap()))
.collect();
let dst_rows = dst_pixels.chunks_exact(width as usize);
for (row, &valid_pixel) in dst_rows.zip(res_pixels.iter()) {
for &pixel in row.iter() {
assert_eq!(pixel.to_le_bytes(), valid_pixel.to_le_bytes());
}
}
// Inplace
let rows: Vec<&mut [u32]> = src_rows.iter_mut().map(|r| r.as_mut()).collect();
let mut image_view = DstImageView::new(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
rows,
PixelType::U8x4,
)
.unwrap();
alpha_mul_div.divide_alpha_inplace(&mut image_view).unwrap();
for (row, &valid_pixel) in src_rows.iter().zip(res_pixels.iter()) {
for &pixel in row.iter() {
assert_eq!(pixel.to_le_bytes(), valid_pixel.to_le_bytes());
}
}
}
#[test]
fn divide_alpha_avx2_test() {
divide_alpha_test(CpuExtensions::Avx2);
}
#[test]
fn divide_alpha_sse2_test() {
divide_alpha_test(CpuExtensions::Sse2);
}
#[test]
fn divide_alpha_native_test() {
divide_alpha_test(CpuExtensions::None);
}
+39
View File
@@ -0,0 +1,39 @@
use std::num::NonZeroU32;
use fast_image_resize::{
CropBox, FilterType, ImageData, PixelType, ResizeAlg, Resizer, SrcImageView,
};
fn resize_lanczos3(src_pixels: &[u8], width: NonZeroU32, height: NonZeroU32) -> Vec<u8> {
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
let src_image = ImageData::new(width, height, src_pixels, PixelType::U8x4).unwrap();
let dst_width = NonZeroU32::new(1024).unwrap();
let dst_height = NonZeroU32::new(768).unwrap();
let mut dst_image = ImageData::new_owned(dst_width, dst_height, src_image.pixel_type());
let src_view = src_image.src_view();
let mut dst_view = dst_image.dst_view();
resizer.resize(&src_view, &mut dst_view);
dst_image.get_buffer().to_owned()
}
fn resize_cropped_image(mut src_view: SrcImageView) -> ImageData<Vec<u8>> {
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
src_view
.set_crop_box(CropBox {
left: 10,
top: 10,
width: NonZeroU32::new(100).unwrap(),
height: NonZeroU32::new(200).unwrap(),
})
.unwrap();
let dst_width = NonZeroU32::new(1024).unwrap();
let dst_height = NonZeroU32::new(768).unwrap();
let mut dst_image = ImageData::new_owned(dst_width, dst_height, src_view.pixel_type());
let mut dst_view = dst_image.dst_view();
resizer.resize(&src_view, &mut dst_view);
dst_image
}
+179
View File
@@ -0,0 +1,179 @@
use std::fs::File;
use std::num::NonZeroU32;
use image::codecs::png::PngEncoder;
use image::io::Reader as ImageReader;
use image::{ColorType, GenericImageView};
use fast_image_resize::ImageData;
use fast_image_resize::{CpuExtensions, FilterType, PixelType, ResizeAlg, Resizer, SrcImageView};
fn get_source_image() -> ImageData<Vec<u8>> {
let img = ImageReader::open("./data/nasa-4928x3279.png")
.unwrap()
.decode()
.unwrap();
let width = img.width();
let height = img.height();
let rgb = img.to_rgba8();
let buf = rgb.as_raw().clone();
ImageData::new(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
buf,
PixelType::U8x4,
)
.unwrap()
}
fn get_small_source_image() -> ImageData<Vec<u8>> {
let img = ImageReader::open("./data/nasa-852x567.png")
.unwrap()
.decode()
.unwrap();
let width = img.width();
let height = img.height();
let rgb = img.to_rgba8();
let buf = rgb.as_raw().clone();
ImageData::new(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
buf,
PixelType::U8x4,
)
.unwrap()
}
fn get_new_height(src_image: &SrcImageView, new_width: u32) -> u32 {
let scale = new_width as f32 / src_image.width().get() as f32;
(src_image.height().get() as f32 * scale).round() as u32
}
const NEW_WIDTH: u32 = 255;
const NEW_BIG_WIDTH: u32 = 5016;
fn save_result(image: &SrcImageView, name: &str) {
std::fs::create_dir_all("./data/result").unwrap();
let mut file = File::create(format!("./data/result/{}.png", name)).unwrap();
let encoder = PngEncoder::new(&mut file);
encoder
.encode(
&image.get_buffer(),
image.width().get(),
image.height().get(),
ColorType::Rgba8,
)
.unwrap();
}
#[test]
fn resample_wo_simd_lanczos3_test() {
let image = get_source_image();
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
unsafe {
resizer.set_cpu_extensions(CpuExtensions::None);
}
let new_height = get_new_height(&image.src_view(), NEW_WIDTH);
let mut result = ImageData::new_owned(
NonZeroU32::new(NEW_WIDTH).unwrap(),
NonZeroU32::new(new_height).unwrap(),
image.pixel_type(),
);
resizer.resize(&image.src_view(), &mut result.dst_view());
save_result(&result.src_view(), "lanczos3_wo_simd");
}
#[test]
fn resample_sse4_lanczos3_test() {
let image = get_source_image();
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
unsafe {
resizer.set_cpu_extensions(CpuExtensions::Sse4_1);
}
let new_height = get_new_height(&image.src_view(), NEW_WIDTH);
let mut result = ImageData::new_owned(
NonZeroU32::new(NEW_WIDTH).unwrap(),
NonZeroU32::new(new_height).unwrap(),
image.pixel_type(),
);
resizer.resize(&image.src_view(), &mut result.dst_view());
save_result(&result.src_view(), "lanczos3_sse4");
}
fn resize_lanczos3(src_pixels: &[u8], width: NonZeroU32, height: NonZeroU32) -> Vec<u8> {
let src_image = ImageData::new(width, height, src_pixels, PixelType::U8x4).unwrap();
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
let dst_width = NonZeroU32::new(1024).unwrap();
let dst_height = NonZeroU32::new(768).unwrap();
let mut dst_image = ImageData::new_owned(dst_width, dst_height, src_image.pixel_type());
resizer.resize(&src_image.src_view(), &mut dst_image.dst_view());
dst_image.get_buffer().to_owned()
}
#[test]
fn resample_avx2_lanczos3_test() {
let image = get_source_image();
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
unsafe {
resizer.set_cpu_extensions(CpuExtensions::Avx2);
}
let new_height = get_new_height(&image.src_view(), NEW_WIDTH);
let mut result = ImageData::new_owned(
NonZeroU32::new(NEW_WIDTH).unwrap(),
NonZeroU32::new(new_height).unwrap(),
image.pixel_type(),
);
resizer.resize(&image.src_view(), &mut result.dst_view());
save_result(&result.src_view(), "lanczos3_avx2");
}
#[test]
fn resample_avx2_lanczos3_upscale_test() {
let image = get_small_source_image();
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
unsafe {
resizer.set_cpu_extensions(CpuExtensions::Avx2);
}
let new_height = get_new_height(&image.src_view(), NEW_BIG_WIDTH);
let mut result = ImageData::new_owned(
NonZeroU32::new(NEW_BIG_WIDTH).unwrap(),
NonZeroU32::new(new_height).unwrap(),
image.pixel_type(),
);
resizer.resize(&image.src_view(), &mut result.dst_view());
save_result(&result.src_view(), "lanczos3_avx2_upscale");
}
#[test]
fn resample_nearest_test() {
let image = get_source_image();
let mut resizer = Resizer::new(ResizeAlg::Nearest);
unsafe {
resizer.set_cpu_extensions(CpuExtensions::None);
}
let new_height = get_new_height(&image.src_view(), NEW_WIDTH);
let mut result = ImageData::new_owned(
NonZeroU32::new(NEW_WIDTH).unwrap(),
NonZeroU32::new(new_height).unwrap(),
image.pixel_type(),
);
resizer.resize(&image.src_view(), &mut result.dst_view());
save_result(&result.src_view(), "nearest_wo_simd");
}
#[test]
fn resample_super_sampling_test() {
let image = get_source_image();
let mut resizer = Resizer::new(ResizeAlg::SuperSampling(FilterType::Lanczos3, 2));
unsafe {
resizer.set_cpu_extensions(CpuExtensions::Avx2);
}
let new_height = get_new_height(&image.src_view(), NEW_WIDTH);
let mut result = ImageData::new_owned(
NonZeroU32::new(NEW_WIDTH).unwrap(),
NonZeroU32::new(new_height).unwrap(),
image.pixel_type(),
);
resizer.resize(&image.src_view(), &mut result.dst_view());
save_result(&result.src_view(), "super_sampling_avx2");
}