- Updated dependencies.

- Optimized convolution algorythm by deleting zero coefficients from start and end of bounds.
This commit is contained in:
Kirill Kuzminykh
2024-10-03 20:24:00 +03:00
parent 75bd369a6b
commit 79a3a73afb
39 changed files with 433 additions and 404 deletions
Generated
+127 -117
View File
@@ -46,9 +46,9 @@ checksum = "4b46cbb362ab8752921c97e041f5e366ee6297bd428a31275b9fcf1e380f7299"
[[package]]
name = "anstream"
version = "0.6.14"
version = "0.6.15"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "418c75fa768af9c03be99d17643f93f79bbba589895012a80e3452a19ddda15b"
checksum = "64e15c1ab1f89faffbf04a634d5e1962e9074f2741eef6d97f3c4e322426d526"
dependencies = [
"anstyle",
"anstyle-parse",
@@ -61,36 +61,36 @@ dependencies = [
[[package]]
name = "anstyle"
version = "1.0.7"
version = "1.0.8"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "038dfcf04a5feb68e9c60b21c9625a54c2c0616e79b72b0fd87075a056ae1d1b"
checksum = "1bec1de6f59aedf83baf9ff929c98f2ad654b97c9510f4e70cf6f661d49fd5b1"
[[package]]
name = "anstyle-parse"
version = "0.2.4"
version = "0.2.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c03a11a9034d92058ceb6ee011ce58af4a9bf61491aa7e1e59ecd24bd40d22d4"
checksum = "eb47de1e80c2b463c735db5b217a0ddc39d612e7ac9e2e96a5aed1f57616c1cb"
dependencies = [
"utf8parse",
]
[[package]]
name = "anstyle-query"
version = "1.1.0"
version = "1.1.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ad186efb764318d35165f1758e7dcef3b10628e26d41a44bc5550652e6804391"
checksum = "6d36fc52c7f6c869915e99412912f22093507da8d9e942ceaf66fe4b7c14422a"
dependencies = [
"windows-sys",
"windows-sys 0.52.0",
]
[[package]]
name = "anstyle-wincon"
version = "3.0.3"
version = "3.0.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "61a38449feb7068f52bb06c12759005cf459ee52bb4adc1d5a7c4322d716fb19"
checksum = "5bf74e1b6e971609db8ca7a9ce79fd5768ab6ae46441c572e46cf596f59e57f8"
dependencies = [
"anstyle",
"windows-sys",
"windows-sys 0.52.0",
]
[[package]]
@@ -113,7 +113,7 @@ checksum = "0ae92a5119aa49cdbcf6b9f893fe4e1d98b04ccbf82ee0584ad948a44a734dea"
dependencies = [
"proc-macro2",
"quote",
"syn 2.0.71",
"syn",
]
[[package]]
@@ -186,9 +186,9 @@ dependencies = [
[[package]]
name = "bstr"
version = "1.9.1"
version = "1.10.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "05efc5cfd9110c8416e471df0e96702d58690178e206e61b7173706673c93706"
checksum = "40723b8fb387abc38f4f4a37c09073622e41dd12327033091ef8950659e6dc0c"
dependencies = [
"memchr",
"serde",
@@ -208,9 +208,9 @@ checksum = "79296716171880943b8470b5f8d03aa55eb2e645a4874bdbb28adb49162e012c"
[[package]]
name = "bytemuck"
version = "1.16.1"
version = "1.16.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b236fc92302c97ed75b38da1f4917b5cdda4984745740f153a5d3059e48d725e"
checksum = "102087e286b4677862ea56cf8fc58bb2cdfa8725c40ffb80fe3a008eb7f2fc83"
[[package]]
name = "byteorder"
@@ -232,9 +232,9 @@ checksum = "37b2a672a2cb129a2e41c10b1224bb368f9f37a2b16b612598138befd7b37eb5"
[[package]]
name = "cc"
version = "1.1.5"
version = "1.1.8"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "324c74f2155653c90b04f25b2a47a8a631360cb908f92a772695f430c7e31052"
checksum = "504bdec147f2cc13c8b57ed9401fd8a147cc66b67ad5cb241394244f2c947549"
dependencies = [
"jobserver",
"libc",
@@ -325,9 +325,9 @@ dependencies = [
[[package]]
name = "clap"
version = "4.5.9"
version = "4.5.13"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "64acc1846d54c1fe936a78dc189c34e28d3f5afc348403f28ecf53660b9b8462"
checksum = "0fbb260a053428790f3de475e304ff84cdbc4face759ea7a3e64c1edd938a7fc"
dependencies = [
"clap_builder",
"clap_derive",
@@ -335,9 +335,9 @@ dependencies = [
[[package]]
name = "clap-verbosity-flag"
version = "2.2.0"
version = "2.2.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "bb9b20c0dd58e4c2e991c8d203bbeb76c11304d1011659686b5b644bc29aa478"
checksum = "63d19864d6b68464c59f7162c9914a0b569ddc2926b4a2d71afe62a9738eff53"
dependencies = [
"clap",
"log",
@@ -345,9 +345,9 @@ dependencies = [
[[package]]
name = "clap_builder"
version = "4.5.9"
version = "4.5.13"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "6fb8393d67ba2e7bfaf28a23458e4e2b543cc73a99595511eb207fdb8aede942"
checksum = "64b17d7ea74e9f833c7dbf2cbe4fb12ff26783eda4782a8975b72f895c9b4d99"
dependencies = [
"anstream",
"anstyle",
@@ -357,21 +357,21 @@ dependencies = [
[[package]]
name = "clap_derive"
version = "4.5.8"
version = "4.5.13"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "2bac35c6dafb060fd4d275d9a4ffae97917c13a6327903a8be2153cd964f7085"
checksum = "501d359d5f3dcaf6ecdeee48833ae73ec6e42723a1e52419c79abf9507eec0a0"
dependencies = [
"heck",
"proc-macro2",
"quote",
"syn 2.0.71",
"syn",
]
[[package]]
name = "clap_lex"
version = "0.7.1"
version = "0.7.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "4b82cf0babdbd58558212896d1a4272303a57bdb245c2bf1147185fb45640e70"
checksum = "1462739cb27611015575c0c11df5df7601141071f07518d56fcc1be504cbec97"
[[package]]
name = "color_quant"
@@ -381,9 +381,9 @@ checksum = "3d7b894f5411737b7867f4827955924d7c254fc9f4d91a6aad6b097804b1018b"
[[package]]
name = "colorchoice"
version = "1.0.1"
version = "1.0.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "0b6a852b24ab71dffc585bcb46eaf7959d175cb865a7152e35b348d1b2960422"
checksum = "d3fd119d74b830634cea2a0f58bbd0d54540518a14397557951e79340abc28c0"
[[package]]
name = "core-foundation-sys"
@@ -517,9 +517,9 @@ checksum = "60b1af1c220855b6ceac025d3f6ecdd2b7c4894bfe9cd9bda4fbb4bc7c0d4cf0"
[[package]]
name = "env_filter"
version = "0.1.0"
version = "0.1.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "a009aa4810eb158359dda09d0c87378e4bbb89b5a801f016885a4707ba24f7ea"
checksum = "4f2c92ceda6ceec50f43169f9ee8424fe2db276791afde7b2cd8bc084cb376ab"
dependencies = [
"log",
"regex",
@@ -527,9 +527,9 @@ dependencies = [
[[package]]
name = "env_logger"
version = "0.11.3"
version = "0.11.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "38b35839ba51819680ba087cd351788c9a3c476841207e0b8cee0b04722343b9"
checksum = "e13fa619b91fb2381732789fc5de83b45675e882f66623b7d8cb4f643017018d"
dependencies = [
"anstream",
"anstyle",
@@ -596,9 +596,9 @@ dependencies = [
[[package]]
name = "flate2"
version = "1.0.30"
version = "1.0.31"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "5f54427cfd1c7829e2a139fcefea601bf088ebca651d2bf53ebc600eac295dae"
checksum = "7f211bbe8e69bbd0cfdea405084f128ae8b4aaa6b0b522fc8f2b009084797920"
dependencies = [
"crc32fast",
"miniz_oxide",
@@ -752,12 +752,12 @@ dependencies = [
[[package]]
name = "image"
version = "0.25.1"
version = "0.25.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "fd54d660e773627692c524beaad361aca785a4f9f5730ce91f42aabe5bce3d11"
checksum = "99314c8a2152b8ddb211f924cdae532d8c5e4c8bb54728e12fff1b0cd5963a10"
dependencies = [
"bytemuck",
"byteorder",
"byteorder-lite",
"color_quant",
"exr",
"gif",
@@ -791,9 +791,9 @@ checksum = "44feda355f4159a7c757171a77de25daf6411e217b4cabd03bd6650690468126"
[[package]]
name = "indexmap"
version = "2.2.6"
version = "2.3.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "168fb715dda47215e360912c096649d23d58bf392ac62f73919e831745e40f26"
checksum = "de3fc2e30ba82dd1b3911c8de1ffc143c74a914a14e99514d7637e3099df5ea0"
dependencies = [
"equivalent",
"hashbrown",
@@ -807,7 +807,7 @@ checksum = "c34819042dc3d3971c46c2190835914dfbe0c3c13f61449b2997f4e9722dfa60"
dependencies = [
"proc-macro2",
"quote",
"syn 2.0.71",
"syn",
]
[[package]]
@@ -818,14 +818,14 @@ checksum = "f23ff5ef2b80d608d61efee834934d862cd92461afc0560dedf493e4c033738b"
dependencies = [
"hermit-abi",
"libc",
"windows-sys",
"windows-sys 0.52.0",
]
[[package]]
name = "is_terminal_polyfill"
version = "1.70.0"
version = "1.70.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f8478577c03552c21db0e2724ffb8986a5ce7af88107e6be5d2ee6e158c12800"
checksum = "7943c866cc5cd64cbc25b2e01621d07fa8eb2a1a23160ee81ce38704e97b8ecf"
[[package]]
name = "itertools"
@@ -862,9 +862,9 @@ checksum = "49f1f14873335454500d59611f1cf4a4b0f786f9ac11f4312a78e4cf2566695b"
[[package]]
name = "jobserver"
version = "0.1.31"
version = "0.1.32"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d2b099aaa34a9751c5bf0878add70444e1ed2dd73f347be99003d4577277de6e"
checksum = "48d1dbcbbeb6a7fec7e059840aa538bd62aaccf972c7346c4d9d2059312853d0"
dependencies = [
"libc",
]
@@ -921,11 +921,11 @@ checksum = "4ec2a862134d2a7d32d7983ddcdd1c4923530833c9f2ea1a44fc5fa473989058"
[[package]]
name = "libvips"
version = "1.4.3"
version = "1.7.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9c6574a02b3823ce436bd70d47546428f4546686031f8d2af4c056d23969ace2"
checksum = "33890b93365ac05b5e6063d41bc014a1598087db4a0b9f75eddaa7dad0f7fc2a"
dependencies = [
"num-derive 0.3.3",
"num-derive",
"num-traits",
]
@@ -967,7 +967,6 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8ea1f30cedd69f0a2954655f7188c6a834246d2bcf1e315e2ac40c4b24dc9519"
dependencies = [
"cfg-if",
"rayon",
]
[[package]]
@@ -1036,17 +1035,6 @@ dependencies = [
"num-traits",
]
[[package]]
name = "num-derive"
version = "0.3.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "876a53fff98e03a936a674b29568b0e605f06b29372c2489ff4de23f1949743d"
dependencies = [
"proc-macro2",
"quote",
"syn 1.0.109",
]
[[package]]
name = "num-derive"
version = "0.4.2"
@@ -1055,7 +1043,7 @@ checksum = "ed3955f1a9c7c0c15e092f9c887db08b1fc683305fdf6eb6684f22555355e202"
dependencies = [
"proc-macro2",
"quote",
"syn 2.0.71",
"syn",
]
[[package]]
@@ -1151,7 +1139,7 @@ dependencies = [
"pest_meta",
"proc-macro2",
"quote",
"syn 2.0.71",
"syn",
]
[[package]]
@@ -1224,9 +1212,12 @@ dependencies = [
[[package]]
name = "ppv-lite86"
version = "0.2.17"
version = "0.2.20"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "5b40af805b3121feab8a3c29f04d8ad262fa8e0561883e7653e024ae4479e6de"
checksum = "77957b295656769bb8ad2b6a6b09d897d94f05c41b069aede1fcdaa675eaea04"
dependencies = [
"zerocopy",
]
[[package]]
name = "proc-macro2"
@@ -1253,7 +1244,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8021cf59c8ec9c432cfc2526ac6b8aa508ecaf29cd415f271b8406c1b851c3fd"
dependencies = [
"quote",
"syn 2.0.71",
"syn",
]
[[package]]
@@ -1331,7 +1322,7 @@ dependencies = [
"maybe-rayon",
"new_debug_unreachable",
"noop_proc_macro",
"num-derive 0.4.2",
"num-derive",
"num-traits",
"once_cell",
"paste",
@@ -1347,16 +1338,15 @@ dependencies = [
[[package]]
name = "ravif"
version = "0.11.8"
version = "0.11.9"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c6ba61c28ba24c0cf8406e025cb29a742637e3f70776e61c27a8a8b72a042d12"
checksum = "5797d09f9bd33604689e87e8380df4951d4912f01b63f71205e2abd4ae25e6b6"
dependencies = [
"avif-serialize",
"imgref",
"loop9",
"quick-error",
"rav1e",
"rayon",
"rgb",
]
@@ -1382,9 +1372,9 @@ dependencies = [
[[package]]
name = "regex"
version = "1.10.5"
version = "1.10.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b91213439dad192326a0d7c6ee3955910425f441d7038e0d6933b0aec5c4517f"
checksum = "4219d74c6b67a3654a9fbebc4b419e22126d13d2f3c4a07ee0cb61ff79a79619"
dependencies = [
"aho-corasick",
"memchr",
@@ -1411,9 +1401,9 @@ checksum = "7a66a03ae7c801facd77a29370b4faec201768915ac14a721ba36f20bc9c209b"
[[package]]
name = "resize"
version = "0.8.4"
version = "0.8.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c3e29f584c07a8396c5e2eee0bd8d7aec5c8d9e0a3c2333806fd2ec1d2a5b080"
checksum = "a84f5827feaf48508b264176bd88e0479695af183738cf0305fadf956c796412"
dependencies = [
"rgb",
]
@@ -1434,9 +1424,9 @@ dependencies = [
[[package]]
name = "rgb"
version = "0.8.45"
version = "0.8.48"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ade4539f42266ded9e755c605bdddf546242b2c961b03b06a7375260788a0523"
checksum = "0f86ae463694029097b846d8f99fd5536740602ae00022c0c50c5600720b2f71"
dependencies = [
"bytemuck",
]
@@ -1479,25 +1469,26 @@ checksum = "e0cd7e117be63d3c3678776753929474f3b04a43a080c744d6b0ae2a8c28e222"
dependencies = [
"proc-macro2",
"quote",
"syn 2.0.71",
"syn",
]
[[package]]
name = "serde_json"
version = "1.0.120"
version = "1.0.122"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "4e0d21c9a8cae1235ad58a00c11cb40d4b1e5c784f1ef2c537876ed6ffd8b7c5"
checksum = "784b6203951c57ff748476b126ccb5e8e2959a5c19e5c617ab1956be3dbc68da"
dependencies = [
"itoa",
"memchr",
"ryu",
"serde",
]
[[package]]
name = "serde_spanned"
version = "0.6.6"
version = "0.6.7"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "79e674e01f999af37c49f70a6ede167a8a60b2503e56c5599532a65baa5969a0"
checksum = "eb5b1b31579f3811bf615c144393417496f152e12ac8b7663bf664f4a815306d"
dependencies = [
"serde",
]
@@ -1567,20 +1558,9 @@ checksum = "7da8b5736845d9f2fcb837ea5d9e2628564b3b043a70948a3f0b778838c5fb4f"
[[package]]
name = "syn"
version = "1.0.109"
version = "2.0.72"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "72b64191b275b66ffe2469e8af2c1cfe3bafa67b529ead792a6d0160888b4237"
dependencies = [
"proc-macro2",
"quote",
"unicode-ident",
]
[[package]]
name = "syn"
version = "2.0.71"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b146dcf730474b4bcd16c311627b31ede9ab149045db4d6088b3becaea046462"
checksum = "dc4b9b9bf2add8093d3f2c0204471e951b2285580335de42f9d2534f3ae7a8af"
dependencies = [
"proc-macro2",
"quote",
@@ -1602,9 +1582,9 @@ dependencies = [
[[package]]
name = "target-lexicon"
version = "0.12.15"
version = "0.12.16"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "4873307b7c257eddcb50c9bedf158eb669578359fb28428bef438fec8e6ba7c2"
checksum = "61c41af27dd6d1e27b1b16b489db798443478cef1f06a660c96db617ba5de3b1"
[[package]]
name = "tera"
@@ -1653,7 +1633,7 @@ checksum = "a4558b58466b9ad7ca0f102865eccc95938dca1a74a856f2b57b6629050da261"
dependencies = [
"proc-macro2",
"quote",
"syn 2.0.71",
"syn",
]
[[package]]
@@ -1679,9 +1659,9 @@ dependencies = [
[[package]]
name = "toml"
version = "0.8.15"
version = "0.8.19"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ac2caab0bf757388c6c0ae23b3293fdb463fee59434529014f85e3263b995c28"
checksum = "a1ed1f98e3fdc28d6d910e6737ae6ab1a93bf1985935a1193e68f93eeb68d24e"
dependencies = [
"serde",
"serde_spanned",
@@ -1691,18 +1671,18 @@ dependencies = [
[[package]]
name = "toml_datetime"
version = "0.6.6"
version = "0.6.8"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "4badfd56924ae69bcc9039335b2e017639ce3f9b001c393c1b2d1ef846ce2cbf"
checksum = "0dd7358ecb8fc2f8d014bf86f6f638ce72ba252a2c3a2572f2a795f1d23efb41"
dependencies = [
"serde",
]
[[package]]
name = "toml_edit"
version = "0.22.16"
version = "0.22.20"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "278f3d518e152219c994ce877758516bca5e118eaed6996192a774fb9fbf0788"
checksum = "583c44c02ad26b0c3f3066fe629275e50627026c51ac2e595cca4c230ce1ce1d"
dependencies = [
"indexmap",
"serde",
@@ -1804,9 +1784,9 @@ checksum = "852e951cb7832cb45cb1169900d19760cfa39b82bc0ea9c0e5a14ae88411c98b"
[[package]]
name = "version_check"
version = "0.9.4"
version = "0.9.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "49874b5167b65d7193b8aba1567f5c7d93d001cafc34600cee003eda787e483f"
checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a"
[[package]]
name = "walkdir"
@@ -1845,7 +1825,7 @@ dependencies = [
"once_cell",
"proc-macro2",
"quote",
"syn 2.0.71",
"syn",
"wasm-bindgen-shared",
]
@@ -1867,7 +1847,7 @@ checksum = "e94f17b526d0a461a191c78ea52bbce64071ed5c04c9ffe424dcb38f74171bb7"
dependencies = [
"proc-macro2",
"quote",
"syn 2.0.71",
"syn",
"wasm-bindgen-backend",
"wasm-bindgen-shared",
]
@@ -1886,11 +1866,11 @@ checksum = "53a85b86a771b1c87058196170769dd264f66c0782acf1ae6cc51bfd64b39082"
[[package]]
name = "winapi-util"
version = "0.1.8"
version = "0.1.9"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "4d4cc384e1e73b93bafa6fb4f1df8c41695c8a91cf9c4c64358067d15a7b6c6b"
checksum = "cf221c93e13a30d793f7645a0e7762c55d169dbb0a49671918a2319d289b10bb"
dependencies = [
"windows-sys",
"windows-sys 0.59.0",
]
[[package]]
@@ -1911,6 +1891,15 @@ dependencies = [
"windows-targets",
]
[[package]]
name = "windows-sys"
version = "0.59.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1e38bc4d79ed67fd075bcc251a1c39b32a1776bbe92e5bef1f0bf1f8c531853b"
dependencies = [
"windows-targets",
]
[[package]]
name = "windows-targets"
version = "0.52.6"
@@ -1977,13 +1966,34 @@ checksum = "589f6da84c646204747d1270a2a5661ea66ed1cced2631d546fdfb155959f9ec"
[[package]]
name = "winnow"
version = "0.6.13"
version = "0.6.18"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "59b5e5f6c299a3c7890b876a2a587f3115162487e704907d9b6cd29473052ba1"
checksum = "68a9bda4691f099d435ad181000724da8e5899daa10713c2d432552b9ccd3a6f"
dependencies = [
"memchr",
]
[[package]]
name = "zerocopy"
version = "0.7.35"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1b9b4fd18abc82b8136838da5d50bae7bdea537c574d8dc1a34ed098d6c166f0"
dependencies = [
"byteorder",
"zerocopy-derive",
]
[[package]]
name = "zerocopy-derive"
version = "0.7.35"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "fa4f8080344d4671fb4e831a13ad1e68092748387dfc4f55e356242fae12ce3e"
dependencies = [
"proc-macro2",
"quote",
"syn",
]
[[package]]
name = "zune-core"
version = "0.4.12"
@@ -2001,9 +2011,9 @@ dependencies = [
[[package]]
name = "zune-jpeg"
version = "0.4.11"
version = "0.4.13"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ec866b44a2a1fd6133d363f073ca1b179f438f99e7e5bfb1e33f7181facfe448"
checksum = "16099418600b4d8f028622f73ff6e3deaabdff330fb9a2a131dea781ee8b0768"
dependencies = [
"zune-core",
]
+4 -4
View File
@@ -25,7 +25,7 @@ num-traits = "0.2.19"
thiserror = "1.0"
document-features = "0.2.10"
# Optional dependencies
image = { version = "0.25.1", optional = true, default-features = false }
image = { version = "0.25.2", optional = true, default-features = false }
bytemuck = { version = "1.16", optional = true }
[features]
@@ -40,8 +40,8 @@ only_u8x4 = ["testing/only_u8x4"] # This can be used to experiment with the cra
[dev-dependencies]
fast_image_resize = { path = ".", features = ["for_testing"] }
resize = { version = "0.8.4", default-features = false, features = ["std"] }
rgb = "0.8.45"
resize = { version = "0.8.5", default-features = false, features = ["std"] }
rgb = "0.8.48"
png = "0.17.13"
serde = { version = "1.0", features = ["serde_derive"] }
serde_json = "1.0"
@@ -57,7 +57,7 @@ nix = { version = "0.29.0", default-features = false, features = ["sched"] }
[target.'cfg(all(not(target_arch = "wasm32"), not(target_os = "windows")))'.dev-dependencies]
libvips = "=1.4.3"
libvips = "1.7"
[[bench]]
+2 -4
View File
@@ -137,8 +137,7 @@ Otherwise, you have to convert such images into supported by the crate image typ
use std::io::BufWriter;
use image::codecs::png::PngEncoder;
use image::io::Reader as ImageReader;
use image::{ExtendedColorType, ImageEncoder};
use image::{ExtendedColorType, ImageEncoder, ImageReader};
use fast_image_resize::{IntoImageView, Resizer};
use fast_image_resize::images::Image;
@@ -181,8 +180,7 @@ fn main() {
```rust
use image::codecs::png::PngEncoder;
use image::io::Reader as ImageReader;
use image::{ColorType, GenericImageView};
use image::{ColorType, ImageReader, GenericImageView};
use fast_image_resize::{IntoImageView, Resizer, ResizeOptions};
use fast_image_resize::images::Image;
+1 -2
View File
@@ -3,8 +3,7 @@ use std::path::PathBuf;
use anyhow::{anyhow, Context, Result};
use clap::Parser;
use image::io::Reader as ImageReader;
use image::ColorType;
use image::{ColorType, ImageReader};
use log::debug;
use once_cell::sync::Lazy;
+33 -15
View File
@@ -284,13 +284,22 @@ impl PixelComponentMapper {
};
}
match_img!(
tables,
(PT::U8, U8, PT::U16, U16),
(PT::U8x2, U8x2, PT::U16x2, U16x2),
(PT::U8x3, U8x3, PT::U16x3, U16x3),
(PT::U8x4, U8x4, PT::U16x4, U16x4),
)
#[cfg(not(feature = "only_u8x4"))]
{
match_img!(
tables,
(PT::U8, U8, PT::U16, U16),
(PT::U8x2, U8x2, PT::U16x2, U16x2),
(PT::U8x3, U8x3, PT::U16x3, U16x3),
(PT::U8x4, U8x4, PT::U16x4, U16x4),
)
}
#[cfg(feature = "only_u8x4")]
match (src_pixel_type, dst_pixel_type) {
(PT::U8x4, PT::U8x4) => tables.u8_u8.map_image::<U8x4, U8x4>(src_image, dst_image),
_ => return Err(MappingError::UnsupportedCombinationOfImageTypes),
}
}
fn map_inplace(
@@ -316,14 +325,23 @@ impl PixelComponentMapper {
};
}
match_img!(
tables,
image,
(PT::U8, U8, PT::U16, U16),
(PT::U8x2, U8x2, PT::U16x2, U16x2),
(PT::U8x3, U8x3, PT::U16x3, U16x3),
(PT::U8x4, U8x4, PT::U16x4, U16x4),
)
#[cfg(not(feature = "only_u8x4"))]
{
match_img!(
tables,
image,
(PT::U8, U8, PT::U16, U16),
(PT::U8x2, U8x2, PT::U16x2, U16x2),
(PT::U8x3, U8x3, PT::U16x3, U16x3),
(PT::U8x4, U8x4, PT::U16x4, U16x4),
)
}
#[cfg(feature = "only_u8x4")]
match pixel_type {
PT::U8x4 => tables.u8_u8.map_image_inplace::<U8x4>(image),
_ => return Err(MappingError::UnsupportedCombinationOfImageTypes),
}
}
/// Mapping in the forward direction of pixel's components of source image
+20 -4
View File
@@ -136,12 +136,28 @@ pub(crate) fn precompute_coefficients(
// (x + 0.5) - in_center => x - (in_center - 0.5) => x - center
let center = in_center - 0.5;
let mut bound_start = x_min;
let mut bound_end = x_max;
// Calculate the weight of each input pixel from the given x-range.
for x in x_min..x_max {
let w: f64 = filter((x as f64 - center) * recip_filter_scale);
coeffs.push(w);
ww += w;
if x == bound_start && w == 0. {
// Don't use zero coefficients at the start of bound;
bound_start += 1;
} else {
coeffs.push(w);
ww += w;
}
}
for &c in coeffs.iter().rev() {
if bound_end <= bound_start || c != 0. {
break;
}
// Don't use zero coefficients at the end of bound;
bound_end -= 1;
}
if ww != 0.0 {
// Normalise values of weights.
// The sum of weights must be equal to 1.0.
@@ -150,8 +166,8 @@ pub(crate) fn precompute_coefficients(
// Remaining values should stay empty if they are used despite x_max.
coeffs.resize(cur_index + window_size, 0.);
bounds.push(Bound {
start: x_min,
size: x_max - x_min,
start: bound_start,
size: bound_end - bound_start,
});
}
+63 -64
View File
@@ -1,4 +1,3 @@
use super::Bound;
use crate::convolution::Coefficients;
// This code is based on C-implementation from Pillow-SIMD package for Python
@@ -31,16 +30,21 @@ const MAX_COEFFS_PRECISION: u8 = 16 - 1;
/// Converts `Vec<f64>` into `Vec<i16>`.
pub(crate) struct Normalizer16 {
values: Vec<i16>,
precision: u8,
window_size: usize,
bounds: Vec<Bound>,
chunks: Vec<CoefficientsI16Chunk>,
}
#[derive(Debug, Clone, Copy)]
pub(crate) struct CoefficientsI16Chunk<'a> {
#[derive(Debug, Clone)]
pub(crate) struct CoefficientsI16Chunk {
pub start: u32,
pub values: &'a [i16],
values: Vec<i16>,
}
impl CoefficientsI16Chunk {
#[inline(always)]
pub fn values(&self) -> &[i16] {
&self.values
}
}
impl Normalizer16 {
@@ -64,34 +68,29 @@ impl Normalizer16 {
}
debug_assert!(precision >= 4); // required for some SIMD optimisations
let mut values_i16 = Vec::with_capacity(coefficients.values.len());
let mut chunks = Vec::with_capacity(coefficients.bounds.len());
if coefficients.window_size > 0 {
let scale = (1 << precision) as f64;
let coef_chunks = coefficients.values.chunks_exact(coefficients.window_size);
for (chunk, bound) in coef_chunks.zip(&coefficients.bounds) {
let chunk_i16: Vec<i16> = chunk
.iter()
.take(bound.size as usize)
.map(|&v| (v * scale).round() as i16)
.collect();
chunks.push(CoefficientsI16Chunk {
start: bound.start,
values: chunk_i16,
});
}
}
let scale = (1 << precision) as f64;
for src in coefficients.values.iter().copied() {
values_i16.push((src * scale).round() as i16);
}
Self {
values: values_i16,
precision,
window_size: coefficients.window_size,
bounds: coefficients.bounds,
}
Self { precision, chunks }
}
#[inline]
pub fn normalized_chunks(&self) -> Vec<CoefficientsI16Chunk> {
let mut cooefs = self.values.as_slice();
let mut res = Vec::with_capacity(self.bounds.len());
for bound in self.bounds.iter() {
let (left, right) = cooefs.split_at(self.window_size);
cooefs = right;
let size = bound.size as usize;
res.push(CoefficientsI16Chunk {
start: bound.start,
values: &left[0..size],
});
}
res
#[inline(always)]
pub fn coefficients(&self) -> &[CoefficientsI16Chunk] {
&self.chunks
}
#[inline]
@@ -112,7 +111,7 @@ impl Normalizer16 {
}
}
// 16 bits for result. Filter can have negative areas.
// 16 bits for a result. Filter can have negative areas.
// In one cases the sum of the coefficients will be negative,
// in the other it will be more than 1.0. That is why we need
// two extra bits for overflow and i64 type.
@@ -120,18 +119,23 @@ const PRECISION16_BITS: u8 = 64 - 16 - 2;
// We use i32 type to store coefficients.
const MAX_COEFFS_PRECISION16: u8 = 32 - 1;
#[derive(Debug, Clone, Copy)]
pub(crate) struct CoefficientsI32Chunk<'a> {
#[derive(Debug, Clone)]
pub(crate) struct CoefficientsI32Chunk {
pub start: u32,
pub values: &'a [i32],
pub values: Vec<i32>,
}
impl CoefficientsI32Chunk {
#[inline(always)]
pub fn values(&self) -> &[i32] {
&self.values
}
}
/// Converts `Vec<f64>` into `Vec<i32>`.
pub(crate) struct Normalizer32 {
values: Vec<i32>,
precision: u8,
window_size: usize,
bounds: Vec<Bound>,
chunks: Vec<CoefficientsI32Chunk>,
}
impl Normalizer32 {
@@ -155,34 +159,29 @@ impl Normalizer32 {
}
debug_assert!(precision >= 4); // required for some SIMD optimisations
let mut values_i32 = Vec::with_capacity(coefficients.values.len());
let mut chunks = Vec::with_capacity(coefficients.bounds.len());
if coefficients.window_size > 0 {
let scale = (1i64 << precision) as f64;
let coef_chunks = coefficients.values.chunks_exact(coefficients.window_size);
for (chunk, bound) in coef_chunks.zip(&coefficients.bounds) {
let chunk_i32: Vec<i32> = chunk
.iter()
.take(bound.size as usize)
.map(|&v| (v * scale).round() as i32)
.collect();
chunks.push(CoefficientsI32Chunk {
start: bound.start,
values: chunk_i32,
});
}
}
let scale = (1i64 << precision) as f64;
for src in coefficients.values.iter().copied() {
values_i32.push((src * scale).round() as i32);
}
Self {
values: values_i32,
precision,
window_size: coefficients.window_size,
bounds: coefficients.bounds,
}
Self { precision, chunks }
}
#[inline]
pub fn normalized_chunks(&self) -> Vec<CoefficientsI32Chunk> {
let mut cooefs = self.values.as_slice();
let mut res = Vec::with_capacity(self.bounds.len());
for bound in self.bounds.iter() {
let (left, right) = cooefs.split_at(self.window_size);
cooefs = right;
let size = bound.size as usize;
res.push(CoefficientsI32Chunk {
start: bound.start,
values: &left[0..size],
});
}
res
#[inline(always)]
pub fn coefficients(&self) -> &[CoefficientsI32Chunk] {
&self.chunks
}
#[inline]
+6 -7
View File
@@ -12,14 +12,13 @@ pub(crate) fn horiz_convolution(
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer32::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_view.height();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
}
}
@@ -28,7 +27,7 @@ pub(crate) fn horiz_convolution(
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
horiz_convolution_one_row(src_row, dst_row, &normalizer);
}
}
}
@@ -42,12 +41,12 @@ pub(crate) fn horiz_convolution(
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U16]; 4],
dst_rows: [&mut [U16]; 4],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
normalizer: &optimisations::Normalizer32,
) {
let precision = normalizer.precision();
let half_error = 1i64 << (precision - 1);
let mut ll_buf = [0i64; 4];
let coefficients_chunks = normalizer.coefficients();
/*
|L0 | |L1 | |L2 | |L3 | |L4 | |L5 | |L6 | |L7 |
@@ -91,7 +90,7 @@ unsafe fn horiz_convolution_four_rows(
let mut x: usize = coeffs_chunk.start as usize;
let mut ll_sum = [_mm256_set1_epi64x(0); 2];
let mut coeffs = coeffs_chunk.values;
let mut coeffs = coeffs_chunk.values();
let coeffs_by_8 = coeffs.chunks_exact(8);
coeffs = coeffs_by_8.remainder();
@@ -201,12 +200,12 @@ unsafe fn horiz_convolution_four_rows(
unsafe fn horiz_convolution_one_row(
src_row: &[U16],
dst_row: &mut [U16],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
normalizer: &optimisations::Normalizer32,
) {
let precision = normalizer.precision();
let half_error = 1i64 << (precision - 1);
let mut ll_buf = [0i64; 4];
let coefficients_chunks = normalizer.coefficients();
/*
|L0 | |L1 | |L2 | |L3 | |L4 | |L5 | |L6 | |L7 |
@@ -249,7 +248,7 @@ unsafe fn horiz_convolution_one_row(
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
let mut ll_sum = _mm256_set1_epi64x(0);
let mut coeffs = coeffs_chunk.values;
let mut coeffs = coeffs_chunk.values();
let coeffs_by_16 = coeffs.chunks_exact(16);
coeffs = coeffs_by_16.remainder();
+3 -3
View File
@@ -11,17 +11,17 @@ pub(crate) fn horiz_convolution(
) {
let normalizer = optimisations::Normalizer32::new(coeffs);
let precision = normalizer.precision();
let coefficients_chunks = normalizer.normalized_chunks();
let coefficients_chunks = normalizer.coefficients();
let initial = 1i64 << (precision - 1);
let src_rows = src_view.iter_rows(offset);
let dst_rows = dst_view.iter_rows_mut(0);
for (dst_row, src_row) in dst_rows.zip(src_rows) {
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
for (coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
let first_x_src = coeffs_chunk.start as usize;
let mut ss = initial;
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
for (&k, src_pixel) in coeffs_chunk.values.iter().zip(src_pixels) {
for (&k, src_pixel) in coeffs_chunk.values().iter().zip(src_pixels) {
ss += src_pixel.0 as i64 * (k as i64);
}
dst_pixel.0 = normalizer.clip(ss);
+6 -7
View File
@@ -12,14 +12,13 @@ pub(crate) fn horiz_convolution(
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer32::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_view.height();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
}
}
@@ -28,7 +27,7 @@ pub(crate) fn horiz_convolution(
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
horiz_convolution_one_row(src_row, dst_row, &normalizer);
}
}
}
@@ -42,12 +41,12 @@ pub(crate) fn horiz_convolution(
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U16]; 4],
dst_rows: [&mut [U16]; 4],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
normalizer: &optimisations::Normalizer32,
) {
let precision = normalizer.precision();
let half_error = 1i64 << (precision - 1);
let mut ll_buf = [0i64; 2];
let coefficients_chunks = normalizer.coefficients();
/*
|L0 | |L1 | |L2 | |L3 | |L4 | |L5 | |L6 | |L7 |
@@ -77,7 +76,7 @@ unsafe fn horiz_convolution_four_rows(
let mut x: usize = coeffs_chunk.start as usize;
let mut ll_sum = [_mm_set1_epi64x(0); 4];
let mut coeffs = coeffs_chunk.values;
let mut coeffs = coeffs_chunk.values();
let coeffs_by_8 = coeffs.chunks_exact(8);
coeffs = coeffs_by_8.remainder();
@@ -169,12 +168,12 @@ unsafe fn horiz_convolution_four_rows(
unsafe fn horiz_convolution_one_row(
src_row: &[U16],
dst_row: &mut [U16],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
normalizer: &optimisations::Normalizer32,
) {
let precision = normalizer.precision();
let half_error = 1i64 << (precision - 1);
let mut ll_buf = [0i64; 2];
let coefficients_chunks = normalizer.coefficients();
/*
|L0 | |L1 | |L2 | |L3 | |L4 | |L5 | |L6 | |L7 |
@@ -203,7 +202,7 @@ unsafe fn horiz_convolution_one_row(
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
let mut ll_sum = _mm_set1_epi64x(0);
let mut coeffs = coeffs_chunk.values;
let mut coeffs = coeffs_chunk.values();
let coeffs_by_8 = coeffs.chunks_exact(8);
coeffs = coeffs_by_8.remainder();
+6 -7
View File
@@ -12,14 +12,13 @@ pub(crate) fn horiz_convolution(
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer32::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_view.height();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
}
}
@@ -28,7 +27,7 @@ pub(crate) fn horiz_convolution(
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
horiz_convolution_one_row(src_row, dst_row, &normalizer);
}
}
}
@@ -42,12 +41,12 @@ pub(crate) fn horiz_convolution(
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U16x2]; 4],
dst_rows: [&mut [U16x2]; 4],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
normalizer: &optimisations::Normalizer32,
) {
let precision = normalizer.precision();
let half_error = 1i64 << (precision - 1);
let mut ll_buf = [0i64; 4];
let coefficients_chunks = normalizer.coefficients();
/*
|L0 A0 | |L1 A1 | |L2 A2 | |L3 A3 |
@@ -91,7 +90,7 @@ unsafe fn horiz_convolution_four_rows(
let mut x: usize = coeffs_chunk.start as usize;
let mut ll_sum = [_mm256_set1_epi64x(half_error); 2];
let mut coeffs = coeffs_chunk.values;
let mut coeffs = coeffs_chunk.values();
let coeffs_by_4 = coeffs.chunks_exact(4);
coeffs = coeffs_by_4.remainder();
@@ -178,12 +177,12 @@ unsafe fn horiz_convolution_four_rows(
unsafe fn horiz_convolution_one_row(
src_row: &[U16x2],
dst_row: &mut [U16x2],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
normalizer: &optimisations::Normalizer32,
) {
let precision = normalizer.precision();
let half_error = 1i64 << (precision - 1);
let mut ll_buf = [0i64; 4];
let coefficients_chunks = normalizer.coefficients();
/*
|L0 A0 | |L1 A1 | |L2 A2 | |L3 A3 |
@@ -226,7 +225,7 @@ unsafe fn horiz_convolution_one_row(
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
let mut ll_sum = _mm256_setzero_si256();
let mut coeffs = coeffs_chunk.values;
let mut coeffs = coeffs_chunk.values();
let coeffs_by_8 = coeffs.chunks_exact(8);
coeffs = coeffs_by_8.remainder();
+3 -3
View File
@@ -11,17 +11,17 @@ pub(crate) fn horiz_convolution(
) {
let normalizer = optimisations::Normalizer32::new(coeffs);
let precision = normalizer.precision();
let coefficients_chunks = normalizer.normalized_chunks();
let coefficients_chunks = normalizer.coefficients();
let initial: i64 = 1 << (precision - 1);
let src_rows = src_view.iter_rows(offset);
let dst_rows = dst_view.iter_rows_mut(0);
for (dst_row, src_row) in dst_rows.zip(src_rows) {
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
for (coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
let first_x_src = coeffs_chunk.start as usize;
let mut ss = [initial; 2];
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
for (&k, src_pixel) in coeffs_chunk.values.iter().zip(src_pixels) {
for (&k, src_pixel) in coeffs_chunk.values().iter().zip(src_pixels) {
for (i, s) in ss.iter_mut().enumerate() {
*s += src_pixel.0[i] as i64 * (k as i64);
}
+6 -7
View File
@@ -12,14 +12,13 @@ pub(crate) fn horiz_convolution(
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer32::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_view.height();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
}
}
@@ -28,7 +27,7 @@ pub(crate) fn horiz_convolution(
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
horiz_convolution_one_row(src_row, dst_row, &normalizer);
}
}
}
@@ -43,12 +42,12 @@ pub(crate) fn horiz_convolution(
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U16x2]; 4],
dst_rows: [&mut [U16x2]; 4],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
normalizer: &optimisations::Normalizer32,
) {
let precision = normalizer.precision();
let half_error = 1i64 << (precision - 1);
let mut ll_buf = [0i64; 2];
let coefficients_chunks = normalizer.coefficients();
/*
|L0 A0 | |L1 A1 | |L2 A2 | |L3 A3 |
@@ -78,7 +77,7 @@ unsafe fn horiz_convolution_four_rows(
let mut x: usize = coeffs_chunk.start as usize;
let mut ll_sum = [_mm_set1_epi64x(half_error); 4];
let mut coeffs = coeffs_chunk.values;
let mut coeffs = coeffs_chunk.values();
let coeffs_by_4 = coeffs.chunks_exact(4);
coeffs = coeffs_by_4.remainder();
@@ -159,12 +158,12 @@ unsafe fn horiz_convolution_four_rows(
unsafe fn horiz_convolution_one_row(
src_row: &[U16x2],
dst_row: &mut [U16x2],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
normalizer: &optimisations::Normalizer32,
) {
let precision = normalizer.precision();
let half_error = 1i64 << (precision - 1);
let mut ll_buf = [0i64; 2];
let coefficients_chunks = normalizer.coefficients();
/*
|L0 A0 | |L1 A1 | |L2 A2 | |L3 A3 |
@@ -193,7 +192,7 @@ unsafe fn horiz_convolution_one_row(
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
let mut ll_sum = _mm_set1_epi64x(half_error);
let mut coeffs = coeffs_chunk.values;
let mut coeffs = coeffs_chunk.values();
let coeffs_by_4 = coeffs.chunks_exact(4);
coeffs = coeffs_by_4.remainder();
+8 -8
View File
@@ -12,14 +12,13 @@ pub(crate) fn horiz_convolution(
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer32::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_view.height();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
}
}
@@ -28,7 +27,7 @@ pub(crate) fn horiz_convolution(
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
horiz_convolution_one_row(src_row, dst_row, &normalizer);
}
}
}
@@ -42,7 +41,7 @@ pub(crate) fn horiz_convolution(
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U16x3]; 4],
dst_rows: [&mut [U16x3]; 4],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
normalizer: &optimisations::Normalizer32,
) {
let precision = normalizer.precision();
@@ -50,6 +49,7 @@ unsafe fn horiz_convolution_four_rows(
let mut rg_buf = [0i64; 4];
let mut rg_bb_buf = [0i64; 4];
let mut bbb_buf = [0i64; 4];
let coefficients_chunks = normalizer.coefficients();
/*
|R G B | |R G B | |R G | - |B | |R G B | |R G B | |R |
@@ -101,7 +101,7 @@ unsafe fn horiz_convolution_four_rows(
let mut rg_bb_sum = [_mm256_set1_epi8(0); 4];
let mut bbb_sum = [_mm256_set1_epi8(0); 4];
let mut coeffs = coeffs_chunk.values;
let mut coeffs = coeffs_chunk.values();
let end_x = x + coeffs.len();
if width - end_x >= 1 {
@@ -177,7 +177,6 @@ unsafe fn horiz_convolution_four_rows(
unsafe fn horiz_convolution_one_row(
src_row: &[U16x3],
dst_row: &mut [U16x3],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
normalizer: &optimisations::Normalizer32,
) {
let precision = normalizer.precision();
@@ -185,6 +184,7 @@ unsafe fn horiz_convolution_one_row(
let mut rg_buf = [0i64; 4];
let mut rg_bb_buf = [0i64; 4];
let mut bbb_buf = [0i64; 4];
let coefficients_chunks = normalizer.coefficients();
/*
|R G B | |R G B | |R G | - |B | |R G B | |R G B | |R |
@@ -232,13 +232,13 @@ unsafe fn horiz_convolution_one_row(
let width = src_row.len();
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
let mut rg_sum = zero_i64x4;
let mut rg_bb_sum = zero_i64x4;
let mut bbb_sum = zero_i64x4;
let mut coeffs = coeffs_chunk.values;
let mut coeffs = coeffs_chunk.values();
let end_x = x + coeffs.len();
if width - end_x >= 1 {
+3 -3
View File
@@ -11,17 +11,17 @@ pub(crate) fn horiz_convolution(
) {
let normalizer = optimisations::Normalizer32::new(coeffs);
let precision = normalizer.precision();
let coefficients_chunks = normalizer.normalized_chunks();
let coefficients_chunks = normalizer.coefficients();
let initial = 1i64 << (precision - 1);
let src_rows = src_view.iter_rows(offset);
let dst_rows = dst_view.iter_rows_mut(0);
for (dst_row, src_row) in dst_rows.zip(src_rows) {
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
for (coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
let first_x_src = coeffs_chunk.start as usize;
let mut ss = [initial; 3];
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
for (&k, src_pixel) in coeffs_chunk.values.iter().zip(src_pixels) {
for (&k, src_pixel) in coeffs_chunk.values().iter().zip(src_pixels) {
for (s, c) in ss.iter_mut().zip(src_pixel.0) {
*s += c as i64 * (k as i64);
}
+7 -8
View File
@@ -12,14 +12,13 @@ pub(crate) fn horiz_convolution(
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer32::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_view.height();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
}
}
@@ -28,7 +27,7 @@ pub(crate) fn horiz_convolution(
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
horiz_convolution_one_row(src_row, dst_row, &normalizer);
}
}
}
@@ -43,13 +42,13 @@ pub(crate) fn horiz_convolution(
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U16x3]; 4],
dst_rows: [&mut [U16x3]; 4],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
normalizer: &optimisations::Normalizer32,
) {
let precision = normalizer.precision();
let half_error = 1i64 << (precision - 1);
let mut rg_buf = [0i64; 2];
let mut bb_buf = [0i64; 2];
let coefficients_chunks = normalizer.coefficients();
/*
|R G B | |R G B | |R G |
@@ -77,7 +76,7 @@ unsafe fn horiz_convolution_four_rows(
let mut rg_sum = [_mm_set1_epi8(0); 4];
let mut bb_sum = [_mm_set1_epi8(0); 4];
let mut coeffs = coeffs_chunk.values;
let mut coeffs = coeffs_chunk.values();
let end_x = x + coeffs.len();
if width - end_x >= 1 {
@@ -138,12 +137,12 @@ unsafe fn horiz_convolution_four_rows(
unsafe fn horiz_convolution_one_row(
src_row: &[U16x3],
dst_row: &mut [U16x3],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
normalizer: &optimisations::Normalizer32,
) {
let precision = normalizer.precision();
let rg_initial = _mm_set1_epi64x(1 << (precision - 1));
let bb_initial = _mm_set1_epi64x(1 << (precision - 2));
let coefficients_chunks = normalizer.coefficients();
/*
|R G B | |R G B | |R G |
@@ -168,13 +167,13 @@ unsafe fn horiz_convolution_one_row(
let width = src_row.len();
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
let mut rg_sum = rg_initial;
let mut bb_sum = bb_initial;
let mut coeffs = coeffs_chunk.values;
let mut coeffs = coeffs_chunk.values();
let end_x = x + coeffs.len();
if width - end_x >= 1 {
+6 -7
View File
@@ -12,14 +12,13 @@ pub(crate) fn horiz_convolution(
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer32::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_view.height();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
}
}
@@ -28,7 +27,7 @@ pub(crate) fn horiz_convolution(
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
horiz_convolution_one_row(src_row, dst_row, &normalizer);
}
}
}
@@ -43,13 +42,13 @@ pub(crate) fn horiz_convolution(
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U16x4]; 4],
dst_rows: [&mut [U16x4]; 4],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
normalizer: &optimisations::Normalizer32,
) {
let precision = normalizer.precision();
let half_error = 1i64 << (precision - 1);
let mut rg_buf = [0i64; 4];
let mut ba_buf = [0i64; 4];
let coefficients_chunks = normalizer.coefficients();
/*
|R0 G0 B0 A0 | |R1 G1 B1 A1 |
@@ -93,7 +92,7 @@ unsafe fn horiz_convolution_four_rows(
let mut rg_sum = [_mm256_set1_epi64x(half_error); 2];
let mut ba_sum = [_mm256_set1_epi64x(half_error); 2];
let mut coeffs = coeffs_chunk.values;
let mut coeffs = coeffs_chunk.values();
let coeffs_by_2 = coeffs.chunks_exact(2);
coeffs = coeffs_by_2.remainder();
@@ -179,13 +178,13 @@ unsafe fn horiz_convolution_four_rows(
unsafe fn horiz_convolution_one_row(
src_row: &[U16x4],
dst_row: &mut [U16x4],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
normalizer: &optimisations::Normalizer32,
) {
let precision = normalizer.precision();
let half_error = 1i64 << (precision - 1);
let mut rg_buf = [0i64; 4];
let mut ba_buf = [0i64; 4];
let coefficients_chunks = normalizer.coefficients();
/*
|R0 G0 B0 A0 | |R1 G1 B1 A1 |
@@ -227,7 +226,7 @@ unsafe fn horiz_convolution_one_row(
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
let mut coeffs = coeffs_chunk.values;
let mut coeffs = coeffs_chunk.values();
let mut rg_sum = _mm256_setzero_si256();
let mut ba_sum = _mm256_setzero_si256();
+3 -3
View File
@@ -11,17 +11,17 @@ pub(crate) fn horiz_convolution(
) {
let normalizer = optimisations::Normalizer32::new(coeffs);
let precision = normalizer.precision();
let coefficients_chunks = normalizer.normalized_chunks();
let coefficients_chunks = normalizer.coefficients();
let initial: i64 = 1 << (precision - 1);
let src_rows = src_view.iter_rows(offset);
let dst_rows = dst_view.iter_rows_mut(0);
for (dst_row, src_row) in dst_rows.zip(src_rows) {
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
for (coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
let first_x_src = coeffs_chunk.start as usize;
let mut ss = [initial; 4];
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
for (&k, src_pixel) in coeffs_chunk.values.iter().zip(src_pixels) {
for (&k, src_pixel) in coeffs_chunk.values().iter().zip(src_pixels) {
for (i, s) in ss.iter_mut().enumerate() {
*s += src_pixel.0[i] as i64 * (k as i64);
}
+6 -7
View File
@@ -12,14 +12,13 @@ pub(crate) fn horiz_convolution(
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer32::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_view.height();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
}
}
@@ -28,7 +27,7 @@ pub(crate) fn horiz_convolution(
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
horiz_convolution_one_row(src_row, dst_row, &normalizer);
}
}
}
@@ -43,13 +42,13 @@ pub(crate) fn horiz_convolution(
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U16x4]; 4],
dst_rows: [&mut [U16x4]; 4],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
normalizer: &optimisations::Normalizer32,
) {
let precision = normalizer.precision();
let half_error = 1i64 << (precision - 1);
let mut rg_buf = [0i64; 2];
let mut ba_buf = [0i64; 2];
let coefficients_chunks = normalizer.coefficients();
/*
|R0 G0 B0 A0 | |R1 G1 B1 A1 |
@@ -80,7 +79,7 @@ unsafe fn horiz_convolution_four_rows(
let mut rg_sum = [_mm_set1_epi64x(half_error); 4];
let mut ba_sum = [_mm_set1_epi64x(half_error); 4];
let mut coeffs = coeffs_chunk.values;
let mut coeffs = coeffs_chunk.values();
let coeffs_by_2 = coeffs.chunks_exact(2);
coeffs = coeffs_by_2.remainder();
@@ -143,13 +142,13 @@ unsafe fn horiz_convolution_four_rows(
unsafe fn horiz_convolution_one_row(
src_row: &[U16x4],
dst_row: &mut [U16x4],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
normalizer: &optimisations::Normalizer32,
) {
let precision = normalizer.precision();
let half_error = 1i64 << (precision - 1);
let mut rg_buf = [0i64; 2];
let mut ba_buf = [0i64; 2];
let coefficients_chunks = normalizer.coefficients();
/*
|R0 G0 B0 A0 | |R1 G1 B1 A1 |
@@ -177,7 +176,7 @@ unsafe fn horiz_convolution_one_row(
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
let mut coeffs = coeffs_chunk.values;
let mut coeffs = coeffs_chunk.values();
let mut rg_sum = _mm_set1_epi64x(half_error);
let mut ba_sum = _mm_set1_epi64x(half_error);
+7 -8
View File
@@ -12,14 +12,13 @@ pub(crate) fn horiz_convolution(
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer16::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_view.height();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
}
}
@@ -28,7 +27,7 @@ pub(crate) fn horiz_convolution(
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
horiz_convolution_one_row(src_row, dst_row, &normalizer);
}
}
}
@@ -44,15 +43,15 @@ pub(crate) fn horiz_convolution(
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U8]; 4],
dst_rows: [&mut [U8]; 4],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
normalizer: &optimisations::Normalizer16,
) {
let zero = _mm_setzero_si128();
// 8 components will be added, use only 1/8 of the error
let initial = _mm256_set1_epi32(1 << (normalizer.precision() - 4));
let coefficients_chunks = normalizer.coefficients();
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let coeffs = coeffs_chunk.values;
let coeffs = coeffs_chunk.values();
let mut x = coeffs_chunk.start as usize;
let mut result_i32x8x4 = [initial, initial, initial, initial];
@@ -113,15 +112,15 @@ unsafe fn horiz_convolution_four_rows(
unsafe fn horiz_convolution_one_row(
src_row: &[U8],
dst_row: &mut [U8],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
normalizer: &optimisations::Normalizer16,
) {
let zero = _mm_setzero_si128();
// 8 components will be added, use only 1/8 of the error
let initial = _mm256_set1_epi32(1 << (normalizer.precision() - 4));
let coefficients_chunks = normalizer.coefficients();
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let coeffs = coeffs_chunk.values;
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let coeffs = coeffs_chunk.values();
let mut x = coeffs_chunk.start as usize;
let mut result_i32x8 = initial;
+3 -3
View File
@@ -11,17 +11,17 @@ pub(crate) fn horiz_convolution(
) {
let normalizer = optimisations::Normalizer16::new(coeffs);
let precision = normalizer.precision();
let coefficients_chunks = normalizer.normalized_chunks();
let coefficients_chunks = normalizer.coefficients();
let initial = 1i32 << (precision - 1);
let src_rows = src_view.iter_rows(offset);
let dst_rows = dst_view.iter_rows_mut(0);
for (dst_row, src_row) in dst_rows.zip(src_rows) {
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
for (coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
let first_x_src = coeffs_chunk.start as usize;
let mut ss = initial;
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
for (&k, &src_pixel) in coeffs_chunk.values.iter().zip(src_pixels) {
for (&k, &src_pixel) in coeffs_chunk.values().iter().zip(src_pixels) {
ss += src_pixel.0 as i32 * (k as i32);
}
dst_pixel.0 = unsafe { normalizer.clip(ss) };
+7 -8
View File
@@ -12,14 +12,13 @@ pub(crate) fn horiz_convolution(
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer16::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_view.height();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
}
}
@@ -28,7 +27,7 @@ pub(crate) fn horiz_convolution(
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
horiz_convolution_one_row(src_row, dst_row, &normalizer);
}
}
}
@@ -44,15 +43,15 @@ pub(crate) fn horiz_convolution(
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U8]; 4],
dst_rows: [&mut [U8]; 4],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
normalizer: &optimisations::Normalizer16,
) {
let zero = _mm_setzero_si128();
let initial = 1 << (normalizer.precision() - 1);
let mut buf = [0, 0, 0, 0, initial];
let coefficients_chunks = normalizer.coefficients();
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let coeffs = coeffs_chunk.values;
let coeffs = coeffs_chunk.values();
let mut x = coeffs_chunk.start as usize;
let mut result_i32x4 = [zero, zero, zero, zero];
@@ -112,15 +111,15 @@ unsafe fn horiz_convolution_four_rows(
unsafe fn horiz_convolution_one_row(
src_row: &[U8],
dst_row: &mut [U8],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
normalizer: &optimisations::Normalizer16,
) {
let zero = _mm_setzero_si128();
let initial = 1 << (normalizer.precision() - 1);
let mut buf = [0, 0, 0, 0, initial];
let coefficients_chunks = normalizer.coefficients();
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let coeffs = coeffs_chunk.values;
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let coeffs = coeffs_chunk.values();
let mut x = coeffs_chunk.start as usize;
let mut result_i32x4 = zero;
+7 -8
View File
@@ -12,14 +12,13 @@ pub(crate) fn horiz_convolution(
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer16::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_view.height();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
}
}
@@ -28,7 +27,7 @@ pub(crate) fn horiz_convolution(
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
horiz_convolution_one_row(src_row, dst_row, &normalizer);
}
}
}
@@ -44,11 +43,11 @@ pub(crate) fn horiz_convolution(
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U8x2]; 4],
dst_rows: [&mut [U8x2]; 4],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
normalizer: &optimisations::Normalizer16,
) {
let precision = normalizer.precision();
let initial = _mm256_set1_epi32(1 << (precision - 2));
let coefficients_chunks = normalizer.coefficients();
/*
|L A | |L A | |L A | |L A | |L A | |L A | |L A | |L A |
@@ -79,7 +78,7 @@ unsafe fn horiz_convolution_four_rows(
let mut sss0 = initial;
let mut sss1 = initial;
let coeffs = coeffs_chunk.values;
let coeffs = coeffs_chunk.values();
let coeffs_by_8 = coeffs.chunks_exact(8);
let reminder = coeffs_by_8.remainder();
@@ -214,7 +213,6 @@ unsafe fn set_dst_pixel(
unsafe fn horiz_convolution_one_row(
src_row: &[U8x2],
dst_row: &mut [U8x2],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
normalizer: &optimisations::Normalizer16,
) {
let precision = normalizer.precision();
@@ -315,10 +313,11 @@ unsafe fn horiz_convolution_one_row(
L: |-1 02| |-1 00|
*/
let pix_sh4 = _mm_set_epi8(-1, 7, -1, 5, -1, 6, -1, 4, -1, 3, -1, 1, -1, 2, -1, 0);
let coefficients_chunks = normalizer.coefficients();
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x = coeffs_chunk.start as usize;
let mut coeffs = coeffs_chunk.values;
let mut coeffs = coeffs_chunk.values();
let mut sss = if coeffs.len() < 16 {
// Lower part will be added to higher, use only half of the error
+3 -3
View File
@@ -10,15 +10,15 @@ pub(crate) fn horiz_convolution(
) {
let normalizer = optimisations::Normalizer16::new(coeffs);
let precision = normalizer.precision();
let coefficients_chunks = normalizer.normalized_chunks();
let coefficients_chunks = normalizer.coefficients();
let initial = 1 << (precision - 1);
let src_rows = src_view.iter_rows(offset);
let dst_rows = dst_view.iter_rows_mut(0);
for (dst_row, src_row) in dst_rows.zip(src_rows) {
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
for (coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
let first_x_src = coeffs_chunk.start as usize;
let ks = coeffs_chunk.values;
let ks = coeffs_chunk.values();
let mut ss = [initial; 2];
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
for (&k, &src_pixel) in ks.iter().zip(src_pixels) {
+7 -8
View File
@@ -12,14 +12,13 @@ pub(crate) fn horiz_convolution(
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer16::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_view.height();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
}
}
@@ -28,7 +27,7 @@ pub(crate) fn horiz_convolution(
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
horiz_convolution_one_row(src_row, dst_row, &normalizer);
}
}
}
@@ -44,11 +43,11 @@ pub(crate) fn horiz_convolution(
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U8x2]; 4],
dst_rows: [&mut [U8x2]; 4],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
normalizer: &optimisations::Normalizer16,
) {
let precision = normalizer.precision();
let initial = _mm_set1_epi32(1 << (precision - 2));
let coefficients_chunks = normalizer.coefficients();
/*
|L A | |L A | |L A | |L A | |L A | |L A | |L A | |L A |
@@ -76,7 +75,7 @@ unsafe fn horiz_convolution_four_rows(
let mut x = coeffs_chunk.start as usize;
let mut sss: [__m128i; 4] = [initial; 4];
let coeffs = coeffs_chunk.values;
let coeffs = coeffs_chunk.values();
let coeffs_by_8 = coeffs.chunks_exact(8);
let reminder = coeffs_by_8.remainder();
@@ -166,7 +165,6 @@ unsafe fn set_dst_pixel(
unsafe fn horiz_convolution_one_row(
src_row: &[U8x2],
dst_row: &mut [U8x2],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
normalizer: &optimisations::Normalizer16,
) {
let precision = normalizer.precision();
@@ -244,10 +242,11 @@ unsafe fn horiz_convolution_one_row(
L: |-1 02| |-1 00|
*/
let pix_sh3 = _mm_set_epi8(-1, 7, -1, 5, -1, 6, -1, 4, -1, 3, -1, 1, -1, 2, -1, 0);
let coefficients_chunks = normalizer.coefficients();
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x = coeffs_chunk.start as usize;
let mut coeffs = coeffs_chunk.values;
let mut coeffs = coeffs_chunk.values();
// Lower part will be added to higher, use only half of the error
let mut sss = _mm_set1_epi32(1 << (precision - 2));
+9 -8
View File
@@ -29,14 +29,13 @@ fn horiz_convolution_p<const PRECISION: i32>(
offset: u32,
normalizer: optimisations::Normalizer16,
) {
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_view.height();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &coefficients_chunks);
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &normalizer);
}
}
@@ -45,7 +44,7 @@ fn horiz_convolution_p<const PRECISION: i32>(
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &coefficients_chunks);
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &normalizer);
}
}
}
@@ -61,11 +60,12 @@ fn horiz_convolution_p<const PRECISION: i32>(
unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
src_rows: [&[U8x3]; 4],
dst_rows: [&mut [U8x3]; 4],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
normalizer: &optimisations::Normalizer16,
) {
let zero = _mm256_setzero_si256();
let initial = _mm256_set1_epi32(1 << (PRECISION - 1));
let src_width = src_rows[0].len();
let coefficients_chunks = normalizer.coefficients();
/*
|R G B | |R G B | |R G B | |R G B | |R G B | |R |
@@ -102,7 +102,7 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
let mut sss0 = initial;
let mut sss1 = initial;
let mut coeffs = coeffs_chunk.values;
let mut coeffs = coeffs_chunk.values();
// (16 bytes) / (3 bytes per pixel) = 5 whole pixels + 1 byte
let max_x = src_width.saturating_sub(5);
@@ -224,7 +224,7 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
src_row: &[U8x3],
dst_row: &mut [U8x3],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
normalizer: &optimisations::Normalizer16,
) {
#[rustfmt::skip]
let sh1 = _mm256_set_epi8(
@@ -271,11 +271,12 @@ unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
*/
let sh7 = _mm_set_epi8(-1, -1, -1, -1, -1, 5, -1, 2, -1, 4, -1, 1, -1, 3, -1, 0);
let src_width = src_row.len();
let coefficients_chunks = normalizer.coefficients();
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let x_start = coeffs_chunk.start as usize;
let mut x = x_start;
let mut coeffs = coeffs_chunk.values;
let mut coeffs = coeffs_chunk.values();
// (16 bytes) / (3 bytes per pixel) = 5 whole pixels + 1 bytes
// 4 + 5 = 9
+3 -3
View File
@@ -11,17 +11,17 @@ pub(crate) fn horiz_convolution(
) {
let normalizer = optimisations::Normalizer16::new(coeffs);
let precision = normalizer.precision();
let coefficients_chunks = normalizer.normalized_chunks();
let coefficients = normalizer.coefficients();
let initial = 1i32 << (precision - 1);
let src_rows = src_view.iter_rows(offset);
let dst_rows = dst_view.iter_rows_mut(0);
for (dst_row, src_row) in dst_rows.zip(src_rows) {
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
for (coeffs_chunk, dst_pixel) in coefficients.iter().zip(dst_row.iter_mut()) {
let first_x_src = coeffs_chunk.start as usize;
let mut ss = [initial; 3];
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
for (&k, src_pixel) in coeffs_chunk.values.iter().zip(src_pixels) {
for (&k, src_pixel) in coeffs_chunk.values().iter().zip(src_pixels) {
for (s, c) in ss.iter_mut().zip(src_pixel.0) {
*s += c as i32 * (k as i32);
}
+9 -8
View File
@@ -29,14 +29,13 @@ fn horiz_convolution_p<const PRECISION: i32>(
offset: u32,
normalizer: optimisations::Normalizer16,
) {
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_view.height();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &coefficients_chunks);
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &normalizer);
}
}
@@ -45,7 +44,7 @@ fn horiz_convolution_p<const PRECISION: i32>(
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &coefficients_chunks);
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &normalizer);
}
}
}
@@ -61,11 +60,12 @@ fn horiz_convolution_p<const PRECISION: i32>(
unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
src_rows: [&[U8x3]; 4],
dst_rows: [&mut [U8x3]; 4],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
normalizer: &optimisations::Normalizer16,
) {
let zero = _mm_setzero_si128();
let initial = _mm_set1_epi32(1 << (PRECISION - 1));
let src_width = src_rows[0].len();
let coefficients_chunks = normalizer.coefficients();
/*
|R G B | |R G B | |R G B | |R G B | |R G B | |R |
@@ -99,7 +99,7 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
let mut x = x_start;
let mut sss_a = [initial; 4];
let mut coeffs = coeffs_chunk.values;
let mut coeffs = coeffs_chunk.values();
// Next block of code will be load source pixels by 16 bytes per time.
// We must guarantee what this process will not go beyond
@@ -187,7 +187,7 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
src_row: &[U8x3],
dst_row: &mut [U8x3],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
normalizer: &optimisations::Normalizer16,
) {
#[rustfmt::skip]
let pix_sh1 = _mm_set_epi8(
@@ -219,11 +219,12 @@ unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
R: |-1 03| |-1 00|
*/
let src_width = src_row.len();
let coefficients_chunks = normalizer.coefficients();
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let x_start = coeffs_chunk.start as usize;
let mut x = x_start;
let mut coeffs = coeffs_chunk.values;
let mut coeffs = coeffs_chunk.values();
let mut sss = _mm_set1_epi32(1 << (PRECISION - 1));
// Next block of code will be load source pixels by 16 bytes per time.
+16 -13
View File
@@ -32,14 +32,13 @@ fn horiz_convolution_p<const PRECISION: i32>(
offset: u32,
normalizer: optimisations::Normalizer16,
) {
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_view.height();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &coefficients_chunks);
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &normalizer);
}
}
@@ -48,7 +47,7 @@ fn horiz_convolution_p<const PRECISION: i32>(
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &coefficients_chunks);
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &normalizer);
}
}
}
@@ -64,7 +63,7 @@ fn horiz_convolution_p<const PRECISION: i32>(
unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
src_rows: [&[U8x4]; 4],
dst_rows: [&mut [U8x4]; 4],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
normalizer: &optimisations::Normalizer16,
) {
let zero = _mm256_setzero_si256();
let initial = _mm256_set1_epi32(1 << (PRECISION - 1));
@@ -80,12 +79,14 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
);
let coefficients_chunks = normalizer.coefficients();
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x = coeffs_chunk.start as usize;
let mut sss0 = initial;
let mut sss1 = initial;
let coeffs = coeffs_chunk.values;
let coeffs = coeffs_chunk.values();
let coeffs_by_4 = coeffs.chunks_exact(4);
let reminder1 = coeffs_by_4.remainder();
@@ -164,13 +165,13 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
sss0 = _mm256_packus_epi16(sss0, zero);
sss1 = _mm256_packus_epi16(sss1, zero);
*dst_rows[0].get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm256_extracti128_si256::<0>(sss0)));
transmute::<i32, U8x4>(_mm_cvtsi128_si32(_mm256_extracti128_si256::<0>(sss0)));
*dst_rows[1].get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm256_extracti128_si256::<1>(sss0)));
transmute::<i32, U8x4>(_mm_cvtsi128_si32(_mm256_extracti128_si256::<1>(sss0)));
*dst_rows[2].get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm256_extracti128_si256::<0>(sss1)));
transmute::<i32, U8x4>(_mm_cvtsi128_si32(_mm256_extracti128_si256::<0>(sss1)));
*dst_rows[3].get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm256_extracti128_si256::<1>(sss1)));
transmute::<i32, U8x4>(_mm_cvtsi128_si32(_mm256_extracti128_si256::<1>(sss1)));
}
}
@@ -184,7 +185,7 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
src_row: &[U8x4],
dst_row: &mut [U8x4],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
normalizer: &optimisations::Normalizer16,
) {
#[rustfmt::skip]
let sh1 = _mm256_set_epi8(
@@ -218,9 +219,11 @@ unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
);
let sh7 = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let coefficients_chunks = normalizer.coefficients();
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x = coeffs_chunk.start as usize;
let mut coeffs = coeffs_chunk.values;
let mut coeffs = coeffs_chunk.values();
let mut sss = if coeffs.len() < 8 {
_mm_set1_epi32(1 << (PRECISION - 1))
@@ -293,6 +296,6 @@ unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
sss = _mm_packs_epi32(sss, sss);
*dst_row.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss, sss)));
transmute::<i32, U8x4>(_mm_cvtsi128_si32(_mm_packus_epi16(sss, sss)));
}
}
+4 -5
View File
@@ -11,17 +11,16 @@ pub(crate) fn horiz_convolution(
) {
let normalizer = optimisations::Normalizer16::new(coeffs);
let precision = normalizer.precision();
let coefficients_chunks = normalizer.normalized_chunks();
let coefficients = normalizer.coefficients();
let initial = 1 << (precision - 1);
let src_rows = src_view.iter_rows(offset);
let dst_rows = dst_view.iter_rows_mut(0);
for (dst_row, src_row) in dst_rows.zip(src_rows) {
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
let first_x_src = coeffs_chunk.start as usize;
for (chunk, dst_pixel) in coefficients.iter().zip(dst_row.iter_mut()) {
let mut ss = [initial; 4];
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
for (&k, &src_pixel) in coeffs_chunk.values.iter().zip(src_pixels) {
let src_pixels = unsafe { src_row.get_unchecked(chunk.start as usize..) };
for (&k, &src_pixel) in chunk.values().iter().zip(src_pixels) {
for (i, s) in ss.iter_mut().enumerate() {
*s += src_pixel.0[i] as i32 * (k as i32);
}
+11 -13
View File
@@ -32,14 +32,13 @@ fn horiz_convolution_p<const PRECISION: i32>(
offset: u32,
normalizer: optimisations::Normalizer16,
) {
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_view.height();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &coefficients_chunks);
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &normalizer);
}
}
@@ -48,7 +47,7 @@ fn horiz_convolution_p<const PRECISION: i32>(
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &coefficients_chunks);
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &normalizer);
}
}
}
@@ -63,23 +62,22 @@ fn horiz_convolution_p<const PRECISION: i32>(
unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
src_rows: [&[U8x4]; 4],
dst_rows: [&mut [U8x4]; 4],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
normalizer: &optimisations::Normalizer16,
) {
let initial = _mm_set1_epi32(1 << (PRECISION - 1));
let mask_lo = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
let mask_hi = _mm_set_epi8(-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8);
let mask = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
for (dst_x, chunk) in normalizer.coefficients().iter().enumerate() {
let mut x: usize = chunk.start as usize;
let mut sss0 = initial;
let mut sss1 = initial;
let mut sss2 = initial;
let mut sss3 = initial;
let coeffs = coeffs_chunk.values;
let coeffs_by_4 = coeffs.chunks_exact(4);
let coeffs_by_4 = chunk.values().chunks_exact(4);
let reminder1 = coeffs_by_4.remainder();
for k in coeffs_by_4 {
@@ -189,7 +187,7 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
src_row: &[U8x4],
dst_row: &mut [U8x4],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
normalizer: &optimisations::Normalizer16,
) {
let initial = _mm_set1_epi32(1 << (PRECISION - 1));
let sh1 = _mm_set_epi8(-1, 11, -1, 3, -1, 10, -1, 2, -1, 9, -1, 1, -1, 8, -1, 0);
@@ -202,11 +200,11 @@ unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
);
let sh7 = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
for (dst_x, chunk) in normalizer.coefficients().iter().enumerate() {
let mut x: usize = chunk.start as usize;
let mut sss = initial;
let coeffs_by_8 = coeffs_chunk.values.chunks_exact(8);
let coeffs_by_8 = chunk.values().chunks_exact(8);
let reminder8 = coeffs_by_8.remainder();
for k in coeffs_by_8 {
@@ -274,6 +272,6 @@ unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
sss = _mm_srai_epi32::<PRECISION>(sss);
sss = _mm_packs_epi32(sss, sss);
*dst_row.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss, sss)));
transmute::<i32, U8x4>(_mm_cvtsi128_si32(_mm_packus_epi16(sss, sss)));
}
}
+3 -3
View File
@@ -13,7 +13,7 @@ pub(crate) fn vert_convolution<T>(
T: InnerPixel<Component = u16>,
{
let normalizer = optimisations::Normalizer32::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let coefficients_chunks = normalizer.coefficients();
let src_x = offset as usize * T::count_of_components();
let dst_rows = dst_view.iter_rows_mut(0);
@@ -29,13 +29,13 @@ pub(crate) unsafe fn vert_convolution_into_one_row_u16<T>(
src_view: &impl ImageView<Pixel = T>,
dst_row: &mut [T],
mut src_x: usize,
coeffs_chunk: optimisations::CoefficientsI32Chunk,
coeffs_chunk: &optimisations::CoefficientsI32Chunk,
normalizer: &optimisations::Normalizer32,
) where
T: InnerPixel<Component = u16>,
{
let y_start = coeffs_chunk.start;
let coeffs = coeffs_chunk.values;
let coeffs = coeffs_chunk.values();
let mut dst_u16 = T::components_mut(dst_row);
/*
+2 -2
View File
@@ -13,7 +13,7 @@ pub(crate) fn vert_convolution<T>(
T: InnerPixel<Component = u16>,
{
let normalizer = optimisations::Normalizer32::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let coefficients_chunks = normalizer.coefficients();
let precision = normalizer.precision();
let initial: i64 = 1 << (precision - 1);
let src_x_initial = offset as usize * T::count_of_components();
@@ -22,7 +22,7 @@ pub(crate) fn vert_convolution<T>(
let coeffs_chunks_iter = coefficients_chunks.into_iter();
for (coeffs_chunk, dst_row) in coeffs_chunks_iter.zip(dst_rows) {
let first_y_src = coeffs_chunk.start;
let ks = coeffs_chunk.values;
let ks = coeffs_chunk.values();
let dst_components = T::components_mut(dst_row);
let mut x_src = src_x_initial;
+3 -3
View File
@@ -15,7 +15,7 @@ pub(crate) fn vert_convolution<T>(
T: InnerPixel<Component = u16>,
{
let normalizer = optimisations::Normalizer32::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let coefficients_chunks = normalizer.coefficients();
let src_x = offset as usize * T::count_of_components();
let dst_rows = dst_view.iter_rows_mut(0);
@@ -31,11 +31,11 @@ unsafe fn vert_convolution_into_one_row_u16<T: InnerPixel<Component = u16>>(
src_view: &impl ImageView<Pixel = T>,
dst_row: &mut [T],
mut src_x: usize,
coeffs_chunk: CoefficientsI32Chunk,
coeffs_chunk: &CoefficientsI32Chunk,
normalizer: &optimisations::Normalizer32,
) {
let y_start = coeffs_chunk.start;
let coeffs = coeffs_chunk.values;
let coeffs = coeffs_chunk.values();
let max_rows = coeffs.len() as u32;
let mut dst_u16 = T::components_mut(dst_row);
+3 -3
View File
@@ -34,7 +34,7 @@ fn vert_convolution_p<T, const PRECISION: i32>(
) where
T: InnerPixel<Component = u8>,
{
let coefficients_chunks = normalizer.normalized_chunks();
let coefficients_chunks = normalizer.coefficients();
let src_x = offset as usize * T::count_of_components();
let dst_rows = dst_view.iter_rows_mut(0);
@@ -57,13 +57,13 @@ unsafe fn vert_convolution_into_one_row<T, const PRECISION: i32>(
src_view: &impl ImageView<Pixel = T>,
dst_row: &mut [T],
mut src_x: usize,
coeffs_chunk: optimisations::CoefficientsI16Chunk,
coeffs_chunk: &optimisations::CoefficientsI16Chunk,
normalizer: &optimisations::Normalizer16,
) where
T: InnerPixel<Component = u8>,
{
let y_start = coeffs_chunk.start;
let coeffs = coeffs_chunk.values;
let coeffs = coeffs_chunk.values();
let max_rows = coeffs.len() as u32;
let initial = _mm_set1_epi32(1 << (PRECISION as u8 - 1));
+3 -3
View File
@@ -13,16 +13,16 @@ pub(crate) fn vert_convolution<T>(
T: InnerPixel<Component = u8>,
{
let normalizer = optimisations::Normalizer16::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let coefficients_chunks = normalizer.coefficients();
let precision = normalizer.precision();
let initial = 1 << (precision - 1);
let src_x_initial = offset as usize * T::count_of_components();
let dst_rows = dst_image.iter_rows_mut(0);
let coeffs_chunks_iter = coefficients_chunks.into_iter();
let coeffs_chunks_iter = coefficients_chunks.iter();
for (coeffs_chunk, dst_row) in coeffs_chunks_iter.zip(dst_rows) {
let first_y_src = coeffs_chunk.start;
let ks = coeffs_chunk.values;
let ks = coeffs_chunk.values();
let mut x_src = src_x_initial;
let dst_components = T::components_mut(dst_row);
+3 -3
View File
@@ -33,7 +33,7 @@ fn vert_convolution_p<T, const PRECISION: i32>(
) where
T: InnerPixel<Component = u8>,
{
let coefficients_chunks = normalizer.normalized_chunks();
let coefficients_chunks = normalizer.coefficients();
let src_x = offset as usize * T::count_of_components();
let dst_rows = dst_view.iter_rows_mut(0);
@@ -55,13 +55,13 @@ unsafe fn vert_convolution_into_one_row<T, const PRECISION: i32>(
src_view: &impl ImageView<Pixel = T>,
dst_row: &mut [T],
mut src_x: usize,
coeffs_chunk: optimisations::CoefficientsI16Chunk,
coeffs_chunk: &optimisations::CoefficientsI16Chunk,
normalizer: &optimisations::Normalizer16,
) where
T: InnerPixel<Component = u8>,
{
let y_start = coeffs_chunk.start;
let coeffs = coeffs_chunk.values;
let coeffs = coeffs_chunk.values();
let max_rows = coeffs.len() as u32;
let mut dst_u8 = T::components_mut(dst_row);
+15 -15
View File
@@ -3,7 +3,7 @@ use std::io::BufReader;
use std::num::NonZeroU32;
use std::ops::Deref;
use image::io::{Reader as ImageReader, Reader};
use image::ImageReader;
use image::{ColorType, ExtendedColorType, ImageBuffer};
use fast_image_resize::images::Image;
@@ -112,7 +112,7 @@ pub trait PixelTestingExt: PixelTrait {
}
fn load_image_buffer(
img_reader: Reader<BufReader<File>>,
img_reader: ImageReader<BufReader<File>>,
) -> ImageBuffer<Self::ImagePixel, Self::Container>;
fn load_big_image() -> ImageBuffer<Self::ImagePixel, Self::Container> {
@@ -172,7 +172,7 @@ pub mod not_u8x4 {
type Container = Vec<u8>;
fn load_image_buffer(
img_reader: Reader<BufReader<File>>,
img_reader: ImageReader<BufReader<File>>,
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
img_reader.decode().unwrap().to_luma8()
}
@@ -187,7 +187,7 @@ pub mod not_u8x4 {
type Container = Vec<u8>;
fn load_image_buffer(
img_reader: Reader<BufReader<File>>,
img_reader: ImageReader<BufReader<File>>,
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
img_reader.decode().unwrap().to_luma_alpha8()
}
@@ -202,7 +202,7 @@ pub mod not_u8x4 {
type Container = Vec<u8>;
fn load_image_buffer(
img_reader: Reader<BufReader<File>>,
img_reader: ImageReader<BufReader<File>>,
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
img_reader.decode().unwrap().to_rgb8()
}
@@ -217,7 +217,7 @@ pub mod not_u8x4 {
type Container = Vec<u16>;
fn load_image_buffer(
img_reader: Reader<BufReader<File>>,
img_reader: ImageReader<BufReader<File>>,
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
img_reader.decode().unwrap().to_luma16()
}
@@ -237,7 +237,7 @@ pub mod not_u8x4 {
type Container = Vec<u16>;
fn load_image_buffer(
img_reader: Reader<BufReader<File>>,
img_reader: ImageReader<BufReader<File>>,
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
img_reader.decode().unwrap().to_luma_alpha16()
}
@@ -252,7 +252,7 @@ pub mod not_u8x4 {
type Container = Vec<u16>;
fn load_image_buffer(
img_reader: Reader<BufReader<File>>,
img_reader: ImageReader<BufReader<File>>,
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
img_reader.decode().unwrap().to_rgb16()
}
@@ -267,7 +267,7 @@ pub mod not_u8x4 {
type Container = Vec<u16>;
fn load_image_buffer(
img_reader: Reader<BufReader<File>>,
img_reader: ImageReader<BufReader<File>>,
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
img_reader.decode().unwrap().to_rgba16()
}
@@ -286,7 +286,7 @@ pub mod not_u8x4 {
}
fn load_image_buffer(
img_reader: Reader<BufReader<File>>,
img_reader: ImageReader<BufReader<File>>,
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
let image_u16 = img_reader.decode().unwrap().to_luma32f();
ImageBuffer::from_fn(image_u16.width(), image_u16.height(), |x, y| {
@@ -308,7 +308,7 @@ pub mod not_u8x4 {
type Container = Vec<f32>;
fn load_image_buffer(
img_reader: Reader<BufReader<File>>,
img_reader: ImageReader<BufReader<File>>,
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
img_reader.decode().unwrap().to_luma32f()
}
@@ -326,7 +326,7 @@ pub mod not_u8x4 {
type Container = Vec<f32>;
fn load_image_buffer(
img_reader: Reader<BufReader<File>>,
img_reader: ImageReader<BufReader<File>>,
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
img_reader.decode().unwrap().to_luma_alpha32f()
}
@@ -344,7 +344,7 @@ pub mod not_u8x4 {
type Container = Vec<f32>;
fn load_image_buffer(
img_reader: Reader<BufReader<File>>,
img_reader: ImageReader<BufReader<File>>,
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
img_reader.decode().unwrap().to_rgb32f()
}
@@ -359,7 +359,7 @@ pub mod not_u8x4 {
type Container = Vec<f32>;
fn load_image_buffer(
img_reader: Reader<BufReader<File>>,
img_reader: ImageReader<BufReader<File>>,
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
img_reader.decode().unwrap().to_rgba32f()
}
@@ -375,7 +375,7 @@ impl PixelTestingExt for U8x4 {
type Container = Vec<u8>;
fn load_image_buffer(
img_reader: Reader<BufReader<File>>,
img_reader: ImageReader<BufReader<File>>,
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
img_reader.decode().unwrap().to_rgba8()
}
+2 -4
View File
@@ -1,8 +1,6 @@
use std::cmp::Ordering;
use std::fmt::Debug;
use image::io::Reader as ImageReader;
use fast_image_resize::images::{Image, TypedImage, TypedImageRef};
use fast_image_resize::pixels::*;
use fast_image_resize::{
@@ -865,9 +863,9 @@ mod not_u8x4 {
}
mod u8x4 {
use std::f64::consts::PI;
use fast_image_resize::ResizeError;
use image::ImageReader;
use std::f64::consts::PI;
use super::*;