mirror of
https://github.com/Cykooz/fast_image_resize.git
synced 2026-10-08 01:11:09 +00:00
- Updated dependencies.
- Optimized convolution algorythm by deleting zero coefficients from start and end of bounds.
This commit is contained in:
Generated
+127
-117
@@ -46,9 +46,9 @@ checksum = "4b46cbb362ab8752921c97e041f5e366ee6297bd428a31275b9fcf1e380f7299"
|
||||
|
||||
[[package]]
|
||||
name = "anstream"
|
||||
version = "0.6.14"
|
||||
version = "0.6.15"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "418c75fa768af9c03be99d17643f93f79bbba589895012a80e3452a19ddda15b"
|
||||
checksum = "64e15c1ab1f89faffbf04a634d5e1962e9074f2741eef6d97f3c4e322426d526"
|
||||
dependencies = [
|
||||
"anstyle",
|
||||
"anstyle-parse",
|
||||
@@ -61,36 +61,36 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "anstyle"
|
||||
version = "1.0.7"
|
||||
version = "1.0.8"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "038dfcf04a5feb68e9c60b21c9625a54c2c0616e79b72b0fd87075a056ae1d1b"
|
||||
checksum = "1bec1de6f59aedf83baf9ff929c98f2ad654b97c9510f4e70cf6f661d49fd5b1"
|
||||
|
||||
[[package]]
|
||||
name = "anstyle-parse"
|
||||
version = "0.2.4"
|
||||
version = "0.2.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "c03a11a9034d92058ceb6ee011ce58af4a9bf61491aa7e1e59ecd24bd40d22d4"
|
||||
checksum = "eb47de1e80c2b463c735db5b217a0ddc39d612e7ac9e2e96a5aed1f57616c1cb"
|
||||
dependencies = [
|
||||
"utf8parse",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "anstyle-query"
|
||||
version = "1.1.0"
|
||||
version = "1.1.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ad186efb764318d35165f1758e7dcef3b10628e26d41a44bc5550652e6804391"
|
||||
checksum = "6d36fc52c7f6c869915e99412912f22093507da8d9e942ceaf66fe4b7c14422a"
|
||||
dependencies = [
|
||||
"windows-sys",
|
||||
"windows-sys 0.52.0",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "anstyle-wincon"
|
||||
version = "3.0.3"
|
||||
version = "3.0.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "61a38449feb7068f52bb06c12759005cf459ee52bb4adc1d5a7c4322d716fb19"
|
||||
checksum = "5bf74e1b6e971609db8ca7a9ce79fd5768ab6ae46441c572e46cf596f59e57f8"
|
||||
dependencies = [
|
||||
"anstyle",
|
||||
"windows-sys",
|
||||
"windows-sys 0.52.0",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -113,7 +113,7 @@ checksum = "0ae92a5119aa49cdbcf6b9f893fe4e1d98b04ccbf82ee0584ad948a44a734dea"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"syn 2.0.71",
|
||||
"syn",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -186,9 +186,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "bstr"
|
||||
version = "1.9.1"
|
||||
version = "1.10.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "05efc5cfd9110c8416e471df0e96702d58690178e206e61b7173706673c93706"
|
||||
checksum = "40723b8fb387abc38f4f4a37c09073622e41dd12327033091ef8950659e6dc0c"
|
||||
dependencies = [
|
||||
"memchr",
|
||||
"serde",
|
||||
@@ -208,9 +208,9 @@ checksum = "79296716171880943b8470b5f8d03aa55eb2e645a4874bdbb28adb49162e012c"
|
||||
|
||||
[[package]]
|
||||
name = "bytemuck"
|
||||
version = "1.16.1"
|
||||
version = "1.16.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b236fc92302c97ed75b38da1f4917b5cdda4984745740f153a5d3059e48d725e"
|
||||
checksum = "102087e286b4677862ea56cf8fc58bb2cdfa8725c40ffb80fe3a008eb7f2fc83"
|
||||
|
||||
[[package]]
|
||||
name = "byteorder"
|
||||
@@ -232,9 +232,9 @@ checksum = "37b2a672a2cb129a2e41c10b1224bb368f9f37a2b16b612598138befd7b37eb5"
|
||||
|
||||
[[package]]
|
||||
name = "cc"
|
||||
version = "1.1.5"
|
||||
version = "1.1.8"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "324c74f2155653c90b04f25b2a47a8a631360cb908f92a772695f430c7e31052"
|
||||
checksum = "504bdec147f2cc13c8b57ed9401fd8a147cc66b67ad5cb241394244f2c947549"
|
||||
dependencies = [
|
||||
"jobserver",
|
||||
"libc",
|
||||
@@ -325,9 +325,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "clap"
|
||||
version = "4.5.9"
|
||||
version = "4.5.13"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "64acc1846d54c1fe936a78dc189c34e28d3f5afc348403f28ecf53660b9b8462"
|
||||
checksum = "0fbb260a053428790f3de475e304ff84cdbc4face759ea7a3e64c1edd938a7fc"
|
||||
dependencies = [
|
||||
"clap_builder",
|
||||
"clap_derive",
|
||||
@@ -335,9 +335,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "clap-verbosity-flag"
|
||||
version = "2.2.0"
|
||||
version = "2.2.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "bb9b20c0dd58e4c2e991c8d203bbeb76c11304d1011659686b5b644bc29aa478"
|
||||
checksum = "63d19864d6b68464c59f7162c9914a0b569ddc2926b4a2d71afe62a9738eff53"
|
||||
dependencies = [
|
||||
"clap",
|
||||
"log",
|
||||
@@ -345,9 +345,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "clap_builder"
|
||||
version = "4.5.9"
|
||||
version = "4.5.13"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "6fb8393d67ba2e7bfaf28a23458e4e2b543cc73a99595511eb207fdb8aede942"
|
||||
checksum = "64b17d7ea74e9f833c7dbf2cbe4fb12ff26783eda4782a8975b72f895c9b4d99"
|
||||
dependencies = [
|
||||
"anstream",
|
||||
"anstyle",
|
||||
@@ -357,21 +357,21 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "clap_derive"
|
||||
version = "4.5.8"
|
||||
version = "4.5.13"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "2bac35c6dafb060fd4d275d9a4ffae97917c13a6327903a8be2153cd964f7085"
|
||||
checksum = "501d359d5f3dcaf6ecdeee48833ae73ec6e42723a1e52419c79abf9507eec0a0"
|
||||
dependencies = [
|
||||
"heck",
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"syn 2.0.71",
|
||||
"syn",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "clap_lex"
|
||||
version = "0.7.1"
|
||||
version = "0.7.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "4b82cf0babdbd58558212896d1a4272303a57bdb245c2bf1147185fb45640e70"
|
||||
checksum = "1462739cb27611015575c0c11df5df7601141071f07518d56fcc1be504cbec97"
|
||||
|
||||
[[package]]
|
||||
name = "color_quant"
|
||||
@@ -381,9 +381,9 @@ checksum = "3d7b894f5411737b7867f4827955924d7c254fc9f4d91a6aad6b097804b1018b"
|
||||
|
||||
[[package]]
|
||||
name = "colorchoice"
|
||||
version = "1.0.1"
|
||||
version = "1.0.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0b6a852b24ab71dffc585bcb46eaf7959d175cb865a7152e35b348d1b2960422"
|
||||
checksum = "d3fd119d74b830634cea2a0f58bbd0d54540518a14397557951e79340abc28c0"
|
||||
|
||||
[[package]]
|
||||
name = "core-foundation-sys"
|
||||
@@ -517,9 +517,9 @@ checksum = "60b1af1c220855b6ceac025d3f6ecdd2b7c4894bfe9cd9bda4fbb4bc7c0d4cf0"
|
||||
|
||||
[[package]]
|
||||
name = "env_filter"
|
||||
version = "0.1.0"
|
||||
version = "0.1.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "a009aa4810eb158359dda09d0c87378e4bbb89b5a801f016885a4707ba24f7ea"
|
||||
checksum = "4f2c92ceda6ceec50f43169f9ee8424fe2db276791afde7b2cd8bc084cb376ab"
|
||||
dependencies = [
|
||||
"log",
|
||||
"regex",
|
||||
@@ -527,9 +527,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "env_logger"
|
||||
version = "0.11.3"
|
||||
version = "0.11.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "38b35839ba51819680ba087cd351788c9a3c476841207e0b8cee0b04722343b9"
|
||||
checksum = "e13fa619b91fb2381732789fc5de83b45675e882f66623b7d8cb4f643017018d"
|
||||
dependencies = [
|
||||
"anstream",
|
||||
"anstyle",
|
||||
@@ -596,9 +596,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "flate2"
|
||||
version = "1.0.30"
|
||||
version = "1.0.31"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "5f54427cfd1c7829e2a139fcefea601bf088ebca651d2bf53ebc600eac295dae"
|
||||
checksum = "7f211bbe8e69bbd0cfdea405084f128ae8b4aaa6b0b522fc8f2b009084797920"
|
||||
dependencies = [
|
||||
"crc32fast",
|
||||
"miniz_oxide",
|
||||
@@ -752,12 +752,12 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "image"
|
||||
version = "0.25.1"
|
||||
version = "0.25.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "fd54d660e773627692c524beaad361aca785a4f9f5730ce91f42aabe5bce3d11"
|
||||
checksum = "99314c8a2152b8ddb211f924cdae532d8c5e4c8bb54728e12fff1b0cd5963a10"
|
||||
dependencies = [
|
||||
"bytemuck",
|
||||
"byteorder",
|
||||
"byteorder-lite",
|
||||
"color_quant",
|
||||
"exr",
|
||||
"gif",
|
||||
@@ -791,9 +791,9 @@ checksum = "44feda355f4159a7c757171a77de25daf6411e217b4cabd03bd6650690468126"
|
||||
|
||||
[[package]]
|
||||
name = "indexmap"
|
||||
version = "2.2.6"
|
||||
version = "2.3.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "168fb715dda47215e360912c096649d23d58bf392ac62f73919e831745e40f26"
|
||||
checksum = "de3fc2e30ba82dd1b3911c8de1ffc143c74a914a14e99514d7637e3099df5ea0"
|
||||
dependencies = [
|
||||
"equivalent",
|
||||
"hashbrown",
|
||||
@@ -807,7 +807,7 @@ checksum = "c34819042dc3d3971c46c2190835914dfbe0c3c13f61449b2997f4e9722dfa60"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"syn 2.0.71",
|
||||
"syn",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -818,14 +818,14 @@ checksum = "f23ff5ef2b80d608d61efee834934d862cd92461afc0560dedf493e4c033738b"
|
||||
dependencies = [
|
||||
"hermit-abi",
|
||||
"libc",
|
||||
"windows-sys",
|
||||
"windows-sys 0.52.0",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "is_terminal_polyfill"
|
||||
version = "1.70.0"
|
||||
version = "1.70.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f8478577c03552c21db0e2724ffb8986a5ce7af88107e6be5d2ee6e158c12800"
|
||||
checksum = "7943c866cc5cd64cbc25b2e01621d07fa8eb2a1a23160ee81ce38704e97b8ecf"
|
||||
|
||||
[[package]]
|
||||
name = "itertools"
|
||||
@@ -862,9 +862,9 @@ checksum = "49f1f14873335454500d59611f1cf4a4b0f786f9ac11f4312a78e4cf2566695b"
|
||||
|
||||
[[package]]
|
||||
name = "jobserver"
|
||||
version = "0.1.31"
|
||||
version = "0.1.32"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d2b099aaa34a9751c5bf0878add70444e1ed2dd73f347be99003d4577277de6e"
|
||||
checksum = "48d1dbcbbeb6a7fec7e059840aa538bd62aaccf972c7346c4d9d2059312853d0"
|
||||
dependencies = [
|
||||
"libc",
|
||||
]
|
||||
@@ -921,11 +921,11 @@ checksum = "4ec2a862134d2a7d32d7983ddcdd1c4923530833c9f2ea1a44fc5fa473989058"
|
||||
|
||||
[[package]]
|
||||
name = "libvips"
|
||||
version = "1.4.3"
|
||||
version = "1.7.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9c6574a02b3823ce436bd70d47546428f4546686031f8d2af4c056d23969ace2"
|
||||
checksum = "33890b93365ac05b5e6063d41bc014a1598087db4a0b9f75eddaa7dad0f7fc2a"
|
||||
dependencies = [
|
||||
"num-derive 0.3.3",
|
||||
"num-derive",
|
||||
"num-traits",
|
||||
]
|
||||
|
||||
@@ -967,7 +967,6 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "8ea1f30cedd69f0a2954655f7188c6a834246d2bcf1e315e2ac40c4b24dc9519"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"rayon",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -1036,17 +1035,6 @@ dependencies = [
|
||||
"num-traits",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "num-derive"
|
||||
version = "0.3.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "876a53fff98e03a936a674b29568b0e605f06b29372c2489ff4de23f1949743d"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"syn 1.0.109",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "num-derive"
|
||||
version = "0.4.2"
|
||||
@@ -1055,7 +1043,7 @@ checksum = "ed3955f1a9c7c0c15e092f9c887db08b1fc683305fdf6eb6684f22555355e202"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"syn 2.0.71",
|
||||
"syn",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -1151,7 +1139,7 @@ dependencies = [
|
||||
"pest_meta",
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"syn 2.0.71",
|
||||
"syn",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -1224,9 +1212,12 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "ppv-lite86"
|
||||
version = "0.2.17"
|
||||
version = "0.2.20"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "5b40af805b3121feab8a3c29f04d8ad262fa8e0561883e7653e024ae4479e6de"
|
||||
checksum = "77957b295656769bb8ad2b6a6b09d897d94f05c41b069aede1fcdaa675eaea04"
|
||||
dependencies = [
|
||||
"zerocopy",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "proc-macro2"
|
||||
@@ -1253,7 +1244,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "8021cf59c8ec9c432cfc2526ac6b8aa508ecaf29cd415f271b8406c1b851c3fd"
|
||||
dependencies = [
|
||||
"quote",
|
||||
"syn 2.0.71",
|
||||
"syn",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -1331,7 +1322,7 @@ dependencies = [
|
||||
"maybe-rayon",
|
||||
"new_debug_unreachable",
|
||||
"noop_proc_macro",
|
||||
"num-derive 0.4.2",
|
||||
"num-derive",
|
||||
"num-traits",
|
||||
"once_cell",
|
||||
"paste",
|
||||
@@ -1347,16 +1338,15 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "ravif"
|
||||
version = "0.11.8"
|
||||
version = "0.11.9"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "c6ba61c28ba24c0cf8406e025cb29a742637e3f70776e61c27a8a8b72a042d12"
|
||||
checksum = "5797d09f9bd33604689e87e8380df4951d4912f01b63f71205e2abd4ae25e6b6"
|
||||
dependencies = [
|
||||
"avif-serialize",
|
||||
"imgref",
|
||||
"loop9",
|
||||
"quick-error",
|
||||
"rav1e",
|
||||
"rayon",
|
||||
"rgb",
|
||||
]
|
||||
|
||||
@@ -1382,9 +1372,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "regex"
|
||||
version = "1.10.5"
|
||||
version = "1.10.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b91213439dad192326a0d7c6ee3955910425f441d7038e0d6933b0aec5c4517f"
|
||||
checksum = "4219d74c6b67a3654a9fbebc4b419e22126d13d2f3c4a07ee0cb61ff79a79619"
|
||||
dependencies = [
|
||||
"aho-corasick",
|
||||
"memchr",
|
||||
@@ -1411,9 +1401,9 @@ checksum = "7a66a03ae7c801facd77a29370b4faec201768915ac14a721ba36f20bc9c209b"
|
||||
|
||||
[[package]]
|
||||
name = "resize"
|
||||
version = "0.8.4"
|
||||
version = "0.8.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "c3e29f584c07a8396c5e2eee0bd8d7aec5c8d9e0a3c2333806fd2ec1d2a5b080"
|
||||
checksum = "a84f5827feaf48508b264176bd88e0479695af183738cf0305fadf956c796412"
|
||||
dependencies = [
|
||||
"rgb",
|
||||
]
|
||||
@@ -1434,9 +1424,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rgb"
|
||||
version = "0.8.45"
|
||||
version = "0.8.48"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ade4539f42266ded9e755c605bdddf546242b2c961b03b06a7375260788a0523"
|
||||
checksum = "0f86ae463694029097b846d8f99fd5536740602ae00022c0c50c5600720b2f71"
|
||||
dependencies = [
|
||||
"bytemuck",
|
||||
]
|
||||
@@ -1479,25 +1469,26 @@ checksum = "e0cd7e117be63d3c3678776753929474f3b04a43a080c744d6b0ae2a8c28e222"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"syn 2.0.71",
|
||||
"syn",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "serde_json"
|
||||
version = "1.0.120"
|
||||
version = "1.0.122"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "4e0d21c9a8cae1235ad58a00c11cb40d4b1e5c784f1ef2c537876ed6ffd8b7c5"
|
||||
checksum = "784b6203951c57ff748476b126ccb5e8e2959a5c19e5c617ab1956be3dbc68da"
|
||||
dependencies = [
|
||||
"itoa",
|
||||
"memchr",
|
||||
"ryu",
|
||||
"serde",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "serde_spanned"
|
||||
version = "0.6.6"
|
||||
version = "0.6.7"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "79e674e01f999af37c49f70a6ede167a8a60b2503e56c5599532a65baa5969a0"
|
||||
checksum = "eb5b1b31579f3811bf615c144393417496f152e12ac8b7663bf664f4a815306d"
|
||||
dependencies = [
|
||||
"serde",
|
||||
]
|
||||
@@ -1567,20 +1558,9 @@ checksum = "7da8b5736845d9f2fcb837ea5d9e2628564b3b043a70948a3f0b778838c5fb4f"
|
||||
|
||||
[[package]]
|
||||
name = "syn"
|
||||
version = "1.0.109"
|
||||
version = "2.0.72"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "72b64191b275b66ffe2469e8af2c1cfe3bafa67b529ead792a6d0160888b4237"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"unicode-ident",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "syn"
|
||||
version = "2.0.71"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b146dcf730474b4bcd16c311627b31ede9ab149045db4d6088b3becaea046462"
|
||||
checksum = "dc4b9b9bf2add8093d3f2c0204471e951b2285580335de42f9d2534f3ae7a8af"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
@@ -1602,9 +1582,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "target-lexicon"
|
||||
version = "0.12.15"
|
||||
version = "0.12.16"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "4873307b7c257eddcb50c9bedf158eb669578359fb28428bef438fec8e6ba7c2"
|
||||
checksum = "61c41af27dd6d1e27b1b16b489db798443478cef1f06a660c96db617ba5de3b1"
|
||||
|
||||
[[package]]
|
||||
name = "tera"
|
||||
@@ -1653,7 +1633,7 @@ checksum = "a4558b58466b9ad7ca0f102865eccc95938dca1a74a856f2b57b6629050da261"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"syn 2.0.71",
|
||||
"syn",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -1679,9 +1659,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "toml"
|
||||
version = "0.8.15"
|
||||
version = "0.8.19"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ac2caab0bf757388c6c0ae23b3293fdb463fee59434529014f85e3263b995c28"
|
||||
checksum = "a1ed1f98e3fdc28d6d910e6737ae6ab1a93bf1985935a1193e68f93eeb68d24e"
|
||||
dependencies = [
|
||||
"serde",
|
||||
"serde_spanned",
|
||||
@@ -1691,18 +1671,18 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "toml_datetime"
|
||||
version = "0.6.6"
|
||||
version = "0.6.8"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "4badfd56924ae69bcc9039335b2e017639ce3f9b001c393c1b2d1ef846ce2cbf"
|
||||
checksum = "0dd7358ecb8fc2f8d014bf86f6f638ce72ba252a2c3a2572f2a795f1d23efb41"
|
||||
dependencies = [
|
||||
"serde",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "toml_edit"
|
||||
version = "0.22.16"
|
||||
version = "0.22.20"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "278f3d518e152219c994ce877758516bca5e118eaed6996192a774fb9fbf0788"
|
||||
checksum = "583c44c02ad26b0c3f3066fe629275e50627026c51ac2e595cca4c230ce1ce1d"
|
||||
dependencies = [
|
||||
"indexmap",
|
||||
"serde",
|
||||
@@ -1804,9 +1784,9 @@ checksum = "852e951cb7832cb45cb1169900d19760cfa39b82bc0ea9c0e5a14ae88411c98b"
|
||||
|
||||
[[package]]
|
||||
name = "version_check"
|
||||
version = "0.9.4"
|
||||
version = "0.9.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "49874b5167b65d7193b8aba1567f5c7d93d001cafc34600cee003eda787e483f"
|
||||
checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a"
|
||||
|
||||
[[package]]
|
||||
name = "walkdir"
|
||||
@@ -1845,7 +1825,7 @@ dependencies = [
|
||||
"once_cell",
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"syn 2.0.71",
|
||||
"syn",
|
||||
"wasm-bindgen-shared",
|
||||
]
|
||||
|
||||
@@ -1867,7 +1847,7 @@ checksum = "e94f17b526d0a461a191c78ea52bbce64071ed5c04c9ffe424dcb38f74171bb7"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"syn 2.0.71",
|
||||
"syn",
|
||||
"wasm-bindgen-backend",
|
||||
"wasm-bindgen-shared",
|
||||
]
|
||||
@@ -1886,11 +1866,11 @@ checksum = "53a85b86a771b1c87058196170769dd264f66c0782acf1ae6cc51bfd64b39082"
|
||||
|
||||
[[package]]
|
||||
name = "winapi-util"
|
||||
version = "0.1.8"
|
||||
version = "0.1.9"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "4d4cc384e1e73b93bafa6fb4f1df8c41695c8a91cf9c4c64358067d15a7b6c6b"
|
||||
checksum = "cf221c93e13a30d793f7645a0e7762c55d169dbb0a49671918a2319d289b10bb"
|
||||
dependencies = [
|
||||
"windows-sys",
|
||||
"windows-sys 0.59.0",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -1911,6 +1891,15 @@ dependencies = [
|
||||
"windows-targets",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "windows-sys"
|
||||
version = "0.59.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "1e38bc4d79ed67fd075bcc251a1c39b32a1776bbe92e5bef1f0bf1f8c531853b"
|
||||
dependencies = [
|
||||
"windows-targets",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "windows-targets"
|
||||
version = "0.52.6"
|
||||
@@ -1977,13 +1966,34 @@ checksum = "589f6da84c646204747d1270a2a5661ea66ed1cced2631d546fdfb155959f9ec"
|
||||
|
||||
[[package]]
|
||||
name = "winnow"
|
||||
version = "0.6.13"
|
||||
version = "0.6.18"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "59b5e5f6c299a3c7890b876a2a587f3115162487e704907d9b6cd29473052ba1"
|
||||
checksum = "68a9bda4691f099d435ad181000724da8e5899daa10713c2d432552b9ccd3a6f"
|
||||
dependencies = [
|
||||
"memchr",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "zerocopy"
|
||||
version = "0.7.35"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "1b9b4fd18abc82b8136838da5d50bae7bdea537c574d8dc1a34ed098d6c166f0"
|
||||
dependencies = [
|
||||
"byteorder",
|
||||
"zerocopy-derive",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "zerocopy-derive"
|
||||
version = "0.7.35"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "fa4f8080344d4671fb4e831a13ad1e68092748387dfc4f55e356242fae12ce3e"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"syn",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "zune-core"
|
||||
version = "0.4.12"
|
||||
@@ -2001,9 +2011,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "zune-jpeg"
|
||||
version = "0.4.11"
|
||||
version = "0.4.13"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ec866b44a2a1fd6133d363f073ca1b179f438f99e7e5bfb1e33f7181facfe448"
|
||||
checksum = "16099418600b4d8f028622f73ff6e3deaabdff330fb9a2a131dea781ee8b0768"
|
||||
dependencies = [
|
||||
"zune-core",
|
||||
]
|
||||
|
||||
+4
-4
@@ -25,7 +25,7 @@ num-traits = "0.2.19"
|
||||
thiserror = "1.0"
|
||||
document-features = "0.2.10"
|
||||
# Optional dependencies
|
||||
image = { version = "0.25.1", optional = true, default-features = false }
|
||||
image = { version = "0.25.2", optional = true, default-features = false }
|
||||
bytemuck = { version = "1.16", optional = true }
|
||||
|
||||
[features]
|
||||
@@ -40,8 +40,8 @@ only_u8x4 = ["testing/only_u8x4"] # This can be used to experiment with the cra
|
||||
|
||||
[dev-dependencies]
|
||||
fast_image_resize = { path = ".", features = ["for_testing"] }
|
||||
resize = { version = "0.8.4", default-features = false, features = ["std"] }
|
||||
rgb = "0.8.45"
|
||||
resize = { version = "0.8.5", default-features = false, features = ["std"] }
|
||||
rgb = "0.8.48"
|
||||
png = "0.17.13"
|
||||
serde = { version = "1.0", features = ["serde_derive"] }
|
||||
serde_json = "1.0"
|
||||
@@ -57,7 +57,7 @@ nix = { version = "0.29.0", default-features = false, features = ["sched"] }
|
||||
|
||||
|
||||
[target.'cfg(all(not(target_arch = "wasm32"), not(target_os = "windows")))'.dev-dependencies]
|
||||
libvips = "=1.4.3"
|
||||
libvips = "1.7"
|
||||
|
||||
|
||||
[[bench]]
|
||||
|
||||
@@ -137,8 +137,7 @@ Otherwise, you have to convert such images into supported by the crate image typ
|
||||
use std::io::BufWriter;
|
||||
|
||||
use image::codecs::png::PngEncoder;
|
||||
use image::io::Reader as ImageReader;
|
||||
use image::{ExtendedColorType, ImageEncoder};
|
||||
use image::{ExtendedColorType, ImageEncoder, ImageReader};
|
||||
|
||||
use fast_image_resize::{IntoImageView, Resizer};
|
||||
use fast_image_resize::images::Image;
|
||||
@@ -181,8 +180,7 @@ fn main() {
|
||||
|
||||
```rust
|
||||
use image::codecs::png::PngEncoder;
|
||||
use image::io::Reader as ImageReader;
|
||||
use image::{ColorType, GenericImageView};
|
||||
use image::{ColorType, ImageReader, GenericImageView};
|
||||
|
||||
use fast_image_resize::{IntoImageView, Resizer, ResizeOptions};
|
||||
use fast_image_resize::images::Image;
|
||||
|
||||
+1
-2
@@ -3,8 +3,7 @@ use std::path::PathBuf;
|
||||
|
||||
use anyhow::{anyhow, Context, Result};
|
||||
use clap::Parser;
|
||||
use image::io::Reader as ImageReader;
|
||||
use image::ColorType;
|
||||
use image::{ColorType, ImageReader};
|
||||
use log::debug;
|
||||
use once_cell::sync::Lazy;
|
||||
|
||||
|
||||
+33
-15
@@ -284,13 +284,22 @@ impl PixelComponentMapper {
|
||||
};
|
||||
}
|
||||
|
||||
match_img!(
|
||||
tables,
|
||||
(PT::U8, U8, PT::U16, U16),
|
||||
(PT::U8x2, U8x2, PT::U16x2, U16x2),
|
||||
(PT::U8x3, U8x3, PT::U16x3, U16x3),
|
||||
(PT::U8x4, U8x4, PT::U16x4, U16x4),
|
||||
)
|
||||
#[cfg(not(feature = "only_u8x4"))]
|
||||
{
|
||||
match_img!(
|
||||
tables,
|
||||
(PT::U8, U8, PT::U16, U16),
|
||||
(PT::U8x2, U8x2, PT::U16x2, U16x2),
|
||||
(PT::U8x3, U8x3, PT::U16x3, U16x3),
|
||||
(PT::U8x4, U8x4, PT::U16x4, U16x4),
|
||||
)
|
||||
}
|
||||
|
||||
#[cfg(feature = "only_u8x4")]
|
||||
match (src_pixel_type, dst_pixel_type) {
|
||||
(PT::U8x4, PT::U8x4) => tables.u8_u8.map_image::<U8x4, U8x4>(src_image, dst_image),
|
||||
_ => return Err(MappingError::UnsupportedCombinationOfImageTypes),
|
||||
}
|
||||
}
|
||||
|
||||
fn map_inplace(
|
||||
@@ -316,14 +325,23 @@ impl PixelComponentMapper {
|
||||
};
|
||||
}
|
||||
|
||||
match_img!(
|
||||
tables,
|
||||
image,
|
||||
(PT::U8, U8, PT::U16, U16),
|
||||
(PT::U8x2, U8x2, PT::U16x2, U16x2),
|
||||
(PT::U8x3, U8x3, PT::U16x3, U16x3),
|
||||
(PT::U8x4, U8x4, PT::U16x4, U16x4),
|
||||
)
|
||||
#[cfg(not(feature = "only_u8x4"))]
|
||||
{
|
||||
match_img!(
|
||||
tables,
|
||||
image,
|
||||
(PT::U8, U8, PT::U16, U16),
|
||||
(PT::U8x2, U8x2, PT::U16x2, U16x2),
|
||||
(PT::U8x3, U8x3, PT::U16x3, U16x3),
|
||||
(PT::U8x4, U8x4, PT::U16x4, U16x4),
|
||||
)
|
||||
}
|
||||
|
||||
#[cfg(feature = "only_u8x4")]
|
||||
match pixel_type {
|
||||
PT::U8x4 => tables.u8_u8.map_image_inplace::<U8x4>(image),
|
||||
_ => return Err(MappingError::UnsupportedCombinationOfImageTypes),
|
||||
}
|
||||
}
|
||||
|
||||
/// Mapping in the forward direction of pixel's components of source image
|
||||
|
||||
+20
-4
@@ -136,12 +136,28 @@ pub(crate) fn precompute_coefficients(
|
||||
// (x + 0.5) - in_center => x - (in_center - 0.5) => x - center
|
||||
let center = in_center - 0.5;
|
||||
|
||||
let mut bound_start = x_min;
|
||||
let mut bound_end = x_max;
|
||||
|
||||
// Calculate the weight of each input pixel from the given x-range.
|
||||
for x in x_min..x_max {
|
||||
let w: f64 = filter((x as f64 - center) * recip_filter_scale);
|
||||
coeffs.push(w);
|
||||
ww += w;
|
||||
if x == bound_start && w == 0. {
|
||||
// Don't use zero coefficients at the start of bound;
|
||||
bound_start += 1;
|
||||
} else {
|
||||
coeffs.push(w);
|
||||
ww += w;
|
||||
}
|
||||
}
|
||||
for &c in coeffs.iter().rev() {
|
||||
if bound_end <= bound_start || c != 0. {
|
||||
break;
|
||||
}
|
||||
// Don't use zero coefficients at the end of bound;
|
||||
bound_end -= 1;
|
||||
}
|
||||
|
||||
if ww != 0.0 {
|
||||
// Normalise values of weights.
|
||||
// The sum of weights must be equal to 1.0.
|
||||
@@ -150,8 +166,8 @@ pub(crate) fn precompute_coefficients(
|
||||
// Remaining values should stay empty if they are used despite x_max.
|
||||
coeffs.resize(cur_index + window_size, 0.);
|
||||
bounds.push(Bound {
|
||||
start: x_min,
|
||||
size: x_max - x_min,
|
||||
start: bound_start,
|
||||
size: bound_end - bound_start,
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
@@ -1,4 +1,3 @@
|
||||
use super::Bound;
|
||||
use crate::convolution::Coefficients;
|
||||
|
||||
// This code is based on C-implementation from Pillow-SIMD package for Python
|
||||
@@ -31,16 +30,21 @@ const MAX_COEFFS_PRECISION: u8 = 16 - 1;
|
||||
|
||||
/// Converts `Vec<f64>` into `Vec<i16>`.
|
||||
pub(crate) struct Normalizer16 {
|
||||
values: Vec<i16>,
|
||||
precision: u8,
|
||||
window_size: usize,
|
||||
bounds: Vec<Bound>,
|
||||
chunks: Vec<CoefficientsI16Chunk>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
pub(crate) struct CoefficientsI16Chunk<'a> {
|
||||
#[derive(Debug, Clone)]
|
||||
pub(crate) struct CoefficientsI16Chunk {
|
||||
pub start: u32,
|
||||
pub values: &'a [i16],
|
||||
values: Vec<i16>,
|
||||
}
|
||||
|
||||
impl CoefficientsI16Chunk {
|
||||
#[inline(always)]
|
||||
pub fn values(&self) -> &[i16] {
|
||||
&self.values
|
||||
}
|
||||
}
|
||||
|
||||
impl Normalizer16 {
|
||||
@@ -64,34 +68,29 @@ impl Normalizer16 {
|
||||
}
|
||||
debug_assert!(precision >= 4); // required for some SIMD optimisations
|
||||
|
||||
let mut values_i16 = Vec::with_capacity(coefficients.values.len());
|
||||
let mut chunks = Vec::with_capacity(coefficients.bounds.len());
|
||||
if coefficients.window_size > 0 {
|
||||
let scale = (1 << precision) as f64;
|
||||
let coef_chunks = coefficients.values.chunks_exact(coefficients.window_size);
|
||||
for (chunk, bound) in coef_chunks.zip(&coefficients.bounds) {
|
||||
let chunk_i16: Vec<i16> = chunk
|
||||
.iter()
|
||||
.take(bound.size as usize)
|
||||
.map(|&v| (v * scale).round() as i16)
|
||||
.collect();
|
||||
chunks.push(CoefficientsI16Chunk {
|
||||
start: bound.start,
|
||||
values: chunk_i16,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
let scale = (1 << precision) as f64;
|
||||
for src in coefficients.values.iter().copied() {
|
||||
values_i16.push((src * scale).round() as i16);
|
||||
}
|
||||
Self {
|
||||
values: values_i16,
|
||||
precision,
|
||||
window_size: coefficients.window_size,
|
||||
bounds: coefficients.bounds,
|
||||
}
|
||||
Self { precision, chunks }
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn normalized_chunks(&self) -> Vec<CoefficientsI16Chunk> {
|
||||
let mut cooefs = self.values.as_slice();
|
||||
let mut res = Vec::with_capacity(self.bounds.len());
|
||||
for bound in self.bounds.iter() {
|
||||
let (left, right) = cooefs.split_at(self.window_size);
|
||||
cooefs = right;
|
||||
let size = bound.size as usize;
|
||||
res.push(CoefficientsI16Chunk {
|
||||
start: bound.start,
|
||||
values: &left[0..size],
|
||||
});
|
||||
}
|
||||
res
|
||||
#[inline(always)]
|
||||
pub fn coefficients(&self) -> &[CoefficientsI16Chunk] {
|
||||
&self.chunks
|
||||
}
|
||||
|
||||
#[inline]
|
||||
@@ -112,7 +111,7 @@ impl Normalizer16 {
|
||||
}
|
||||
}
|
||||
|
||||
// 16 bits for result. Filter can have negative areas.
|
||||
// 16 bits for a result. Filter can have negative areas.
|
||||
// In one cases the sum of the coefficients will be negative,
|
||||
// in the other it will be more than 1.0. That is why we need
|
||||
// two extra bits for overflow and i64 type.
|
||||
@@ -120,18 +119,23 @@ const PRECISION16_BITS: u8 = 64 - 16 - 2;
|
||||
// We use i32 type to store coefficients.
|
||||
const MAX_COEFFS_PRECISION16: u8 = 32 - 1;
|
||||
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
pub(crate) struct CoefficientsI32Chunk<'a> {
|
||||
#[derive(Debug, Clone)]
|
||||
pub(crate) struct CoefficientsI32Chunk {
|
||||
pub start: u32,
|
||||
pub values: &'a [i32],
|
||||
pub values: Vec<i32>,
|
||||
}
|
||||
|
||||
impl CoefficientsI32Chunk {
|
||||
#[inline(always)]
|
||||
pub fn values(&self) -> &[i32] {
|
||||
&self.values
|
||||
}
|
||||
}
|
||||
|
||||
/// Converts `Vec<f64>` into `Vec<i32>`.
|
||||
pub(crate) struct Normalizer32 {
|
||||
values: Vec<i32>,
|
||||
precision: u8,
|
||||
window_size: usize,
|
||||
bounds: Vec<Bound>,
|
||||
chunks: Vec<CoefficientsI32Chunk>,
|
||||
}
|
||||
|
||||
impl Normalizer32 {
|
||||
@@ -155,34 +159,29 @@ impl Normalizer32 {
|
||||
}
|
||||
debug_assert!(precision >= 4); // required for some SIMD optimisations
|
||||
|
||||
let mut values_i32 = Vec::with_capacity(coefficients.values.len());
|
||||
let mut chunks = Vec::with_capacity(coefficients.bounds.len());
|
||||
if coefficients.window_size > 0 {
|
||||
let scale = (1i64 << precision) as f64;
|
||||
let coef_chunks = coefficients.values.chunks_exact(coefficients.window_size);
|
||||
for (chunk, bound) in coef_chunks.zip(&coefficients.bounds) {
|
||||
let chunk_i32: Vec<i32> = chunk
|
||||
.iter()
|
||||
.take(bound.size as usize)
|
||||
.map(|&v| (v * scale).round() as i32)
|
||||
.collect();
|
||||
chunks.push(CoefficientsI32Chunk {
|
||||
start: bound.start,
|
||||
values: chunk_i32,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
let scale = (1i64 << precision) as f64;
|
||||
for src in coefficients.values.iter().copied() {
|
||||
values_i32.push((src * scale).round() as i32);
|
||||
}
|
||||
Self {
|
||||
values: values_i32,
|
||||
precision,
|
||||
window_size: coefficients.window_size,
|
||||
bounds: coefficients.bounds,
|
||||
}
|
||||
Self { precision, chunks }
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn normalized_chunks(&self) -> Vec<CoefficientsI32Chunk> {
|
||||
let mut cooefs = self.values.as_slice();
|
||||
let mut res = Vec::with_capacity(self.bounds.len());
|
||||
for bound in self.bounds.iter() {
|
||||
let (left, right) = cooefs.split_at(self.window_size);
|
||||
cooefs = right;
|
||||
let size = bound.size as usize;
|
||||
res.push(CoefficientsI32Chunk {
|
||||
start: bound.start,
|
||||
values: &left[0..size],
|
||||
});
|
||||
}
|
||||
res
|
||||
#[inline(always)]
|
||||
pub fn coefficients(&self) -> &[CoefficientsI32Chunk] {
|
||||
&self.chunks
|
||||
}
|
||||
|
||||
#[inline]
|
||||
|
||||
@@ -12,14 +12,13 @@ pub(crate) fn horiz_convolution(
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -28,7 +27,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_one_row(src_row, dst_row, &normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -42,12 +41,12 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U16]; 4],
|
||||
dst_rows: [&mut [U16]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 4];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|L0 | |L1 | |L2 | |L3 | |L4 | |L5 | |L6 | |L7 |
|
||||
@@ -91,7 +90,7 @@ unsafe fn horiz_convolution_four_rows(
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut ll_sum = [_mm256_set1_epi64x(0); 2];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
coeffs = coeffs_by_8.remainder();
|
||||
@@ -201,12 +200,12 @@ unsafe fn horiz_convolution_four_rows(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16],
|
||||
dst_row: &mut [U16],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 4];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|L0 | |L1 | |L2 | |L3 | |L4 | |L5 | |L6 | |L7 |
|
||||
@@ -249,7 +248,7 @@ unsafe fn horiz_convolution_one_row(
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut ll_sum = _mm256_set1_epi64x(0);
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_16 = coeffs.chunks_exact(16);
|
||||
coeffs = coeffs_by_16.remainder();
|
||||
|
||||
@@ -11,17 +11,17 @@ pub(crate) fn horiz_convolution(
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
let initial = 1i64 << (precision - 1);
|
||||
|
||||
let src_rows = src_view.iter_rows(offset);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
for (dst_row, src_row) in dst_rows.zip(src_rows) {
|
||||
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
|
||||
for (coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
|
||||
let first_x_src = coeffs_chunk.start as usize;
|
||||
let mut ss = initial;
|
||||
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
|
||||
for (&k, src_pixel) in coeffs_chunk.values.iter().zip(src_pixels) {
|
||||
for (&k, src_pixel) in coeffs_chunk.values().iter().zip(src_pixels) {
|
||||
ss += src_pixel.0 as i64 * (k as i64);
|
||||
}
|
||||
dst_pixel.0 = normalizer.clip(ss);
|
||||
|
||||
@@ -12,14 +12,13 @@ pub(crate) fn horiz_convolution(
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -28,7 +27,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_one_row(src_row, dst_row, &normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -42,12 +41,12 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U16]; 4],
|
||||
dst_rows: [&mut [U16]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 2];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|L0 | |L1 | |L2 | |L3 | |L4 | |L5 | |L6 | |L7 |
|
||||
@@ -77,7 +76,7 @@ unsafe fn horiz_convolution_four_rows(
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut ll_sum = [_mm_set1_epi64x(0); 4];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
coeffs = coeffs_by_8.remainder();
|
||||
@@ -169,12 +168,12 @@ unsafe fn horiz_convolution_four_rows(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16],
|
||||
dst_row: &mut [U16],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 2];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|L0 | |L1 | |L2 | |L3 | |L4 | |L5 | |L6 | |L7 |
|
||||
@@ -203,7 +202,7 @@ unsafe fn horiz_convolution_one_row(
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut ll_sum = _mm_set1_epi64x(0);
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
coeffs = coeffs_by_8.remainder();
|
||||
|
||||
@@ -12,14 +12,13 @@ pub(crate) fn horiz_convolution(
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -28,7 +27,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_one_row(src_row, dst_row, &normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -42,12 +41,12 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U16x2]; 4],
|
||||
dst_rows: [&mut [U16x2]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 4];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|L0 A0 | |L1 A1 | |L2 A2 | |L3 A3 |
|
||||
@@ -91,7 +90,7 @@ unsafe fn horiz_convolution_four_rows(
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut ll_sum = [_mm256_set1_epi64x(half_error); 2];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
coeffs = coeffs_by_4.remainder();
|
||||
@@ -178,12 +177,12 @@ unsafe fn horiz_convolution_four_rows(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16x2],
|
||||
dst_row: &mut [U16x2],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 4];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|L0 A0 | |L1 A1 | |L2 A2 | |L3 A3 |
|
||||
@@ -226,7 +225,7 @@ unsafe fn horiz_convolution_one_row(
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut ll_sum = _mm256_setzero_si256();
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
coeffs = coeffs_by_8.remainder();
|
||||
|
||||
@@ -11,17 +11,17 @@ pub(crate) fn horiz_convolution(
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
let initial: i64 = 1 << (precision - 1);
|
||||
|
||||
let src_rows = src_view.iter_rows(offset);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
for (dst_row, src_row) in dst_rows.zip(src_rows) {
|
||||
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
|
||||
for (coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
|
||||
let first_x_src = coeffs_chunk.start as usize;
|
||||
let mut ss = [initial; 2];
|
||||
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
|
||||
for (&k, src_pixel) in coeffs_chunk.values.iter().zip(src_pixels) {
|
||||
for (&k, src_pixel) in coeffs_chunk.values().iter().zip(src_pixels) {
|
||||
for (i, s) in ss.iter_mut().enumerate() {
|
||||
*s += src_pixel.0[i] as i64 * (k as i64);
|
||||
}
|
||||
|
||||
@@ -12,14 +12,13 @@ pub(crate) fn horiz_convolution(
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -28,7 +27,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_one_row(src_row, dst_row, &normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -43,12 +42,12 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U16x2]; 4],
|
||||
dst_rows: [&mut [U16x2]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 2];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|L0 A0 | |L1 A1 | |L2 A2 | |L3 A3 |
|
||||
@@ -78,7 +77,7 @@ unsafe fn horiz_convolution_four_rows(
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut ll_sum = [_mm_set1_epi64x(half_error); 4];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
coeffs = coeffs_by_4.remainder();
|
||||
@@ -159,12 +158,12 @@ unsafe fn horiz_convolution_four_rows(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16x2],
|
||||
dst_row: &mut [U16x2],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 2];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|L0 A0 | |L1 A1 | |L2 A2 | |L3 A3 |
|
||||
@@ -193,7 +192,7 @@ unsafe fn horiz_convolution_one_row(
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut ll_sum = _mm_set1_epi64x(half_error);
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
coeffs = coeffs_by_4.remainder();
|
||||
|
||||
@@ -12,14 +12,13 @@ pub(crate) fn horiz_convolution(
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -28,7 +27,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_one_row(src_row, dst_row, &normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -42,7 +41,7 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U16x3]; 4],
|
||||
dst_rows: [&mut [U16x3]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
@@ -50,6 +49,7 @@ unsafe fn horiz_convolution_four_rows(
|
||||
let mut rg_buf = [0i64; 4];
|
||||
let mut rg_bb_buf = [0i64; 4];
|
||||
let mut bbb_buf = [0i64; 4];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|R G B | |R G B | |R G | - |B | |R G B | |R G B | |R |
|
||||
@@ -101,7 +101,7 @@ unsafe fn horiz_convolution_four_rows(
|
||||
let mut rg_bb_sum = [_mm256_set1_epi8(0); 4];
|
||||
let mut bbb_sum = [_mm256_set1_epi8(0); 4];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
let end_x = x + coeffs.len();
|
||||
|
||||
if width - end_x >= 1 {
|
||||
@@ -177,7 +177,6 @@ unsafe fn horiz_convolution_four_rows(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16x3],
|
||||
dst_row: &mut [U16x3],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
@@ -185,6 +184,7 @@ unsafe fn horiz_convolution_one_row(
|
||||
let mut rg_buf = [0i64; 4];
|
||||
let mut rg_bb_buf = [0i64; 4];
|
||||
let mut bbb_buf = [0i64; 4];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|R G B | |R G B | |R G | - |B | |R G B | |R G B | |R |
|
||||
@@ -232,13 +232,13 @@ unsafe fn horiz_convolution_one_row(
|
||||
|
||||
let width = src_row.len();
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut rg_sum = zero_i64x4;
|
||||
let mut rg_bb_sum = zero_i64x4;
|
||||
let mut bbb_sum = zero_i64x4;
|
||||
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
let end_x = x + coeffs.len();
|
||||
|
||||
if width - end_x >= 1 {
|
||||
|
||||
@@ -11,17 +11,17 @@ pub(crate) fn horiz_convolution(
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
let initial = 1i64 << (precision - 1);
|
||||
|
||||
let src_rows = src_view.iter_rows(offset);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
for (dst_row, src_row) in dst_rows.zip(src_rows) {
|
||||
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
|
||||
for (coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
|
||||
let first_x_src = coeffs_chunk.start as usize;
|
||||
let mut ss = [initial; 3];
|
||||
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
|
||||
for (&k, src_pixel) in coeffs_chunk.values.iter().zip(src_pixels) {
|
||||
for (&k, src_pixel) in coeffs_chunk.values().iter().zip(src_pixels) {
|
||||
for (s, c) in ss.iter_mut().zip(src_pixel.0) {
|
||||
*s += c as i64 * (k as i64);
|
||||
}
|
||||
|
||||
@@ -12,14 +12,13 @@ pub(crate) fn horiz_convolution(
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -28,7 +27,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_one_row(src_row, dst_row, &normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -43,13 +42,13 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U16x3]; 4],
|
||||
dst_rows: [&mut [U16x3]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut rg_buf = [0i64; 2];
|
||||
let mut bb_buf = [0i64; 2];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|R G B | |R G B | |R G |
|
||||
@@ -77,7 +76,7 @@ unsafe fn horiz_convolution_four_rows(
|
||||
let mut rg_sum = [_mm_set1_epi8(0); 4];
|
||||
let mut bb_sum = [_mm_set1_epi8(0); 4];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
let end_x = x + coeffs.len();
|
||||
|
||||
if width - end_x >= 1 {
|
||||
@@ -138,12 +137,12 @@ unsafe fn horiz_convolution_four_rows(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16x3],
|
||||
dst_row: &mut [U16x3],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let rg_initial = _mm_set1_epi64x(1 << (precision - 1));
|
||||
let bb_initial = _mm_set1_epi64x(1 << (precision - 2));
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|R G B | |R G B | |R G |
|
||||
@@ -168,13 +167,13 @@ unsafe fn horiz_convolution_one_row(
|
||||
|
||||
let width = src_row.len();
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
|
||||
let mut rg_sum = rg_initial;
|
||||
let mut bb_sum = bb_initial;
|
||||
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
let end_x = x + coeffs.len();
|
||||
|
||||
if width - end_x >= 1 {
|
||||
|
||||
@@ -12,14 +12,13 @@ pub(crate) fn horiz_convolution(
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -28,7 +27,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_one_row(src_row, dst_row, &normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -43,13 +42,13 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U16x4]; 4],
|
||||
dst_rows: [&mut [U16x4]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut rg_buf = [0i64; 4];
|
||||
let mut ba_buf = [0i64; 4];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|R0 G0 B0 A0 | |R1 G1 B1 A1 |
|
||||
@@ -93,7 +92,7 @@ unsafe fn horiz_convolution_four_rows(
|
||||
let mut rg_sum = [_mm256_set1_epi64x(half_error); 2];
|
||||
let mut ba_sum = [_mm256_set1_epi64x(half_error); 2];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_2 = coeffs.chunks_exact(2);
|
||||
coeffs = coeffs_by_2.remainder();
|
||||
@@ -179,13 +178,13 @@ unsafe fn horiz_convolution_four_rows(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16x4],
|
||||
dst_row: &mut [U16x4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut rg_buf = [0i64; 4];
|
||||
let mut ba_buf = [0i64; 4];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|R0 G0 B0 A0 | |R1 G1 B1 A1 |
|
||||
@@ -227,7 +226,7 @@ unsafe fn horiz_convolution_one_row(
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
let mut rg_sum = _mm256_setzero_si256();
|
||||
let mut ba_sum = _mm256_setzero_si256();
|
||||
|
||||
|
||||
@@ -11,17 +11,17 @@ pub(crate) fn horiz_convolution(
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
let initial: i64 = 1 << (precision - 1);
|
||||
|
||||
let src_rows = src_view.iter_rows(offset);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
for (dst_row, src_row) in dst_rows.zip(src_rows) {
|
||||
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
|
||||
for (coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
|
||||
let first_x_src = coeffs_chunk.start as usize;
|
||||
let mut ss = [initial; 4];
|
||||
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
|
||||
for (&k, src_pixel) in coeffs_chunk.values.iter().zip(src_pixels) {
|
||||
for (&k, src_pixel) in coeffs_chunk.values().iter().zip(src_pixels) {
|
||||
for (i, s) in ss.iter_mut().enumerate() {
|
||||
*s += src_pixel.0[i] as i64 * (k as i64);
|
||||
}
|
||||
|
||||
@@ -12,14 +12,13 @@ pub(crate) fn horiz_convolution(
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -28,7 +27,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_one_row(src_row, dst_row, &normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -43,13 +42,13 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U16x4]; 4],
|
||||
dst_rows: [&mut [U16x4]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut rg_buf = [0i64; 2];
|
||||
let mut ba_buf = [0i64; 2];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|R0 G0 B0 A0 | |R1 G1 B1 A1 |
|
||||
@@ -80,7 +79,7 @@ unsafe fn horiz_convolution_four_rows(
|
||||
let mut rg_sum = [_mm_set1_epi64x(half_error); 4];
|
||||
let mut ba_sum = [_mm_set1_epi64x(half_error); 4];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_2 = coeffs.chunks_exact(2);
|
||||
coeffs = coeffs_by_2.remainder();
|
||||
@@ -143,13 +142,13 @@ unsafe fn horiz_convolution_four_rows(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16x4],
|
||||
dst_row: &mut [U16x4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut rg_buf = [0i64; 2];
|
||||
let mut ba_buf = [0i64; 2];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|R0 G0 B0 A0 | |R1 G1 B1 A1 |
|
||||
@@ -177,7 +176,7 @@ unsafe fn horiz_convolution_one_row(
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
let mut rg_sum = _mm_set1_epi64x(half_error);
|
||||
let mut ba_sum = _mm_set1_epi64x(half_error);
|
||||
|
||||
|
||||
@@ -12,14 +12,13 @@ pub(crate) fn horiz_convolution(
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -28,7 +27,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_one_row(src_row, dst_row, &normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -44,15 +43,15 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U8]; 4],
|
||||
dst_rows: [&mut [U8]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
let zero = _mm_setzero_si128();
|
||||
// 8 components will be added, use only 1/8 of the error
|
||||
let initial = _mm256_set1_epi32(1 << (normalizer.precision() - 4));
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let coeffs = coeffs_chunk.values;
|
||||
let coeffs = coeffs_chunk.values();
|
||||
let mut x = coeffs_chunk.start as usize;
|
||||
let mut result_i32x8x4 = [initial, initial, initial, initial];
|
||||
|
||||
@@ -113,15 +112,15 @@ unsafe fn horiz_convolution_four_rows(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U8],
|
||||
dst_row: &mut [U8],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
let zero = _mm_setzero_si128();
|
||||
// 8 components will be added, use only 1/8 of the error
|
||||
let initial = _mm256_set1_epi32(1 << (normalizer.precision() - 4));
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let coeffs = coeffs_chunk.values;
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let coeffs = coeffs_chunk.values();
|
||||
let mut x = coeffs_chunk.start as usize;
|
||||
let mut result_i32x8 = initial;
|
||||
|
||||
|
||||
@@ -11,17 +11,17 @@ pub(crate) fn horiz_convolution(
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
let initial = 1i32 << (precision - 1);
|
||||
|
||||
let src_rows = src_view.iter_rows(offset);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
for (dst_row, src_row) in dst_rows.zip(src_rows) {
|
||||
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
|
||||
for (coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
|
||||
let first_x_src = coeffs_chunk.start as usize;
|
||||
let mut ss = initial;
|
||||
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
|
||||
for (&k, &src_pixel) in coeffs_chunk.values.iter().zip(src_pixels) {
|
||||
for (&k, &src_pixel) in coeffs_chunk.values().iter().zip(src_pixels) {
|
||||
ss += src_pixel.0 as i32 * (k as i32);
|
||||
}
|
||||
dst_pixel.0 = unsafe { normalizer.clip(ss) };
|
||||
|
||||
@@ -12,14 +12,13 @@ pub(crate) fn horiz_convolution(
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -28,7 +27,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_one_row(src_row, dst_row, &normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -44,15 +43,15 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U8]; 4],
|
||||
dst_rows: [&mut [U8]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
let zero = _mm_setzero_si128();
|
||||
let initial = 1 << (normalizer.precision() - 1);
|
||||
let mut buf = [0, 0, 0, 0, initial];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let coeffs = coeffs_chunk.values;
|
||||
let coeffs = coeffs_chunk.values();
|
||||
let mut x = coeffs_chunk.start as usize;
|
||||
let mut result_i32x4 = [zero, zero, zero, zero];
|
||||
|
||||
@@ -112,15 +111,15 @@ unsafe fn horiz_convolution_four_rows(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U8],
|
||||
dst_row: &mut [U8],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
let zero = _mm_setzero_si128();
|
||||
let initial = 1 << (normalizer.precision() - 1);
|
||||
let mut buf = [0, 0, 0, 0, initial];
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let coeffs = coeffs_chunk.values;
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let coeffs = coeffs_chunk.values();
|
||||
let mut x = coeffs_chunk.start as usize;
|
||||
let mut result_i32x4 = zero;
|
||||
|
||||
|
||||
@@ -12,14 +12,13 @@ pub(crate) fn horiz_convolution(
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -28,7 +27,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_one_row(src_row, dst_row, &normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -44,11 +43,11 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U8x2]; 4],
|
||||
dst_rows: [&mut [U8x2]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let initial = _mm256_set1_epi32(1 << (precision - 2));
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|L A | |L A | |L A | |L A | |L A | |L A | |L A | |L A |
|
||||
@@ -79,7 +78,7 @@ unsafe fn horiz_convolution_four_rows(
|
||||
|
||||
let mut sss0 = initial;
|
||||
let mut sss1 = initial;
|
||||
let coeffs = coeffs_chunk.values;
|
||||
let coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
let reminder = coeffs_by_8.remainder();
|
||||
@@ -214,7 +213,6 @@ unsafe fn set_dst_pixel(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U8x2],
|
||||
dst_row: &mut [U8x2],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
@@ -315,10 +313,11 @@ unsafe fn horiz_convolution_one_row(
|
||||
L: |-1 02| |-1 00|
|
||||
*/
|
||||
let pix_sh4 = _mm_set_epi8(-1, 7, -1, 5, -1, 6, -1, 4, -1, 3, -1, 1, -1, 2, -1, 0);
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x = coeffs_chunk.start as usize;
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let mut sss = if coeffs.len() < 16 {
|
||||
// Lower part will be added to higher, use only half of the error
|
||||
|
||||
@@ -10,15 +10,15 @@ pub(crate) fn horiz_convolution(
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
let initial = 1 << (precision - 1);
|
||||
|
||||
let src_rows = src_view.iter_rows(offset);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
for (dst_row, src_row) in dst_rows.zip(src_rows) {
|
||||
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
|
||||
for (coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
|
||||
let first_x_src = coeffs_chunk.start as usize;
|
||||
let ks = coeffs_chunk.values;
|
||||
let ks = coeffs_chunk.values();
|
||||
let mut ss = [initial; 2];
|
||||
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
|
||||
for (&k, &src_pixel) in ks.iter().zip(src_pixels) {
|
||||
|
||||
@@ -12,14 +12,13 @@ pub(crate) fn horiz_convolution(
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -28,7 +27,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
|
||||
horiz_convolution_one_row(src_row, dst_row, &normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -44,11 +43,11 @@ pub(crate) fn horiz_convolution(
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U8x2]; 4],
|
||||
dst_rows: [&mut [U8x2]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let initial = _mm_set1_epi32(1 << (precision - 2));
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|L A | |L A | |L A | |L A | |L A | |L A | |L A | |L A |
|
||||
@@ -76,7 +75,7 @@ unsafe fn horiz_convolution_four_rows(
|
||||
let mut x = coeffs_chunk.start as usize;
|
||||
|
||||
let mut sss: [__m128i; 4] = [initial; 4];
|
||||
let coeffs = coeffs_chunk.values;
|
||||
let coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
let reminder = coeffs_by_8.remainder();
|
||||
@@ -166,7 +165,6 @@ unsafe fn set_dst_pixel(
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U8x2],
|
||||
dst_row: &mut [U8x2],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
@@ -244,10 +242,11 @@ unsafe fn horiz_convolution_one_row(
|
||||
L: |-1 02| |-1 00|
|
||||
*/
|
||||
let pix_sh3 = _mm_set_epi8(-1, 7, -1, 5, -1, 6, -1, 4, -1, 3, -1, 1, -1, 2, -1, 0);
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x = coeffs_chunk.start as usize;
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
// Lower part will be added to higher, use only half of the error
|
||||
let mut sss = _mm_set1_epi32(1 << (precision - 2));
|
||||
|
||||
@@ -29,14 +29,13 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
offset: u32,
|
||||
normalizer: optimisations::Normalizer16,
|
||||
) {
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &coefficients_chunks);
|
||||
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -45,7 +44,7 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &coefficients_chunks);
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -61,11 +60,12 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
src_rows: [&[U8x3]; 4],
|
||||
dst_rows: [&mut [U8x3]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
let zero = _mm256_setzero_si256();
|
||||
let initial = _mm256_set1_epi32(1 << (PRECISION - 1));
|
||||
let src_width = src_rows[0].len();
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|R G B | |R G B | |R G B | |R G B | |R G B | |R |
|
||||
@@ -102,7 +102,7 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
|
||||
let mut sss0 = initial;
|
||||
let mut sss1 = initial;
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
// (16 bytes) / (3 bytes per pixel) = 5 whole pixels + 1 byte
|
||||
let max_x = src_width.saturating_sub(5);
|
||||
@@ -224,7 +224,7 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
src_row: &[U8x3],
|
||||
dst_row: &mut [U8x3],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
#[rustfmt::skip]
|
||||
let sh1 = _mm256_set_epi8(
|
||||
@@ -271,11 +271,12 @@ unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
*/
|
||||
let sh7 = _mm_set_epi8(-1, -1, -1, -1, -1, 5, -1, 2, -1, 4, -1, 1, -1, 3, -1, 0);
|
||||
let src_width = src_row.len();
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let x_start = coeffs_chunk.start as usize;
|
||||
let mut x = x_start;
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
// (16 bytes) / (3 bytes per pixel) = 5 whole pixels + 1 bytes
|
||||
// 4 + 5 = 9
|
||||
|
||||
@@ -11,17 +11,17 @@ pub(crate) fn horiz_convolution(
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let coefficients = normalizer.coefficients();
|
||||
let initial = 1i32 << (precision - 1);
|
||||
|
||||
let src_rows = src_view.iter_rows(offset);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
for (dst_row, src_row) in dst_rows.zip(src_rows) {
|
||||
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
|
||||
for (coeffs_chunk, dst_pixel) in coefficients.iter().zip(dst_row.iter_mut()) {
|
||||
let first_x_src = coeffs_chunk.start as usize;
|
||||
let mut ss = [initial; 3];
|
||||
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
|
||||
for (&k, src_pixel) in coeffs_chunk.values.iter().zip(src_pixels) {
|
||||
for (&k, src_pixel) in coeffs_chunk.values().iter().zip(src_pixels) {
|
||||
for (s, c) in ss.iter_mut().zip(src_pixel.0) {
|
||||
*s += c as i32 * (k as i32);
|
||||
}
|
||||
|
||||
@@ -29,14 +29,13 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
offset: u32,
|
||||
normalizer: optimisations::Normalizer16,
|
||||
) {
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &coefficients_chunks);
|
||||
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -45,7 +44,7 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &coefficients_chunks);
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -61,11 +60,12 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
src_rows: [&[U8x3]; 4],
|
||||
dst_rows: [&mut [U8x3]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
let zero = _mm_setzero_si128();
|
||||
let initial = _mm_set1_epi32(1 << (PRECISION - 1));
|
||||
let src_width = src_rows[0].len();
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
/*
|
||||
|R G B | |R G B | |R G B | |R G B | |R G B | |R |
|
||||
@@ -99,7 +99,7 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
let mut x = x_start;
|
||||
|
||||
let mut sss_a = [initial; 4];
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
// Next block of code will be load source pixels by 16 bytes per time.
|
||||
// We must guarantee what this process will not go beyond
|
||||
@@ -187,7 +187,7 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
src_row: &[U8x3],
|
||||
dst_row: &mut [U8x3],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
#[rustfmt::skip]
|
||||
let pix_sh1 = _mm_set_epi8(
|
||||
@@ -219,11 +219,12 @@ unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
R: |-1 03| |-1 00|
|
||||
*/
|
||||
let src_width = src_row.len();
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let x_start = coeffs_chunk.start as usize;
|
||||
let mut x = x_start;
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
let mut sss = _mm_set1_epi32(1 << (PRECISION - 1));
|
||||
|
||||
// Next block of code will be load source pixels by 16 bytes per time.
|
||||
|
||||
@@ -32,14 +32,13 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
offset: u32,
|
||||
normalizer: optimisations::Normalizer16,
|
||||
) {
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &coefficients_chunks);
|
||||
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -48,7 +47,7 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &coefficients_chunks);
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -64,7 +63,7 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
src_rows: [&[U8x4]; 4],
|
||||
dst_rows: [&mut [U8x4]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
let zero = _mm256_setzero_si256();
|
||||
let initial = _mm256_set1_epi32(1 << (PRECISION - 1));
|
||||
@@ -80,12 +79,14 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
|
||||
);
|
||||
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x = coeffs_chunk.start as usize;
|
||||
|
||||
let mut sss0 = initial;
|
||||
let mut sss1 = initial;
|
||||
let coeffs = coeffs_chunk.values;
|
||||
let coeffs = coeffs_chunk.values();
|
||||
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
let reminder1 = coeffs_by_4.remainder();
|
||||
@@ -164,13 +165,13 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
sss0 = _mm256_packus_epi16(sss0, zero);
|
||||
sss1 = _mm256_packus_epi16(sss1, zero);
|
||||
*dst_rows[0].get_unchecked_mut(dst_x) =
|
||||
transmute(_mm_cvtsi128_si32(_mm256_extracti128_si256::<0>(sss0)));
|
||||
transmute::<i32, U8x4>(_mm_cvtsi128_si32(_mm256_extracti128_si256::<0>(sss0)));
|
||||
*dst_rows[1].get_unchecked_mut(dst_x) =
|
||||
transmute(_mm_cvtsi128_si32(_mm256_extracti128_si256::<1>(sss0)));
|
||||
transmute::<i32, U8x4>(_mm_cvtsi128_si32(_mm256_extracti128_si256::<1>(sss0)));
|
||||
*dst_rows[2].get_unchecked_mut(dst_x) =
|
||||
transmute(_mm_cvtsi128_si32(_mm256_extracti128_si256::<0>(sss1)));
|
||||
transmute::<i32, U8x4>(_mm_cvtsi128_si32(_mm256_extracti128_si256::<0>(sss1)));
|
||||
*dst_rows[3].get_unchecked_mut(dst_x) =
|
||||
transmute(_mm_cvtsi128_si32(_mm256_extracti128_si256::<1>(sss1)));
|
||||
transmute::<i32, U8x4>(_mm_cvtsi128_si32(_mm256_extracti128_si256::<1>(sss1)));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -184,7 +185,7 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
src_row: &[U8x4],
|
||||
dst_row: &mut [U8x4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
#[rustfmt::skip]
|
||||
let sh1 = _mm256_set_epi8(
|
||||
@@ -218,9 +219,11 @@ unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
);
|
||||
let sh7 = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x = coeffs_chunk.start as usize;
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut coeffs = coeffs_chunk.values();
|
||||
|
||||
let mut sss = if coeffs.len() < 8 {
|
||||
_mm_set1_epi32(1 << (PRECISION - 1))
|
||||
@@ -293,6 +296,6 @@ unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
|
||||
sss = _mm_packs_epi32(sss, sss);
|
||||
*dst_row.get_unchecked_mut(dst_x) =
|
||||
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss, sss)));
|
||||
transmute::<i32, U8x4>(_mm_cvtsi128_si32(_mm_packus_epi16(sss, sss)));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -11,17 +11,16 @@ pub(crate) fn horiz_convolution(
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let coefficients = normalizer.coefficients();
|
||||
let initial = 1 << (precision - 1);
|
||||
|
||||
let src_rows = src_view.iter_rows(offset);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
for (dst_row, src_row) in dst_rows.zip(src_rows) {
|
||||
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
|
||||
let first_x_src = coeffs_chunk.start as usize;
|
||||
for (chunk, dst_pixel) in coefficients.iter().zip(dst_row.iter_mut()) {
|
||||
let mut ss = [initial; 4];
|
||||
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
|
||||
for (&k, &src_pixel) in coeffs_chunk.values.iter().zip(src_pixels) {
|
||||
let src_pixels = unsafe { src_row.get_unchecked(chunk.start as usize..) };
|
||||
for (&k, &src_pixel) in chunk.values().iter().zip(src_pixels) {
|
||||
for (i, s) in ss.iter_mut().enumerate() {
|
||||
*s += src_pixel.0[i] as i32 * (k as i32);
|
||||
}
|
||||
|
||||
@@ -32,14 +32,13 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
offset: u32,
|
||||
normalizer: optimisations::Normalizer16,
|
||||
) {
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &coefficients_chunks);
|
||||
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -48,7 +47,7 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &coefficients_chunks);
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -63,23 +62,22 @@ fn horiz_convolution_p<const PRECISION: i32>(
|
||||
unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
src_rows: [&[U8x4]; 4],
|
||||
dst_rows: [&mut [U8x4]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
let initial = _mm_set1_epi32(1 << (PRECISION - 1));
|
||||
let mask_lo = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
|
||||
let mask_hi = _mm_set_epi8(-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8);
|
||||
let mask = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
for (dst_x, chunk) in normalizer.coefficients().iter().enumerate() {
|
||||
let mut x: usize = chunk.start as usize;
|
||||
|
||||
let mut sss0 = initial;
|
||||
let mut sss1 = initial;
|
||||
let mut sss2 = initial;
|
||||
let mut sss3 = initial;
|
||||
|
||||
let coeffs = coeffs_chunk.values;
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
let coeffs_by_4 = chunk.values().chunks_exact(4);
|
||||
let reminder1 = coeffs_by_4.remainder();
|
||||
|
||||
for k in coeffs_by_4 {
|
||||
@@ -189,7 +187,7 @@ unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
src_row: &[U8x4],
|
||||
dst_row: &mut [U8x4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
let initial = _mm_set1_epi32(1 << (PRECISION - 1));
|
||||
let sh1 = _mm_set_epi8(-1, 11, -1, 3, -1, 10, -1, 2, -1, 9, -1, 1, -1, 8, -1, 0);
|
||||
@@ -202,11 +200,11 @@ unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
);
|
||||
let sh7 = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
for (dst_x, chunk) in normalizer.coefficients().iter().enumerate() {
|
||||
let mut x: usize = chunk.start as usize;
|
||||
let mut sss = initial;
|
||||
|
||||
let coeffs_by_8 = coeffs_chunk.values.chunks_exact(8);
|
||||
let coeffs_by_8 = chunk.values().chunks_exact(8);
|
||||
let reminder8 = coeffs_by_8.remainder();
|
||||
|
||||
for k in coeffs_by_8 {
|
||||
@@ -274,6 +272,6 @@ unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
sss = _mm_srai_epi32::<PRECISION>(sss);
|
||||
sss = _mm_packs_epi32(sss, sss);
|
||||
*dst_row.get_unchecked_mut(dst_x) =
|
||||
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss, sss)));
|
||||
transmute::<i32, U8x4>(_mm_cvtsi128_si32(_mm_packus_epi16(sss, sss)));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -13,7 +13,7 @@ pub(crate) fn vert_convolution<T>(
|
||||
T: InnerPixel<Component = u16>,
|
||||
{
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
let src_x = offset as usize * T::count_of_components();
|
||||
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
@@ -29,13 +29,13 @@ pub(crate) unsafe fn vert_convolution_into_one_row_u16<T>(
|
||||
src_view: &impl ImageView<Pixel = T>,
|
||||
dst_row: &mut [T],
|
||||
mut src_x: usize,
|
||||
coeffs_chunk: optimisations::CoefficientsI32Chunk,
|
||||
coeffs_chunk: &optimisations::CoefficientsI32Chunk,
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) where
|
||||
T: InnerPixel<Component = u16>,
|
||||
{
|
||||
let y_start = coeffs_chunk.start;
|
||||
let coeffs = coeffs_chunk.values;
|
||||
let coeffs = coeffs_chunk.values();
|
||||
let mut dst_u16 = T::components_mut(dst_row);
|
||||
|
||||
/*
|
||||
|
||||
@@ -13,7 +13,7 @@ pub(crate) fn vert_convolution<T>(
|
||||
T: InnerPixel<Component = u16>,
|
||||
{
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
let precision = normalizer.precision();
|
||||
let initial: i64 = 1 << (precision - 1);
|
||||
let src_x_initial = offset as usize * T::count_of_components();
|
||||
@@ -22,7 +22,7 @@ pub(crate) fn vert_convolution<T>(
|
||||
let coeffs_chunks_iter = coefficients_chunks.into_iter();
|
||||
for (coeffs_chunk, dst_row) in coeffs_chunks_iter.zip(dst_rows) {
|
||||
let first_y_src = coeffs_chunk.start;
|
||||
let ks = coeffs_chunk.values;
|
||||
let ks = coeffs_chunk.values();
|
||||
let dst_components = T::components_mut(dst_row);
|
||||
let mut x_src = src_x_initial;
|
||||
|
||||
|
||||
@@ -15,7 +15,7 @@ pub(crate) fn vert_convolution<T>(
|
||||
T: InnerPixel<Component = u16>,
|
||||
{
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
let src_x = offset as usize * T::count_of_components();
|
||||
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
@@ -31,11 +31,11 @@ unsafe fn vert_convolution_into_one_row_u16<T: InnerPixel<Component = u16>>(
|
||||
src_view: &impl ImageView<Pixel = T>,
|
||||
dst_row: &mut [T],
|
||||
mut src_x: usize,
|
||||
coeffs_chunk: CoefficientsI32Chunk,
|
||||
coeffs_chunk: &CoefficientsI32Chunk,
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let y_start = coeffs_chunk.start;
|
||||
let coeffs = coeffs_chunk.values;
|
||||
let coeffs = coeffs_chunk.values();
|
||||
let max_rows = coeffs.len() as u32;
|
||||
let mut dst_u16 = T::components_mut(dst_row);
|
||||
|
||||
|
||||
@@ -34,7 +34,7 @@ fn vert_convolution_p<T, const PRECISION: i32>(
|
||||
) where
|
||||
T: InnerPixel<Component = u8>,
|
||||
{
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
let src_x = offset as usize * T::count_of_components();
|
||||
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
@@ -57,13 +57,13 @@ unsafe fn vert_convolution_into_one_row<T, const PRECISION: i32>(
|
||||
src_view: &impl ImageView<Pixel = T>,
|
||||
dst_row: &mut [T],
|
||||
mut src_x: usize,
|
||||
coeffs_chunk: optimisations::CoefficientsI16Chunk,
|
||||
coeffs_chunk: &optimisations::CoefficientsI16Chunk,
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) where
|
||||
T: InnerPixel<Component = u8>,
|
||||
{
|
||||
let y_start = coeffs_chunk.start;
|
||||
let coeffs = coeffs_chunk.values;
|
||||
let coeffs = coeffs_chunk.values();
|
||||
let max_rows = coeffs.len() as u32;
|
||||
|
||||
let initial = _mm_set1_epi32(1 << (PRECISION as u8 - 1));
|
||||
|
||||
@@ -13,16 +13,16 @@ pub(crate) fn vert_convolution<T>(
|
||||
T: InnerPixel<Component = u8>,
|
||||
{
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
let precision = normalizer.precision();
|
||||
let initial = 1 << (precision - 1);
|
||||
let src_x_initial = offset as usize * T::count_of_components();
|
||||
|
||||
let dst_rows = dst_image.iter_rows_mut(0);
|
||||
let coeffs_chunks_iter = coefficients_chunks.into_iter();
|
||||
let coeffs_chunks_iter = coefficients_chunks.iter();
|
||||
for (coeffs_chunk, dst_row) in coeffs_chunks_iter.zip(dst_rows) {
|
||||
let first_y_src = coeffs_chunk.start;
|
||||
let ks = coeffs_chunk.values;
|
||||
let ks = coeffs_chunk.values();
|
||||
let mut x_src = src_x_initial;
|
||||
let dst_components = T::components_mut(dst_row);
|
||||
|
||||
|
||||
@@ -33,7 +33,7 @@ fn vert_convolution_p<T, const PRECISION: i32>(
|
||||
) where
|
||||
T: InnerPixel<Component = u8>,
|
||||
{
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let coefficients_chunks = normalizer.coefficients();
|
||||
let src_x = offset as usize * T::count_of_components();
|
||||
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
@@ -55,13 +55,13 @@ unsafe fn vert_convolution_into_one_row<T, const PRECISION: i32>(
|
||||
src_view: &impl ImageView<Pixel = T>,
|
||||
dst_row: &mut [T],
|
||||
mut src_x: usize,
|
||||
coeffs_chunk: optimisations::CoefficientsI16Chunk,
|
||||
coeffs_chunk: &optimisations::CoefficientsI16Chunk,
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) where
|
||||
T: InnerPixel<Component = u8>,
|
||||
{
|
||||
let y_start = coeffs_chunk.start;
|
||||
let coeffs = coeffs_chunk.values;
|
||||
let coeffs = coeffs_chunk.values();
|
||||
let max_rows = coeffs.len() as u32;
|
||||
let mut dst_u8 = T::components_mut(dst_row);
|
||||
|
||||
|
||||
+15
-15
@@ -3,7 +3,7 @@ use std::io::BufReader;
|
||||
use std::num::NonZeroU32;
|
||||
use std::ops::Deref;
|
||||
|
||||
use image::io::{Reader as ImageReader, Reader};
|
||||
use image::ImageReader;
|
||||
use image::{ColorType, ExtendedColorType, ImageBuffer};
|
||||
|
||||
use fast_image_resize::images::Image;
|
||||
@@ -112,7 +112,7 @@ pub trait PixelTestingExt: PixelTrait {
|
||||
}
|
||||
|
||||
fn load_image_buffer(
|
||||
img_reader: Reader<BufReader<File>>,
|
||||
img_reader: ImageReader<BufReader<File>>,
|
||||
) -> ImageBuffer<Self::ImagePixel, Self::Container>;
|
||||
|
||||
fn load_big_image() -> ImageBuffer<Self::ImagePixel, Self::Container> {
|
||||
@@ -172,7 +172,7 @@ pub mod not_u8x4 {
|
||||
type Container = Vec<u8>;
|
||||
|
||||
fn load_image_buffer(
|
||||
img_reader: Reader<BufReader<File>>,
|
||||
img_reader: ImageReader<BufReader<File>>,
|
||||
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
|
||||
img_reader.decode().unwrap().to_luma8()
|
||||
}
|
||||
@@ -187,7 +187,7 @@ pub mod not_u8x4 {
|
||||
type Container = Vec<u8>;
|
||||
|
||||
fn load_image_buffer(
|
||||
img_reader: Reader<BufReader<File>>,
|
||||
img_reader: ImageReader<BufReader<File>>,
|
||||
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
|
||||
img_reader.decode().unwrap().to_luma_alpha8()
|
||||
}
|
||||
@@ -202,7 +202,7 @@ pub mod not_u8x4 {
|
||||
type Container = Vec<u8>;
|
||||
|
||||
fn load_image_buffer(
|
||||
img_reader: Reader<BufReader<File>>,
|
||||
img_reader: ImageReader<BufReader<File>>,
|
||||
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
|
||||
img_reader.decode().unwrap().to_rgb8()
|
||||
}
|
||||
@@ -217,7 +217,7 @@ pub mod not_u8x4 {
|
||||
type Container = Vec<u16>;
|
||||
|
||||
fn load_image_buffer(
|
||||
img_reader: Reader<BufReader<File>>,
|
||||
img_reader: ImageReader<BufReader<File>>,
|
||||
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
|
||||
img_reader.decode().unwrap().to_luma16()
|
||||
}
|
||||
@@ -237,7 +237,7 @@ pub mod not_u8x4 {
|
||||
type Container = Vec<u16>;
|
||||
|
||||
fn load_image_buffer(
|
||||
img_reader: Reader<BufReader<File>>,
|
||||
img_reader: ImageReader<BufReader<File>>,
|
||||
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
|
||||
img_reader.decode().unwrap().to_luma_alpha16()
|
||||
}
|
||||
@@ -252,7 +252,7 @@ pub mod not_u8x4 {
|
||||
type Container = Vec<u16>;
|
||||
|
||||
fn load_image_buffer(
|
||||
img_reader: Reader<BufReader<File>>,
|
||||
img_reader: ImageReader<BufReader<File>>,
|
||||
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
|
||||
img_reader.decode().unwrap().to_rgb16()
|
||||
}
|
||||
@@ -267,7 +267,7 @@ pub mod not_u8x4 {
|
||||
type Container = Vec<u16>;
|
||||
|
||||
fn load_image_buffer(
|
||||
img_reader: Reader<BufReader<File>>,
|
||||
img_reader: ImageReader<BufReader<File>>,
|
||||
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
|
||||
img_reader.decode().unwrap().to_rgba16()
|
||||
}
|
||||
@@ -286,7 +286,7 @@ pub mod not_u8x4 {
|
||||
}
|
||||
|
||||
fn load_image_buffer(
|
||||
img_reader: Reader<BufReader<File>>,
|
||||
img_reader: ImageReader<BufReader<File>>,
|
||||
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
|
||||
let image_u16 = img_reader.decode().unwrap().to_luma32f();
|
||||
ImageBuffer::from_fn(image_u16.width(), image_u16.height(), |x, y| {
|
||||
@@ -308,7 +308,7 @@ pub mod not_u8x4 {
|
||||
type Container = Vec<f32>;
|
||||
|
||||
fn load_image_buffer(
|
||||
img_reader: Reader<BufReader<File>>,
|
||||
img_reader: ImageReader<BufReader<File>>,
|
||||
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
|
||||
img_reader.decode().unwrap().to_luma32f()
|
||||
}
|
||||
@@ -326,7 +326,7 @@ pub mod not_u8x4 {
|
||||
type Container = Vec<f32>;
|
||||
|
||||
fn load_image_buffer(
|
||||
img_reader: Reader<BufReader<File>>,
|
||||
img_reader: ImageReader<BufReader<File>>,
|
||||
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
|
||||
img_reader.decode().unwrap().to_luma_alpha32f()
|
||||
}
|
||||
@@ -344,7 +344,7 @@ pub mod not_u8x4 {
|
||||
type Container = Vec<f32>;
|
||||
|
||||
fn load_image_buffer(
|
||||
img_reader: Reader<BufReader<File>>,
|
||||
img_reader: ImageReader<BufReader<File>>,
|
||||
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
|
||||
img_reader.decode().unwrap().to_rgb32f()
|
||||
}
|
||||
@@ -359,7 +359,7 @@ pub mod not_u8x4 {
|
||||
type Container = Vec<f32>;
|
||||
|
||||
fn load_image_buffer(
|
||||
img_reader: Reader<BufReader<File>>,
|
||||
img_reader: ImageReader<BufReader<File>>,
|
||||
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
|
||||
img_reader.decode().unwrap().to_rgba32f()
|
||||
}
|
||||
@@ -375,7 +375,7 @@ impl PixelTestingExt for U8x4 {
|
||||
type Container = Vec<u8>;
|
||||
|
||||
fn load_image_buffer(
|
||||
img_reader: Reader<BufReader<File>>,
|
||||
img_reader: ImageReader<BufReader<File>>,
|
||||
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
|
||||
img_reader.decode().unwrap().to_rgba8()
|
||||
}
|
||||
|
||||
@@ -1,8 +1,6 @@
|
||||
use std::cmp::Ordering;
|
||||
use std::fmt::Debug;
|
||||
|
||||
use image::io::Reader as ImageReader;
|
||||
|
||||
use fast_image_resize::images::{Image, TypedImage, TypedImageRef};
|
||||
use fast_image_resize::pixels::*;
|
||||
use fast_image_resize::{
|
||||
@@ -865,9 +863,9 @@ mod not_u8x4 {
|
||||
}
|
||||
|
||||
mod u8x4 {
|
||||
use std::f64::consts::PI;
|
||||
|
||||
use fast_image_resize::ResizeError;
|
||||
use image::ImageReader;
|
||||
use std::f64::consts::PI;
|
||||
|
||||
use super::*;
|
||||
|
||||
|
||||
Reference in New Issue
Block a user