- Significantly improved (4.5 times on x86_64) speed of vertical convolution pass implemented

in native Rust for `U8`, `U8x2`, `U8x3` and `U8x4` images.
- Changed order of convolution passes for `U8`, `U8x2`, `U8x3` and `U8x4` images.
  Now the vertical pass is the first and the horizontal pass is the second.
- **BREAKING**: Changed internal data type for `U8x4` structure. Not it is `[u8; 4]` instead of `u32`.
This commit is contained in:
Kirill Kuzminykh
2023-12-14 22:45:26 +03:00
parent 63369009ec
commit c195c4ff43
18 changed files with 471 additions and 178 deletions
+5
View File
@@ -4,6 +4,11 @@
- Slightly improved (about 3%) speed of `AVX2` implementation of `Convolution` trait
for `U8x3` and `U8x4` images.
- Significantly improved (4.5 times on `x86_64`) speed of vertical convolution pass implemented
in native Rust for `U8`, `U8x2`, `U8x3` and `U8x4` images.
- Changed order of convolution passes for `U8`, `U8x2`, `U8x3` and `U8x4` images.
Now the vertical pass is the first and the horizontal pass is the second.
- **BREAKING**: Changed internal data type for `U8x4` structure. Not it is `[u8; 4]` instead of `u32`.
## [2.7.3] - 2023-05-07
+1
View File
@@ -107,6 +107,7 @@ harness = false
[profile.dev.package.'*']
opt-level = 3
debug = false
[profile.release]
+17 -17
View File
@@ -66,12 +66,12 @@ Pipeline:
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
| image | 28.98 | - | 85.03 | 146.36 | 191.18 |
| resize | - | 26.83 | 53.33 | 97.71 | 144.57 |
| libvips | 7.73 | 59.65 | 19.86 | 30.40 | 39.81 |
| fir rust | 0.28 | 23.63 | 39.82 | 74.52 | 107.93 |
| fir sse4.1 | 0.28 | 7.89 | 9.46 | 14.01 | 19.64 |
| fir avx2 | 0.28 | 6.67 | 7.44 | 9.66 | 13.92 |
| image | 31.83 | - | 90.73 | 157.56 | 210.36 |
| resize | - | 26.82 | 54.07 | 97.90 | 144.95 |
| libvips | 7.65 | 59.54 | 19.80 | 30.02 | 39.42 |
| fir rust | 0.28 | 9.79 | 15.47 | 27.34 | 39.58 |
| fir sse4.1 | 0.28 | 3.96 | 5.68 | 10.12 | 15.86 |
| fir avx2 | 0.28 | 2.73 | 3.59 | 6.82 | 13.16 |
<!-- bench_compare_rgb end -->
<!-- bench_compare_rgba start -->
@@ -88,11 +88,11 @@ Pipeline:
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|------------|:-------:|:------:|:--------:|:-------:|:--------:|
| resize | - | 42.84 | 85.25 | 147.44 | 211.69 |
| libvips | 9.87 | 122.96 | 190.30 | 339.61 | 501.24 |
| fir rust | 0.19 | 60.54 | 100.92 | 189.20 | 287.01 |
| fir sse4.1 | 0.19 | 11.09 | 12.94 | 17.08 | 22.32 |
| fir avx2 | 0.19 | 8.69 | 9.40 | 11.84 | 15.86 |
| resize | - | 43.26 | 85.63 | 147.73 | 211.88 |
| libvips | 10.14 | 121.46 | 190.25 | 337.97 | 500.06 |
| fir rust | 0.19 | 20.22 | 27.17 | 41.57 | 56.84 |
| fir sse4.1 | 0.19 | 10.04 | 12.29 | 18.50 | 25.14 |
| fir avx2 | 0.19 | 7.12 | 8.20 | 13.86 | 22.36 |
<!-- bench_compare_rgba end -->
<!-- bench_compare_l start -->
@@ -108,12 +108,12 @@ Pipeline:
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
| image | 26.15 | - | 57.74 | 85.31 | 113.69 |
| resize | - | 10.86 | 18.68 | 37.91 | 65.64 |
| libvips | 4.66 | 25.00 | 9.58 | 13.10 | 17.94 |
| fir rust | 0.15 | 12.35 | 13.42 | 15.05 | 22.06 |
| fir sse4.1 | 0.15 | 5.67 | 5.23 | 5.91 | 8.44 |
| fir avx2 | 0.15 | 6.63 | 6.20 | 4.52 | 7.81 |
| image | 25.71 | - | 57.23 | 86.20 | 113.09 |
| resize | - | 10.85 | 18.90 | 38.28 | 65.62 |
| libvips | 4.69 | 25.00 | 9.63 | 13.40 | 17.98 |
| fir rust | 0.16 | 4.29 | 5.27 | 7.50 | 11.41 |
| fir sse4.1 | 0.16 | 1.88 | 2.32 | 3.61 | 5.95 |
| fir avx2 | 0.16 | 1.70 | 1.92 | 2.35 | 4.47 |
<!-- bench_compare_l end -->
## Examples
+105 -19
View File
@@ -1,16 +1,16 @@
use fast_image_resize::pixels::*;
use fast_image_resize::Image;
use fast_image_resize::{CpuExtensions, FilterType, PixelType, ResizeAlg, Resizer};
use std::num::NonZeroU32;
use testing::{cpu_ext_into_str, nonzero, PixelTestingExt};
mod utils;
const NEW_WIDTH: u32 = 852;
const NEW_HEIGHT: u32 = 567;
const NEW_SIZE: u32 = 695;
fn native_nearest_u8x4_bench(bench_group: &mut utils::BenchGroup) {
let image = U8x4::load_big_src_image();
let mut res_image = Image::new(nonzero(NEW_WIDTH), nonzero(NEW_HEIGHT), image.pixel_type());
let image = U8x4::load_big_square_src_image();
let mut res_image = Image::new(nonzero(NEW_SIZE), nonzero(NEW_SIZE), image.pixel_type());
let src_image = image.view();
let mut dst_image = res_image.view_mut();
let mut resizer = Resizer::new(ResizeAlg::Nearest);
@@ -25,8 +25,8 @@ fn native_nearest_u8x4_bench(bench_group: &mut utils::BenchGroup) {
}
fn native_nearest_u8_bench(bench_group: &mut utils::BenchGroup) {
let image = U8::load_big_src_image();
let mut res_image = Image::new(nonzero(NEW_WIDTH), nonzero(NEW_HEIGHT), image.pixel_type());
let image = U8::load_big_square_src_image();
let mut res_image = Image::new(nonzero(NEW_SIZE), nonzero(NEW_SIZE), image.pixel_type());
let src_image = image.view();
let mut dst_image = res_image.view_mut();
let mut resizer = Resizer::new(ResizeAlg::Nearest);
@@ -45,19 +45,27 @@ fn downscale_bench(
image: &Image<'static>,
cpu_extensions: CpuExtensions,
filter_type: FilterType,
dst_width: NonZeroU32,
dst_height: NonZeroU32,
name_prefix: &str,
) {
let mut res_image = Image::new(nonzero(NEW_WIDTH), nonzero(NEW_HEIGHT), image.pixel_type());
let mut res_image = Image::new(dst_width, dst_height, image.pixel_type());
let src_image = image.view();
let mut dst_image = res_image.view_mut();
let mut resizer = Resizer::new(ResizeAlg::Convolution(filter_type));
unsafe {
resizer.set_cpu_extensions(cpu_extensions);
}
let prefix = if name_prefix.is_empty() {
"".to_string()
} else {
format!(" {}", name_prefix)
};
utils::bench(
bench_group,
100,
&format!("{:?} {:?}", image.pixel_type(), filter_type),
cpu_ext_into_str(cpu_extensions),
&format!("{}{}", cpu_ext_into_str(cpu_extensions), prefix),
|bencher| {
bencher.iter(|| {
resizer.resize(&src_image, &mut dst_image).unwrap();
@@ -66,6 +74,71 @@ fn downscale_bench(
);
}
pub fn resize_in_one_dimension_bench(bench_group: &mut utils::BenchGroup) {
let pixel_types = [
PixelType::U8,
PixelType::U8x2,
PixelType::U8x3,
PixelType::U8x4,
PixelType::U16,
PixelType::U16x2,
PixelType::U16x3,
PixelType::U16x4,
];
let mut cpu_extensions = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions.push(CpuExtensions::Sse4_1);
cpu_extensions.push(CpuExtensions::Avx2);
}
#[cfg(target_arch = "aarch64")]
{
cpu_extensions.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions.push(CpuExtensions::Simd128);
}
for pixel_type in pixel_types {
#[cfg(feature = "only_u8x4")]
if pixel_type != PixelType::U8x4 {
continue;
}
for &cpu_extension in cpu_extensions.iter() {
let image = match pixel_type {
PixelType::U8 => U8::load_big_square_src_image(),
PixelType::U8x2 => U8x2::load_big_square_src_image(),
PixelType::U8x3 => U8x3::load_big_square_src_image(),
PixelType::U8x4 => U8x4::load_big_square_src_image(),
PixelType::U16 => U16::load_big_square_src_image(),
PixelType::U16x2 => U16x2::load_big_square_src_image(),
PixelType::U16x3 => U16x3::load_big_square_src_image(),
PixelType::U16x4 => U16x4::load_big_square_src_image(),
PixelType::I32 => I32::load_big_square_src_image(),
_ => unreachable!(),
};
downscale_bench(
bench_group,
&image,
cpu_extension,
FilterType::Lanczos3,
nonzero(NEW_SIZE),
image.height(),
"H",
);
downscale_bench(
bench_group,
&image,
cpu_extension,
FilterType::Lanczos3,
image.height(),
nonzero(NEW_SIZE),
"V",
);
}
}
}
pub fn resize_bench(bench_group: &mut utils::BenchGroup) {
let pixel_types = [
PixelType::U8,
@@ -99,18 +172,26 @@ pub fn resize_bench(bench_group: &mut utils::BenchGroup) {
}
for &cpu_extension in cpu_extensions.iter() {
let image = match pixel_type {
PixelType::U8 => U8::load_big_src_image(),
PixelType::U8x2 => U8x2::load_big_src_image(),
PixelType::U8x3 => U8x3::load_big_src_image(),
PixelType::U8x4 => U8x4::load_big_src_image(),
PixelType::U16 => U16::load_big_src_image(),
PixelType::U16x2 => U16x2::load_big_src_image(),
PixelType::U16x3 => U16x3::load_big_src_image(),
PixelType::U16x4 => U16x4::load_big_src_image(),
PixelType::I32 => I32::load_big_src_image(),
PixelType::U8 => U8::load_big_square_src_image(),
PixelType::U8x2 => U8x2::load_big_square_src_image(),
PixelType::U8x3 => U8x3::load_big_square_src_image(),
PixelType::U8x4 => U8x4::load_big_square_src_image(),
PixelType::U16 => U16::load_big_square_src_image(),
PixelType::U16x2 => U16x2::load_big_square_src_image(),
PixelType::U16x3 => U16x3::load_big_square_src_image(),
PixelType::U16x4 => U16x4::load_big_square_src_image(),
PixelType::I32 => I32::load_big_square_src_image(),
_ => unreachable!(),
};
downscale_bench(bench_group, &image, cpu_extension, FilterType::Lanczos3);
downscale_bench(
bench_group,
&image,
cpu_extension,
FilterType::Lanczos3,
nonzero(NEW_SIZE),
nonzero(NEW_SIZE),
"",
);
}
}
@@ -119,7 +200,12 @@ pub fn resize_bench(bench_group: &mut utils::BenchGroup) {
native_nearest_u8_bench(bench_group);
}
fn main() {
fn main1() {
let results = utils::run_bench(resize_bench, "Resize");
println!("{}", utils::build_md_table(&results));
}
fn main() {
let results = utils::run_bench(resize_in_one_dimension_bench, "Resize one dimension");
println!("{}", utils::build_md_table(&results));
}
+42 -42
View File
@@ -39,12 +39,12 @@ Pipeline:
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
| image | 28.98 | - | 85.03 | 146.36 | 191.18 |
| resize | - | 26.83 | 53.33 | 97.71 | 144.57 |
| libvips | 7.73 | 59.65 | 19.86 | 30.40 | 39.81 |
| fir rust | 0.28 | 23.63 | 39.82 | 74.52 | 107.93 |
| fir sse4.1 | 0.28 | 7.89 | 9.46 | 14.01 | 19.64 |
| fir avx2 | 0.28 | 6.67 | 7.44 | 9.66 | 13.92 |
| image | 31.83 | - | 90.73 | 157.56 | 210.36 |
| resize | - | 26.82 | 54.07 | 97.90 | 144.95 |
| libvips | 7.65 | 59.54 | 19.80 | 30.02 | 39.42 |
| fir rust | 0.28 | 9.79 | 15.47 | 27.34 | 39.58 |
| fir sse4.1 | 0.28 | 3.96 | 5.68 | 10.12 | 15.86 |
| fir avx2 | 0.28 | 2.73 | 3.59 | 6.82 | 13.16 |
<!-- bench_compare_rgb end -->
<!-- bench_compare_rgba start -->
@@ -61,11 +61,11 @@ Pipeline:
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|------------|:-------:|:------:|:--------:|:-------:|:--------:|
| resize | - | 42.84 | 85.25 | 147.44 | 211.69 |
| libvips | 9.87 | 122.96 | 190.30 | 339.61 | 501.24 |
| fir rust | 0.19 | 60.54 | 100.92 | 189.20 | 287.01 |
| fir sse4.1 | 0.19 | 11.09 | 12.94 | 17.08 | 22.32 |
| fir avx2 | 0.19 | 8.69 | 9.40 | 11.84 | 15.86 |
| resize | - | 43.26 | 85.63 | 147.73 | 211.88 |
| libvips | 10.14 | 121.46 | 190.25 | 337.97 | 500.06 |
| fir rust | 0.19 | 20.22 | 27.17 | 41.57 | 56.84 |
| fir sse4.1 | 0.19 | 10.04 | 12.29 | 18.50 | 25.14 |
| fir avx2 | 0.19 | 7.12 | 8.20 | 13.86 | 22.36 |
<!-- bench_compare_rgba end -->
<!-- bench_compare_l start -->
@@ -81,12 +81,12 @@ Pipeline:
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
| image | 26.15 | - | 57.74 | 85.31 | 113.69 |
| resize | - | 10.86 | 18.68 | 37.91 | 65.64 |
| libvips | 4.66 | 25.00 | 9.58 | 13.10 | 17.94 |
| fir rust | 0.15 | 12.35 | 13.42 | 15.05 | 22.06 |
| fir sse4.1 | 0.15 | 5.67 | 5.23 | 5.91 | 8.44 |
| fir avx2 | 0.15 | 6.63 | 6.20 | 4.52 | 7.81 |
| image | 25.71 | - | 57.23 | 86.20 | 113.09 |
| resize | - | 10.85 | 18.90 | 38.28 | 65.62 |
| libvips | 4.69 | 25.00 | 9.63 | 13.40 | 17.98 |
| fir rust | 0.16 | 4.29 | 5.27 | 7.50 | 11.41 |
| fir sse4.1 | 0.16 | 1.88 | 2.32 | 3.61 | 5.95 |
| fir avx2 | 0.16 | 1.70 | 1.92 | 2.35 | 4.47 |
<!-- bench_compare_l end -->
<!-- bench_compare_la start -->
@@ -105,10 +105,10 @@ Pipeline:
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
| libvips | 6.45 | 73.14 | 118.34 | 205.71 | 293.96 |
| fir rust | 0.17 | 26.02 | 24.81 | 28.73 | 40.06 |
| fir sse4.1 | 0.17 | 11.84 | 12.46 | 14.53 | 17.99 |
| fir avx2 | 0.17 | 8.09 | 8.58 | 9.80 | 12.38 |
| libvips | 6.44 | 72.80 | 117.74 | 205.61 | 293.02 |
| fir rust | 0.17 | 11.10 | 13.06 | 17.49 | 24.11 |
| fir sse4.1 | 0.17 | 6.24 | 7.32 | 9.88 | 13.85 |
| fir avx2 | 0.17 | 4.13 | 4.74 | 6.44 | 9.22 |
<!-- bench_compare_la end -->
<!-- bench_compare_rgb16 start -->
@@ -124,12 +124,12 @@ Pipeline:
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
| image | 28.82 | - | 91.33 | 160.30 | 214.84 |
| resize | - | 27.10 | 49.64 | 95.91 | 141.28 |
| libvips | 16.00 | 62.78 | 54.24 | 102.54 | 125.42 |
| fir rust | 0.35 | 25.55 | 42.63 | 76.47 | 108.68 |
| fir sse4.1 | 0.35 | 16.07 | 23.03 | 36.73 | 51.97 |
| fir avx2 | 0.34 | 13.98 | 19.57 | 30.88 | 38.30 |
| image | 29.30 | - | 83.82 | 135.59 | 187.38 |
| resize | - | 26.35 | 50.23 | 96.71 | 143.53 |
| libvips | 16.00 | 63.15 | 54.30 | 101.65 | 125.21 |
| fir rust | 0.35 | 24.72 | 41.75 | 75.50 | 107.55 |
| fir sse4.1 | 0.35 | 16.26 | 23.22 | 36.94 | 52.17 |
| fir avx2 | 0.35 | 14.07 | 19.69 | 30.93 | 38.58 |
<!-- bench_compare_rgb16 end -->
<!-- bench_compare_rgba16 start -->
@@ -146,11 +146,11 @@ Pipeline:
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|------------|:-------:|:------:|:--------:|:-------:|:--------:|
| resize | - | 43.49 | 83.75 | 144.57 | 206.78 |
| libvips | 22.61 | 129.33 | 205.19 | 364.93 | 535.68 |
| fir rust | 0.38 | 61.06 | 79.63 | 118.03 | 159.90 |
| fir sse4.1 | 0.38 | 31.79 | 42.33 | 64.26 | 86.26 |
| fir avx2 | 0.38 | 20.43 | 25.94 | 36.59 | 47.92 |
| resize | - | 43.62 | 84.14 | 144.98 | 207.50 |
| libvips | 22.68 | 128.33 | 205.29 | 364.70 | 535.40 |
| fir rust | 0.38 | 60.18 | 78.69 | 116.18 | 155.10 |
| fir sse4.1 | 0.38 | 31.96 | 42.41 | 64.24 | 86.47 |
| fir avx2 | 0.38 | 20.40 | 26.04 | 36.97 | 48.46 |
<!-- bench_compare_rgba16 end -->
<!-- bench_compare_l16 start -->
@@ -166,12 +166,12 @@ Pipeline:
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
| image | 26.23 | - | 58.57 | 86.74 | 115.76 |
| resize | - | 9.94 | 16.08 | 33.55 | 58.73 |
| libvips | 7.89 | 26.07 | 21.67 | 36.36 | 46.28 |
| fir rust | 0.17 | 14.22 | 19.82 | 29.13 | 41.53 |
| fir sse4.1 | 0.17 | 5.57 | 7.73 | 13.10 | 19.02 |
| fir avx2 | 0.17 | 5.67 | 6.47 | 8.78 | 13.78 |
| image | 26.44 | - | 59.57 | 88.45 | 118.20 |
| resize | - | 10.07 | 16.27 | 33.82 | 59.02 |
| libvips | 7.71 | 26.87 | 21.95 | 37.06 | 46.97 |
| fir rust | 0.18 | 14.41 | 21.24 | 30.19 | 40.49 |
| fir sse4.1 | 0.18 | 5.72 | 7.92 | 13.43 | 19.29 |
| fir avx2 | 0.18 | 5.77 | 6.68 | 8.97 | 13.78 |
<!-- bench_compare_l16 end -->
<!-- bench_compare_la16 start -->
@@ -190,8 +190,8 @@ Pipeline:
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
| libvips | 12.51 | 79.89 | 133.82 | 231.42 | 327.75 |
| fir rust | 0.19 | 26.81 | 35.29 | 53.20 | 73.16 |
| fir sse4.1 | 0.19 | 15.29 | 21.43 | 33.53 | 46.13 |
| fir avx2 | 0.19 | 11.59 | 14.76 | 21.73 | 29.05 |
| libvips | 12.54 | 79.43 | 133.76 | 231.51 | 328.53 |
| fir rust | 0.19 | 26.33 | 34.62 | 52.63 | 72.48 |
| fir sse4.1 | 0.19 | 15.64 | 21.74 | 33.77 | 46.25 |
| fir avx2 | 0.19 | 11.71 | 14.89 | 21.91 | 29.22 |
<!-- bench_compare_la16 end -->
+12 -14
View File
@@ -35,14 +35,13 @@ pub(crate) fn multiply_alpha_row_inplace(row: &mut [U8x4]) {
#[inline(always)]
fn multiply_alpha_pixel(mut pixel: U8x4) -> U8x4 {
let components: [u8; 4] = pixel.0.to_le_bytes();
let alpha = components[3];
pixel.0 = u32::from_le_bytes([
mul_div_255(components[0], alpha),
mul_div_255(components[1], alpha),
mul_div_255(components[2], alpha),
let alpha = pixel.0[3];
pixel.0 = [
mul_div_255(pixel.0[0], alpha),
mul_div_255(pixel.0[1], alpha),
mul_div_255(pixel.0[2], alpha),
alpha,
]);
];
pixel
}
@@ -76,14 +75,13 @@ pub(crate) fn divide_alpha_row(src_row: &[U8x4], dst_row: &mut [U8x4]) {
#[inline(always)]
fn divide_alpha_pixel(mut pixel: U8x4) -> U8x4 {
let components: [u8; 4] = pixel.0.to_le_bytes();
let alpha = components[3];
let alpha = pixel.0[3];
let recip_alpha = RECIP_ALPHA[alpha as usize];
pixel.0 = u32::from_le_bytes([
div_and_clip(components[0], recip_alpha),
div_and_clip(components[1], recip_alpha),
div_and_clip(components[2], recip_alpha),
pixel.0 = [
div_and_clip(pixel.0[0], recip_alpha),
div_and_clip(pixel.0[1], recip_alpha),
div_and_clip(pixel.0[2], recip_alpha),
alpha,
]);
];
pixel
}
+4 -4
View File
@@ -137,13 +137,13 @@ pub(crate) unsafe fn divide_alpha_row(src_row: &[U8x4], dst_row: &mut [U8x4]) {
if !src_remainder.is_empty() {
let dst_reminder = dst_chunks.into_remainder();
let mut src_buffer = [U8x4::new(0); 4];
let mut src_buffer = [U8x4::new([0; 4]); 4];
src_buffer
.iter_mut()
.zip(src_remainder)
.for_each(|(d, s)| *d = *s);
let mut dst_buffer = [U8x4::new(0); 4];
let mut dst_buffer = [U8x4::new([0; 4]); 4];
let src_pixels = _mm_loadu_si128(src_buffer.as_ptr() as *const __m128i);
let dst_pixels = divide_alpha_4_pixels(src_pixels);
_mm_storeu_si128(dst_buffer.as_mut_ptr() as *mut __m128i, dst_pixels);
@@ -174,13 +174,13 @@ pub(crate) unsafe fn divide_alpha_row_inplace(row: &mut [U8x4]) {
let tail = chunks.into_remainder();
if !tail.is_empty() {
let mut src_buffer = [U8x4::new(0); 4];
let mut src_buffer = [U8x4::new([0; 4]); 4];
src_buffer
.iter_mut()
.zip(tail.iter())
.for_each(|(d, s)| *d = *s);
let mut dst_buffer = [U8x4::new(0); 4];
let mut dst_buffer = [U8x4::new([0; 4]); 4];
let src_pixels = _mm_loadu_si128(src_buffer.as_ptr() as *const __m128i);
let dst_pixels = divide_alpha_4_pixels(src_pixels);
_mm_storeu_si128(dst_buffer.as_mut_ptr() as *mut __m128i, dst_pixels);
+4 -4
View File
@@ -12,19 +12,19 @@ pub(crate) fn horiz_convolution(
let normalizer = optimisations::Normalizer32::new(coeffs);
let precision = normalizer.precision();
let coefficients_chunks = normalizer.normalized_chunks();
let initial: i64 = 1 << (precision - 1);
let initial = 1i64 << (precision - 1);
let src_rows = src_image.iter_rows(offset);
let dst_rows = dst_image.iter_rows_mut();
for (dst_row, src_row) in dst_rows.zip(src_rows) {
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
let first_x_src = coeffs_chunk.start as usize;
let mut sum = initial;
let mut ss = initial;
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
for (&k, src_pixel) in coeffs_chunk.values.iter().zip(src_pixels) {
sum += src_pixel.0 as i64 * (k as i64);
ss += src_pixel.0 as i64 * (k as i64);
}
dst_pixel.0 = normalizer.clip(sum);
dst_pixel.0 = normalizer.clip(ss);
}
}
}
+1 -1
View File
@@ -12,7 +12,7 @@ pub(crate) fn horiz_convolution(
let normalizer = optimisations::Normalizer32::new(coeffs);
let precision = normalizer.precision();
let coefficients_chunks = normalizer.normalized_chunks();
let initial: i64 = 1 << (precision - 1);
let initial = 1i64 << (precision - 1);
let src_rows = src_image.iter_rows(offset);
let dst_rows = dst_image.iter_rows_mut();
+2 -4
View File
@@ -12,18 +12,16 @@ pub(crate) fn horiz_convolution(
let normalizer = optimisations::Normalizer16::new(coeffs);
let precision = normalizer.precision();
let coefficients_chunks = normalizer.normalized_chunks();
let initial = 1 << (precision - 1);
let initial = 1i32 << (precision - 1);
let src_rows = src_image.iter_rows(offset);
let dst_rows = dst_image.iter_rows_mut();
for (dst_row, src_row) in dst_rows.zip(src_rows) {
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
let first_x_src = coeffs_chunk.start as usize;
let ks = coeffs_chunk.values;
let mut ss = initial;
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
for (&k, &src_pixel) in ks.iter().zip(src_pixels) {
for (&k, &src_pixel) in coeffs_chunk.values.iter().zip(src_pixels) {
ss += src_pixel.0 as i32 * (k as i32);
}
dst_pixel.0 = unsafe { normalizer.clip(ss) };
+1 -1
View File
@@ -12,7 +12,7 @@ pub(crate) fn horiz_convolution(
let normalizer = optimisations::Normalizer16::new(coeffs);
let precision = normalizer.precision();
let coefficients_chunks = normalizer.normalized_chunks();
let initial = 1 << (precision - 1);
let initial = 1i32 << (precision - 1);
let src_rows = src_image.iter_rows(offset);
let dst_rows = dst_image.iter_rows_mut();
+8 -5
View File
@@ -2,6 +2,7 @@ use crate::convolution::{optimisations, Coefficients};
use crate::pixels::U8x4;
use crate::{ImageView, ImageViewMut};
#[inline(always)]
pub(crate) fn horiz_convolution(
src_image: &ImageView<U8x4>,
dst_image: &mut ImageViewMut<U8x4>,
@@ -18,16 +19,18 @@ pub(crate) fn horiz_convolution(
for (dst_row, src_row) in dst_rows.zip(src_rows) {
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
let first_x_src = coeffs_chunk.start as usize;
let ks = coeffs_chunk.values;
let mut ss = [initial; 4];
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
for (&k, &src_pixel) in ks.iter().zip(src_pixels) {
let components: [u8; 4] = src_pixel.0.to_le_bytes();
for (&k, &src_pixel) in coeffs_chunk.values.iter().zip(src_pixels) {
for (i, s) in ss.iter_mut().enumerate() {
*s += components[i] as i32 * (k as i32);
*s += src_pixel.0[i] as i32 * (k as i32);
}
}
dst_pixel.0 = u32::from_le_bytes(ss.map(|v| unsafe { normalizer.clip(v) }));
for (i, s) in ss.iter().copied().enumerate() {
dst_pixel.0[i] = unsafe { normalizer.clip(s) };
}
}
}
}
+64 -5
View File
@@ -1,19 +1,22 @@
use crate::convolution::{optimisations, Coefficients};
use crate::pixels::PixelExt;
use crate::utils::foreach_with_pre_reading;
use crate::{ImageView, ImageViewMut};
#[inline(always)]
pub(crate) fn vert_convolution<T: PixelExt<Component = u16>>(
pub(crate) fn vert_convolution<T>(
src_image: &ImageView<T>,
dst_image: &mut ImageViewMut<T>,
offset: u32,
coeffs: Coefficients,
) {
) where
T: PixelExt<Component = u16>,
{
let normalizer = optimisations::Normalizer32::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let precision = normalizer.precision();
let initial: i64 = 1 << (precision - 1);
let x_src = offset as usize * T::count_of_components();
let src_x_initial = offset as usize * T::count_of_components();
let dst_rows = dst_image.iter_rows_mut();
let coeffs_chunks_iter = coefficients_chunks.into_iter();
@@ -21,16 +24,30 @@ pub(crate) fn vert_convolution<T: PixelExt<Component = u16>>(
let first_y_src = coeffs_chunk.start;
let ks = coeffs_chunk.values;
let dst_components = T::components_mut(dst_row);
let mut x_src = src_x_initial;
convolution_by_u16(
let (_, dst_chunks, tail) = unsafe { dst_components.align_to_mut::<[u16; 16]>() };
x_src = convolution_by_chunks(
src_image,
&normalizer,
initial,
dst_components,
dst_chunks,
x_src,
first_y_src,
ks,
);
if !tail.is_empty() {
convolution_by_u16(
src_image,
&normalizer,
initial,
tail,
x_src,
first_y_src,
ks,
);
}
}
}
@@ -59,3 +76,45 @@ pub(crate) fn convolution_by_u16<T: PixelExt<Component = u16>>(
}
x_src
}
#[inline(always)]
fn convolution_by_chunks<T, const CHUNK_SIZE: usize>(
src_image: &ImageView<T>,
normalizer: &optimisations::Normalizer32,
initial: i64,
dst_chunks: &mut [[u16; CHUNK_SIZE]],
mut x_src: usize,
first_y_src: u32,
ks: &[i32],
) -> usize
where
T: PixelExt<Component = u16>,
{
for dst_chunk in dst_chunks {
let mut ss = [initial; CHUNK_SIZE];
let src_rows = src_image.iter_rows(first_y_src);
foreach_with_pre_reading(
ks.iter().zip(src_rows),
|(&k, src_row)| {
let src_ptr = src_row.as_ptr() as *const u16;
let src_chunk = unsafe {
let ptr = src_ptr.add(x_src) as *const [u16; CHUNK_SIZE];
ptr.read_unaligned()
};
(src_chunk, k)
},
|(src_chunk, k)| {
for (s, c) in ss.iter_mut().zip(src_chunk) {
*s += c as i64 * (k as i64);
}
},
);
for (i, s) in ss.iter().copied().enumerate() {
dst_chunk[i] = normalizer.clip(s);
}
x_src += CHUNK_SIZE;
}
x_src
}
+92 -29
View File
@@ -1,5 +1,6 @@
use crate::convolution::{optimisations, Coefficients};
use crate::pixels::PixelExt;
use crate::utils::foreach_with_pre_reading;
use crate::{ImageView, ImageViewMut};
#[inline(always)]
@@ -24,40 +25,60 @@ pub(crate) fn vert_convolution<T>(
let ks = coeffs_chunk.values;
let mut x_src = src_x_initial;
let dst_components = T::components_mut(dst_row);
let (head, dst_chunks, tail) = unsafe { dst_components.align_to_mut::<u32>() };
if !head.is_empty() {
x_src = convolution_by_u8(
src_image,
&normalizer,
initial,
head,
x_src,
first_y_src,
ks,
);
let (_, dst_chunks, tail) = unsafe { dst_components.align_to_mut::<[u8; 32]>() };
x_src = convolution_by_chunks(
src_image,
&normalizer,
initial,
dst_chunks,
x_src,
first_y_src,
ks,
);
if tail.is_empty() {
continue;
}
// Convolution by u8x4
for dst_chunk in dst_chunks {
let mut ss = [initial; 4];
let src_rows = src_image.iter_rows(first_y_src);
for (&k, src_row) in ks.iter().zip(src_rows) {
let src_ptr = src_row.as_ptr() as *const u8;
let src_chunk = unsafe {
let ptr = src_ptr.add(x_src) as *const u32;
ptr.read_unaligned()
};
let components: [u8; 4] = src_chunk.to_le_bytes();
for (s, c) in ss.iter_mut().zip(components) {
*s += c as i32 * (k as i32);
}
}
*dst_chunk = u32::from_le_bytes(ss.map(|v| unsafe { normalizer.clip(v) }));
x_src += 4;
let (_, dst_chunks, tail) = unsafe { tail.align_to_mut::<[u8; 16]>() };
x_src = convolution_by_chunks(
src_image,
&normalizer,
initial,
dst_chunks,
x_src,
first_y_src,
ks,
);
if tail.is_empty() {
continue;
}
let (_, dst_chunks, tail) = unsafe { tail.align_to_mut::<[u8; 8]>() };
x_src = convolution_by_chunks(
src_image,
&normalizer,
initial,
dst_chunks,
x_src,
first_y_src,
ks,
);
if tail.is_empty() {
continue;
}
let (_, dst_chunks, tail) = unsafe { tail.align_to_mut::<[u8; 4]>() };
x_src = convolution_by_chunks(
src_image,
&normalizer,
initial,
dst_chunks,
x_src,
first_y_src,
ks,
);
if !tail.is_empty() {
convolution_by_u8(
src_image,
@@ -98,3 +119,45 @@ where
}
x_src
}
#[inline(always)]
fn convolution_by_chunks<T, const CHUNK_SIZE: usize>(
src_image: &ImageView<T>,
normalizer: &optimisations::Normalizer16,
initial: i32,
dst_chunks: &mut [[u8; CHUNK_SIZE]],
mut x_src: usize,
first_y_src: u32,
ks: &[i16],
) -> usize
where
T: PixelExt<Component = u8>,
{
for dst_chunk in dst_chunks {
let mut ss = [initial; CHUNK_SIZE];
let src_rows = src_image.iter_rows(first_y_src);
foreach_with_pre_reading(
ks.iter().zip(src_rows),
|(&k, src_row)| {
let src_ptr = src_row.as_ptr() as *const u8;
let src_chunk = unsafe {
let ptr = src_ptr.add(x_src) as *const [u8; CHUNK_SIZE];
ptr.read_unaligned()
};
(src_chunk, k)
},
|(src_chunk, k)| {
for (s, c) in ss.iter_mut().zip(src_chunk) {
*s += c as i32 * (k as i32);
}
},
);
for (i, s) in ss.iter().copied().enumerate() {
dst_chunk[i] = unsafe { normalizer.clip(s) };
}
x_src += CHUNK_SIZE;
}
x_src
}
+1 -1
View File
@@ -230,7 +230,7 @@ pixel_struct!(
);
pixel_struct!(
U8x4,
u32,
[u8; 4],
u8,
4,
PixelType::U8x4,
+61 -27
View File
@@ -330,34 +330,68 @@ fn resample_convolution<P>(
});
match (horiz_coeffs, vert_coeffs) {
(Some(horiz_coeffs), Some(mut vert_coeffs)) => {
let y_first = vert_coeffs.bounds[0].start;
// Last used row in the source image
let last_y_bound = vert_coeffs.bounds.last().unwrap();
let y_last = last_y_bound.start + last_y_bound.size;
let temp_height = NonZeroU32::new(y_last - y_first).unwrap();
let mut temp_image = get_temp_image_from_buffer(temp_buffer, dst_width, temp_height);
let mut tmp_dst_view = temp_image.dst_view();
P::horiz_convolution(
src_image,
&mut tmp_dst_view,
y_first,
horiz_coeffs,
cpu_extensions,
);
(Some(mut horiz_coeffs), Some(mut vert_coeffs)) => {
if P::count_of_component_values() > 256 {
let y_first = vert_coeffs.bounds[0].start;
// Last used row in the source image
let last_y_bound = vert_coeffs.bounds.last().unwrap();
let y_last = last_y_bound.start + last_y_bound.size;
let temp_height = NonZeroU32::new(y_last - y_first).unwrap();
let mut temp_image =
get_temp_image_from_buffer(temp_buffer, dst_width, temp_height);
let mut tmp_dst_view = temp_image.dst_view();
P::horiz_convolution(
src_image,
&mut tmp_dst_view,
y_first,
horiz_coeffs,
cpu_extensions,
);
// Shift bounds for vertical pass
vert_coeffs
.bounds
.iter_mut()
.for_each(|b| b.start -= y_first);
P::vert_convolution(
&tmp_dst_view.into(),
dst_image,
0,
vert_coeffs,
cpu_extensions,
);
// Shift bounds for vertical pass
vert_coeffs
.bounds
.iter_mut()
.for_each(|b| b.start -= y_first);
P::vert_convolution(
&tmp_dst_view.into(),
dst_image,
0,
vert_coeffs,
cpu_extensions,
);
} else {
let y_first = vert_coeffs.bounds[0].start;
let x_first = horiz_coeffs.bounds[0].start;
// Last used col in the source image
let last_x_bound = horiz_coeffs.bounds.last().unwrap();
let x_last = last_x_bound.start + last_x_bound.size;
let temp_width = NonZeroU32::new(x_last - x_first).unwrap();
let mut temp_image =
get_temp_image_from_buffer(temp_buffer, temp_width, dst_height);
let mut tmp_dst_view = temp_image.dst_view();
P::vert_convolution(
src_image,
&mut tmp_dst_view,
y_first,
vert_coeffs,
cpu_extensions,
);
// Shift bounds for horizontal pass
horiz_coeffs
.bounds
.iter_mut()
.for_each(|b| b.start -= x_first);
P::horiz_convolution(
&tmp_dst_view.into(),
dst_image,
0,
horiz_coeffs,
cpu_extensions,
);
}
}
(Some(horiz_coeffs), None) => {
P::horiz_convolution(
+50 -4
View File
@@ -55,11 +55,29 @@ pub trait PixelTestingExt: PixelExt {
.unwrap()
}
fn load_big_square_image() -> DynamicImage {
ImageReader::open("./data/nasa-4019x4019.png")
.unwrap()
.decode()
.unwrap()
}
fn load_big_src_image() -> Image<'static> {
let img = Self::load_big_image();
Image::from_vec_u8(
NonZeroU32::new(img.width()).unwrap(),
NonZeroU32::new(img.height()).unwrap(),
nonzero(img.width()),
nonzero(img.height()),
Self::img_into_bytes(img),
Self::pixel_type(),
)
.unwrap()
}
fn load_big_square_src_image() -> Image<'static> {
let img = Self::load_big_square_image();
Image::from_vec_u8(
nonzero(img.width()),
nonzero(img.height()),
Self::img_into_bytes(img),
Self::pixel_type(),
)
@@ -76,8 +94,8 @@ pub trait PixelTestingExt: PixelExt {
fn load_small_src_image() -> Image<'static> {
let img = Self::load_small_image();
Image::from_vec_u8(
NonZeroU32::new(img.width()).unwrap(),
NonZeroU32::new(img.height()).unwrap(),
nonzero(img.width()),
nonzero(img.height()),
Self::img_into_bytes(img),
Self::pixel_type(),
)
@@ -101,6 +119,13 @@ impl PixelTestingExt for U8x2 {
.unwrap()
}
fn load_big_square_image() -> DynamicImage {
ImageReader::open("./data/nasa-4019x4019-rgba.png")
.unwrap()
.decode()
.unwrap()
}
fn load_small_image() -> DynamicImage {
ImageReader::open("./data/nasa-852x567-rgba.png")
.unwrap()
@@ -127,6 +152,13 @@ impl PixelTestingExt for U8x4 {
.unwrap()
}
fn load_big_square_image() -> DynamicImage {
ImageReader::open("./data/nasa-4019x4019-rgba.png")
.unwrap()
.decode()
.unwrap()
}
fn load_small_image() -> DynamicImage {
ImageReader::open("./data/nasa-852x567-rgba.png")
.unwrap()
@@ -164,6 +196,13 @@ impl PixelTestingExt for U16x2 {
.unwrap()
}
fn load_big_square_image() -> DynamicImage {
ImageReader::open("./data/nasa-4019x4019-rgba.png")
.unwrap()
.decode()
.unwrap()
}
fn load_small_image() -> DynamicImage {
ImageReader::open("./data/nasa-852x567-rgba.png")
.unwrap()
@@ -198,6 +237,13 @@ impl PixelTestingExt for U16x4 {
.unwrap()
}
fn load_big_square_image() -> DynamicImage {
ImageReader::open("./data/nasa-4019x4019-rgba.png")
.unwrap()
.decode()
.unwrap()
}
fn load_small_image() -> DynamicImage {
ImageReader::open("./data/nasa-852x567-rgba.png")
.unwrap()
+1 -1
View File
@@ -120,7 +120,7 @@ mod u8x4 {
use fast_image_resize::pixels::U8x4;
const fn p4(r: u8, g: u8, b: u8, a: u8) -> U8x4 {
U8x4::new(u32::from_le_bytes([r, g, b, a]))
U8x4::new([r, g, b, a])
}
#[cfg(test)]