mirror of
https://github.com/Cykooz/fast_image_resize.git
synced 2026-10-08 01:11:09 +00:00
- Improved vertical vor f32 pixels.
- Updated f32 benchmarks for x86_64 arch.
This commit is contained in:
+20
-20
@@ -226,12 +226,12 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
|
||||
| image | 24.55 | - | 49.54 | 78.44 | 104.51 |
|
||||
| resize | 5.03 | 8.64 | 13.85 | 30.14 | 45.81 |
|
||||
| libvips | 5.96 | 25.74 | 21.52 | 42.24 | 71.71 |
|
||||
| fir rust | 0.20 | 9.87 | 14.93 | 29.66 | 51.71 |
|
||||
| fir sse4.1 | 0.20 | 5.06 | 7.31 | 11.59 | 16.89 |
|
||||
| fir avx2 | 0.20 | 4.75 | 5.55 | 7.13 | 10.69 |
|
||||
| image | 24.21 | - | 48.61 | 79.04 | 104.73 |
|
||||
| resize | 5.26 | 8.83 | 13.71 | 30.19 | 45.85 |
|
||||
| libvips | 6.21 | 27.94 | 19.81 | 43.10 | 69.32 |
|
||||
| fir rust | 0.19 | 8.32 | 12.65 | 26.87 | 41.19 |
|
||||
| fir sse4.1 | 0.19 | 5.28 | 7.28 | 11.67 | 17.07 |
|
||||
| fir avx2 | 0.19 | 4.68 | 5.46 | 7.11 | 10.73 |
|
||||
|
||||
<!-- bench_compare_l32f end -->
|
||||
|
||||
@@ -252,10 +252,10 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
|
||||
| libvips | 11.58 | 70.05 | 101.70 | 176.48 | 252.74 |
|
||||
| fir rust | 0.38 | 22.75 | 30.35 | 49.41 | 71.81 |
|
||||
| fir sse4.1 | 0.38 | 17.63 | 21.97 | 31.34 | 41.09 |
|
||||
| fir avx2 | 0.38 | 15.91 | 18.46 | 23.27 | 28.78 |
|
||||
| libvips | 13.72 | 72.74 | 100.78 | 176.30 | 252.88 |
|
||||
| fir rust | 0.35 | 21.99 | 29.09 | 47.76 | 70.52 |
|
||||
| fir sse4.1 | 0.35 | 16.25 | 20.94 | 30.33 | 40.11 |
|
||||
| fir avx2 | 0.35 | 15.73 | 17.36 | 22.92 | 27.79 |
|
||||
|
||||
<!-- bench_compare_la32f end -->
|
||||
|
||||
@@ -273,12 +273,12 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|------------|:-------:|:-----:|:--------:|:-------:|:--------:|
|
||||
| image | 26.32 | - | 62.99 | 104.89 | 147.02 |
|
||||
| resize | 8.97 | 16.30 | 24.52 | 48.25 | 72.16 |
|
||||
| libvips | 11.75 | 60.36 | 52.63 | 113.20 | 197.67 |
|
||||
| fir rust | 0.83 | 16.39 | 26.95 | 50.36 | 75.58 |
|
||||
| fir sse4.1 | 0.83 | 12.65 | 19.17 | 32.15 | 46.64 |
|
||||
| fir avx2 | 0.83 | 11.03 | 14.10 | 20.86 | 29.43 |
|
||||
| image | 26.67 | - | 64.09 | 108.10 | 155.20 |
|
||||
| resize | 9.30 | 16.45 | 24.68 | 48.28 | 72.52 |
|
||||
| libvips | 13.46 | 62.91 | 52.79 | 114.29 | 199.08 |
|
||||
| fir rust | 0.86 | 16.49 | 27.10 | 51.23 | 76.20 |
|
||||
| fir sse4.1 | 0.86 | 12.38 | 19.56 | 32.72 | 47.73 |
|
||||
| fir avx2 | 0.86 | 11.23 | 14.21 | 20.39 | 29.32 |
|
||||
|
||||
<!-- bench_compare_rgb32f end -->
|
||||
|
||||
@@ -300,9 +300,9 @@ Pipeline:
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|------------|:-------:|:------:|:--------:|:-------:|:--------:|
|
||||
| libvips | 23.22 | 111.26 | 140.16 | 249.97 | 381.64 |
|
||||
| fir rust | 1.01 | 35.88 | 45.64 | 70.56 | 93.29 |
|
||||
| fir sse4.1 | 1.01 | 32.41 | 39.49 | 57.88 | 77.38 |
|
||||
| fir avx2 | 1.01 | 28.39 | 31.49 | 40.73 | 49.47 |
|
||||
| libvips | 24.44 | 113.81 | 138.06 | 249.74 | 379.76 |
|
||||
| fir rust | 1.01 | 36.20 | 46.05 | 71.26 | 93.95 |
|
||||
| fir sse4.1 | 1.01 | 31.81 | 39.68 | 58.07 | 77.59 |
|
||||
| fir avx2 | 1.01 | 28.64 | 30.94 | 40.70 | 49.71 |
|
||||
|
||||
<!-- bench_compare_rgba32f end -->
|
||||
|
||||
@@ -14,12 +14,35 @@ pub(crate) fn horiz_convolution(
|
||||
for (dst_row, src_row) in dst_rows.zip(src_rows) {
|
||||
for (dst_pixel, coeffs_chunk) in dst_row.iter_mut().zip(&coefficients_chunks) {
|
||||
let first_x_src = coeffs_chunk.start as usize;
|
||||
let end_x_src = first_x_src + coeffs_chunk.values.len();
|
||||
let mut ss = 0.;
|
||||
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
|
||||
for (&k, &pixel) in coeffs_chunk.values.iter().zip(src_pixels) {
|
||||
let mut src_pixels = unsafe { src_row.get_unchecked(first_x_src..end_x_src) };
|
||||
let mut coefs = coeffs_chunk.values;
|
||||
|
||||
(coefs, src_pixels) = convolution_by_chunks::<8>(coefs, src_pixels, &mut ss);
|
||||
|
||||
for (&k, &pixel) in coefs.iter().zip(src_pixels) {
|
||||
ss += pixel.0 as f64 * k;
|
||||
}
|
||||
dst_pixel.0 = ss as f32;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
fn convolution_by_chunks<'a, 'b, const CHUNK_SIZE: usize>(
|
||||
coefs: &'a [f64],
|
||||
src_pixels: &'b [F32],
|
||||
ss: &mut f64,
|
||||
) -> (&'a [f64], &'b [F32]) {
|
||||
let coef_chunks = coefs.chunks_exact(CHUNK_SIZE);
|
||||
let coefs = coef_chunks.remainder();
|
||||
let pixel_chunks = src_pixels.chunks_exact(CHUNK_SIZE);
|
||||
let src_pixels = pixel_chunks.remainder();
|
||||
for (ks, pixels) in coef_chunks.zip(pixel_chunks) {
|
||||
for (&k, &pixel) in ks.iter().zip(pixels) {
|
||||
*ss += pixel.0 as f64 * k;
|
||||
}
|
||||
}
|
||||
(coefs, src_pixels)
|
||||
}
|
||||
|
||||
@@ -20,14 +20,39 @@ pub(crate) fn vert_convolution<T>(
|
||||
for (coeffs_chunk, dst_row) in coeffs_chunks_iter.zip(dst_rows) {
|
||||
let first_y_src = coeffs_chunk.start;
|
||||
let ks = coeffs_chunk.values;
|
||||
let dst_components = T::components_mut(dst_row);
|
||||
let mut dst_components = T::components_mut(dst_row);
|
||||
let mut x_src = src_x_initial;
|
||||
|
||||
let (_, dst_chunks, tail) = unsafe { dst_components.align_to_mut::<[f32; 8]>() };
|
||||
x_src = convolution_by_chunks(src_view, dst_chunks, x_src, first_y_src, ks);
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
{
|
||||
(dst_components, x_src) =
|
||||
convolution_by_chunks::<_, 16>(src_view, dst_components, x_src, first_y_src, ks);
|
||||
}
|
||||
|
||||
if !tail.is_empty() {
|
||||
convolution_by_f32(src_view, tail, x_src, first_y_src, ks);
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
{
|
||||
if !dst_components.is_empty() {
|
||||
(dst_components, x_src) =
|
||||
convolution_by_chunks::<_, 8>(src_view, dst_components, x_src, first_y_src, ks);
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
{
|
||||
if !dst_components.is_empty() {
|
||||
(dst_components, x_src) =
|
||||
crate::convolution::vertical_f32::native::convolution_by_chunks::<_, 4>(
|
||||
src_view,
|
||||
dst_components,
|
||||
x_src,
|
||||
first_y_src,
|
||||
ks,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
if !dst_components.is_empty() {
|
||||
convolution_by_f32(src_view, dst_components, x_src, first_y_src, ks);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -57,17 +82,19 @@ pub(crate) fn convolution_by_f32<T: InnerPixel<Component = f32>>(
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
fn convolution_by_chunks<T, const CHUNK_SIZE: usize>(
|
||||
fn convolution_by_chunks<'a, T, const CHUNK_SIZE: usize>(
|
||||
src_view: &impl ImageView<Pixel = T>,
|
||||
dst_chunks: &mut [[f32; CHUNK_SIZE]],
|
||||
dst_components: &'a mut [f32],
|
||||
mut x_src: usize,
|
||||
first_y_src: u32,
|
||||
ks: &[f64],
|
||||
) -> usize
|
||||
) -> (&'a mut [f32], usize)
|
||||
where
|
||||
T: InnerPixel<Component = f32>,
|
||||
{
|
||||
for dst_chunk in dst_chunks {
|
||||
let mut dst_chunks = dst_components.chunks_exact_mut(CHUNK_SIZE);
|
||||
|
||||
for dst_chunk in &mut dst_chunks {
|
||||
let mut ss = [0.; CHUNK_SIZE];
|
||||
let src_rows = src_view.iter_rows(first_y_src);
|
||||
|
||||
@@ -93,5 +120,6 @@ where
|
||||
}
|
||||
x_src += CHUNK_SIZE;
|
||||
}
|
||||
x_src
|
||||
|
||||
(dst_chunks.into_remainder(), x_src)
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user