Fixed a "divide by zero" error in case of using multithreading to resize images with particular sizes (#55).

This commit is contained in:
Kirill Kuzminykh
2025-08-29 22:33:03 +03:00
parent ae5139df07
commit f0d54cb009
4 changed files with 96 additions and 33 deletions
+7
View File
@@ -1,3 +1,10 @@
## [Unreleased] - ReleaseDate
### Fixed
- Fixed a "divide by zero" error in case of using multithreading to resize images
with particular sizes ([#55](https://github.com/Cykooz/fast_image_resize/issues/55)).
## [5.2.1] - 2025-07-27
### Changed
+9 -3
View File
@@ -9,7 +9,7 @@ name = "fast_image_resize"
version = "5.2.1"
authors = ["Kirill Kuzminykh <[email protected]>"]
edition = "2021"
rust-version = " 1.87.0"
rust-version = "1.87.0"
license = "MIT OR Apache-2.0"
description = "Library for fast image resizing with using of SIMD instructions"
readme = "README.md"
@@ -36,7 +36,7 @@ rayon = { version = "1.10", optional = true }
## [DynamicImage](https://docs.rs/image/latest/image/enum.DynamicImage.html)
## type from the `image` crate.
image = ["dep:image", "dep:bytemuck"]
## This feature enables image processing in `rayon` thread pool.
## This feature enables image processing in the ` rayon ` thread pool.
rayon = ["dep:rayon", "resize/rayon", "image/rayon"]
for_testing = ["image", "image/png"]
only_u8x4 = [] # This can be used to experiment with the crate's code.
@@ -181,5 +181,11 @@ name = "bench_color_mapper"
harness = false
# Header of next release in CHANGELOG.md:
[[bench]]
name = "bench_threads"
harness = false
# Header of the next release in CHANGELOG.md:
# ## [Unreleased] - ReleaseDate
+60
View File
@@ -0,0 +1,60 @@
use fast_image_resize::images::Image;
use fast_image_resize::pixels::U8x3;
use fast_image_resize::{CpuExtensions, FilterType, ResizeAlg, ResizeOptions, Resizer};
use utils::testing::PixelTestingExt;
use crate::utils::{bench, build_md_table, BenchGroup};
mod utils;
pub fn fir_resize<P: PixelTestingExt>(bench_group: &mut BenchGroup, use_alpha: bool) {
let src_sizes: Vec<u32> = vec![2, 10, 100, 500, 1000, 5000, 10000, 65536];
let mut resizer = Resizer::new();
unsafe {
resizer.set_cpu_extensions(CpuExtensions::None);
}
let resize_options = ResizeOptions::new()
.resize_alg(ResizeAlg::Convolution(FilterType::Bilinear))
.use_alpha(false);
for &src_width in &src_sizes {
for &src_height in &src_sizes {
let dst_width = src_width / 2;
let src_image = Image::new(src_width, src_height, P::pixel_type());
let mut dst_image = Image::new(dst_width, src_height, src_image.pixel_type());
for thread_count in 1..=8 {
bench(
bench_group,
10,
format!("{src_width}x{src_height}"),
format!("{thread_count}"),
|bencher| {
let mut builder = rayon::ThreadPoolBuilder::new();
builder = builder.num_threads(thread_count);
let pool = builder.build().unwrap();
pool.install(|| {
bencher.iter(|| {
resizer
.resize(&src_image, &mut dst_image, &resize_options)
.unwrap()
})
})
},
);
}
}
}
}
pub fn bench_threads(bench_group: &mut BenchGroup) {
type P = U8x3;
fir_resize::<P>(bench_group, false);
}
fn main() {
let res = utils::run_bench(bench_threads, "Compare resize by width images with threads");
let md_table = build_md_table(&res);
println!("{}", md_table);
}
+20 -30
View File
@@ -1,8 +1,10 @@
use crate::pixels::InnerPixel;
use crate::{ImageView, ImageViewMut};
use std::num::NonZeroU32;
use rayon::current_num_threads;
use rayon::prelude::*;
use std::num::NonZeroU32;
use crate::pixels::InnerPixel;
use crate::{ImageView, ImageViewMut};
#[inline]
pub(crate) fn split_h_two_images_for_threading<'a, P: InnerPixel>(
@@ -21,7 +23,7 @@ pub(crate) fn split_h_two_images_for_threading<'a, P: InnerPixel>(
let dst_width = dst_view.width();
let dst_height = dst_view.height();
let max_num_parts = calculate_max_h_parts_number(dst_width, dst_height);
let max_num_parts = calculate_max_number_of_horizonal_parts(dst_width, dst_height).get();
let num_threads = current_num_threads() as u32;
if num_threads > 1 && max_num_parts > 1 {
@@ -44,7 +46,7 @@ pub(crate) fn split_h_one_image_for_threading<P: InnerPixel>(
) -> Option<impl ParallelIterator<Item = impl ImageViewMut<Pixel = P> + '_>> {
let width = image_view.width();
let height = image_view.height();
let max_num_parts = calculate_max_h_parts_number(width, height);
let max_num_parts = calculate_max_number_of_horizonal_parts(width, height).get();
let num_threads = current_num_threads() as u32;
if num_threads > 1 && max_num_parts > 1 {
@@ -56,19 +58,6 @@ pub(crate) fn split_h_one_image_for_threading<P: InnerPixel>(
None
}
/// It is not optimal to split images on too small parts.
/// We have to calculate minimal height of one part.
/// For small images, it is equal to `constant / area`.
/// For tall images, it is equal to `height / 256`.
fn calculate_max_h_parts_number(width: u32, height: u32) -> u32 {
if width == 0 || height == 0 {
return 1;
}
let area = height * height.max(width);
let min_height = ((1 << 14) / area).max(height / 256);
height / min_height.max(1)
}
#[inline]
pub(crate) fn split_v_two_images_for_threading<'a, P: InnerPixel>(
src_view: &'a impl ImageView<Pixel = P>,
@@ -86,7 +75,7 @@ pub(crate) fn split_v_two_images_for_threading<'a, P: InnerPixel>(
let dst_width = dst_view.width();
let dst_height = dst_view.height();
let max_num_parts = calculate_max_v_parts_number(dst_width, dst_height);
let max_num_parts = calculate_max_number_of_vertical_parts(dst_width, dst_height).get();
let num_threads = current_num_threads() as u32;
if num_threads > 1 && max_num_parts > 1 {
@@ -103,15 +92,16 @@ pub(crate) fn split_v_two_images_for_threading<'a, P: InnerPixel>(
None
}
/// It is not optimal to split images on too small parts.
/// We have to calculate minimal width of one part.
/// For small images, it is equal to `constant / area`.
/// For wide images, it is equal to `width / 256`.
fn calculate_max_v_parts_number(width: u32, height: u32) -> u32 {
if width == 0 || height == 0 {
return 1;
}
let area = width * height.max(width);
let min_width = ((1 << 14) / area).max(width / 256);
width / min_width.max(1)
const PIXELS_PER_THREAD: u64 = 1_024; // It was selected as a result of simple benchmarking.
fn calculate_max_number_of_horizonal_parts(width: u32, height: u32) -> NonZeroU32 {
let area = width as u64 * height as u64;
let num_parts = (area / PIXELS_PER_THREAD).min(height as _) as u32;
NonZeroU32::new(num_parts).unwrap_or(NonZeroU32::MIN)
}
fn calculate_max_number_of_vertical_parts(width: u32, height: u32) -> NonZeroU32 {
let area = width as u64 * height as u64;
let num_parts = (area / PIXELS_PER_THREAD).min(width as _) as u32;
NonZeroU32::new(num_parts).unwrap_or(NonZeroU32::MIN)
}