mirror of
https://github.com/Cykooz/fast_image_resize.git
synced 2026-10-07 17:01:09 +00:00
Fixed a "divide by zero" error in case of using multithreading to resize images with particular sizes (#55).
This commit is contained in:
@@ -1,3 +1,10 @@
|
||||
## [Unreleased] - ReleaseDate
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed a "divide by zero" error in case of using multithreading to resize images
|
||||
with particular sizes ([#55](https://github.com/Cykooz/fast_image_resize/issues/55)).
|
||||
|
||||
## [5.2.1] - 2025-07-27
|
||||
|
||||
### Changed
|
||||
|
||||
+9
-3
@@ -9,7 +9,7 @@ name = "fast_image_resize"
|
||||
version = "5.2.1"
|
||||
authors = ["Kirill Kuzminykh <[email protected]>"]
|
||||
edition = "2021"
|
||||
rust-version = " 1.87.0"
|
||||
rust-version = "1.87.0"
|
||||
license = "MIT OR Apache-2.0"
|
||||
description = "Library for fast image resizing with using of SIMD instructions"
|
||||
readme = "README.md"
|
||||
@@ -36,7 +36,7 @@ rayon = { version = "1.10", optional = true }
|
||||
## [DynamicImage](https://docs.rs/image/latest/image/enum.DynamicImage.html)
|
||||
## type from the `image` crate.
|
||||
image = ["dep:image", "dep:bytemuck"]
|
||||
## This feature enables image processing in `rayon` thread pool.
|
||||
## This feature enables image processing in the ` rayon ` thread pool.
|
||||
rayon = ["dep:rayon", "resize/rayon", "image/rayon"]
|
||||
for_testing = ["image", "image/png"]
|
||||
only_u8x4 = [] # This can be used to experiment with the crate's code.
|
||||
@@ -181,5 +181,11 @@ name = "bench_color_mapper"
|
||||
harness = false
|
||||
|
||||
|
||||
# Header of next release in CHANGELOG.md:
|
||||
[[bench]]
|
||||
name = "bench_threads"
|
||||
harness = false
|
||||
|
||||
|
||||
|
||||
# Header of the next release in CHANGELOG.md:
|
||||
# ## [Unreleased] - ReleaseDate
|
||||
|
||||
@@ -0,0 +1,60 @@
|
||||
use fast_image_resize::images::Image;
|
||||
use fast_image_resize::pixels::U8x3;
|
||||
use fast_image_resize::{CpuExtensions, FilterType, ResizeAlg, ResizeOptions, Resizer};
|
||||
use utils::testing::PixelTestingExt;
|
||||
|
||||
use crate::utils::{bench, build_md_table, BenchGroup};
|
||||
|
||||
mod utils;
|
||||
|
||||
pub fn fir_resize<P: PixelTestingExt>(bench_group: &mut BenchGroup, use_alpha: bool) {
|
||||
let src_sizes: Vec<u32> = vec![2, 10, 100, 500, 1000, 5000, 10000, 65536];
|
||||
|
||||
let mut resizer = Resizer::new();
|
||||
unsafe {
|
||||
resizer.set_cpu_extensions(CpuExtensions::None);
|
||||
}
|
||||
let resize_options = ResizeOptions::new()
|
||||
.resize_alg(ResizeAlg::Convolution(FilterType::Bilinear))
|
||||
.use_alpha(false);
|
||||
|
||||
for &src_width in &src_sizes {
|
||||
for &src_height in &src_sizes {
|
||||
let dst_width = src_width / 2;
|
||||
let src_image = Image::new(src_width, src_height, P::pixel_type());
|
||||
let mut dst_image = Image::new(dst_width, src_height, src_image.pixel_type());
|
||||
|
||||
for thread_count in 1..=8 {
|
||||
bench(
|
||||
bench_group,
|
||||
10,
|
||||
format!("{src_width}x{src_height}"),
|
||||
format!("{thread_count}"),
|
||||
|bencher| {
|
||||
let mut builder = rayon::ThreadPoolBuilder::new();
|
||||
builder = builder.num_threads(thread_count);
|
||||
let pool = builder.build().unwrap();
|
||||
pool.install(|| {
|
||||
bencher.iter(|| {
|
||||
resizer
|
||||
.resize(&src_image, &mut dst_image, &resize_options)
|
||||
.unwrap()
|
||||
})
|
||||
})
|
||||
},
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub fn bench_threads(bench_group: &mut BenchGroup) {
|
||||
type P = U8x3;
|
||||
fir_resize::<P>(bench_group, false);
|
||||
}
|
||||
|
||||
fn main() {
|
||||
let res = utils::run_bench(bench_threads, "Compare resize by width images with threads");
|
||||
let md_table = build_md_table(&res);
|
||||
println!("{}", md_table);
|
||||
}
|
||||
+20
-30
@@ -1,8 +1,10 @@
|
||||
use crate::pixels::InnerPixel;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
use std::num::NonZeroU32;
|
||||
|
||||
use rayon::current_num_threads;
|
||||
use rayon::prelude::*;
|
||||
use std::num::NonZeroU32;
|
||||
|
||||
use crate::pixels::InnerPixel;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn split_h_two_images_for_threading<'a, P: InnerPixel>(
|
||||
@@ -21,7 +23,7 @@ pub(crate) fn split_h_two_images_for_threading<'a, P: InnerPixel>(
|
||||
|
||||
let dst_width = dst_view.width();
|
||||
let dst_height = dst_view.height();
|
||||
let max_num_parts = calculate_max_h_parts_number(dst_width, dst_height);
|
||||
let max_num_parts = calculate_max_number_of_horizonal_parts(dst_width, dst_height).get();
|
||||
|
||||
let num_threads = current_num_threads() as u32;
|
||||
if num_threads > 1 && max_num_parts > 1 {
|
||||
@@ -44,7 +46,7 @@ pub(crate) fn split_h_one_image_for_threading<P: InnerPixel>(
|
||||
) -> Option<impl ParallelIterator<Item = impl ImageViewMut<Pixel = P> + '_>> {
|
||||
let width = image_view.width();
|
||||
let height = image_view.height();
|
||||
let max_num_parts = calculate_max_h_parts_number(width, height);
|
||||
let max_num_parts = calculate_max_number_of_horizonal_parts(width, height).get();
|
||||
|
||||
let num_threads = current_num_threads() as u32;
|
||||
if num_threads > 1 && max_num_parts > 1 {
|
||||
@@ -56,19 +58,6 @@ pub(crate) fn split_h_one_image_for_threading<P: InnerPixel>(
|
||||
None
|
||||
}
|
||||
|
||||
/// It is not optimal to split images on too small parts.
|
||||
/// We have to calculate minimal height of one part.
|
||||
/// For small images, it is equal to `constant / area`.
|
||||
/// For tall images, it is equal to `height / 256`.
|
||||
fn calculate_max_h_parts_number(width: u32, height: u32) -> u32 {
|
||||
if width == 0 || height == 0 {
|
||||
return 1;
|
||||
}
|
||||
let area = height * height.max(width);
|
||||
let min_height = ((1 << 14) / area).max(height / 256);
|
||||
height / min_height.max(1)
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn split_v_two_images_for_threading<'a, P: InnerPixel>(
|
||||
src_view: &'a impl ImageView<Pixel = P>,
|
||||
@@ -86,7 +75,7 @@ pub(crate) fn split_v_two_images_for_threading<'a, P: InnerPixel>(
|
||||
|
||||
let dst_width = dst_view.width();
|
||||
let dst_height = dst_view.height();
|
||||
let max_num_parts = calculate_max_v_parts_number(dst_width, dst_height);
|
||||
let max_num_parts = calculate_max_number_of_vertical_parts(dst_width, dst_height).get();
|
||||
|
||||
let num_threads = current_num_threads() as u32;
|
||||
if num_threads > 1 && max_num_parts > 1 {
|
||||
@@ -103,15 +92,16 @@ pub(crate) fn split_v_two_images_for_threading<'a, P: InnerPixel>(
|
||||
None
|
||||
}
|
||||
|
||||
/// It is not optimal to split images on too small parts.
|
||||
/// We have to calculate minimal width of one part.
|
||||
/// For small images, it is equal to `constant / area`.
|
||||
/// For wide images, it is equal to `width / 256`.
|
||||
fn calculate_max_v_parts_number(width: u32, height: u32) -> u32 {
|
||||
if width == 0 || height == 0 {
|
||||
return 1;
|
||||
}
|
||||
let area = width * height.max(width);
|
||||
let min_width = ((1 << 14) / area).max(width / 256);
|
||||
width / min_width.max(1)
|
||||
const PIXELS_PER_THREAD: u64 = 1_024; // It was selected as a result of simple benchmarking.
|
||||
|
||||
fn calculate_max_number_of_horizonal_parts(width: u32, height: u32) -> NonZeroU32 {
|
||||
let area = width as u64 * height as u64;
|
||||
let num_parts = (area / PIXELS_PER_THREAD).min(height as _) as u32;
|
||||
NonZeroU32::new(num_parts).unwrap_or(NonZeroU32::MIN)
|
||||
}
|
||||
|
||||
fn calculate_max_number_of_vertical_parts(width: u32, height: u32) -> NonZeroU32 {
|
||||
let area = width as u64 * height as u64;
|
||||
let num_parts = (area / PIXELS_PER_THREAD).min(width as _) as u32;
|
||||
NonZeroU32::new(num_parts).unwrap_or(NonZeroU32::MIN)
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user