Added basic support for the new pixel type PixelType::F32x3 (#30).

This commit is contained in:
Kirill Kuzminykh
2024-06-18 02:10:49 +03:00
parent 05d31af22f
commit 91c28d4539
16 changed files with 250 additions and 13 deletions
+1
View File
@@ -4,6 +4,7 @@
- Added support for the new pixel type `PixelType::F32x2` with
optimizations for SSE4.1 and AVX2 (#30).
- Added basic support for the new pixel type `PixelType::F32x3` (#30).
## [4.0.0] - 2024-05-13
+6 -1
View File
@@ -80,6 +80,11 @@ name = "bench_compare_rgb16"
harness = false
[[bench]]
name = "bench_compare_rgb32f"
harness = false
[[bench]]
name = "bench_compare_rgba"
harness = false
@@ -111,7 +116,7 @@ harness = false
[[bench]]
name = "bench_compare_la_f32"
name = "bench_compare_la32f"
harness = false
[[bench]]
+2 -1
View File
@@ -22,7 +22,8 @@ Supported pixel formats and available optimisations:
| U16x4 | Four `u16` components per pixel (e.g. RGBA16, RGBx16, CMYK16) | + | + | + | + |
| I32 | One `i32` component per pixel (e.g. L) | - | - | - | - |
| F32 | One `f32` component per pixel (e.g. L) | - | - | - | - |
| F32x2 | Two `f32` components per pixel (e.g. LA) | + | + | - | - |
| F32x2 | Two `f32` components per pixel (e.g. LA32f) | + | + | - | - |
| F32x3 | Three `f32` components per pixel (e.g. RGB32F) | - | - | - | - |
## Colorspace
@@ -2,13 +2,13 @@ use fast_image_resize::pixels::F32x2;
mod utils;
pub fn bench_downscale_la_f32(bench_group: &mut utils::BenchGroup) {
pub fn bench_downscale_la32f(bench_group: &mut utils::BenchGroup) {
type P = F32x2;
utils::libvips_resize::<P>(bench_group, true);
utils::fir_resize::<P>(bench_group, true);
}
fn main() {
let res = utils::run_bench(bench_downscale_la_f32, "Compare resize of LA-F32 image");
let res = utils::run_bench(bench_downscale_la32f, "Compare resize of LA32F image");
utils::print_and_write_compare_result(&res);
}
+27
View File
@@ -0,0 +1,27 @@
use resize::Pixel::RGBF32;
use rgb::FromSlice;
use fast_image_resize::pixels::F32x3;
use testing::PixelTestingExt;
mod utils;
pub fn bench_downscale_rgb32f(bench_group: &mut utils::BenchGroup) {
type P = F32x3;
let src_image = P::load_big_image();
utils::image_resize(bench_group, &src_image);
utils::resize_resize(
bench_group,
RGBF32,
src_image.as_raw().as_rgb(),
src_image.width(),
src_image.height(),
);
utils::libvips_resize::<P>(bench_group, false);
utils::fir_resize::<P>(bench_group, false);
}
fn main() {
let res = utils::run_bench(bench_downscale_rgb32f, "Compare resize of RGB32F image");
utils::print_and_write_compare_result(&res);
}
@@ -0,0 +1,11 @@
### Resize RGB16F image (F32x3) 4928x3279 => 852x567
Pipeline:
`src_image => resize => dst_image`
- Source image [nasa-4928x3279.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279.png)
has converted into RGB32F image.
- Numbers in the table mean a duration of image resizing in milliseconds.
{{ compare_results -}}
+28 -5
View File
@@ -1,4 +1,5 @@
<!-- introduction start -->
## Benchmarks of fast_image_resize crate for x86_64 architecture
Environment:
@@ -10,14 +11,12 @@ Environment:
- criterion = "0.5.1"
- fast_image_resize = "4.0.0"
Other libraries used to compare of resizing speed:
- image = "0.25.1" (<https://crates.io/crates/image>)
- resize = "0.8.4" (<https://crates.io/crates/resize>)
- libvips = "8.12.1" (single-threaded mode, cache disabled)
Resize algorithms:
- Nearest
@@ -25,6 +24,7 @@ Resize algorithms:
- Bilinear - convolution with minimal kernel size 2x2 px
- Bicubic (CatmullRom) - convolution with minimal kernel size 4x4 px
- Lanczos3 - convolution with minimal kernel size 6x6 px
<!-- introduction end -->
<!-- bench_compare_rgb start -->
@@ -212,8 +212,9 @@ Pipeline:
<!-- bench_compare_la16 end -->
<!-- bench_compare_la_f32 start -->
### Resize LA-F32 (luma with alpha channel) image (F32x2) 4928x3279 => 852x567
<!-- bench_compare_la32f start -->
### Resize LA32F (luma with alpha channel) image (F32x2) 4928x3279 => 852x567
Pipeline:
@@ -232,4 +233,26 @@ Pipeline:
| fir rust | 0.38 | 21.26 | 28.82 | 47.50 | 70.34 |
| fir sse4.1 | 0.38 | 16.23 | 20.91 | 30.35 | 40.12 |
| fir avx2 | 0.39 | 15.05 | 17.18 | 22.47 | 27.89 |
<!-- bench_compare_la_f32 end -->
<!-- bench_compare_la32f end -->
<!-- bench_compare_rgb32f start -->
### Resize RGB16F image (F32x3) 4928x3279 => 852x567
Pipeline:
`src_image => resize => dst_image`
- Source image [nasa-4928x3279.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279.png)
has converted into RGB32F image.
- Numbers in the table mean a duration of image resizing in milliseconds.
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|----------|:-------:|:-----:|:--------:|:-------:|:--------:|
| image | 26.52 | - | 62.98 | 105.56 | 147.74 |
| resize | 8.76 | 14.03 | 23.62 | 47.69 | 70.36 |
| libvips | 11.91 | 60.28 | 53.55 | 112.62 | 200.59 |
| fir rust | 0.87 | 17.12 | 27.34 | 51.51 | 75.67 |
<!-- bench_compare_rgb32f end -->
+1
View File
@@ -60,3 +60,4 @@ impl AlphaMulDiv for pixels::U16 {}
impl AlphaMulDiv for pixels::U16x3 {}
impl AlphaMulDiv for pixels::I32 {}
impl AlphaMulDiv for pixels::F32 {}
impl AlphaMulDiv for pixels::F32x3 {}
+23 -3
View File
@@ -1,5 +1,6 @@
use crate::pixels::{
F32x2, InnerPixel, IntoPixelComponent, U16x2, U16x3, U16x4, U8x2, U8x3, U8x4, F32, I32, U16, U8,
F32x2, F32x3, InnerPixel, IntoPixelComponent, U16x2, U16x3, U16x4, U8x2, U8x3, U8x4, F32, I32,
U16, U8,
};
use crate::{
try_pixel_type, DifferentDimensionsError, ImageView, ImageViewMut, IntoImageView,
@@ -46,7 +47,13 @@ pub fn change_type_of_pixel_components(
(PT::U16x2, U16x2),
(PT::F32x2, F32x2)
),
PixelType::U8x3 => map_dst!(U8x3, dst_pixel_type, (PT::U8x3, U8x3), (PT::U16x3, U16x3)),
PixelType::U8x3 => map_dst!(
U8x3,
dst_pixel_type,
(PT::U8x3, U8x3),
(PT::U16x3, U16x3),
(PT::F32x3, F32x3)
),
PixelType::U8x4 => map_dst!(U8x4, dst_pixel_type, (PT::U8x4, U8x4), (PT::U16x4, U16x4)),
PixelType::U16 => map_dst!(
U16,
@@ -63,7 +70,13 @@ pub fn change_type_of_pixel_components(
(PT::U16x2, U16x2),
(PT::F32x2, F32x2)
),
PixelType::U16x3 => map_dst!(U16x3, dst_pixel_type, (PT::U8x3, U8x3), (PT::U16x3, U16x3)),
PixelType::U16x3 => map_dst!(
U16x3,
dst_pixel_type,
(PT::U8x3, U8x3),
(PT::U16x3, U16x3),
(PT::F32x3, F32x3)
),
PixelType::U16x4 => map_dst!(U16x4, dst_pixel_type, (PT::U8x4, U8x4), (PT::U16x4, U16x4)),
PixelType::I32 => map_dst!(
I32,
@@ -88,6 +101,13 @@ pub fn change_type_of_pixel_components(
(PT::U16x2, U16x2),
(PT::F32x2, F32x2)
),
PixelType::F32x3 => map_dst!(
F32x3,
dst_pixel_type,
(PT::U8x3, U8x3),
(PT::U16x3, U16x3),
(PT::F32x3, F32x3)
),
}
}
+48
View File
@@ -0,0 +1,48 @@
use crate::convolution::vertical_f32::vert_convolution_f32;
use crate::cpu_extensions::CpuExtensions;
use crate::pixels::F32x3;
use crate::{ImageView, ImageViewMut};
use super::{Coefficients, Convolution};
// #[cfg(target_arch = "x86_64")]
// mod avx2;
mod native;
// #[cfg(target_arch = "aarch64")]
// mod neon;
// #[cfg(target_arch = "x86_64")]
// mod sse4;
// #[cfg(target_arch = "wasm32")]
// mod wasm32;
impl Convolution for F32x3 {
fn horiz_convolution(
src_view: &impl ImageView<Pixel = Self>,
dst_view: &mut impl ImageViewMut<Pixel = Self>,
offset: u32,
coeffs: Coefficients,
cpu_extensions: CpuExtensions,
) {
match cpu_extensions {
// #[cfg(target_arch = "x86_64")]
// CpuExtensions::Avx2 => avx2::horiz_convolution(src_view, dst_view, offset, coeffs),
// #[cfg(target_arch = "x86_64")]
// CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, coeffs),
#[cfg(target_arch = "aarch64")]
CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
_ => native::horiz_convolution(src_view, dst_view, offset, coeffs),
}
}
fn vert_convolution(
src_view: &impl ImageView<Pixel = Self>,
dst_view: &mut impl ImageViewMut<Pixel = Self>,
offset: u32,
coeffs: Coefficients,
cpu_extensions: CpuExtensions,
) {
vert_convolution_f32(src_view, dst_view, offset, coeffs, cpu_extensions);
}
}
+27
View File
@@ -0,0 +1,27 @@
use crate::convolution::Coefficients;
use crate::pixels::F32x3;
use crate::{ImageView, ImageViewMut};
pub(crate) fn horiz_convolution(
src_view: &impl ImageView<Pixel = F32x3>,
dst_view: &mut impl ImageViewMut<Pixel = F32x3>,
offset: u32,
coeffs: Coefficients,
) {
let coefficients_chunks = coeffs.get_chunks();
let src_rows = src_view.iter_rows(offset);
let dst_rows = dst_view.iter_rows_mut(0);
for (dst_row, src_row) in dst_rows.zip(src_rows) {
for (dst_pixel, coeffs_chunk) in dst_row.iter_mut().zip(&coefficients_chunks) {
let first_x_src = coeffs_chunk.start as usize;
let mut ss = [0.; 3];
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
for (&k, &src_pixel) in coeffs_chunk.values.iter().zip(src_pixels) {
for (s, c) in ss.iter_mut().zip(src_pixel.0) {
*s += c as f64 * k;
}
}
dst_pixel.0 = ss.map(|v| v as f32);
}
}
}
+1
View File
@@ -22,6 +22,7 @@ cfg_if::cfg_if! {
mod i32x1;
mod f32x1;
mod f32x2;
mod f32x3;
mod vertical_u16;
mod vertical_f32;
}
+11
View File
@@ -18,6 +18,7 @@ pub enum PixelType {
I32,
F32,
F32x2,
F32x3,
}
impl PixelType {
@@ -32,6 +33,7 @@ impl PixelType {
Self::U16x3 => 6,
Self::U16x4 => 8,
Self::F32x2 => 8,
Self::F32x3 => 12,
_ => 4,
}
}
@@ -50,6 +52,7 @@ impl PixelType {
Self::I32 => unsafe { buffer.align_to::<I32>().0.is_empty() },
Self::F32 => unsafe { buffer.align_to::<F32>().0.is_empty() },
Self::F32x2 => unsafe { buffer.align_to::<F32x2>().0.is_empty() },
Self::F32x3 => unsafe { buffer.align_to::<F32x3>().0.is_empty() },
}
}
}
@@ -310,6 +313,14 @@ pixel_struct!(
PixelType::F32x2,
"Two `f32` component per pixel (e.g. LA-F32)"
);
pixel_struct!(
F32x3,
[f32; 3],
f32,
3,
PixelType::F32x3,
"Three `f32` components per pixel (e.g. RGB-F32)"
);
pub trait IntoPixelComponent<Out: PixelComponent>
where
+1
View File
@@ -178,6 +178,7 @@ impl Resizer {
(PT::I32, pixels::I32),
(PT::F32, pixels::F32),
(PT::F32x2, pixels::F32x2),
(PT::F32x3, pixels::F32x3),
);
#[cfg(feature = "only_u8x4")]
+24 -1
View File
@@ -60,6 +60,7 @@ pub trait PixelTestingExt: PixelTrait {
PixelType::I32 => "i32",
PixelType::F32 => "f32",
PixelType::F32x2 => "f32x2",
PixelType::F32x3 => "f32x3",
_ => unreachable!(),
}
}
@@ -89,7 +90,8 @@ pub trait PixelTestingExt: PixelTrait {
| PixelType::U16
| PixelType::U16x3
| PixelType::I32
| PixelType::F32 => (
| PixelType::F32
| PixelType::F32x3 => (
"./data/nasa-4928x3279.png",
"./data/nasa-4019x4019.png",
"./data/nasa-852x567.png",
@@ -350,6 +352,21 @@ impl PixelTestingExt for F32x2 {
}
}
impl PixelTestingExt for F32x3 {
type ImagePixel = image::Rgb<f32>;
type Container = Vec<f32>;
fn load_image_buffer(
img_reader: Reader<BufReader<File>>,
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
img_reader.decode().unwrap().to_rgb32f()
}
fn img_into_bytes(img: ImageBuffer<Self::ImagePixel, Self::Container>) -> Vec<u8> {
img.as_raw().iter().flat_map(|&c| c.to_le_bytes()).collect()
}
}
pub fn save_result(image: &Image, name: &str) {
if std::env::var("SAVE_RESULT").unwrap_or_else(|_| "".to_owned()) == "" {
return;
@@ -378,6 +395,12 @@ pub fn save_result(image: &Image, name: &str) {
save_result(&image_u16, name);
return;
}
PixelType::F32x3 => {
let mut image_u16 = Image::new(image.width(), image.height(), PixelType::U16x3);
change_type_of_pixel_components(image, &mut image_u16).unwrap();
save_result(&image_u16, name);
return;
}
_ => panic!("Unsupported type of pixels"),
};
image::save_buffer(
+37
View File
@@ -717,6 +717,43 @@ mod not_u8x4 {
}
}
// F32x3
#[test]
fn downscale_f32x3() {
type P = F32x3;
P::downscale_test(
ResizeAlg::Nearest,
CpuExtensions::None,
[28271357515050, 34102344731602, 34154875278897],
);
for cpu_extensions in P::cpu_extensions() {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[41676905227663, 40314876714373, 40146438250679],
);
}
}
#[test]
fn upscale_f32x3() {
type P = F32x3;
P::upscale_test(
ResizeAlg::Nearest,
CpuExtensions::None,
[10945976142696393, 13185868359050104, 13307431189096686],
);
for cpu_extensions in P::cpu_extensions() {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
cpu_extensions,
[12058434766196853, 14081191473477964, 14079890133382920],
);
}
}
#[test]
fn fractional_cropping() {
let mut src_buf = [0, 0, 0, 0, 255, 0, 0, 0, 0];