mirror of
https://github.com/Cykooz/fast_image_resize.git
synced 2026-10-07 17:01:09 +00:00
Added basic support for the new pixel type PixelType::F32x3 (#30).
This commit is contained in:
@@ -4,6 +4,7 @@
|
||||
|
||||
- Added support for the new pixel type `PixelType::F32x2` with
|
||||
optimizations for SSE4.1 and AVX2 (#30).
|
||||
- Added basic support for the new pixel type `PixelType::F32x3` (#30).
|
||||
|
||||
## [4.0.0] - 2024-05-13
|
||||
|
||||
|
||||
+6
-1
@@ -80,6 +80,11 @@ name = "bench_compare_rgb16"
|
||||
harness = false
|
||||
|
||||
|
||||
[[bench]]
|
||||
name = "bench_compare_rgb32f"
|
||||
harness = false
|
||||
|
||||
|
||||
[[bench]]
|
||||
name = "bench_compare_rgba"
|
||||
harness = false
|
||||
@@ -111,7 +116,7 @@ harness = false
|
||||
|
||||
|
||||
[[bench]]
|
||||
name = "bench_compare_la_f32"
|
||||
name = "bench_compare_la32f"
|
||||
harness = false
|
||||
|
||||
[[bench]]
|
||||
|
||||
@@ -22,7 +22,8 @@ Supported pixel formats and available optimisations:
|
||||
| U16x4 | Four `u16` components per pixel (e.g. RGBA16, RGBx16, CMYK16) | + | + | + | + |
|
||||
| I32 | One `i32` component per pixel (e.g. L) | - | - | - | - |
|
||||
| F32 | One `f32` component per pixel (e.g. L) | - | - | - | - |
|
||||
| F32x2 | Two `f32` components per pixel (e.g. LA) | + | + | - | - |
|
||||
| F32x2 | Two `f32` components per pixel (e.g. LA32f) | + | + | - | - |
|
||||
| F32x3 | Three `f32` components per pixel (e.g. RGB32F) | - | - | - | - |
|
||||
|
||||
## Colorspace
|
||||
|
||||
|
||||
@@ -2,13 +2,13 @@ use fast_image_resize::pixels::F32x2;
|
||||
|
||||
mod utils;
|
||||
|
||||
pub fn bench_downscale_la_f32(bench_group: &mut utils::BenchGroup) {
|
||||
pub fn bench_downscale_la32f(bench_group: &mut utils::BenchGroup) {
|
||||
type P = F32x2;
|
||||
utils::libvips_resize::<P>(bench_group, true);
|
||||
utils::fir_resize::<P>(bench_group, true);
|
||||
}
|
||||
|
||||
fn main() {
|
||||
let res = utils::run_bench(bench_downscale_la_f32, "Compare resize of LA-F32 image");
|
||||
let res = utils::run_bench(bench_downscale_la32f, "Compare resize of LA32F image");
|
||||
utils::print_and_write_compare_result(&res);
|
||||
}
|
||||
@@ -0,0 +1,27 @@
|
||||
use resize::Pixel::RGBF32;
|
||||
use rgb::FromSlice;
|
||||
|
||||
use fast_image_resize::pixels::F32x3;
|
||||
use testing::PixelTestingExt;
|
||||
|
||||
mod utils;
|
||||
|
||||
pub fn bench_downscale_rgb32f(bench_group: &mut utils::BenchGroup) {
|
||||
type P = F32x3;
|
||||
let src_image = P::load_big_image();
|
||||
utils::image_resize(bench_group, &src_image);
|
||||
utils::resize_resize(
|
||||
bench_group,
|
||||
RGBF32,
|
||||
src_image.as_raw().as_rgb(),
|
||||
src_image.width(),
|
||||
src_image.height(),
|
||||
);
|
||||
utils::libvips_resize::<P>(bench_group, false);
|
||||
utils::fir_resize::<P>(bench_group, false);
|
||||
}
|
||||
|
||||
fn main() {
|
||||
let res = utils::run_bench(bench_downscale_rgb32f, "Compare resize of RGB32F image");
|
||||
utils::print_and_write_compare_result(&res);
|
||||
}
|
||||
@@ -0,0 +1,11 @@
|
||||
### Resize RGB16F image (F32x3) 4928x3279 => 852x567
|
||||
|
||||
Pipeline:
|
||||
|
||||
`src_image => resize => dst_image`
|
||||
|
||||
- Source image [nasa-4928x3279.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279.png)
|
||||
has converted into RGB32F image.
|
||||
- Numbers in the table mean a duration of image resizing in milliseconds.
|
||||
|
||||
{{ compare_results -}}
|
||||
+28
-5
@@ -1,4 +1,5 @@
|
||||
<!-- introduction start -->
|
||||
|
||||
## Benchmarks of fast_image_resize crate for x86_64 architecture
|
||||
|
||||
Environment:
|
||||
@@ -10,14 +11,12 @@ Environment:
|
||||
- criterion = "0.5.1"
|
||||
- fast_image_resize = "4.0.0"
|
||||
|
||||
|
||||
Other libraries used to compare of resizing speed:
|
||||
|
||||
- image = "0.25.1" (<https://crates.io/crates/image>)
|
||||
- resize = "0.8.4" (<https://crates.io/crates/resize>)
|
||||
- libvips = "8.12.1" (single-threaded mode, cache disabled)
|
||||
|
||||
|
||||
Resize algorithms:
|
||||
|
||||
- Nearest
|
||||
@@ -25,6 +24,7 @@ Resize algorithms:
|
||||
- Bilinear - convolution with minimal kernel size 2x2 px
|
||||
- Bicubic (CatmullRom) - convolution with minimal kernel size 4x4 px
|
||||
- Lanczos3 - convolution with minimal kernel size 6x6 px
|
||||
|
||||
<!-- introduction end -->
|
||||
|
||||
<!-- bench_compare_rgb start -->
|
||||
@@ -212,8 +212,9 @@ Pipeline:
|
||||
|
||||
<!-- bench_compare_la16 end -->
|
||||
|
||||
<!-- bench_compare_la_f32 start -->
|
||||
### Resize LA-F32 (luma with alpha channel) image (F32x2) 4928x3279 => 852x567
|
||||
<!-- bench_compare_la32f start -->
|
||||
|
||||
### Resize LA32F (luma with alpha channel) image (F32x2) 4928x3279 => 852x567
|
||||
|
||||
Pipeline:
|
||||
|
||||
@@ -232,4 +233,26 @@ Pipeline:
|
||||
| fir rust | 0.38 | 21.26 | 28.82 | 47.50 | 70.34 |
|
||||
| fir sse4.1 | 0.38 | 16.23 | 20.91 | 30.35 | 40.12 |
|
||||
| fir avx2 | 0.39 | 15.05 | 17.18 | 22.47 | 27.89 |
|
||||
<!-- bench_compare_la_f32 end -->
|
||||
|
||||
<!-- bench_compare_la32f end -->
|
||||
|
||||
<!-- bench_compare_rgb32f start -->
|
||||
|
||||
### Resize RGB16F image (F32x3) 4928x3279 => 852x567
|
||||
|
||||
Pipeline:
|
||||
|
||||
`src_image => resize => dst_image`
|
||||
|
||||
- Source image [nasa-4928x3279.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279.png)
|
||||
has converted into RGB32F image.
|
||||
- Numbers in the table mean a duration of image resizing in milliseconds.
|
||||
|
||||
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|
||||
|----------|:-------:|:-----:|:--------:|:-------:|:--------:|
|
||||
| image | 26.52 | - | 62.98 | 105.56 | 147.74 |
|
||||
| resize | 8.76 | 14.03 | 23.62 | 47.69 | 70.36 |
|
||||
| libvips | 11.91 | 60.28 | 53.55 | 112.62 | 200.59 |
|
||||
| fir rust | 0.87 | 17.12 | 27.34 | 51.51 | 75.67 |
|
||||
|
||||
<!-- bench_compare_rgb32f end -->
|
||||
|
||||
@@ -60,3 +60,4 @@ impl AlphaMulDiv for pixels::U16 {}
|
||||
impl AlphaMulDiv for pixels::U16x3 {}
|
||||
impl AlphaMulDiv for pixels::I32 {}
|
||||
impl AlphaMulDiv for pixels::F32 {}
|
||||
impl AlphaMulDiv for pixels::F32x3 {}
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
use crate::pixels::{
|
||||
F32x2, InnerPixel, IntoPixelComponent, U16x2, U16x3, U16x4, U8x2, U8x3, U8x4, F32, I32, U16, U8,
|
||||
F32x2, F32x3, InnerPixel, IntoPixelComponent, U16x2, U16x3, U16x4, U8x2, U8x3, U8x4, F32, I32,
|
||||
U16, U8,
|
||||
};
|
||||
use crate::{
|
||||
try_pixel_type, DifferentDimensionsError, ImageView, ImageViewMut, IntoImageView,
|
||||
@@ -46,7 +47,13 @@ pub fn change_type_of_pixel_components(
|
||||
(PT::U16x2, U16x2),
|
||||
(PT::F32x2, F32x2)
|
||||
),
|
||||
PixelType::U8x3 => map_dst!(U8x3, dst_pixel_type, (PT::U8x3, U8x3), (PT::U16x3, U16x3)),
|
||||
PixelType::U8x3 => map_dst!(
|
||||
U8x3,
|
||||
dst_pixel_type,
|
||||
(PT::U8x3, U8x3),
|
||||
(PT::U16x3, U16x3),
|
||||
(PT::F32x3, F32x3)
|
||||
),
|
||||
PixelType::U8x4 => map_dst!(U8x4, dst_pixel_type, (PT::U8x4, U8x4), (PT::U16x4, U16x4)),
|
||||
PixelType::U16 => map_dst!(
|
||||
U16,
|
||||
@@ -63,7 +70,13 @@ pub fn change_type_of_pixel_components(
|
||||
(PT::U16x2, U16x2),
|
||||
(PT::F32x2, F32x2)
|
||||
),
|
||||
PixelType::U16x3 => map_dst!(U16x3, dst_pixel_type, (PT::U8x3, U8x3), (PT::U16x3, U16x3)),
|
||||
PixelType::U16x3 => map_dst!(
|
||||
U16x3,
|
||||
dst_pixel_type,
|
||||
(PT::U8x3, U8x3),
|
||||
(PT::U16x3, U16x3),
|
||||
(PT::F32x3, F32x3)
|
||||
),
|
||||
PixelType::U16x4 => map_dst!(U16x4, dst_pixel_type, (PT::U8x4, U8x4), (PT::U16x4, U16x4)),
|
||||
PixelType::I32 => map_dst!(
|
||||
I32,
|
||||
@@ -88,6 +101,13 @@ pub fn change_type_of_pixel_components(
|
||||
(PT::U16x2, U16x2),
|
||||
(PT::F32x2, F32x2)
|
||||
),
|
||||
PixelType::F32x3 => map_dst!(
|
||||
F32x3,
|
||||
dst_pixel_type,
|
||||
(PT::U8x3, U8x3),
|
||||
(PT::U16x3, U16x3),
|
||||
(PT::F32x3, F32x3)
|
||||
),
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,48 @@
|
||||
use crate::convolution::vertical_f32::vert_convolution_f32;
|
||||
use crate::cpu_extensions::CpuExtensions;
|
||||
use crate::pixels::F32x3;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
use super::{Coefficients, Convolution};
|
||||
|
||||
// #[cfg(target_arch = "x86_64")]
|
||||
// mod avx2;
|
||||
mod native;
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// mod neon;
|
||||
// #[cfg(target_arch = "x86_64")]
|
||||
// mod sse4;
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
// mod wasm32;
|
||||
|
||||
impl Convolution for F32x3 {
|
||||
fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = Self>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
// #[cfg(target_arch = "x86_64")]
|
||||
// CpuExtensions::Avx2 => avx2::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
// #[cfg(target_arch = "x86_64")]
|
||||
// CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "aarch64")]
|
||||
CpuExtensions::Neon => neon::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Simd128 => wasm32::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
_ => native::horiz_convolution(src_view, dst_view, offset, coeffs),
|
||||
}
|
||||
}
|
||||
|
||||
fn vert_convolution(
|
||||
src_view: &impl ImageView<Pixel = Self>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = Self>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
vert_convolution_f32(src_view, dst_view, offset, coeffs, cpu_extensions);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,27 @@
|
||||
use crate::convolution::Coefficients;
|
||||
use crate::pixels::F32x3;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
pub(crate) fn horiz_convolution(
|
||||
src_view: &impl ImageView<Pixel = F32x3>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = F32x3>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let coefficients_chunks = coeffs.get_chunks();
|
||||
let src_rows = src_view.iter_rows(offset);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
for (dst_row, src_row) in dst_rows.zip(src_rows) {
|
||||
for (dst_pixel, coeffs_chunk) in dst_row.iter_mut().zip(&coefficients_chunks) {
|
||||
let first_x_src = coeffs_chunk.start as usize;
|
||||
let mut ss = [0.; 3];
|
||||
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
|
||||
for (&k, &src_pixel) in coeffs_chunk.values.iter().zip(src_pixels) {
|
||||
for (s, c) in ss.iter_mut().zip(src_pixel.0) {
|
||||
*s += c as f64 * k;
|
||||
}
|
||||
}
|
||||
dst_pixel.0 = ss.map(|v| v as f32);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -22,6 +22,7 @@ cfg_if::cfg_if! {
|
||||
mod i32x1;
|
||||
mod f32x1;
|
||||
mod f32x2;
|
||||
mod f32x3;
|
||||
mod vertical_u16;
|
||||
mod vertical_f32;
|
||||
}
|
||||
|
||||
@@ -18,6 +18,7 @@ pub enum PixelType {
|
||||
I32,
|
||||
F32,
|
||||
F32x2,
|
||||
F32x3,
|
||||
}
|
||||
|
||||
impl PixelType {
|
||||
@@ -32,6 +33,7 @@ impl PixelType {
|
||||
Self::U16x3 => 6,
|
||||
Self::U16x4 => 8,
|
||||
Self::F32x2 => 8,
|
||||
Self::F32x3 => 12,
|
||||
_ => 4,
|
||||
}
|
||||
}
|
||||
@@ -50,6 +52,7 @@ impl PixelType {
|
||||
Self::I32 => unsafe { buffer.align_to::<I32>().0.is_empty() },
|
||||
Self::F32 => unsafe { buffer.align_to::<F32>().0.is_empty() },
|
||||
Self::F32x2 => unsafe { buffer.align_to::<F32x2>().0.is_empty() },
|
||||
Self::F32x3 => unsafe { buffer.align_to::<F32x3>().0.is_empty() },
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -310,6 +313,14 @@ pixel_struct!(
|
||||
PixelType::F32x2,
|
||||
"Two `f32` component per pixel (e.g. LA-F32)"
|
||||
);
|
||||
pixel_struct!(
|
||||
F32x3,
|
||||
[f32; 3],
|
||||
f32,
|
||||
3,
|
||||
PixelType::F32x3,
|
||||
"Three `f32` components per pixel (e.g. RGB-F32)"
|
||||
);
|
||||
|
||||
pub trait IntoPixelComponent<Out: PixelComponent>
|
||||
where
|
||||
|
||||
@@ -178,6 +178,7 @@ impl Resizer {
|
||||
(PT::I32, pixels::I32),
|
||||
(PT::F32, pixels::F32),
|
||||
(PT::F32x2, pixels::F32x2),
|
||||
(PT::F32x3, pixels::F32x3),
|
||||
);
|
||||
|
||||
#[cfg(feature = "only_u8x4")]
|
||||
|
||||
+24
-1
@@ -60,6 +60,7 @@ pub trait PixelTestingExt: PixelTrait {
|
||||
PixelType::I32 => "i32",
|
||||
PixelType::F32 => "f32",
|
||||
PixelType::F32x2 => "f32x2",
|
||||
PixelType::F32x3 => "f32x3",
|
||||
_ => unreachable!(),
|
||||
}
|
||||
}
|
||||
@@ -89,7 +90,8 @@ pub trait PixelTestingExt: PixelTrait {
|
||||
| PixelType::U16
|
||||
| PixelType::U16x3
|
||||
| PixelType::I32
|
||||
| PixelType::F32 => (
|
||||
| PixelType::F32
|
||||
| PixelType::F32x3 => (
|
||||
"./data/nasa-4928x3279.png",
|
||||
"./data/nasa-4019x4019.png",
|
||||
"./data/nasa-852x567.png",
|
||||
@@ -350,6 +352,21 @@ impl PixelTestingExt for F32x2 {
|
||||
}
|
||||
}
|
||||
|
||||
impl PixelTestingExt for F32x3 {
|
||||
type ImagePixel = image::Rgb<f32>;
|
||||
type Container = Vec<f32>;
|
||||
|
||||
fn load_image_buffer(
|
||||
img_reader: Reader<BufReader<File>>,
|
||||
) -> ImageBuffer<Self::ImagePixel, Self::Container> {
|
||||
img_reader.decode().unwrap().to_rgb32f()
|
||||
}
|
||||
|
||||
fn img_into_bytes(img: ImageBuffer<Self::ImagePixel, Self::Container>) -> Vec<u8> {
|
||||
img.as_raw().iter().flat_map(|&c| c.to_le_bytes()).collect()
|
||||
}
|
||||
}
|
||||
|
||||
pub fn save_result(image: &Image, name: &str) {
|
||||
if std::env::var("SAVE_RESULT").unwrap_or_else(|_| "".to_owned()) == "" {
|
||||
return;
|
||||
@@ -378,6 +395,12 @@ pub fn save_result(image: &Image, name: &str) {
|
||||
save_result(&image_u16, name);
|
||||
return;
|
||||
}
|
||||
PixelType::F32x3 => {
|
||||
let mut image_u16 = Image::new(image.width(), image.height(), PixelType::U16x3);
|
||||
change_type_of_pixel_components(image, &mut image_u16).unwrap();
|
||||
save_result(&image_u16, name);
|
||||
return;
|
||||
}
|
||||
_ => panic!("Unsupported type of pixels"),
|
||||
};
|
||||
image::save_buffer(
|
||||
|
||||
@@ -717,6 +717,43 @@ mod not_u8x4 {
|
||||
}
|
||||
}
|
||||
|
||||
// F32x3
|
||||
#[test]
|
||||
fn downscale_f32x3() {
|
||||
type P = F32x3;
|
||||
P::downscale_test(
|
||||
ResizeAlg::Nearest,
|
||||
CpuExtensions::None,
|
||||
[28271357515050, 34102344731602, 34154875278897],
|
||||
);
|
||||
|
||||
for cpu_extensions in P::cpu_extensions() {
|
||||
P::downscale_test(
|
||||
ResizeAlg::Convolution(FilterType::Lanczos3),
|
||||
cpu_extensions,
|
||||
[41676905227663, 40314876714373, 40146438250679],
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn upscale_f32x3() {
|
||||
type P = F32x3;
|
||||
P::upscale_test(
|
||||
ResizeAlg::Nearest,
|
||||
CpuExtensions::None,
|
||||
[10945976142696393, 13185868359050104, 13307431189096686],
|
||||
);
|
||||
|
||||
for cpu_extensions in P::cpu_extensions() {
|
||||
P::upscale_test(
|
||||
ResizeAlg::Convolution(FilterType::Lanczos3),
|
||||
cpu_extensions,
|
||||
[12058434766196853, 14081191473477964, 14079890133382920],
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn fractional_cropping() {
|
||||
let mut src_buf = [0, 0, 0, 0, 255, 0, 0, 0, 0];
|
||||
|
||||
Reference in New Issue
Block a user