mirror of
https://github.com/Cykooz/fast_image_resize.git
synced 2026-10-07 17:01:09 +00:00
- Added support of new type of pixels PixelType::U8x3 (with auto-vectorization for SSE4.1).
- Exposed module `fast_image_resize::pixels` with types `U8x3`, `U8x4`, `F32`, `I32`, `U8` used as wrappers for represent type of one pixel of image. - Some optimisations in code of convolution written in Rust (without intrinsics for SIMD). - Added variant `U8x3` into the enum `PixelType`. - Changed internal tuple structures inside of variant of `ImageRows` and `ImageRowsMut` enums.
This commit is contained in:
+15
-1
@@ -1,10 +1,24 @@
|
||||
## [Unreleased] - ReleaseDate
|
||||
|
||||
- Added support of new type of pixels `PixelType::U8x3` (with
|
||||
auto-vectorization for SSE4.1).
|
||||
- Exposed module `fast_image_resize::pixels` with types `U8x3`,
|
||||
`U8x4`, `F32`, `I32`, `U8` used as wrappers for represent type of
|
||||
one pixel of image.
|
||||
- Some optimisations in code of convolution written in Rust (without
|
||||
intrinsics for SIMD).
|
||||
- Breaking changes:
|
||||
- Added variant `U8x3` into the enum `PixelType`.
|
||||
- Changed internal tuple structures inside of variant of `ImageRows`
|
||||
and `ImageRowsMut` enums.
|
||||
|
||||
## [0.4.1] - 2021-11-13
|
||||
|
||||
- Added optimisation of convolution grayscale images (U8) with helps of ``AVX2`` instructions.
|
||||
|
||||
## [0.4.0] - 2021-10-23
|
||||
|
||||
- Added support of new type of pixels `U8` (without forced SIMD).
|
||||
- Added support of new type of pixels `PixelType::U8` (without forced SIMD).
|
||||
- Breaking changes:
|
||||
- ``ImageData`` renamed into ``Image``.
|
||||
- ``SrcImageView`` and ``DstImageView`` replaced by ``ImageView``
|
||||
|
||||
+7
-2
@@ -40,6 +40,11 @@ name = "bench_compare_rgb"
|
||||
harness = false
|
||||
|
||||
|
||||
[[bench]]
|
||||
name = "bench_compare_rgbx"
|
||||
harness = false
|
||||
|
||||
|
||||
[[bench]]
|
||||
name = "bench_compare_rgba"
|
||||
harness = false
|
||||
@@ -56,9 +61,9 @@ opt-level = 3
|
||||
|
||||
[profile.release]
|
||||
#debug = true
|
||||
#lto = true
|
||||
lto = true
|
||||
opt-level = 3
|
||||
#codegen-units = 1
|
||||
codegen-units = 1
|
||||
|
||||
|
||||
[package.metadata.release]
|
||||
|
||||
@@ -5,7 +5,13 @@ Rust library for fast image resizing with using of SIMD instructions.
|
||||
[CHANGELOG](https://github.com/Cykooz/fast_image_resize/blob/main/CHANGELOG.md)
|
||||
|
||||
Supported pixel formats and available optimisations:
|
||||
- `U8x4` - four `u8` components per pixel (RGB, RGBA, CMYK and other):
|
||||
- `U8` - one `u8` component per pixel:
|
||||
- native Rust-code without forced SIMD
|
||||
- AVX2
|
||||
- `U8x3` - three `u8` components per pixel (e.g. RGB):
|
||||
- native Rust-code without forced SIMD
|
||||
- SSE4.1 (auto-vectorization)
|
||||
- `U8x4` - four `u8` components per pixel (RGBA, RGBx, CMYK and other):
|
||||
- native Rust-code without forced SIMD
|
||||
- SSE4.1
|
||||
- AVX2
|
||||
@@ -13,9 +19,6 @@ Supported pixel formats and available optimisations:
|
||||
- native Rust-code without forced SIMD
|
||||
- `F32` - one `f32` component per pixel:
|
||||
- native Rust-code without forced SIMD
|
||||
- `U8` - one `u8` component per pixel:
|
||||
- native Rust-code without forced SIMD
|
||||
- AVX2
|
||||
|
||||
## Benchmarks
|
||||
|
||||
@@ -24,8 +27,9 @@ Environment:
|
||||
- RAM: DDR4 3000 MHz
|
||||
- Ubuntu 20.04 (linux 5.11)
|
||||
- Rust 1.56.1
|
||||
- fast_image_resize = "0.4"
|
||||
- fast_image_resize = "0.5"
|
||||
- glassbench = "0.3.0"
|
||||
- `rustflags = ["-C", "llvm-args=-x86-branches-within-32B-boundaries"]`
|
||||
|
||||
Other Rust libraries used to compare of resizing speed:
|
||||
- image = "0.23.14" (<https://crates.io/crates/image>)
|
||||
@@ -37,7 +41,7 @@ Resize algorithms:
|
||||
- Convolution with CatmullRom filter
|
||||
- Convolution with Lanczos3 filter
|
||||
|
||||
### Resize RGB image 4928x3279 => 852x567
|
||||
### Resize RGB image (U8x3) 4928x3279 => 852x567
|
||||
|
||||
Pipeline:
|
||||
|
||||
@@ -48,13 +52,12 @@ Pipeline:
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|------------|:-------:|:--------:|:----------:|:--------:|
|
||||
| image | 106.320 | 199.150 | 288.609 | 380.830 |
|
||||
| resize | 15.550 | 72.122 | 132.152 | 192.081 |
|
||||
| fir rust | 0.476 | 56.451 | 86.984 | 119.357 |
|
||||
| fir sse4.1 | - | 11.798 | 17.768 | 25.296 |
|
||||
| fir avx2 | - | 8.995 | 13.533 | 19.525 |
|
||||
| image | 108.064 | 196.203 | 279.562 | 363.843 |
|
||||
| resize | 15.607 | 72.011 | 132.167 | 205.827 |
|
||||
| fir rust | 0.481 | 53.753 | 86.047 | 117.852 |
|
||||
| fir sse4.1 | - | 43.236 | 54.124 | 76.111 |
|
||||
|
||||
### Resize RGBA image 4928x3279 => 852x567
|
||||
### Resize RGBA image (U8x4) 4928x3279 => 852x567
|
||||
|
||||
Pipeline:
|
||||
|
||||
@@ -65,11 +68,11 @@ Pipeline:
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|------------|:-------:|:--------:|:----------:|:--------:|
|
||||
| image | 107.186 | 191.834 | 281.246 | 372.796 |
|
||||
| resize | 18.163 | 79.871 | 149.159 | 218.684 |
|
||||
| fir rust | 13.630 | 69.949 | 100.425 | 133.011 |
|
||||
| fir sse4.1 | 12.034 | 23.566 | 29.809 | 37.232 |
|
||||
| fir avx2 | 6.890 | 15.015 | 18.394 | 23.827 |
|
||||
| image | 110.485 | 191.373 | 267.640 | 348.590 |
|
||||
| resize | 18.169 | 81.034 | 152.473 | 219.331 |
|
||||
| fir rust | 13.236 | 63.711 | 88.811 | 117.468 |
|
||||
| fir sse4.1 | 11.760 | 23.090 | 29.461 | 36.958 |
|
||||
| fir avx2 | 6.952 | 15.563 | 18.769 | 24.088 |
|
||||
|
||||
### Resize grayscale image (U8) 4928x3279 => 852x567
|
||||
|
||||
@@ -83,10 +86,10 @@ Pipeline:
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|----------|:-------:|:--------:|:----------:|:--------:|
|
||||
| image | 92.171 | 141.153 | 184.442 | 230.455 |
|
||||
| resize | 9.890 | 26.205 | 53.054 | 81.181 |
|
||||
| fir rust | 0.197 | 26.333 | 29.450 | 43.266 |
|
||||
| fir avx2 | - | 14.766 | 12.567 | 16.641 |
|
||||
| image | 94.548 | 140.978 | 178.725 | 218.875 |
|
||||
| resize | 9.884 | 26.831 | 54.274 | 82.708 |
|
||||
| fir rust | 0.196 | 22.045 | 24.734 | 35.630 |
|
||||
| fir avx2 | - | 9.623 | 7.869 | 11.832 |
|
||||
|
||||
## Examples
|
||||
|
||||
@@ -117,7 +120,7 @@ fn resize_image_example() {
|
||||
img.to_rgba8().into_raw(),
|
||||
fr::PixelType::U8x4,
|
||||
)
|
||||
.unwrap();
|
||||
.unwrap();
|
||||
|
||||
// Create MulDiv instance
|
||||
let alpha_mul_div: fr::MulDiv = Default::default();
|
||||
@@ -126,7 +129,7 @@ fn resize_image_example() {
|
||||
.multiply_alpha_inplace(&mut src_image.view_mut())
|
||||
.unwrap();
|
||||
|
||||
// Create wrapper that own data of destination image
|
||||
// Create container for data of destination image
|
||||
let dst_width = NonZeroU32::new(1024).unwrap();
|
||||
let dst_height = NonZeroU32::new(768).unwrap();
|
||||
let mut dst_image = fr::Image::new(dst_width, dst_height, src_image.pixel_type());
|
||||
|
||||
@@ -67,20 +67,20 @@ pub fn bench_downscale_rgb(bench: &mut Bench) {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
{
|
||||
cpu_ext_and_name.push((CpuExtensions::Sse4_1, "sse4.1"));
|
||||
cpu_ext_and_name.push((CpuExtensions::Avx2, "avx2"));
|
||||
// cpu_ext_and_name.push((CpuExtensions::Avx2, "avx2"));
|
||||
}
|
||||
for (cpu_ext, ext_name) in cpu_ext_and_name {
|
||||
for alg_name in alg_names {
|
||||
let src_rgba_image = utils::get_big_rgba_image();
|
||||
let src_buffer = src_image.as_raw();
|
||||
let src_image_data = Image::from_vec_u8(
|
||||
NonZeroU32::new(src_image.width()).unwrap(),
|
||||
NonZeroU32::new(src_image.height()).unwrap(),
|
||||
src_rgba_image.into_raw(),
|
||||
PixelType::U8x4,
|
||||
src_buffer.clone(),
|
||||
PixelType::U8x3,
|
||||
)
|
||||
.unwrap();
|
||||
let src_view = src_image_data.view();
|
||||
let mut dst_image = Image::new(new_width, new_height, PixelType::U8x4);
|
||||
let mut dst_image = Image::new(new_width, new_height, PixelType::U8x3);
|
||||
let mut dst_view = dst_image.view_mut();
|
||||
|
||||
let resize_alg = match alg_name {
|
||||
|
||||
@@ -6,8 +6,7 @@ use resize::px::RGBA;
|
||||
use resize::Pixel::RGBA8;
|
||||
use rgb::FromSlice;
|
||||
|
||||
use fast_image_resize::{CpuExtensions, FilterType, PixelType, ResizeAlg, Resizer};
|
||||
use fast_image_resize::{Image, MulDiv};
|
||||
use fast_image_resize::{CpuExtensions, FilterType, Image, MulDiv, PixelType, ResizeAlg, Resizer};
|
||||
|
||||
mod utils;
|
||||
|
||||
|
||||
@@ -0,0 +1,63 @@
|
||||
use std::num::NonZeroU32;
|
||||
|
||||
use glassbench::*;
|
||||
|
||||
use fast_image_resize::Image;
|
||||
use fast_image_resize::{CpuExtensions, FilterType, PixelType, ResizeAlg, Resizer};
|
||||
|
||||
mod utils;
|
||||
|
||||
pub fn bench_downscale_rgbx(bench: &mut Bench) {
|
||||
let src_image = utils::get_big_rgb_image();
|
||||
let new_width = NonZeroU32::new(852).unwrap();
|
||||
let new_height = NonZeroU32::new(567).unwrap();
|
||||
|
||||
let alg_names = ["Nearest", "Bilinear", "CatmullRom", "Lanczos3"];
|
||||
|
||||
// fast_image_resize crate;
|
||||
let mut cpu_ext_and_name = vec![(CpuExtensions::None, "rust")];
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
{
|
||||
cpu_ext_and_name.push((CpuExtensions::Sse4_1, "sse4.1"));
|
||||
cpu_ext_and_name.push((CpuExtensions::Avx2, "avx2"));
|
||||
}
|
||||
for (cpu_ext, ext_name) in cpu_ext_and_name {
|
||||
for alg_name in alg_names {
|
||||
let src_rgba_image = utils::get_big_rgba_image();
|
||||
let src_image_data = Image::from_vec_u8(
|
||||
NonZeroU32::new(src_image.width()).unwrap(),
|
||||
NonZeroU32::new(src_image.height()).unwrap(),
|
||||
src_rgba_image.into_raw(),
|
||||
PixelType::U8x4,
|
||||
)
|
||||
.unwrap();
|
||||
let src_view = src_image_data.view();
|
||||
let mut dst_image = Image::new(new_width, new_height, PixelType::U8x4);
|
||||
let mut dst_view = dst_image.view_mut();
|
||||
|
||||
let resize_alg = match alg_name {
|
||||
"Nearest" => ResizeAlg::Nearest,
|
||||
"Bilinear" => ResizeAlg::Convolution(FilterType::Bilinear),
|
||||
"CatmullRom" => ResizeAlg::Convolution(FilterType::CatmullRom),
|
||||
"Lanczos3" => ResizeAlg::Convolution(FilterType::Lanczos3),
|
||||
_ => return,
|
||||
};
|
||||
let mut fast_resizer = Resizer::new(resize_alg);
|
||||
|
||||
unsafe {
|
||||
fast_resizer.reset_internal_buffers();
|
||||
fast_resizer.set_cpu_extensions(cpu_ext);
|
||||
}
|
||||
|
||||
bench.task(format!("fir {} - {}", ext_name, alg_name), |task| {
|
||||
task.iter(|| {
|
||||
fast_resizer.resize(&src_view, &mut dst_view).unwrap();
|
||||
})
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
utils::print_md_table(bench);
|
||||
}
|
||||
|
||||
glassbench!("Compare resize of RGBx image", bench_downscale_rgbx,);
|
||||
@@ -67,7 +67,6 @@ pub fn bench_downscale_u8(bench: &mut Bench) {
|
||||
let mut cpu_ext_and_name = vec![(CpuExtensions::None, "rust")];
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
{
|
||||
// cpu_ext_and_name.push((CpuExtensions::Sse4_1, "sse4.1"));
|
||||
cpu_ext_and_name.push((CpuExtensions::Avx2, "avx2"));
|
||||
}
|
||||
for (cpu_ext, ext_name) in cpu_ext_and_name {
|
||||
|
||||
+52
-74
@@ -26,6 +26,19 @@ fn get_big_source_image() -> Image<'static> {
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
fn get_big_u8x3_source_image() -> Image<'static> {
|
||||
let img = utils::get_big_rgb_image();
|
||||
let width = img.width();
|
||||
let height = img.height();
|
||||
Image::from_vec_u8(
|
||||
NonZeroU32::new(width).unwrap(),
|
||||
NonZeroU32::new(height).unwrap(),
|
||||
img.into_raw(),
|
||||
PixelType::U8x3,
|
||||
)
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
fn get_big_i32_image() -> Image<'static> {
|
||||
let img = utils::get_big_luma16_image();
|
||||
let img_data: Vec<u32> = img
|
||||
@@ -90,7 +103,7 @@ fn native_nearest_bench(bench: &mut Bench) {
|
||||
});
|
||||
}
|
||||
|
||||
fn native_lanczos3_bench(bench: &mut Bench) {
|
||||
fn u8x4_lanczos3_bench(bench: &mut Bench, cpu_extensions: CpuExtensions, name: &str) {
|
||||
let image = get_big_source_image();
|
||||
let mut res_image = Image::new(
|
||||
NonZeroU32::new(NEW_WIDTH).unwrap(),
|
||||
@@ -101,51 +114,9 @@ fn native_lanczos3_bench(bench: &mut Bench) {
|
||||
let mut dst_image = res_image.view_mut();
|
||||
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
|
||||
unsafe {
|
||||
resizer.set_cpu_extensions(CpuExtensions::None);
|
||||
resizer.set_cpu_extensions(cpu_extensions);
|
||||
}
|
||||
bench.task("lanczos3 wo SIMD", |task| {
|
||||
task.iter(|| {
|
||||
resizer.resize(&src_image, &mut dst_image).unwrap();
|
||||
})
|
||||
});
|
||||
}
|
||||
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
fn sse4_lanczos3_bench(bench: &mut Bench) {
|
||||
let image = get_big_source_image();
|
||||
let mut res_image = Image::new(
|
||||
NonZeroU32::new(NEW_WIDTH).unwrap(),
|
||||
NonZeroU32::new(NEW_HEIGHT).unwrap(),
|
||||
image.pixel_type(),
|
||||
);
|
||||
let src_image = image.view();
|
||||
let mut dst_image = res_image.view_mut();
|
||||
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
|
||||
unsafe {
|
||||
resizer.set_cpu_extensions(CpuExtensions::Sse4_1);
|
||||
}
|
||||
bench.task("lanczos3 sse4", |task| {
|
||||
task.iter(|| {
|
||||
resizer.resize(&src_image, &mut dst_image).unwrap();
|
||||
})
|
||||
});
|
||||
}
|
||||
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
fn avx2_lanczos3_bench(bench: &mut Bench) {
|
||||
let image = get_big_source_image();
|
||||
let mut res_image = Image::new(
|
||||
NonZeroU32::new(NEW_WIDTH).unwrap(),
|
||||
NonZeroU32::new(NEW_HEIGHT).unwrap(),
|
||||
image.pixel_type(),
|
||||
);
|
||||
let src_image = image.view();
|
||||
let mut dst_image = res_image.view_mut();
|
||||
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
|
||||
unsafe {
|
||||
resizer.set_cpu_extensions(CpuExtensions::Avx2);
|
||||
}
|
||||
bench.task("lanczos3 avx2", |task| {
|
||||
bench.task(name, |task| {
|
||||
task.iter(|| {
|
||||
resizer.resize(&src_image, &mut dst_image).unwrap();
|
||||
})
|
||||
@@ -214,7 +185,7 @@ fn native_lanczos3_i32_bench(bench: &mut Bench) {
|
||||
});
|
||||
}
|
||||
|
||||
fn avx2_lanczos3_u8_bench(bench: &mut Bench) {
|
||||
fn u8_lanczos3_bench(bench: &mut Bench, cpu_extensions: CpuExtensions, name: &str) {
|
||||
let image = get_big_u8_image();
|
||||
let mut res_image = Image::new(
|
||||
NonZeroU32::new(NEW_WIDTH).unwrap(),
|
||||
@@ -225,29 +196,9 @@ fn avx2_lanczos3_u8_bench(bench: &mut Bench) {
|
||||
let mut dst_image = res_image.view_mut();
|
||||
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
|
||||
unsafe {
|
||||
resizer.set_cpu_extensions(CpuExtensions::Avx2);
|
||||
resizer.set_cpu_extensions(cpu_extensions);
|
||||
}
|
||||
bench.task("u8 lanczos3 avx2", |task| {
|
||||
task.iter(|| {
|
||||
resizer.resize(&src_image, &mut dst_image).unwrap();
|
||||
})
|
||||
});
|
||||
}
|
||||
|
||||
fn native_lanczos3_u8_bench(bench: &mut Bench) {
|
||||
let image = get_big_u8_image();
|
||||
let mut res_image = Image::new(
|
||||
NonZeroU32::new(NEW_WIDTH).unwrap(),
|
||||
NonZeroU32::new(NEW_HEIGHT).unwrap(),
|
||||
image.pixel_type(),
|
||||
);
|
||||
let src_image = image.view();
|
||||
let mut dst_image = res_image.view_mut();
|
||||
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
|
||||
unsafe {
|
||||
resizer.set_cpu_extensions(CpuExtensions::None);
|
||||
}
|
||||
bench.task("u8 lanczos3 wo SIMD", |task| {
|
||||
bench.task(name, |task| {
|
||||
task.iter(|| {
|
||||
resizer.resize(&src_image, &mut dst_image).unwrap();
|
||||
})
|
||||
@@ -274,6 +225,26 @@ fn native_nearest_u8_bench(bench: &mut Bench) {
|
||||
});
|
||||
}
|
||||
|
||||
fn u8x3_lanczos3_bench(bench: &mut Bench, cpu_extensions: CpuExtensions, name: &str) {
|
||||
let image = get_big_u8x3_source_image();
|
||||
let mut res_image = Image::new(
|
||||
NonZeroU32::new(NEW_WIDTH).unwrap(),
|
||||
NonZeroU32::new(NEW_HEIGHT).unwrap(),
|
||||
image.pixel_type(),
|
||||
);
|
||||
let src_image = image.view();
|
||||
let mut dst_image = res_image.view_mut();
|
||||
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
|
||||
unsafe {
|
||||
resizer.set_cpu_extensions(cpu_extensions);
|
||||
}
|
||||
bench.task(name, |task| {
|
||||
task.iter(|| {
|
||||
resizer.resize(&src_image, &mut dst_image).unwrap();
|
||||
})
|
||||
});
|
||||
}
|
||||
|
||||
pub fn main() {
|
||||
use glassbench::*;
|
||||
let name = env!("CARGO_CRATE_NAME");
|
||||
@@ -281,17 +252,24 @@ pub fn main() {
|
||||
if cmd.include_bench(name) {
|
||||
let mut bench = create_bench(name, "Resize", &cmd);
|
||||
native_nearest_bench(&mut bench);
|
||||
native_lanczos3_bench(&mut bench);
|
||||
native_lanczos3_i32_bench(&mut bench);
|
||||
native_lanczos3_u8_bench(&mut bench);
|
||||
native_nearest_u8_bench(&mut bench);
|
||||
|
||||
u8_lanczos3_bench(&mut bench, CpuExtensions::None, "u8 lanczos3 wo SIMD");
|
||||
u8x3_lanczos3_bench(&mut bench, CpuExtensions::None, "u8x3 lanczos3 wo SIMD");
|
||||
u8x4_lanczos3_bench(&mut bench, CpuExtensions::None, "u8x4 lanczos3 wo SIMD");
|
||||
native_lanczos3_i32_bench(&mut bench);
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
{
|
||||
sse4_lanczos3_bench(&mut bench);
|
||||
u8_lanczos3_bench(&mut bench, CpuExtensions::Avx2, "u8 lanczos3 avx2");
|
||||
|
||||
u8x3_lanczos3_bench(&mut bench, CpuExtensions::Sse4_1, "u8x3 lanczos3 sse4.1");
|
||||
// u8x3_lanczos3_bench(&mut bench, CpuExtensions::Avx2, "u8x3 lanczos3 avx2");
|
||||
|
||||
u8x4_lanczos3_bench(&mut bench, CpuExtensions::Sse4_1, "u8x4 lanczos3 sse4.1");
|
||||
u8x4_lanczos3_bench(&mut bench, CpuExtensions::Avx2, "u8x4 lanczos3 avx2");
|
||||
|
||||
avx2_supersampling_lanczos3_bench(&mut bench);
|
||||
avx2_lanczos3_upscale_bench(&mut bench);
|
||||
avx2_lanczos3_bench(&mut bench);
|
||||
avx2_lanczos3_u8_bench(&mut bench);
|
||||
}
|
||||
if let Err(e) = after_bench(&mut bench, &cmd) {
|
||||
eprintln!("{:?}", e);
|
||||
|
||||
@@ -16,7 +16,7 @@ pub fn get_big_rgb_image() -> RgbImage {
|
||||
|
||||
pub fn get_big_rgba_image() -> RgbaImage {
|
||||
let cur_dir = env::current_dir().unwrap();
|
||||
let img = Reader::open(cur_dir.join("data/nasa-4928x3279.png"))
|
||||
let img = Reader::open(cur_dir.join("data/nasa-4928x3279-rgba.png"))
|
||||
.unwrap()
|
||||
.decode()
|
||||
.unwrap();
|
||||
|
||||
Binary file not shown.
|
After Width: | Height: | Size: 13 MiB |
@@ -10,7 +10,7 @@ pub(crate) fn divide_alpha_avx2(
|
||||
mut dst_image: TypedImageViewMut<U8x4>,
|
||||
) {
|
||||
let width = src_image.width().get();
|
||||
let src_rows = src_image.iter_rows(0, src_image.height().get());
|
||||
let src_rows = src_image.iter_rows(0);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
@@ -31,7 +31,7 @@ pub(crate) fn divide_alpha_inplace_avx2(mut image: TypedImageViewMut<U8x4>) {
|
||||
}
|
||||
|
||||
#[target_feature(enable = "avx2")]
|
||||
unsafe fn divide_alpha_row_avx2(src_row: &[u32], dst_row: &mut [u32], width: usize) {
|
||||
unsafe fn divide_alpha_row_avx2(src_row: &[U8x4], dst_row: &mut [U8x4], width: usize) {
|
||||
let mut x: usize = 0;
|
||||
let zero = _mm256_setzero_si256();
|
||||
#[rustfmt::skip]
|
||||
|
||||
@@ -10,7 +10,7 @@ pub(crate) fn multiply_alpha_avx2(
|
||||
mut dst_image: TypedImageViewMut<U8x4>,
|
||||
) {
|
||||
let width = src_image.width().get() as usize;
|
||||
let src_rows = src_image.iter_rows(0, src_image.height().get());
|
||||
let src_rows = src_image.iter_rows(0);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
@@ -32,7 +32,7 @@ pub(crate) fn multiply_alpha_inplace_avx2(mut image: TypedImageViewMut<U8x4>) {
|
||||
|
||||
/// https://github.com/Wizermil/premultiply_alpha/blob/master/premultiply_alpha/premultiply_alpha.hpp#L232
|
||||
#[target_feature(enable = "avx2")]
|
||||
unsafe fn multiply_alpha_row_avx2(src_row: &[u32], dst_row: &mut [u32], width: usize) {
|
||||
unsafe fn multiply_alpha_row_avx2(src_row: &[U8x4], dst_row: &mut [U8x4], width: usize) {
|
||||
let mask_alpha_color_odd_255 = _mm256_set1_epi32(0xff000000u32 as i32);
|
||||
let div_255 = _mm256_set1_epi16(0x8081u16 as i16);
|
||||
#[rustfmt::skip]
|
||||
|
||||
+2
-1
@@ -21,7 +21,8 @@ mod sse2;
|
||||
///
|
||||
/// ```
|
||||
/// use std::num::NonZeroU32;
|
||||
/// use fast_image_resize::{Image, MulDiv, PixelType};
|
||||
/// use fast_image_resize::pixels::PixelType;
|
||||
/// use fast_image_resize::{Image, MulDiv};
|
||||
///
|
||||
/// let width = NonZeroU32::new(10).unwrap();
|
||||
/// let height = NonZeroU32::new(7).unwrap();
|
||||
|
||||
@@ -5,7 +5,7 @@ pub(crate) fn divide_alpha_native(
|
||||
src_image: TypedImageView<U8x4>,
|
||||
mut dst_image: TypedImageViewMut<U8x4>,
|
||||
) {
|
||||
let src_rows = src_image.iter_rows(0, src_image.height().get());
|
||||
let src_rows = src_image.iter_rows(0);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
@@ -27,20 +27,19 @@ pub(crate) fn div_and_clip(v: u8, rev_alpha: f32) -> u8 {
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) fn divide_alpha_row_native(src_row: &[u32], dst_row: &mut [u32]) {
|
||||
pub(crate) fn divide_alpha_row_native(src_row: &[U8x4], dst_row: &mut [U8x4]) {
|
||||
src_row
|
||||
.iter()
|
||||
.zip(dst_row)
|
||||
.for_each(|(src_pixel, dst_pixel)| {
|
||||
let components: [u8; 4] = src_pixel.to_le_bytes();
|
||||
let components: [u8; 4] = src_pixel.0.to_le_bytes();
|
||||
let alpha = components[3];
|
||||
let recip_alpha = if alpha == 0 { 0. } else { 255. / alpha as f32 };
|
||||
let res = [
|
||||
dst_pixel.0 = u32::from_le_bytes([
|
||||
div_and_clip(components[0], recip_alpha),
|
||||
div_and_clip(components[1], recip_alpha),
|
||||
div_and_clip(components[2], recip_alpha),
|
||||
alpha,
|
||||
];
|
||||
*dst_pixel = u32::from_le_bytes(res);
|
||||
]);
|
||||
});
|
||||
}
|
||||
|
||||
@@ -5,7 +5,7 @@ pub(crate) fn multiply_alpha_native(
|
||||
src_image: TypedImageView<U8x4>,
|
||||
mut dst_image: TypedImageViewMut<U8x4>,
|
||||
) {
|
||||
let src_rows = src_image.iter_rows(0, src_image.height().get());
|
||||
let src_rows = src_image.iter_rows(0);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
@@ -21,17 +21,16 @@ pub(crate) fn multiply_alpha_inplace_native(mut image: TypedImageViewMut<U8x4>)
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) fn multiply_alpha_row_native(src_row: &[u32], dst_row: &mut [u32]) {
|
||||
pub(crate) fn multiply_alpha_row_native(src_row: &[U8x4], dst_row: &mut [U8x4]) {
|
||||
for (src_pixel, dst_pixel) in src_row.iter().zip(dst_row) {
|
||||
let components: [u8; 4] = src_pixel.to_le_bytes();
|
||||
let components: [u8; 4] = src_pixel.0.to_le_bytes();
|
||||
let alpha = components[3];
|
||||
let res: [u8; 4] = [
|
||||
dst_pixel.0 = u32::from_le_bytes([
|
||||
mul_div_255(components[0], alpha),
|
||||
mul_div_255(components[1], alpha),
|
||||
mul_div_255(components[2], alpha),
|
||||
alpha,
|
||||
];
|
||||
*dst_pixel = u32::from_le_bytes(res);
|
||||
]);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -10,7 +10,7 @@ pub(crate) fn divide_alpha_sse2(
|
||||
mut dst_image: TypedImageViewMut<U8x4>,
|
||||
) {
|
||||
let width = src_image.width().get() as usize;
|
||||
let src_rows = src_image.iter_rows(0, src_image.height().get());
|
||||
let src_rows = src_image.iter_rows(0);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
@@ -31,7 +31,7 @@ pub(crate) fn divide_alpha_inplace_sse2(mut image: TypedImageViewMut<U8x4>) {
|
||||
}
|
||||
|
||||
#[target_feature(enable = "sse2")]
|
||||
unsafe fn divide_alpha_row_sse2(src_row: &[u32], dst_row: &mut [u32], width: usize) {
|
||||
unsafe fn divide_alpha_row_sse2(src_row: &[U8x4], dst_row: &mut [U8x4], width: usize) {
|
||||
let zero = _mm_setzero_si128();
|
||||
let alpha_mask = _mm_set1_epi32(0xff000000u32 as i32);
|
||||
let shuffle0 = _mm_set_epi8(5, 4, 5, 4, 5, 4, 5, 4, 1, 0, 1, 0, 1, 0, 1, 0);
|
||||
|
||||
@@ -11,7 +11,7 @@ pub(crate) fn multiply_alpha_sse2(
|
||||
mut dst_image: TypedImageViewMut<U8x4>,
|
||||
) {
|
||||
let width = src_image.width().get() as usize;
|
||||
let src_rows = src_image.iter_rows(0, src_image.height().get());
|
||||
let src_rows = src_image.iter_rows(0);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
@@ -25,7 +25,7 @@ pub(crate) fn multiply_alpha_sse2(
|
||||
/// This implementation is twice slowly than native version.
|
||||
#[allow(dead_code)]
|
||||
#[target_feature(enable = "sse2")]
|
||||
unsafe fn multiply_alpha_row_sse2(src_row: &[u32], dst_row: &mut [u32], width: usize) {
|
||||
unsafe fn multiply_alpha_row_sse2(src_row: &[U8x4], dst_row: &mut [U8x4], width: usize) {
|
||||
let mask_alpha_color_odd_255 = _mm_set1_epi32(0xff000000u32 as i32);
|
||||
let div_255 = _mm_set1_epi16(0x8081u16 as i16);
|
||||
|
||||
|
||||
@@ -9,19 +9,18 @@ pub(crate) fn horiz_convolution(
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let coefficients_chunks = coeffs.get_chunks();
|
||||
let mut y_src = offset;
|
||||
|
||||
for out_row in dst_image.iter_rows_mut() {
|
||||
for (out_pixel, coeffs_chunk) in out_row.iter_mut().zip(&coefficients_chunks) {
|
||||
let first_x_src = coeffs_chunk.start;
|
||||
let src_rows = src_image.iter_rows(offset);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for (dst_row, src_row) in dst_rows.zip(src_rows) {
|
||||
for (dst_pixel, coeffs_chunk) in dst_row.iter_mut().zip(&coefficients_chunks) {
|
||||
let first_x_src = coeffs_chunk.start as usize;
|
||||
let mut ss = 0.;
|
||||
let pixels = src_image.iter_horiz(first_x_src, y_src);
|
||||
for (&k, &pixel) in coeffs_chunk.values.iter().zip(pixels) {
|
||||
ss += pixel as f64 * k;
|
||||
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
|
||||
for (&k, &pixel) in coeffs_chunk.values.iter().zip(src_pixels) {
|
||||
ss += pixel.0 as f64 * k;
|
||||
}
|
||||
*out_pixel = ss.round() as f32;
|
||||
dst_pixel.0 = ss.round() as f32;
|
||||
}
|
||||
y_src += 1;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -31,18 +30,17 @@ pub(crate) fn vert_convolution(
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let coefficients_chunks = coeffs.get_chunks();
|
||||
|
||||
for (out_row, coeffs_chunk) in dst_image.iter_rows_mut().zip(coefficients_chunks) {
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for (&coeffs_chunk, dst_row) in coefficients_chunks.iter().zip(dst_rows) {
|
||||
let first_y_src = coeffs_chunk.start;
|
||||
for (x_src, out_pixel) in out_row.iter_mut().enumerate() {
|
||||
for (x_src, dst_pixel) in dst_row.iter_mut().enumerate() {
|
||||
let mut ss = 0.;
|
||||
let mut y_src = first_y_src;
|
||||
for &k in coeffs_chunk.values.iter() {
|
||||
let pixel = src_image.get_pixel(x_src as u32, y_src);
|
||||
ss += pixel as f64 * k;
|
||||
y_src += 1;
|
||||
let src_rows = src_image.iter_rows(first_y_src);
|
||||
for (src_row, &k) in src_rows.zip(coeffs_chunk.values) {
|
||||
let src_pixel = unsafe { src_row.get_unchecked(x_src as usize) };
|
||||
ss += src_pixel.0 as f64 * k;
|
||||
}
|
||||
*out_pixel = ss.round() as f32;
|
||||
dst_pixel.0 = ss.round() as f32;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -9,19 +9,18 @@ pub(crate) fn horiz_convolution(
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let coefficients_chunks = coeffs.get_chunks();
|
||||
let mut y_src = offset;
|
||||
|
||||
for out_row in dst_image.iter_rows_mut() {
|
||||
for (out_pixel, coeffs_chunk) in out_row.iter_mut().zip(&coefficients_chunks) {
|
||||
let first_x_src = coeffs_chunk.start;
|
||||
let src_rows = src_image.iter_rows(offset);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for (dst_row, src_row) in dst_rows.zip(src_rows) {
|
||||
for (dst_pixel, coeffs_chunk) in dst_row.iter_mut().zip(&coefficients_chunks) {
|
||||
let first_x_src = coeffs_chunk.start as usize;
|
||||
let mut ss = 0.;
|
||||
let pixels = src_image.iter_horiz(first_x_src, y_src);
|
||||
for (&k, &pixel) in coeffs_chunk.values.iter().zip(pixels) {
|
||||
ss += pixel as f64 * k;
|
||||
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
|
||||
for (&k, &pixel) in coeffs_chunk.values.iter().zip(src_pixels) {
|
||||
ss += pixel.0 as f64 * k;
|
||||
}
|
||||
*out_pixel = ss.round() as i32;
|
||||
dst_pixel.0 = ss.round() as i32;
|
||||
}
|
||||
y_src += 1;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -31,18 +30,17 @@ pub(crate) fn vert_convolution(
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let coefficients_chunks = coeffs.get_chunks();
|
||||
|
||||
for (out_row, coeffs_chunk) in dst_image.iter_rows_mut().zip(coefficients_chunks) {
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for (&coeffs_chunk, dst_row) in coefficients_chunks.iter().zip(dst_rows) {
|
||||
let first_y_src = coeffs_chunk.start;
|
||||
for (x_src, out_pixel) in out_row.iter_mut().enumerate() {
|
||||
for (x_src, dst_pixel) in dst_row.iter_mut().enumerate() {
|
||||
let mut ss = 0.;
|
||||
let mut y_src = first_y_src;
|
||||
for &k in coeffs_chunk.values.iter() {
|
||||
let pixel = src_image.get_pixel(x_src as u32, y_src);
|
||||
ss += pixel as f64 * k;
|
||||
y_src += 1;
|
||||
let src_rows = src_image.iter_rows(first_y_src);
|
||||
for (src_row, &k) in src_rows.zip(coeffs_chunk.values) {
|
||||
let src_pixel = unsafe { src_row.get_unchecked(x_src as usize) };
|
||||
ss += src_pixel.0 as f64 * k;
|
||||
}
|
||||
*out_pixel = ss.round() as i32;
|
||||
dst_pixel.0 = ss.round() as i32;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -13,6 +13,7 @@ mod filters;
|
||||
mod i32x1;
|
||||
mod optimisations;
|
||||
mod u8x1;
|
||||
mod u8x3;
|
||||
mod u8x4;
|
||||
|
||||
pub(crate) trait Convolution
|
||||
|
||||
@@ -74,8 +74,8 @@ pub(crate) fn vert_convolution(
|
||||
#[inline]
|
||||
#[target_feature(enable = "avx2")]
|
||||
unsafe fn horiz_convolution_8u4x(
|
||||
src_rows: FourRows<u8>,
|
||||
dst_rows: FourRowsMut<u8>,
|
||||
src_rows: FourRows<U8>,
|
||||
dst_rows: FourRowsMut<U8>,
|
||||
coefficients_chunks: &[CoefficientsI16Chunk],
|
||||
precision: u8,
|
||||
) {
|
||||
@@ -124,14 +124,14 @@ unsafe fn horiz_convolution_8u4x(
|
||||
for &coeff in reminder8 {
|
||||
let coeff_i32 = coeff as i32;
|
||||
for i in 0..4 {
|
||||
result_i32x4[i] += s_rows[i].get_unchecked(x).to_owned() as i32 * coeff_i32;
|
||||
result_i32x4[i] += s_rows[i].get_unchecked(x).0.to_owned() as i32 * coeff_i32;
|
||||
}
|
||||
x += 1;
|
||||
}
|
||||
|
||||
let result_u8x4 = result_i32x4.map(|v| optimisations::clip8(v, precision));
|
||||
for i in 0..4 {
|
||||
*d_rows[i].get_unchecked_mut(dst_x) = result_u8x4[i];
|
||||
d_rows[i].get_unchecked_mut(dst_x).0 = result_u8x4[i];
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -144,8 +144,8 @@ unsafe fn horiz_convolution_8u4x(
|
||||
#[inline]
|
||||
#[target_feature(enable = "avx2")]
|
||||
unsafe fn horiz_convolution_8u(
|
||||
src_row: &[u8],
|
||||
dst_row: &mut [u8],
|
||||
src_row: &[U8],
|
||||
dst_row: &mut [U8],
|
||||
coefficients_chunks: &[CoefficientsI16Chunk],
|
||||
precision: u8,
|
||||
) {
|
||||
@@ -188,11 +188,11 @@ unsafe fn horiz_convolution_8u(
|
||||
|
||||
for &coeff in reminder8 {
|
||||
let coeff_i32 = coeff as i32;
|
||||
result_i32 += *src_row.get_unchecked(x) as i32 * coeff_i32;
|
||||
result_i32 += src_row.get_unchecked(x).0 as i32 * coeff_i32;
|
||||
x += 1;
|
||||
}
|
||||
|
||||
*dst_row.get_unchecked_mut(dst_x) = optimisations::clip8(result_i32, precision);
|
||||
dst_row.get_unchecked_mut(dst_x).0 = optimisations::clip8(result_i32, precision);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -200,7 +200,7 @@ unsafe fn horiz_convolution_8u(
|
||||
#[target_feature(enable = "avx2")]
|
||||
unsafe fn vert_convolution_8u(
|
||||
src_img: &TypedImageView<U8>,
|
||||
dst_row: &mut [u8],
|
||||
dst_row: &mut [U8],
|
||||
coeffs: &[i16],
|
||||
bound: Bound,
|
||||
precision: u8,
|
||||
@@ -244,7 +244,7 @@ unsafe fn vert_convolution_8u(
|
||||
y += 2;
|
||||
}
|
||||
|
||||
for s_row in src_img.iter_rows(y_start + y, y_start + y_size) {
|
||||
if let Some(s_row) = src_img.get_row(y_start + y) {
|
||||
let one_coeff = _mm256_set1_epi32(coeffs[y as usize] as i32);
|
||||
|
||||
let row1 = simd_utils::loadu_si256(s_row, x); // top line
|
||||
@@ -261,8 +261,6 @@ unsafe fn vert_convolution_8u(
|
||||
sss2 = _mm256_add_epi32(sss2, _mm256_madd_epi16(lo_hi, one_coeff));
|
||||
let hi_hi = _mm256_unpackhi_epi8(hi_pixels, zero_256);
|
||||
sss3 = _mm256_add_epi32(sss3, _mm256_madd_epi16(hi_hi, one_coeff));
|
||||
|
||||
y += 1;
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
@@ -305,7 +303,7 @@ unsafe fn vert_convolution_8u(
|
||||
y += 2;
|
||||
}
|
||||
|
||||
for s_row in src_img.iter_rows(y_start + y, y_start + y_size) {
|
||||
if let Some(s_row) = src_img.get_row(y_start + y) {
|
||||
let one_coeff = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
|
||||
|
||||
let row1 = simd_utils::loadl_epi64(s_row, x); // top line
|
||||
@@ -316,8 +314,6 @@ unsafe fn vert_convolution_8u(
|
||||
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(lo_pixels, one_coeff));
|
||||
let hi_pixels = _mm_unpackhi_epi8(pixels, zero_128);
|
||||
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(hi_pixels, one_coeff));
|
||||
|
||||
y += 1;
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
@@ -353,12 +349,10 @@ unsafe fn vert_convolution_8u(
|
||||
y += 2;
|
||||
}
|
||||
|
||||
for s_row in src_img.iter_rows(y_start + y, y_start + y_size) {
|
||||
if let Some(s_row) = src_img.get_row(y_start + y) {
|
||||
let pix = simd_utils::mm_cvtepu8_epi32_from_u8(s_row, x);
|
||||
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
|
||||
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
y += 1;
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
@@ -370,10 +364,10 @@ unsafe fn vert_convolution_8u(
|
||||
|
||||
sss = _mm_packs_epi32(sss, sss);
|
||||
let u8x4: [u8; 4] = _mm_cvtsi128_si32(_mm_packus_epi16(sss, sss)).to_le_bytes();
|
||||
*dst_row.get_unchecked_mut(x) = u8x4[0];
|
||||
*dst_row.get_unchecked_mut(x + 1) = u8x4[1];
|
||||
*dst_row.get_unchecked_mut(x + 2) = u8x4[2];
|
||||
*dst_row.get_unchecked_mut(x + 3) = u8x4[3];
|
||||
dst_row.get_unchecked_mut(x).0 = u8x4[0];
|
||||
dst_row.get_unchecked_mut(x + 1).0 = u8x4[1];
|
||||
dst_row.get_unchecked_mut(x + 2).0 = u8x4[2];
|
||||
dst_row.get_unchecked_mut(x + 3).0 = u8x4[3];
|
||||
x += 4;
|
||||
}
|
||||
|
||||
@@ -381,9 +375,9 @@ unsafe fn vert_convolution_8u(
|
||||
let mut ss0 = 1 << (precision - 1);
|
||||
for (dy, &k) in coeffs.iter().take(y_size as usize).enumerate() {
|
||||
let src_pixel = src_img.get_pixel(x as u32, y_start + dy as u32);
|
||||
ss0 += src_pixel as i32 * (k as i32);
|
||||
ss0 += src_pixel.0 as i32 * (k as i32);
|
||||
}
|
||||
*dst_pixel = optimisations::clip8(ss0, precision);
|
||||
dst_pixel.0 = optimisations::clip8(ss0, precision);
|
||||
x += 1;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2,6 +2,7 @@ use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::image_view::{TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::U8;
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) fn horiz_convolution(
|
||||
src_image: TypedImageView<U8>,
|
||||
mut dst_image: TypedImageViewMut<U8>,
|
||||
@@ -13,25 +14,26 @@ pub(crate) fn horiz_convolution(
|
||||
let normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks = normalizer_guard.normalized_i16_chunks(window_size, &bounds);
|
||||
let initial = 1 << (precision - 1);
|
||||
|
||||
let src_rows = src_image.iter_rows(offset);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for (y_dst, dst_row) in dst_rows.enumerate() {
|
||||
let y_src = y_dst as u32 + offset;
|
||||
|
||||
for (dst_row, src_row) in dst_rows.zip(src_rows) {
|
||||
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
|
||||
let first_x_src = coeffs_chunk.start;
|
||||
let first_x_src = coeffs_chunk.start as usize;
|
||||
let ks = coeffs_chunk.values;
|
||||
|
||||
let mut ss0 = 1 << (precision - 1);
|
||||
let src_pixels = src_image.iter_horiz(first_x_src, y_src);
|
||||
let mut ss = initial;
|
||||
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
|
||||
for (&k, &src_pixel) in ks.iter().zip(src_pixels) {
|
||||
ss0 += src_pixel as i32 * (k as i32);
|
||||
ss += src_pixel.0 as i32 * (k as i32);
|
||||
}
|
||||
*dst_pixel = unsafe { optimisations::clip8(ss0, precision) };
|
||||
dst_pixel.0 = unsafe { optimisations::clip8(ss, precision) };
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) fn vert_convolution(
|
||||
src_image: TypedImageView<U8>,
|
||||
mut dst_image: TypedImageViewMut<U8>,
|
||||
@@ -42,6 +44,7 @@ pub(crate) fn vert_convolution(
|
||||
let normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks = normalizer_guard.normalized_i16_chunks(window_size, &bounds);
|
||||
let initial = 1 << (precision - 1);
|
||||
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for (&coeffs_chunk, dst_row) in coefficients_chunks.iter().zip(dst_rows) {
|
||||
@@ -49,12 +52,13 @@ pub(crate) fn vert_convolution(
|
||||
let ks = coeffs_chunk.values;
|
||||
|
||||
for (x_src, dst_pixel) in dst_row.iter_mut().enumerate() {
|
||||
let mut ss0 = 1 << (precision - 1);
|
||||
for (dy, &k) in ks.iter().enumerate() {
|
||||
let src_pixel = src_image.get_pixel(x_src as u32, first_y_src + dy as u32);
|
||||
ss0 += src_pixel as i32 * (k as i32);
|
||||
let mut ss = initial;
|
||||
let src_rows = src_image.iter_rows(first_y_src);
|
||||
for (&k, src_row) in ks.iter().zip(src_rows) {
|
||||
let src_pixel = unsafe { src_row.get_unchecked(x_src as usize) };
|
||||
ss += src_pixel.0 as i32 * (k as i32);
|
||||
}
|
||||
*dst_pixel = unsafe { optimisations::clip8(ss0, precision) };
|
||||
dst_pixel.0 = unsafe { optimisations::clip8(ss, precision) };
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,40 @@
|
||||
use super::{Coefficients, Convolution};
|
||||
use crate::image_view::{TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::U8x3;
|
||||
use crate::CpuExtensions;
|
||||
|
||||
mod native;
|
||||
mod sse4;
|
||||
|
||||
impl Convolution for U8x3 {
|
||||
fn horiz_convolution(
|
||||
src_image: TypedImageView<Self>,
|
||||
dst_image: TypedImageViewMut<Self>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 | CpuExtensions::Sse4_1 => unsafe {
|
||||
sse4::horiz_convolution(src_image, dst_image, offset, coeffs)
|
||||
},
|
||||
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
|
||||
}
|
||||
}
|
||||
|
||||
fn vert_convolution(
|
||||
src_image: TypedImageView<Self>,
|
||||
dst_image: TypedImageViewMut<Self>,
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 | CpuExtensions::Sse4_1 => unsafe {
|
||||
sse4::vert_convolution(src_image, dst_image, coeffs)
|
||||
},
|
||||
_ => native::vert_convolution(src_image, dst_image, coeffs),
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,70 @@
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::image_view::{TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::U8x3;
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) fn horiz_convolution(
|
||||
src_image: TypedImageView<U8x3>,
|
||||
mut dst_image: TypedImageViewMut<U8x3>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks = normalizer_guard.normalized_i16_chunks(window_size, &bounds);
|
||||
let initial = 1 << (precision - 1);
|
||||
|
||||
let src_rows = src_image.iter_rows(offset);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for (dst_row, src_row) in dst_rows.zip(src_rows) {
|
||||
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
|
||||
let first_x_src = coeffs_chunk.start as usize;
|
||||
let mut ss = [initial; 3];
|
||||
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
|
||||
for (&k, src_pixel) in coeffs_chunk.values.iter().zip(src_pixels) {
|
||||
for (i, s) in ss.iter_mut().enumerate() {
|
||||
*s += src_pixel.0[i] as i32 * (k as i32);
|
||||
}
|
||||
}
|
||||
for (i, s) in ss.iter().copied().enumerate() {
|
||||
dst_pixel.0[i] = unsafe { optimisations::clip8(s, precision) };
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) fn vert_convolution(
|
||||
src_image: TypedImageView<U8x3>,
|
||||
mut dst_image: TypedImageViewMut<U8x3>,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks = normalizer_guard.normalized_i16_chunks(window_size, &bounds);
|
||||
let initial = 1 << (precision - 1);
|
||||
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for (&coeffs_chunk, dst_row) in coefficients_chunks.iter().zip(dst_rows) {
|
||||
let first_y_src = coeffs_chunk.start;
|
||||
let ks = coeffs_chunk.values;
|
||||
|
||||
for (x_src, dst_pixel) in dst_row.iter_mut().enumerate() {
|
||||
let mut ss = [initial; 3];
|
||||
let src_rows = src_image.iter_rows(first_y_src);
|
||||
for (&k, src_row) in ks.iter().zip(src_rows) {
|
||||
let src_pixel = unsafe { src_row.get_unchecked(x_src as usize) };
|
||||
for (i, s) in ss.iter_mut().enumerate() {
|
||||
*s += src_pixel.0[i] as i32 * (k as i32);
|
||||
}
|
||||
}
|
||||
for (i, s) in ss.iter().copied().enumerate() {
|
||||
dst_pixel.0[i] = unsafe { optimisations::clip8(s, precision) };
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,24 @@
|
||||
use crate::convolution::Coefficients;
|
||||
use crate::image_view::{TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::U8x3;
|
||||
|
||||
use super::native;
|
||||
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
pub(crate) unsafe fn horiz_convolution(
|
||||
src_image: TypedImageView<U8x3>,
|
||||
dst_image: TypedImageViewMut<U8x3>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
native::horiz_convolution(src_image, dst_image, offset, coeffs);
|
||||
}
|
||||
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
pub(crate) unsafe fn vert_convolution(
|
||||
src_image: TypedImageView<U8x3>,
|
||||
dst_image: TypedImageViewMut<U8x3>,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
native::vert_convolution(src_image, dst_image, coeffs);
|
||||
}
|
||||
@@ -78,8 +78,8 @@ pub(crate) fn vert_convolution(
|
||||
#[inline]
|
||||
#[target_feature(enable = "avx2")]
|
||||
unsafe fn horiz_convolution_8u4x(
|
||||
src_rows: FourRows<u32>,
|
||||
dst_rows: FourRowsMut<u32>,
|
||||
src_rows: FourRows<U8x4>,
|
||||
dst_rows: FourRowsMut<U8x4>,
|
||||
coefficients_chunks: &[CoefficientsI16Chunk],
|
||||
precision: u8,
|
||||
) {
|
||||
@@ -208,8 +208,8 @@ unsafe fn horiz_convolution_8u4x(
|
||||
#[inline]
|
||||
#[target_feature(enable = "avx2")]
|
||||
unsafe fn horiz_convolution_8u(
|
||||
src_row: &[u32],
|
||||
dst_row: &mut [u32],
|
||||
src_row: &[U8x4],
|
||||
dst_row: &mut [U8x4],
|
||||
coefficients_chunks: &[CoefficientsI16Chunk],
|
||||
precision: u8,
|
||||
) {
|
||||
@@ -336,7 +336,7 @@ unsafe fn horiz_convolution_8u(
|
||||
#[target_feature(enable = "avx2")]
|
||||
unsafe fn vert_convolution_8u(
|
||||
src_img: &TypedImageView<U8x4>,
|
||||
dst_row: &mut [u32],
|
||||
dst_row: &mut [U8x4],
|
||||
coeffs: &[i16],
|
||||
bound: Bound,
|
||||
precision: u8,
|
||||
@@ -379,7 +379,7 @@ unsafe fn vert_convolution_8u(
|
||||
y += 2;
|
||||
}
|
||||
|
||||
for s_row in src_img.iter_rows(y_start + y, y_start + y_size) {
|
||||
if let Some(s_row) = src_img.get_row(y_start + y) {
|
||||
let mmk = _mm256_set1_epi32(coeffs[y as usize] as i32);
|
||||
|
||||
let source1 = simd_utils::loadu_si256(s_row, x); // top line
|
||||
@@ -396,8 +396,6 @@ unsafe fn vert_convolution_8u(
|
||||
sss2 = _mm256_add_epi32(sss2, _mm256_madd_epi16(pix, mmk));
|
||||
pix = _mm256_unpackhi_epi8(source, _mm256_setzero_si256());
|
||||
sss3 = _mm256_add_epi32(sss3, _mm256_madd_epi16(pix, mmk));
|
||||
|
||||
y += 1;
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
@@ -440,7 +438,7 @@ unsafe fn vert_convolution_8u(
|
||||
y += 2;
|
||||
}
|
||||
|
||||
for s_row in src_img.iter_rows(y_start + y, y_start + y_size) {
|
||||
if let Some(s_row) = src_img.get_row(y_start + y) {
|
||||
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
|
||||
|
||||
let source1 = simd_utils::loadl_epi64(s_row, x); // top line
|
||||
@@ -451,8 +449,6 @@ unsafe fn vert_convolution_8u(
|
||||
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
|
||||
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
|
||||
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
y += 1;
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
@@ -471,7 +467,7 @@ unsafe fn vert_convolution_8u(
|
||||
x += 2;
|
||||
}
|
||||
|
||||
while x < src_width {
|
||||
if x < src_width {
|
||||
let mut sss = initial;
|
||||
let mut y: u32 = 0;
|
||||
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
|
||||
@@ -488,12 +484,10 @@ unsafe fn vert_convolution_8u(
|
||||
y += 2;
|
||||
}
|
||||
|
||||
for s_row in src_img.iter_rows(y_start + y, y_start + y_size) {
|
||||
if let Some(s_row) = src_img.get_row(y_start + y) {
|
||||
let pix = simd_utils::mm_cvtepu8_epi32(s_row, x);
|
||||
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
|
||||
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
y += 1;
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
@@ -505,7 +499,5 @@ unsafe fn vert_convolution_8u(
|
||||
|
||||
sss = _mm_packs_epi32(sss, sss);
|
||||
*dst_row.get_unchecked_mut(x) = transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss, sss)));
|
||||
|
||||
x += 1;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -13,36 +13,24 @@ pub(crate) fn horiz_convolution(
|
||||
let normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks = normalizer_guard.normalized_i16_chunks(window_size, &bounds);
|
||||
let initial = 1 << (precision - 1);
|
||||
|
||||
let src_rows = src_image.iter_rows(offset);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for (y_dst, dst_row) in dst_rows.enumerate() {
|
||||
let y_src = y_dst as u32 + offset;
|
||||
|
||||
for (dst_row, src_row) in dst_rows.zip(src_rows) {
|
||||
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
|
||||
let first_x_src = coeffs_chunk.start;
|
||||
let first_x_src = coeffs_chunk.start as usize;
|
||||
let ks = coeffs_chunk.values;
|
||||
|
||||
let mut ss0 = 1 << (precision - 1);
|
||||
let mut ss1 = ss0;
|
||||
let mut ss2 = ss0;
|
||||
let mut ss3 = ss0;
|
||||
let src_pixels = src_image.iter_horiz(first_x_src, y_src);
|
||||
let mut ss = [initial; 4];
|
||||
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
|
||||
for (&k, &src_pixel) in ks.iter().zip(src_pixels) {
|
||||
let components: [u8; 4] = src_pixel.to_le_bytes();
|
||||
ss0 += components[0] as i32 * (k as i32);
|
||||
ss1 += components[1] as i32 * (k as i32);
|
||||
ss2 += components[2] as i32 * (k as i32);
|
||||
ss3 += components[3] as i32 * (k as i32);
|
||||
let components: [u8; 4] = src_pixel.0.to_le_bytes();
|
||||
for (i, s) in ss.iter_mut().enumerate() {
|
||||
*s += components[i] as i32 * (k as i32);
|
||||
}
|
||||
}
|
||||
let t: [u8; 4] = unsafe {
|
||||
[
|
||||
optimisations::clip8(ss0, precision),
|
||||
optimisations::clip8(ss1, precision),
|
||||
optimisations::clip8(ss2, precision),
|
||||
optimisations::clip8(ss3, precision),
|
||||
]
|
||||
};
|
||||
*dst_pixel = u32::from_le_bytes(t);
|
||||
dst_pixel.0 =
|
||||
u32::from_le_bytes(ss.map(|v| unsafe { optimisations::clip8(v, precision) }));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -57,34 +45,25 @@ pub(crate) fn vert_convolution(
|
||||
let normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks = normalizer_guard.normalized_i16_chunks(window_size, &bounds);
|
||||
let initial = 1 << (precision - 1);
|
||||
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for (&coeffs_chunk, dst_row) in coefficients_chunks.iter().zip(dst_rows) {
|
||||
let first_y_src = coeffs_chunk.start;
|
||||
let ks = coeffs_chunk.values;
|
||||
|
||||
for (x_src, out_pixel) in dst_row.iter_mut().enumerate() {
|
||||
let mut ss0 = 1 << (precision - 1);
|
||||
let mut ss1 = ss0;
|
||||
let mut ss2 = ss0;
|
||||
let mut ss3 = ss0;
|
||||
for (dy, &k) in ks.iter().enumerate() {
|
||||
let pixel = src_image.get_pixel(x_src as u32, first_y_src + dy as u32);
|
||||
let components: [u8; 4] = pixel.to_le_bytes();
|
||||
ss0 += components[0] as i32 * (k as i32);
|
||||
ss1 += components[1] as i32 * (k as i32);
|
||||
ss2 += components[2] as i32 * (k as i32);
|
||||
ss3 += components[3] as i32 * (k as i32);
|
||||
for (x_src, dst_pixel) in dst_row.iter_mut().enumerate() {
|
||||
let mut ss = [initial; 4];
|
||||
let src_rows = src_image.iter_rows(first_y_src);
|
||||
for (&k, src_row) in ks.iter().zip(src_rows) {
|
||||
let src_pixel = unsafe { src_row.get_unchecked(x_src as usize) };
|
||||
let components: [u8; 4] = src_pixel.0.to_le_bytes();
|
||||
for (i, s) in ss.iter_mut().enumerate() {
|
||||
*s += components[i] as i32 * (k as i32);
|
||||
}
|
||||
}
|
||||
let t: [u8; 4] = unsafe {
|
||||
[
|
||||
optimisations::clip8(ss0, precision),
|
||||
optimisations::clip8(ss1, precision),
|
||||
optimisations::clip8(ss2, precision),
|
||||
optimisations::clip8(ss3, precision),
|
||||
]
|
||||
};
|
||||
*out_pixel = u32::from_le_bytes(t);
|
||||
dst_pixel.0 =
|
||||
u32::from_le_bytes(ss.map(|v| unsafe { optimisations::clip8(v, precision) }));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -77,8 +77,8 @@ pub(crate) fn vert_convolution(
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
unsafe fn horiz_convolution_8u4x(
|
||||
src_rows: FourRows<u32>,
|
||||
dst_rows: FourRowsMut<u32>,
|
||||
src_rows: FourRows<U8x4>,
|
||||
dst_rows: FourRowsMut<U8x4>,
|
||||
coefficients_chunks: &[CoefficientsI16Chunk],
|
||||
precision: u8,
|
||||
) {
|
||||
@@ -163,7 +163,7 @@ unsafe fn horiz_convolution_8u4x(
|
||||
x += 2;
|
||||
}
|
||||
|
||||
for &k in reminder2 {
|
||||
if let Some(&k) = reminder2.get(0) {
|
||||
// [16] xx k0 xx k0 xx k0 xx k0
|
||||
let mmk = _mm_set1_epi32(k as i32);
|
||||
// [16] xx a0 xx b0 xx g0 xx r0
|
||||
@@ -178,8 +178,6 @@ unsafe fn horiz_convolution_8u4x(
|
||||
|
||||
pix = simd_utils::mm_cvtepu8_epi32(s_row3, x);
|
||||
sss3 = _mm_add_epi32(sss3, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
x += 1;
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
@@ -214,8 +212,8 @@ unsafe fn horiz_convolution_8u4x(
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
unsafe fn horiz_convolution_8u(
|
||||
src_row: &[u32],
|
||||
dst_row: &mut [u32],
|
||||
src_row: &[U8x4],
|
||||
dst_row: &mut [U8x4],
|
||||
coefficients_chunks: &[CoefficientsI16Chunk],
|
||||
precision: u8,
|
||||
) {
|
||||
@@ -231,15 +229,12 @@ unsafe fn horiz_convolution_8u(
|
||||
let sh7 = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
// for (dst_x, (&bound, k)) in bounds.iter().zip(coeffs_chunks).enumerate() {
|
||||
let x_start = coeffs_chunk.start as usize;
|
||||
let mut x: usize = 0;
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
|
||||
let mut sss = initial;
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
let reminder1 = coeffs_by_8.remainder();
|
||||
let coeffs_by_8 = coeffs_chunk.values.chunks_exact(8);
|
||||
let reminder8 = coeffs_by_8.remainder();
|
||||
|
||||
for k in coeffs_by_8 {
|
||||
let ksource = simd_utils::loadu_si128(k, 0);
|
||||
@@ -267,8 +262,8 @@ unsafe fn horiz_convolution_8u(
|
||||
x += 8;
|
||||
}
|
||||
|
||||
let coeffs_by_4 = reminder1.chunks_exact(4);
|
||||
coeffs = coeffs_by_4.remainder();
|
||||
let coeffs_by_4 = reminder8.chunks_exact(4);
|
||||
let reminder4 = coeffs_by_4.remainder();
|
||||
|
||||
for k in coeffs_by_4 {
|
||||
let source = simd_utils::loadu_si128(src_row, x + x_start);
|
||||
@@ -285,8 +280,8 @@ unsafe fn horiz_convolution_8u(
|
||||
x += 4;
|
||||
}
|
||||
|
||||
let coeffs_by_2 = coeffs.chunks_exact(2);
|
||||
let reminder1 = coeffs_by_2.remainder();
|
||||
let coeffs_by_2 = reminder4.chunks_exact(2);
|
||||
let reminder2 = coeffs_by_2.remainder();
|
||||
|
||||
for k in coeffs_by_2 {
|
||||
let mmk = simd_utils::ptr_i16_to_set1_epi32(k, 0);
|
||||
@@ -297,12 +292,10 @@ unsafe fn horiz_convolution_8u(
|
||||
x += 2
|
||||
}
|
||||
|
||||
for &k in reminder1 {
|
||||
if let Some(&k) = reminder2.get(0) {
|
||||
let pix = simd_utils::mm_cvtepu8_epi32(src_row, x + x_start);
|
||||
let mmk = _mm_set1_epi32(k as i32);
|
||||
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
x += 1;
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
@@ -321,7 +314,7 @@ unsafe fn horiz_convolution_8u(
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
pub(crate) unsafe fn vert_convolution_8u(
|
||||
src_img: &TypedImageView<U8x4>,
|
||||
dst_row: &mut [u32],
|
||||
dst_row: &mut [U8x4],
|
||||
coeffs: &[i16],
|
||||
bound: Bound,
|
||||
precision: u8,
|
||||
@@ -382,7 +375,7 @@ pub(crate) unsafe fn vert_convolution_8u(
|
||||
y += 2;
|
||||
}
|
||||
|
||||
for s_row in src_img.iter_rows(y_start + y, y_start + y_size) {
|
||||
if let Some(s_row) = src_img.get_row(y_start + y) {
|
||||
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
|
||||
|
||||
let mut source1 = simd_utils::loadu_si128(s_row, xx); // top line
|
||||
@@ -412,8 +405,6 @@ pub(crate) unsafe fn vert_convolution_8u(
|
||||
sss6 = _mm_add_epi32(sss6, _mm_madd_epi16(pix, mmk));
|
||||
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
|
||||
sss7 = _mm_add_epi32(sss7, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
y += 1;
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
@@ -465,18 +456,16 @@ pub(crate) unsafe fn vert_convolution_8u(
|
||||
y += 2;
|
||||
}
|
||||
|
||||
for s_row1 in src_img.iter_rows(y_start + y, y_start + y_size) {
|
||||
if let Some(s_row) = src_img.get_row(y_start + y) {
|
||||
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
|
||||
|
||||
let source1 = simd_utils::loadl_epi64(s_row1, xx); // top line
|
||||
let source1 = simd_utils::loadl_epi64(s_row, xx); // top line
|
||||
|
||||
let source = _mm_unpacklo_epi8(source1, _mm_setzero_si128());
|
||||
let mut pix = _mm_unpacklo_epi8(source, _mm_setzero_si128());
|
||||
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
|
||||
pix = _mm_unpackhi_epi8(source, _mm_setzero_si128());
|
||||
sss1 = _mm_add_epi32(sss1, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
y += 1;
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
@@ -496,7 +485,7 @@ pub(crate) unsafe fn vert_convolution_8u(
|
||||
xx += 2;
|
||||
}
|
||||
|
||||
while xx < src_width {
|
||||
if xx < src_width {
|
||||
let mut sss = initial;
|
||||
let mut y: u32 = 0;
|
||||
|
||||
@@ -514,12 +503,10 @@ pub(crate) unsafe fn vert_convolution_8u(
|
||||
y += 2;
|
||||
}
|
||||
|
||||
for s_row in src_img.iter_rows(y_start + y, y_start + y_size) {
|
||||
if let Some(s_row) = src_img.get_row(y_start + y) {
|
||||
let pix = simd_utils::mm_cvtepu8_epi32(s_row, xx);
|
||||
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
|
||||
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
y += 1;
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
@@ -531,7 +518,5 @@ pub(crate) unsafe fn vert_convolution_8u(
|
||||
|
||||
sss = _mm_packs_epi32(sss, sss);
|
||||
*dst_row.get_unchecked_mut(xx) = transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss, sss)));
|
||||
|
||||
xx += 1;
|
||||
}
|
||||
}
|
||||
|
||||
+40
-24
@@ -1,8 +1,8 @@
|
||||
use std::num::NonZeroU32;
|
||||
|
||||
use crate::image_view::{ImageRows, ImageRowsMut, TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::Pixel;
|
||||
use crate::{ImageBufferError, ImageView, ImageViewMut, InvalidBufferSizeError, PixelType};
|
||||
use crate::pixels::{Pixel, PixelType, U8x3, U8x4, F32, I32, U8};
|
||||
use crate::{ImageBufferError, ImageView, ImageViewMut, InvalidBufferSizeError};
|
||||
|
||||
#[derive(Debug)]
|
||||
enum PixelsContainer<'a> {
|
||||
@@ -24,11 +24,13 @@ pub struct Image<'a> {
|
||||
impl<'a> Image<'a> {
|
||||
/// Create empty image with given dimensions and pixel type.
|
||||
pub fn new(width: NonZeroU32, height: NonZeroU32, pixel_type: PixelType) -> Self {
|
||||
let size = (width.get() * height.get()) as usize;
|
||||
let pixels = if let PixelType::U8 = pixel_type {
|
||||
PixelsContainer::VecU8(vec![0; size])
|
||||
} else {
|
||||
PixelsContainer::VecU32(vec![0; size])
|
||||
let pixels_count = (width.get() * height.get()) as usize;
|
||||
let pixels = match pixel_type {
|
||||
PixelType::U8x3 => PixelsContainer::VecU8(vec![0; pixels_count * U8x3::size()]),
|
||||
PixelType::U8x4 | PixelType::I32 | PixelType::F32 => {
|
||||
PixelsContainer::VecU32(vec![0; pixels_count])
|
||||
}
|
||||
PixelType::U8 => PixelsContainer::VecU8(vec![0; pixels_count]),
|
||||
};
|
||||
Self {
|
||||
width,
|
||||
@@ -156,19 +158,26 @@ impl<'a> Image<'a> {
|
||||
pub fn view(&self) -> ImageView {
|
||||
let buffer = self.buffer();
|
||||
let rows = match self.pixel_type {
|
||||
PixelType::U8x3 => {
|
||||
let pixels = unsafe { buffer.align_to::<U8x3>().1 };
|
||||
ImageRows::U8x3(pixels.chunks_exact(self.width.get() as usize).collect())
|
||||
}
|
||||
PixelType::U8x4 => {
|
||||
let pixels = unsafe { buffer.align_to::<u32>().1 };
|
||||
ImageRows::U8x4(pixels.chunks(self.width.get() as usize).collect())
|
||||
let pixels = unsafe { buffer.align_to::<U8x4>().1 };
|
||||
ImageRows::U8x4(pixels.chunks_exact(self.width.get() as usize).collect())
|
||||
}
|
||||
PixelType::I32 => {
|
||||
let pixels = unsafe { buffer.align_to::<i32>().1 };
|
||||
ImageRows::I32(pixels.chunks(self.width.get() as usize).collect())
|
||||
let pixels = unsafe { buffer.align_to::<I32>().1 };
|
||||
ImageRows::I32(pixels.chunks_exact(self.width.get() as usize).collect())
|
||||
}
|
||||
PixelType::F32 => {
|
||||
let pixels = unsafe { buffer.align_to::<f32>().1 };
|
||||
ImageRows::F32(pixels.chunks(self.width.get() as usize).collect())
|
||||
let pixels = unsafe { buffer.align_to::<F32>().1 };
|
||||
ImageRows::F32(pixels.chunks_exact(self.width.get() as usize).collect())
|
||||
}
|
||||
PixelType::U8 => {
|
||||
let pixels = unsafe { buffer.align_to::<U8>().1 };
|
||||
ImageRows::U8(pixels.chunks_exact(self.width.get() as usize).collect())
|
||||
}
|
||||
PixelType::U8 => ImageRows::U8(buffer.chunks(self.width.get() as usize).collect()),
|
||||
};
|
||||
ImageView::new(self.width, self.height, rows).unwrap()
|
||||
}
|
||||
@@ -180,19 +189,26 @@ impl<'a> Image<'a> {
|
||||
let height = self.height;
|
||||
let buffer = self.buffer_mut();
|
||||
let rows = match pixel_type {
|
||||
PixelType::U8x3 => {
|
||||
let pixels = unsafe { buffer.align_to_mut::<U8x3>().1 };
|
||||
ImageRowsMut::U8x3(pixels.chunks_exact_mut(width.get() as usize).collect())
|
||||
}
|
||||
PixelType::U8x4 => {
|
||||
let pixels = unsafe { buffer.align_to_mut::<u32>().1 };
|
||||
ImageRowsMut::U8x4(pixels.chunks_mut(width.get() as usize).collect())
|
||||
let pixels = unsafe { buffer.align_to_mut::<U8x4>().1 };
|
||||
ImageRowsMut::U8x4(pixels.chunks_exact_mut(width.get() as usize).collect())
|
||||
}
|
||||
PixelType::I32 => {
|
||||
let pixels = unsafe { buffer.align_to_mut::<i32>().1 };
|
||||
ImageRowsMut::I32(pixels.chunks_mut(width.get() as usize).collect())
|
||||
let pixels = unsafe { buffer.align_to_mut::<I32>().1 };
|
||||
ImageRowsMut::I32(pixels.chunks_exact_mut(width.get() as usize).collect())
|
||||
}
|
||||
PixelType::F32 => {
|
||||
let pixels = unsafe { buffer.align_to_mut::<f32>().1 };
|
||||
ImageRowsMut::F32(pixels.chunks_mut(width.get() as usize).collect())
|
||||
let pixels = unsafe { buffer.align_to_mut::<F32>().1 };
|
||||
ImageRowsMut::F32(pixels.chunks_exact_mut(width.get() as usize).collect())
|
||||
}
|
||||
PixelType::U8 => {
|
||||
let pixels = unsafe { buffer.align_to_mut::<U8>().1 };
|
||||
ImageRowsMut::U8(pixels.chunks_exact_mut(width.get() as usize).collect())
|
||||
}
|
||||
PixelType::U8 => ImageRowsMut::U8(buffer.chunks_mut(width.get() as usize).collect()),
|
||||
};
|
||||
ImageViewMut::new(width, height, rows).unwrap()
|
||||
}
|
||||
@@ -205,14 +221,14 @@ where
|
||||
{
|
||||
width: NonZeroU32,
|
||||
height: NonZeroU32,
|
||||
rows: Vec<&'a mut [P::Type]>,
|
||||
rows: Vec<&'a mut [P]>,
|
||||
}
|
||||
|
||||
impl<'a, P> InnerImage<'a, P>
|
||||
where
|
||||
P: Pixel,
|
||||
{
|
||||
pub fn new(width: NonZeroU32, height: NonZeroU32, pixels: &'a mut [P::Type]) -> Self {
|
||||
pub fn new(width: NonZeroU32, height: NonZeroU32, pixels: &'a mut [P]) -> Self {
|
||||
let rows = pixels.chunks_mut(width.get() as usize).collect();
|
||||
Self {
|
||||
width,
|
||||
@@ -224,7 +240,7 @@ where
|
||||
#[inline(always)]
|
||||
pub fn src_view<'s>(&'s self) -> TypedImageView<'s, 'a, P> {
|
||||
let rows = self.rows.as_slice();
|
||||
let rows: &[&[P::Type]] = unsafe { std::mem::transmute(rows) };
|
||||
let rows: &[&[P]] = unsafe { std::mem::transmute(rows) };
|
||||
TypedImageView::new(self.width, self.height, rows)
|
||||
}
|
||||
|
||||
|
||||
+76
-47
@@ -2,7 +2,7 @@ use std::num::NonZeroU32;
|
||||
use std::slice;
|
||||
|
||||
use crate::errors::{CropBoxError, ImageBufferError, ImageRowsError};
|
||||
use crate::pixels::{Pixel, PixelType, U8x4, F32, I32, U8};
|
||||
use crate::pixels::{Pixel, PixelType, U8x3, U8x4, F32, I32, U8};
|
||||
|
||||
pub(crate) type RowMut<'a, 'b, T> = &'a mut &'b mut [T];
|
||||
pub(crate) type TwoRows<'a, T> = (&'a [T], &'a [T]);
|
||||
@@ -26,10 +26,11 @@ pub struct CropBox {
|
||||
/// An immutable rows of image.
|
||||
#[derive(Debug, Clone)]
|
||||
pub enum ImageRows<'a> {
|
||||
U8x4(Vec<&'a [u32]>),
|
||||
I32(Vec<&'a [i32]>),
|
||||
F32(Vec<&'a [f32]>),
|
||||
U8(Vec<&'a [u8]>),
|
||||
U8x3(Vec<&'a [U8x3]>),
|
||||
U8x4(Vec<&'a [U8x4]>),
|
||||
I32(Vec<&'a [I32]>),
|
||||
F32(Vec<&'a [F32]>),
|
||||
U8(Vec<&'a [U8]>),
|
||||
}
|
||||
|
||||
impl<'a> ImageRows<'a> {
|
||||
@@ -39,6 +40,7 @@ impl<'a> ImageRows<'a> {
|
||||
height: NonZeroU32,
|
||||
) -> Result<(), ImageRowsError> {
|
||||
match self {
|
||||
ImageRows::U8x3(rows) => check_rows_count_and_size(width, height, rows),
|
||||
ImageRows::U8x4(rows) => check_rows_count_and_size(width, height, rows),
|
||||
ImageRows::I32(rows) => check_rows_count_and_size(width, height, rows),
|
||||
ImageRows::F32(rows) => check_rows_count_and_size(width, height, rows),
|
||||
@@ -48,6 +50,7 @@ impl<'a> ImageRows<'a> {
|
||||
|
||||
pub fn pixel_type(&self) -> PixelType {
|
||||
match self {
|
||||
Self::U8x3(_) => PixelType::U8x3,
|
||||
Self::U8x4(_) => PixelType::U8x4,
|
||||
Self::I32(_) => PixelType::I32,
|
||||
Self::F32(_) => PixelType::F32,
|
||||
@@ -59,10 +62,11 @@ impl<'a> ImageRows<'a> {
|
||||
/// A mutable rows of image.
|
||||
#[derive(Debug)]
|
||||
pub enum ImageRowsMut<'a> {
|
||||
U8x4(Vec<&'a mut [u32]>),
|
||||
I32(Vec<&'a mut [i32]>),
|
||||
F32(Vec<&'a mut [f32]>),
|
||||
U8(Vec<&'a mut [u8]>),
|
||||
U8x3(Vec<&'a mut [U8x3]>),
|
||||
U8x4(Vec<&'a mut [U8x4]>),
|
||||
I32(Vec<&'a mut [I32]>),
|
||||
F32(Vec<&'a mut [F32]>),
|
||||
U8(Vec<&'a mut [U8]>),
|
||||
}
|
||||
|
||||
impl<'a> ImageRowsMut<'a> {
|
||||
@@ -72,6 +76,7 @@ impl<'a> ImageRowsMut<'a> {
|
||||
height: NonZeroU32,
|
||||
) -> Result<(), ImageRowsError> {
|
||||
match self {
|
||||
Self::U8x3(rows) => check_rows_count_and_size(width, height, rows),
|
||||
Self::U8x4(rows) => check_rows_count_and_size(width, height, rows),
|
||||
Self::I32(rows) => check_rows_count_and_size(width, height, rows),
|
||||
Self::F32(rows) => check_rows_count_and_size(width, height, rows),
|
||||
@@ -81,6 +86,7 @@ impl<'a> ImageRowsMut<'a> {
|
||||
|
||||
pub fn pixel_type(&self) -> PixelType {
|
||||
match self {
|
||||
Self::U8x3(_) => PixelType::U8x3,
|
||||
Self::U8x4(_) => PixelType::U8x4,
|
||||
Self::I32(_) => PixelType::I32,
|
||||
Self::F32(_) => PixelType::F32,
|
||||
@@ -129,19 +135,26 @@ impl<'a> ImageView<'a> {
|
||||
return Err(ImageBufferError::InvalidBufferSize);
|
||||
}
|
||||
let rows = match pixel_type {
|
||||
PixelType::U8x3 => {
|
||||
let pixels = align_buffer_to(buffer)?;
|
||||
ImageRows::U8x3(pixels.chunks_exact(width.get() as usize).collect())
|
||||
}
|
||||
PixelType::U8x4 => {
|
||||
let pixels = align_buffer_to(buffer)?;
|
||||
ImageRows::U8x4(pixels.chunks(width.get() as usize).collect())
|
||||
ImageRows::U8x4(pixels.chunks_exact(width.get() as usize).collect())
|
||||
}
|
||||
PixelType::I32 => {
|
||||
let pixels = align_buffer_to(buffer)?;
|
||||
ImageRows::I32(pixels.chunks(width.get() as usize).collect())
|
||||
ImageRows::I32(pixels.chunks_exact(width.get() as usize).collect())
|
||||
}
|
||||
PixelType::F32 => {
|
||||
let pixels = align_buffer_to(buffer)?;
|
||||
ImageRows::F32(pixels.chunks(width.get() as usize).collect())
|
||||
ImageRows::F32(pixels.chunks_exact(width.get() as usize).collect())
|
||||
}
|
||||
PixelType::U8 => {
|
||||
let pixels = align_buffer_to(buffer)?;
|
||||
ImageRows::U8(pixels.chunks_exact(width.get() as usize).collect())
|
||||
}
|
||||
PixelType::U8 => ImageRows::U8(buffer.chunks(width.get() as usize).collect()),
|
||||
};
|
||||
Ok(Self {
|
||||
width,
|
||||
@@ -251,6 +264,19 @@ impl<'a> ImageView<'a> {
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
pub(crate) fn u8x3_image(&self) -> Option<TypedImageView<U8x3>> {
|
||||
if let ImageRows::U8x3(ref rows) = self.rows {
|
||||
Some(TypedImageView {
|
||||
width: self.width,
|
||||
height: self.height,
|
||||
crop_box: self.crop_box,
|
||||
rows,
|
||||
})
|
||||
} else {
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn u32_image(&self) -> Option<TypedImageView<U8x4>> {
|
||||
if let ImageRows::U8x4(ref rows) = self.rows {
|
||||
Some(TypedImageView {
|
||||
@@ -312,14 +338,14 @@ where
|
||||
width: NonZeroU32,
|
||||
height: NonZeroU32,
|
||||
crop_box: CropBox,
|
||||
rows: &'a [&'b [P::Type]],
|
||||
rows: &'a [&'b [P]],
|
||||
}
|
||||
|
||||
impl<'a, 'b, P> TypedImageView<'a, 'b, P>
|
||||
where
|
||||
P: Pixel,
|
||||
{
|
||||
pub fn new(width: NonZeroU32, height: NonZeroU32, rows: &'a [&'b [P::Type]]) -> Self {
|
||||
pub fn new(width: NonZeroU32, height: NonZeroU32, rows: &'a [&'b [P]]) -> Self {
|
||||
Self {
|
||||
width,
|
||||
height,
|
||||
@@ -349,7 +375,7 @@ where
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn get_pixel(&self, x: u32, y: u32) -> P::Type {
|
||||
pub(crate) fn get_pixel(&self, x: u32, y: u32) -> P {
|
||||
self.rows[y as usize][x as usize]
|
||||
}
|
||||
|
||||
@@ -358,7 +384,7 @@ where
|
||||
&'s self,
|
||||
start_y: u32,
|
||||
max_y: u32,
|
||||
) -> impl Iterator<Item = FourRows<'b, P::Type>> + 's {
|
||||
) -> impl Iterator<Item = FourRows<'b, P>> + 's {
|
||||
let start_y = start_y as usize;
|
||||
let max_y = max_y.min(self.height.get()) as usize;
|
||||
let rows = self.rows.get(start_y..max_y).unwrap_or_else(|| &[]);
|
||||
@@ -373,7 +399,7 @@ where
|
||||
&'s self,
|
||||
start_y: u32,
|
||||
max_y: u32,
|
||||
) -> impl Iterator<Item = TwoRows<'b, P::Type>> + 's {
|
||||
) -> impl Iterator<Item = TwoRows<'b, P>> + 's {
|
||||
let start_y = start_y as usize;
|
||||
let max_y = max_y.min(self.height.get()) as usize;
|
||||
let rows = self.rows.get(start_y..max_y).unwrap_or_else(|| &[]);
|
||||
@@ -384,30 +410,14 @@ where
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) fn iter_rows<'s>(
|
||||
&'s self,
|
||||
start_y: u32,
|
||||
max_y: u32,
|
||||
) -> impl Iterator<Item = &'b [P::Type]> + 's {
|
||||
pub(crate) fn iter_rows<'s>(&'s self, start_y: u32) -> impl Iterator<Item = &'b [P]> + 's {
|
||||
let start_y = start_y as usize;
|
||||
let max_y = max_y.min(self.height.get()) as usize;
|
||||
let rows = self.rows.get(start_y..max_y).unwrap_or_else(|| &[]);
|
||||
let rows = self.rows.get(start_y..).unwrap_or_else(|| &[]);
|
||||
rows.iter().copied()
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) fn iter_horiz(&self, x: u32, y: u32) -> &'b [P::Type] {
|
||||
if let Some(&row) = self.rows.get(y as usize) {
|
||||
let start_pos = x as usize;
|
||||
if let Some(res) = row.get(start_pos..) {
|
||||
return res;
|
||||
}
|
||||
}
|
||||
&[]
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) fn get_row(&self, y: u32) -> Option<&'b [P::Type]> {
|
||||
pub(crate) fn get_row(&self, y: u32) -> Option<&'b [P]> {
|
||||
self.rows.get(y as usize).copied()
|
||||
}
|
||||
|
||||
@@ -417,7 +427,7 @@ where
|
||||
mut y: f64,
|
||||
step: f64,
|
||||
max_count: usize,
|
||||
) -> impl Iterator<Item = &'b [P::Type]> + 's {
|
||||
) -> impl Iterator<Item = &'b [P]> + 's {
|
||||
let steps = (self.height.get() as f64 - y) / step;
|
||||
let steps = (steps.max(0.).ceil() as usize).min(max_count);
|
||||
(0..steps).map(move |_| {
|
||||
@@ -462,19 +472,26 @@ impl<'a> ImageViewMut<'a> {
|
||||
return Err(ImageBufferError::InvalidBufferSize);
|
||||
}
|
||||
let rows = match pixel_type {
|
||||
PixelType::U8x3 => {
|
||||
let pixels = align_buffer_to_mut(buffer)?;
|
||||
ImageRowsMut::U8x3(pixels.chunks_exact_mut(width.get() as usize).collect())
|
||||
}
|
||||
PixelType::U8x4 => {
|
||||
let pixels = align_buffer_to_mut(buffer)?;
|
||||
ImageRowsMut::U8x4(pixels.chunks_mut(width.get() as usize).collect())
|
||||
ImageRowsMut::U8x4(pixels.chunks_exact_mut(width.get() as usize).collect())
|
||||
}
|
||||
PixelType::I32 => {
|
||||
let pixels = align_buffer_to_mut(buffer)?;
|
||||
ImageRowsMut::I32(pixels.chunks_mut(width.get() as usize).collect())
|
||||
ImageRowsMut::I32(pixels.chunks_exact_mut(width.get() as usize).collect())
|
||||
}
|
||||
PixelType::F32 => {
|
||||
let pixels = align_buffer_to_mut(buffer)?;
|
||||
ImageRowsMut::F32(pixels.chunks_mut(width.get() as usize).collect())
|
||||
ImageRowsMut::F32(pixels.chunks_exact_mut(width.get() as usize).collect())
|
||||
}
|
||||
PixelType::U8 => {
|
||||
let pixels = align_buffer_to_mut(buffer)?;
|
||||
ImageRowsMut::U8(pixels.chunks_exact_mut(width.get() as usize).collect())
|
||||
}
|
||||
PixelType::U8 => ImageRowsMut::U8(buffer.chunks_mut(width.get() as usize).collect()),
|
||||
};
|
||||
Ok(Self {
|
||||
width,
|
||||
@@ -498,6 +515,18 @@ impl<'a> ImageViewMut<'a> {
|
||||
self.height
|
||||
}
|
||||
|
||||
pub(crate) fn u8x3_image<'s>(&'s mut self) -> Option<TypedImageViewMut<'s, 'a, U8x3>> {
|
||||
if let ImageRowsMut::U8x3(rows) = &mut self.rows {
|
||||
Some(TypedImageViewMut {
|
||||
width: self.width,
|
||||
height: self.height,
|
||||
rows,
|
||||
})
|
||||
} else {
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn u32_image<'s>(&'s mut self) -> Option<TypedImageViewMut<'s, 'a, U8x4>> {
|
||||
if let ImageRowsMut::U8x4(rows) = &mut self.rows {
|
||||
Some(TypedImageViewMut {
|
||||
@@ -554,14 +583,14 @@ where
|
||||
{
|
||||
width: NonZeroU32,
|
||||
height: NonZeroU32,
|
||||
rows: &'a mut [&'b mut [P::Type]],
|
||||
rows: &'a mut [&'b mut [P]],
|
||||
}
|
||||
|
||||
impl<'a, 'b, P> TypedImageViewMut<'a, 'b, P>
|
||||
where
|
||||
P: Pixel,
|
||||
{
|
||||
pub fn new(width: NonZeroU32, height: NonZeroU32, rows: &'a mut [&'b mut [P::Type]]) -> Self {
|
||||
pub fn new(width: NonZeroU32, height: NonZeroU32, rows: &'a mut [&'b mut [P]]) -> Self {
|
||||
Self {
|
||||
width,
|
||||
height,
|
||||
@@ -580,12 +609,12 @@ where
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub fn iter_rows_mut(&mut self) -> slice::IterMut<&'b mut [P::Type]> {
|
||||
pub fn iter_rows_mut(&mut self) -> slice::IterMut<&'b mut [P]> {
|
||||
self.rows.iter_mut()
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub fn iter_4_rows_mut<'s>(&'s mut self) -> impl Iterator<Item = FourRowsMut<'s, 'b, P::Type>> {
|
||||
pub fn iter_4_rows_mut<'s>(&'s mut self) -> impl Iterator<Item = FourRowsMut<'s, 'b, P>> {
|
||||
self.rows.chunks_exact_mut(4).map(|rows| match rows {
|
||||
[a, b, c, d] => (a, b, c, d),
|
||||
_ => unreachable!(),
|
||||
@@ -593,7 +622,7 @@ where
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub fn get_row_mut<'s>(&'s mut self, y: u32) -> Option<RowMut<'s, 'b, P::Type>> {
|
||||
pub fn get_row_mut<'s>(&'s mut self, y: u32) -> Option<RowMut<'s, 'b, P>> {
|
||||
self.rows.get_mut(y as usize)
|
||||
}
|
||||
}
|
||||
|
||||
+1
-1
@@ -14,7 +14,7 @@ mod convolution;
|
||||
mod errors;
|
||||
mod image;
|
||||
mod image_view;
|
||||
mod pixels;
|
||||
pub mod pixels;
|
||||
mod resizer;
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod simd_utils;
|
||||
|
||||
+46
-19
@@ -1,7 +1,9 @@
|
||||
//! Contains types of pixels.
|
||||
use std::mem::size_of;
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq)]
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum PixelType {
|
||||
U8x3,
|
||||
U8x4,
|
||||
I32,
|
||||
F32,
|
||||
@@ -11,38 +13,52 @@ pub enum PixelType {
|
||||
impl PixelType {
|
||||
pub(crate) fn size(&self) -> usize {
|
||||
match self {
|
||||
Self::U8x3 => 3,
|
||||
Self::U8 => 1,
|
||||
_ => 4,
|
||||
}
|
||||
}
|
||||
|
||||
/// Returns `true` is given buffer is aligned by the alignment of pixel.
|
||||
pub(crate) fn is_aligned(&self, buffer: &[u8]) -> bool {
|
||||
match self {
|
||||
Self::U8x4 => unsafe { buffer.align_to::<u32>().0.is_empty() },
|
||||
Self::I32 => unsafe { buffer.align_to::<i32>().0.is_empty() },
|
||||
Self::F32 => unsafe { buffer.align_to::<f32>().0.is_empty() },
|
||||
Self::U8x3 => unsafe { buffer.align_to::<U8x3>().0.is_empty() },
|
||||
Self::U8x4 => unsafe { buffer.align_to::<U8x4>().0.is_empty() },
|
||||
Self::I32 => unsafe { buffer.align_to::<I32>().0.is_empty() },
|
||||
Self::F32 => unsafe { buffer.align_to::<F32>().0.is_empty() },
|
||||
Self::U8 => true,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) trait Pixel {
|
||||
type Type: Copy;
|
||||
|
||||
fn size() -> usize {
|
||||
size_of::<Self::Type>()
|
||||
}
|
||||
|
||||
/// Additional information about pixel type.
|
||||
pub trait Pixel
|
||||
where
|
||||
Self: Copy + Sized,
|
||||
{
|
||||
fn pixel_type() -> PixelType;
|
||||
|
||||
/// Size of pixel in bytes
|
||||
///
|
||||
/// Example:
|
||||
/// ```
|
||||
/// # use fast_image_resize::pixels::{U8x3, U8, Pixel};
|
||||
/// assert_eq!(U8x3::size(), 3);
|
||||
/// assert_eq!(U8::size(), 1);
|
||||
/// ```
|
||||
fn size() -> usize {
|
||||
size_of::<Self>()
|
||||
}
|
||||
}
|
||||
|
||||
macro_rules! pixel_struct {
|
||||
($name:ident, $type:tt, $pixel_type:expr) => {
|
||||
pub struct $name;
|
||||
($name:ident, $type:tt, $pixel_type:expr, $doc:expr) => {
|
||||
#[doc = $doc]
|
||||
#[derive(Debug, Clone, Copy, PartialEq)]
|
||||
#[repr(C)]
|
||||
pub struct $name(pub $type);
|
||||
|
||||
impl Pixel for $name {
|
||||
type Type = $type;
|
||||
|
||||
fn pixel_type() -> PixelType {
|
||||
$pixel_type
|
||||
}
|
||||
@@ -50,7 +66,18 @@ macro_rules! pixel_struct {
|
||||
};
|
||||
}
|
||||
|
||||
pixel_struct!(U8x4, u32, PixelType::U8x4);
|
||||
pixel_struct!(I32, i32, PixelType::I32);
|
||||
pixel_struct!(F32, f32, PixelType::F32);
|
||||
pixel_struct!(U8, u8, PixelType::U8);
|
||||
pixel_struct!(U8, u8, PixelType::U8, "One byte per pixel");
|
||||
pixel_struct!(
|
||||
U8x3,
|
||||
[u8; 3],
|
||||
PixelType::U8x3,
|
||||
"Three bytes per pixel (e.g. RGB)"
|
||||
);
|
||||
pixel_struct!(
|
||||
U8x4,
|
||||
u32,
|
||||
PixelType::U8x4,
|
||||
"Four bytes per pixel (RGBA, RGBx, CMYK and other)"
|
||||
);
|
||||
pixel_struct!(I32, i32, PixelType::I32, "One `i32` component per pixel");
|
||||
pixel_struct!(F32, f32, PixelType::F32, "One `f32` component per pixel");
|
||||
|
||||
+9
-2
@@ -6,7 +6,7 @@ use crate::image::InnerImage;
|
||||
use crate::image_view::{ImageView, ImageViewMut, TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::{Pixel, PixelType};
|
||||
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum CpuExtensions {
|
||||
None,
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
@@ -87,6 +87,13 @@ impl Resizer {
|
||||
return Err(DifferentTypesOfPixelsError);
|
||||
}
|
||||
match src_image.pixel_type() {
|
||||
PixelType::U8x3 => {
|
||||
if let Some(src_rows) = src_image.u8x3_image() {
|
||||
if let Some(dst_rows) = dst_image.u8x3_image() {
|
||||
self.resize_inner(src_rows, dst_rows);
|
||||
}
|
||||
}
|
||||
}
|
||||
PixelType::U8x4 => {
|
||||
if let Some(src_rows) = src_image.u32_image() {
|
||||
if let Some(dst_rows) = dst_image.u32_image() {
|
||||
@@ -193,7 +200,7 @@ fn get_temp_image_from_buffer<P: Pixel>(
|
||||
if buffer.len() < buf_size {
|
||||
buffer.resize(buf_size, 0);
|
||||
}
|
||||
let pixels = unsafe { buffer.align_to_mut::<P::Type>().1 };
|
||||
let pixels = unsafe { buffer.align_to_mut::<P>().1 };
|
||||
InnerImage::new(width, height, &mut pixels[0..pixels_count])
|
||||
}
|
||||
|
||||
|
||||
+5
-4
@@ -1,3 +1,4 @@
|
||||
use crate::pixels::{U8x4, U8};
|
||||
use std::arch::x86_64::*;
|
||||
use std::intrinsics::transmute;
|
||||
|
||||
@@ -17,25 +18,25 @@ pub unsafe fn loadl_epi64<T>(buf: &[T], index: usize) -> __m128i {
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub unsafe fn mm_cvtepu8_epi32(buf: &[u32], index: usize) -> __m128i {
|
||||
pub unsafe fn mm_cvtepu8_epi32(buf: &[U8x4], index: usize) -> __m128i {
|
||||
let v: i32 = transmute(*buf.get_unchecked(index));
|
||||
_mm_cvtepu8_epi32(_mm_cvtsi32_si128(v))
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub unsafe fn mm_cvtepu8_epi32_from_u8(buf: &[u8], index: usize) -> __m128i {
|
||||
pub unsafe fn mm_cvtepu8_epi32_from_u8(buf: &[U8], index: usize) -> __m128i {
|
||||
let ptr = buf.get_unchecked(index..).as_ptr() as *const i32;
|
||||
_mm_cvtepu8_epi32(_mm_cvtsi32_si128(*ptr))
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub unsafe fn mm_cvtsi32_si128_from_u32(buf: &[u32], index: usize) -> __m128i {
|
||||
pub unsafe fn mm_cvtsi32_si128_from_u32(buf: &[U8x4], index: usize) -> __m128i {
|
||||
let v: i32 = transmute(*buf.get_unchecked(index));
|
||||
_mm_cvtsi32_si128(v)
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub unsafe fn mm_cvtsi32_si128_from_u8(buf: &[u8], index: usize) -> __m128i {
|
||||
pub unsafe fn mm_cvtsi32_si128_from_u8(buf: &[U8], index: usize) -> __m128i {
|
||||
let ptr = buf.get_unchecked(index..).as_ptr() as *const i32;
|
||||
_mm_cvtsi32_si128(*ptr)
|
||||
}
|
||||
|
||||
+13
-12
@@ -1,11 +1,12 @@
|
||||
use std::num::NonZeroU32;
|
||||
|
||||
use fast_image_resize::pixels::U8x4;
|
||||
use fast_image_resize::{
|
||||
CpuExtensions, Image, ImageRows, ImageRowsMut, ImageView, ImageViewMut, MulDiv, PixelType,
|
||||
};
|
||||
|
||||
const fn p(r: u8, g: u8, b: u8, a: u8) -> u32 {
|
||||
u32::from_le_bytes([r, g, b, a])
|
||||
const fn p(r: u8, g: u8, b: u8, a: u8) -> U8x4 {
|
||||
U8x4(u32::from_le_bytes([r, g, b, a]))
|
||||
}
|
||||
|
||||
// Multiplies by alpha
|
||||
@@ -17,13 +18,13 @@ fn multiply_alpha_test(cpu_extensions: CpuExtensions) {
|
||||
let src_pixels = [p(255, 128, 0, 128), p(255, 128, 0, 255), p(255, 128, 0, 0)];
|
||||
let res_pixels = [p(128, 64, 0, 128), p(255, 128, 0, 255), p(0, 0, 0, 0)];
|
||||
|
||||
let mut src_rows: [Vec<u32>; 3] = [
|
||||
let mut src_rows: [Vec<U8x4>; 3] = [
|
||||
vec![src_pixels[0]; width as usize],
|
||||
vec![src_pixels[1]; width as usize],
|
||||
vec![src_pixels[2]; width as usize],
|
||||
];
|
||||
|
||||
let rows: Vec<&[u32]> = src_rows.iter().map(|r| r.as_ref()).collect();
|
||||
let rows: Vec<&[U8x4]> = src_rows.iter().map(|r| r.as_ref()).collect();
|
||||
let src_image_view = ImageView::new(
|
||||
NonZeroU32::new(width).unwrap(),
|
||||
NonZeroU32::new(height).unwrap(),
|
||||
@@ -51,12 +52,12 @@ fn multiply_alpha_test(cpu_extensions: CpuExtensions) {
|
||||
let dst_rows = dst_pixels.chunks_exact(width as usize);
|
||||
for (row, &valid_pixel) in dst_rows.zip(res_pixels.iter()) {
|
||||
for &pixel in row.iter() {
|
||||
assert_eq!(pixel.to_le_bytes(), valid_pixel.to_le_bytes());
|
||||
assert_eq!(pixel, valid_pixel.0);
|
||||
}
|
||||
}
|
||||
|
||||
// Inplace
|
||||
let rows: Vec<&mut [u32]> = src_rows.iter_mut().map(|r| r.as_mut()).collect();
|
||||
let rows: Vec<&mut [U8x4]> = src_rows.iter_mut().map(|r| r.as_mut()).collect();
|
||||
let mut image_view = ImageViewMut::new(
|
||||
NonZeroU32::new(width).unwrap(),
|
||||
NonZeroU32::new(height).unwrap(),
|
||||
@@ -69,7 +70,7 @@ fn multiply_alpha_test(cpu_extensions: CpuExtensions) {
|
||||
|
||||
for (row, &valid_pixel) in src_rows.iter().zip(res_pixels.iter()) {
|
||||
for &pixel in row.iter() {
|
||||
assert_eq!(pixel.to_le_bytes(), valid_pixel.to_le_bytes());
|
||||
assert_eq!(pixel, valid_pixel);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -98,13 +99,13 @@ fn divide_alpha_test(cpu_extensions: CpuExtensions) {
|
||||
let src_pixels = [p(128, 64, 0, 128), p(255, 128, 0, 255), p(255, 128, 0, 0)];
|
||||
let res_pixels = [p(255, 127, 0, 128), p(255, 128, 0, 255), p(0, 0, 0, 0)];
|
||||
|
||||
let mut src_rows: [Vec<u32>; 3] = [
|
||||
let mut src_rows: [Vec<U8x4>; 3] = [
|
||||
vec![src_pixels[0]; width as usize],
|
||||
vec![src_pixels[1]; width as usize],
|
||||
vec![src_pixels[2]; width as usize],
|
||||
];
|
||||
|
||||
let rows: Vec<&[u32]> = src_rows.iter().map(|r| r.as_ref()).collect();
|
||||
let rows: Vec<&[U8x4]> = src_rows.iter().map(|r| r.as_ref()).collect();
|
||||
let src_image_view = ImageView::new(
|
||||
NonZeroU32::new(width).unwrap(),
|
||||
NonZeroU32::new(height).unwrap(),
|
||||
@@ -132,12 +133,12 @@ fn divide_alpha_test(cpu_extensions: CpuExtensions) {
|
||||
let dst_rows = dst_pixels.chunks_exact(width as usize);
|
||||
for (row, &valid_pixel) in dst_rows.zip(res_pixels.iter()) {
|
||||
for &pixel in row.iter() {
|
||||
assert_eq!(pixel.to_le_bytes(), valid_pixel.to_le_bytes());
|
||||
assert_eq!(pixel, valid_pixel.0);
|
||||
}
|
||||
}
|
||||
|
||||
// Inplace
|
||||
let rows: Vec<&mut [u32]> = src_rows.iter_mut().map(|r| r.as_mut()).collect();
|
||||
let rows: Vec<&mut [U8x4]> = src_rows.iter_mut().map(|r| r.as_mut()).collect();
|
||||
let mut image_view = ImageViewMut::new(
|
||||
NonZeroU32::new(width).unwrap(),
|
||||
NonZeroU32::new(height).unwrap(),
|
||||
@@ -148,7 +149,7 @@ fn divide_alpha_test(cpu_extensions: CpuExtensions) {
|
||||
|
||||
for (row, &valid_pixel) in src_rows.iter().zip(res_pixels.iter()) {
|
||||
for &pixel in row.iter() {
|
||||
assert_eq!(pixel.to_le_bytes(), valid_pixel.to_le_bytes());
|
||||
assert_eq!(pixel, valid_pixel);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+1
-1
@@ -31,7 +31,7 @@ fn resize_image_example() {
|
||||
.multiply_alpha_inplace(&mut src_image.view_mut())
|
||||
.unwrap();
|
||||
|
||||
// Create wrapper that own data of destination image
|
||||
// Create container for data of destination image
|
||||
let dst_width = NonZeroU32::new(1024).unwrap();
|
||||
let dst_height = NonZeroU32::new(768).unwrap();
|
||||
let mut dst_image = fr::Image::new(dst_width, dst_height, src_image.pixel_type());
|
||||
|
||||
+136
-152
@@ -3,8 +3,9 @@ use std::num::NonZeroU32;
|
||||
|
||||
use image::codecs::png::PngEncoder;
|
||||
use image::io::Reader as ImageReader;
|
||||
use image::{ColorType, GenericImageView};
|
||||
use image::{ColorType, DynamicImage, GenericImageView};
|
||||
|
||||
use fast_image_resize::pixels::*;
|
||||
use fast_image_resize::{
|
||||
CpuExtensions, DifferentTypesOfPixelsError, FilterType, Image, ImageView, PixelType, ResizeAlg,
|
||||
Resizer,
|
||||
@@ -26,22 +27,6 @@ fn get_source_image_u8x4() -> Image<'static> {
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
fn get_source_image_u8x1() -> Image<'static> {
|
||||
let img = ImageReader::open("./data/nasa-4928x3279.png")
|
||||
.unwrap()
|
||||
.decode()
|
||||
.unwrap();
|
||||
let width = img.width();
|
||||
let height = img.height();
|
||||
Image::from_vec_u8(
|
||||
NonZeroU32::new(width).unwrap(),
|
||||
NonZeroU32::new(height).unwrap(),
|
||||
img.to_luma8().into_raw(),
|
||||
PixelType::U8,
|
||||
)
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
fn get_small_source_image() -> Image<'static> {
|
||||
let img = ImageReader::open("./data/nasa-852x567.png")
|
||||
.unwrap()
|
||||
@@ -70,6 +55,7 @@ fn save_result(image: &Image, name: &str) {
|
||||
std::fs::create_dir_all("./data/result").unwrap();
|
||||
let mut file = File::create(format!("./data/result/{}.png", name)).unwrap();
|
||||
let color_type = match image.pixel_type() {
|
||||
PixelType::U8x3 => ColorType::Rgb8,
|
||||
PixelType::U8x4 => ColorType::Rgba8,
|
||||
PixelType::U8 => ColorType::L8,
|
||||
_ => panic!("Unsupported type of pixels"),
|
||||
@@ -84,63 +70,6 @@ fn save_result(image: &Image, name: &str) {
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn resize_wo_simd_lanczos3_test() {
|
||||
let image = get_source_image_u8x4();
|
||||
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
|
||||
unsafe {
|
||||
resizer.set_cpu_extensions(CpuExtensions::None);
|
||||
}
|
||||
let new_height = get_new_height(&image.view(), NEW_WIDTH);
|
||||
let mut result = Image::new(
|
||||
NonZeroU32::new(NEW_WIDTH).unwrap(),
|
||||
NonZeroU32::new(new_height).unwrap(),
|
||||
image.pixel_type(),
|
||||
);
|
||||
assert!(resizer
|
||||
.resize(&image.view(), &mut result.view_mut())
|
||||
.is_ok());
|
||||
save_result(&result, "u8x4-lanczos3-native");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn resize_sse4_lanczos3_test() {
|
||||
let image = get_source_image_u8x4();
|
||||
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
|
||||
unsafe {
|
||||
resizer.set_cpu_extensions(CpuExtensions::Sse4_1);
|
||||
}
|
||||
let new_height = get_new_height(&image.view(), NEW_WIDTH);
|
||||
let mut result = Image::new(
|
||||
NonZeroU32::new(NEW_WIDTH).unwrap(),
|
||||
NonZeroU32::new(new_height).unwrap(),
|
||||
image.pixel_type(),
|
||||
);
|
||||
assert!(resizer
|
||||
.resize(&image.view(), &mut result.view_mut())
|
||||
.is_ok());
|
||||
save_result(&result, "u8x4-lanczos3-sse4");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn resize_avx2_lanczos3_test() {
|
||||
let image = get_source_image_u8x4();
|
||||
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
|
||||
unsafe {
|
||||
resizer.set_cpu_extensions(CpuExtensions::Avx2);
|
||||
}
|
||||
let new_height = get_new_height(&image.view(), NEW_WIDTH);
|
||||
let mut result = Image::new(
|
||||
NonZeroU32::new(NEW_WIDTH).unwrap(),
|
||||
NonZeroU32::new(new_height).unwrap(),
|
||||
image.pixel_type(),
|
||||
);
|
||||
assert!(resizer
|
||||
.resize(&image.view(), &mut result.view_mut())
|
||||
.is_ok());
|
||||
save_result(&result, "u8x4-lanczos3-avx2");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn resize_avx2_lanczos3_upscale_test() {
|
||||
let image = get_small_source_image();
|
||||
@@ -160,44 +89,6 @@ fn resize_avx2_lanczos3_upscale_test() {
|
||||
save_result(&result, "u8x4-lanczos3_upscale-avx2");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn resize_nearest_test() {
|
||||
let image = get_source_image_u8x4();
|
||||
let mut resizer = Resizer::new(ResizeAlg::Nearest);
|
||||
unsafe {
|
||||
resizer.set_cpu_extensions(CpuExtensions::None);
|
||||
}
|
||||
let new_height = get_new_height(&image.view(), NEW_WIDTH);
|
||||
let mut result = Image::new(
|
||||
NonZeroU32::new(NEW_WIDTH).unwrap(),
|
||||
NonZeroU32::new(new_height).unwrap(),
|
||||
image.pixel_type(),
|
||||
);
|
||||
assert!(resizer
|
||||
.resize(&image.view(), &mut result.view_mut())
|
||||
.is_ok());
|
||||
save_result(&result, "u8x4-nearest-native");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn resize_super_sampling_test() {
|
||||
let image = get_source_image_u8x4();
|
||||
let mut resizer = Resizer::new(ResizeAlg::SuperSampling(FilterType::Lanczos3, 2));
|
||||
unsafe {
|
||||
resizer.set_cpu_extensions(CpuExtensions::Avx2);
|
||||
}
|
||||
let new_height = get_new_height(&image.view(), NEW_WIDTH);
|
||||
let mut result = Image::new(
|
||||
NonZeroU32::new(NEW_WIDTH).unwrap(),
|
||||
NonZeroU32::new(new_height).unwrap(),
|
||||
image.pixel_type(),
|
||||
);
|
||||
assert!(resizer
|
||||
.resize(&image.view(), &mut result.view_mut())
|
||||
.is_ok());
|
||||
save_result(&result, "u8x4-super_sampling-avx2");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn try_resize_to_other_pixel_type() {
|
||||
let src_image = get_source_image_u8x4();
|
||||
@@ -213,14 +104,81 @@ fn try_resize_to_other_pixel_type() {
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn resize_nearest_u8x1() {
|
||||
let image = get_source_image_u8x1();
|
||||
assert!(matches!(image.pixel_type(), PixelType::U8));
|
||||
trait PixelExt: Pixel {
|
||||
fn pixel_type_str() -> &'static str {
|
||||
match Self::pixel_type() {
|
||||
PixelType::U8 => "u8",
|
||||
PixelType::U8x3 => "u8x3",
|
||||
PixelType::U8x4 => "u8x4",
|
||||
PixelType::I32 => "i32",
|
||||
PixelType::F32 => "f32",
|
||||
}
|
||||
}
|
||||
|
||||
let mut resizer = Resizer::new(ResizeAlg::Nearest);
|
||||
fn load_src_image() -> Image<'static> {
|
||||
let img = ImageReader::open("./data/nasa-4928x3279.png")
|
||||
.unwrap()
|
||||
.decode()
|
||||
.unwrap();
|
||||
Image::from_vec_u8(
|
||||
NonZeroU32::new(img.width()).unwrap(),
|
||||
NonZeroU32::new(img.height()).unwrap(),
|
||||
Self::img_into_bytes(img),
|
||||
Self::pixel_type(),
|
||||
)
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
fn img_into_bytes(img: DynamicImage) -> Vec<u8>;
|
||||
}
|
||||
|
||||
impl PixelExt for U8 {
|
||||
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
|
||||
img.to_luma8().into_raw()
|
||||
}
|
||||
}
|
||||
|
||||
impl PixelExt for U8x3 {
|
||||
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
|
||||
img.to_rgb8().into_raw()
|
||||
}
|
||||
}
|
||||
|
||||
impl PixelExt for U8x4 {
|
||||
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
|
||||
img.to_rgba8().into_raw()
|
||||
}
|
||||
}
|
||||
|
||||
impl PixelExt for I32 {
|
||||
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
|
||||
img.to_luma16()
|
||||
.as_raw()
|
||||
.iter()
|
||||
.map(|&p| p as u32 * (i16::MAX as u32 + 1))
|
||||
.flat_map(|val| val.to_le_bytes())
|
||||
.collect()
|
||||
}
|
||||
}
|
||||
|
||||
impl PixelExt for F32 {
|
||||
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
|
||||
img.to_luma16()
|
||||
.as_raw()
|
||||
.iter()
|
||||
.map(|&p| p as f32 * (i16::MAX as f32 + 1.0))
|
||||
.flat_map(|val| val.to_le_bytes())
|
||||
.collect()
|
||||
}
|
||||
}
|
||||
|
||||
fn resize_test<P: PixelExt>(resize_alg: ResizeAlg, cpu_extensions: CpuExtensions) {
|
||||
let image = P::load_src_image();
|
||||
assert_eq!(image.pixel_type(), P::pixel_type());
|
||||
|
||||
let mut resizer = Resizer::new(resize_alg);
|
||||
unsafe {
|
||||
resizer.set_cpu_extensions(CpuExtensions::None);
|
||||
resizer.set_cpu_extensions(cpu_extensions);
|
||||
}
|
||||
let new_height = get_new_height(&image.view(), NEW_WIDTH);
|
||||
let mut result = Image::new(
|
||||
@@ -231,47 +189,73 @@ fn resize_nearest_u8x1() {
|
||||
assert!(resizer
|
||||
.resize(&image.view(), &mut result.view_mut())
|
||||
.is_ok());
|
||||
save_result(&result, "u8x1-nearest-native");
|
||||
|
||||
let alg_name = match resize_alg {
|
||||
ResizeAlg::Nearest => "nearest",
|
||||
ResizeAlg::Convolution(filter) => match filter {
|
||||
FilterType::Box => "box",
|
||||
FilterType::Bilinear => "bilinear",
|
||||
FilterType::Hamming => "hamming",
|
||||
FilterType::Mitchell => "mitchell",
|
||||
FilterType::CatmullRom => "catmullrom",
|
||||
FilterType::Lanczos3 => "lanczos3",
|
||||
_ => "unknown",
|
||||
},
|
||||
ResizeAlg::SuperSampling(_, _) => "supersampling",
|
||||
_ => "unknown",
|
||||
};
|
||||
|
||||
let ext_name = match cpu_extensions {
|
||||
CpuExtensions::None => "native",
|
||||
CpuExtensions::Sse2 => "sse2",
|
||||
CpuExtensions::Sse4_1 => "sse41",
|
||||
CpuExtensions::Avx2 => "avx2",
|
||||
};
|
||||
|
||||
let name = format!("{}-{}-{}", P::pixel_type_str(), alg_name, ext_name);
|
||||
save_result(&result, &name);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn resize_lanczos3_u8x1_native() {
|
||||
let image = get_source_image_u8x1();
|
||||
assert!(matches!(image.pixel_type(), PixelType::U8));
|
||||
|
||||
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
|
||||
unsafe {
|
||||
resizer.set_cpu_extensions(CpuExtensions::None);
|
||||
fn resize_u8() {
|
||||
type P = U8;
|
||||
resize_test::<P>(ResizeAlg::Nearest, CpuExtensions::None);
|
||||
for cpu_extensions in [CpuExtensions::None, CpuExtensions::Avx2] {
|
||||
resize_test::<P>(ResizeAlg::Convolution(FilterType::Lanczos3), cpu_extensions);
|
||||
}
|
||||
let new_height = get_new_height(&image.view(), NEW_WIDTH);
|
||||
let mut result = Image::new(
|
||||
NonZeroU32::new(NEW_WIDTH).unwrap(),
|
||||
NonZeroU32::new(new_height).unwrap(),
|
||||
image.pixel_type(),
|
||||
);
|
||||
assert!(resizer
|
||||
.resize(&image.view(), &mut result.view_mut())
|
||||
.is_ok());
|
||||
save_result(&result, "u8x1-lanczos3-native");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn resize_lanczos3_u8x1_avx2() {
|
||||
let image = get_source_image_u8x1();
|
||||
assert!(matches!(image.pixel_type(), PixelType::U8));
|
||||
fn resize_u8x3() {
|
||||
type P = U8x3;
|
||||
resize_test::<P>(ResizeAlg::Nearest, CpuExtensions::None);
|
||||
for cpu_extensions in [CpuExtensions::None] {
|
||||
resize_test::<P>(ResizeAlg::Convolution(FilterType::Lanczos3), cpu_extensions);
|
||||
}
|
||||
}
|
||||
|
||||
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
|
||||
unsafe {
|
||||
resizer.set_cpu_extensions(CpuExtensions::Avx2);
|
||||
#[test]
|
||||
fn resize_u8x4() {
|
||||
type P = U8x4;
|
||||
resize_test::<P>(ResizeAlg::Nearest, CpuExtensions::None);
|
||||
for cpu_extensions in [
|
||||
CpuExtensions::None,
|
||||
CpuExtensions::Sse4_1,
|
||||
CpuExtensions::Avx2,
|
||||
] {
|
||||
resize_test::<P>(ResizeAlg::Convolution(FilterType::Lanczos3), cpu_extensions);
|
||||
resize_test::<P>(
|
||||
ResizeAlg::SuperSampling(FilterType::Lanczos3, 2),
|
||||
cpu_extensions,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
// #[test]
|
||||
fn _resize_i32() {
|
||||
type P = I32;
|
||||
resize_test::<P>(ResizeAlg::Nearest, CpuExtensions::None);
|
||||
for cpu_extensions in [CpuExtensions::None] {
|
||||
resize_test::<P>(ResizeAlg::Convolution(FilterType::Lanczos3), cpu_extensions);
|
||||
}
|
||||
let new_height = get_new_height(&image.view(), NEW_WIDTH);
|
||||
let mut result = Image::new(
|
||||
NonZeroU32::new(NEW_WIDTH).unwrap(),
|
||||
NonZeroU32::new(new_height).unwrap(),
|
||||
image.pixel_type(),
|
||||
);
|
||||
assert!(resizer
|
||||
.resize(&image.view(), &mut result.view_mut())
|
||||
.is_ok());
|
||||
save_result(&result, "u8x1-lanczos3-avx2");
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user