mirror of
https://github.com/Cykooz/fast_image_resize.git
synced 2026-10-08 01:11:09 +00:00
- Added support of new type of pixels PixelType::U16x3.
- Added variant `U16x3` into the enum `PixelType`.
This commit is contained in:
@@ -1,3 +1,9 @@
|
||||
## [Unreleased] - ReleaseDate
|
||||
|
||||
- Added support of new type of pixels `PixelType::U16x3`.
|
||||
- Breaking changes:
|
||||
- Added variant `U16x3` into the enum `PixelType`.
|
||||
|
||||
## [0.6.0] - 2022-01-12
|
||||
|
||||
- Added optimisation of multiplying and dividing image by alpha channel with helps
|
||||
|
||||
Generated
+89
-2
@@ -308,6 +308,15 @@ dependencies = [
|
||||
"byteorder",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "deflate"
|
||||
version = "0.9.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "5f95bf05dffba6e6cce8dfbb30def788154949ccd9aed761b472119c21e01c70"
|
||||
dependencies = [
|
||||
"adler32",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "directories-next"
|
||||
version = "2.0.0"
|
||||
@@ -335,6 +344,70 @@ version = "1.6.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e78d4f1cc4ae33bbfc157ed5d5a5ef3bc29227303d595861deb238fcec4e9457"
|
||||
|
||||
[[package]]
|
||||
name = "encoding"
|
||||
version = "0.2.33"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "6b0d943856b990d12d3b55b359144ff341533e516d94098b1d3fc1ac666d36ec"
|
||||
dependencies = [
|
||||
"encoding-index-japanese",
|
||||
"encoding-index-korean",
|
||||
"encoding-index-simpchinese",
|
||||
"encoding-index-singlebyte",
|
||||
"encoding-index-tradchinese",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "encoding-index-japanese"
|
||||
version = "1.20141219.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "04e8b2ff42e9a05335dbf8b5c6f7567e5591d0d916ccef4e0b1710d32a0d0c91"
|
||||
dependencies = [
|
||||
"encoding_index_tests",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "encoding-index-korean"
|
||||
version = "1.20141219.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "4dc33fb8e6bcba213fe2f14275f0963fd16f0a02c878e3095ecfdf5bee529d81"
|
||||
dependencies = [
|
||||
"encoding_index_tests",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "encoding-index-simpchinese"
|
||||
version = "1.20141219.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d87a7194909b9118fc707194baa434a4e3b0fb6a5a757c73c3adb07aa25031f7"
|
||||
dependencies = [
|
||||
"encoding_index_tests",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "encoding-index-singlebyte"
|
||||
version = "1.20141219.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3351d5acffb224af9ca265f435b859c7c01537c0849754d3db3fdf2bfe2ae84a"
|
||||
dependencies = [
|
||||
"encoding_index_tests",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "encoding-index-tradchinese"
|
||||
version = "1.20141219.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "fd0e20d5688ce3cab59eb3ef3a2083a5c77bf496cb798dc6fcdb75f323890c18"
|
||||
dependencies = [
|
||||
"encoding_index_tests",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "encoding_index_tests"
|
||||
version = "0.1.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "a246d82be1c9d791c5dfde9a2bd045fc3cbba3fa2b11ad558f27d01712f00569"
|
||||
|
||||
[[package]]
|
||||
name = "fallible-iterator"
|
||||
version = "0.2.0"
|
||||
@@ -363,6 +436,7 @@ dependencies = [
|
||||
"glassbench",
|
||||
"image",
|
||||
"num-traits",
|
||||
"png 0.17.2",
|
||||
"resize",
|
||||
"rgb",
|
||||
"thiserror",
|
||||
@@ -514,7 +588,7 @@ dependencies = [
|
||||
"num-iter",
|
||||
"num-rational",
|
||||
"num-traits",
|
||||
"png",
|
||||
"png 0.16.8",
|
||||
"scoped_threadpool",
|
||||
"tiff",
|
||||
]
|
||||
@@ -821,10 +895,23 @@ checksum = "3c3287920cb847dee3de33d301c463fba14dda99db24214ddf93f83d3021f4c6"
|
||||
dependencies = [
|
||||
"bitflags",
|
||||
"crc32fast",
|
||||
"deflate",
|
||||
"deflate 0.8.6",
|
||||
"miniz_oxide 0.3.7",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "png"
|
||||
version = "0.17.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "c845088517daa61e8a57eee40309347cea13f273694d1385c553e7a57127763b"
|
||||
dependencies = [
|
||||
"bitflags",
|
||||
"crc32fast",
|
||||
"deflate 0.9.1",
|
||||
"encoding",
|
||||
"miniz_oxide 0.4.4",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "proc-macro2"
|
||||
version = "1.0.36"
|
||||
|
||||
@@ -23,6 +23,7 @@ glassbench = "0.3.1"
|
||||
image = "0.23.14"
|
||||
resize = "0.7.2"
|
||||
rgb = "0.8.31"
|
||||
png = "0.17.2"
|
||||
|
||||
|
||||
[[bench]]
|
||||
@@ -40,6 +41,11 @@ name = "bench_compare_rgb"
|
||||
harness = false
|
||||
|
||||
|
||||
[[bench]]
|
||||
name = "bench_compare_rgb16"
|
||||
harness = false
|
||||
|
||||
|
||||
[[bench]]
|
||||
name = "bench_compare_rgbx"
|
||||
harness = false
|
||||
|
||||
@@ -3,8 +3,8 @@
|
||||
Rust library for fast image resizing with using of SIMD instructions.
|
||||
|
||||
_Note: This library does not support converting image color spaces.
|
||||
If it is important for you to resize images with a non-linear color space
|
||||
(e.g. sRGB) correctly, then you need to convert it to a linear color space
|
||||
If it is important for you to resize images with a non-linear color space
|
||||
(e.g. sRGB) correctly, then you need to convert it to a linear color space
|
||||
before resizing. [Read more](https://legacy.imagemagick.org/Usage/resize/#resize_colorspace)
|
||||
about resizing with respect to color space._
|
||||
|
||||
@@ -22,6 +22,8 @@ Supported pixel formats and available optimisations:
|
||||
- native Rust-code without forced SIMD
|
||||
- SSE4.1
|
||||
- AVX2
|
||||
- `U16x3` - three `u16` components per pixel (e.g. RGB):
|
||||
- native Rust-code without forced SIMD
|
||||
- `I32` - one `i32` component per pixel:
|
||||
- native Rust-code without forced SIMD
|
||||
- `F32` - one `f32` component per pixel:
|
||||
@@ -33,7 +35,7 @@ Environment:
|
||||
- CPU: Intel(R) Core(TM) i7-6700K CPU @ 4.00GHz
|
||||
- RAM: DDR4 3000 MHz
|
||||
- Ubuntu 20.04 (linux 5.11)
|
||||
- Rust 1.57.0
|
||||
- Rust 1.57.1
|
||||
- fast_image_resize = "0.6.0"
|
||||
- glassbench = "0.3.1"
|
||||
- `rustflags = ["-C", "llvm-args=-x86-branches-within-32B-boundaries"]`
|
||||
@@ -59,11 +61,11 @@ Pipeline:
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|------------|:-------:|:--------:|:----------:|:--------:|
|
||||
| image | 92.300 | 180.063 | 257.043 | 342.268 |
|
||||
| resize | 15.429 | 71.589 | 131.447 | 191.048 |
|
||||
| fir rust | 0.481 | 55.347 | 89.512 | 121.270 |
|
||||
| fir sse4.1 | - | 44.841 | 54.630 | 75.284 |
|
||||
| fir avx2 | - | 10.814 | 14.408 | 20.897 |
|
||||
| image | 96.466 | 186.243 | 268.888 | 358.235 |
|
||||
| resize | 15.812 | 68.793 | 125.291 | 181.471 |
|
||||
| fir rust | 0.495 | 56.372 | 93.182 | 127.870 |
|
||||
| fir sse4.1 | - | 44.775 | 56.014 | 78.759 |
|
||||
| fir avx2 | - | 11.290 | 14.731 | 20.678 |
|
||||
|
||||
### Resize RGBA image (U8x4) 4928x3279 => 852x567
|
||||
|
||||
@@ -76,11 +78,11 @@ Pipeline:
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|------------|:-------:|:--------:|:----------:|:--------:|
|
||||
| image | 98.598 | 177.427 | 253.149 | 336.999 |
|
||||
| resize | 17.988 | 79.891 | 149.178 | 218.593 |
|
||||
| fir rust | 11.952 | 63.214 | 89.397 | 118.714 |
|
||||
| fir sse4.1 | 9.169 | 20.664 | 26.442 | 34.326 |
|
||||
| fir avx2 | 7.353 | 15.514 | 18.936 | 24.624 |
|
||||
| image | 102.809 | 185.787 | 266.163 | 356.038 |
|
||||
| resize | 18.723 | 83.072 | 156.063 | 229.400 |
|
||||
| fir rust | 12.518 | 65.821 | 92.371 | 122.981 |
|
||||
| fir sse4.1 | 9.529 | 21.008 | 27.444 | 35.534 |
|
||||
| fir avx2 | 7.622 | 15.755 | 19.369 | 25.184 |
|
||||
|
||||
### Resize grayscale image (U8) 4928x3279 => 852x567
|
||||
|
||||
@@ -94,10 +96,25 @@ Pipeline:
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|----------|:-------:|:--------:|:----------:|:--------:|
|
||||
| image | 77.209 | 128.704 | 166.905 | 210.503 |
|
||||
| resize | 9.694 | 24.477 | 47.722 | 80.824 |
|
||||
| fir rust | 0.198 | 21.457 | 24.256 | 36.893 |
|
||||
| fir avx2 | - | 9.758 | 7.787 | 12.099 |
|
||||
| image | 80.937 | 132.497 | 174.721 | 219.534 |
|
||||
| resize | 10.107 | 25.514 | 49.871 | 84.692 |
|
||||
| fir rust | 0.208 | 23.255 | 26.471 | 37.642 |
|
||||
| fir avx2 | - | 9.927 | 8.054 | 12.298 |
|
||||
|
||||
### Resize RGB16 image (U16x3) 4928x3279 => 852x567
|
||||
|
||||
Pipeline:
|
||||
|
||||
`src_image => resize => dst_image`
|
||||
|
||||
- Source image [nasa-4928x3279.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279.png)
|
||||
- Numbers in table is mean duration of image resizing in milliseconds.
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|----------|:-------:|:--------:|:----------:|:--------:|
|
||||
| image | 98.962 | 180.516 | 255.045 | 335.265 |
|
||||
| resize | 16.504 | 66.861 | 120.973 | 174.806 |
|
||||
| fir rust | 0.800 | 53.874 | 89.453 | 123.963 |
|
||||
|
||||
## Examples
|
||||
|
||||
@@ -128,7 +145,7 @@ fn resize_image_example() {
|
||||
img.to_rgba8().into_raw(),
|
||||
fr::PixelType::U8x4,
|
||||
)
|
||||
.unwrap();
|
||||
.unwrap();
|
||||
|
||||
// Create MulDiv instance
|
||||
let alpha_mul_div = fr::MulDiv::default();
|
||||
@@ -141,9 +158,9 @@ fn resize_image_example() {
|
||||
let dst_width = NonZeroU32::new(1024).unwrap();
|
||||
let dst_height = NonZeroU32::new(768).unwrap();
|
||||
let mut dst_image = fr::Image::new(
|
||||
dst_width,
|
||||
dst_height,
|
||||
src_image.pixel_type(),
|
||||
dst_width,
|
||||
dst_height,
|
||||
src_image.pixel_type(),
|
||||
);
|
||||
|
||||
// Get mutable view of destination image data
|
||||
|
||||
@@ -0,0 +1,116 @@
|
||||
use std::num::NonZeroU32;
|
||||
|
||||
use glassbench::*;
|
||||
use image::imageops;
|
||||
use resize::Pixel::{RGB16, RGB8};
|
||||
use rgb::{FromSlice, RGB};
|
||||
|
||||
use fast_image_resize::Image;
|
||||
use fast_image_resize::{CpuExtensions, FilterType, PixelType, ResizeAlg, Resizer};
|
||||
|
||||
mod utils;
|
||||
|
||||
pub fn bench_downscale_rgb16(bench: &mut Bench) {
|
||||
let src_image = utils::get_big_rgb16_image();
|
||||
let new_width = NonZeroU32::new(852).unwrap();
|
||||
let new_height = NonZeroU32::new(567).unwrap();
|
||||
|
||||
let alg_names = ["Nearest", "Bilinear", "CatmullRom", "Lanczos3"];
|
||||
|
||||
// image crate
|
||||
// https://crates.io/crates/image
|
||||
for alg_name in alg_names {
|
||||
let filter = match alg_name {
|
||||
"Nearest" => imageops::Nearest,
|
||||
"Bilinear" => imageops::Triangle,
|
||||
"CatmullRom" => imageops::CatmullRom,
|
||||
"Lanczos3" => imageops::Lanczos3,
|
||||
_ => continue,
|
||||
};
|
||||
bench.task(format!("image - {}", alg_name), |task| {
|
||||
task.iter(|| {
|
||||
imageops::resize(&src_image, new_width.get(), new_height.get(), filter);
|
||||
})
|
||||
});
|
||||
}
|
||||
|
||||
// resize crate
|
||||
// https://crates.io/crates/resize
|
||||
for alg_name in alg_names {
|
||||
let resize_src_image = src_image.as_raw().as_rgb();
|
||||
let mut dst =
|
||||
vec![RGB::new(0u16, 0u16, 0u16); (new_width.get() * new_height.get()) as usize];
|
||||
bench.task(format!("resize - {}", alg_name), |task| {
|
||||
let filter = match alg_name {
|
||||
"Nearest" => resize::Type::Point,
|
||||
"Bilinear" => resize::Type::Triangle,
|
||||
"CatmullRom" => resize::Type::Catrom,
|
||||
"Lanczos3" => resize::Type::Lanczos3,
|
||||
_ => return,
|
||||
};
|
||||
let mut resize = resize::new(
|
||||
src_image.width() as usize,
|
||||
src_image.height() as usize,
|
||||
new_width.get() as usize,
|
||||
new_height.get() as usize,
|
||||
RGB16,
|
||||
filter,
|
||||
)
|
||||
.unwrap();
|
||||
task.iter(|| {
|
||||
resize.resize(resize_src_image, &mut dst).unwrap();
|
||||
})
|
||||
});
|
||||
}
|
||||
|
||||
// fast_image_resize crate;
|
||||
let src_buffer: Vec<u8> = src_image
|
||||
.as_raw()
|
||||
.iter()
|
||||
.flat_map(|&c| c.to_le_bytes())
|
||||
.collect();
|
||||
let mut cpu_ext_and_name = vec![(CpuExtensions::None, "rust")];
|
||||
// #[cfg(target_arch = "x86_64")]
|
||||
// {
|
||||
// cpu_ext_and_name.push((CpuExtensions::Sse4_1, "sse4.1"));
|
||||
// cpu_ext_and_name.push((CpuExtensions::Avx2, "avx2"));
|
||||
// }
|
||||
for (cpu_ext, ext_name) in cpu_ext_and_name {
|
||||
for alg_name in alg_names {
|
||||
let src_image_data = Image::from_vec_u8(
|
||||
NonZeroU32::new(src_image.width()).unwrap(),
|
||||
NonZeroU32::new(src_image.height()).unwrap(),
|
||||
src_buffer.clone(),
|
||||
PixelType::U16x3,
|
||||
)
|
||||
.unwrap();
|
||||
let src_view = src_image_data.view();
|
||||
let mut dst_image = Image::new(new_width, new_height, src_image_data.pixel_type());
|
||||
let mut dst_view = dst_image.view_mut();
|
||||
|
||||
let resize_alg = match alg_name {
|
||||
"Nearest" => ResizeAlg::Nearest,
|
||||
"Bilinear" => ResizeAlg::Convolution(FilterType::Bilinear),
|
||||
"CatmullRom" => ResizeAlg::Convolution(FilterType::CatmullRom),
|
||||
"Lanczos3" => ResizeAlg::Convolution(FilterType::Lanczos3),
|
||||
_ => return,
|
||||
};
|
||||
let mut fast_resizer = Resizer::new(resize_alg);
|
||||
|
||||
unsafe {
|
||||
fast_resizer.reset_internal_buffers();
|
||||
fast_resizer.set_cpu_extensions(cpu_ext);
|
||||
}
|
||||
|
||||
bench.task(format!("fir {} - {}", ext_name, alg_name), |task| {
|
||||
task.iter(|| {
|
||||
fast_resizer.resize(&src_view, &mut dst_view).unwrap();
|
||||
})
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
utils::print_md_table(bench);
|
||||
}
|
||||
|
||||
glassbench!("Compare resize of RGB16 image", bench_downscale_rgb16,);
|
||||
+38
-3
@@ -39,6 +39,19 @@ fn get_big_u8x3_source_image() -> Image<'static> {
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
fn get_big_u16x3_source_image() -> Image<'static> {
|
||||
let img = utils::get_big_rgb16_image();
|
||||
let width = img.width();
|
||||
let height = img.height();
|
||||
Image::from_vec_u8(
|
||||
NonZeroU32::new(width).unwrap(),
|
||||
NonZeroU32::new(height).unwrap(),
|
||||
img.as_raw().iter().flat_map(|&c| c.to_le_bytes()).collect(),
|
||||
PixelType::U16x3,
|
||||
)
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
fn get_big_i32_image() -> Image<'static> {
|
||||
let img = utils::get_big_luma16_image();
|
||||
let img_data: Vec<u32> = img
|
||||
@@ -83,7 +96,7 @@ fn get_small_source_image() -> Image<'static> {
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
fn native_nearest_bench(bench: &mut Bench) {
|
||||
fn native_nearest_u8x4_bench(bench: &mut Bench) {
|
||||
let image = get_big_source_image();
|
||||
let mut res_image = Image::new(
|
||||
NonZeroU32::new(NEW_WIDTH).unwrap(),
|
||||
@@ -245,25 +258,47 @@ fn u8x3_lanczos3_bench(bench: &mut Bench, cpu_extensions: CpuExtensions, name: &
|
||||
});
|
||||
}
|
||||
|
||||
fn u16x3_lanczos3_bench(bench: &mut Bench, cpu_extensions: CpuExtensions, name: &str) {
|
||||
let image = get_big_u16x3_source_image();
|
||||
let mut res_image = Image::new(
|
||||
NonZeroU32::new(NEW_WIDTH).unwrap(),
|
||||
NonZeroU32::new(NEW_HEIGHT).unwrap(),
|
||||
image.pixel_type(),
|
||||
);
|
||||
let src_image = image.view();
|
||||
let mut dst_image = res_image.view_mut();
|
||||
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
|
||||
unsafe {
|
||||
resizer.set_cpu_extensions(cpu_extensions);
|
||||
}
|
||||
bench.task(name, |task| {
|
||||
task.iter(|| {
|
||||
resizer.resize(&src_image, &mut dst_image).unwrap();
|
||||
})
|
||||
});
|
||||
}
|
||||
|
||||
pub fn main() {
|
||||
use glassbench::*;
|
||||
let name = env!("CARGO_CRATE_NAME");
|
||||
let cmd = Command::read();
|
||||
if cmd.include_bench(name) {
|
||||
let mut bench = create_bench(name, "Resize", &cmd);
|
||||
native_nearest_bench(&mut bench);
|
||||
native_nearest_u8x4_bench(&mut bench);
|
||||
native_nearest_u8_bench(&mut bench);
|
||||
|
||||
u8_lanczos3_bench(&mut bench, CpuExtensions::None, "u8 lanczos3 wo SIMD");
|
||||
u8x3_lanczos3_bench(&mut bench, CpuExtensions::None, "u8x3 lanczos3 wo SIMD");
|
||||
u8x4_lanczos3_bench(&mut bench, CpuExtensions::None, "u8x4 lanczos3 wo SIMD");
|
||||
u16x3_lanczos3_bench(&mut bench, CpuExtensions::None, "u16x3 lanczos3 wo SIMD");
|
||||
native_lanczos3_i32_bench(&mut bench);
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
{
|
||||
u8_lanczos3_bench(&mut bench, CpuExtensions::Avx2, "u8 lanczos3 avx2");
|
||||
|
||||
u8x3_lanczos3_bench(&mut bench, CpuExtensions::Sse4_1, "u8x3 lanczos3 sse4.1");
|
||||
// u8x3_lanczos3_bench(&mut bench, CpuExtensions::Avx2, "u8x3 lanczos3 avx2");
|
||||
u8x3_lanczos3_bench(&mut bench, CpuExtensions::Avx2, "u8x3 lanczos3 avx2");
|
||||
u16x3_lanczos3_bench(&mut bench, CpuExtensions::Avx2, "u16x3 lanczos3 avx2");
|
||||
|
||||
u8x4_lanczos3_bench(&mut bench, CpuExtensions::Sse4_1, "u8x4 lanczos3 sse4.1");
|
||||
u8x4_lanczos3_bench(&mut bench, CpuExtensions::Avx2, "u8x4 lanczos3 avx2");
|
||||
|
||||
+12
-1
@@ -3,7 +3,9 @@ use std::env;
|
||||
|
||||
use glassbench::*;
|
||||
use image::io::Reader;
|
||||
use image::{GrayImage, ImageBuffer, Luma, RgbImage, RgbaImage};
|
||||
use image::{GrayImage, ImageBuffer, Luma, Rgb, RgbImage, RgbaImage};
|
||||
|
||||
pub type Rgb16Image = ImageBuffer<Rgb<u16>, Vec<u16>>;
|
||||
|
||||
pub fn get_big_rgb_image() -> RgbImage {
|
||||
let cur_dir = env::current_dir().unwrap();
|
||||
@@ -14,6 +16,15 @@ pub fn get_big_rgb_image() -> RgbImage {
|
||||
img.to_rgb8()
|
||||
}
|
||||
|
||||
pub fn get_big_rgb16_image() -> Rgb16Image {
|
||||
let cur_dir = env::current_dir().unwrap();
|
||||
let img = Reader::open(cur_dir.join("data/nasa-4928x3279.png"))
|
||||
.unwrap()
|
||||
.decode()
|
||||
.unwrap();
|
||||
img.to_rgb16()
|
||||
}
|
||||
|
||||
pub fn get_big_rgba_image() -> RgbaImage {
|
||||
let cur_dir = env::current_dir().unwrap();
|
||||
let img = Reader::open(cur_dir.join("data/nasa-4928x3279-rgba.png"))
|
||||
|
||||
+3
-3
@@ -134,10 +134,10 @@ fn assert_images<'s, 'd, 'da>(
|
||||
MulDivImagesError,
|
||||
> {
|
||||
let src_image_u8x4 = src_image
|
||||
.u32_image()
|
||||
.u8x4_image()
|
||||
.ok_or(MulDivImagesError::UnsupportedPixelType)?;
|
||||
let dst_image_u8x4 = dst_image
|
||||
.u32_image()
|
||||
.u8x4_image()
|
||||
.ok_or(MulDivImagesError::UnsupportedPixelType)?;
|
||||
if src_image_u8x4.width() != dst_image_u8x4.width()
|
||||
|| src_image_u8x4.height() != dst_image_u8x4.height()
|
||||
@@ -152,6 +152,6 @@ fn assert_image<'a, 'b>(
|
||||
image: &'a mut ImageViewMut<'b>,
|
||||
) -> Result<TypedImageViewMut<'a, 'b, U8x4>, MulDivImageError> {
|
||||
image
|
||||
.u32_image()
|
||||
.u8x4_image()
|
||||
.ok_or(MulDivImageError::UnsupportedPixelType)
|
||||
}
|
||||
|
||||
@@ -12,6 +12,7 @@ mod f32x1;
|
||||
mod filters;
|
||||
mod i32x1;
|
||||
mod optimisations;
|
||||
mod u16x3;
|
||||
mod u8x1;
|
||||
mod u8x3;
|
||||
mod u8x4;
|
||||
|
||||
@@ -5,62 +5,22 @@ use super::Bound;
|
||||
// This code is based on C-implementation from Pillow-SIMD package for Python
|
||||
// https://github.com/uploadcare/pillow-simd
|
||||
|
||||
const fn get_clip_table() -> [u8; 1280] {
|
||||
let mut table = [0u8; 1280];
|
||||
let mut i: usize = 640;
|
||||
while i < 640 + 255 {
|
||||
table[i] = (i - 640) as u8;
|
||||
i += 1;
|
||||
}
|
||||
while i < 1280 {
|
||||
table[i] = 255;
|
||||
i += 1;
|
||||
}
|
||||
table
|
||||
}
|
||||
|
||||
// Handles values form -640 to 639.
|
||||
const CLIP8_LOOKUPS: [u8; 1280] = [
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
|
||||
0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25,
|
||||
26, 27, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 47, 48, 49,
|
||||
50, 51, 52, 53, 54, 55, 56, 57, 58, 59, 60, 61, 62, 63, 64, 65, 66, 67, 68, 69, 70, 71, 72, 73,
|
||||
74, 75, 76, 77, 78, 79, 80, 81, 82, 83, 84, 85, 86, 87, 88, 89, 90, 91, 92, 93, 94, 95, 96, 97,
|
||||
98, 99, 100, 101, 102, 103, 104, 105, 106, 107, 108, 109, 110, 111, 112, 113, 114, 115, 116,
|
||||
117, 118, 119, 120, 121, 122, 123, 124, 125, 126, 127, 128, 129, 130, 131, 132, 133, 134, 135,
|
||||
136, 137, 138, 139, 140, 141, 142, 143, 144, 145, 146, 147, 148, 149, 150, 151, 152, 153, 154,
|
||||
155, 156, 157, 158, 159, 160, 161, 162, 163, 164, 165, 166, 167, 168, 169, 170, 171, 172, 173,
|
||||
174, 175, 176, 177, 178, 179, 180, 181, 182, 183, 184, 185, 186, 187, 188, 189, 190, 191, 192,
|
||||
193, 194, 195, 196, 197, 198, 199, 200, 201, 202, 203, 204, 205, 206, 207, 208, 209, 210, 211,
|
||||
212, 213, 214, 215, 216, 217, 218, 219, 220, 221, 222, 223, 224, 225, 226, 227, 228, 229, 230,
|
||||
231, 232, 233, 234, 235, 236, 237, 238, 239, 240, 241, 242, 243, 244, 245, 246, 247, 248, 249,
|
||||
250, 251, 252, 253, 254, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
|
||||
];
|
||||
const CLIP8_LOOKUPS: [u8; 1280] = get_clip_table();
|
||||
|
||||
// 8 bits for result. Filter can have negative areas.
|
||||
// In one cases the sum of the coefficients will be negative,
|
||||
@@ -70,20 +30,9 @@ const PRECISION_BITS: u8 = 32 - 8 - 2;
|
||||
// We use i16 type to store coefficients.
|
||||
const MAX_COEFS_PRECISION: u8 = 16 - 1;
|
||||
|
||||
/// # Safety
|
||||
/// The function must be used with the `v` and `precision` values
|
||||
/// such that the expression `v >> precision`
|
||||
/// produces a result in the range `[-512, 511]`.
|
||||
#[inline(always)]
|
||||
pub(crate) unsafe fn clip8(v: i32, precision: u8) -> u8 {
|
||||
let index = (640 + (v >> precision)) as usize;
|
||||
// index must be in range [(640-512)..(640+511)]
|
||||
*CLIP8_LOOKUPS.get_unchecked(index)
|
||||
}
|
||||
|
||||
/// Converts `Vec<f64>` into `&[i16]` without additional memory allocations.
|
||||
/// The memory buffer from `Vec<f64>` uses as `[i16]` .
|
||||
pub struct NormalizerGuard {
|
||||
pub struct NormalizerGuard16 {
|
||||
values: Vec<f64>,
|
||||
precision: u8,
|
||||
}
|
||||
@@ -94,7 +43,7 @@ pub struct CoefficientsI16Chunk<'a> {
|
||||
pub values: &'a [i16],
|
||||
}
|
||||
|
||||
impl NormalizerGuard {
|
||||
impl NormalizerGuard16 {
|
||||
#[inline]
|
||||
pub fn new(mut values: Vec<f64>) -> Self {
|
||||
let max_weight = values
|
||||
@@ -127,14 +76,7 @@ impl NormalizerGuard {
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn normalized_i16(&self) -> &[i16] {
|
||||
let len = self.values.len();
|
||||
let ptr = self.values.as_ptr();
|
||||
unsafe { slice::from_raw_parts(ptr as *const i16, len) }
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn normalized_i16_chunks(
|
||||
pub fn normalized_chunks(
|
||||
&self,
|
||||
window_size: usize,
|
||||
bounds: &[Bound],
|
||||
@@ -159,6 +101,103 @@ impl NormalizerGuard {
|
||||
pub fn precision(&self) -> u8 {
|
||||
self.precision
|
||||
}
|
||||
|
||||
/// # Safety
|
||||
/// The function must be used with the `v`
|
||||
/// such that the expression `v >> self.precision`
|
||||
/// produces a result in the range `[-512, 511]`.
|
||||
#[inline(always)]
|
||||
pub unsafe fn clip(&self, v: i32) -> u8 {
|
||||
let index = (640 + (v >> self.precision)) as usize;
|
||||
// index must be in range [(640-512)..(640+511)]
|
||||
*CLIP8_LOOKUPS.get_unchecked(index)
|
||||
}
|
||||
}
|
||||
|
||||
// 16 bits for result. Filter can have negative areas.
|
||||
// In one cases the sum of the coefficients will be negative,
|
||||
// in the other it will be more than 1.0. That is why we need
|
||||
// two extra bits for overflow and i64 type.
|
||||
const PRECISION16_BITS: u8 = 64 - 16 - 2;
|
||||
// We use i32 type to store coefficients.
|
||||
const MAX_COEFS_PRECISION16: u8 = 32 - 1;
|
||||
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
pub struct CoefficientsI32Chunk<'a> {
|
||||
pub start: u32,
|
||||
pub values: &'a [i32],
|
||||
}
|
||||
|
||||
/// Converts `Vec<f64>` into `&[i32]` without additional memory allocations.
|
||||
/// The memory buffer from `Vec<f64>` uses as `[i32]` .
|
||||
pub struct NormalizerGuard32 {
|
||||
values: Vec<f64>,
|
||||
precision: u8,
|
||||
}
|
||||
|
||||
impl NormalizerGuard32 {
|
||||
#[inline]
|
||||
pub fn new(mut values: Vec<f64>) -> Self {
|
||||
let max_weight = values
|
||||
.iter()
|
||||
.max_by(|&x, &y| x.partial_cmp(y).unwrap())
|
||||
.unwrap_or(&0.0)
|
||||
.to_owned();
|
||||
|
||||
let mut precision = 0u8;
|
||||
for cur_precision in 0..PRECISION16_BITS {
|
||||
precision = cur_precision;
|
||||
let next_value: i64 = (max_weight * (1i64 << (precision + 1)) as f64).round() as i64;
|
||||
// The next value will be outside the range, so just stop
|
||||
if next_value >= (1i64 << MAX_COEFS_PRECISION16) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
debug_assert!(precision >= 4); // required for some SIMD optimisations
|
||||
|
||||
let len = values.len();
|
||||
let ptr = values.as_mut_ptr();
|
||||
// Size of `[i32]` always will be not greater than `[f64]` with same number of items
|
||||
let values_i32 = unsafe { slice::from_raw_parts_mut(ptr as *mut i32, len) };
|
||||
|
||||
let scale = (1i64 << precision) as f64;
|
||||
for (&src, dst) in values.iter().zip(values_i32.iter_mut()) {
|
||||
*dst = (src * scale).round() as i32;
|
||||
}
|
||||
Self { values, precision }
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn normalized_chunks(
|
||||
&self,
|
||||
window_size: usize,
|
||||
bounds: &[Bound],
|
||||
) -> Vec<CoefficientsI32Chunk> {
|
||||
let len = self.values.len();
|
||||
let ptr = self.values.as_ptr();
|
||||
let mut cooefs = unsafe { slice::from_raw_parts(ptr as *const i32, len) };
|
||||
let mut res = Vec::with_capacity(bounds.len());
|
||||
for bound in bounds {
|
||||
let (left, right) = cooefs.split_at(window_size);
|
||||
cooefs = right;
|
||||
let size = bound.size as usize;
|
||||
res.push(CoefficientsI32Chunk {
|
||||
start: bound.start,
|
||||
values: &left[0..size],
|
||||
});
|
||||
}
|
||||
res
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn precision(&self) -> u8 {
|
||||
self.precision
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub fn clip(&self, v: i64) -> u16 {
|
||||
(v >> self.precision).min(u16::MAX as i64).max(0) as u16
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
@@ -167,7 +206,10 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn test_minimal_precision() {
|
||||
assert!(NormalizerGuard::new(vec![0.0]).precision() >= 4);
|
||||
assert!(NormalizerGuard::new(vec![2.0]).precision() >= 4);
|
||||
// required for some SIMD optimisations
|
||||
assert!(NormalizerGuard16::new(vec![0.0]).precision() >= 4);
|
||||
assert!(NormalizerGuard16::new(vec![2.0]).precision() >= 4);
|
||||
assert!(NormalizerGuard32::new(vec![0.0]).precision() >= 4);
|
||||
assert!(NormalizerGuard32::new(vec![2.0]).precision() >= 4);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,27 @@
|
||||
use super::{Coefficients, Convolution};
|
||||
use crate::image_view::{TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::U16x3;
|
||||
use crate::CpuExtensions;
|
||||
|
||||
mod native;
|
||||
|
||||
impl Convolution for U16x3 {
|
||||
fn horiz_convolution(
|
||||
src_image: TypedImageView<Self>,
|
||||
dst_image: TypedImageViewMut<Self>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
native::horiz_convolution(src_image, dst_image, offset, coeffs);
|
||||
}
|
||||
|
||||
fn vert_convolution(
|
||||
src_image: TypedImageView<Self>,
|
||||
dst_image: TypedImageViewMut<Self>,
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
native::vert_convolution(src_image, dst_image, coeffs);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,70 @@
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::image_view::{TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::U16x3;
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) fn horiz_convolution(
|
||||
src_image: TypedImageView<U16x3>,
|
||||
mut dst_image: TypedImageViewMut<U16x3>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard32::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
|
||||
let initial: i64 = 1 << (precision - 1);
|
||||
|
||||
let src_rows = src_image.iter_rows(offset);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for (dst_row, src_row) in dst_rows.zip(src_rows) {
|
||||
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
|
||||
let first_x_src = coeffs_chunk.start as usize;
|
||||
let mut ss = [initial; 3];
|
||||
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
|
||||
for (&k, src_pixel) in coeffs_chunk.values.iter().zip(src_pixels) {
|
||||
for (i, s) in ss.iter_mut().enumerate() {
|
||||
*s += src_pixel.0[i] as i64 * (k as i64);
|
||||
}
|
||||
}
|
||||
for (i, s) in ss.iter().copied().enumerate() {
|
||||
dst_pixel.0[i] = normalizer_guard.clip(s);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) fn vert_convolution(
|
||||
src_image: TypedImageView<U16x3>,
|
||||
mut dst_image: TypedImageViewMut<U16x3>,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard32::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
|
||||
let initial = 1 << (precision - 1);
|
||||
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for (&coeffs_chunk, dst_row) in coefficients_chunks.iter().zip(dst_rows) {
|
||||
let first_y_src = coeffs_chunk.start;
|
||||
let ks = coeffs_chunk.values;
|
||||
|
||||
for (x_src, dst_pixel) in dst_row.iter_mut().enumerate() {
|
||||
let mut ss = [initial; 3];
|
||||
let src_rows = src_image.iter_rows(first_y_src);
|
||||
for (&k, src_row) in ks.iter().zip(src_rows) {
|
||||
let src_pixel = unsafe { src_row.get_unchecked(x_src as usize) };
|
||||
for (i, s) in ss.iter_mut().enumerate() {
|
||||
*s += src_pixel.0[i] as i64 * (k as i64);
|
||||
}
|
||||
}
|
||||
for (i, s) in ss.iter().copied().enumerate() {
|
||||
dst_pixel.0[i] = normalizer_guard.clip(s);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,7 +1,7 @@
|
||||
use std::arch::x86_64::*;
|
||||
|
||||
use crate::convolution::optimisations::CoefficientsI16Chunk;
|
||||
use crate::convolution::{optimisations, Bound, Coefficients};
|
||||
use crate::convolution::optimisations::{CoefficientsI16Chunk, NormalizerGuard16};
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::image_view::{FourRows, FourRowsMut, TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::U8;
|
||||
use crate::simd_utils;
|
||||
@@ -16,17 +16,15 @@ pub(crate) fn horiz_convolution(
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks =
|
||||
normalizer_guard.normalized_i16_chunks(window_size, &bounds_per_pixel);
|
||||
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_8u4x(src_rows, dst_rows, &coefficients_chunks, precision);
|
||||
horiz_convolution_8u4x(src_rows, dst_rows, &coefficients_chunks, &normalizer_guard);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -37,7 +35,7 @@ pub(crate) fn horiz_convolution(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
&coefficients_chunks,
|
||||
precision,
|
||||
&normalizer_guard,
|
||||
);
|
||||
}
|
||||
yy += 1;
|
||||
@@ -50,17 +48,16 @@ pub(crate) fn vert_convolution(
|
||||
mut dst_image: TypedImageViewMut<U8>,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coeffs_i16 = normalizer_guard.normalized_i16();
|
||||
let coeffs_chunks = coeffs_i16.chunks(window_size);
|
||||
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for ((&bound, k), dst_row) in bounds.iter().zip(coeffs_chunks).zip(dst_rows) {
|
||||
for (dst_row, coeffs_chunk) in dst_rows.zip(coefficients_chunks) {
|
||||
unsafe {
|
||||
vert_convolution_8u(&src_image, dst_row, k, bound, precision);
|
||||
vert_convolution_8u(&src_image, dst_row, coeffs_chunk, &normalizer_guard);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -77,13 +74,13 @@ unsafe fn horiz_convolution_8u4x(
|
||||
src_rows: FourRows<U8>,
|
||||
dst_rows: FourRowsMut<U8>,
|
||||
coefficients_chunks: &[CoefficientsI16Chunk],
|
||||
precision: u8,
|
||||
normalizer_guard: &NormalizerGuard16,
|
||||
) {
|
||||
let s_rows = [src_rows.0, src_rows.1, src_rows.2, src_rows.3];
|
||||
let d_rows = [dst_rows.0, dst_rows.1, dst_rows.2, dst_rows.3];
|
||||
let zero = _mm_setzero_si128();
|
||||
// 8 components will be added, use only 1/8 of the error
|
||||
let initial = _mm256_set1_epi32(1 << (precision - 4));
|
||||
let initial = _mm256_set1_epi32(1 << (normalizer_guard.precision() - 4));
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let coeffs = coeffs_chunk.values;
|
||||
@@ -130,7 +127,7 @@ unsafe fn horiz_convolution_8u4x(
|
||||
x += 1;
|
||||
}
|
||||
|
||||
let result_u8x4 = result_i32x4.map(|v| optimisations::clip8(v, precision));
|
||||
let result_u8x4 = result_i32x4.map(|v| normalizer_guard.clip(v));
|
||||
for i in 0..4 {
|
||||
d_rows[i].get_unchecked_mut(dst_x).0 = result_u8x4[i];
|
||||
}
|
||||
@@ -148,11 +145,11 @@ unsafe fn horiz_convolution_8u(
|
||||
src_row: &[U8],
|
||||
dst_row: &mut [U8],
|
||||
coefficients_chunks: &[CoefficientsI16Chunk],
|
||||
precision: u8,
|
||||
normalizer_guard: &NormalizerGuard16,
|
||||
) {
|
||||
let zero = _mm_setzero_si128();
|
||||
// 8 components will be added, use only 1/8 of the error
|
||||
let initial = _mm256_set1_epi32(1 << (precision - 4));
|
||||
let initial = _mm256_set1_epi32(1 << (normalizer_guard.precision() - 4));
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let coeffs = coeffs_chunk.values;
|
||||
@@ -194,7 +191,7 @@ unsafe fn horiz_convolution_8u(
|
||||
x += 1;
|
||||
}
|
||||
|
||||
dst_row.get_unchecked_mut(dst_x).0 = optimisations::clip8(result_i32, precision);
|
||||
dst_row.get_unchecked_mut(dst_x).0 = normalizer_guard.clip(result_i32);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -203,13 +200,14 @@ unsafe fn horiz_convolution_8u(
|
||||
unsafe fn vert_convolution_8u(
|
||||
src_img: &TypedImageView<U8>,
|
||||
dst_row: &mut [U8],
|
||||
coeffs: &[i16],
|
||||
bound: Bound,
|
||||
precision: u8,
|
||||
coeffs_chunk: CoefficientsI16Chunk,
|
||||
normalizer_guard: &NormalizerGuard16,
|
||||
) {
|
||||
let src_width = src_img.width().get() as usize;
|
||||
let y_start = bound.start;
|
||||
let y_size = bound.size;
|
||||
let y_start = coeffs_chunk.start;
|
||||
let coeffs = coeffs_chunk.values;
|
||||
let max_y = y_start + coeffs.len() as u32;
|
||||
let precision = normalizer_guard.precision();
|
||||
|
||||
let initial = _mm_set1_epi32(1 << (precision - 1));
|
||||
let initial_256 = _mm256_set1_epi32(1 << (precision - 1));
|
||||
@@ -225,7 +223,7 @@ unsafe fn vert_convolution_8u(
|
||||
|
||||
let mut y: u32 = 0;
|
||||
|
||||
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
|
||||
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, max_y) {
|
||||
// Load two coefficients at once
|
||||
let two_coeffs = simd_utils::ptr_i16_to_256set1_epi32(coeffs, y as usize);
|
||||
let row1 = simd_utils::loadu_si256(s_row1, x); // top line
|
||||
@@ -246,8 +244,9 @@ unsafe fn vert_convolution_8u(
|
||||
y += 2;
|
||||
}
|
||||
|
||||
if let Some(s_row) = src_img.get_row(y_start + y) {
|
||||
let one_coeff = _mm256_set1_epi32(coeffs[y as usize] as i32);
|
||||
if let Some(&k) = coeffs.get(y as usize) {
|
||||
let s_row = src_img.get_row(y_start + y).unwrap();
|
||||
let one_coeff = _mm256_set1_epi32(k as i32);
|
||||
|
||||
let row1 = simd_utils::loadu_si256(s_row, x); // top line
|
||||
let row2 = _mm256_setzero_si256(); // bottom line is empty
|
||||
@@ -289,7 +288,7 @@ unsafe fn vert_convolution_8u(
|
||||
let mut sss1 = initial; // right row
|
||||
let mut y: u32 = 0;
|
||||
|
||||
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
|
||||
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, max_y) {
|
||||
// Load two coefficients at once
|
||||
let two_coeffs = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
|
||||
|
||||
@@ -305,8 +304,9 @@ unsafe fn vert_convolution_8u(
|
||||
y += 2;
|
||||
}
|
||||
|
||||
if let Some(s_row) = src_img.get_row(y_start + y) {
|
||||
let one_coeff = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
|
||||
if let Some(&k) = coeffs.get(y as usize) {
|
||||
let s_row = src_img.get_row(y_start + y).unwrap();
|
||||
let one_coeff = _mm_set1_epi32(k as i32);
|
||||
|
||||
let row1 = simd_utils::loadl_epi64(s_row, x); // top line
|
||||
let row2 = _mm_setzero_si128(); // bottom line is empty
|
||||
@@ -337,7 +337,7 @@ unsafe fn vert_convolution_8u(
|
||||
while x < src_width.saturating_sub(3) {
|
||||
let mut sss = initial;
|
||||
let mut y: u32 = 0;
|
||||
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
|
||||
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, max_y) {
|
||||
// Load two coefficients at once
|
||||
let two_coeffs = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
|
||||
|
||||
@@ -351,9 +351,10 @@ unsafe fn vert_convolution_8u(
|
||||
y += 2;
|
||||
}
|
||||
|
||||
if let Some(s_row) = src_img.get_row(y_start + y) {
|
||||
if let Some(&k) = coeffs.get(y as usize) {
|
||||
let s_row = src_img.get_row(y_start + y).unwrap();
|
||||
let pix = simd_utils::mm_cvtepu8_epi32_from_u8(s_row, x);
|
||||
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
|
||||
let mmk = _mm_set1_epi32(k as i32);
|
||||
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
|
||||
}
|
||||
|
||||
@@ -375,11 +376,11 @@ unsafe fn vert_convolution_8u(
|
||||
|
||||
for dst_pixel in dst_row.iter_mut().skip(x) {
|
||||
let mut ss0 = 1 << (precision - 1);
|
||||
for (dy, &k) in coeffs.iter().take(y_size as usize).enumerate() {
|
||||
for (dy, &k) in coeffs.iter().enumerate() {
|
||||
let src_pixel = src_img.get_pixel(x as u32, y_start + dy as u32);
|
||||
ss0 += src_pixel.0 as i32 * (k as i32);
|
||||
}
|
||||
dst_pixel.0 = optimisations::clip8(ss0, precision);
|
||||
dst_pixel.0 = normalizer_guard.clip(ss0);
|
||||
x += 1;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -11,9 +11,9 @@ pub(crate) fn horiz_convolution(
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks = normalizer_guard.normalized_i16_chunks(window_size, &bounds);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
|
||||
let initial = 1 << (precision - 1);
|
||||
|
||||
let src_rows = src_image.iter_rows(offset);
|
||||
@@ -28,7 +28,7 @@ pub(crate) fn horiz_convolution(
|
||||
for (&k, &src_pixel) in ks.iter().zip(src_pixels) {
|
||||
ss += src_pixel.0 as i32 * (k as i32);
|
||||
}
|
||||
dst_pixel.0 = unsafe { optimisations::clip8(ss, precision) };
|
||||
dst_pixel.0 = unsafe { normalizer_guard.clip(ss) };
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -41,9 +41,9 @@ pub(crate) fn vert_convolution(
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks = normalizer_guard.normalized_i16_chunks(window_size, &bounds);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
|
||||
let initial = 1 << (precision - 1);
|
||||
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
@@ -58,7 +58,7 @@ pub(crate) fn vert_convolution(
|
||||
let src_pixel = unsafe { src_row.get_unchecked(x_src as usize) };
|
||||
ss += src_pixel.0 as i32 * (k as i32);
|
||||
}
|
||||
dst_pixel.0 = unsafe { optimisations::clip8(ss, precision) };
|
||||
dst_pixel.0 = unsafe { normalizer_guard.clip(ss) };
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
use std::arch::x86_64::*;
|
||||
use std::intrinsics::transmute;
|
||||
|
||||
use crate::convolution::optimisations::CoefficientsI16Chunk;
|
||||
use crate::convolution::{optimisations, Bound, Coefficients};
|
||||
use crate::convolution::optimisations::{CoefficientsI16Chunk, NormalizerGuard16};
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::image_view::{FourRows, FourRowsMut, TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::{Pixel, U8x3};
|
||||
use crate::simd_utils;
|
||||
@@ -17,10 +17,9 @@ pub(crate) fn horiz_convolution(
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks =
|
||||
normalizer_guard.normalized_i16_chunks(window_size, &bounds_per_pixel);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
@@ -51,17 +50,16 @@ pub(crate) fn vert_convolution(
|
||||
mut dst_image: TypedImageViewMut<U8x3>,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coeffs_i16 = normalizer_guard.normalized_i16();
|
||||
let coeffs_chunks = coeffs_i16.chunks(window_size);
|
||||
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for ((&bound, k), dst_row) in bounds.iter().zip(coeffs_chunks).zip(dst_rows) {
|
||||
for (dst_row, coeffs_chunk) in dst_rows.zip(coefficients_chunks) {
|
||||
unsafe {
|
||||
vert_convolution_8u(&src_image, dst_row, k, bound, precision);
|
||||
vert_convolution_8u(&src_image, dst_row, coeffs_chunk, &normalizer_guard);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -410,13 +408,14 @@ unsafe fn horiz_convolution_8u(
|
||||
unsafe fn vert_convolution_8u(
|
||||
src_img: &TypedImageView<U8x3>,
|
||||
dst_row: &mut [U8x3],
|
||||
coeffs: &[i16],
|
||||
bound: Bound,
|
||||
precision: u8,
|
||||
coeffs_chunk: CoefficientsI16Chunk,
|
||||
normalizer_guard: &NormalizerGuard16,
|
||||
) {
|
||||
let src_width = src_img.width().get() as usize;
|
||||
let y_start = bound.start;
|
||||
let y_size = bound.size;
|
||||
let y_start = coeffs_chunk.start;
|
||||
let coeffs = coeffs_chunk.values;
|
||||
let max_y = y_start + coeffs.len() as u32;
|
||||
let precision = normalizer_guard.precision();
|
||||
|
||||
let initial = _mm_set1_epi32(1 << (precision - 1));
|
||||
let initial_256 = _mm256_set1_epi32(1 << (precision - 1));
|
||||
@@ -433,7 +432,7 @@ unsafe fn vert_convolution_8u(
|
||||
|
||||
let mut y: u32 = 0;
|
||||
|
||||
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
|
||||
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, max_y) {
|
||||
// Load two coefficients at once
|
||||
let mmk = simd_utils::ptr_i16_to_256set1_epi32(coeffs, y as usize);
|
||||
|
||||
@@ -455,8 +454,9 @@ unsafe fn vert_convolution_8u(
|
||||
y += 2;
|
||||
}
|
||||
|
||||
if let Some(s_row) = src_img.get_row(y_start + y) {
|
||||
let mmk = _mm256_set1_epi32(coeffs[y as usize] as i32);
|
||||
if let Some(&k) = coeffs.get(y as usize) {
|
||||
let s_row = src_img.get_row(y_start + y).unwrap();
|
||||
let mmk = _mm256_set1_epi32(k as i32);
|
||||
|
||||
let source1 = simd_utils::loadu_si256_raw(s_row, x_in_bytes); // top line
|
||||
let source2 = _mm256_setzero_si256(); // bottom line is empty
|
||||
@@ -499,7 +499,7 @@ unsafe fn vert_convolution_8u(
|
||||
let mut sss1 = initial; // right row
|
||||
let mut y: u32 = 0;
|
||||
|
||||
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
|
||||
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, max_y) {
|
||||
// Load two coefficients at once
|
||||
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
|
||||
|
||||
@@ -515,8 +515,9 @@ unsafe fn vert_convolution_8u(
|
||||
y += 2;
|
||||
}
|
||||
|
||||
if let Some(s_row) = src_img.get_row(y_start + y) {
|
||||
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
|
||||
if let Some(&k) = coeffs.get(y as usize) {
|
||||
let s_row = src_img.get_row(y_start + y).unwrap();
|
||||
let mmk = _mm_set1_epi32(k as i32);
|
||||
|
||||
let source1 = simd_utils::loadl_epi64_raw(s_row, x_in_bytes); // top line
|
||||
let source2 = _mm_setzero_si128(); // bottom line is empty
|
||||
@@ -548,7 +549,7 @@ unsafe fn vert_convolution_8u(
|
||||
while x_in_bytes < width_in_bytes.saturating_sub(3) {
|
||||
let mut sss = initial;
|
||||
let mut y: u32 = 0;
|
||||
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
|
||||
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, max_y) {
|
||||
// Load two coefficients at once
|
||||
let two_coeffs = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
|
||||
|
||||
@@ -562,9 +563,10 @@ unsafe fn vert_convolution_8u(
|
||||
y += 2;
|
||||
}
|
||||
|
||||
if let Some(s_row) = src_img.get_row(y_start + y) {
|
||||
if let Some(&k) = coeffs.get(y as usize) {
|
||||
let s_row = src_img.get_row(y_start + y).unwrap();
|
||||
let pix = simd_utils::mm_cvtepu8_epi32_from_raw(s_row, x_in_bytes);
|
||||
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
|
||||
let mmk = _mm_set1_epi32(k as i32);
|
||||
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
|
||||
}
|
||||
|
||||
@@ -588,14 +590,14 @@ unsafe fn vert_convolution_8u(
|
||||
|
||||
for dst_pixel in dst_u8 {
|
||||
let mut ss0 = 1 << (precision - 1);
|
||||
for (dy, &k) in coeffs.iter().take(y_size as usize).enumerate() {
|
||||
for (dy, &k) in coeffs.iter().enumerate() {
|
||||
if let Some(src_row) = src_img.get_row(y_start + dy as u32) {
|
||||
let src_ptr = src_row.as_ptr() as *const u8;
|
||||
let src_component = *src_ptr.add(x_in_bytes);
|
||||
ss0 += src_component as i32 * (k as i32);
|
||||
}
|
||||
}
|
||||
*dst_pixel = optimisations::clip8(ss0, precision);
|
||||
*dst_pixel = normalizer_guard.clip(ss0);
|
||||
x_in_bytes += 1;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -11,9 +11,9 @@ pub(crate) fn horiz_convolution(
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks = normalizer_guard.normalized_i16_chunks(window_size, &bounds);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
|
||||
let initial = 1 << (precision - 1);
|
||||
|
||||
let src_rows = src_image.iter_rows(offset);
|
||||
@@ -29,7 +29,7 @@ pub(crate) fn horiz_convolution(
|
||||
}
|
||||
}
|
||||
for (i, s) in ss.iter().copied().enumerate() {
|
||||
dst_pixel.0[i] = unsafe { optimisations::clip8(s, precision) };
|
||||
dst_pixel.0[i] = unsafe { normalizer_guard.clip(s) };
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -43,9 +43,9 @@ pub(crate) fn vert_convolution(
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks = normalizer_guard.normalized_i16_chunks(window_size, &bounds);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
|
||||
let initial = 1 << (precision - 1);
|
||||
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
@@ -63,7 +63,7 @@ pub(crate) fn vert_convolution(
|
||||
}
|
||||
}
|
||||
for (i, s) in ss.iter().copied().enumerate() {
|
||||
dst_pixel.0[i] = unsafe { optimisations::clip8(s, precision) };
|
||||
dst_pixel.0[i] = unsafe { normalizer_guard.clip(s) };
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
use std::arch::x86_64::*;
|
||||
use std::intrinsics::transmute;
|
||||
|
||||
use crate::convolution::optimisations::CoefficientsI16Chunk;
|
||||
use crate::convolution::{optimisations, Bound, Coefficients};
|
||||
use crate::convolution::optimisations::{CoefficientsI16Chunk, NormalizerGuard16};
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::image_view::{FourRows, FourRowsMut, TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::U8x4;
|
||||
use crate::simd_utils;
|
||||
@@ -20,10 +20,9 @@ pub(crate) fn horiz_convolution(
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks =
|
||||
normalizer_guard.normalized_i16_chunks(window_size, &bounds_per_pixel);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
@@ -54,17 +53,16 @@ pub(crate) fn vert_convolution(
|
||||
mut dst_image: TypedImageViewMut<U8x4>,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coeffs_i16 = normalizer_guard.normalized_i16();
|
||||
let coeffs_chunks = coeffs_i16.chunks(window_size);
|
||||
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for ((&bound, k), dst_row) in bounds.iter().zip(coeffs_chunks).zip(dst_rows) {
|
||||
for (dst_row, coeffs_chunk) in dst_rows.zip(coefficients_chunks) {
|
||||
unsafe {
|
||||
vert_convolution_8u(&src_image, dst_row, k, bound, precision);
|
||||
vert_convolution_8u(&src_image, dst_row, coeffs_chunk, &normalizer_guard);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -332,13 +330,14 @@ unsafe fn horiz_convolution_8u(
|
||||
unsafe fn vert_convolution_8u(
|
||||
src_img: &TypedImageView<U8x4>,
|
||||
dst_row: &mut [U8x4],
|
||||
coeffs: &[i16],
|
||||
bound: Bound,
|
||||
precision: u8,
|
||||
coeffs_chunk: CoefficientsI16Chunk,
|
||||
normalizer_guard: &NormalizerGuard16,
|
||||
) {
|
||||
let src_width = src_img.width().get() as usize;
|
||||
let y_start = bound.start;
|
||||
let y_size = bound.size;
|
||||
let y_start = coeffs_chunk.start;
|
||||
let coeffs = coeffs_chunk.values;
|
||||
let max_y = y_start + coeffs.len() as u32;
|
||||
let precision = normalizer_guard.precision();
|
||||
|
||||
let initial = _mm_set1_epi32(1 << (precision - 1));
|
||||
let initial_256 = _mm256_set1_epi32(1 << (precision - 1));
|
||||
@@ -352,7 +351,7 @@ unsafe fn vert_convolution_8u(
|
||||
|
||||
let mut y: u32 = 0;
|
||||
|
||||
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
|
||||
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, max_y) {
|
||||
// Load two coefficients at once
|
||||
let mmk = simd_utils::ptr_i16_to_256set1_epi32(coeffs, y as usize);
|
||||
|
||||
@@ -374,8 +373,9 @@ unsafe fn vert_convolution_8u(
|
||||
y += 2;
|
||||
}
|
||||
|
||||
if let Some(s_row) = src_img.get_row(y_start + y) {
|
||||
let mmk = _mm256_set1_epi32(coeffs[y as usize] as i32);
|
||||
if let Some(&k) = coeffs.get(y as usize) {
|
||||
let s_row = src_img.get_row(y_start + y).unwrap();
|
||||
let mmk = _mm256_set1_epi32(k as i32);
|
||||
|
||||
let source1 = simd_utils::loadu_si256(s_row, x); // top line
|
||||
let source2 = _mm256_setzero_si256(); // bottom line is empty
|
||||
@@ -417,7 +417,7 @@ unsafe fn vert_convolution_8u(
|
||||
let mut sss1 = initial; // right row
|
||||
let mut y: u32 = 0;
|
||||
|
||||
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
|
||||
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, max_y) {
|
||||
// Load two coefficients at once
|
||||
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
|
||||
|
||||
@@ -433,8 +433,9 @@ unsafe fn vert_convolution_8u(
|
||||
y += 2;
|
||||
}
|
||||
|
||||
if let Some(s_row) = src_img.get_row(y_start + y) {
|
||||
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
|
||||
if let Some(&k) = coeffs.get(y as usize) {
|
||||
let s_row = src_img.get_row(y_start + y).unwrap();
|
||||
let mmk = _mm_set1_epi32(k as i32);
|
||||
|
||||
let source1 = simd_utils::loadl_epi64(s_row, x); // top line
|
||||
let source2 = _mm_setzero_si128(); // bottom line is empty
|
||||
@@ -465,7 +466,7 @@ unsafe fn vert_convolution_8u(
|
||||
if x < src_width {
|
||||
let mut sss = initial;
|
||||
let mut y: u32 = 0;
|
||||
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
|
||||
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, max_y) {
|
||||
// Load two coefficients at once
|
||||
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
|
||||
|
||||
@@ -479,9 +480,10 @@ unsafe fn vert_convolution_8u(
|
||||
y += 2;
|
||||
}
|
||||
|
||||
if let Some(s_row) = src_img.get_row(y_start + y) {
|
||||
if let Some(&k) = coeffs.get(y as usize) {
|
||||
let s_row = src_img.get_row(y_start + y).unwrap();
|
||||
let pix = simd_utils::mm_cvtepu8_epi32(s_row, x);
|
||||
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
|
||||
let mmk = _mm_set1_epi32(k as i32);
|
||||
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
|
||||
}
|
||||
|
||||
|
||||
@@ -10,9 +10,9 @@ pub(crate) fn horiz_convolution(
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks = normalizer_guard.normalized_i16_chunks(window_size, &bounds);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
|
||||
let initial = 1 << (precision - 1);
|
||||
|
||||
let src_rows = src_image.iter_rows(offset);
|
||||
@@ -29,8 +29,7 @@ pub(crate) fn horiz_convolution(
|
||||
*s += components[i] as i32 * (k as i32);
|
||||
}
|
||||
}
|
||||
dst_pixel.0 =
|
||||
u32::from_le_bytes(ss.map(|v| unsafe { optimisations::clip8(v, precision) }));
|
||||
dst_pixel.0 = u32::from_le_bytes(ss.map(|v| unsafe { normalizer_guard.clip(v) }));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -42,9 +41,9 @@ pub(crate) fn vert_convolution(
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks = normalizer_guard.normalized_i16_chunks(window_size, &bounds);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
|
||||
let initial = 1 << (precision - 1);
|
||||
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
@@ -62,8 +61,7 @@ pub(crate) fn vert_convolution(
|
||||
*s += components[i] as i32 * (k as i32);
|
||||
}
|
||||
}
|
||||
dst_pixel.0 =
|
||||
u32::from_le_bytes(ss.map(|v| unsafe { optimisations::clip8(v, precision) }));
|
||||
dst_pixel.0 = u32::from_le_bytes(ss.map(|v| unsafe { normalizer_guard.clip(v) }));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
use std::arch::x86_64::*;
|
||||
use std::intrinsics::transmute;
|
||||
|
||||
use crate::convolution::optimisations::CoefficientsI16Chunk;
|
||||
use crate::convolution::{optimisations, Bound, Coefficients};
|
||||
use crate::convolution::optimisations::{CoefficientsI16Chunk, NormalizerGuard16};
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::image_view::{FourRows, FourRowsMut, TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::U8x4;
|
||||
use crate::simd_utils;
|
||||
@@ -20,10 +20,9 @@ pub(crate) fn horiz_convolution(
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks =
|
||||
normalizer_guard.normalized_i16_chunks(window_size, &bounds_per_pixel);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
@@ -54,17 +53,16 @@ pub(crate) fn vert_convolution(
|
||||
mut dst_image: TypedImageViewMut<U8x4>,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coeffs_i16 = normalizer_guard.normalized_i16();
|
||||
let coeffs_chunks = coeffs_i16.chunks(window_size);
|
||||
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for ((&bound, k), dst_row) in bounds.iter().zip(coeffs_chunks).zip(dst_rows) {
|
||||
for (dst_row, coeffs_chunk) in dst_rows.zip(coefficients_chunks) {
|
||||
unsafe {
|
||||
vert_convolution_8u(&src_image, dst_row, k, bound, precision);
|
||||
vert_convolution_8u(&src_image, dst_row, coeffs_chunk, &normalizer_guard);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -313,14 +311,15 @@ unsafe fn horiz_convolution_8u(
|
||||
pub(crate) unsafe fn vert_convolution_8u(
|
||||
src_img: &TypedImageView<U8x4>,
|
||||
dst_row: &mut [U8x4],
|
||||
coeffs: &[i16],
|
||||
bound: Bound,
|
||||
precision: u8,
|
||||
coeffs_chunk: CoefficientsI16Chunk,
|
||||
normalizer_guard: &NormalizerGuard16,
|
||||
) {
|
||||
let mut xx: usize = 0;
|
||||
let src_width = src_img.width().get() as usize;
|
||||
let y_start = bound.start;
|
||||
let y_size = bound.size;
|
||||
let y_start = coeffs_chunk.start;
|
||||
let coeffs = coeffs_chunk.values;
|
||||
let max_y = y_start + coeffs.len() as u32;
|
||||
let precision = normalizer_guard.precision();
|
||||
|
||||
let initial = _mm_set1_epi32(1 << (precision - 1));
|
||||
|
||||
@@ -336,7 +335,7 @@ pub(crate) unsafe fn vert_convolution_8u(
|
||||
|
||||
let mut y: u32 = 0;
|
||||
|
||||
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
|
||||
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, max_y) {
|
||||
// Load two coefficients at once
|
||||
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
|
||||
|
||||
@@ -373,8 +372,9 @@ pub(crate) unsafe fn vert_convolution_8u(
|
||||
y += 2;
|
||||
}
|
||||
|
||||
if let Some(s_row) = src_img.get_row(y_start + y) {
|
||||
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
|
||||
if let Some(&k) = coeffs.get(y as usize) {
|
||||
let s_row = src_img.get_row(y_start + y).unwrap();
|
||||
let mmk = _mm_set1_epi32(k as i32);
|
||||
|
||||
let mut source1 = simd_utils::loadu_si128(s_row, xx); // top line
|
||||
|
||||
@@ -438,7 +438,7 @@ pub(crate) unsafe fn vert_convolution_8u(
|
||||
let mut sss1 = initial; // right row
|
||||
let mut y: u32 = 0;
|
||||
|
||||
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
|
||||
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, max_y) {
|
||||
// Load two coefficients at once
|
||||
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
|
||||
|
||||
@@ -454,8 +454,9 @@ pub(crate) unsafe fn vert_convolution_8u(
|
||||
y += 2;
|
||||
}
|
||||
|
||||
if let Some(s_row) = src_img.get_row(y_start + y) {
|
||||
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
|
||||
if let Some(&k) = coeffs.get(y as usize) {
|
||||
let s_row = src_img.get_row(y_start + y).unwrap();
|
||||
let mmk = _mm_set1_epi32(k as i32);
|
||||
|
||||
let source1 = simd_utils::loadl_epi64(s_row, xx); // top line
|
||||
|
||||
@@ -479,7 +480,6 @@ pub(crate) unsafe fn vert_convolution_8u(
|
||||
let dst_ptr = dst_row.get_unchecked_mut(xx..).as_mut_ptr() as *mut __m128i;
|
||||
_mm_storel_epi64(dst_ptr, sss0);
|
||||
|
||||
//
|
||||
xx += 2;
|
||||
}
|
||||
|
||||
@@ -487,7 +487,7 @@ pub(crate) unsafe fn vert_convolution_8u(
|
||||
let mut sss = initial;
|
||||
let mut y: u32 = 0;
|
||||
|
||||
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
|
||||
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, max_y) {
|
||||
// Load two coefficients at once
|
||||
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
|
||||
|
||||
@@ -501,9 +501,10 @@ pub(crate) unsafe fn vert_convolution_8u(
|
||||
y += 2;
|
||||
}
|
||||
|
||||
if let Some(s_row) = src_img.get_row(y_start + y) {
|
||||
if let Some(&k) = coeffs.get(y as usize) {
|
||||
let s_row = src_img.get_row(y_start + y).unwrap();
|
||||
let pix = simd_utils::mm_cvtepu8_epi32(s_row, xx);
|
||||
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
|
||||
let mmk = _mm_set1_epi32(k as i32);
|
||||
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
|
||||
}
|
||||
|
||||
|
||||
+10
-1
@@ -1,7 +1,7 @@
|
||||
use std::num::NonZeroU32;
|
||||
|
||||
use crate::image_view::{ImageRows, ImageRowsMut, TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::{Pixel, PixelType, U8x3, U8x4, F32, I32, U8};
|
||||
use crate::pixels::{Pixel, PixelType, U16x3, U8x3, U8x4, F32, I32, U8};
|
||||
use crate::{ImageBufferError, ImageView, ImageViewMut, InvalidBufferSizeError};
|
||||
|
||||
#[derive(Debug)]
|
||||
@@ -27,6 +27,7 @@ impl<'a> Image<'a> {
|
||||
let pixels_count = (width.get() * height.get()) as usize;
|
||||
let pixels = match pixel_type {
|
||||
PixelType::U8x3 => PixelsContainer::VecU8(vec![0; pixels_count * U8x3::size()]),
|
||||
PixelType::U16x3 => PixelsContainer::VecU8(vec![0; pixels_count * U16x3::size()]),
|
||||
PixelType::U8x4 | PixelType::I32 | PixelType::F32 => {
|
||||
PixelsContainer::VecU32(vec![0; pixels_count])
|
||||
}
|
||||
@@ -166,6 +167,10 @@ impl<'a> Image<'a> {
|
||||
let pixels = unsafe { buffer.align_to::<U8x4>().1 };
|
||||
ImageRows::U8x4(pixels.chunks_exact(self.width.get() as usize).collect())
|
||||
}
|
||||
PixelType::U16x3 => {
|
||||
let pixels = unsafe { buffer.align_to::<U16x3>().1 };
|
||||
ImageRows::U16x3(pixels.chunks_exact(self.width.get() as usize).collect())
|
||||
}
|
||||
PixelType::I32 => {
|
||||
let pixels = unsafe { buffer.align_to::<I32>().1 };
|
||||
ImageRows::I32(pixels.chunks_exact(self.width.get() as usize).collect())
|
||||
@@ -197,6 +202,10 @@ impl<'a> Image<'a> {
|
||||
let pixels = unsafe { buffer.align_to_mut::<U8x4>().1 };
|
||||
ImageRowsMut::U8x4(pixels.chunks_exact_mut(width.get() as usize).collect())
|
||||
}
|
||||
PixelType::U16x3 => {
|
||||
let pixels = unsafe { buffer.align_to_mut::<U16x3>().1 };
|
||||
ImageRowsMut::U16x3(pixels.chunks_exact_mut(width.get() as usize).collect())
|
||||
}
|
||||
PixelType::I32 => {
|
||||
let pixels = unsafe { buffer.align_to_mut::<I32>().1 };
|
||||
ImageRowsMut::I32(pixels.chunks_exact_mut(width.get() as usize).collect())
|
||||
|
||||
+42
-3
@@ -2,7 +2,7 @@ use std::num::NonZeroU32;
|
||||
use std::slice;
|
||||
|
||||
use crate::errors::{CropBoxError, ImageBufferError, ImageRowsError};
|
||||
use crate::pixels::{Pixel, PixelType, U8x3, U8x4, F32, I32, U8};
|
||||
use crate::pixels::{Pixel, PixelType, U16x3, U8x3, U8x4, F32, I32, U8};
|
||||
|
||||
pub(crate) type RowMut<'a, 'b, T> = &'a mut &'b mut [T];
|
||||
pub(crate) type TwoRows<'a, T> = (&'a [T], &'a [T]);
|
||||
@@ -28,6 +28,7 @@ pub struct CropBox {
|
||||
pub enum ImageRows<'a> {
|
||||
U8x3(Vec<&'a [U8x3]>),
|
||||
U8x4(Vec<&'a [U8x4]>),
|
||||
U16x3(Vec<&'a [U16x3]>),
|
||||
I32(Vec<&'a [I32]>),
|
||||
F32(Vec<&'a [F32]>),
|
||||
U8(Vec<&'a [U8]>),
|
||||
@@ -42,6 +43,7 @@ impl<'a> ImageRows<'a> {
|
||||
match self {
|
||||
ImageRows::U8x3(rows) => check_rows_count_and_size(width, height, rows),
|
||||
ImageRows::U8x4(rows) => check_rows_count_and_size(width, height, rows),
|
||||
ImageRows::U16x3(rows) => check_rows_count_and_size(width, height, rows),
|
||||
ImageRows::I32(rows) => check_rows_count_and_size(width, height, rows),
|
||||
ImageRows::F32(rows) => check_rows_count_and_size(width, height, rows),
|
||||
ImageRows::U8(rows) => check_rows_count_and_size(width, height, rows),
|
||||
@@ -52,6 +54,7 @@ impl<'a> ImageRows<'a> {
|
||||
match self {
|
||||
Self::U8x3(_) => PixelType::U8x3,
|
||||
Self::U8x4(_) => PixelType::U8x4,
|
||||
Self::U16x3(_) => PixelType::U16x3,
|
||||
Self::I32(_) => PixelType::I32,
|
||||
Self::F32(_) => PixelType::F32,
|
||||
Self::U8(_) => PixelType::U8,
|
||||
@@ -64,6 +67,7 @@ impl<'a> ImageRows<'a> {
|
||||
pub enum ImageRowsMut<'a> {
|
||||
U8x3(Vec<&'a mut [U8x3]>),
|
||||
U8x4(Vec<&'a mut [U8x4]>),
|
||||
U16x3(Vec<&'a mut [U16x3]>),
|
||||
I32(Vec<&'a mut [I32]>),
|
||||
F32(Vec<&'a mut [F32]>),
|
||||
U8(Vec<&'a mut [U8]>),
|
||||
@@ -78,6 +82,7 @@ impl<'a> ImageRowsMut<'a> {
|
||||
match self {
|
||||
Self::U8x3(rows) => check_rows_count_and_size(width, height, rows),
|
||||
Self::U8x4(rows) => check_rows_count_and_size(width, height, rows),
|
||||
Self::U16x3(rows) => check_rows_count_and_size(width, height, rows),
|
||||
Self::I32(rows) => check_rows_count_and_size(width, height, rows),
|
||||
Self::F32(rows) => check_rows_count_and_size(width, height, rows),
|
||||
Self::U8(rows) => check_rows_count_and_size(width, height, rows),
|
||||
@@ -88,6 +93,7 @@ impl<'a> ImageRowsMut<'a> {
|
||||
match self {
|
||||
Self::U8x3(_) => PixelType::U8x3,
|
||||
Self::U8x4(_) => PixelType::U8x4,
|
||||
Self::U16x3(_) => PixelType::U16x3,
|
||||
Self::I32(_) => PixelType::I32,
|
||||
Self::F32(_) => PixelType::F32,
|
||||
Self::U8(_) => PixelType::U8,
|
||||
@@ -143,6 +149,10 @@ impl<'a> ImageView<'a> {
|
||||
let pixels = align_buffer_to(buffer)?;
|
||||
ImageRows::U8x4(pixels.chunks_exact(width.get() as usize).collect())
|
||||
}
|
||||
PixelType::U16x3 => {
|
||||
let pixels = align_buffer_to(buffer)?;
|
||||
ImageRows::U16x3(pixels.chunks_exact(width.get() as usize).collect())
|
||||
}
|
||||
PixelType::I32 => {
|
||||
let pixels = align_buffer_to(buffer)?;
|
||||
ImageRows::I32(pixels.chunks_exact(width.get() as usize).collect())
|
||||
@@ -277,7 +287,7 @@ impl<'a> ImageView<'a> {
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn u32_image(&self) -> Option<TypedImageView<U8x4>> {
|
||||
pub(crate) fn u8x4_image(&self) -> Option<TypedImageView<U8x4>> {
|
||||
if let ImageRows::U8x4(ref rows) = self.rows {
|
||||
Some(TypedImageView {
|
||||
width: self.width,
|
||||
@@ -290,6 +300,19 @@ impl<'a> ImageView<'a> {
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn u16x3_image(&self) -> Option<TypedImageView<U16x3>> {
|
||||
if let ImageRows::U16x3(ref rows) = self.rows {
|
||||
Some(TypedImageView {
|
||||
width: self.width,
|
||||
height: self.height,
|
||||
crop_box: self.crop_box,
|
||||
rows,
|
||||
})
|
||||
} else {
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn i32_image(&self) -> Option<TypedImageView<I32>> {
|
||||
if let ImageRows::I32(ref rows) = self.rows {
|
||||
Some(TypedImageView {
|
||||
@@ -480,6 +503,10 @@ impl<'a> ImageViewMut<'a> {
|
||||
let pixels = align_buffer_to_mut(buffer)?;
|
||||
ImageRowsMut::U8x4(pixels.chunks_exact_mut(width.get() as usize).collect())
|
||||
}
|
||||
PixelType::U16x3 => {
|
||||
let pixels = align_buffer_to_mut(buffer)?;
|
||||
ImageRowsMut::U16x3(pixels.chunks_exact_mut(width.get() as usize).collect())
|
||||
}
|
||||
PixelType::I32 => {
|
||||
let pixels = align_buffer_to_mut(buffer)?;
|
||||
ImageRowsMut::I32(pixels.chunks_exact_mut(width.get() as usize).collect())
|
||||
@@ -527,7 +554,7 @@ impl<'a> ImageViewMut<'a> {
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn u32_image<'s>(&'s mut self) -> Option<TypedImageViewMut<'s, 'a, U8x4>> {
|
||||
pub(crate) fn u8x4_image<'s>(&'s mut self) -> Option<TypedImageViewMut<'s, 'a, U8x4>> {
|
||||
if let ImageRowsMut::U8x4(rows) = &mut self.rows {
|
||||
Some(TypedImageViewMut {
|
||||
width: self.width,
|
||||
@@ -539,6 +566,18 @@ impl<'a> ImageViewMut<'a> {
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn u16x3_image<'s>(&'s mut self) -> Option<TypedImageViewMut<'s, 'a, U16x3>> {
|
||||
if let ImageRowsMut::U16x3(rows) = &mut self.rows {
|
||||
Some(TypedImageViewMut {
|
||||
width: self.width,
|
||||
height: self.height,
|
||||
rows,
|
||||
})
|
||||
} else {
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn i32_image<'s>(&'s mut self) -> Option<TypedImageViewMut<'s, 'a, I32>> {
|
||||
if let ImageRowsMut::I32(rows) = &mut self.rows {
|
||||
Some(TypedImageViewMut {
|
||||
|
||||
@@ -5,6 +5,7 @@ use std::mem::size_of;
|
||||
pub enum PixelType {
|
||||
U8x3,
|
||||
U8x4,
|
||||
U16x3,
|
||||
I32,
|
||||
F32,
|
||||
U8,
|
||||
@@ -14,6 +15,7 @@ impl PixelType {
|
||||
pub(crate) fn size(&self) -> usize {
|
||||
match self {
|
||||
Self::U8x3 => 3,
|
||||
Self::U16x3 => 6,
|
||||
Self::U8 => 1,
|
||||
_ => 4,
|
||||
}
|
||||
@@ -24,6 +26,7 @@ impl PixelType {
|
||||
match self {
|
||||
Self::U8x3 => unsafe { buffer.align_to::<U8x3>().0.is_empty() },
|
||||
Self::U8x4 => unsafe { buffer.align_to::<U8x4>().0.is_empty() },
|
||||
Self::U16x3 => unsafe { buffer.align_to::<U16x3>().0.is_empty() },
|
||||
Self::I32 => unsafe { buffer.align_to::<I32>().0.is_empty() },
|
||||
Self::F32 => unsafe { buffer.align_to::<F32>().0.is_empty() },
|
||||
Self::U8 => true,
|
||||
@@ -79,5 +82,11 @@ pixel_struct!(
|
||||
PixelType::U8x4,
|
||||
"Four bytes per pixel (RGBA, RGBx, CMYK and other)"
|
||||
);
|
||||
pixel_struct!(
|
||||
U16x3,
|
||||
[u16; 3],
|
||||
PixelType::U16x3,
|
||||
"Three `u16` components per pixel (e.g. RGB)"
|
||||
);
|
||||
pixel_struct!(I32, i32, PixelType::I32, "One `i32` component per pixel");
|
||||
pixel_struct!(F32, f32, PixelType::F32, "One `f32` component per pixel");
|
||||
|
||||
+9
-2
@@ -91,8 +91,15 @@ impl Resizer {
|
||||
}
|
||||
}
|
||||
PixelType::U8x4 => {
|
||||
if let Some(src_rows) = src_image.u32_image() {
|
||||
if let Some(dst_rows) = dst_image.u32_image() {
|
||||
if let Some(src_rows) = src_image.u8x4_image() {
|
||||
if let Some(dst_rows) = dst_image.u8x4_image() {
|
||||
self.resize_inner(src_rows, dst_rows);
|
||||
}
|
||||
}
|
||||
}
|
||||
PixelType::U16x3 => {
|
||||
if let Some(src_rows) = src_image.u16x3_image() {
|
||||
if let Some(dst_rows) = dst_image.u16x3_image() {
|
||||
self.resize_inner(src_rows, dst_rows);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -204,6 +204,56 @@ fn upscale_u8x3() {
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn downscale_u16x3() {
|
||||
type P = U16x3;
|
||||
let buffer = downscale_test::<P>(ResizeAlg::Nearest, CpuExtensions::None);
|
||||
assert_eq!(
|
||||
utils::image_u16_checksum::<3>(&buffer),
|
||||
[755050580, 756962660, 740848503]
|
||||
);
|
||||
|
||||
let mut cpu_extensions_vec = vec![CpuExtensions::None];
|
||||
// #[cfg(target_arch = "x86_64")]
|
||||
// {
|
||||
// cpu_extensions_vec.push(CpuExtensions::Sse4_1);
|
||||
// cpu_extensions_vec.push(CpuExtensions::Avx2);
|
||||
// }
|
||||
for cpu_extensions in cpu_extensions_vec {
|
||||
let buffer =
|
||||
downscale_test::<P>(ResizeAlg::Convolution(FilterType::Lanczos3), cpu_extensions);
|
||||
assert_eq!(
|
||||
utils::image_u16_checksum::<3>(&buffer),
|
||||
[756269847, 757632467, 741478612]
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn upscale_u16x3() {
|
||||
type P = U16x3;
|
||||
let buffer = upscale_test::<P>(ResizeAlg::Nearest, CpuExtensions::None);
|
||||
assert_eq!(
|
||||
utils::image_u16_checksum::<3>(&buffer),
|
||||
[297094122820, 297713401842, 291717497780]
|
||||
);
|
||||
|
||||
let mut cpu_extensions_vec = vec![CpuExtensions::None];
|
||||
// #[cfg(target_arch = "x86_64")]
|
||||
// {
|
||||
// cpu_extensions_vec.push(CpuExtensions::Sse4_1);
|
||||
// cpu_extensions_vec.push(CpuExtensions::Avx2);
|
||||
// }
|
||||
for cpu_extensions in cpu_extensions_vec {
|
||||
let buffer =
|
||||
upscale_test::<P>(ResizeAlg::Convolution(FilterType::Lanczos3), cpu_extensions);
|
||||
assert_eq!(
|
||||
utils::image_u16_checksum::<3>(&buffer),
|
||||
[297122154090, 297723994984, 291725294637]
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn downscale_u8x4() {
|
||||
type P = U8x4;
|
||||
|
||||
+30
-11
@@ -1,7 +1,5 @@
|
||||
use std::fs::File;
|
||||
use std::num::NonZeroU32;
|
||||
|
||||
use image::codecs::png::PngEncoder;
|
||||
use image::io::Reader as ImageReader;
|
||||
use image::{ColorType, DynamicImage, GenericImageView};
|
||||
|
||||
@@ -16,12 +14,22 @@ pub fn image_checksum<const N: usize>(buffer: &[u8]) -> [u32; N] {
|
||||
res
|
||||
}
|
||||
|
||||
pub fn image_u16_checksum<const N: usize>(buffer: &[u8]) -> [u64; N] {
|
||||
let buffer_u16 = unsafe { buffer.align_to::<u16>().1 };
|
||||
let mut res = [0u64; N];
|
||||
for pixel in buffer_u16.chunks_exact(N) {
|
||||
res.iter_mut().zip(pixel).for_each(|(d, &s)| *d += s as u64);
|
||||
}
|
||||
res
|
||||
}
|
||||
|
||||
pub trait PixelExt: Pixel {
|
||||
fn pixel_type_str() -> &'static str {
|
||||
match Self::pixel_type() {
|
||||
PixelType::U8 => "u8",
|
||||
PixelType::U8x3 => "u8x3",
|
||||
PixelType::U8x4 => "u8x4",
|
||||
PixelType::U16x3 => "u16x3",
|
||||
PixelType::I32 => "i32",
|
||||
PixelType::F32 => "f32",
|
||||
}
|
||||
@@ -90,6 +98,16 @@ impl PixelExt for U8x4 {
|
||||
}
|
||||
}
|
||||
|
||||
impl PixelExt for U16x3 {
|
||||
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
|
||||
img.to_rgb8()
|
||||
.as_raw()
|
||||
.iter()
|
||||
.flat_map(|&c| [c, c])
|
||||
.collect()
|
||||
}
|
||||
}
|
||||
|
||||
impl PixelExt for I32 {
|
||||
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
|
||||
img.to_luma16()
|
||||
@@ -117,21 +135,22 @@ pub fn save_result(image: &Image, name: &str) {
|
||||
return;
|
||||
}
|
||||
std::fs::create_dir_all("./data/result").unwrap();
|
||||
let mut file = File::create(format!("./data/result/{}.png", name)).unwrap();
|
||||
let path = format!("./data/result/{}.png", name);
|
||||
let color_type = match image.pixel_type() {
|
||||
PixelType::U8x3 => ColorType::Rgb8,
|
||||
PixelType::U8x4 => ColorType::Rgba8,
|
||||
PixelType::U16x3 => ColorType::Rgb16,
|
||||
PixelType::U8 => ColorType::L8,
|
||||
_ => panic!("Unsupported type of pixels"),
|
||||
};
|
||||
PngEncoder::new(&mut file)
|
||||
.encode(
|
||||
image.buffer(),
|
||||
image.width().get(),
|
||||
image.height().get(),
|
||||
color_type,
|
||||
)
|
||||
.unwrap();
|
||||
image::save_buffer(
|
||||
&path,
|
||||
image.buffer(),
|
||||
image.width().get(),
|
||||
image.height().get(),
|
||||
color_type,
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
pub fn cpu_ext_into_str(cpu_extensions: CpuExtensions) -> &'static str {
|
||||
|
||||
Reference in New Issue
Block a user