- Added support of new type of pixels PixelType::U16x3.

- Added variant `U16x3` into the enum `PixelType`.
This commit is contained in:
Kirill Kuzminykh
2022-01-27 20:25:44 +03:00
parent 1fbe296460
commit 146777ddbc
25 changed files with 818 additions and 263 deletions
+6
View File
@@ -1,3 +1,9 @@
## [Unreleased] - ReleaseDate
- Added support of new type of pixels `PixelType::U16x3`.
- Breaking changes:
- Added variant `U16x3` into the enum `PixelType`.
## [0.6.0] - 2022-01-12
- Added optimisation of multiplying and dividing image by alpha channel with helps
Generated
+89 -2
View File
@@ -308,6 +308,15 @@ dependencies = [
"byteorder",
]
[[package]]
name = "deflate"
version = "0.9.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "5f95bf05dffba6e6cce8dfbb30def788154949ccd9aed761b472119c21e01c70"
dependencies = [
"adler32",
]
[[package]]
name = "directories-next"
version = "2.0.0"
@@ -335,6 +344,70 @@ version = "1.6.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e78d4f1cc4ae33bbfc157ed5d5a5ef3bc29227303d595861deb238fcec4e9457"
[[package]]
name = "encoding"
version = "0.2.33"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "6b0d943856b990d12d3b55b359144ff341533e516d94098b1d3fc1ac666d36ec"
dependencies = [
"encoding-index-japanese",
"encoding-index-korean",
"encoding-index-simpchinese",
"encoding-index-singlebyte",
"encoding-index-tradchinese",
]
[[package]]
name = "encoding-index-japanese"
version = "1.20141219.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "04e8b2ff42e9a05335dbf8b5c6f7567e5591d0d916ccef4e0b1710d32a0d0c91"
dependencies = [
"encoding_index_tests",
]
[[package]]
name = "encoding-index-korean"
version = "1.20141219.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "4dc33fb8e6bcba213fe2f14275f0963fd16f0a02c878e3095ecfdf5bee529d81"
dependencies = [
"encoding_index_tests",
]
[[package]]
name = "encoding-index-simpchinese"
version = "1.20141219.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d87a7194909b9118fc707194baa434a4e3b0fb6a5a757c73c3adb07aa25031f7"
dependencies = [
"encoding_index_tests",
]
[[package]]
name = "encoding-index-singlebyte"
version = "1.20141219.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "3351d5acffb224af9ca265f435b859c7c01537c0849754d3db3fdf2bfe2ae84a"
dependencies = [
"encoding_index_tests",
]
[[package]]
name = "encoding-index-tradchinese"
version = "1.20141219.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "fd0e20d5688ce3cab59eb3ef3a2083a5c77bf496cb798dc6fcdb75f323890c18"
dependencies = [
"encoding_index_tests",
]
[[package]]
name = "encoding_index_tests"
version = "0.1.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "a246d82be1c9d791c5dfde9a2bd045fc3cbba3fa2b11ad558f27d01712f00569"
[[package]]
name = "fallible-iterator"
version = "0.2.0"
@@ -363,6 +436,7 @@ dependencies = [
"glassbench",
"image",
"num-traits",
"png 0.17.2",
"resize",
"rgb",
"thiserror",
@@ -514,7 +588,7 @@ dependencies = [
"num-iter",
"num-rational",
"num-traits",
"png",
"png 0.16.8",
"scoped_threadpool",
"tiff",
]
@@ -821,10 +895,23 @@ checksum = "3c3287920cb847dee3de33d301c463fba14dda99db24214ddf93f83d3021f4c6"
dependencies = [
"bitflags",
"crc32fast",
"deflate",
"deflate 0.8.6",
"miniz_oxide 0.3.7",
]
[[package]]
name = "png"
version = "0.17.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c845088517daa61e8a57eee40309347cea13f273694d1385c553e7a57127763b"
dependencies = [
"bitflags",
"crc32fast",
"deflate 0.9.1",
"encoding",
"miniz_oxide 0.4.4",
]
[[package]]
name = "proc-macro2"
version = "1.0.36"
+6
View File
@@ -23,6 +23,7 @@ glassbench = "0.3.1"
image = "0.23.14"
resize = "0.7.2"
rgb = "0.8.31"
png = "0.17.2"
[[bench]]
@@ -40,6 +41,11 @@ name = "bench_compare_rgb"
harness = false
[[bench]]
name = "bench_compare_rgb16"
harness = false
[[bench]]
name = "bench_compare_rgbx"
harness = false
+38 -21
View File
@@ -3,8 +3,8 @@
Rust library for fast image resizing with using of SIMD instructions.
_Note: This library does not support converting image color spaces.
If it is important for you to resize images with a non-linear color space
(e.g. sRGB) correctly, then you need to convert it to a linear color space
If it is important for you to resize images with a non-linear color space
(e.g. sRGB) correctly, then you need to convert it to a linear color space
before resizing. [Read more](https://legacy.imagemagick.org/Usage/resize/#resize_colorspace)
about resizing with respect to color space._
@@ -22,6 +22,8 @@ Supported pixel formats and available optimisations:
- native Rust-code without forced SIMD
- SSE4.1
- AVX2
- `U16x3` - three `u16` components per pixel (e.g. RGB):
- native Rust-code without forced SIMD
- `I32` - one `i32` component per pixel:
- native Rust-code without forced SIMD
- `F32` - one `f32` component per pixel:
@@ -33,7 +35,7 @@ Environment:
- CPU: Intel(R) Core(TM) i7-6700K CPU @ 4.00GHz
- RAM: DDR4 3000 MHz
- Ubuntu 20.04 (linux 5.11)
- Rust 1.57.0
- Rust 1.57.1
- fast_image_resize = "0.6.0"
- glassbench = "0.3.1"
- `rustflags = ["-C", "llvm-args=-x86-branches-within-32B-boundaries"]`
@@ -59,11 +61,11 @@ Pipeline:
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|------------|:-------:|:--------:|:----------:|:--------:|
| image | 92.300 | 180.063 | 257.043 | 342.268 |
| resize | 15.429 | 71.589 | 131.447 | 191.048 |
| fir rust | 0.481 | 55.347 | 89.512 | 121.270 |
| fir sse4.1 | - | 44.841 | 54.630 | 75.284 |
| fir avx2 | - | 10.814 | 14.408 | 20.897 |
| image | 96.466 | 186.243 | 268.888 | 358.235 |
| resize | 15.812 | 68.793 | 125.291 | 181.471 |
| fir rust | 0.495 | 56.372 | 93.182 | 127.870 |
| fir sse4.1 | - | 44.775 | 56.014 | 78.759 |
| fir avx2 | - | 11.290 | 14.731 | 20.678 |
### Resize RGBA image (U8x4) 4928x3279 => 852x567
@@ -76,11 +78,11 @@ Pipeline:
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|------------|:-------:|:--------:|:----------:|:--------:|
| image | 98.598 | 177.427 | 253.149 | 336.999 |
| resize | 17.988 | 79.891 | 149.178 | 218.593 |
| fir rust | 11.952 | 63.214 | 89.397 | 118.714 |
| fir sse4.1 | 9.169 | 20.664 | 26.442 | 34.326 |
| fir avx2 | 7.353 | 15.514 | 18.936 | 24.624 |
| image | 102.809 | 185.787 | 266.163 | 356.038 |
| resize | 18.723 | 83.072 | 156.063 | 229.400 |
| fir rust | 12.518 | 65.821 | 92.371 | 122.981 |
| fir sse4.1 | 9.529 | 21.008 | 27.444 | 35.534 |
| fir avx2 | 7.622 | 15.755 | 19.369 | 25.184 |
### Resize grayscale image (U8) 4928x3279 => 852x567
@@ -94,10 +96,25 @@ Pipeline:
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|----------|:-------:|:--------:|:----------:|:--------:|
| image | 77.209 | 128.704 | 166.905 | 210.503 |
| resize | 9.694 | 24.477 | 47.722 | 80.824 |
| fir rust | 0.198 | 21.457 | 24.256 | 36.893 |
| fir avx2 | - | 9.758 | 7.787 | 12.099 |
| image | 80.937 | 132.497 | 174.721 | 219.534 |
| resize | 10.107 | 25.514 | 49.871 | 84.692 |
| fir rust | 0.208 | 23.255 | 26.471 | 37.642 |
| fir avx2 | - | 9.927 | 8.054 | 12.298 |
### Resize RGB16 image (U16x3) 4928x3279 => 852x567
Pipeline:
`src_image => resize => dst_image`
- Source image [nasa-4928x3279.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279.png)
- Numbers in table is mean duration of image resizing in milliseconds.
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|----------|:-------:|:--------:|:----------:|:--------:|
| image | 98.962 | 180.516 | 255.045 | 335.265 |
| resize | 16.504 | 66.861 | 120.973 | 174.806 |
| fir rust | 0.800 | 53.874 | 89.453 | 123.963 |
## Examples
@@ -128,7 +145,7 @@ fn resize_image_example() {
img.to_rgba8().into_raw(),
fr::PixelType::U8x4,
)
.unwrap();
.unwrap();
// Create MulDiv instance
let alpha_mul_div = fr::MulDiv::default();
@@ -141,9 +158,9 @@ fn resize_image_example() {
let dst_width = NonZeroU32::new(1024).unwrap();
let dst_height = NonZeroU32::new(768).unwrap();
let mut dst_image = fr::Image::new(
dst_width,
dst_height,
src_image.pixel_type(),
dst_width,
dst_height,
src_image.pixel_type(),
);
// Get mutable view of destination image data
+116
View File
@@ -0,0 +1,116 @@
use std::num::NonZeroU32;
use glassbench::*;
use image::imageops;
use resize::Pixel::{RGB16, RGB8};
use rgb::{FromSlice, RGB};
use fast_image_resize::Image;
use fast_image_resize::{CpuExtensions, FilterType, PixelType, ResizeAlg, Resizer};
mod utils;
pub fn bench_downscale_rgb16(bench: &mut Bench) {
let src_image = utils::get_big_rgb16_image();
let new_width = NonZeroU32::new(852).unwrap();
let new_height = NonZeroU32::new(567).unwrap();
let alg_names = ["Nearest", "Bilinear", "CatmullRom", "Lanczos3"];
// image crate
// https://crates.io/crates/image
for alg_name in alg_names {
let filter = match alg_name {
"Nearest" => imageops::Nearest,
"Bilinear" => imageops::Triangle,
"CatmullRom" => imageops::CatmullRom,
"Lanczos3" => imageops::Lanczos3,
_ => continue,
};
bench.task(format!("image - {}", alg_name), |task| {
task.iter(|| {
imageops::resize(&src_image, new_width.get(), new_height.get(), filter);
})
});
}
// resize crate
// https://crates.io/crates/resize
for alg_name in alg_names {
let resize_src_image = src_image.as_raw().as_rgb();
let mut dst =
vec![RGB::new(0u16, 0u16, 0u16); (new_width.get() * new_height.get()) as usize];
bench.task(format!("resize - {}", alg_name), |task| {
let filter = match alg_name {
"Nearest" => resize::Type::Point,
"Bilinear" => resize::Type::Triangle,
"CatmullRom" => resize::Type::Catrom,
"Lanczos3" => resize::Type::Lanczos3,
_ => return,
};
let mut resize = resize::new(
src_image.width() as usize,
src_image.height() as usize,
new_width.get() as usize,
new_height.get() as usize,
RGB16,
filter,
)
.unwrap();
task.iter(|| {
resize.resize(resize_src_image, &mut dst).unwrap();
})
});
}
// fast_image_resize crate;
let src_buffer: Vec<u8> = src_image
.as_raw()
.iter()
.flat_map(|&c| c.to_le_bytes())
.collect();
let mut cpu_ext_and_name = vec![(CpuExtensions::None, "rust")];
// #[cfg(target_arch = "x86_64")]
// {
// cpu_ext_and_name.push((CpuExtensions::Sse4_1, "sse4.1"));
// cpu_ext_and_name.push((CpuExtensions::Avx2, "avx2"));
// }
for (cpu_ext, ext_name) in cpu_ext_and_name {
for alg_name in alg_names {
let src_image_data = Image::from_vec_u8(
NonZeroU32::new(src_image.width()).unwrap(),
NonZeroU32::new(src_image.height()).unwrap(),
src_buffer.clone(),
PixelType::U16x3,
)
.unwrap();
let src_view = src_image_data.view();
let mut dst_image = Image::new(new_width, new_height, src_image_data.pixel_type());
let mut dst_view = dst_image.view_mut();
let resize_alg = match alg_name {
"Nearest" => ResizeAlg::Nearest,
"Bilinear" => ResizeAlg::Convolution(FilterType::Bilinear),
"CatmullRom" => ResizeAlg::Convolution(FilterType::CatmullRom),
"Lanczos3" => ResizeAlg::Convolution(FilterType::Lanczos3),
_ => return,
};
let mut fast_resizer = Resizer::new(resize_alg);
unsafe {
fast_resizer.reset_internal_buffers();
fast_resizer.set_cpu_extensions(cpu_ext);
}
bench.task(format!("fir {} - {}", ext_name, alg_name), |task| {
task.iter(|| {
fast_resizer.resize(&src_view, &mut dst_view).unwrap();
})
});
}
}
utils::print_md_table(bench);
}
glassbench!("Compare resize of RGB16 image", bench_downscale_rgb16,);
+38 -3
View File
@@ -39,6 +39,19 @@ fn get_big_u8x3_source_image() -> Image<'static> {
.unwrap()
}
fn get_big_u16x3_source_image() -> Image<'static> {
let img = utils::get_big_rgb16_image();
let width = img.width();
let height = img.height();
Image::from_vec_u8(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
img.as_raw().iter().flat_map(|&c| c.to_le_bytes()).collect(),
PixelType::U16x3,
)
.unwrap()
}
fn get_big_i32_image() -> Image<'static> {
let img = utils::get_big_luma16_image();
let img_data: Vec<u32> = img
@@ -83,7 +96,7 @@ fn get_small_source_image() -> Image<'static> {
.unwrap()
}
fn native_nearest_bench(bench: &mut Bench) {
fn native_nearest_u8x4_bench(bench: &mut Bench) {
let image = get_big_source_image();
let mut res_image = Image::new(
NonZeroU32::new(NEW_WIDTH).unwrap(),
@@ -245,25 +258,47 @@ fn u8x3_lanczos3_bench(bench: &mut Bench, cpu_extensions: CpuExtensions, name: &
});
}
fn u16x3_lanczos3_bench(bench: &mut Bench, cpu_extensions: CpuExtensions, name: &str) {
let image = get_big_u16x3_source_image();
let mut res_image = Image::new(
NonZeroU32::new(NEW_WIDTH).unwrap(),
NonZeroU32::new(NEW_HEIGHT).unwrap(),
image.pixel_type(),
);
let src_image = image.view();
let mut dst_image = res_image.view_mut();
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
unsafe {
resizer.set_cpu_extensions(cpu_extensions);
}
bench.task(name, |task| {
task.iter(|| {
resizer.resize(&src_image, &mut dst_image).unwrap();
})
});
}
pub fn main() {
use glassbench::*;
let name = env!("CARGO_CRATE_NAME");
let cmd = Command::read();
if cmd.include_bench(name) {
let mut bench = create_bench(name, "Resize", &cmd);
native_nearest_bench(&mut bench);
native_nearest_u8x4_bench(&mut bench);
native_nearest_u8_bench(&mut bench);
u8_lanczos3_bench(&mut bench, CpuExtensions::None, "u8 lanczos3 wo SIMD");
u8x3_lanczos3_bench(&mut bench, CpuExtensions::None, "u8x3 lanczos3 wo SIMD");
u8x4_lanczos3_bench(&mut bench, CpuExtensions::None, "u8x4 lanczos3 wo SIMD");
u16x3_lanczos3_bench(&mut bench, CpuExtensions::None, "u16x3 lanczos3 wo SIMD");
native_lanczos3_i32_bench(&mut bench);
#[cfg(target_arch = "x86_64")]
{
u8_lanczos3_bench(&mut bench, CpuExtensions::Avx2, "u8 lanczos3 avx2");
u8x3_lanczos3_bench(&mut bench, CpuExtensions::Sse4_1, "u8x3 lanczos3 sse4.1");
// u8x3_lanczos3_bench(&mut bench, CpuExtensions::Avx2, "u8x3 lanczos3 avx2");
u8x3_lanczos3_bench(&mut bench, CpuExtensions::Avx2, "u8x3 lanczos3 avx2");
u16x3_lanczos3_bench(&mut bench, CpuExtensions::Avx2, "u16x3 lanczos3 avx2");
u8x4_lanczos3_bench(&mut bench, CpuExtensions::Sse4_1, "u8x4 lanczos3 sse4.1");
u8x4_lanczos3_bench(&mut bench, CpuExtensions::Avx2, "u8x4 lanczos3 avx2");
+12 -1
View File
@@ -3,7 +3,9 @@ use std::env;
use glassbench::*;
use image::io::Reader;
use image::{GrayImage, ImageBuffer, Luma, RgbImage, RgbaImage};
use image::{GrayImage, ImageBuffer, Luma, Rgb, RgbImage, RgbaImage};
pub type Rgb16Image = ImageBuffer<Rgb<u16>, Vec<u16>>;
pub fn get_big_rgb_image() -> RgbImage {
let cur_dir = env::current_dir().unwrap();
@@ -14,6 +16,15 @@ pub fn get_big_rgb_image() -> RgbImage {
img.to_rgb8()
}
pub fn get_big_rgb16_image() -> Rgb16Image {
let cur_dir = env::current_dir().unwrap();
let img = Reader::open(cur_dir.join("data/nasa-4928x3279.png"))
.unwrap()
.decode()
.unwrap();
img.to_rgb16()
}
pub fn get_big_rgba_image() -> RgbaImage {
let cur_dir = env::current_dir().unwrap();
let img = Reader::open(cur_dir.join("data/nasa-4928x3279-rgba.png"))
+3 -3
View File
@@ -134,10 +134,10 @@ fn assert_images<'s, 'd, 'da>(
MulDivImagesError,
> {
let src_image_u8x4 = src_image
.u32_image()
.u8x4_image()
.ok_or(MulDivImagesError::UnsupportedPixelType)?;
let dst_image_u8x4 = dst_image
.u32_image()
.u8x4_image()
.ok_or(MulDivImagesError::UnsupportedPixelType)?;
if src_image_u8x4.width() != dst_image_u8x4.width()
|| src_image_u8x4.height() != dst_image_u8x4.height()
@@ -152,6 +152,6 @@ fn assert_image<'a, 'b>(
image: &'a mut ImageViewMut<'b>,
) -> Result<TypedImageViewMut<'a, 'b, U8x4>, MulDivImageError> {
image
.u32_image()
.u8x4_image()
.ok_or(MulDivImageError::UnsupportedPixelType)
}
+1
View File
@@ -12,6 +12,7 @@ mod f32x1;
mod filters;
mod i32x1;
mod optimisations;
mod u16x3;
mod u8x1;
mod u8x3;
mod u8x4;
+120 -78
View File
@@ -5,62 +5,22 @@ use super::Bound;
// This code is based on C-implementation from Pillow-SIMD package for Python
// https://github.com/uploadcare/pillow-simd
const fn get_clip_table() -> [u8; 1280] {
let mut table = [0u8; 1280];
let mut i: usize = 640;
while i < 640 + 255 {
table[i] = (i - 640) as u8;
i += 1;
}
while i < 1280 {
table[i] = 255;
i += 1;
}
table
}
// Handles values form -640 to 639.
const CLIP8_LOOKUPS: [u8; 1280] = [
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25,
26, 27, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, 41, 42, 43, 44, 45, 46, 47, 48, 49,
50, 51, 52, 53, 54, 55, 56, 57, 58, 59, 60, 61, 62, 63, 64, 65, 66, 67, 68, 69, 70, 71, 72, 73,
74, 75, 76, 77, 78, 79, 80, 81, 82, 83, 84, 85, 86, 87, 88, 89, 90, 91, 92, 93, 94, 95, 96, 97,
98, 99, 100, 101, 102, 103, 104, 105, 106, 107, 108, 109, 110, 111, 112, 113, 114, 115, 116,
117, 118, 119, 120, 121, 122, 123, 124, 125, 126, 127, 128, 129, 130, 131, 132, 133, 134, 135,
136, 137, 138, 139, 140, 141, 142, 143, 144, 145, 146, 147, 148, 149, 150, 151, 152, 153, 154,
155, 156, 157, 158, 159, 160, 161, 162, 163, 164, 165, 166, 167, 168, 169, 170, 171, 172, 173,
174, 175, 176, 177, 178, 179, 180, 181, 182, 183, 184, 185, 186, 187, 188, 189, 190, 191, 192,
193, 194, 195, 196, 197, 198, 199, 200, 201, 202, 203, 204, 205, 206, 207, 208, 209, 210, 211,
212, 213, 214, 215, 216, 217, 218, 219, 220, 221, 222, 223, 224, 225, 226, 227, 228, 229, 230,
231, 232, 233, 234, 235, 236, 237, 238, 239, 240, 241, 242, 243, 244, 245, 246, 247, 248, 249,
250, 251, 252, 253, 254, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
255, 255, 255, 255, 255, 255, 255, 255, 255, 255,
];
const CLIP8_LOOKUPS: [u8; 1280] = get_clip_table();
// 8 bits for result. Filter can have negative areas.
// In one cases the sum of the coefficients will be negative,
@@ -70,20 +30,9 @@ const PRECISION_BITS: u8 = 32 - 8 - 2;
// We use i16 type to store coefficients.
const MAX_COEFS_PRECISION: u8 = 16 - 1;
/// # Safety
/// The function must be used with the `v` and `precision` values
/// such that the expression `v >> precision`
/// produces a result in the range `[-512, 511]`.
#[inline(always)]
pub(crate) unsafe fn clip8(v: i32, precision: u8) -> u8 {
let index = (640 + (v >> precision)) as usize;
// index must be in range [(640-512)..(640+511)]
*CLIP8_LOOKUPS.get_unchecked(index)
}
/// Converts `Vec<f64>` into `&[i16]` without additional memory allocations.
/// The memory buffer from `Vec<f64>` uses as `[i16]` .
pub struct NormalizerGuard {
pub struct NormalizerGuard16 {
values: Vec<f64>,
precision: u8,
}
@@ -94,7 +43,7 @@ pub struct CoefficientsI16Chunk<'a> {
pub values: &'a [i16],
}
impl NormalizerGuard {
impl NormalizerGuard16 {
#[inline]
pub fn new(mut values: Vec<f64>) -> Self {
let max_weight = values
@@ -127,14 +76,7 @@ impl NormalizerGuard {
}
#[inline]
pub fn normalized_i16(&self) -> &[i16] {
let len = self.values.len();
let ptr = self.values.as_ptr();
unsafe { slice::from_raw_parts(ptr as *const i16, len) }
}
#[inline]
pub fn normalized_i16_chunks(
pub fn normalized_chunks(
&self,
window_size: usize,
bounds: &[Bound],
@@ -159,6 +101,103 @@ impl NormalizerGuard {
pub fn precision(&self) -> u8 {
self.precision
}
/// # Safety
/// The function must be used with the `v`
/// such that the expression `v >> self.precision`
/// produces a result in the range `[-512, 511]`.
#[inline(always)]
pub unsafe fn clip(&self, v: i32) -> u8 {
let index = (640 + (v >> self.precision)) as usize;
// index must be in range [(640-512)..(640+511)]
*CLIP8_LOOKUPS.get_unchecked(index)
}
}
// 16 bits for result. Filter can have negative areas.
// In one cases the sum of the coefficients will be negative,
// in the other it will be more than 1.0. That is why we need
// two extra bits for overflow and i64 type.
const PRECISION16_BITS: u8 = 64 - 16 - 2;
// We use i32 type to store coefficients.
const MAX_COEFS_PRECISION16: u8 = 32 - 1;
#[derive(Debug, Clone, Copy)]
pub struct CoefficientsI32Chunk<'a> {
pub start: u32,
pub values: &'a [i32],
}
/// Converts `Vec<f64>` into `&[i32]` without additional memory allocations.
/// The memory buffer from `Vec<f64>` uses as `[i32]` .
pub struct NormalizerGuard32 {
values: Vec<f64>,
precision: u8,
}
impl NormalizerGuard32 {
#[inline]
pub fn new(mut values: Vec<f64>) -> Self {
let max_weight = values
.iter()
.max_by(|&x, &y| x.partial_cmp(y).unwrap())
.unwrap_or(&0.0)
.to_owned();
let mut precision = 0u8;
for cur_precision in 0..PRECISION16_BITS {
precision = cur_precision;
let next_value: i64 = (max_weight * (1i64 << (precision + 1)) as f64).round() as i64;
// The next value will be outside the range, so just stop
if next_value >= (1i64 << MAX_COEFS_PRECISION16) {
break;
}
}
debug_assert!(precision >= 4); // required for some SIMD optimisations
let len = values.len();
let ptr = values.as_mut_ptr();
// Size of `[i32]` always will be not greater than `[f64]` with same number of items
let values_i32 = unsafe { slice::from_raw_parts_mut(ptr as *mut i32, len) };
let scale = (1i64 << precision) as f64;
for (&src, dst) in values.iter().zip(values_i32.iter_mut()) {
*dst = (src * scale).round() as i32;
}
Self { values, precision }
}
#[inline]
pub fn normalized_chunks(
&self,
window_size: usize,
bounds: &[Bound],
) -> Vec<CoefficientsI32Chunk> {
let len = self.values.len();
let ptr = self.values.as_ptr();
let mut cooefs = unsafe { slice::from_raw_parts(ptr as *const i32, len) };
let mut res = Vec::with_capacity(bounds.len());
for bound in bounds {
let (left, right) = cooefs.split_at(window_size);
cooefs = right;
let size = bound.size as usize;
res.push(CoefficientsI32Chunk {
start: bound.start,
values: &left[0..size],
});
}
res
}
#[inline]
pub fn precision(&self) -> u8 {
self.precision
}
#[inline(always)]
pub fn clip(&self, v: i64) -> u16 {
(v >> self.precision).min(u16::MAX as i64).max(0) as u16
}
}
#[cfg(test)]
@@ -167,7 +206,10 @@ mod tests {
#[test]
fn test_minimal_precision() {
assert!(NormalizerGuard::new(vec![0.0]).precision() >= 4);
assert!(NormalizerGuard::new(vec![2.0]).precision() >= 4);
// required for some SIMD optimisations
assert!(NormalizerGuard16::new(vec![0.0]).precision() >= 4);
assert!(NormalizerGuard16::new(vec![2.0]).precision() >= 4);
assert!(NormalizerGuard32::new(vec![0.0]).precision() >= 4);
assert!(NormalizerGuard32::new(vec![2.0]).precision() >= 4);
}
}
+27
View File
@@ -0,0 +1,27 @@
use super::{Coefficients, Convolution};
use crate::image_view::{TypedImageView, TypedImageViewMut};
use crate::pixels::U16x3;
use crate::CpuExtensions;
mod native;
impl Convolution for U16x3 {
fn horiz_convolution(
src_image: TypedImageView<Self>,
dst_image: TypedImageViewMut<Self>,
offset: u32,
coeffs: Coefficients,
cpu_extensions: CpuExtensions,
) {
native::horiz_convolution(src_image, dst_image, offset, coeffs);
}
fn vert_convolution(
src_image: TypedImageView<Self>,
dst_image: TypedImageViewMut<Self>,
coeffs: Coefficients,
cpu_extensions: CpuExtensions,
) {
native::vert_convolution(src_image, dst_image, coeffs);
}
}
+70
View File
@@ -0,0 +1,70 @@
use crate::convolution::{optimisations, Coefficients};
use crate::image_view::{TypedImageView, TypedImageViewMut};
use crate::pixels::U16x3;
#[inline(always)]
pub(crate) fn horiz_convolution(
src_image: TypedImageView<U16x3>,
mut dst_image: TypedImageViewMut<U16x3>,
offset: u32,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard32::new(values);
let precision = normalizer_guard.precision();
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
let initial: i64 = 1 << (precision - 1);
let src_rows = src_image.iter_rows(offset);
let dst_rows = dst_image.iter_rows_mut();
for (dst_row, src_row) in dst_rows.zip(src_rows) {
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
let first_x_src = coeffs_chunk.start as usize;
let mut ss = [initial; 3];
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
for (&k, src_pixel) in coeffs_chunk.values.iter().zip(src_pixels) {
for (i, s) in ss.iter_mut().enumerate() {
*s += src_pixel.0[i] as i64 * (k as i64);
}
}
for (i, s) in ss.iter().copied().enumerate() {
dst_pixel.0[i] = normalizer_guard.clip(s);
}
}
}
}
#[inline(always)]
pub(crate) fn vert_convolution(
src_image: TypedImageView<U16x3>,
mut dst_image: TypedImageViewMut<U16x3>,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard32::new(values);
let precision = normalizer_guard.precision();
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
let initial = 1 << (precision - 1);
let dst_rows = dst_image.iter_rows_mut();
for (&coeffs_chunk, dst_row) in coefficients_chunks.iter().zip(dst_rows) {
let first_y_src = coeffs_chunk.start;
let ks = coeffs_chunk.values;
for (x_src, dst_pixel) in dst_row.iter_mut().enumerate() {
let mut ss = [initial; 3];
let src_rows = src_image.iter_rows(first_y_src);
for (&k, src_row) in ks.iter().zip(src_rows) {
let src_pixel = unsafe { src_row.get_unchecked(x_src as usize) };
for (i, s) in ss.iter_mut().enumerate() {
*s += src_pixel.0[i] as i64 * (k as i64);
}
}
for (i, s) in ss.iter().copied().enumerate() {
dst_pixel.0[i] = normalizer_guard.clip(s);
}
}
}
}
+38 -37
View File
@@ -1,7 +1,7 @@
use std::arch::x86_64::*;
use crate::convolution::optimisations::CoefficientsI16Chunk;
use crate::convolution::{optimisations, Bound, Coefficients};
use crate::convolution::optimisations::{CoefficientsI16Chunk, NormalizerGuard16};
use crate::convolution::{optimisations, Coefficients};
use crate::image_view::{FourRows, FourRowsMut, TypedImageView, TypedImageViewMut};
use crate::pixels::U8;
use crate::simd_utils;
@@ -16,17 +16,15 @@ pub(crate) fn horiz_convolution(
let (values, window_size, bounds_per_pixel) =
(coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coefficients_chunks =
normalizer_guard.normalized_i16_chunks(window_size, &bounds_per_pixel);
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
let dst_height = dst_image.height().get();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_image.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_8u4x(src_rows, dst_rows, &coefficients_chunks, precision);
horiz_convolution_8u4x(src_rows, dst_rows, &coefficients_chunks, &normalizer_guard);
}
}
@@ -37,7 +35,7 @@ pub(crate) fn horiz_convolution(
src_image.get_row(yy + offset).unwrap(),
dst_image.get_row_mut(yy).unwrap(),
&coefficients_chunks,
precision,
&normalizer_guard,
);
}
yy += 1;
@@ -50,17 +48,16 @@ pub(crate) fn vert_convolution(
mut dst_image: TypedImageViewMut<U8>,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let (values, window_size, bounds_per_pixel) =
(coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coeffs_i16 = normalizer_guard.normalized_i16();
let coeffs_chunks = coeffs_i16.chunks(window_size);
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
let dst_rows = dst_image.iter_rows_mut();
for ((&bound, k), dst_row) in bounds.iter().zip(coeffs_chunks).zip(dst_rows) {
for (dst_row, coeffs_chunk) in dst_rows.zip(coefficients_chunks) {
unsafe {
vert_convolution_8u(&src_image, dst_row, k, bound, precision);
vert_convolution_8u(&src_image, dst_row, coeffs_chunk, &normalizer_guard);
}
}
}
@@ -77,13 +74,13 @@ unsafe fn horiz_convolution_8u4x(
src_rows: FourRows<U8>,
dst_rows: FourRowsMut<U8>,
coefficients_chunks: &[CoefficientsI16Chunk],
precision: u8,
normalizer_guard: &NormalizerGuard16,
) {
let s_rows = [src_rows.0, src_rows.1, src_rows.2, src_rows.3];
let d_rows = [dst_rows.0, dst_rows.1, dst_rows.2, dst_rows.3];
let zero = _mm_setzero_si128();
// 8 components will be added, use only 1/8 of the error
let initial = _mm256_set1_epi32(1 << (precision - 4));
let initial = _mm256_set1_epi32(1 << (normalizer_guard.precision() - 4));
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let coeffs = coeffs_chunk.values;
@@ -130,7 +127,7 @@ unsafe fn horiz_convolution_8u4x(
x += 1;
}
let result_u8x4 = result_i32x4.map(|v| optimisations::clip8(v, precision));
let result_u8x4 = result_i32x4.map(|v| normalizer_guard.clip(v));
for i in 0..4 {
d_rows[i].get_unchecked_mut(dst_x).0 = result_u8x4[i];
}
@@ -148,11 +145,11 @@ unsafe fn horiz_convolution_8u(
src_row: &[U8],
dst_row: &mut [U8],
coefficients_chunks: &[CoefficientsI16Chunk],
precision: u8,
normalizer_guard: &NormalizerGuard16,
) {
let zero = _mm_setzero_si128();
// 8 components will be added, use only 1/8 of the error
let initial = _mm256_set1_epi32(1 << (precision - 4));
let initial = _mm256_set1_epi32(1 << (normalizer_guard.precision() - 4));
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let coeffs = coeffs_chunk.values;
@@ -194,7 +191,7 @@ unsafe fn horiz_convolution_8u(
x += 1;
}
dst_row.get_unchecked_mut(dst_x).0 = optimisations::clip8(result_i32, precision);
dst_row.get_unchecked_mut(dst_x).0 = normalizer_guard.clip(result_i32);
}
}
@@ -203,13 +200,14 @@ unsafe fn horiz_convolution_8u(
unsafe fn vert_convolution_8u(
src_img: &TypedImageView<U8>,
dst_row: &mut [U8],
coeffs: &[i16],
bound: Bound,
precision: u8,
coeffs_chunk: CoefficientsI16Chunk,
normalizer_guard: &NormalizerGuard16,
) {
let src_width = src_img.width().get() as usize;
let y_start = bound.start;
let y_size = bound.size;
let y_start = coeffs_chunk.start;
let coeffs = coeffs_chunk.values;
let max_y = y_start + coeffs.len() as u32;
let precision = normalizer_guard.precision();
let initial = _mm_set1_epi32(1 << (precision - 1));
let initial_256 = _mm256_set1_epi32(1 << (precision - 1));
@@ -225,7 +223,7 @@ unsafe fn vert_convolution_8u(
let mut y: u32 = 0;
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, max_y) {
// Load two coefficients at once
let two_coeffs = simd_utils::ptr_i16_to_256set1_epi32(coeffs, y as usize);
let row1 = simd_utils::loadu_si256(s_row1, x); // top line
@@ -246,8 +244,9 @@ unsafe fn vert_convolution_8u(
y += 2;
}
if let Some(s_row) = src_img.get_row(y_start + y) {
let one_coeff = _mm256_set1_epi32(coeffs[y as usize] as i32);
if let Some(&k) = coeffs.get(y as usize) {
let s_row = src_img.get_row(y_start + y).unwrap();
let one_coeff = _mm256_set1_epi32(k as i32);
let row1 = simd_utils::loadu_si256(s_row, x); // top line
let row2 = _mm256_setzero_si256(); // bottom line is empty
@@ -289,7 +288,7 @@ unsafe fn vert_convolution_8u(
let mut sss1 = initial; // right row
let mut y: u32 = 0;
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, max_y) {
// Load two coefficients at once
let two_coeffs = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
@@ -305,8 +304,9 @@ unsafe fn vert_convolution_8u(
y += 2;
}
if let Some(s_row) = src_img.get_row(y_start + y) {
let one_coeff = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
if let Some(&k) = coeffs.get(y as usize) {
let s_row = src_img.get_row(y_start + y).unwrap();
let one_coeff = _mm_set1_epi32(k as i32);
let row1 = simd_utils::loadl_epi64(s_row, x); // top line
let row2 = _mm_setzero_si128(); // bottom line is empty
@@ -337,7 +337,7 @@ unsafe fn vert_convolution_8u(
while x < src_width.saturating_sub(3) {
let mut sss = initial;
let mut y: u32 = 0;
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, max_y) {
// Load two coefficients at once
let two_coeffs = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
@@ -351,9 +351,10 @@ unsafe fn vert_convolution_8u(
y += 2;
}
if let Some(s_row) = src_img.get_row(y_start + y) {
if let Some(&k) = coeffs.get(y as usize) {
let s_row = src_img.get_row(y_start + y).unwrap();
let pix = simd_utils::mm_cvtepu8_epi32_from_u8(s_row, x);
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
let mmk = _mm_set1_epi32(k as i32);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
}
@@ -375,11 +376,11 @@ unsafe fn vert_convolution_8u(
for dst_pixel in dst_row.iter_mut().skip(x) {
let mut ss0 = 1 << (precision - 1);
for (dy, &k) in coeffs.iter().take(y_size as usize).enumerate() {
for (dy, &k) in coeffs.iter().enumerate() {
let src_pixel = src_img.get_pixel(x as u32, y_start + dy as u32);
ss0 += src_pixel.0 as i32 * (k as i32);
}
dst_pixel.0 = optimisations::clip8(ss0, precision);
dst_pixel.0 = normalizer_guard.clip(ss0);
x += 1;
}
}
+6 -6
View File
@@ -11,9 +11,9 @@ pub(crate) fn horiz_convolution(
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
let precision = normalizer_guard.precision();
let coefficients_chunks = normalizer_guard.normalized_i16_chunks(window_size, &bounds);
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
let initial = 1 << (precision - 1);
let src_rows = src_image.iter_rows(offset);
@@ -28,7 +28,7 @@ pub(crate) fn horiz_convolution(
for (&k, &src_pixel) in ks.iter().zip(src_pixels) {
ss += src_pixel.0 as i32 * (k as i32);
}
dst_pixel.0 = unsafe { optimisations::clip8(ss, precision) };
dst_pixel.0 = unsafe { normalizer_guard.clip(ss) };
}
}
}
@@ -41,9 +41,9 @@ pub(crate) fn vert_convolution(
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
let precision = normalizer_guard.precision();
let coefficients_chunks = normalizer_guard.normalized_i16_chunks(window_size, &bounds);
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
let initial = 1 << (precision - 1);
let dst_rows = dst_image.iter_rows_mut();
@@ -58,7 +58,7 @@ pub(crate) fn vert_convolution(
let src_pixel = unsafe { src_row.get_unchecked(x_src as usize) };
ss += src_pixel.0 as i32 * (k as i32);
}
dst_pixel.0 = unsafe { optimisations::clip8(ss, precision) };
dst_pixel.0 = unsafe { normalizer_guard.clip(ss) };
}
}
}
+30 -28
View File
@@ -1,8 +1,8 @@
use std::arch::x86_64::*;
use std::intrinsics::transmute;
use crate::convolution::optimisations::CoefficientsI16Chunk;
use crate::convolution::{optimisations, Bound, Coefficients};
use crate::convolution::optimisations::{CoefficientsI16Chunk, NormalizerGuard16};
use crate::convolution::{optimisations, Coefficients};
use crate::image_view::{FourRows, FourRowsMut, TypedImageView, TypedImageViewMut};
use crate::pixels::{Pixel, U8x3};
use crate::simd_utils;
@@ -17,10 +17,9 @@ pub(crate) fn horiz_convolution(
let (values, window_size, bounds_per_pixel) =
(coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
let precision = normalizer_guard.precision();
let coefficients_chunks =
normalizer_guard.normalized_i16_chunks(window_size, &bounds_per_pixel);
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
let dst_height = dst_image.height().get();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
@@ -51,17 +50,16 @@ pub(crate) fn vert_convolution(
mut dst_image: TypedImageViewMut<U8x3>,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let (values, window_size, bounds_per_pixel) =
(coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coeffs_i16 = normalizer_guard.normalized_i16();
let coeffs_chunks = coeffs_i16.chunks(window_size);
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
let dst_rows = dst_image.iter_rows_mut();
for ((&bound, k), dst_row) in bounds.iter().zip(coeffs_chunks).zip(dst_rows) {
for (dst_row, coeffs_chunk) in dst_rows.zip(coefficients_chunks) {
unsafe {
vert_convolution_8u(&src_image, dst_row, k, bound, precision);
vert_convolution_8u(&src_image, dst_row, coeffs_chunk, &normalizer_guard);
}
}
}
@@ -410,13 +408,14 @@ unsafe fn horiz_convolution_8u(
unsafe fn vert_convolution_8u(
src_img: &TypedImageView<U8x3>,
dst_row: &mut [U8x3],
coeffs: &[i16],
bound: Bound,
precision: u8,
coeffs_chunk: CoefficientsI16Chunk,
normalizer_guard: &NormalizerGuard16,
) {
let src_width = src_img.width().get() as usize;
let y_start = bound.start;
let y_size = bound.size;
let y_start = coeffs_chunk.start;
let coeffs = coeffs_chunk.values;
let max_y = y_start + coeffs.len() as u32;
let precision = normalizer_guard.precision();
let initial = _mm_set1_epi32(1 << (precision - 1));
let initial_256 = _mm256_set1_epi32(1 << (precision - 1));
@@ -433,7 +432,7 @@ unsafe fn vert_convolution_8u(
let mut y: u32 = 0;
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, max_y) {
// Load two coefficients at once
let mmk = simd_utils::ptr_i16_to_256set1_epi32(coeffs, y as usize);
@@ -455,8 +454,9 @@ unsafe fn vert_convolution_8u(
y += 2;
}
if let Some(s_row) = src_img.get_row(y_start + y) {
let mmk = _mm256_set1_epi32(coeffs[y as usize] as i32);
if let Some(&k) = coeffs.get(y as usize) {
let s_row = src_img.get_row(y_start + y).unwrap();
let mmk = _mm256_set1_epi32(k as i32);
let source1 = simd_utils::loadu_si256_raw(s_row, x_in_bytes); // top line
let source2 = _mm256_setzero_si256(); // bottom line is empty
@@ -499,7 +499,7 @@ unsafe fn vert_convolution_8u(
let mut sss1 = initial; // right row
let mut y: u32 = 0;
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, max_y) {
// Load two coefficients at once
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
@@ -515,8 +515,9 @@ unsafe fn vert_convolution_8u(
y += 2;
}
if let Some(s_row) = src_img.get_row(y_start + y) {
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
if let Some(&k) = coeffs.get(y as usize) {
let s_row = src_img.get_row(y_start + y).unwrap();
let mmk = _mm_set1_epi32(k as i32);
let source1 = simd_utils::loadl_epi64_raw(s_row, x_in_bytes); // top line
let source2 = _mm_setzero_si128(); // bottom line is empty
@@ -548,7 +549,7 @@ unsafe fn vert_convolution_8u(
while x_in_bytes < width_in_bytes.saturating_sub(3) {
let mut sss = initial;
let mut y: u32 = 0;
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, max_y) {
// Load two coefficients at once
let two_coeffs = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
@@ -562,9 +563,10 @@ unsafe fn vert_convolution_8u(
y += 2;
}
if let Some(s_row) = src_img.get_row(y_start + y) {
if let Some(&k) = coeffs.get(y as usize) {
let s_row = src_img.get_row(y_start + y).unwrap();
let pix = simd_utils::mm_cvtepu8_epi32_from_raw(s_row, x_in_bytes);
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
let mmk = _mm_set1_epi32(k as i32);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
}
@@ -588,14 +590,14 @@ unsafe fn vert_convolution_8u(
for dst_pixel in dst_u8 {
let mut ss0 = 1 << (precision - 1);
for (dy, &k) in coeffs.iter().take(y_size as usize).enumerate() {
for (dy, &k) in coeffs.iter().enumerate() {
if let Some(src_row) = src_img.get_row(y_start + dy as u32) {
let src_ptr = src_row.as_ptr() as *const u8;
let src_component = *src_ptr.add(x_in_bytes);
ss0 += src_component as i32 * (k as i32);
}
}
*dst_pixel = optimisations::clip8(ss0, precision);
*dst_pixel = normalizer_guard.clip(ss0);
x_in_bytes += 1;
}
}
+6 -6
View File
@@ -11,9 +11,9 @@ pub(crate) fn horiz_convolution(
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
let precision = normalizer_guard.precision();
let coefficients_chunks = normalizer_guard.normalized_i16_chunks(window_size, &bounds);
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
let initial = 1 << (precision - 1);
let src_rows = src_image.iter_rows(offset);
@@ -29,7 +29,7 @@ pub(crate) fn horiz_convolution(
}
}
for (i, s) in ss.iter().copied().enumerate() {
dst_pixel.0[i] = unsafe { optimisations::clip8(s, precision) };
dst_pixel.0[i] = unsafe { normalizer_guard.clip(s) };
}
}
}
@@ -43,9 +43,9 @@ pub(crate) fn vert_convolution(
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
let precision = normalizer_guard.precision();
let coefficients_chunks = normalizer_guard.normalized_i16_chunks(window_size, &bounds);
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
let initial = 1 << (precision - 1);
let dst_rows = dst_image.iter_rows_mut();
@@ -63,7 +63,7 @@ pub(crate) fn vert_convolution(
}
}
for (i, s) in ss.iter().copied().enumerate() {
dst_pixel.0[i] = unsafe { optimisations::clip8(s, precision) };
dst_pixel.0[i] = unsafe { normalizer_guard.clip(s) };
}
}
}
+28 -26
View File
@@ -1,8 +1,8 @@
use std::arch::x86_64::*;
use std::intrinsics::transmute;
use crate::convolution::optimisations::CoefficientsI16Chunk;
use crate::convolution::{optimisations, Bound, Coefficients};
use crate::convolution::optimisations::{CoefficientsI16Chunk, NormalizerGuard16};
use crate::convolution::{optimisations, Coefficients};
use crate::image_view::{FourRows, FourRowsMut, TypedImageView, TypedImageViewMut};
use crate::pixels::U8x4;
use crate::simd_utils;
@@ -20,10 +20,9 @@ pub(crate) fn horiz_convolution(
let (values, window_size, bounds_per_pixel) =
(coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
let precision = normalizer_guard.precision();
let coefficients_chunks =
normalizer_guard.normalized_i16_chunks(window_size, &bounds_per_pixel);
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
let dst_height = dst_image.height().get();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
@@ -54,17 +53,16 @@ pub(crate) fn vert_convolution(
mut dst_image: TypedImageViewMut<U8x4>,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let (values, window_size, bounds_per_pixel) =
(coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coeffs_i16 = normalizer_guard.normalized_i16();
let coeffs_chunks = coeffs_i16.chunks(window_size);
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
let dst_rows = dst_image.iter_rows_mut();
for ((&bound, k), dst_row) in bounds.iter().zip(coeffs_chunks).zip(dst_rows) {
for (dst_row, coeffs_chunk) in dst_rows.zip(coefficients_chunks) {
unsafe {
vert_convolution_8u(&src_image, dst_row, k, bound, precision);
vert_convolution_8u(&src_image, dst_row, coeffs_chunk, &normalizer_guard);
}
}
}
@@ -332,13 +330,14 @@ unsafe fn horiz_convolution_8u(
unsafe fn vert_convolution_8u(
src_img: &TypedImageView<U8x4>,
dst_row: &mut [U8x4],
coeffs: &[i16],
bound: Bound,
precision: u8,
coeffs_chunk: CoefficientsI16Chunk,
normalizer_guard: &NormalizerGuard16,
) {
let src_width = src_img.width().get() as usize;
let y_start = bound.start;
let y_size = bound.size;
let y_start = coeffs_chunk.start;
let coeffs = coeffs_chunk.values;
let max_y = y_start + coeffs.len() as u32;
let precision = normalizer_guard.precision();
let initial = _mm_set1_epi32(1 << (precision - 1));
let initial_256 = _mm256_set1_epi32(1 << (precision - 1));
@@ -352,7 +351,7 @@ unsafe fn vert_convolution_8u(
let mut y: u32 = 0;
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, max_y) {
// Load two coefficients at once
let mmk = simd_utils::ptr_i16_to_256set1_epi32(coeffs, y as usize);
@@ -374,8 +373,9 @@ unsafe fn vert_convolution_8u(
y += 2;
}
if let Some(s_row) = src_img.get_row(y_start + y) {
let mmk = _mm256_set1_epi32(coeffs[y as usize] as i32);
if let Some(&k) = coeffs.get(y as usize) {
let s_row = src_img.get_row(y_start + y).unwrap();
let mmk = _mm256_set1_epi32(k as i32);
let source1 = simd_utils::loadu_si256(s_row, x); // top line
let source2 = _mm256_setzero_si256(); // bottom line is empty
@@ -417,7 +417,7 @@ unsafe fn vert_convolution_8u(
let mut sss1 = initial; // right row
let mut y: u32 = 0;
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, max_y) {
// Load two coefficients at once
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
@@ -433,8 +433,9 @@ unsafe fn vert_convolution_8u(
y += 2;
}
if let Some(s_row) = src_img.get_row(y_start + y) {
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
if let Some(&k) = coeffs.get(y as usize) {
let s_row = src_img.get_row(y_start + y).unwrap();
let mmk = _mm_set1_epi32(k as i32);
let source1 = simd_utils::loadl_epi64(s_row, x); // top line
let source2 = _mm_setzero_si128(); // bottom line is empty
@@ -465,7 +466,7 @@ unsafe fn vert_convolution_8u(
if x < src_width {
let mut sss = initial;
let mut y: u32 = 0;
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, max_y) {
// Load two coefficients at once
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
@@ -479,9 +480,10 @@ unsafe fn vert_convolution_8u(
y += 2;
}
if let Some(s_row) = src_img.get_row(y_start + y) {
if let Some(&k) = coeffs.get(y as usize) {
let s_row = src_img.get_row(y_start + y).unwrap();
let pix = simd_utils::mm_cvtepu8_epi32(s_row, x);
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
let mmk = _mm_set1_epi32(k as i32);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
}
+6 -8
View File
@@ -10,9 +10,9 @@ pub(crate) fn horiz_convolution(
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
let precision = normalizer_guard.precision();
let coefficients_chunks = normalizer_guard.normalized_i16_chunks(window_size, &bounds);
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
let initial = 1 << (precision - 1);
let src_rows = src_image.iter_rows(offset);
@@ -29,8 +29,7 @@ pub(crate) fn horiz_convolution(
*s += components[i] as i32 * (k as i32);
}
}
dst_pixel.0 =
u32::from_le_bytes(ss.map(|v| unsafe { optimisations::clip8(v, precision) }));
dst_pixel.0 = u32::from_le_bytes(ss.map(|v| unsafe { normalizer_guard.clip(v) }));
}
}
}
@@ -42,9 +41,9 @@ pub(crate) fn vert_convolution(
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
let precision = normalizer_guard.precision();
let coefficients_chunks = normalizer_guard.normalized_i16_chunks(window_size, &bounds);
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
let initial = 1 << (precision - 1);
let dst_rows = dst_image.iter_rows_mut();
@@ -62,8 +61,7 @@ pub(crate) fn vert_convolution(
*s += components[i] as i32 * (k as i32);
}
}
dst_pixel.0 =
u32::from_le_bytes(ss.map(|v| unsafe { optimisations::clip8(v, precision) }));
dst_pixel.0 = u32::from_le_bytes(ss.map(|v| unsafe { normalizer_guard.clip(v) }));
}
}
}
+28 -27
View File
@@ -1,8 +1,8 @@
use std::arch::x86_64::*;
use std::intrinsics::transmute;
use crate::convolution::optimisations::CoefficientsI16Chunk;
use crate::convolution::{optimisations, Bound, Coefficients};
use crate::convolution::optimisations::{CoefficientsI16Chunk, NormalizerGuard16};
use crate::convolution::{optimisations, Coefficients};
use crate::image_view::{FourRows, FourRowsMut, TypedImageView, TypedImageViewMut};
use crate::pixels::U8x4;
use crate::simd_utils;
@@ -20,10 +20,9 @@ pub(crate) fn horiz_convolution(
let (values, window_size, bounds_per_pixel) =
(coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
let precision = normalizer_guard.precision();
let coefficients_chunks =
normalizer_guard.normalized_i16_chunks(window_size, &bounds_per_pixel);
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
let dst_height = dst_image.height().get();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
@@ -54,17 +53,16 @@ pub(crate) fn vert_convolution(
mut dst_image: TypedImageViewMut<U8x4>,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let (values, window_size, bounds_per_pixel) =
(coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coeffs_i16 = normalizer_guard.normalized_i16();
let coeffs_chunks = coeffs_i16.chunks(window_size);
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
let dst_rows = dst_image.iter_rows_mut();
for ((&bound, k), dst_row) in bounds.iter().zip(coeffs_chunks).zip(dst_rows) {
for (dst_row, coeffs_chunk) in dst_rows.zip(coefficients_chunks) {
unsafe {
vert_convolution_8u(&src_image, dst_row, k, bound, precision);
vert_convolution_8u(&src_image, dst_row, coeffs_chunk, &normalizer_guard);
}
}
}
@@ -313,14 +311,15 @@ unsafe fn horiz_convolution_8u(
pub(crate) unsafe fn vert_convolution_8u(
src_img: &TypedImageView<U8x4>,
dst_row: &mut [U8x4],
coeffs: &[i16],
bound: Bound,
precision: u8,
coeffs_chunk: CoefficientsI16Chunk,
normalizer_guard: &NormalizerGuard16,
) {
let mut xx: usize = 0;
let src_width = src_img.width().get() as usize;
let y_start = bound.start;
let y_size = bound.size;
let y_start = coeffs_chunk.start;
let coeffs = coeffs_chunk.values;
let max_y = y_start + coeffs.len() as u32;
let precision = normalizer_guard.precision();
let initial = _mm_set1_epi32(1 << (precision - 1));
@@ -336,7 +335,7 @@ pub(crate) unsafe fn vert_convolution_8u(
let mut y: u32 = 0;
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, max_y) {
// Load two coefficients at once
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
@@ -373,8 +372,9 @@ pub(crate) unsafe fn vert_convolution_8u(
y += 2;
}
if let Some(s_row) = src_img.get_row(y_start + y) {
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
if let Some(&k) = coeffs.get(y as usize) {
let s_row = src_img.get_row(y_start + y).unwrap();
let mmk = _mm_set1_epi32(k as i32);
let mut source1 = simd_utils::loadu_si128(s_row, xx); // top line
@@ -438,7 +438,7 @@ pub(crate) unsafe fn vert_convolution_8u(
let mut sss1 = initial; // right row
let mut y: u32 = 0;
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, max_y) {
// Load two coefficients at once
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
@@ -454,8 +454,9 @@ pub(crate) unsafe fn vert_convolution_8u(
y += 2;
}
if let Some(s_row) = src_img.get_row(y_start + y) {
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
if let Some(&k) = coeffs.get(y as usize) {
let s_row = src_img.get_row(y_start + y).unwrap();
let mmk = _mm_set1_epi32(k as i32);
let source1 = simd_utils::loadl_epi64(s_row, xx); // top line
@@ -479,7 +480,6 @@ pub(crate) unsafe fn vert_convolution_8u(
let dst_ptr = dst_row.get_unchecked_mut(xx..).as_mut_ptr() as *mut __m128i;
_mm_storel_epi64(dst_ptr, sss0);
//
xx += 2;
}
@@ -487,7 +487,7 @@ pub(crate) unsafe fn vert_convolution_8u(
let mut sss = initial;
let mut y: u32 = 0;
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, y_start + y_size) {
for (s_row1, s_row2) in src_img.iter_2_rows(y_start, max_y) {
// Load two coefficients at once
let mmk = simd_utils::ptr_i16_to_set1_epi32(coeffs, y as usize);
@@ -501,9 +501,10 @@ pub(crate) unsafe fn vert_convolution_8u(
y += 2;
}
if let Some(s_row) = src_img.get_row(y_start + y) {
if let Some(&k) = coeffs.get(y as usize) {
let s_row = src_img.get_row(y_start + y).unwrap();
let pix = simd_utils::mm_cvtepu8_epi32(s_row, xx);
let mmk = _mm_set1_epi32(*coeffs.get_unchecked(y as usize) as i32);
let mmk = _mm_set1_epi32(k as i32);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
}
+10 -1
View File
@@ -1,7 +1,7 @@
use std::num::NonZeroU32;
use crate::image_view::{ImageRows, ImageRowsMut, TypedImageView, TypedImageViewMut};
use crate::pixels::{Pixel, PixelType, U8x3, U8x4, F32, I32, U8};
use crate::pixels::{Pixel, PixelType, U16x3, U8x3, U8x4, F32, I32, U8};
use crate::{ImageBufferError, ImageView, ImageViewMut, InvalidBufferSizeError};
#[derive(Debug)]
@@ -27,6 +27,7 @@ impl<'a> Image<'a> {
let pixels_count = (width.get() * height.get()) as usize;
let pixels = match pixel_type {
PixelType::U8x3 => PixelsContainer::VecU8(vec![0; pixels_count * U8x3::size()]),
PixelType::U16x3 => PixelsContainer::VecU8(vec![0; pixels_count * U16x3::size()]),
PixelType::U8x4 | PixelType::I32 | PixelType::F32 => {
PixelsContainer::VecU32(vec![0; pixels_count])
}
@@ -166,6 +167,10 @@ impl<'a> Image<'a> {
let pixels = unsafe { buffer.align_to::<U8x4>().1 };
ImageRows::U8x4(pixels.chunks_exact(self.width.get() as usize).collect())
}
PixelType::U16x3 => {
let pixels = unsafe { buffer.align_to::<U16x3>().1 };
ImageRows::U16x3(pixels.chunks_exact(self.width.get() as usize).collect())
}
PixelType::I32 => {
let pixels = unsafe { buffer.align_to::<I32>().1 };
ImageRows::I32(pixels.chunks_exact(self.width.get() as usize).collect())
@@ -197,6 +202,10 @@ impl<'a> Image<'a> {
let pixels = unsafe { buffer.align_to_mut::<U8x4>().1 };
ImageRowsMut::U8x4(pixels.chunks_exact_mut(width.get() as usize).collect())
}
PixelType::U16x3 => {
let pixels = unsafe { buffer.align_to_mut::<U16x3>().1 };
ImageRowsMut::U16x3(pixels.chunks_exact_mut(width.get() as usize).collect())
}
PixelType::I32 => {
let pixels = unsafe { buffer.align_to_mut::<I32>().1 };
ImageRowsMut::I32(pixels.chunks_exact_mut(width.get() as usize).collect())
+42 -3
View File
@@ -2,7 +2,7 @@ use std::num::NonZeroU32;
use std::slice;
use crate::errors::{CropBoxError, ImageBufferError, ImageRowsError};
use crate::pixels::{Pixel, PixelType, U8x3, U8x4, F32, I32, U8};
use crate::pixels::{Pixel, PixelType, U16x3, U8x3, U8x4, F32, I32, U8};
pub(crate) type RowMut<'a, 'b, T> = &'a mut &'b mut [T];
pub(crate) type TwoRows<'a, T> = (&'a [T], &'a [T]);
@@ -28,6 +28,7 @@ pub struct CropBox {
pub enum ImageRows<'a> {
U8x3(Vec<&'a [U8x3]>),
U8x4(Vec<&'a [U8x4]>),
U16x3(Vec<&'a [U16x3]>),
I32(Vec<&'a [I32]>),
F32(Vec<&'a [F32]>),
U8(Vec<&'a [U8]>),
@@ -42,6 +43,7 @@ impl<'a> ImageRows<'a> {
match self {
ImageRows::U8x3(rows) => check_rows_count_and_size(width, height, rows),
ImageRows::U8x4(rows) => check_rows_count_and_size(width, height, rows),
ImageRows::U16x3(rows) => check_rows_count_and_size(width, height, rows),
ImageRows::I32(rows) => check_rows_count_and_size(width, height, rows),
ImageRows::F32(rows) => check_rows_count_and_size(width, height, rows),
ImageRows::U8(rows) => check_rows_count_and_size(width, height, rows),
@@ -52,6 +54,7 @@ impl<'a> ImageRows<'a> {
match self {
Self::U8x3(_) => PixelType::U8x3,
Self::U8x4(_) => PixelType::U8x4,
Self::U16x3(_) => PixelType::U16x3,
Self::I32(_) => PixelType::I32,
Self::F32(_) => PixelType::F32,
Self::U8(_) => PixelType::U8,
@@ -64,6 +67,7 @@ impl<'a> ImageRows<'a> {
pub enum ImageRowsMut<'a> {
U8x3(Vec<&'a mut [U8x3]>),
U8x4(Vec<&'a mut [U8x4]>),
U16x3(Vec<&'a mut [U16x3]>),
I32(Vec<&'a mut [I32]>),
F32(Vec<&'a mut [F32]>),
U8(Vec<&'a mut [U8]>),
@@ -78,6 +82,7 @@ impl<'a> ImageRowsMut<'a> {
match self {
Self::U8x3(rows) => check_rows_count_and_size(width, height, rows),
Self::U8x4(rows) => check_rows_count_and_size(width, height, rows),
Self::U16x3(rows) => check_rows_count_and_size(width, height, rows),
Self::I32(rows) => check_rows_count_and_size(width, height, rows),
Self::F32(rows) => check_rows_count_and_size(width, height, rows),
Self::U8(rows) => check_rows_count_and_size(width, height, rows),
@@ -88,6 +93,7 @@ impl<'a> ImageRowsMut<'a> {
match self {
Self::U8x3(_) => PixelType::U8x3,
Self::U8x4(_) => PixelType::U8x4,
Self::U16x3(_) => PixelType::U16x3,
Self::I32(_) => PixelType::I32,
Self::F32(_) => PixelType::F32,
Self::U8(_) => PixelType::U8,
@@ -143,6 +149,10 @@ impl<'a> ImageView<'a> {
let pixels = align_buffer_to(buffer)?;
ImageRows::U8x4(pixels.chunks_exact(width.get() as usize).collect())
}
PixelType::U16x3 => {
let pixels = align_buffer_to(buffer)?;
ImageRows::U16x3(pixels.chunks_exact(width.get() as usize).collect())
}
PixelType::I32 => {
let pixels = align_buffer_to(buffer)?;
ImageRows::I32(pixels.chunks_exact(width.get() as usize).collect())
@@ -277,7 +287,7 @@ impl<'a> ImageView<'a> {
}
}
pub(crate) fn u32_image(&self) -> Option<TypedImageView<U8x4>> {
pub(crate) fn u8x4_image(&self) -> Option<TypedImageView<U8x4>> {
if let ImageRows::U8x4(ref rows) = self.rows {
Some(TypedImageView {
width: self.width,
@@ -290,6 +300,19 @@ impl<'a> ImageView<'a> {
}
}
pub(crate) fn u16x3_image(&self) -> Option<TypedImageView<U16x3>> {
if let ImageRows::U16x3(ref rows) = self.rows {
Some(TypedImageView {
width: self.width,
height: self.height,
crop_box: self.crop_box,
rows,
})
} else {
None
}
}
pub(crate) fn i32_image(&self) -> Option<TypedImageView<I32>> {
if let ImageRows::I32(ref rows) = self.rows {
Some(TypedImageView {
@@ -480,6 +503,10 @@ impl<'a> ImageViewMut<'a> {
let pixels = align_buffer_to_mut(buffer)?;
ImageRowsMut::U8x4(pixels.chunks_exact_mut(width.get() as usize).collect())
}
PixelType::U16x3 => {
let pixels = align_buffer_to_mut(buffer)?;
ImageRowsMut::U16x3(pixels.chunks_exact_mut(width.get() as usize).collect())
}
PixelType::I32 => {
let pixels = align_buffer_to_mut(buffer)?;
ImageRowsMut::I32(pixels.chunks_exact_mut(width.get() as usize).collect())
@@ -527,7 +554,7 @@ impl<'a> ImageViewMut<'a> {
}
}
pub(crate) fn u32_image<'s>(&'s mut self) -> Option<TypedImageViewMut<'s, 'a, U8x4>> {
pub(crate) fn u8x4_image<'s>(&'s mut self) -> Option<TypedImageViewMut<'s, 'a, U8x4>> {
if let ImageRowsMut::U8x4(rows) = &mut self.rows {
Some(TypedImageViewMut {
width: self.width,
@@ -539,6 +566,18 @@ impl<'a> ImageViewMut<'a> {
}
}
pub(crate) fn u16x3_image<'s>(&'s mut self) -> Option<TypedImageViewMut<'s, 'a, U16x3>> {
if let ImageRowsMut::U16x3(rows) = &mut self.rows {
Some(TypedImageViewMut {
width: self.width,
height: self.height,
rows,
})
} else {
None
}
}
pub(crate) fn i32_image<'s>(&'s mut self) -> Option<TypedImageViewMut<'s, 'a, I32>> {
if let ImageRowsMut::I32(rows) = &mut self.rows {
Some(TypedImageViewMut {
+9
View File
@@ -5,6 +5,7 @@ use std::mem::size_of;
pub enum PixelType {
U8x3,
U8x4,
U16x3,
I32,
F32,
U8,
@@ -14,6 +15,7 @@ impl PixelType {
pub(crate) fn size(&self) -> usize {
match self {
Self::U8x3 => 3,
Self::U16x3 => 6,
Self::U8 => 1,
_ => 4,
}
@@ -24,6 +26,7 @@ impl PixelType {
match self {
Self::U8x3 => unsafe { buffer.align_to::<U8x3>().0.is_empty() },
Self::U8x4 => unsafe { buffer.align_to::<U8x4>().0.is_empty() },
Self::U16x3 => unsafe { buffer.align_to::<U16x3>().0.is_empty() },
Self::I32 => unsafe { buffer.align_to::<I32>().0.is_empty() },
Self::F32 => unsafe { buffer.align_to::<F32>().0.is_empty() },
Self::U8 => true,
@@ -79,5 +82,11 @@ pixel_struct!(
PixelType::U8x4,
"Four bytes per pixel (RGBA, RGBx, CMYK and other)"
);
pixel_struct!(
U16x3,
[u16; 3],
PixelType::U16x3,
"Three `u16` components per pixel (e.g. RGB)"
);
pixel_struct!(I32, i32, PixelType::I32, "One `i32` component per pixel");
pixel_struct!(F32, f32, PixelType::F32, "One `f32` component per pixel");
+9 -2
View File
@@ -91,8 +91,15 @@ impl Resizer {
}
}
PixelType::U8x4 => {
if let Some(src_rows) = src_image.u32_image() {
if let Some(dst_rows) = dst_image.u32_image() {
if let Some(src_rows) = src_image.u8x4_image() {
if let Some(dst_rows) = dst_image.u8x4_image() {
self.resize_inner(src_rows, dst_rows);
}
}
}
PixelType::U16x3 => {
if let Some(src_rows) = src_image.u16x3_image() {
if let Some(dst_rows) = dst_image.u16x3_image() {
self.resize_inner(src_rows, dst_rows);
}
}
+50
View File
@@ -204,6 +204,56 @@ fn upscale_u8x3() {
}
}
#[test]
fn downscale_u16x3() {
type P = U16x3;
let buffer = downscale_test::<P>(ResizeAlg::Nearest, CpuExtensions::None);
assert_eq!(
utils::image_u16_checksum::<3>(&buffer),
[755050580, 756962660, 740848503]
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
// #[cfg(target_arch = "x86_64")]
// {
// cpu_extensions_vec.push(CpuExtensions::Sse4_1);
// cpu_extensions_vec.push(CpuExtensions::Avx2);
// }
for cpu_extensions in cpu_extensions_vec {
let buffer =
downscale_test::<P>(ResizeAlg::Convolution(FilterType::Lanczos3), cpu_extensions);
assert_eq!(
utils::image_u16_checksum::<3>(&buffer),
[756269847, 757632467, 741478612]
);
}
}
#[test]
fn upscale_u16x3() {
type P = U16x3;
let buffer = upscale_test::<P>(ResizeAlg::Nearest, CpuExtensions::None);
assert_eq!(
utils::image_u16_checksum::<3>(&buffer),
[297094122820, 297713401842, 291717497780]
);
let mut cpu_extensions_vec = vec![CpuExtensions::None];
// #[cfg(target_arch = "x86_64")]
// {
// cpu_extensions_vec.push(CpuExtensions::Sse4_1);
// cpu_extensions_vec.push(CpuExtensions::Avx2);
// }
for cpu_extensions in cpu_extensions_vec {
let buffer =
upscale_test::<P>(ResizeAlg::Convolution(FilterType::Lanczos3), cpu_extensions);
assert_eq!(
utils::image_u16_checksum::<3>(&buffer),
[297122154090, 297723994984, 291725294637]
);
}
}
#[test]
fn downscale_u8x4() {
type P = U8x4;
+30 -11
View File
@@ -1,7 +1,5 @@
use std::fs::File;
use std::num::NonZeroU32;
use image::codecs::png::PngEncoder;
use image::io::Reader as ImageReader;
use image::{ColorType, DynamicImage, GenericImageView};
@@ -16,12 +14,22 @@ pub fn image_checksum<const N: usize>(buffer: &[u8]) -> [u32; N] {
res
}
pub fn image_u16_checksum<const N: usize>(buffer: &[u8]) -> [u64; N] {
let buffer_u16 = unsafe { buffer.align_to::<u16>().1 };
let mut res = [0u64; N];
for pixel in buffer_u16.chunks_exact(N) {
res.iter_mut().zip(pixel).for_each(|(d, &s)| *d += s as u64);
}
res
}
pub trait PixelExt: Pixel {
fn pixel_type_str() -> &'static str {
match Self::pixel_type() {
PixelType::U8 => "u8",
PixelType::U8x3 => "u8x3",
PixelType::U8x4 => "u8x4",
PixelType::U16x3 => "u16x3",
PixelType::I32 => "i32",
PixelType::F32 => "f32",
}
@@ -90,6 +98,16 @@ impl PixelExt for U8x4 {
}
}
impl PixelExt for U16x3 {
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
img.to_rgb8()
.as_raw()
.iter()
.flat_map(|&c| [c, c])
.collect()
}
}
impl PixelExt for I32 {
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
img.to_luma16()
@@ -117,21 +135,22 @@ pub fn save_result(image: &Image, name: &str) {
return;
}
std::fs::create_dir_all("./data/result").unwrap();
let mut file = File::create(format!("./data/result/{}.png", name)).unwrap();
let path = format!("./data/result/{}.png", name);
let color_type = match image.pixel_type() {
PixelType::U8x3 => ColorType::Rgb8,
PixelType::U8x4 => ColorType::Rgba8,
PixelType::U16x3 => ColorType::Rgb16,
PixelType::U8 => ColorType::L8,
_ => panic!("Unsupported type of pixels"),
};
PngEncoder::new(&mut file)
.encode(
image.buffer(),
image.width().get(),
image.height().get(),
color_type,
)
.unwrap();
image::save_buffer(
&path,
image.buffer(),
image.width().get(),
image.height().get(),
color_type,
)
.unwrap();
}
pub fn cpu_ext_into_str(cpu_extensions: CpuExtensions) -> &'static str {