mirror of
https://github.com/Cykooz/fast_image_resize.git
synced 2026-10-08 01:11:09 +00:00
Added support of new type of pixels PixelType::U16.
This commit is contained in:
@@ -1,3 +1,7 @@
|
||||
## [Unreleased] - ReleaseDate
|
||||
|
||||
- Added support of new type of pixels `PixelType::U16`.
|
||||
|
||||
## [0.9.2] - 2022-05-19
|
||||
|
||||
- Added optimisation for convolution of `U8x2` images with helps of `SSE4.1`.
|
||||
|
||||
+6
-1
@@ -58,7 +58,7 @@ harness = false
|
||||
|
||||
|
||||
[[bench]]
|
||||
name = "bench_compare_u8"
|
||||
name = "bench_compare_l"
|
||||
harness = false
|
||||
|
||||
|
||||
@@ -67,6 +67,11 @@ name = "bench_compare_la"
|
||||
harness = false
|
||||
|
||||
|
||||
[[bench]]
|
||||
name = "bench_compare_l16"
|
||||
harness = false
|
||||
|
||||
|
||||
[profile.dev.package.'*']
|
||||
opt-level = 3
|
||||
|
||||
|
||||
@@ -28,6 +28,10 @@ Supported pixel formats and available optimisations:
|
||||
- native Rust-code without forced SIMD
|
||||
- SSE4.1
|
||||
- AVX2
|
||||
- `U16` - one `u16` components per pixel (e.g. L16):
|
||||
- native Rust-code without forced SIMD
|
||||
- SSE4.1
|
||||
- AVX2
|
||||
- `U16x3` - three `u16` components per pixel (e.g. RGB):
|
||||
- native Rust-code without forced SIMD
|
||||
- SSE4.1
|
||||
@@ -45,7 +49,7 @@ Environment:
|
||||
- RAM: DDR4 3800 MHz
|
||||
- Ubuntu 22.04 (linux 5.15.0)
|
||||
- Rust 1.61.0
|
||||
- fast_image_resize = "0.9.2"
|
||||
- fast_image_resize = "0.10.0"
|
||||
- glassbench = "0.3.1"
|
||||
- `rustflags = ["-C", "llvm-args=-x86-branches-within-32B-boundaries"]`
|
||||
|
||||
@@ -151,6 +155,24 @@ Pipeline:
|
||||
| fir sse4.1 | 0.31 | 22.14 | 35.69 | 50.40 |
|
||||
| fir avx2 | 0.31 | 19.12 | 28.25 | 33.86 |
|
||||
|
||||
### Resize grayscale image with 16 bits per pixel (U16) 4928x3279 => 852x567
|
||||
|
||||
Pipeline:
|
||||
|
||||
`src_image => resize => dst_image`
|
||||
|
||||
- Source image [nasa-4928x3279.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279.png)
|
||||
has converted into grayscale image with two bytes per pixel.
|
||||
- Numbers in table is mean duration of image resizing in milliseconds.
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|------------|:-------:|:--------:|:----------:|:--------:|
|
||||
| image | 15.18 | 44.51 | 70.23 | 96.64 |
|
||||
| resize | - | 15.41 | 30.13 | 52.52 |
|
||||
| fir rust | 0.16 | 16.96 | 25.51 | 34.87 |
|
||||
| fir sse4.1 | 0.16 | 7.36 | 11.96 | 17.40 |
|
||||
| fir avx2 | 0.16 | 15.33 | 10.47 | 16.89 |
|
||||
|
||||
## Examples
|
||||
|
||||
### Resize RGBA8 image
|
||||
|
||||
@@ -11,7 +11,7 @@ use fast_image_resize::{CpuExtensions, FilterType, PixelType, ResizeAlg, Resizer
|
||||
|
||||
mod utils;
|
||||
|
||||
pub fn bench_downscale_u8(bench: &mut Bench) {
|
||||
pub fn bench_downscale_l(bench: &mut Bench) {
|
||||
let src_image = utils::get_big_luma8_image();
|
||||
let new_width = NonZeroU32::new(852).unwrap();
|
||||
let new_height = NonZeroU32::new(567).unwrap();
|
||||
@@ -113,4 +113,4 @@ pub fn bench_downscale_u8(bench: &mut Bench) {
|
||||
utils::print_md_table(bench);
|
||||
}
|
||||
|
||||
bench_main!("Compare resize of U8 image", bench_downscale_u8,);
|
||||
bench_main!("Compare resize of U8 image", bench_downscale_l,);
|
||||
@@ -0,0 +1,119 @@
|
||||
use std::num::NonZeroU32;
|
||||
|
||||
use glassbench::*;
|
||||
use image::imageops;
|
||||
use resize::Pixel::Gray16;
|
||||
use rgb::alt::Gray;
|
||||
use rgb::FromSlice;
|
||||
|
||||
use fast_image_resize::Image;
|
||||
use fast_image_resize::{CpuExtensions, FilterType, PixelType, ResizeAlg, Resizer};
|
||||
|
||||
mod utils;
|
||||
|
||||
pub fn bench_downscale_l16(bench: &mut Bench) {
|
||||
let src_image = utils::get_big_luma16_image();
|
||||
let new_width = NonZeroU32::new(852).unwrap();
|
||||
let new_height = NonZeroU32::new(567).unwrap();
|
||||
|
||||
let alg_names = ["Nearest", "Bilinear", "CatmullRom", "Lanczos3"];
|
||||
|
||||
// image crate
|
||||
// https://crates.io/crates/image
|
||||
for alg_name in alg_names {
|
||||
let filter = match alg_name {
|
||||
"Nearest" => imageops::Nearest,
|
||||
"Bilinear" => imageops::Triangle,
|
||||
"CatmullRom" => imageops::CatmullRom,
|
||||
"Lanczos3" => imageops::Lanczos3,
|
||||
_ => continue,
|
||||
};
|
||||
bench.task(format!("image - {}", alg_name), |task| {
|
||||
task.iter(|| {
|
||||
imageops::resize(&src_image, new_width.get(), new_height.get(), filter);
|
||||
})
|
||||
});
|
||||
}
|
||||
|
||||
// resize crate
|
||||
// https://crates.io/crates/resize
|
||||
for alg_name in alg_names {
|
||||
let resize_src_image = src_image.as_raw().as_gray();
|
||||
let mut dst = vec![Gray(0u16); (new_width.get() * new_height.get()) as usize];
|
||||
bench.task(format!("resize - {}", alg_name), |task| {
|
||||
let filter = match alg_name {
|
||||
"Nearest" => {
|
||||
// resizer doesn't support "nearest" algorithm
|
||||
task.iter(|| {});
|
||||
return;
|
||||
}
|
||||
"Bilinear" => resize::Type::Triangle,
|
||||
"CatmullRom" => resize::Type::Catrom,
|
||||
"Lanczos3" => resize::Type::Lanczos3,
|
||||
_ => return,
|
||||
};
|
||||
let mut resize = resize::new(
|
||||
src_image.width() as usize,
|
||||
src_image.height() as usize,
|
||||
new_width.get() as usize,
|
||||
new_height.get() as usize,
|
||||
Gray16,
|
||||
filter,
|
||||
)
|
||||
.unwrap();
|
||||
task.iter(|| {
|
||||
resize.resize(resize_src_image, &mut dst).unwrap();
|
||||
})
|
||||
});
|
||||
}
|
||||
|
||||
// fast_image_resize crate;
|
||||
let mut cpu_ext_and_name = vec![(CpuExtensions::None, "rust")];
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
{
|
||||
cpu_ext_and_name.push((CpuExtensions::Sse4_1, "sse4.1"));
|
||||
cpu_ext_and_name.push((CpuExtensions::Avx2, "avx2"));
|
||||
}
|
||||
for (cpu_ext, ext_name) in cpu_ext_and_name {
|
||||
for alg_name in alg_names {
|
||||
let src_image_data = Image::from_vec_u8(
|
||||
NonZeroU32::new(src_image.width()).unwrap(),
|
||||
NonZeroU32::new(src_image.height()).unwrap(),
|
||||
src_image
|
||||
.as_raw()
|
||||
.iter()
|
||||
.flat_map(|&c| c.to_le_bytes())
|
||||
.collect(),
|
||||
PixelType::U16,
|
||||
)
|
||||
.unwrap();
|
||||
let src_view = src_image_data.view();
|
||||
let mut dst_image = Image::new(new_width, new_height, src_image_data.pixel_type());
|
||||
let mut dst_view = dst_image.view_mut();
|
||||
|
||||
let resize_alg = match alg_name {
|
||||
"Nearest" => ResizeAlg::Nearest,
|
||||
"Bilinear" => ResizeAlg::Convolution(FilterType::Bilinear),
|
||||
"CatmullRom" => ResizeAlg::Convolution(FilterType::CatmullRom),
|
||||
"Lanczos3" => ResizeAlg::Convolution(FilterType::Lanczos3),
|
||||
_ => return,
|
||||
};
|
||||
let mut fast_resizer = Resizer::new(resize_alg);
|
||||
|
||||
unsafe {
|
||||
fast_resizer.reset_internal_buffers();
|
||||
fast_resizer.set_cpu_extensions(cpu_ext);
|
||||
}
|
||||
|
||||
bench.task(format!("fir {} - {}", ext_name, alg_name), |task| {
|
||||
task.iter(|| {
|
||||
fast_resizer.resize(&src_view, &mut dst_view).unwrap();
|
||||
})
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
utils::print_md_table(bench);
|
||||
}
|
||||
|
||||
bench_main!("Compare resize of U16 image", bench_downscale_l16,);
|
||||
@@ -12,6 +12,7 @@ mod f32x1;
|
||||
mod filters;
|
||||
mod i32x1;
|
||||
mod optimisations;
|
||||
mod u16x1;
|
||||
mod u16x3;
|
||||
mod u8x1;
|
||||
mod u8x2;
|
||||
|
||||
@@ -0,0 +1,369 @@
|
||||
use std::arch::x86_64::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::image_view::{FourRows, FourRowsMut, TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::U16;
|
||||
use crate::simd_utils;
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn horiz_convolution(
|
||||
src_image: TypedImageView<U16>,
|
||||
mut dst_image: TypedImageViewMut<U16>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard32::new(values);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(
|
||||
src_rows,
|
||||
dst_rows,
|
||||
&coefficients_chunks,
|
||||
&normalizer_guard,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
let mut yy = dst_height - dst_height % 4;
|
||||
while yy < dst_height {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
&coefficients_chunks,
|
||||
&normalizer_guard,
|
||||
);
|
||||
}
|
||||
yy += 1;
|
||||
}
|
||||
}
|
||||
|
||||
/// For safety, it is necessary to ensure the following conditions:
|
||||
/// - length of all rows in src_rows must be equal
|
||||
/// - length of all rows in dst_rows must be equal
|
||||
/// - coefficients_chunks.len() == dst_rows.0.len()
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
|
||||
#[target_feature(enable = "avx2")]
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: FourRows<U16>,
|
||||
dst_rows: FourRowsMut<U16>,
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer_guard: &optimisations::NormalizerGuard32,
|
||||
) {
|
||||
let (s_row0, s_row1, s_row2, s_row3) = src_rows;
|
||||
let s_rows = [s_row0, s_row1, s_row2, s_row3];
|
||||
let (d_row0, d_row1, d_row2, d_row3) = dst_rows;
|
||||
let d_rows = [d_row0, d_row1, d_row2, d_row3];
|
||||
let precision = normalizer_guard.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 4];
|
||||
|
||||
/*
|
||||
|L0 | |L1 | |L2 | |L3 | |L4 | |L5 | |L6 | |L7 |
|
||||
|0001| |0203| |0405| |0607| |0809| |1011| |1213| |1415|
|
||||
|
||||
Shuffle to extract L0 and L1 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 3, 2, -1, -1, -1, -1, -1, -1, 1, 0
|
||||
|
||||
Shuffle to extract L2 and L3 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 7, 6, -1, -1, -1, -1, -1, -1, 5, 4
|
||||
|
||||
Shuffle to extract L4 and L5 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 11, 10, -1, -1, -1, -1, -1, -1, 9, 8
|
||||
|
||||
Shuffle to extract L6 and L7 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12
|
||||
*/
|
||||
|
||||
#[rustfmt::skip]
|
||||
let l0l1_shuffle = _mm256_set_epi8(
|
||||
-1, -1, -1, -1, -1, -1, 3, 2, -1, -1, -1, -1, -1, -1, 1, 0,
|
||||
-1, -1, -1, -1, -1, -1, 3, 2, -1, -1, -1, -1, -1, -1, 1, 0,
|
||||
);
|
||||
#[rustfmt::skip]
|
||||
let l2l3_shuffle = _mm256_set_epi8(
|
||||
-1, -1, -1, -1, -1, -1, 7, 6, -1, -1, -1, -1, -1, -1, 5, 4,
|
||||
-1, -1, -1, -1, -1, -1, 7, 6, -1, -1, -1, -1, -1, -1, 5, 4,
|
||||
);
|
||||
#[rustfmt::skip]
|
||||
let l4l5_shuffle = _mm256_set_epi8(
|
||||
-1, -1, -1, -1, -1, -1, 11, 10, -1, -1, -1, -1, -1, -1, 9, 8,
|
||||
-1, -1, -1, -1, -1, -1, 11, 10, -1, -1, -1, -1, -1, -1, 9, 8,
|
||||
);
|
||||
#[rustfmt::skip]
|
||||
let l6l7_shuffle = _mm256_set_epi8(
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12,
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut ll_sum = [_mm256_set1_epi64x(0); 4];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
|
||||
let coeffs_by_16 = coeffs.chunks_exact(16);
|
||||
coeffs = coeffs_by_16.remainder();
|
||||
|
||||
for k in coeffs_by_16 {
|
||||
let coeff0189_i64x4 =
|
||||
_mm256_set_epi64x(k[9] as i64, k[8] as i64, k[1] as i64, k[0] as i64);
|
||||
let coeff23ab_i64x2 =
|
||||
_mm256_set_epi64x(k[11] as i64, k[10] as i64, k[3] as i64, k[2] as i64);
|
||||
let coeff45cd_i64x2 =
|
||||
_mm256_set_epi64x(k[13] as i64, k[12] as i64, k[5] as i64, k[4] as i64);
|
||||
let coeff67ef_i64x2 =
|
||||
_mm256_set_epi64x(k[15] as i64, k[14] as i64, k[7] as i64, k[6] as i64);
|
||||
|
||||
for i in 0..4 {
|
||||
let mut sum = ll_sum[i];
|
||||
let source = simd_utils::loadu_si256(s_rows[i], x);
|
||||
|
||||
let l0l1_i64x4 = _mm256_shuffle_epi8(source, l0l1_shuffle);
|
||||
sum = _mm256_add_epi64(sum, _mm256_mul_epi32(l0l1_i64x4, coeff0189_i64x4));
|
||||
|
||||
let l2l3_i64x4 = _mm256_shuffle_epi8(source, l2l3_shuffle);
|
||||
sum = _mm256_add_epi64(sum, _mm256_mul_epi32(l2l3_i64x4, coeff23ab_i64x2));
|
||||
|
||||
let l4l5_i64x4 = _mm256_shuffle_epi8(source, l4l5_shuffle);
|
||||
sum = _mm256_add_epi64(sum, _mm256_mul_epi32(l4l5_i64x4, coeff45cd_i64x2));
|
||||
|
||||
let l6l7_i64x4 = _mm256_shuffle_epi8(source, l6l7_shuffle);
|
||||
sum = _mm256_add_epi64(sum, _mm256_mul_epi32(l6l7_i64x4, coeff67ef_i64x2));
|
||||
|
||||
ll_sum[i] = sum;
|
||||
}
|
||||
x += 16;
|
||||
}
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
coeffs = coeffs_by_8.remainder();
|
||||
|
||||
for k in coeffs_by_8 {
|
||||
let coeff0145_i64x4 =
|
||||
_mm256_set_epi64x(k[5] as i64, k[4] as i64, k[1] as i64, k[0] as i64);
|
||||
let coeff2367_i64x2 =
|
||||
_mm256_set_epi64x(k[7] as i64, k[6] as i64, k[3] as i64, k[2] as i64);
|
||||
|
||||
for i in 0..4 {
|
||||
let mut sum = ll_sum[i];
|
||||
let source = _mm256_set_m128i(
|
||||
simd_utils::loadl_epi64(s_rows[i], x + 4),
|
||||
simd_utils::loadl_epi64(s_rows[i], x),
|
||||
);
|
||||
|
||||
let l0l1_i64x4 = _mm256_shuffle_epi8(source, l0l1_shuffle);
|
||||
sum = _mm256_add_epi64(sum, _mm256_mul_epi32(l0l1_i64x4, coeff0145_i64x4));
|
||||
|
||||
let l2l3_i64x4 = _mm256_shuffle_epi8(source, l2l3_shuffle);
|
||||
sum = _mm256_add_epi64(sum, _mm256_mul_epi32(l2l3_i64x4, coeff2367_i64x2));
|
||||
|
||||
ll_sum[i] = sum;
|
||||
}
|
||||
x += 8;
|
||||
}
|
||||
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
coeffs = coeffs_by_4.remainder();
|
||||
|
||||
for k in coeffs_by_4 {
|
||||
let coeff0123_i64x4 =
|
||||
_mm256_set_epi64x(k[3] as i64, k[2] as i64, k[1] as i64, k[0] as i64);
|
||||
|
||||
for i in 0..4 {
|
||||
let source = _mm256_set_m128i(
|
||||
simd_utils::loadl_epi32(s_rows[i], x + 2),
|
||||
simd_utils::loadl_epi32(s_rows[i], x),
|
||||
);
|
||||
|
||||
let l0l1_i64x4 = _mm256_shuffle_epi8(source, l0l1_shuffle);
|
||||
ll_sum[i] =
|
||||
_mm256_add_epi64(ll_sum[i], _mm256_mul_epi32(l0l1_i64x4, coeff0123_i64x4));
|
||||
}
|
||||
x += 4;
|
||||
}
|
||||
|
||||
if !coeffs.is_empty() {
|
||||
let mut coeffs_x3: [i64; 4] = [0; 4];
|
||||
for (d_coeff, &s_coeff) in coeffs_x3.iter_mut().zip(coeffs) {
|
||||
*d_coeff = s_coeff as i64;
|
||||
}
|
||||
let coeff0123_i64x4 = simd_utils::loadu_si256(&coeffs_x3, 0);
|
||||
|
||||
for i in 0..4 {
|
||||
let mut pixels: [i64; 4] = [0; 4];
|
||||
let src_row = s_rows[i];
|
||||
for (i, pixel) in pixels.iter_mut().take(coeffs.len()).enumerate() {
|
||||
*pixel = (*src_row.get_unchecked(x + i)).0 as i64;
|
||||
}
|
||||
let source = simd_utils::loadu_si256(&pixels, 0);
|
||||
ll_sum[i] = _mm256_add_epi64(ll_sum[i], _mm256_mul_epi32(source, coeff0123_i64x4));
|
||||
}
|
||||
}
|
||||
|
||||
for i in 0..4 {
|
||||
_mm256_storeu_si256((&mut ll_buf).as_mut_ptr() as *mut __m256i, ll_sum[i]);
|
||||
let dst_pixel = d_rows[i].get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = normalizer_guard.clip(ll_buf.iter().sum::<i64>() + half_error);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// For safety, it is necessary to ensure the following conditions:
|
||||
/// - bounds.len() == dst_row.len()
|
||||
/// - coefficients_chunks.len() == dst_row.len()
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
|
||||
#[target_feature(enable = "avx2")]
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16],
|
||||
dst_row: &mut [U16],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer_guard: &optimisations::NormalizerGuard32,
|
||||
) {
|
||||
let precision = normalizer_guard.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 4];
|
||||
|
||||
/*
|
||||
|L0 | |L1 | |L2 | |L3 | |L4 | |L5 | |L6 | |L7 |
|
||||
|0001| |0203| |0405| |0607| |0809| |1011| |1213| |1415|
|
||||
|
||||
Shuffle to extract L0 and L1 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 3, 2, -1, -1, -1, -1, -1, -1, 1, 0
|
||||
|
||||
Shuffle to extract L2 and L3 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 7, 6, -1, -1, -1, -1, -1, -1, 5, 4
|
||||
|
||||
Shuffle to extract L4 and L5 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 11, 10, -1, -1, -1, -1, -1, -1, 9, 8
|
||||
|
||||
Shuffle to extract L6 and L7 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12
|
||||
*/
|
||||
|
||||
#[rustfmt::skip]
|
||||
let l0l1_shuffle = _mm256_set_epi8(
|
||||
-1, -1, -1, -1, -1, -1, 3, 2, -1, -1, -1, -1, -1, -1, 1, 0,
|
||||
-1, -1, -1, -1, -1, -1, 3, 2, -1, -1, -1, -1, -1, -1, 1, 0,
|
||||
);
|
||||
#[rustfmt::skip]
|
||||
let l2l3_shuffle = _mm256_set_epi8(
|
||||
-1, -1, -1, -1, -1, -1, 7, 6, -1, -1, -1, -1, -1, -1, 5, 4,
|
||||
-1, -1, -1, -1, -1, -1, 7, 6, -1, -1, -1, -1, -1, -1, 5, 4,
|
||||
);
|
||||
#[rustfmt::skip]
|
||||
let l4l5_shuffle = _mm256_set_epi8(
|
||||
-1, -1, -1, -1, -1, -1, 11, 10, -1, -1, -1, -1, -1, -1, 9, 8,
|
||||
-1, -1, -1, -1, -1, -1, 11, 10, -1, -1, -1, -1, -1, -1, 9, 8,
|
||||
);
|
||||
#[rustfmt::skip]
|
||||
let l6l7_shuffle = _mm256_set_epi8(
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12,
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut ll_sum = _mm256_set1_epi64x(0);
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
|
||||
let coeffs_by_16 = coeffs.chunks_exact(16);
|
||||
coeffs = coeffs_by_16.remainder();
|
||||
|
||||
for k in coeffs_by_16 {
|
||||
let coeff0189_i64x4 =
|
||||
_mm256_set_epi64x(k[9] as i64, k[8] as i64, k[1] as i64, k[0] as i64);
|
||||
let coeff23ab_i64x2 =
|
||||
_mm256_set_epi64x(k[11] as i64, k[10] as i64, k[3] as i64, k[2] as i64);
|
||||
let coeff45cd_i64x2 =
|
||||
_mm256_set_epi64x(k[13] as i64, k[12] as i64, k[5] as i64, k[4] as i64);
|
||||
let coeff67ef_i64x2 =
|
||||
_mm256_set_epi64x(k[15] as i64, k[14] as i64, k[7] as i64, k[6] as i64);
|
||||
|
||||
let source = simd_utils::loadu_si256(src_row, x);
|
||||
|
||||
let l0l1_i64x4 = _mm256_shuffle_epi8(source, l0l1_shuffle);
|
||||
ll_sum = _mm256_add_epi64(ll_sum, _mm256_mul_epi32(l0l1_i64x4, coeff0189_i64x4));
|
||||
|
||||
let l2l3_i64x4 = _mm256_shuffle_epi8(source, l2l3_shuffle);
|
||||
ll_sum = _mm256_add_epi64(ll_sum, _mm256_mul_epi32(l2l3_i64x4, coeff23ab_i64x2));
|
||||
|
||||
let l4l5_i64x4 = _mm256_shuffle_epi8(source, l4l5_shuffle);
|
||||
ll_sum = _mm256_add_epi64(ll_sum, _mm256_mul_epi32(l4l5_i64x4, coeff45cd_i64x2));
|
||||
|
||||
let l6l7_i64x4 = _mm256_shuffle_epi8(source, l6l7_shuffle);
|
||||
ll_sum = _mm256_add_epi64(ll_sum, _mm256_mul_epi32(l6l7_i64x4, coeff67ef_i64x2));
|
||||
|
||||
x += 16;
|
||||
}
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
coeffs = coeffs_by_8.remainder();
|
||||
|
||||
for k in coeffs_by_8 {
|
||||
let coeff0145_i64x4 =
|
||||
_mm256_set_epi64x(k[5] as i64, k[4] as i64, k[1] as i64, k[0] as i64);
|
||||
let coeff2367_i64x2 =
|
||||
_mm256_set_epi64x(k[7] as i64, k[6] as i64, k[3] as i64, k[2] as i64);
|
||||
|
||||
let source = _mm256_set_m128i(
|
||||
simd_utils::loadl_epi64(src_row, x + 4),
|
||||
simd_utils::loadl_epi64(src_row, x),
|
||||
);
|
||||
|
||||
let l0l1_i64x4 = _mm256_shuffle_epi8(source, l0l1_shuffle);
|
||||
ll_sum = _mm256_add_epi64(ll_sum, _mm256_mul_epi32(l0l1_i64x4, coeff0145_i64x4));
|
||||
|
||||
let l2l3_i64x4 = _mm256_shuffle_epi8(source, l2l3_shuffle);
|
||||
ll_sum = _mm256_add_epi64(ll_sum, _mm256_mul_epi32(l2l3_i64x4, coeff2367_i64x2));
|
||||
|
||||
x += 8;
|
||||
}
|
||||
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
coeffs = coeffs_by_4.remainder();
|
||||
|
||||
for k in coeffs_by_4 {
|
||||
let coeff0123_i64x4 =
|
||||
_mm256_set_epi64x(k[3] as i64, k[2] as i64, k[1] as i64, k[0] as i64);
|
||||
|
||||
let source = _mm256_set_m128i(
|
||||
simd_utils::loadl_epi32(src_row, x + 2),
|
||||
simd_utils::loadl_epi32(src_row, x),
|
||||
);
|
||||
|
||||
let l0l1_i64x4 = _mm256_shuffle_epi8(source, l0l1_shuffle);
|
||||
ll_sum = _mm256_add_epi64(ll_sum, _mm256_mul_epi32(l0l1_i64x4, coeff0123_i64x4));
|
||||
|
||||
x += 4;
|
||||
}
|
||||
|
||||
if !coeffs.is_empty() {
|
||||
let mut coeffs_x4: [i64; 4] = [0; 4];
|
||||
for (d_coeff, &s_coeff) in coeffs_x4.iter_mut().zip(coeffs) {
|
||||
*d_coeff = s_coeff as i64;
|
||||
}
|
||||
let coeff0123_i64x4 = simd_utils::loadu_si256(&coeffs_x4, 0);
|
||||
|
||||
let mut pixels: [i64; 4] = [0; 4];
|
||||
for pixel in pixels.iter_mut().take(coeffs.len()) {
|
||||
*pixel = (*src_row.get_unchecked(x)).0 as i64;
|
||||
x += 1;
|
||||
}
|
||||
let source = simd_utils::loadu_si256(&pixels, 0);
|
||||
ll_sum = _mm256_add_epi64(ll_sum, _mm256_mul_epi32(source, coeff0123_i64x4));
|
||||
}
|
||||
|
||||
_mm256_storeu_si256((&mut ll_buf).as_mut_ptr() as *mut __m256i, ll_sum);
|
||||
let dst_pixel = dst_row.get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = normalizer_guard.clip(ll_buf.iter().sum::<i64>() + half_error);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,38 @@
|
||||
use super::{Coefficients, Convolution};
|
||||
use crate::convolution::vertical_u16::vert_convolution_u16;
|
||||
use crate::image_view::{TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::U16;
|
||||
use crate::CpuExtensions;
|
||||
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod avx2;
|
||||
mod native;
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod sse4;
|
||||
|
||||
impl Convolution for U16 {
|
||||
fn horiz_convolution(
|
||||
src_image: TypedImageView<Self>,
|
||||
dst_image: TypedImageViewMut<Self>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
match cpu_extensions {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::horiz_convolution(src_image, dst_image, offset, coeffs),
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_image, dst_image, offset, coeffs),
|
||||
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
|
||||
}
|
||||
}
|
||||
|
||||
fn vert_convolution(
|
||||
src_image: TypedImageView<Self>,
|
||||
dst_image: TypedImageViewMut<Self>,
|
||||
coeffs: Coefficients,
|
||||
cpu_extensions: CpuExtensions,
|
||||
) {
|
||||
vert_convolution_u16(src_image, dst_image, coeffs, cpu_extensions);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,32 @@
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::image_view::{TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::U16;
|
||||
|
||||
#[inline(always)]
|
||||
pub(crate) fn horiz_convolution(
|
||||
src_image: TypedImageView<U16>,
|
||||
mut dst_image: TypedImageViewMut<U16>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard32::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
|
||||
let initial: i64 = 1 << (precision - 1);
|
||||
|
||||
let src_rows = src_image.iter_rows(offset);
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for (dst_row, src_row) in dst_rows.zip(src_rows) {
|
||||
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
|
||||
let first_x_src = coeffs_chunk.start as usize;
|
||||
let mut sum = initial;
|
||||
let src_pixels = unsafe { src_row.get_unchecked(first_x_src..) };
|
||||
for (&k, src_pixel) in coeffs_chunk.values.iter().zip(src_pixels) {
|
||||
sum += src_pixel.0 as i64 * (k as i64);
|
||||
}
|
||||
dst_pixel.0 = normalizer_guard.clip(sum);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,293 @@
|
||||
use std::arch::x86_64::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::image_view::{FourRows, FourRowsMut, TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::U16;
|
||||
use crate::simd_utils;
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn horiz_convolution(
|
||||
src_image: TypedImageView<U16>,
|
||||
mut dst_image: TypedImageViewMut<U16>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard32::new(values);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(
|
||||
src_rows,
|
||||
dst_rows,
|
||||
&coefficients_chunks,
|
||||
&normalizer_guard,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
let mut yy = dst_height - dst_height % 4;
|
||||
while yy < dst_height {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
&coefficients_chunks,
|
||||
&normalizer_guard,
|
||||
);
|
||||
}
|
||||
yy += 1;
|
||||
}
|
||||
}
|
||||
|
||||
/// For safety, it is necessary to ensure the following conditions:
|
||||
/// - length of all rows in src_rows must be equal
|
||||
/// - length of all rows in dst_rows must be equal
|
||||
/// - coefficients_chunks.len() == dst_rows.0.len()
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: FourRows<U16>,
|
||||
dst_rows: FourRowsMut<U16>,
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer_guard: &optimisations::NormalizerGuard32,
|
||||
) {
|
||||
let (s_row0, s_row1, s_row2, s_row3) = src_rows;
|
||||
let s_rows = [s_row0, s_row1, s_row2, s_row3];
|
||||
let (d_row0, d_row1, d_row2, d_row3) = dst_rows;
|
||||
let d_rows = [d_row0, d_row1, d_row2, d_row3];
|
||||
let precision = normalizer_guard.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 2];
|
||||
|
||||
/*
|
||||
|L0 | |L1 | |L2 | |L3 | |L4 | |L5 | |L6 | |L7 |
|
||||
|0001| |0203| |0405| |0607| |0809| |1011| |1213| |1415|
|
||||
|
||||
Shuffle to extract L0 and L1 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 3, 2, -1, -1, -1, -1, -1, -1, 1, 0
|
||||
|
||||
Shuffle to extract L2 and L3 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 7, 6, -1, -1, -1, -1, -1, -1, 5, 4
|
||||
|
||||
Shuffle to extract L4 and L5 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 11, 10, -1, -1, -1, -1, -1, -1, 9, 8
|
||||
|
||||
Shuffle to extract L6 and L7 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12
|
||||
*/
|
||||
|
||||
let l0l1_shuffle = _mm_set_epi8(-1, -1, -1, -1, -1, -1, 3, 2, -1, -1, -1, -1, -1, -1, 1, 0);
|
||||
let l2l3_shuffle = _mm_set_epi8(-1, -1, -1, -1, -1, -1, 7, 6, -1, -1, -1, -1, -1, -1, 5, 4);
|
||||
let l4l5_shuffle = _mm_set_epi8(-1, -1, -1, -1, -1, -1, 11, 10, -1, -1, -1, -1, -1, -1, 9, 8);
|
||||
let l6l7_shuffle = _mm_set_epi8(
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut ll_sum = [_mm_set1_epi64x(0); 4];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
coeffs = coeffs_by_8.remainder();
|
||||
|
||||
for k in coeffs_by_8 {
|
||||
let coeff01_i64x2 = _mm_set_epi64x(k[1] as i64, k[0] as i64);
|
||||
let coeff23_i64x2 = _mm_set_epi64x(k[3] as i64, k[2] as i64);
|
||||
let coeff45_i64x2 = _mm_set_epi64x(k[5] as i64, k[4] as i64);
|
||||
let coeff67_i64x2 = _mm_set_epi64x(k[7] as i64, k[6] as i64);
|
||||
|
||||
for i in 0..4 {
|
||||
let mut sum = ll_sum[i];
|
||||
let source = simd_utils::loadu_si128(s_rows[i], x);
|
||||
|
||||
let l0l1_i64x4 = _mm_shuffle_epi8(source, l0l1_shuffle);
|
||||
sum = _mm_add_epi64(sum, _mm_mul_epi32(l0l1_i64x4, coeff01_i64x2));
|
||||
|
||||
let l2l3_i64x4 = _mm_shuffle_epi8(source, l2l3_shuffle);
|
||||
sum = _mm_add_epi64(sum, _mm_mul_epi32(l2l3_i64x4, coeff23_i64x2));
|
||||
|
||||
let l4l5_i64x4 = _mm_shuffle_epi8(source, l4l5_shuffle);
|
||||
sum = _mm_add_epi64(sum, _mm_mul_epi32(l4l5_i64x4, coeff45_i64x2));
|
||||
|
||||
let l6l7_i64x4 = _mm_shuffle_epi8(source, l6l7_shuffle);
|
||||
sum = _mm_add_epi64(sum, _mm_mul_epi32(l6l7_i64x4, coeff67_i64x2));
|
||||
|
||||
ll_sum[i] = sum;
|
||||
}
|
||||
x += 8;
|
||||
}
|
||||
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
coeffs = coeffs_by_4.remainder();
|
||||
|
||||
for k in coeffs_by_4 {
|
||||
let coeff01_i64x2 = _mm_set_epi64x(k[1] as i64, k[0] as i64);
|
||||
let coeff23_i64x2 = _mm_set_epi64x(k[3] as i64, k[2] as i64);
|
||||
|
||||
for i in 0..4 {
|
||||
let mut sum = ll_sum[i];
|
||||
let source = simd_utils::loadl_epi64(s_rows[i], x);
|
||||
|
||||
let l0l1_i64x4 = _mm_shuffle_epi8(source, l0l1_shuffle);
|
||||
sum = _mm_add_epi64(sum, _mm_mul_epi32(l0l1_i64x4, coeff01_i64x2));
|
||||
|
||||
let l2l3_i64x4 = _mm_shuffle_epi8(source, l2l3_shuffle);
|
||||
sum = _mm_add_epi64(sum, _mm_mul_epi32(l2l3_i64x4, coeff23_i64x2));
|
||||
|
||||
ll_sum[i] = sum;
|
||||
}
|
||||
x += 4;
|
||||
}
|
||||
|
||||
let coeffs_by_2 = coeffs.chunks_exact(2);
|
||||
coeffs = coeffs_by_2.remainder();
|
||||
|
||||
for k in coeffs_by_2 {
|
||||
let coeff01_i64x2 = _mm_set_epi64x(k[1] as i64, k[0] as i64);
|
||||
for i in 0..4 {
|
||||
let source = simd_utils::loadl_epi32(s_rows[i], x);
|
||||
let l_i64x2 = _mm_shuffle_epi8(source, l0l1_shuffle);
|
||||
ll_sum[i] = _mm_add_epi64(ll_sum[i], _mm_mul_epi32(l_i64x2, coeff01_i64x2));
|
||||
}
|
||||
x += 2;
|
||||
}
|
||||
|
||||
if let Some(&k) = coeffs.get(0) {
|
||||
let coeff01_i64x2 = _mm_set_epi64x(0, k as i64);
|
||||
for i in 0..4 {
|
||||
let pixel = (*s_rows[i].get_unchecked(x)).0 as i64;
|
||||
let source = _mm_set_epi64x(0, pixel);
|
||||
ll_sum[i] = _mm_add_epi64(ll_sum[i], _mm_mul_epi32(source, coeff01_i64x2));
|
||||
}
|
||||
}
|
||||
|
||||
for i in 0..4 {
|
||||
_mm_storeu_si128((&mut ll_buf).as_mut_ptr() as *mut __m128i, ll_sum[i]);
|
||||
let dst_pixel = d_rows[i].get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = normalizer_guard.clip(ll_buf.iter().sum::<i64>() + half_error);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// For safety, it is necessary to ensure the following conditions:
|
||||
/// - bounds.len() == dst_row.len()
|
||||
/// - coefficients_chunks.len() == dst_row.len()
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16],
|
||||
dst_row: &mut [U16],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer_guard: &optimisations::NormalizerGuard32,
|
||||
) {
|
||||
let precision = normalizer_guard.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 2];
|
||||
|
||||
/*
|
||||
|L0 | |L1 | |L2 | |L3 | |L4 | |L5 | |L6 | |L7 |
|
||||
|0001| |0203| |0405| |0607| |0809| |1011| |1213| |1415|
|
||||
|
||||
Shuffle to extract L0 and L1 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 3, 2, -1, -1, -1, -1, -1, -1, 1, 0
|
||||
|
||||
Shuffle to extract L2 and L3 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 7, 6, -1, -1, -1, -1, -1, -1, 5, 4
|
||||
|
||||
Shuffle to extract L4 and L5 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 11, 10, -1, -1, -1, -1, -1, -1, 9, 8
|
||||
|
||||
Shuffle to extract L6 and L7 as i64:
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12
|
||||
*/
|
||||
|
||||
let l01_shuffle = _mm_set_epi8(-1, -1, -1, -1, -1, -1, 3, 2, -1, -1, -1, -1, -1, -1, 1, 0);
|
||||
let l23_shuffle = _mm_set_epi8(-1, -1, -1, -1, -1, -1, 7, 6, -1, -1, -1, -1, -1, -1, 5, 4);
|
||||
let l45_shuffle = _mm_set_epi8(-1, -1, -1, -1, -1, -1, 11, 10, -1, -1, -1, -1, -1, -1, 9, 8);
|
||||
let l67_shuffle = _mm_set_epi8(
|
||||
-1, -1, -1, -1, -1, -1, 15, 14, -1, -1, -1, -1, -1, -1, 13, 12,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut ll_sum = _mm_set1_epi64x(0);
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
coeffs = coeffs_by_8.remainder();
|
||||
|
||||
for k in coeffs_by_8 {
|
||||
let coeff01_i64x2 = _mm_set_epi64x(k[1] as i64, k[0] as i64);
|
||||
let coeff23_i64x2 = _mm_set_epi64x(k[3] as i64, k[2] as i64);
|
||||
let coeff45_i64x2 = _mm_set_epi64x(k[5] as i64, k[4] as i64);
|
||||
let coeff67_i64x2 = _mm_set_epi64x(k[7] as i64, k[6] as i64);
|
||||
|
||||
let source = simd_utils::loadu_si128(src_row, x);
|
||||
|
||||
let l_i64x2 = _mm_shuffle_epi8(source, l01_shuffle);
|
||||
ll_sum = _mm_add_epi64(ll_sum, _mm_mul_epi32(l_i64x2, coeff01_i64x2));
|
||||
|
||||
let l_i64x2 = _mm_shuffle_epi8(source, l23_shuffle);
|
||||
ll_sum = _mm_add_epi64(ll_sum, _mm_mul_epi32(l_i64x2, coeff23_i64x2));
|
||||
|
||||
let l_i64x2 = _mm_shuffle_epi8(source, l45_shuffle);
|
||||
ll_sum = _mm_add_epi64(ll_sum, _mm_mul_epi32(l_i64x2, coeff45_i64x2));
|
||||
|
||||
let l_i64x2 = _mm_shuffle_epi8(source, l67_shuffle);
|
||||
ll_sum = _mm_add_epi64(ll_sum, _mm_mul_epi32(l_i64x2, coeff67_i64x2));
|
||||
|
||||
x += 8;
|
||||
}
|
||||
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
coeffs = coeffs_by_4.remainder();
|
||||
|
||||
for k in coeffs_by_4 {
|
||||
let coeff01_i64x2 = _mm_set_epi64x(k[1] as i64, k[0] as i64);
|
||||
let coeff23_i64x2 = _mm_set_epi64x(k[3] as i64, k[2] as i64);
|
||||
|
||||
let source = simd_utils::loadl_epi64(src_row, x);
|
||||
|
||||
let l_i64x2 = _mm_shuffle_epi8(source, l01_shuffle);
|
||||
ll_sum = _mm_add_epi64(ll_sum, _mm_mul_epi32(l_i64x2, coeff01_i64x2));
|
||||
|
||||
let l_i64x2 = _mm_shuffle_epi8(source, l23_shuffle);
|
||||
ll_sum = _mm_add_epi64(ll_sum, _mm_mul_epi32(l_i64x2, coeff23_i64x2));
|
||||
|
||||
x += 4;
|
||||
}
|
||||
|
||||
let coeffs_by_2 = coeffs.chunks_exact(2);
|
||||
coeffs = coeffs_by_2.remainder();
|
||||
|
||||
for k in coeffs_by_2 {
|
||||
let coeff01_i64x2 = _mm_set_epi64x(k[1] as i64, k[0] as i64);
|
||||
let source = simd_utils::loadl_epi32(src_row, x);
|
||||
|
||||
let l_i64x2 = _mm_shuffle_epi8(source, l01_shuffle);
|
||||
ll_sum = _mm_add_epi64(ll_sum, _mm_mul_epi32(l_i64x2, coeff01_i64x2));
|
||||
|
||||
x += 2;
|
||||
}
|
||||
|
||||
if let Some(&k) = coeffs.get(0) {
|
||||
let coeff01_i64x2 = _mm_set_epi64x(0, k as i64);
|
||||
let pixel = (*src_row.get_unchecked(x)).0 as i64;
|
||||
let source = _mm_set_epi64x(0, pixel);
|
||||
ll_sum = _mm_add_epi64(ll_sum, _mm_mul_epi32(source, coeff01_i64x2));
|
||||
}
|
||||
|
||||
_mm_storeu_si128((&mut ll_buf).as_mut_ptr() as *mut __m128i, ll_sum);
|
||||
let dst_pixel = dst_row.get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = normalizer_guard.clip(ll_buf[0] + ll_buf[1] + half_error);
|
||||
}
|
||||
}
|
||||
+20
-1
@@ -1,7 +1,7 @@
|
||||
use std::num::NonZeroU32;
|
||||
|
||||
use crate::image_view::{ImageRows, ImageRowsMut, TypedImageView, TypedImageViewMut};
|
||||
use crate::pixels::{Pixel, PixelType, U16x3, U8x2, U8x3, U8x4, F32, I32, U8};
|
||||
use crate::pixels::{Pixel, PixelType, U16x3, U8x2, U8x3, U8x4, F32, I32, U16, U8};
|
||||
use crate::{ImageBufferError, ImageView, ImageViewMut};
|
||||
|
||||
#[derive(Debug)]
|
||||
@@ -26,6 +26,7 @@ impl<'a> Image<'a> {
|
||||
let pixels = match pixel_type {
|
||||
PixelType::U8x2 => PixelsContainer::VecU8(vec![0; pixels_count * U8x2::size()]),
|
||||
PixelType::U8x3 => PixelsContainer::VecU8(vec![0; pixels_count * U8x3::size()]),
|
||||
PixelType::U16 => PixelsContainer::VecU8(vec![0; pixels_count * U16::size()]),
|
||||
PixelType::U16x3 => PixelsContainer::VecU8(vec![0; pixels_count * U16x3::size()]),
|
||||
PixelType::U8x4 => PixelsContainer::VecU8(vec![0; pixels_count * U8x4::size()]),
|
||||
PixelType::I32 => PixelsContainer::VecU8(vec![0; pixels_count * I32::size()]),
|
||||
@@ -155,6 +156,15 @@ impl<'a> Image<'a> {
|
||||
.collect(),
|
||||
)
|
||||
}
|
||||
PixelType::U16 => {
|
||||
let pixels = unsafe { buffer.align_to::<U16>().1 };
|
||||
ImageRows::U16(
|
||||
pixels
|
||||
.chunks_exact(self.width.get() as usize)
|
||||
.take(rows_count)
|
||||
.collect(),
|
||||
)
|
||||
}
|
||||
PixelType::U16x3 => {
|
||||
let pixels = unsafe { buffer.align_to::<U16x3>().1 };
|
||||
ImageRows::U16x3(
|
||||
@@ -230,6 +240,15 @@ impl<'a> Image<'a> {
|
||||
.collect(),
|
||||
)
|
||||
}
|
||||
PixelType::U16 => {
|
||||
let pixels = unsafe { buffer.align_to_mut::<U16>().1 };
|
||||
ImageRowsMut::U16(
|
||||
pixels
|
||||
.chunks_exact_mut(width.get() as usize)
|
||||
.take(rows_count)
|
||||
.collect(),
|
||||
)
|
||||
}
|
||||
PixelType::U16x3 => {
|
||||
let pixels = unsafe { buffer.align_to_mut::<U16x3>().1 };
|
||||
ImageRowsMut::U16x3(
|
||||
|
||||
+50
-1
@@ -2,7 +2,7 @@ use std::num::NonZeroU32;
|
||||
use std::slice;
|
||||
|
||||
use crate::errors::{CropBoxError, ImageBufferError, ImageRowsError};
|
||||
use crate::pixels::{Pixel, PixelType, U16x3, U8x2, U8x3, U8x4, F32, I32, U8};
|
||||
use crate::pixels::{Pixel, PixelType, U16x3, U8x2, U8x3, U8x4, F32, I32, U16, U8};
|
||||
|
||||
pub(crate) type RowMut<'a, 'b, T> = &'a mut &'b mut [T];
|
||||
pub(crate) type TwoRows<'a, T> = (&'a [T], &'a [T]);
|
||||
@@ -31,6 +31,7 @@ pub enum ImageRows<'a> {
|
||||
U8x2(Vec<&'a [U8x2]>),
|
||||
U8x3(Vec<&'a [U8x3]>),
|
||||
U8x4(Vec<&'a [U8x4]>),
|
||||
U16(Vec<&'a [U16]>),
|
||||
U16x3(Vec<&'a [U16x3]>),
|
||||
I32(Vec<&'a [I32]>),
|
||||
F32(Vec<&'a [F32]>),
|
||||
@@ -47,6 +48,7 @@ impl<'a> ImageRows<'a> {
|
||||
ImageRows::U8x2(rows) => check_rows_count_and_size(width, height, rows),
|
||||
ImageRows::U8x3(rows) => check_rows_count_and_size(width, height, rows),
|
||||
ImageRows::U8x4(rows) => check_rows_count_and_size(width, height, rows),
|
||||
ImageRows::U16(rows) => check_rows_count_and_size(width, height, rows),
|
||||
ImageRows::U16x3(rows) => check_rows_count_and_size(width, height, rows),
|
||||
ImageRows::I32(rows) => check_rows_count_and_size(width, height, rows),
|
||||
ImageRows::F32(rows) => check_rows_count_and_size(width, height, rows),
|
||||
@@ -59,6 +61,7 @@ impl<'a> ImageRows<'a> {
|
||||
Self::U8x2(_) => PixelType::U8x2,
|
||||
Self::U8x3(_) => PixelType::U8x3,
|
||||
Self::U8x4(_) => PixelType::U8x4,
|
||||
Self::U16(_) => PixelType::U16,
|
||||
Self::U16x3(_) => PixelType::U16x3,
|
||||
Self::I32(_) => PixelType::I32,
|
||||
Self::F32(_) => PixelType::F32,
|
||||
@@ -91,6 +94,7 @@ pub enum ImageRowsMut<'a> {
|
||||
U8x2(Vec<&'a mut [U8x2]>),
|
||||
U8x3(Vec<&'a mut [U8x3]>),
|
||||
U8x4(Vec<&'a mut [U8x4]>),
|
||||
U16(Vec<&'a mut [U16]>),
|
||||
U16x3(Vec<&'a mut [U16x3]>),
|
||||
I32(Vec<&'a mut [I32]>),
|
||||
F32(Vec<&'a mut [F32]>),
|
||||
@@ -107,6 +111,7 @@ impl<'a> ImageRowsMut<'a> {
|
||||
Self::U8x2(rows) => check_rows_count_and_size(width, height, rows),
|
||||
Self::U8x3(rows) => check_rows_count_and_size(width, height, rows),
|
||||
Self::U8x4(rows) => check_rows_count_and_size(width, height, rows),
|
||||
Self::U16(rows) => check_rows_count_and_size(width, height, rows),
|
||||
Self::U16x3(rows) => check_rows_count_and_size(width, height, rows),
|
||||
Self::I32(rows) => check_rows_count_and_size(width, height, rows),
|
||||
Self::F32(rows) => check_rows_count_and_size(width, height, rows),
|
||||
@@ -119,6 +124,7 @@ impl<'a> ImageRowsMut<'a> {
|
||||
Self::U8x2(_) => PixelType::U8x2,
|
||||
Self::U8x3(_) => PixelType::U8x3,
|
||||
Self::U8x4(_) => PixelType::U8x4,
|
||||
Self::U16(_) => PixelType::U16,
|
||||
Self::U16x3(_) => PixelType::U16x3,
|
||||
Self::I32(_) => PixelType::I32,
|
||||
Self::F32(_) => PixelType::F32,
|
||||
@@ -195,6 +201,15 @@ impl<'a> ImageView<'a> {
|
||||
.collect(),
|
||||
)
|
||||
}
|
||||
PixelType::U16 => {
|
||||
let pixels = align_buffer_to(buffer)?;
|
||||
ImageRows::U16(
|
||||
pixels
|
||||
.chunks_exact(width.get() as usize)
|
||||
.take(rows_count)
|
||||
.collect(),
|
||||
)
|
||||
}
|
||||
PixelType::U16x3 => {
|
||||
let pixels = align_buffer_to(buffer)?;
|
||||
ImageRows::U16x3(
|
||||
@@ -379,6 +394,19 @@ impl<'a> ImageView<'a> {
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn u16_image(&self) -> Option<TypedImageView<U16>> {
|
||||
if let ImageRows::U16(ref rows) = self.rows {
|
||||
Some(TypedImageView {
|
||||
width: self.width,
|
||||
height: self.height,
|
||||
crop_box: self.crop_box,
|
||||
rows,
|
||||
})
|
||||
} else {
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn u16x3_image(&self) -> Option<TypedImageView<U16x3>> {
|
||||
if let ImageRows::U16x3(ref rows) = self.rows {
|
||||
Some(TypedImageView {
|
||||
@@ -597,6 +625,15 @@ impl<'a> ImageViewMut<'a> {
|
||||
.collect(),
|
||||
)
|
||||
}
|
||||
PixelType::U16 => {
|
||||
let pixels = align_buffer_to_mut(buffer)?;
|
||||
ImageRowsMut::U16(
|
||||
pixels
|
||||
.chunks_exact_mut(width.get() as usize)
|
||||
.take(rows_count)
|
||||
.collect(),
|
||||
)
|
||||
}
|
||||
PixelType::U16x3 => {
|
||||
let pixels = align_buffer_to_mut(buffer)?;
|
||||
ImageRowsMut::U16x3(
|
||||
@@ -692,6 +729,18 @@ impl<'a> ImageViewMut<'a> {
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn u16_image<'s>(&'s mut self) -> Option<TypedImageViewMut<'s, 'a, U16>> {
|
||||
if let ImageRowsMut::U16(rows) = &mut self.rows {
|
||||
Some(TypedImageViewMut {
|
||||
width: self.width,
|
||||
height: self.height,
|
||||
rows,
|
||||
})
|
||||
} else {
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn u16x3_image<'s>(&'s mut self) -> Option<TypedImageViewMut<'s, 'a, U16x3>> {
|
||||
if let ImageRowsMut::U16x3(rows) = &mut self.rows {
|
||||
Some(TypedImageViewMut {
|
||||
|
||||
@@ -9,6 +9,7 @@ pub enum PixelType {
|
||||
U8x2,
|
||||
U8x3,
|
||||
U8x4,
|
||||
U16,
|
||||
U16x3,
|
||||
I32,
|
||||
F32,
|
||||
@@ -21,6 +22,7 @@ impl PixelType {
|
||||
Self::U8 => 1,
|
||||
Self::U8x2 => 2,
|
||||
Self::U8x3 => 3,
|
||||
Self::U16 => 2,
|
||||
Self::U16x3 => 6,
|
||||
_ => 4,
|
||||
}
|
||||
@@ -33,6 +35,7 @@ impl PixelType {
|
||||
Self::U8x2 => unsafe { buffer.align_to::<U8x2>().0.is_empty() },
|
||||
Self::U8x3 => unsafe { buffer.align_to::<U8x3>().0.is_empty() },
|
||||
Self::U8x4 => unsafe { buffer.align_to::<U8x4>().0.is_empty() },
|
||||
Self::U16 => unsafe { buffer.align_to::<U16>().0.is_empty() },
|
||||
Self::U16x3 => unsafe { buffer.align_to::<U16x3>().0.is_empty() },
|
||||
Self::I32 => unsafe { buffer.align_to::<I32>().0.is_empty() },
|
||||
Self::F32 => unsafe { buffer.align_to::<F32>().0.is_empty() },
|
||||
@@ -127,6 +130,14 @@ pixel_struct!(
|
||||
PixelType::U8x4,
|
||||
"Four bytes per pixel (RGBA, RGBx, CMYK and other)"
|
||||
);
|
||||
pixel_struct!(
|
||||
U16,
|
||||
u16,
|
||||
u16,
|
||||
1,
|
||||
PixelType::U16,
|
||||
"One `u16` component per pixel (e.g. L16)"
|
||||
);
|
||||
pixel_struct!(
|
||||
U16x3,
|
||||
[u16; 3],
|
||||
|
||||
@@ -104,6 +104,13 @@ impl Resizer {
|
||||
}
|
||||
}
|
||||
}
|
||||
PixelType::U16 => {
|
||||
if let Some(src_rows) = src_image.u16_image() {
|
||||
if let Some(dst_rows) = dst_image.u16_image() {
|
||||
self.resize_inner(src_rows, dst_rows);
|
||||
}
|
||||
}
|
||||
}
|
||||
PixelType::U16x3 => {
|
||||
if let Some(src_rows) = src_image.u16x3_image() {
|
||||
if let Some(dst_rows) = dst_image.u16x3_image() {
|
||||
|
||||
@@ -304,6 +304,54 @@ fn upscale_u8x4() {
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn downscale_u16() {
|
||||
type P = U16;
|
||||
let buffer = downscale_test::<P>(ResizeAlg::Nearest, CpuExtensions::None);
|
||||
assert_eq!(utils::image_u16_checksum::<1>(&buffer), [750529436]);
|
||||
|
||||
let mut cpu_extensions_vec = vec![CpuExtensions::None];
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
|
||||
cpu_extensions_vec.push(CpuExtensions::Avx2);
|
||||
}
|
||||
for cpu_extensions in cpu_extensions_vec {
|
||||
let buffer =
|
||||
downscale_test::<P>(ResizeAlg::Convolution(FilterType::Lanczos3), cpu_extensions);
|
||||
assert_eq!(
|
||||
utils::image_u16_checksum::<1>(&buffer),
|
||||
[751401243],
|
||||
"Error in checksum for {:?}",
|
||||
cpu_extensions
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn upscale_u16() {
|
||||
type P = U16;
|
||||
let buffer = upscale_test::<P>(ResizeAlg::Nearest, CpuExtensions::None);
|
||||
assert_eq!(utils::image_u16_checksum::<1>(&buffer), [295229780570]);
|
||||
|
||||
let mut cpu_extensions_vec = vec![CpuExtensions::None];
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Sse4_1);
|
||||
cpu_extensions_vec.push(CpuExtensions::Avx2);
|
||||
}
|
||||
for cpu_extensions in cpu_extensions_vec {
|
||||
let buffer =
|
||||
upscale_test::<P>(ResizeAlg::Convolution(FilterType::Lanczos3), cpu_extensions);
|
||||
assert_eq!(
|
||||
utils::image_u16_checksum::<1>(&buffer),
|
||||
[295246940755],
|
||||
"Error in checksum for {:?}",
|
||||
cpu_extensions
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn downscale_u16x3() {
|
||||
type P = U16x3;
|
||||
|
||||
@@ -30,6 +30,7 @@ pub trait PixelExt: Pixel {
|
||||
PixelType::U8x2 => "u8x2",
|
||||
PixelType::U8x3 => "u8x3",
|
||||
PixelType::U8x4 => "u8x4",
|
||||
PixelType::U16 => "u16",
|
||||
PixelType::U16x3 => "u16x3",
|
||||
PixelType::I32 => "i32",
|
||||
PixelType::F32 => "f32",
|
||||
@@ -106,6 +107,23 @@ impl PixelExt for U8x4 {
|
||||
}
|
||||
}
|
||||
|
||||
impl PixelExt for U16 {
|
||||
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
|
||||
// img.to_luma16()
|
||||
// .as_raw()
|
||||
// .iter()
|
||||
// .enumerate()
|
||||
// .flat_map(|(i, &c)| ((i & 0xffff) as u16).to_le_bytes())
|
||||
// .collect()
|
||||
|
||||
img.to_luma16()
|
||||
.as_raw()
|
||||
.iter()
|
||||
.flat_map(|&c| c.to_le_bytes())
|
||||
.collect()
|
||||
}
|
||||
}
|
||||
|
||||
impl PixelExt for U16x3 {
|
||||
fn img_into_bytes(img: DynamicImage) -> Vec<u8> {
|
||||
img.to_rgb8()
|
||||
@@ -148,6 +166,7 @@ pub fn save_result(image: &Image, name: &str) {
|
||||
PixelType::U8x2 => ColorType::La8,
|
||||
PixelType::U8x3 => ColorType::Rgb8,
|
||||
PixelType::U8x4 => ColorType::Rgba8,
|
||||
PixelType::U16 => ColorType::L16,
|
||||
PixelType::U16x3 => ColorType::Rgb16,
|
||||
PixelType::U8 => ColorType::L8,
|
||||
_ => panic!("Unsupported type of pixels"),
|
||||
|
||||
Reference in New Issue
Block a user