mirror of
https://github.com/Cykooz/fast_image_resize.git
synced 2026-10-08 01:11:09 +00:00
Added full support of optimisation with help of SSE4.1 for convolution of U8x3 images.
This commit is contained in:
+2
-1
@@ -12,7 +12,7 @@
|
||||
- associated method `component_count_of_values` renamed into `count_of_component_values`.
|
||||
- All pixel types (`U8`, `U8x2`, ...) replaced by type aliases for new
|
||||
generic structure `Pixel`. Use method `new()` to create
|
||||
instance of one pixel.
|
||||
instance of one pixel.
|
||||
- Added module `color` for working with colorspace and gamma:
|
||||
- Added mapper `SRGB_TO_RGB`. It is lazy static instance of `PixelComponentMapper` to
|
||||
convert images from SRGB colorspace to linear RGB and back.
|
||||
@@ -24,6 +24,7 @@
|
||||
components in whole image.
|
||||
- Added generic trait `IntoPixelComponent<Out: PixelComponent>`.
|
||||
- Added generic structure `Pixel` for create all types of pixels.
|
||||
- Added full support of optimisation with help of `SSE4.1` for convolution of `U8x3` images.
|
||||
|
||||
### Example application
|
||||
|
||||
|
||||
@@ -14,7 +14,7 @@ Supported pixel formats and available optimisations:
|
||||
|:------:|:--------------------------------------------------------------|:-----------:|:-------:|:----:|
|
||||
| U8 | One `u8` component per pixel (e.g. L) | + | partial | + |
|
||||
| U8x2 | Two `u8` components per pixel (e.g. LA) | + | + | + |
|
||||
| U8x3 | Three `u8` components per pixel (e.g. RGB) | + | partial | + |
|
||||
| U8x3 | Three `u8` components per pixel (e.g. RGB) | + | + | + |
|
||||
| U8x4 | Four `u8` components per pixel (e.g. RGBA, RGBx, CMYK) | + | + | + |
|
||||
| U16 | One `u16` components per pixel (e.g. L16) | + | + | + |
|
||||
| U16x2 | Two `u16` components per pixel (e.g. LA16) | + | + | + |
|
||||
|
||||
@@ -1,4 +1,6 @@
|
||||
use std::num::NonZeroU32;
|
||||
use std::thread::sleep;
|
||||
use std::time::Duration;
|
||||
|
||||
use glassbench::*;
|
||||
use image::imageops;
|
||||
@@ -46,7 +48,7 @@ pub fn bench_downscale_l(bench: &mut Bench) {
|
||||
let filter = match alg_name {
|
||||
"Nearest" => {
|
||||
// resizer doesn't support "nearest" algorithm
|
||||
task.iter(|| {});
|
||||
task.iter(|| sleep(Duration::new(0, 1)));
|
||||
return;
|
||||
}
|
||||
"Bilinear" => resize::Type::Triangle,
|
||||
|
||||
@@ -1,4 +1,6 @@
|
||||
use std::num::NonZeroU32;
|
||||
use std::thread::sleep;
|
||||
use std::time::Duration;
|
||||
|
||||
use glassbench::*;
|
||||
use image::imageops;
|
||||
@@ -46,7 +48,7 @@ pub fn bench_downscale_l16(bench: &mut Bench) {
|
||||
let filter = match alg_name {
|
||||
"Nearest" => {
|
||||
// resizer doesn't support "nearest" algorithm
|
||||
task.iter(|| {});
|
||||
task.iter(|| sleep(Duration::new(0, 1)));
|
||||
return;
|
||||
}
|
||||
"Bilinear" => resize::Type::Triangle,
|
||||
|
||||
@@ -1,4 +1,6 @@
|
||||
use std::num::NonZeroU32;
|
||||
use std::thread::sleep;
|
||||
use std::time::Duration;
|
||||
|
||||
use glassbench::*;
|
||||
use image::imageops;
|
||||
@@ -45,7 +47,7 @@ pub fn bench_downscale_rgb(bench: &mut Bench) {
|
||||
let filter = match alg_name {
|
||||
"Nearest" => {
|
||||
// resizer doesn't support "nearest" algorithm
|
||||
task.iter(|| {});
|
||||
task.iter(|| sleep(Duration::new(0, 1)));
|
||||
return;
|
||||
}
|
||||
"Bilinear" => resize::Type::Triangle,
|
||||
|
||||
@@ -1,4 +1,6 @@
|
||||
use std::num::NonZeroU32;
|
||||
use std::thread::sleep;
|
||||
use std::time::Duration;
|
||||
|
||||
use glassbench::*;
|
||||
use image::imageops;
|
||||
@@ -46,7 +48,7 @@ pub fn bench_downscale_rgb16(bench: &mut Bench) {
|
||||
let filter = match alg_name {
|
||||
"Nearest" => {
|
||||
// resizer doesn't support "nearest" algorithm
|
||||
task.iter(|| {});
|
||||
task.iter(|| sleep(Duration::new(0, 1)));
|
||||
return;
|
||||
}
|
||||
"Bilinear" => resize::Type::Triangle,
|
||||
|
||||
@@ -1,4 +1,6 @@
|
||||
use std::num::NonZeroU32;
|
||||
use std::thread::sleep;
|
||||
use std::time::Duration;
|
||||
|
||||
use glassbench::*;
|
||||
use resize::px::RGBA;
|
||||
@@ -27,7 +29,7 @@ pub fn bench_downscale_rgba(bench: &mut Bench) {
|
||||
let filter = match alg_name {
|
||||
"Nearest" => {
|
||||
// resizer doesn't support "nearest" algorithm
|
||||
task.iter(|| {});
|
||||
task.iter(|| sleep(Duration::new(0, 1)));
|
||||
return;
|
||||
}
|
||||
"Bilinear" => resize::Type::Triangle,
|
||||
|
||||
@@ -3,6 +3,8 @@ use resize::px::RGBA;
|
||||
use resize::Pixel::RGBA16P;
|
||||
use rgb::FromSlice;
|
||||
use std::num::NonZeroU32;
|
||||
use std::thread::sleep;
|
||||
use std::time::Duration;
|
||||
|
||||
use fast_image_resize::pixels::U16x4;
|
||||
use fast_image_resize::{CpuExtensions, FilterType, Image, MulDiv, ResizeAlg, Resizer};
|
||||
@@ -27,7 +29,7 @@ pub fn bench_downscale_rgba16(bench: &mut Bench) {
|
||||
let filter = match alg_name {
|
||||
"Nearest" => {
|
||||
// resizer doesn't support "nearest" algorithm
|
||||
task.iter(|| {});
|
||||
task.iter(|| sleep(Duration::new(0, 1)));
|
||||
return;
|
||||
}
|
||||
"Bilinear" => resize::Type::Triangle,
|
||||
|
||||
@@ -25,10 +25,10 @@ pub fn print_md_table(bench: &Bench) {
|
||||
res_map.insert(crate_name.clone(), Vec::new());
|
||||
}
|
||||
if let Some(values) = res_map.get_mut(&crate_name) {
|
||||
let s_value = format!("{:.2}", value);
|
||||
if s_value == "0.00" {
|
||||
if value < 0.10 {
|
||||
values.push("-".to_string());
|
||||
} else {
|
||||
let s_value = format!("{:.2}", value);
|
||||
values.push(s_value);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -23,9 +23,7 @@ impl Convolution for U8x3 {
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => avx2::horiz_convolution(src_image, dst_image, offset, coeffs),
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe {
|
||||
sse4::horiz_convolution(src_image, dst_image, offset, coeffs)
|
||||
},
|
||||
CpuExtensions::Sse4_1 => sse4::horiz_convolution(src_image, dst_image, offset, coeffs),
|
||||
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,15 +1,297 @@
|
||||
use crate::convolution::Coefficients;
|
||||
use std::arch::x86_64::*;
|
||||
use std::intrinsics::transmute;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::image_view::{FourRows, FourRowsMut};
|
||||
use crate::pixels::U8x3;
|
||||
use crate::simd_utils;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
use super::native;
|
||||
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
pub(crate) unsafe fn horiz_convolution(
|
||||
#[inline]
|
||||
pub(crate) fn horiz_convolution(
|
||||
src_image: &ImageView<U8x3>,
|
||||
dst_image: &mut ImageViewMut<U8x3>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
native::horiz_convolution(src_image, dst_image, offset, coeffs);
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_8u4x(src_rows, dst_rows, &coefficients_chunks, precision);
|
||||
}
|
||||
}
|
||||
|
||||
let mut yy = dst_height - dst_height % 4;
|
||||
while yy < dst_height {
|
||||
unsafe {
|
||||
horiz_convolution_8u(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
&coefficients_chunks,
|
||||
precision,
|
||||
);
|
||||
}
|
||||
yy += 1;
|
||||
}
|
||||
}
|
||||
|
||||
/// For safety, it is necessary to ensure the following conditions:
|
||||
/// - length of all rows in src_rows must be equal
|
||||
/// - length of all rows in dst_rows must be equal
|
||||
/// - coefficients_chunks.len() == dst_rows.0.len()
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[inline]
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
unsafe fn horiz_convolution_8u4x(
|
||||
src_rows: FourRows<U8x3>,
|
||||
dst_rows: FourRowsMut<U8x3>,
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
precision: u8,
|
||||
) {
|
||||
let (s_row0, s_row1, s_row2, s_row3) = src_rows;
|
||||
let s_rows = [s_row0, s_row1, s_row2, s_row3];
|
||||
let (d_row0, d_row1, d_row2, d_row3) = dst_rows;
|
||||
let d_rows = [d_row0, d_row1, d_row2, d_row3];
|
||||
let zero = _mm_setzero_si128();
|
||||
let initial = _mm_set1_epi32(1 << (precision - 1));
|
||||
let src_width = s_row0.len();
|
||||
|
||||
/*
|
||||
|R G B | |R G B | |R G B | |R G B | |R G B | |R |
|
||||
|00 01 02| |03 04 05| |06 07 08| |09 10 11| |12 13 14| |15|
|
||||
|
||||
Ignore 12-15 bytes in register and
|
||||
shuffle other components with converting from u8 into i16:
|
||||
|
||||
x: |-1 -1| |-1 -1|
|
||||
B: |-1 05| |-1 02|
|
||||
G: |-1 04| |-1 01|
|
||||
R: |-1 03| |-1 00|
|
||||
*/
|
||||
#[rustfmt::skip]
|
||||
let sh_lo = _mm_set_epi8(
|
||||
-1, -1, -1, -1, -1, 5, -1, 2, -1, 4, -1, 1, -1, 3, -1, 0,
|
||||
);
|
||||
/*
|
||||
x: |-1 -1| |-1 -1|
|
||||
B: |-1 11| |-1 08|
|
||||
G: |-1 10| |-1 07|
|
||||
R: |-1 09| |-1 06|
|
||||
*/
|
||||
#[rustfmt::skip]
|
||||
let sh_hi = _mm_set_epi8(
|
||||
-1, -1, -1, -1, -1, 11, -1, 8, -1, 10, -1, 7, -1, 9, -1, 6,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let x_start = coeffs_chunk.start as usize;
|
||||
let mut x = x_start;
|
||||
|
||||
let mut sss_a = [initial; 4];
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
|
||||
// Next block of code will be load source pixels by 16 bytes per time.
|
||||
// We must guarantee what this process will not go beyond
|
||||
// the one row of image.
|
||||
// (16 bytes) / (3 bytes per pixel) = 5 whole pixels + 1 byte
|
||||
let max_x = src_width.saturating_sub(5);
|
||||
if x < max_x {
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
|
||||
for k in coeffs_by_4 {
|
||||
let mmk0 = simd_utils::ptr_i16_to_set1_epi32(k, 0);
|
||||
let mmk1 = simd_utils::ptr_i16_to_set1_epi32(k, 2);
|
||||
for i in 0..4 {
|
||||
let source = simd_utils::loadu_si128(s_rows[i], x);
|
||||
let pix = _mm_shuffle_epi8(source, sh_lo);
|
||||
let mut sss = sss_a[i];
|
||||
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk0));
|
||||
let pix = _mm_shuffle_epi8(source, sh_hi);
|
||||
sss_a[i] = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk1));
|
||||
}
|
||||
|
||||
x += 4;
|
||||
if x >= max_x {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Next block of code will be load source pixels by 8 bytes per time.
|
||||
// We must guarantee what this process will not go beyond
|
||||
// the one row of image.
|
||||
// (8 bytes) / (3 bytes per pixel) = 2 whole pixels + 2 bytes
|
||||
let max_x = src_width.saturating_sub(2);
|
||||
if x < max_x {
|
||||
let coeffs_by_2 = coeffs[x - x_start..].chunks_exact(2);
|
||||
|
||||
for k in coeffs_by_2 {
|
||||
let mmk = simd_utils::ptr_i16_to_set1_epi32(k, 0);
|
||||
|
||||
for i in 0..4 {
|
||||
let source = simd_utils::loadl_epi64(s_rows[i], x);
|
||||
let pix = _mm_shuffle_epi8(source, sh_lo);
|
||||
sss_a[i] = _mm_add_epi32(sss_a[i], _mm_madd_epi16(pix, mmk));
|
||||
}
|
||||
|
||||
x += 2;
|
||||
if x >= max_x {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
coeffs = coeffs.split_at(x - x_start).1;
|
||||
for &k in coeffs {
|
||||
let mmk = _mm_set1_epi32(k as i32);
|
||||
for i in 0..4 {
|
||||
let pix = simd_utils::mm_cvtepu8_epi32_u8x3(s_rows[i], x);
|
||||
sss_a[i] = _mm_add_epi32(sss_a[i], _mm_madd_epi16(pix, mmk));
|
||||
}
|
||||
|
||||
x += 1;
|
||||
}
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss_a[0] = _mm_srai_epi32::<$imm8>(sss_a[0]);
|
||||
sss_a[1] = _mm_srai_epi32::<$imm8>(sss_a[1]);
|
||||
sss_a[2] = _mm_srai_epi32::<$imm8>(sss_a[2]);
|
||||
sss_a[3] = _mm_srai_epi32::<$imm8>(sss_a[3]);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
|
||||
for i in 0..4 {
|
||||
let sss = _mm_packs_epi32(sss_a[i], zero);
|
||||
let pixel: u32 = transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss, zero)));
|
||||
let bytes = pixel.to_le_bytes();
|
||||
d_rows[i].get_unchecked_mut(dst_x).0 = [bytes[0], bytes[1], bytes[2]];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// For safety, it is necessary to ensure the following conditions:
|
||||
/// - bounds.len() == dst_row.len()
|
||||
/// - coeffs.len() == dst_rows.0.len() * window_size
|
||||
/// - max(bound.start + bound.size for bound in bounds) <= src_row.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[inline]
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
unsafe fn horiz_convolution_8u(
|
||||
src_row: &[U8x3],
|
||||
dst_row: &mut [U8x3],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
precision: u8,
|
||||
) {
|
||||
#[rustfmt::skip]
|
||||
let pix_sh1 = _mm_set_epi8(
|
||||
-1, -1, -1, -1, -1, 5, -1, 2, -1, 4, -1, 1, -1, 3, -1, 0,
|
||||
);
|
||||
#[rustfmt::skip]
|
||||
let coef_sh1 = _mm_set_epi8(
|
||||
3, 2, 1, 0, 3, 2, 1, 0, 3, 2, 1, 0, 3, 2, 1, 0,
|
||||
);
|
||||
#[rustfmt::skip]
|
||||
let pix_sh2 = _mm_set_epi8(
|
||||
-1, -1, -1, -1, -1, 11, -1, 8, -1, 10, -1, 7, -1, 9, -1, 6,
|
||||
);
|
||||
#[rustfmt::skip]
|
||||
let coef_sh2 = _mm_set_epi8(
|
||||
7, 6, 5, 4, 7, 6, 5, 4, 7, 6, 5, 4, 7, 6, 5, 4,
|
||||
);
|
||||
/*
|
||||
Load 8 bytes from memory into low half of 16-bytes register:
|
||||
|R G B | |R G B | |R G |
|
||||
|00 01 02| |03 04 05| |06 07| 08 09 10 11 12 13 14 15
|
||||
|
||||
Ignore 06-16 bytes in 16-bytes register and
|
||||
shuffle other components with converting from u8 into i16:
|
||||
|
||||
x: |-1 -1| |-1 -1|
|
||||
B: |-1 05| |-1 02|
|
||||
G: |-1 04| |-1 01|
|
||||
R: |-1 03| |-1 00|
|
||||
*/
|
||||
let src_width = src_row.len();
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let x_start = coeffs_chunk.start as usize;
|
||||
let mut x = x_start;
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut sss = _mm_set1_epi32(1 << (precision - 1));
|
||||
|
||||
// Next block of code will be load source pixels by 16 bytes per time.
|
||||
// We must guarantee what this process will not go beyond
|
||||
// the one row of image.
|
||||
// (16 bytes) / (3 bytes per pixel) = 5 whole pixels + 1 bytes
|
||||
let max_x = src_width.saturating_sub(5);
|
||||
if x < max_x {
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
for k in coeffs_by_4 {
|
||||
let ksource = simd_utils::loadl_epi64(k, 0);
|
||||
let source = simd_utils::loadu_si128(src_row, x);
|
||||
|
||||
let pix = _mm_shuffle_epi8(source, pix_sh1);
|
||||
let mmk = _mm_shuffle_epi8(ksource, coef_sh1);
|
||||
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
let pix = _mm_shuffle_epi8(source, pix_sh2);
|
||||
let mmk = _mm_shuffle_epi8(ksource, coef_sh2);
|
||||
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
x += 4;
|
||||
if x >= max_x {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Next block of code will be load source pixels by 8 bytes per time.
|
||||
// We must guarantee what this process will not go beyond
|
||||
// the one row of image.
|
||||
// (8 bytes) / (3 bytes per pixel) = 2 whole pixels + 2 bytes
|
||||
let max_x = src_width.saturating_sub(2);
|
||||
if x < max_x {
|
||||
let coeffs_by_2 = coeffs[x - x_start..].chunks_exact(2);
|
||||
|
||||
for k in coeffs_by_2 {
|
||||
let mmk = simd_utils::ptr_i16_to_set1_epi32(k, 0);
|
||||
let source = simd_utils::loadl_epi64(src_row, x);
|
||||
let pix = _mm_shuffle_epi8(source, pix_sh1);
|
||||
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
|
||||
|
||||
x += 2;
|
||||
if x >= max_x {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
coeffs = coeffs.split_at(x - x_start).1;
|
||||
for &k in coeffs {
|
||||
let pix = simd_utils::mm_cvtepu8_epi32_u8x3(src_row, x);
|
||||
let mmk = _mm_set1_epi32(k as i32);
|
||||
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
|
||||
x += 1;
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss = _mm_srai_epi32::<$imm8>(sss);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
|
||||
sss = _mm_packs_epi32(sss, sss);
|
||||
let pixel: u32 = transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss, sss)));
|
||||
let bytes = pixel.to_le_bytes();
|
||||
dst_row.get_unchecked_mut(dst_x).0 = [bytes[0], bytes[1], bytes[2]];
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user