Small optimisation of resizing of images with U8x4 pixel format.

This commit is contained in:
Kirill Kuzminykh
2021-07-13 00:13:17 +03:00
parent c3be3bc77a
commit dec46ebb49
16 changed files with 258 additions and 210 deletions
+3 -10
View File
@@ -10,16 +10,9 @@ const fn p(r: u8, g: u8, b: u8, a: u8) -> u32 {
// Multiplies by alpha
fn get_src_image(width: NonZeroU32, height: NonZeroU32, pixel: u32) -> ImageData<Vec<u8>> {
let rgba: [u8; 4] = pixel.to_le_bytes();
let buf_size = (width.get() * height.get()) as usize * 4;
let mut buffer = vec![0u8; buf_size];
buffer.chunks_exact_mut(4).for_each(|c| {
c[0] = rgba[0];
c[1] = rgba[1];
c[2] = rgba[2];
c[3] = rgba[3];
});
fn get_src_image(width: NonZeroU32, height: NonZeroU32, pixel: u32) -> ImageData<Vec<u32>> {
let buf_size = (width.get() * height.get()) as usize;
let buffer = vec![pixel; buf_size];
ImageData::new(width, height, buffer, PixelType::U8x4).unwrap()
}
+6 -1
View File
@@ -77,10 +77,15 @@ pub fn bench_downscale_rgb(bench: &mut Bench) {
};
for alg_name in alg_names {
let src_rgba_image = utils::get_big_rgba_image();
let buf: Vec<u32> = src_rgba_image
.as_raw()
.chunks_exact(4)
.map(|p| u32::from_le_bytes([p[0], p[1], p[2], p[3]]))
.collect();
let src_image_data = ImageData::new(
NonZeroU32::new(src_image.width()).unwrap(),
NonZeroU32::new(src_image.height()).unwrap(),
src_rgba_image.as_raw(),
buf,
PixelType::U8x4,
)
.unwrap();
+6 -1
View File
@@ -84,10 +84,15 @@ pub fn bench_downscale_rgba(bench: &mut Bench) {
"lanczos3" => ResizeAlg::Convolution(FilterType::Lanczos3),
_ => return,
};
let buf: Vec<u32> = src_image
.as_raw()
.chunks_exact(4)
.map(|p| u32::from_le_bytes([p[0], p[1], p[2], p[3]]))
.collect();
let src_image_data = ImageData::new(
NonZeroU32::new(src_image.width()).unwrap(),
NonZeroU32::new(src_image.height()).unwrap(),
src_image.as_raw(),
buf,
PixelType::U8x4,
)
.unwrap();
+12 -4
View File
@@ -13,11 +13,15 @@ const NEW_HEIGHT: u32 = 567;
const NEW_BIG_WIDTH: u32 = 4928;
const NEW_BIG_HEIGHT: u32 = 3279;
fn get_big_source_image() -> ImageData<Vec<u8>> {
fn get_big_source_image() -> ImageData<Vec<u32>> {
let img = utils::get_big_rgba_image();
let width = img.width();
let height = img.height();
let buf = img.as_raw().clone();
let buf = img
.as_raw()
.chunks_exact(4)
.map(|p| u32::from_le_bytes([p[0], p[1], p[2], p[3]]))
.collect();
ImageData::new(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
@@ -27,11 +31,15 @@ fn get_big_source_image() -> ImageData<Vec<u8>> {
.unwrap()
}
fn get_small_source_image() -> ImageData<Vec<u8>> {
fn get_small_source_image() -> ImageData<Vec<u32>> {
let img = utils::get_small_rgba_image();
let width = img.width();
let height = img.height();
let buf = img.as_raw().clone();
let buf = img
.as_raw()
.chunks_exact(4)
.map(|p| u32::from_le_bytes([p[0], p[1], p[2], p[3]]))
.collect();
ImageData::new(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
+55 -54
View File
@@ -1,7 +1,7 @@
use std::arch::x86_64::*;
use std::intrinsics::transmute;
use crate::convolution::{Bound, Coefficients, Convolution};
use crate::convolution::{Bound, Coefficients, CoefficientsChunk, Convolution};
use crate::image_view::{DstImageView, FourRows, FourRowsMut, SrcImageView};
use crate::{optimisations, simd_utils};
@@ -24,9 +24,7 @@ impl Avx2 {
&self,
src_rows: FourRows,
dst_rows: FourRowsMut,
coeffs: &[i16],
window_size: usize,
bounds: &[Bound],
coefficients_chunks: &[CoefficientsChunk],
precision: u8,
) {
let (s_row0, s_row1, s_row2, s_row3) = src_rows;
@@ -35,28 +33,30 @@ impl Avx2 {
let initial = _mm256_set1_epi32(1 << (precision - 1));
#[rustfmt::skip]
let sh1 = _mm256_set_epi8(
let sh1 = _mm256_set_epi8(
-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0,
-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0,
);
#[rustfmt::skip]
let sh2 = _mm256_set_epi8(
let sh2 = _mm256_set_epi8(
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
);
let coeffs_chunks = coeffs.chunks(window_size);
for (dst_x, (&bound, k)) in bounds.iter().zip(coeffs_chunks).enumerate() {
let x_start = bound.start as usize;
let x_size = bound.size as usize;
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let x_start = coeffs_chunk.start as usize;
let mut x: usize = 0;
let mut sss0 = initial;
let mut sss1 = initial;
let coeffs = coeffs_chunk.values;
while x < x_size.saturating_sub(3) {
let mmk0 = simd_utils::ptr_i16_to_256set1_epi32(k, x);
let mmk1 = simd_utils::ptr_i16_to_256set1_epi32(k, x + 2);
let coeffs_by_4 = coeffs.chunks_exact(4);
let reminder1 = coeffs_by_4.remainder();
for k in coeffs_by_4 {
let mmk0 = simd_utils::ptr_i16_to_256set1_epi32(k, 0);
let mmk1 = simd_utils::ptr_i16_to_256set1_epi32(k, 2);
let mut source = _mm256_inserti128_si256(
_mm256_castsi128_si256(simd_utils::loadu_si128(s_row0, x + x_start)),
@@ -81,8 +81,11 @@ impl Avx2 {
x += 4;
}
while x < x_size.saturating_sub(1) {
let mmk = simd_utils::ptr_i16_to_256set1_epi32(k, x);
let coeffs_by_2 = reminder1.chunks_exact(2);
let reminder2 = coeffs_by_2.remainder();
for k in coeffs_by_2 {
let mmk = simd_utils::ptr_i16_to_256set1_epi32(k, 0);
let mut pix = _mm256_inserti128_si256(
_mm256_castsi128_si256(simd_utils::loadl_epi64(s_row0, x + x_start)),
@@ -103,9 +106,9 @@ impl Avx2 {
x += 2;
}
while x < x_size {
for &k in reminder2 {
// [16] xx k0 xx k0 xx k0 xx k0 xx k0 xx k0 xx k0 xx k0
let mmk = _mm256_set1_epi32(*k.get_unchecked(x) as i32);
let mmk = _mm256_set1_epi32(k as i32);
// [16] xx a0 xx b0 xx g0 xx r0 xx a0 xx b0 xx g0 xx r0
let mut pix = _mm256_inserti128_si256(
@@ -158,57 +161,57 @@ impl Avx2 {
&self,
src_row: &[u32],
dst_row: &mut [u32],
coeffs: &[i16],
window_size: usize,
bounds: &[Bound],
coefficients_chunks: &[CoefficientsChunk],
precision: u8,
) {
#[rustfmt::skip]
let sh1 = _mm256_set_epi8(
let sh1 = _mm256_set_epi8(
-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0,
-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0,
);
#[rustfmt::skip]
let sh2 = _mm256_set_epi8(
let sh2 = _mm256_set_epi8(
11, 10, 9, 8, 11, 10, 9, 8, 11, 10, 9, 8, 11, 10, 9, 8,
3, 2, 1, 0, 3, 2, 1, 0, 3, 2, 1, 0, 3, 2, 1, 0,
);
#[rustfmt::skip]
let sh3 = _mm256_set_epi8(
let sh3 = _mm256_set_epi8(
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
);
#[rustfmt::skip]
let sh4 = _mm256_set_epi8(
let sh4 = _mm256_set_epi8(
15, 14, 13, 12, 15, 14, 13, 12, 15, 14, 13, 12, 15, 14, 13, 12,
7, 6, 5, 4, 7, 6, 5, 4, 7, 6, 5, 4, 7, 6, 5, 4,
);
#[rustfmt::skip]
let sh5 = _mm256_set_epi8(
let sh5 = _mm256_set_epi8(
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0,
);
#[rustfmt::skip]
let sh6 = _mm256_set_epi8(
let sh6 = _mm256_set_epi8(
7, 6, 5, 4, 7, 6, 5, 4, 7, 6, 5, 4, 7, 6, 5, 4,
3, 2, 1, 0, 3, 2, 1, 0, 3, 2, 1, 0, 3, 2, 1, 0,
);
let sh7 = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
let coeffs_chunks = coeffs.chunks(window_size);
for (xx, (&bound, k)) in bounds.iter().zip(coeffs_chunks).enumerate() {
let x_start = bound.start as usize;
let x_size = bound.size as usize;
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let x_start = coeffs_chunk.start as usize;
let mut x: usize = 0;
let mut coeffs = coeffs_chunk.values;
let mut sss: __m128i = if x_size < 8 {
let mut sss: __m128i = if coeffs.len() < 8 {
_mm_set1_epi32(1 << (precision - 1))
} else {
// Lower part will be added to higher, use only half of the error
let mut sss256 = _mm256_set1_epi32(1 << (precision - 2));
while x < x_size.saturating_sub(7) {
let tmp = simd_utils::loadu_si128(k, x);
let coeffs_by_8 = coeffs.chunks_exact(8);
let reminder1 = coeffs_by_8.remainder();
for k in coeffs_by_8 {
let tmp = simd_utils::loadu_si128(k, 0);
let ksource = _mm256_insertf128_si256(_mm256_castsi128_si256(tmp), tmp, 1);
let source = simd_utils::loadu_si256(src_row, x + x_start);
@@ -224,8 +227,11 @@ impl Avx2 {
x += 8;
}
while x < x_size.saturating_sub(3) {
let tmp = simd_utils::loadl_epi64(k, x);
let coeffs_by_4 = reminder1.chunks_exact(4);
coeffs = coeffs_by_4.remainder();
for k in coeffs_by_4 {
let tmp = simd_utils::loadl_epi64(k, 0);
let ksource = _mm256_insertf128_si256(_mm256_castsi128_si256(tmp), tmp, 1);
let tmp = simd_utils::loadu_si128(src_row, x + x_start);
@@ -244,8 +250,11 @@ impl Avx2 {
)
};
while x < x_size.saturating_sub(1) {
let mmk = simd_utils::ptr_i16_to_set1_epi32(k, x);
let coeffs_by_2 = coeffs.chunks_exact(2);
let reminder1 = coeffs_by_2.remainder();
for k in coeffs_by_2 {
let mmk = simd_utils::ptr_i16_to_set1_epi32(k, 0);
let source = simd_utils::loadl_epi64(src_row, x + x_start);
let pix = _mm_shuffle_epi8(source, sh7);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
@@ -253,9 +262,9 @@ impl Avx2 {
x += 2
}
while x < x_size {
for &k in reminder1 {
let pix = simd_utils::mm_cvtepu8_epi32(src_row, x + x_start);
let mmk = _mm_set1_epi32(*k.get_unchecked(x) as i32);
let mmk = _mm_set1_epi32(k as i32);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
x += 1;
@@ -269,7 +278,7 @@ impl Avx2 {
constify_imm8!(precision, call);
sss = _mm_packs_epi32(sss, sss);
*dst_row.get_unchecked_mut(xx) =
*dst_row.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss, sss)));
}
}
@@ -467,23 +476,17 @@ impl Convolution for Avx2 {
let (values, window_size, bounds_per_pixel) =
(coeffs.values, coeffs.window_size, coeffs.bounds);
let mut normalizer_guard = optimisations::NormalizerGuard::new(values);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coeffs_i16 = normalizer_guard.normalized();
let coefficients_chunks =
normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
let dst_height = dst_image.height().get();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_image.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
self.horiz_convolution_8u4x(
src_rows,
dst_rows,
coeffs_i16,
window_size,
&bounds_per_pixel,
precision,
);
self.horiz_convolution_8u4x(src_rows, dst_rows, &coefficients_chunks, precision);
}
}
@@ -493,9 +496,7 @@ impl Convolution for Avx2 {
self.horiz_convolution_8u(
src_image.get_row(yy + offset).unwrap(),
dst_image.get_row_mut(yy).unwrap(),
coeffs_i16,
window_size,
&bounds_per_pixel,
&coefficients_chunks,
precision,
);
}
@@ -512,7 +513,7 @@ impl Convolution for Avx2 {
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let mut normalizer_guard = optimisations::NormalizerGuard::new(values);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coeffs_i16 = normalizer_guard.normalized();
let coeffs_chunks = coeffs_i16.chunks(window_size);
+6
View File
@@ -38,6 +38,12 @@ pub struct Bound {
pub size: u32,
}
#[derive(Debug, Clone, Copy)]
pub struct CoefficientsChunk<'a> {
pub start: u32,
pub values: &'a [i16],
}
#[derive(Debug, Clone)]
pub struct Coefficients {
pub values: Vec<f64>,
+10 -14
View File
@@ -16,19 +16,17 @@ impl Convolution for NativeU8x4 {
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let mut normalizer_guard = optimisations::NormalizerGuard::new(values);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coeffs_i16 = normalizer_guard.normalized();
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
let dst_rows = dst_image.iter_rows_mut();
for (y_dst, dst_row) in dst_rows.enumerate() {
let y_src = y_dst as u32 + offset;
for (x_dst, (&bound, dst_pixel)) in bounds.iter().zip(dst_row.iter_mut()).enumerate() {
let first_x_src = bound.start;
let start_index = window_size * x_dst;
let end_index = start_index + bound.size as usize;
let ks = &coeffs_i16[start_index..end_index];
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
let first_x_src = coeffs_chunk.start;
let ks = coeffs_chunk.values;
let mut ss0 = 1 << (precision - 1);
let mut ss1 = ss0;
@@ -63,16 +61,14 @@ impl Convolution for NativeU8x4 {
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let mut normalizer_guard = optimisations::NormalizerGuard::new(values);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coeffs_i16 = normalizer_guard.normalized();
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
let dst_rows = dst_image.iter_rows_mut();
for (y_dst, (&bound, dst_row)) in bounds.iter().zip(dst_rows).enumerate() {
let first_y_src = bound.start;
let start_index = window_size * y_dst;
let end_index = start_index + bound.size as usize;
let ks = &coeffs_i16[start_index..end_index];
for (&coeffs_chunk, dst_row) in coefficients_chunks.iter().zip(dst_rows) {
let first_y_src = coeffs_chunk.start;
let ks = coeffs_chunk.values;
for (x_src, out_pixel) in dst_row.iter_mut().enumerate() {
let mut ss0 = 1 << (precision - 1);
+52 -49
View File
@@ -1,7 +1,7 @@
use std::arch::x86_64::*;
use std::intrinsics::transmute;
use crate::convolution::{Bound, Coefficients, Convolution};
use crate::convolution::{Bound, Coefficients, CoefficientsChunk, Convolution};
use crate::image_view::{DstImageView, FourRows, FourRowsMut, SrcImageView};
use crate::{optimisations, simd_utils};
@@ -23,9 +23,7 @@ impl Sse4 {
&self,
src_rows: FourRows,
dst_rows: FourRowsMut,
coeffs: &[i16],
window_size: usize,
bounds: &[Bound],
coefficients_chunks: &[CoefficientsChunk],
precision: u8,
) {
let (s_row0, s_row1, s_row2, s_row3) = src_rows;
@@ -35,10 +33,8 @@ impl Sse4 {
let mask_hi = _mm_set_epi8(-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8);
let mask = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
let coeffs_chunks = coeffs.chunks(window_size);
for (xx, (&bound, k)) in bounds.iter().zip(coeffs_chunks).enumerate() {
let x_start = bound.start as usize;
let x_size = bound.size as usize;
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let x_start = coeffs_chunk.start as usize;
let mut x: usize = 0;
let mut sss0 = initial;
@@ -46,9 +42,13 @@ impl Sse4 {
let mut sss2 = initial;
let mut sss3 = initial;
while x < x_size.saturating_sub(3) {
let mmk_lo = simd_utils::ptr_i16_to_set1_epi32(k, x);
let mmk_hi = simd_utils::ptr_i16_to_set1_epi32(k, x + 2);
let coeffs = coeffs_chunk.values;
let coeffs_by_4 = coeffs.chunks_exact(4);
let reminder1 = coeffs_by_4.remainder();
for k in coeffs_by_4 {
let mmk_lo = simd_utils::ptr_i16_to_set1_epi32(k, 0);
let mmk_hi = simd_utils::ptr_i16_to_set1_epi32(k, 2);
// [8] a3 b3 g3 r3 a2 b2 g2 r2 a1 b1 g1 r1 a0 b0 g0 r0
let mut source = simd_utils::loadu_si128(s_row0, x + x_start);
@@ -79,9 +79,12 @@ impl Sse4 {
x += 4;
}
while x < x_size.saturating_sub(1) {
let coeffs_by_2 = reminder1.chunks_exact(2);
let reminder2 = coeffs_by_2.remainder();
for k in coeffs_by_2 {
// [16] k1 k0 k1 k0 k1 k0 k1 k0
let mmk = simd_utils::ptr_i16_to_set1_epi32(k, x);
let mmk = simd_utils::ptr_i16_to_set1_epi32(k, 0);
// [8] x x x x x x x x a1 b1 g1 r1 a0 b0 g0 r0
let mut pix = simd_utils::loadl_epi64(s_row0, x + x_start);
@@ -104,9 +107,9 @@ impl Sse4 {
x += 2;
}
while x < x_size {
for &k in reminder2 {
// [16] xx k0 xx k0 xx k0 xx k0
let mmk = _mm_set1_epi32(*k.get_unchecked(x) as i32);
let mmk = _mm_set1_epi32(k as i32);
// [16] xx a0 xx b0 xx g0 xx r0
let mut pix = simd_utils::mm_cvtepu8_epi32(s_row0, x);
sss0 = _mm_add_epi32(sss0, _mm_madd_epi16(pix, mmk));
@@ -137,13 +140,13 @@ impl Sse4 {
sss1 = _mm_packs_epi32(sss1, sss1);
sss2 = _mm_packs_epi32(sss2, sss2);
sss3 = _mm_packs_epi32(sss3, sss3);
*d_row0.get_unchecked_mut(xx) =
*d_row0.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss0, sss0)));
*d_row1.get_unchecked_mut(xx) =
*d_row1.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss1, sss1)));
*d_row2.get_unchecked_mut(xx) =
*d_row2.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss2, sss2)));
*d_row3.get_unchecked_mut(xx) =
*d_row3.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss3, sss3)));
}
}
@@ -158,12 +161,9 @@ impl Sse4 {
&self,
src_row: &[u32],
dst_row: &mut [u32],
coeffs: &[i16],
window_size: usize,
bounds: &[Bound],
coefficients_chunks: &[CoefficientsChunk],
precision: u8,
) {
let coeffs_chunks = coeffs.chunks(window_size);
let initial = _mm_set1_epi32(1 << (precision - 1));
let sh1 = _mm_set_epi8(-1, 11, -1, 3, -1, 10, -1, 2, -1, 9, -1, 1, -1, 8, -1, 0);
let sh2 = _mm_set_epi8(5, 4, 1, 0, 5, 4, 1, 0, 5, 4, 1, 0, 5, 4, 1, 0);
@@ -175,14 +175,19 @@ impl Sse4 {
);
let sh7 = _mm_set_epi8(-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0);
for (xx, (&bound, k)) in bounds.iter().zip(coeffs_chunks).enumerate() {
let x_start = bound.start as usize;
let x_size = bound.size as usize;
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
// for (dst_x, (&bound, k)) in bounds.iter().zip(coeffs_chunks).enumerate() {
let x_start = coeffs_chunk.start as usize;
let mut x: usize = 0;
let mut coeffs = coeffs_chunk.values;
let mut sss = initial;
while x < x_size.saturating_sub(7) {
let ksource = simd_utils::loadu_si128(k, x);
let coeffs_by_8 = coeffs.chunks_exact(8);
let reminder1 = coeffs_by_8.remainder();
for k in coeffs_by_8 {
let ksource = simd_utils::loadu_si128(k, 0);
let mut source = simd_utils::loadu_si128(src_row, x + x_start);
@@ -207,9 +212,12 @@ impl Sse4 {
x += 8;
}
while x < x_size.saturating_sub(3) {
let coeffs_by_4 = reminder1.chunks_exact(4);
coeffs = coeffs_by_4.remainder();
for k in coeffs_by_4 {
let source = simd_utils::loadu_si128(src_row, x + x_start);
let ksource = simd_utils::loadl_epi64(k, x);
let ksource = simd_utils::loadl_epi64(k, 0);
let mut pix = _mm_shuffle_epi8(source, sh1);
let mut mmk = _mm_shuffle_epi8(ksource, sh2);
@@ -222,8 +230,11 @@ impl Sse4 {
x += 4;
}
while x < x_size.saturating_sub(1) {
let mmk = simd_utils::ptr_i16_to_set1_epi32(k, x);
let coeffs_by_2 = coeffs.chunks_exact(2);
let reminder1 = coeffs_by_2.remainder();
for k in coeffs_by_2 {
let mmk = simd_utils::ptr_i16_to_set1_epi32(k, 0);
let source = simd_utils::loadl_epi64(src_row, x + x_start);
let pix = _mm_shuffle_epi8(source, sh7);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
@@ -231,9 +242,9 @@ impl Sse4 {
x += 2
}
while x < x_size {
for &k in reminder1 {
let pix = simd_utils::mm_cvtepu8_epi32(src_row, x + x_start);
let mmk = _mm_set1_epi32(*k.get_unchecked(x) as i32);
let mmk = _mm_set1_epi32(k as i32);
sss = _mm_add_epi32(sss, _mm_madd_epi16(pix, mmk));
x += 1;
@@ -247,7 +258,7 @@ impl Sse4 {
constify_imm8!(precision, call);
sss = _mm_packs_epi32(sss, sss);
*dst_row.get_unchecked_mut(xx) =
*dst_row.get_unchecked_mut(dst_x) =
transmute(_mm_cvtsi128_si32(_mm_packus_epi16(sss, sss)));
}
}
@@ -485,23 +496,17 @@ impl Convolution for Sse4 {
let (values, window_size, bounds_per_pixel) =
(coeffs.values, coeffs.window_size, coeffs.bounds);
let mut normalizer_guard = optimisations::NormalizerGuard::new(values);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coeffs_i16 = normalizer_guard.normalized();
let coefficients_chunks =
normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
let dst_height = dst_image.height().get();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_image.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
self.horiz_convolution_8u4x(
src_rows,
dst_rows,
coeffs_i16,
window_size,
&bounds_per_pixel,
precision,
);
self.horiz_convolution_8u4x(src_rows, dst_rows, &coefficients_chunks, precision);
}
}
@@ -511,9 +516,7 @@ impl Convolution for Sse4 {
self.horiz_convolution_8u(
src_image.get_row(yy + offset).unwrap(),
dst_image.get_row_mut(yy).unwrap(),
coeffs_i16,
window_size,
&bounds_per_pixel,
&coefficients_chunks,
precision,
);
}
@@ -530,7 +533,7 @@ impl Convolution for Sse4 {
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let mut normalizer_guard = optimisations::NormalizerGuard::new(values);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coeffs_i16 = normalizer_guard.normalized();
let coeffs_chunks = coeffs_i16.chunks(window_size);
+12 -2
View File
@@ -1,9 +1,19 @@
use thiserror::Error;
#[derive(Error, Debug, Clone, Copy)]
pub enum ImageError {
#[error("Buffer size don't corresponds to image dimensions")]
pub enum ImageRowsError {
#[error("Count of rows don't match to image height")]
InvalidRowsCount,
#[error("Size of row don't match to image width")]
InvalidRowSize,
}
#[derive(Error, Debug, Clone, Copy)]
pub enum ImageBufferError {
#[error("Size of buffer don't match to image dimensions")]
InvalidBufferSize,
#[error("Alignment of buffer don't match to alignment of u32")]
InvalidBufferAlignment,
}
#[derive(Error, Debug, Clone, Copy)]
+14 -14
View File
@@ -1,25 +1,25 @@
use std::num::NonZeroU32;
use crate::{DstImageView, ImageError, PixelType, SrcImageView};
use crate::{DstImageView, ImageBufferError, PixelType, SrcImageView};
#[derive(Debug, Clone)]
pub struct ImageData<T: AsRef<[u8]>> {
pub struct ImageData<T: AsRef<[u32]>> {
width: NonZeroU32,
height: NonZeroU32,
pixels: T,
pixel_type: PixelType,
}
impl<T: AsRef<[u8]>> ImageData<T> {
impl<T: AsRef<[u32]>> ImageData<T> {
pub fn new(
width: NonZeroU32,
height: NonZeroU32,
pixels: T,
pixel_type: PixelType,
) -> Result<Self, ImageError> {
let size = (width.get() * height.get()) as usize * 4;
) -> Result<Self, ImageBufferError> {
let size = (width.get() * height.get()) as usize;
if pixels.as_ref().len() != size {
return Err(ImageError::InvalidBufferSize);
return Err(ImageBufferError::InvalidBufferSize);
}
Ok(Self {
width,
@@ -45,30 +45,30 @@ impl<T: AsRef<[u8]>> ImageData<T> {
}
#[inline(always)]
pub fn get_buffer(&self) -> &[u8] {
pub fn get_buffer(&self) -> &[u32] {
self.pixels.as_ref()
}
#[inline(always)]
pub fn src_view(&self) -> SrcImageView {
let pixels = unsafe { self.pixels.as_ref().align_to::<u32>().1 };
let pixels = self.pixels.as_ref();
let rows = pixels.chunks(self.width.get() as usize).collect();
SrcImageView::new(self.width, self.height, rows, self.pixel_type).unwrap()
SrcImageView::from_rows(self.width, self.height, rows, self.pixel_type).unwrap()
}
}
impl<T: AsRef<[u8]> + AsMut<[u8]>> ImageData<T> {
impl<T: AsRef<[u32]> + AsMut<[u32]>> ImageData<T> {
#[inline(always)]
pub fn dst_view(&mut self) -> DstImageView {
let pixels = unsafe { self.pixels.as_mut().align_to_mut::<u32>().1 };
let pixels = self.pixels.as_mut();
let rows = pixels.chunks_mut(self.width.get() as usize).collect();
DstImageView::new(self.width, self.height, rows, self.pixel_type).unwrap()
DstImageView::from_rows(self.width, self.height, rows, self.pixel_type).unwrap()
}
}
impl ImageData<Vec<u8>> {
impl ImageData<Vec<u32>> {
pub fn new_owned(width: NonZeroU32, height: NonZeroU32, pixel_type: PixelType) -> Self {
let size = (width.get() * height.get()) as usize * 4;
let size = (width.get() * height.get()) as usize;
let pixels = vec![0; size];
Self {
width,
+28 -39
View File
@@ -2,7 +2,7 @@ use std::mem::transmute;
use std::num::NonZeroU32;
use std::slice;
use crate::errors::{CropBoxError, ImageError};
use crate::errors::{CropBoxError, ImageBufferError, ImageRowsError};
pub type TwoRows<'a> = (&'a [u32], &'a [u32]);
pub type FourRows<'a> = (&'a [u32], &'a [u32], &'a [u32], &'a [u32]);
@@ -49,19 +49,18 @@ pub struct DstImageView<'a> {
}
impl<'a> SrcImageView<'a> {
#[inline(always)]
pub fn new(
pub fn from_rows(
width: NonZeroU32,
height: NonZeroU32,
rows: Vec<&'a [u32]>,
pixel_type: PixelType,
) -> Result<Self, ImageError> {
) -> Result<Self, ImageRowsError> {
if rows.len() != height.get() as usize {
return Err(ImageError::InvalidBufferSize);
return Err(ImageRowsError::InvalidRowsCount);
}
let row_size = width.get() as usize;
if rows.iter().any(|row| row.len() != row_size) {
return Err(ImageError::InvalidBufferSize);
return Err(ImageRowsError::InvalidRowSize);
}
Ok(Self {
width,
@@ -77,6 +76,25 @@ impl<'a> SrcImageView<'a> {
})
}
pub fn from_buffer(
width: NonZeroU32,
height: NonZeroU32,
buffer: &'a [u8],
pixel_type: PixelType,
) -> Result<Self, ImageBufferError> {
let (head, pixels, _) = unsafe { buffer.align_to::<u32>() };
if !head.is_empty() {
return Err(ImageBufferError::InvalidBufferAlignment);
}
let size = (width.get() * height.get()) as usize;
if pixels.len() != size {
return Err(ImageBufferError::InvalidBufferSize);
}
let rows = pixels.chunks(width.get() as usize).collect();
Ok(Self::from_rows(width, height, rows, pixel_type).unwrap())
}
#[inline(always)]
pub fn pixel_type(&self) -> PixelType {
self.pixel_type
@@ -224,18 +242,18 @@ impl<'a> SrcImageView<'a> {
impl<'a> DstImageView<'a> {
#[inline(always)]
pub fn new(
pub fn from_rows(
width: NonZeroU32,
height: NonZeroU32,
rows: Vec<&'a mut [u32]>,
pixel_type: PixelType,
) -> Result<Self, ImageError> {
) -> Result<Self, ImageRowsError> {
if rows.len() != height.get() as usize {
return Err(ImageError::InvalidBufferSize);
return Err(ImageRowsError::InvalidRowsCount);
}
let row_size = width.get() as usize;
if rows.iter().any(|row| row.len() != row_size) {
return Err(ImageError::InvalidBufferSize);
return Err(ImageRowsError::InvalidRowSize);
}
Ok(Self {
width,
@@ -278,32 +296,3 @@ impl<'a> DstImageView<'a> {
self.rows.get_mut(y as usize)
}
}
// pub struct FourRowsIterator<'a> {
// chunks: slice::ChunksExact<'a, &'a [u32]>,
// }
//
// impl<'a> FourRowsIterator<'a> {
// #[inline(always)]
// fn new(image: &'a SrcImageView<'a>, start_y: u32, max_y: u32) -> Self {
// let start_y = start_y as usize;
// let max_y = max_y as usize;
// let max_y = max_y.min(image.height.get() as usize);
// let rows = unsafe { image.rows.get_unchecked(start_y..max_y) };
// Self {
// chunks: rows.chunks_exact(4),
// }
// }
// }
//
// impl<'a> Iterator for FourRowsIterator<'a> {
// type Item = FourRows<'a>;
//
// #[inline(always)]
// fn next(&mut self) -> Option<Self::Item> {
// match self.chunks.next() {
// Some(&[r0, r1, r2, r3]) => Some((r0, r1, r2, r3)),
// _ => None,
// }
// }
// }
+1 -1
View File
@@ -1,6 +1,6 @@
pub use alpha::{MulDiv, MulDivImageError, MulDivImagesError};
pub use convolution::FilterType;
pub use errors::{CropBoxError, ImageError};
pub use errors::{CropBoxError, ImageBufferError, ImageRowsError};
pub use image_data::ImageData;
pub use image_view::{CropBox, DstImageView, PixelType, SrcImageView};
pub use resizer::{CpuExtensions, ResizeAlg, Resizer};
+26 -3
View File
@@ -1,3 +1,4 @@
use crate::convolution::{Bound, CoefficientsChunk};
use std::slice;
// This code is based on C-implementation from Pillow-SIMD package for Python
@@ -118,10 +119,32 @@ impl NormalizerGuard {
}
#[inline]
pub fn normalized(&mut self) -> &[i16] {
pub fn normalized(&self) -> &[i16] {
let len = self.values.len();
let ptr = self.values.as_mut_ptr();
unsafe { slice::from_raw_parts_mut(ptr as *mut i16, len) }
let ptr = self.values.as_ptr();
unsafe { slice::from_raw_parts(ptr as *const i16, len) }
}
#[inline]
pub fn normalized_chunks(
&self,
window_size: usize,
bounds: &[Bound],
) -> Vec<CoefficientsChunk> {
let len = self.values.len();
let ptr = self.values.as_ptr();
let mut cooefs = unsafe { slice::from_raw_parts(ptr as *const i16, len) };
let mut res = Vec::with_capacity(bounds.len());
for bound in bounds {
let (left, right) = cooefs.split_at(window_size);
cooefs = right;
let size = bound.size as usize;
res.push(CoefficientsChunk {
start: bound.start,
values: &left[0..size],
});
}
res
}
#[inline]
+10 -9
View File
@@ -60,8 +60,8 @@ impl Default for ResizeAlg {
pub struct Resizer {
pub algorithm: ResizeAlg,
cpu_extensions: CpuExtensions,
convolution_buffer: Vec<u8>,
super_sampling_buffer: Vec<u8>,
convolution_buffer: Vec<u32>,
super_sampling_buffer: Vec<u32>,
}
impl Resizer {
@@ -114,7 +114,8 @@ impl Resizer {
/// Returns the size of internal buffers used to store the results of
/// intermediate resizing steps.
pub fn size_of_internal_buffers(&self) -> usize {
self.convolution_buffer.len() + self.super_sampling_buffer.len()
(self.convolution_buffer.capacity() + self.super_sampling_buffer.capacity())
* std::mem::size_of::<u32>()
}
/// Deallocates the internal buffers used to store the results of
@@ -142,12 +143,12 @@ impl Resizer {
}
fn get_temp_image_from_buffer(
buffer: &mut Vec<u8>,
buffer: &mut Vec<u32>,
width: NonZeroU32,
height: NonZeroU32,
pixel_type: PixelType,
) -> ImageData<&mut [u8]> {
let buf_size = (width.get() * height.get()) as usize * 4;
) -> ImageData<&mut [u32]> {
let buf_size = (width.get() * height.get()) as usize;
if buffer.len() < buf_size {
buffer.resize(buf_size, 0);
}
@@ -186,7 +187,7 @@ fn resample_convolution(
dst_image: &mut DstImageView,
filter_type: FilterType,
cpu_extensions: CpuExtensions,
temp_buffer: &mut Vec<u8>,
temp_buffer: &mut Vec<u32>,
) {
let crop_box = src_image.crop_box();
let dst_width = dst_image.width();
@@ -258,8 +259,8 @@ fn resample_super_sampling(
filter_type: FilterType,
multiplicity: u8,
cpu_extensions: CpuExtensions,
temp_buffer: &mut Vec<u8>,
convolution_temp_buffer: &mut Vec<u8>,
temp_buffer: &mut Vec<u32>,
convolution_temp_buffer: &mut Vec<u32>,
) {
let crop_box = src_image.crop_box();
let dst_width = dst_image.width().get();
+4 -4
View File
@@ -22,7 +22,7 @@ fn multiply_alpha_test(cpu_extensions: CpuExtensions) {
];
let rows: Vec<&[u32]> = src_rows.iter().map(|r| r.as_ref()).collect();
let src_image_view = SrcImageView::new(
let src_image_view = SrcImageView::from_rows(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
rows,
@@ -60,7 +60,7 @@ fn multiply_alpha_test(cpu_extensions: CpuExtensions) {
// Inplace
let rows: Vec<&mut [u32]> = src_rows.iter_mut().map(|r| r.as_mut()).collect();
let mut image_view = DstImageView::new(
let mut image_view = DstImageView::from_rows(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
rows,
@@ -109,7 +109,7 @@ fn divide_alpha_test(cpu_extensions: CpuExtensions) {
];
let rows: Vec<&[u32]> = src_rows.iter().map(|r| r.as_ref()).collect();
let src_image_view = SrcImageView::new(
let src_image_view = SrcImageView::from_rows(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
rows,
@@ -147,7 +147,7 @@ fn divide_alpha_test(cpu_extensions: CpuExtensions) {
// Inplace
let rows: Vec<&mut [u32]> = src_rows.iter_mut().map(|r| r.as_mut()).collect();
let mut image_view = DstImageView::new(
let mut image_view = DstImageView::from_rows(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
rows,
+13 -5
View File
@@ -8,7 +8,7 @@ use image::{ColorType, GenericImageView};
use fast_image_resize::ImageData;
use fast_image_resize::{CpuExtensions, FilterType, PixelType, ResizeAlg, Resizer, SrcImageView};
fn get_source_image() -> ImageData<Vec<u8>> {
fn get_source_image() -> ImageData<Vec<u32>> {
let img = ImageReader::open("./data/nasa-4928x3279.png")
.unwrap()
.decode()
@@ -16,7 +16,11 @@ fn get_source_image() -> ImageData<Vec<u8>> {
let width = img.width();
let height = img.height();
let rgb = img.to_rgba8();
let buf = rgb.as_raw().clone();
let buf = rgb
.as_raw()
.chunks_exact(4)
.map(|p| u32::from_le_bytes([p[0], p[1], p[2], p[3]]))
.collect();
ImageData::new(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
@@ -26,7 +30,7 @@ fn get_source_image() -> ImageData<Vec<u8>> {
.unwrap()
}
fn get_small_source_image() -> ImageData<Vec<u8>> {
fn get_small_source_image() -> ImageData<Vec<u32>> {
let img = ImageReader::open("./data/nasa-852x567.png")
.unwrap()
.decode()
@@ -34,7 +38,11 @@ fn get_small_source_image() -> ImageData<Vec<u8>> {
let width = img.width();
let height = img.height();
let rgb = img.to_rgba8();
let buf = rgb.as_raw().clone();
let buf = rgb
.as_raw()
.chunks_exact(4)
.map(|p| u32::from_le_bytes([p[0], p[1], p[2], p[3]]))
.collect();
ImageData::new(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
@@ -100,7 +108,7 @@ fn resample_sse4_lanczos3_test() {
save_result(&result.src_view(), "lanczos3_sse4");
}
fn resize_lanczos3(src_pixels: &[u8], width: NonZeroU32, height: NonZeroU32) -> Vec<u8> {
fn resize_lanczos3(src_pixels: &[u32], width: NonZeroU32, height: NonZeroU32) -> Vec<u32> {
let src_image = ImageData::new(width, height, src_pixels, PixelType::U8x4).unwrap();
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
let dst_width = NonZeroU32::new(1024).unwrap();