Added benchmark for resizing of I32 images.

This commit is contained in:
Kirill Kuzminykh
2021-10-09 13:06:34 +03:00
parent ffb12f39d9
commit 52acf313c2
15 changed files with 341 additions and 270 deletions
+3 -3
View File
@@ -12,7 +12,7 @@ use fast_image_resize::{ImageData, MulDiv};
mod utils;
pub fn bench_downscale_rgba(bench: &mut Bench) {
let src_image = utils::get_big_rgba_image();
let src_image = &utils::get_big_rgba_image();
let new_width = NonZeroU32::new(852).unwrap();
let new_height = NonZeroU32::new(567).unwrap();
@@ -30,7 +30,7 @@ pub fn bench_downscale_rgba(bench: &mut Bench) {
};
bench.task(format!("image - {}", alg_name), |task| {
task.iter(|| {
imageops::resize(&src_image, new_width.get(), new_height.get(), filter);
imageops::resize(src_image, new_width.get(), new_height.get(), filter);
})
});
}
@@ -86,7 +86,7 @@ pub fn bench_downscale_rgba(bench: &mut Bench) {
let src_image_data = ImageData::from_vec_u8(
NonZeroU32::new(src_image.width()).unwrap(),
NonZeroU32::new(src_image.height()).unwrap(),
src_image.into_raw(),
src_image.as_raw().clone(),
PixelType::U8x4,
)
.unwrap();
+39
View File
@@ -26,6 +26,24 @@ fn get_big_source_image() -> ImageData<'static> {
.unwrap()
}
fn get_big_i32_image() -> ImageData<'static> {
let img = utils::get_big_luma_image();
let img_data: Vec<u32> = img
.as_raw()
.iter()
.map(|&p| p as u32 * (i16::MAX as u32 + 1))
.collect();
let width = img.width();
let height = img.height();
ImageData::from_vec_u32(
NonZeroU32::new(width).unwrap(),
NonZeroU32::new(height).unwrap(),
img_data,
PixelType::I32,
)
.unwrap()
}
fn get_small_source_image() -> ImageData<'static> {
let img = utils::get_small_rgba_image();
let width = img.width();
@@ -159,6 +177,26 @@ fn avx2_lanczos3_upscale_bench(bench: &mut Bench) {
});
}
fn native_lanczos3_i32_bench(bench: &mut Bench) {
let image = get_big_i32_image();
let mut res_image = ImageData::new(
NonZeroU32::new(NEW_WIDTH).unwrap(),
NonZeroU32::new(NEW_HEIGHT).unwrap(),
image.pixel_type(),
);
let src_image = image.src_view();
let mut dst_image = res_image.dst_view();
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
unsafe {
resizer.set_cpu_extensions(CpuExtensions::None);
}
bench.task("i32 lanczos3 wo SIMD", |task| {
task.iter(|| {
resizer.resize(&src_image, &mut dst_image);
})
});
}
glassbench!(
"Resize",
nearest_wo_simd_bench,
@@ -167,4 +205,5 @@ glassbench!(
avx2_lanczos3_bench,
avx2_supersampling_lanczos3_bench,
avx2_lanczos3_upscale_bench,
native_lanczos3_i32_bench,
);
+10 -1
View File
@@ -3,7 +3,7 @@ use std::env;
use glassbench::*;
use image::io::Reader;
use image::{RgbImage, RgbaImage};
use image::{ImageBuffer, Luma, RgbImage, RgbaImage};
pub fn get_big_rgb_image() -> RgbImage {
let cur_dir = env::current_dir().unwrap();
@@ -23,6 +23,15 @@ pub fn get_big_rgba_image() -> RgbaImage {
img.to_rgba8()
}
pub fn get_big_luma_image() -> ImageBuffer<Luma<u16>, Vec<u16>> {
let cur_dir = env::current_dir().unwrap();
let img = Reader::open(cur_dir.join("data/nasa-4928x3279.png"))
.unwrap()
.decode()
.unwrap();
img.to_luma16()
}
pub fn get_small_rgba_image() -> RgbaImage {
let cur_dir = env::current_dir().unwrap();
let img = Reader::open(cur_dir.join("data/nasa-852x567.png"))
+3
View File
@@ -0,0 +1,3 @@
pub use u8x4::Avx2U8x4;
mod u8x4;
@@ -1,9 +1,10 @@
use std::arch::x86_64::*;
use std::intrinsics::transmute;
use crate::convolution::{Bound, Coefficients, CoefficientsChunk, Convolution};
use crate::convolution::optimisations::CoefficientsI16Chunk;
use crate::convolution::{optimisations, Bound, Coefficients, Convolution};
use crate::image_view::{DstImageView, FourRows, FourRowsMut, SrcImageView};
use crate::{optimisations, simd_utils};
use crate::simd_utils;
pub struct Avx2U8x4;
@@ -22,7 +23,7 @@ impl Avx2U8x4 {
&self,
src_rows: FourRows,
dst_rows: FourRowsMut,
coefficients_chunks: &[CoefficientsChunk],
coefficients_chunks: &[CoefficientsI16Chunk],
precision: u8,
) {
let (s_row0, s_row1, s_row2, s_row3) = src_rows;
@@ -31,12 +32,12 @@ impl Avx2U8x4 {
let initial = _mm256_set1_epi32(1 << (precision - 1));
#[rustfmt::skip]
let sh1 = _mm256_set_epi8(
let sh1 = _mm256_set_epi8(
-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0,
-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0,
);
#[rustfmt::skip]
let sh2 = _mm256_set_epi8(
let sh2 = _mm256_set_epi8(
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
);
@@ -153,7 +154,7 @@ impl Avx2U8x4 {
&self,
src_row: &[u32],
dst_row: &mut [u32],
coefficients_chunks: &[CoefficientsChunk],
coefficients_chunks: &[CoefficientsI16Chunk],
precision: u8,
) {
#[rustfmt::skip]
@@ -471,7 +472,7 @@ impl Convolution for Avx2U8x4 {
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coefficients_chunks =
normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
normalizer_guard.normalized_i16_chunks(window_size, &bounds_per_pixel);
let dst_height = dst_image.height().get();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
@@ -507,7 +508,7 @@ impl Convolution for Avx2U8x4 {
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coeffs_i16 = normalizer_guard.normalized();
let coeffs_i16 = normalizer_guard.normalized_i16();
let coeffs_chunks = coeffs_i16.chunks(window_size);
let dst_rows = dst_image.iter_rows_mut();
+24 -6
View File
@@ -13,6 +13,7 @@ mod macros;
mod avx2;
mod filters;
mod native;
mod optimisations;
mod sse4;
pub trait Convolution {
@@ -38,12 +39,6 @@ pub struct Bound {
pub size: u32,
}
#[derive(Debug, Clone, Copy)]
pub struct CoefficientsChunk<'a> {
pub start: u32,
pub values: &'a [i16],
}
#[derive(Debug, Clone)]
pub struct Coefficients {
pub values: Vec<f64>,
@@ -51,6 +46,29 @@ pub struct Coefficients {
pub bounds: Vec<Bound>,
}
#[derive(Debug, Clone, Copy)]
pub struct CoefficientsChunk<'a> {
pub start: u32,
pub values: &'a [f64],
}
impl Coefficients {
pub fn get_chunks(&self) -> Vec<CoefficientsChunk> {
let mut coeffs = self.values.as_slice();
let mut res = Vec::with_capacity(self.bounds.len());
for bound in &self.bounds {
let (left, right) = coeffs.split_at(self.window_size);
coeffs = right;
let size = bound.size as usize;
res.push(CoefficientsChunk {
start: bound.start,
values: &left[0..size],
});
}
res
}
}
pub fn precompute_coefficients(
in_size: NonZeroU32,
in0: f64, // Left border for cropping
-240
View File
@@ -1,240 +0,0 @@
use std::slice;
use crate::convolution::{Coefficients, Convolution};
use crate::image_view::{DstImageView, SrcImageView};
use crate::optimisations;
pub struct NativeU8x4;
impl Convolution for NativeU8x4 {
fn horiz_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
offset: u32,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
let dst_rows = dst_image.iter_rows_mut();
for (y_dst, dst_row) in dst_rows.enumerate() {
let y_src = y_dst as u32 + offset;
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
let first_x_src = coeffs_chunk.start;
let ks = coeffs_chunk.values;
let mut ss0 = 1 << (precision - 1);
let mut ss1 = ss0;
let mut ss2 = ss0;
let mut ss3 = ss0;
let src_pixels = src_image.iter_horiz(first_x_src, y_src);
for (&k, &src_pixel) in ks.iter().zip(src_pixels) {
let components: [u8; 4] = src_pixel.to_le_bytes();
ss0 += components[0] as i32 * (k as i32);
ss1 += components[1] as i32 * (k as i32);
ss2 += components[2] as i32 * (k as i32);
ss3 += components[3] as i32 * (k as i32);
}
let t: [u8; 4] = unsafe {
[
optimisations::clip8(ss0, precision),
optimisations::clip8(ss1, precision),
optimisations::clip8(ss2, precision),
optimisations::clip8(ss3, precision),
]
};
*dst_pixel = u32::from_le_bytes(t);
}
}
}
fn vert_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
let dst_rows = dst_image.iter_rows_mut();
for (&coeffs_chunk, dst_row) in coefficients_chunks.iter().zip(dst_rows) {
let first_y_src = coeffs_chunk.start;
let ks = coeffs_chunk.values;
for (x_src, out_pixel) in dst_row.iter_mut().enumerate() {
let mut ss0 = 1 << (precision - 1);
let mut ss1 = ss0;
let mut ss2 = ss0;
let mut ss3 = ss0;
for (dy, &k) in ks.iter().enumerate() {
let pixel = src_image.get_pixel_u32(x_src as u32, first_y_src + dy as u32);
let components: [u8; 4] = pixel.to_le_bytes();
ss0 += components[0] as i32 * (k as i32);
ss1 += components[1] as i32 * (k as i32);
ss2 += components[2] as i32 * (k as i32);
ss3 += components[3] as i32 * (k as i32);
}
let t: [u8; 4] = unsafe {
[
optimisations::clip8(ss0, precision),
optimisations::clip8(ss1, precision),
optimisations::clip8(ss2, precision),
optimisations::clip8(ss3, precision),
]
};
*out_pixel = u32::from_le_bytes(t);
}
}
}
}
pub struct NativeI32;
impl Convolution for NativeI32 {
fn horiz_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
offset: u32,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
for y_dst in 0..dst_image.height().get() {
let y_src = y_dst + offset;
if let Some(out_row) = dst_image.get_row_mut(y_dst) {
let out_row_i32 = unsafe {
let len = out_row.len();
let ptr = out_row.as_mut_ptr();
slice::from_raw_parts_mut(ptr as *mut i32, len)
};
for (x_dst, (&bound, out_pixel)) in bounds.iter().zip(out_row_i32).enumerate() {
let first_x_src = bound.start;
let start_index = window_size * x_dst;
let end_index = start_index + bound.size as usize;
let ks = &values[start_index..end_index];
let mut ss = 0.;
let pixels = src_image.iter_horiz_i32(first_x_src, y_src);
for (&k, &pixel) in ks.iter().zip(pixels) {
ss += pixel as f64 * k;
}
*out_pixel = ss.round() as i32;
}
}
}
}
fn vert_convolution(
&self,
image: &SrcImageView,
out_image: &mut DstImageView,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
for (y_dst, &bound) in bounds.iter().enumerate() {
let first_y_src = bound.start;
let start_index = window_size * y_dst;
let end_index = start_index + bound.size as usize;
let ks = &values[start_index..end_index];
if let Some(out_row) = out_image.get_row_mut(y_dst as u32) {
let out_row_i32 = unsafe {
let len = out_row.len();
let ptr = out_row.as_mut_ptr();
slice::from_raw_parts_mut(ptr as *mut i32, len)
};
for (x_src, out_pixel) in out_row_i32.iter_mut().enumerate() {
let mut ss = 0.;
for (dy, &k) in ks.iter().enumerate() {
let pixel = image.get_pixel_i32(x_src as u32, first_y_src + dy as u32);
ss += pixel as f64 * k;
}
*out_pixel = ss.round() as i32;
}
}
}
}
}
pub struct NativeF32;
impl Convolution for NativeF32 {
fn horiz_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
offset: u32,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
for y_dst in 0..dst_image.height().get() {
let y_src = y_dst + offset;
if let Some(out_row) = dst_image.get_row_mut(y_dst) {
let out_row_f32 = unsafe {
let len = out_row.len();
let ptr = out_row.as_mut_ptr();
slice::from_raw_parts_mut(ptr as *mut f32, len)
};
for (x_dst, (&bound, out_pixel)) in bounds.iter().zip(out_row_f32).enumerate() {
let first_x_src = bound.start;
let start_index = window_size * x_dst;
let end_index = start_index + bound.size as usize;
let ks = &values[start_index..end_index];
let mut ss = 0.;
let pixels = src_image.iter_horiz_f32(first_x_src, y_src);
for (&k, &pixel) in ks.iter().zip(pixels) {
ss += pixel as f64 * k;
}
*out_pixel = ss as f32;
}
}
}
}
fn vert_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
for (y_dst, &bound) in bounds.iter().enumerate() {
let first_y_src = bound.start;
let start_index = window_size * y_dst;
let end_index = start_index + bound.size as usize;
let ks = &values[start_index..end_index];
if let Some(out_row) = dst_image.get_row_mut(y_dst as u32) {
let out_row_f32 = unsafe {
let len = out_row.len();
let ptr = out_row.as_mut_ptr();
slice::from_raw_parts_mut(ptr as *mut f32, len)
};
for (x_src, out_pixel) in out_row_f32.iter_mut().enumerate() {
let mut ss = 0.;
for (dy, &k) in ks.iter().enumerate() {
let pixel = src_image.get_pixel_f32(x_src as u32, first_y_src + dy as u32);
ss += pixel as f64 * k;
}
*out_pixel = ss as f32;
}
}
}
}
}
+75
View File
@@ -0,0 +1,75 @@
use std::slice;
use crate::convolution::{Coefficients, Convolution};
use crate::{DstImageView, SrcImageView};
pub struct NativeF32;
impl Convolution for NativeF32 {
fn horiz_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
offset: u32,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
for y_dst in 0..dst_image.height().get() {
let y_src = y_dst + offset;
if let Some(out_row) = dst_image.get_row_mut(y_dst) {
let out_row_f32 = unsafe {
let len = out_row.len();
let ptr = out_row.as_mut_ptr();
slice::from_raw_parts_mut(ptr as *mut f32, len)
};
for (x_dst, (&bound, out_pixel)) in bounds.iter().zip(out_row_f32).enumerate() {
let first_x_src = bound.start;
let start_index = window_size * x_dst;
let end_index = start_index + bound.size as usize;
let ks = &values[start_index..end_index];
let mut ss = 0.;
let pixels = src_image.iter_horiz_f32(first_x_src, y_src);
for (&k, &pixel) in ks.iter().zip(pixels) {
ss += pixel as f64 * k;
}
*out_pixel = ss as f32;
}
}
}
}
fn vert_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
for (y_dst, &bound) in bounds.iter().enumerate() {
let first_y_src = bound.start;
let start_index = window_size * y_dst;
let end_index = start_index + bound.size as usize;
let ks = &values[start_index..end_index];
if let Some(out_row) = dst_image.get_row_mut(y_dst as u32) {
let out_row_f32 = unsafe {
let len = out_row.len();
let ptr = out_row.as_mut_ptr();
slice::from_raw_parts_mut(ptr as *mut f32, len)
};
for (x_src, out_pixel) in out_row_f32.iter_mut().enumerate() {
let mut ss = 0.;
for (dy, &k) in ks.iter().enumerate() {
let pixel = src_image.get_pixel_f32(x_src as u32, first_y_src + dy as u32);
ss += pixel as f64 * k;
}
*out_pixel = ss as f32;
}
}
}
}
}
+53
View File
@@ -0,0 +1,53 @@
use crate::convolution::{Coefficients, Convolution};
use crate::{DstImageView, SrcImageView};
pub struct NativeI32;
impl Convolution for NativeI32 {
fn horiz_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
offset: u32,
coeffs: Coefficients,
) {
let coefficients_chunks = coeffs.get_chunks();
let mut y_src = offset;
for out_row in dst_image.iter_rows_mut() {
for (out_pixel, coeffs_chunk) in out_row.iter_mut().zip(&coefficients_chunks) {
let first_x_src = coeffs_chunk.start;
let mut ss = 0.;
let pixels = src_image.iter_horiz_i32(first_x_src, y_src);
for (&k, &pixel) in coeffs_chunk.values.iter().zip(pixels) {
ss += pixel as f64 * k;
}
*out_pixel = ss.round() as i32 as u32;
}
y_src += 1;
}
}
fn vert_convolution(
&self,
image: &SrcImageView,
out_image: &mut DstImageView,
coeffs: Coefficients,
) {
let coefficients_chunks = coeffs.get_chunks();
for (out_row, coeffs_chunk) in out_image.iter_rows_mut().zip(coefficients_chunks) {
let first_y_src = coeffs_chunk.start;
for (x_src, out_pixel) in out_row.iter_mut().enumerate() {
let mut ss = 0.;
let mut y_src = first_y_src;
for &k in coeffs_chunk.values.iter() {
let pixel = image.get_pixel_i32(x_src as u32, y_src);
ss += pixel as f64 * k;
y_src += 1;
}
*out_pixel = ss.round() as i32 as u32;
}
}
}
}
+7
View File
@@ -0,0 +1,7 @@
pub use f32x1::NativeF32;
pub use i32x1::NativeI32;
pub use u8x4::NativeU8x4;
mod f32x1;
mod i32x1;
mod u8x4;
+95
View File
@@ -0,0 +1,95 @@
use crate::convolution::{optimisations, Coefficients, Convolution};
use crate::{DstImageView, SrcImageView};
pub struct NativeU8x4;
impl Convolution for NativeU8x4 {
fn horiz_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
offset: u32,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coefficients_chunks = normalizer_guard.normalized_i16_chunks(window_size, &bounds);
let dst_rows = dst_image.iter_rows_mut();
for (y_dst, dst_row) in dst_rows.enumerate() {
let y_src = y_dst as u32 + offset;
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
let first_x_src = coeffs_chunk.start;
let ks = coeffs_chunk.values;
let mut ss0 = 1 << (precision - 1);
let mut ss1 = ss0;
let mut ss2 = ss0;
let mut ss3 = ss0;
let src_pixels = src_image.iter_horiz(first_x_src, y_src);
for (&k, &src_pixel) in ks.iter().zip(src_pixels) {
let components: [u8; 4] = src_pixel.to_le_bytes();
ss0 += components[0] as i32 * (k as i32);
ss1 += components[1] as i32 * (k as i32);
ss2 += components[2] as i32 * (k as i32);
ss3 += components[3] as i32 * (k as i32);
}
let t: [u8; 4] = unsafe {
[
optimisations::clip8(ss0, precision),
optimisations::clip8(ss1, precision),
optimisations::clip8(ss2, precision),
optimisations::clip8(ss3, precision),
]
};
*dst_pixel = u32::from_le_bytes(t);
}
}
}
fn vert_convolution(
&self,
src_image: &SrcImageView,
dst_image: &mut DstImageView,
coeffs: Coefficients,
) {
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coefficients_chunks = normalizer_guard.normalized_i16_chunks(window_size, &bounds);
let dst_rows = dst_image.iter_rows_mut();
for (&coeffs_chunk, dst_row) in coefficients_chunks.iter().zip(dst_rows) {
let first_y_src = coeffs_chunk.start;
let ks = coeffs_chunk.values;
for (x_src, out_pixel) in dst_row.iter_mut().enumerate() {
let mut ss0 = 1 << (precision - 1);
let mut ss1 = ss0;
let mut ss2 = ss0;
let mut ss3 = ss0;
for (dy, &k) in ks.iter().enumerate() {
let pixel = src_image.get_pixel_u32(x_src as u32, first_y_src + dy as u32);
let components: [u8; 4] = pixel.to_le_bytes();
ss0 += components[0] as i32 * (k as i32);
ss1 += components[1] as i32 * (k as i32);
ss2 += components[2] as i32 * (k as i32);
ss3 += components[3] as i32 * (k as i32);
}
let t: [u8; 4] = unsafe {
[
optimisations::clip8(ss0, precision),
optimisations::clip8(ss1, precision),
optimisations::clip8(ss2, precision),
optimisations::clip8(ss3, precision),
]
};
*out_pixel = u32::from_le_bytes(t);
}
}
}
}
@@ -1,6 +1,7 @@
use crate::convolution::{Bound, CoefficientsChunk};
use std::slice;
use super::Bound;
// This code is based on C-implementation from Pillow-SIMD package for Python
// https://github.com/uploadcare/pillow-simd
@@ -87,6 +88,12 @@ pub struct NormalizerGuard {
precision: u8,
}
#[derive(Debug, Clone, Copy)]
pub struct CoefficientsI16Chunk<'a> {
pub start: u32,
pub values: &'a [i16],
}
impl NormalizerGuard {
#[inline]
pub fn new(mut values: Vec<f64>) -> Self {
@@ -119,18 +126,18 @@ impl NormalizerGuard {
}
#[inline]
pub fn normalized(&self) -> &[i16] {
pub fn normalized_i16(&self) -> &[i16] {
let len = self.values.len();
let ptr = self.values.as_ptr();
unsafe { slice::from_raw_parts(ptr as *const i16, len) }
}
#[inline]
pub fn normalized_chunks(
pub fn normalized_i16_chunks(
&self,
window_size: usize,
bounds: &[Bound],
) -> Vec<CoefficientsChunk> {
) -> Vec<CoefficientsI16Chunk> {
let len = self.values.len();
let ptr = self.values.as_ptr();
let mut cooefs = unsafe { slice::from_raw_parts(ptr as *const i16, len) };
@@ -139,7 +146,7 @@ impl NormalizerGuard {
let (left, right) = cooefs.split_at(window_size);
cooefs = right;
let size = bound.size as usize;
res.push(CoefficientsChunk {
res.push(CoefficientsI16Chunk {
start: bound.start,
values: &left[0..size],
});
+3
View File
@@ -0,0 +1,3 @@
pub use u8x4::Sse4U8x4;
mod u8x4;
@@ -1,9 +1,10 @@
use std::arch::x86_64::*;
use std::intrinsics::transmute;
use crate::convolution::{Bound, Coefficients, CoefficientsChunk, Convolution};
use crate::convolution::optimisations::CoefficientsI16Chunk;
use crate::convolution::{optimisations, Bound, Coefficients, Convolution};
use crate::image_view::{DstImageView, FourRows, FourRowsMut, SrcImageView};
use crate::{optimisations, simd_utils};
use crate::simd_utils;
pub struct Sse4U8x4;
@@ -21,7 +22,7 @@ impl Sse4U8x4 {
&self,
src_rows: FourRows,
dst_rows: FourRowsMut,
coefficients_chunks: &[CoefficientsChunk],
coefficients_chunks: &[CoefficientsI16Chunk],
precision: u8,
) {
let (s_row0, s_row1, s_row2, s_row3) = src_rows;
@@ -159,7 +160,7 @@ impl Sse4U8x4 {
&self,
src_row: &[u32],
dst_row: &mut [u32],
coefficients_chunks: &[CoefficientsChunk],
coefficients_chunks: &[CoefficientsI16Chunk],
precision: u8,
) {
let initial = _mm_set1_epi32(1 << (precision - 1));
@@ -497,7 +498,7 @@ impl Convolution for Sse4U8x4 {
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coefficients_chunks =
normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
normalizer_guard.normalized_i16_chunks(window_size, &bounds_per_pixel);
let dst_height = dst_image.height().get();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
@@ -533,7 +534,7 @@ impl Convolution for Sse4U8x4 {
let normalizer_guard = optimisations::NormalizerGuard::new(values);
let precision = normalizer_guard.precision();
let coeffs_i16 = normalizer_guard.normalized();
let coeffs_i16 = normalizer_guard.normalized_i16();
let coeffs_chunks = coeffs_i16.chunks(window_size);
let dst_rows = dst_image.iter_rows_mut();
+1 -1
View File
@@ -1,4 +1,5 @@
#![doc = include_str!("../README.md")]
pub use alpha::{MulDiv, MulDivImageError, MulDivImagesError};
pub use convolution::FilterType;
pub use errors::{CropBoxError, ImageBufferError, ImageRowsError, InvalidBufferSizeError};
@@ -11,6 +12,5 @@ mod convolution;
mod errors;
mod image_data;
mod image_view;
mod optimisations;
mod resizer;
mod simd_utils;