mirror of
https://github.com/Cykooz/fast_image_resize.git
synced 2026-10-08 01:11:09 +00:00
Added benchmark for resizing of I32 images.
This commit is contained in:
@@ -12,7 +12,7 @@ use fast_image_resize::{ImageData, MulDiv};
|
||||
mod utils;
|
||||
|
||||
pub fn bench_downscale_rgba(bench: &mut Bench) {
|
||||
let src_image = utils::get_big_rgba_image();
|
||||
let src_image = &utils::get_big_rgba_image();
|
||||
let new_width = NonZeroU32::new(852).unwrap();
|
||||
let new_height = NonZeroU32::new(567).unwrap();
|
||||
|
||||
@@ -30,7 +30,7 @@ pub fn bench_downscale_rgba(bench: &mut Bench) {
|
||||
};
|
||||
bench.task(format!("image - {}", alg_name), |task| {
|
||||
task.iter(|| {
|
||||
imageops::resize(&src_image, new_width.get(), new_height.get(), filter);
|
||||
imageops::resize(src_image, new_width.get(), new_height.get(), filter);
|
||||
})
|
||||
});
|
||||
}
|
||||
@@ -86,7 +86,7 @@ pub fn bench_downscale_rgba(bench: &mut Bench) {
|
||||
let src_image_data = ImageData::from_vec_u8(
|
||||
NonZeroU32::new(src_image.width()).unwrap(),
|
||||
NonZeroU32::new(src_image.height()).unwrap(),
|
||||
src_image.into_raw(),
|
||||
src_image.as_raw().clone(),
|
||||
PixelType::U8x4,
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
@@ -26,6 +26,24 @@ fn get_big_source_image() -> ImageData<'static> {
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
fn get_big_i32_image() -> ImageData<'static> {
|
||||
let img = utils::get_big_luma_image();
|
||||
let img_data: Vec<u32> = img
|
||||
.as_raw()
|
||||
.iter()
|
||||
.map(|&p| p as u32 * (i16::MAX as u32 + 1))
|
||||
.collect();
|
||||
let width = img.width();
|
||||
let height = img.height();
|
||||
ImageData::from_vec_u32(
|
||||
NonZeroU32::new(width).unwrap(),
|
||||
NonZeroU32::new(height).unwrap(),
|
||||
img_data,
|
||||
PixelType::I32,
|
||||
)
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
fn get_small_source_image() -> ImageData<'static> {
|
||||
let img = utils::get_small_rgba_image();
|
||||
let width = img.width();
|
||||
@@ -159,6 +177,26 @@ fn avx2_lanczos3_upscale_bench(bench: &mut Bench) {
|
||||
});
|
||||
}
|
||||
|
||||
fn native_lanczos3_i32_bench(bench: &mut Bench) {
|
||||
let image = get_big_i32_image();
|
||||
let mut res_image = ImageData::new(
|
||||
NonZeroU32::new(NEW_WIDTH).unwrap(),
|
||||
NonZeroU32::new(NEW_HEIGHT).unwrap(),
|
||||
image.pixel_type(),
|
||||
);
|
||||
let src_image = image.src_view();
|
||||
let mut dst_image = res_image.dst_view();
|
||||
let mut resizer = Resizer::new(ResizeAlg::Convolution(FilterType::Lanczos3));
|
||||
unsafe {
|
||||
resizer.set_cpu_extensions(CpuExtensions::None);
|
||||
}
|
||||
bench.task("i32 lanczos3 wo SIMD", |task| {
|
||||
task.iter(|| {
|
||||
resizer.resize(&src_image, &mut dst_image);
|
||||
})
|
||||
});
|
||||
}
|
||||
|
||||
glassbench!(
|
||||
"Resize",
|
||||
nearest_wo_simd_bench,
|
||||
@@ -167,4 +205,5 @@ glassbench!(
|
||||
avx2_lanczos3_bench,
|
||||
avx2_supersampling_lanczos3_bench,
|
||||
avx2_lanczos3_upscale_bench,
|
||||
native_lanczos3_i32_bench,
|
||||
);
|
||||
|
||||
+10
-1
@@ -3,7 +3,7 @@ use std::env;
|
||||
|
||||
use glassbench::*;
|
||||
use image::io::Reader;
|
||||
use image::{RgbImage, RgbaImage};
|
||||
use image::{ImageBuffer, Luma, RgbImage, RgbaImage};
|
||||
|
||||
pub fn get_big_rgb_image() -> RgbImage {
|
||||
let cur_dir = env::current_dir().unwrap();
|
||||
@@ -23,6 +23,15 @@ pub fn get_big_rgba_image() -> RgbaImage {
|
||||
img.to_rgba8()
|
||||
}
|
||||
|
||||
pub fn get_big_luma_image() -> ImageBuffer<Luma<u16>, Vec<u16>> {
|
||||
let cur_dir = env::current_dir().unwrap();
|
||||
let img = Reader::open(cur_dir.join("data/nasa-4928x3279.png"))
|
||||
.unwrap()
|
||||
.decode()
|
||||
.unwrap();
|
||||
img.to_luma16()
|
||||
}
|
||||
|
||||
pub fn get_small_rgba_image() -> RgbaImage {
|
||||
let cur_dir = env::current_dir().unwrap();
|
||||
let img = Reader::open(cur_dir.join("data/nasa-852x567.png"))
|
||||
|
||||
@@ -0,0 +1,3 @@
|
||||
pub use u8x4::Avx2U8x4;
|
||||
|
||||
mod u8x4;
|
||||
@@ -1,9 +1,10 @@
|
||||
use std::arch::x86_64::*;
|
||||
use std::intrinsics::transmute;
|
||||
|
||||
use crate::convolution::{Bound, Coefficients, CoefficientsChunk, Convolution};
|
||||
use crate::convolution::optimisations::CoefficientsI16Chunk;
|
||||
use crate::convolution::{optimisations, Bound, Coefficients, Convolution};
|
||||
use crate::image_view::{DstImageView, FourRows, FourRowsMut, SrcImageView};
|
||||
use crate::{optimisations, simd_utils};
|
||||
use crate::simd_utils;
|
||||
|
||||
pub struct Avx2U8x4;
|
||||
|
||||
@@ -22,7 +23,7 @@ impl Avx2U8x4 {
|
||||
&self,
|
||||
src_rows: FourRows,
|
||||
dst_rows: FourRowsMut,
|
||||
coefficients_chunks: &[CoefficientsChunk],
|
||||
coefficients_chunks: &[CoefficientsI16Chunk],
|
||||
precision: u8,
|
||||
) {
|
||||
let (s_row0, s_row1, s_row2, s_row3) = src_rows;
|
||||
@@ -31,12 +32,12 @@ impl Avx2U8x4 {
|
||||
let initial = _mm256_set1_epi32(1 << (precision - 1));
|
||||
|
||||
#[rustfmt::skip]
|
||||
let sh1 = _mm256_set_epi8(
|
||||
let sh1 = _mm256_set_epi8(
|
||||
-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0,
|
||||
-1, 7, -1, 3, -1, 6, -1, 2, -1, 5, -1, 1, -1, 4, -1, 0,
|
||||
);
|
||||
#[rustfmt::skip]
|
||||
let sh2 = _mm256_set_epi8(
|
||||
let sh2 = _mm256_set_epi8(
|
||||
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
|
||||
-1, 15, -1, 11, -1, 14, -1, 10, -1, 13, -1, 9, -1, 12, -1, 8,
|
||||
);
|
||||
@@ -153,7 +154,7 @@ impl Avx2U8x4 {
|
||||
&self,
|
||||
src_row: &[u32],
|
||||
dst_row: &mut [u32],
|
||||
coefficients_chunks: &[CoefficientsChunk],
|
||||
coefficients_chunks: &[CoefficientsI16Chunk],
|
||||
precision: u8,
|
||||
) {
|
||||
#[rustfmt::skip]
|
||||
@@ -471,7 +472,7 @@ impl Convolution for Avx2U8x4 {
|
||||
let normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks =
|
||||
normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
normalizer_guard.normalized_i16_chunks(window_size, &bounds_per_pixel);
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
@@ -507,7 +508,7 @@ impl Convolution for Avx2U8x4 {
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coeffs_i16 = normalizer_guard.normalized();
|
||||
let coeffs_i16 = normalizer_guard.normalized_i16();
|
||||
let coeffs_chunks = coeffs_i16.chunks(window_size);
|
||||
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
+24
-6
@@ -13,6 +13,7 @@ mod macros;
|
||||
mod avx2;
|
||||
mod filters;
|
||||
mod native;
|
||||
mod optimisations;
|
||||
mod sse4;
|
||||
|
||||
pub trait Convolution {
|
||||
@@ -38,12 +39,6 @@ pub struct Bound {
|
||||
pub size: u32,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
pub struct CoefficientsChunk<'a> {
|
||||
pub start: u32,
|
||||
pub values: &'a [i16],
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct Coefficients {
|
||||
pub values: Vec<f64>,
|
||||
@@ -51,6 +46,29 @@ pub struct Coefficients {
|
||||
pub bounds: Vec<Bound>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
pub struct CoefficientsChunk<'a> {
|
||||
pub start: u32,
|
||||
pub values: &'a [f64],
|
||||
}
|
||||
|
||||
impl Coefficients {
|
||||
pub fn get_chunks(&self) -> Vec<CoefficientsChunk> {
|
||||
let mut coeffs = self.values.as_slice();
|
||||
let mut res = Vec::with_capacity(self.bounds.len());
|
||||
for bound in &self.bounds {
|
||||
let (left, right) = coeffs.split_at(self.window_size);
|
||||
coeffs = right;
|
||||
let size = bound.size as usize;
|
||||
res.push(CoefficientsChunk {
|
||||
start: bound.start,
|
||||
values: &left[0..size],
|
||||
});
|
||||
}
|
||||
res
|
||||
}
|
||||
}
|
||||
|
||||
pub fn precompute_coefficients(
|
||||
in_size: NonZeroU32,
|
||||
in0: f64, // Left border for cropping
|
||||
|
||||
@@ -1,240 +0,0 @@
|
||||
use std::slice;
|
||||
|
||||
use crate::convolution::{Coefficients, Convolution};
|
||||
use crate::image_view::{DstImageView, SrcImageView};
|
||||
use crate::optimisations;
|
||||
|
||||
pub struct NativeU8x4;
|
||||
|
||||
impl Convolution for NativeU8x4 {
|
||||
fn horiz_convolution(
|
||||
&self,
|
||||
src_image: &SrcImageView,
|
||||
dst_image: &mut DstImageView,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
|
||||
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for (y_dst, dst_row) in dst_rows.enumerate() {
|
||||
let y_src = y_dst as u32 + offset;
|
||||
|
||||
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
|
||||
let first_x_src = coeffs_chunk.start;
|
||||
let ks = coeffs_chunk.values;
|
||||
|
||||
let mut ss0 = 1 << (precision - 1);
|
||||
let mut ss1 = ss0;
|
||||
let mut ss2 = ss0;
|
||||
let mut ss3 = ss0;
|
||||
let src_pixels = src_image.iter_horiz(first_x_src, y_src);
|
||||
for (&k, &src_pixel) in ks.iter().zip(src_pixels) {
|
||||
let components: [u8; 4] = src_pixel.to_le_bytes();
|
||||
ss0 += components[0] as i32 * (k as i32);
|
||||
ss1 += components[1] as i32 * (k as i32);
|
||||
ss2 += components[2] as i32 * (k as i32);
|
||||
ss3 += components[3] as i32 * (k as i32);
|
||||
}
|
||||
let t: [u8; 4] = unsafe {
|
||||
[
|
||||
optimisations::clip8(ss0, precision),
|
||||
optimisations::clip8(ss1, precision),
|
||||
optimisations::clip8(ss2, precision),
|
||||
optimisations::clip8(ss3, precision),
|
||||
]
|
||||
};
|
||||
*dst_pixel = u32::from_le_bytes(t);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn vert_convolution(
|
||||
&self,
|
||||
src_image: &SrcImageView,
|
||||
dst_image: &mut DstImageView,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
|
||||
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for (&coeffs_chunk, dst_row) in coefficients_chunks.iter().zip(dst_rows) {
|
||||
let first_y_src = coeffs_chunk.start;
|
||||
let ks = coeffs_chunk.values;
|
||||
|
||||
for (x_src, out_pixel) in dst_row.iter_mut().enumerate() {
|
||||
let mut ss0 = 1 << (precision - 1);
|
||||
let mut ss1 = ss0;
|
||||
let mut ss2 = ss0;
|
||||
let mut ss3 = ss0;
|
||||
for (dy, &k) in ks.iter().enumerate() {
|
||||
let pixel = src_image.get_pixel_u32(x_src as u32, first_y_src + dy as u32);
|
||||
let components: [u8; 4] = pixel.to_le_bytes();
|
||||
ss0 += components[0] as i32 * (k as i32);
|
||||
ss1 += components[1] as i32 * (k as i32);
|
||||
ss2 += components[2] as i32 * (k as i32);
|
||||
ss3 += components[3] as i32 * (k as i32);
|
||||
}
|
||||
let t: [u8; 4] = unsafe {
|
||||
[
|
||||
optimisations::clip8(ss0, precision),
|
||||
optimisations::clip8(ss1, precision),
|
||||
optimisations::clip8(ss2, precision),
|
||||
optimisations::clip8(ss3, precision),
|
||||
]
|
||||
};
|
||||
*out_pixel = u32::from_le_bytes(t);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub struct NativeI32;
|
||||
|
||||
impl Convolution for NativeI32 {
|
||||
fn horiz_convolution(
|
||||
&self,
|
||||
src_image: &SrcImageView,
|
||||
dst_image: &mut DstImageView,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
for y_dst in 0..dst_image.height().get() {
|
||||
let y_src = y_dst + offset;
|
||||
if let Some(out_row) = dst_image.get_row_mut(y_dst) {
|
||||
let out_row_i32 = unsafe {
|
||||
let len = out_row.len();
|
||||
let ptr = out_row.as_mut_ptr();
|
||||
slice::from_raw_parts_mut(ptr as *mut i32, len)
|
||||
};
|
||||
for (x_dst, (&bound, out_pixel)) in bounds.iter().zip(out_row_i32).enumerate() {
|
||||
let first_x_src = bound.start;
|
||||
let start_index = window_size * x_dst;
|
||||
let end_index = start_index + bound.size as usize;
|
||||
|
||||
let ks = &values[start_index..end_index];
|
||||
|
||||
let mut ss = 0.;
|
||||
let pixels = src_image.iter_horiz_i32(first_x_src, y_src);
|
||||
for (&k, &pixel) in ks.iter().zip(pixels) {
|
||||
ss += pixel as f64 * k;
|
||||
}
|
||||
*out_pixel = ss.round() as i32;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn vert_convolution(
|
||||
&self,
|
||||
image: &SrcImageView,
|
||||
out_image: &mut DstImageView,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
for (y_dst, &bound) in bounds.iter().enumerate() {
|
||||
let first_y_src = bound.start;
|
||||
let start_index = window_size * y_dst;
|
||||
let end_index = start_index + bound.size as usize;
|
||||
let ks = &values[start_index..end_index];
|
||||
|
||||
if let Some(out_row) = out_image.get_row_mut(y_dst as u32) {
|
||||
let out_row_i32 = unsafe {
|
||||
let len = out_row.len();
|
||||
let ptr = out_row.as_mut_ptr();
|
||||
slice::from_raw_parts_mut(ptr as *mut i32, len)
|
||||
};
|
||||
for (x_src, out_pixel) in out_row_i32.iter_mut().enumerate() {
|
||||
let mut ss = 0.;
|
||||
for (dy, &k) in ks.iter().enumerate() {
|
||||
let pixel = image.get_pixel_i32(x_src as u32, first_y_src + dy as u32);
|
||||
ss += pixel as f64 * k;
|
||||
}
|
||||
*out_pixel = ss.round() as i32;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub struct NativeF32;
|
||||
|
||||
impl Convolution for NativeF32 {
|
||||
fn horiz_convolution(
|
||||
&self,
|
||||
src_image: &SrcImageView,
|
||||
dst_image: &mut DstImageView,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
for y_dst in 0..dst_image.height().get() {
|
||||
let y_src = y_dst + offset;
|
||||
if let Some(out_row) = dst_image.get_row_mut(y_dst) {
|
||||
let out_row_f32 = unsafe {
|
||||
let len = out_row.len();
|
||||
let ptr = out_row.as_mut_ptr();
|
||||
slice::from_raw_parts_mut(ptr as *mut f32, len)
|
||||
};
|
||||
for (x_dst, (&bound, out_pixel)) in bounds.iter().zip(out_row_f32).enumerate() {
|
||||
let first_x_src = bound.start;
|
||||
let start_index = window_size * x_dst;
|
||||
let end_index = start_index + bound.size as usize;
|
||||
|
||||
let ks = &values[start_index..end_index];
|
||||
|
||||
let mut ss = 0.;
|
||||
let pixels = src_image.iter_horiz_f32(first_x_src, y_src);
|
||||
for (&k, &pixel) in ks.iter().zip(pixels) {
|
||||
ss += pixel as f64 * k;
|
||||
}
|
||||
*out_pixel = ss as f32;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn vert_convolution(
|
||||
&self,
|
||||
src_image: &SrcImageView,
|
||||
dst_image: &mut DstImageView,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
for (y_dst, &bound) in bounds.iter().enumerate() {
|
||||
let first_y_src = bound.start;
|
||||
let start_index = window_size * y_dst;
|
||||
let end_index = start_index + bound.size as usize;
|
||||
let ks = &values[start_index..end_index];
|
||||
|
||||
if let Some(out_row) = dst_image.get_row_mut(y_dst as u32) {
|
||||
let out_row_f32 = unsafe {
|
||||
let len = out_row.len();
|
||||
let ptr = out_row.as_mut_ptr();
|
||||
slice::from_raw_parts_mut(ptr as *mut f32, len)
|
||||
};
|
||||
for (x_src, out_pixel) in out_row_f32.iter_mut().enumerate() {
|
||||
let mut ss = 0.;
|
||||
for (dy, &k) in ks.iter().enumerate() {
|
||||
let pixel = src_image.get_pixel_f32(x_src as u32, first_y_src + dy as u32);
|
||||
ss += pixel as f64 * k;
|
||||
}
|
||||
*out_pixel = ss as f32;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,75 @@
|
||||
use std::slice;
|
||||
|
||||
use crate::convolution::{Coefficients, Convolution};
|
||||
use crate::{DstImageView, SrcImageView};
|
||||
|
||||
pub struct NativeF32;
|
||||
|
||||
impl Convolution for NativeF32 {
|
||||
fn horiz_convolution(
|
||||
&self,
|
||||
src_image: &SrcImageView,
|
||||
dst_image: &mut DstImageView,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
for y_dst in 0..dst_image.height().get() {
|
||||
let y_src = y_dst + offset;
|
||||
if let Some(out_row) = dst_image.get_row_mut(y_dst) {
|
||||
let out_row_f32 = unsafe {
|
||||
let len = out_row.len();
|
||||
let ptr = out_row.as_mut_ptr();
|
||||
slice::from_raw_parts_mut(ptr as *mut f32, len)
|
||||
};
|
||||
for (x_dst, (&bound, out_pixel)) in bounds.iter().zip(out_row_f32).enumerate() {
|
||||
let first_x_src = bound.start;
|
||||
let start_index = window_size * x_dst;
|
||||
let end_index = start_index + bound.size as usize;
|
||||
|
||||
let ks = &values[start_index..end_index];
|
||||
|
||||
let mut ss = 0.;
|
||||
let pixels = src_image.iter_horiz_f32(first_x_src, y_src);
|
||||
for (&k, &pixel) in ks.iter().zip(pixels) {
|
||||
ss += pixel as f64 * k;
|
||||
}
|
||||
*out_pixel = ss as f32;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn vert_convolution(
|
||||
&self,
|
||||
src_image: &SrcImageView,
|
||||
dst_image: &mut DstImageView,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
for (y_dst, &bound) in bounds.iter().enumerate() {
|
||||
let first_y_src = bound.start;
|
||||
let start_index = window_size * y_dst;
|
||||
let end_index = start_index + bound.size as usize;
|
||||
let ks = &values[start_index..end_index];
|
||||
|
||||
if let Some(out_row) = dst_image.get_row_mut(y_dst as u32) {
|
||||
let out_row_f32 = unsafe {
|
||||
let len = out_row.len();
|
||||
let ptr = out_row.as_mut_ptr();
|
||||
slice::from_raw_parts_mut(ptr as *mut f32, len)
|
||||
};
|
||||
for (x_src, out_pixel) in out_row_f32.iter_mut().enumerate() {
|
||||
let mut ss = 0.;
|
||||
for (dy, &k) in ks.iter().enumerate() {
|
||||
let pixel = src_image.get_pixel_f32(x_src as u32, first_y_src + dy as u32);
|
||||
ss += pixel as f64 * k;
|
||||
}
|
||||
*out_pixel = ss as f32;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,53 @@
|
||||
use crate::convolution::{Coefficients, Convolution};
|
||||
use crate::{DstImageView, SrcImageView};
|
||||
|
||||
pub struct NativeI32;
|
||||
|
||||
impl Convolution for NativeI32 {
|
||||
fn horiz_convolution(
|
||||
&self,
|
||||
src_image: &SrcImageView,
|
||||
dst_image: &mut DstImageView,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let coefficients_chunks = coeffs.get_chunks();
|
||||
let mut y_src = offset;
|
||||
|
||||
for out_row in dst_image.iter_rows_mut() {
|
||||
for (out_pixel, coeffs_chunk) in out_row.iter_mut().zip(&coefficients_chunks) {
|
||||
let first_x_src = coeffs_chunk.start;
|
||||
let mut ss = 0.;
|
||||
let pixels = src_image.iter_horiz_i32(first_x_src, y_src);
|
||||
for (&k, &pixel) in coeffs_chunk.values.iter().zip(pixels) {
|
||||
ss += pixel as f64 * k;
|
||||
}
|
||||
*out_pixel = ss.round() as i32 as u32;
|
||||
}
|
||||
y_src += 1;
|
||||
}
|
||||
}
|
||||
|
||||
fn vert_convolution(
|
||||
&self,
|
||||
image: &SrcImageView,
|
||||
out_image: &mut DstImageView,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let coefficients_chunks = coeffs.get_chunks();
|
||||
|
||||
for (out_row, coeffs_chunk) in out_image.iter_rows_mut().zip(coefficients_chunks) {
|
||||
let first_y_src = coeffs_chunk.start;
|
||||
for (x_src, out_pixel) in out_row.iter_mut().enumerate() {
|
||||
let mut ss = 0.;
|
||||
let mut y_src = first_y_src;
|
||||
for &k in coeffs_chunk.values.iter() {
|
||||
let pixel = image.get_pixel_i32(x_src as u32, y_src);
|
||||
ss += pixel as f64 * k;
|
||||
y_src += 1;
|
||||
}
|
||||
*out_pixel = ss.round() as i32 as u32;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,7 @@
|
||||
pub use f32x1::NativeF32;
|
||||
pub use i32x1::NativeI32;
|
||||
pub use u8x4::NativeU8x4;
|
||||
|
||||
mod f32x1;
|
||||
mod i32x1;
|
||||
mod u8x4;
|
||||
@@ -0,0 +1,95 @@
|
||||
use crate::convolution::{optimisations, Coefficients, Convolution};
|
||||
use crate::{DstImageView, SrcImageView};
|
||||
|
||||
pub struct NativeU8x4;
|
||||
|
||||
impl Convolution for NativeU8x4 {
|
||||
fn horiz_convolution(
|
||||
&self,
|
||||
src_image: &SrcImageView,
|
||||
dst_image: &mut DstImageView,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks = normalizer_guard.normalized_i16_chunks(window_size, &bounds);
|
||||
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for (y_dst, dst_row) in dst_rows.enumerate() {
|
||||
let y_src = y_dst as u32 + offset;
|
||||
|
||||
for (&coeffs_chunk, dst_pixel) in coefficients_chunks.iter().zip(dst_row.iter_mut()) {
|
||||
let first_x_src = coeffs_chunk.start;
|
||||
let ks = coeffs_chunk.values;
|
||||
|
||||
let mut ss0 = 1 << (precision - 1);
|
||||
let mut ss1 = ss0;
|
||||
let mut ss2 = ss0;
|
||||
let mut ss3 = ss0;
|
||||
let src_pixels = src_image.iter_horiz(first_x_src, y_src);
|
||||
for (&k, &src_pixel) in ks.iter().zip(src_pixels) {
|
||||
let components: [u8; 4] = src_pixel.to_le_bytes();
|
||||
ss0 += components[0] as i32 * (k as i32);
|
||||
ss1 += components[1] as i32 * (k as i32);
|
||||
ss2 += components[2] as i32 * (k as i32);
|
||||
ss3 += components[3] as i32 * (k as i32);
|
||||
}
|
||||
let t: [u8; 4] = unsafe {
|
||||
[
|
||||
optimisations::clip8(ss0, precision),
|
||||
optimisations::clip8(ss1, precision),
|
||||
optimisations::clip8(ss2, precision),
|
||||
optimisations::clip8(ss3, precision),
|
||||
]
|
||||
};
|
||||
*dst_pixel = u32::from_le_bytes(t);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn vert_convolution(
|
||||
&self,
|
||||
src_image: &SrcImageView,
|
||||
dst_image: &mut DstImageView,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks = normalizer_guard.normalized_i16_chunks(window_size, &bounds);
|
||||
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for (&coeffs_chunk, dst_row) in coefficients_chunks.iter().zip(dst_rows) {
|
||||
let first_y_src = coeffs_chunk.start;
|
||||
let ks = coeffs_chunk.values;
|
||||
|
||||
for (x_src, out_pixel) in dst_row.iter_mut().enumerate() {
|
||||
let mut ss0 = 1 << (precision - 1);
|
||||
let mut ss1 = ss0;
|
||||
let mut ss2 = ss0;
|
||||
let mut ss3 = ss0;
|
||||
for (dy, &k) in ks.iter().enumerate() {
|
||||
let pixel = src_image.get_pixel_u32(x_src as u32, first_y_src + dy as u32);
|
||||
let components: [u8; 4] = pixel.to_le_bytes();
|
||||
ss0 += components[0] as i32 * (k as i32);
|
||||
ss1 += components[1] as i32 * (k as i32);
|
||||
ss2 += components[2] as i32 * (k as i32);
|
||||
ss3 += components[3] as i32 * (k as i32);
|
||||
}
|
||||
let t: [u8; 4] = unsafe {
|
||||
[
|
||||
optimisations::clip8(ss0, precision),
|
||||
optimisations::clip8(ss1, precision),
|
||||
optimisations::clip8(ss2, precision),
|
||||
optimisations::clip8(ss3, precision),
|
||||
]
|
||||
};
|
||||
*out_pixel = u32::from_le_bytes(t);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,6 +1,7 @@
|
||||
use crate::convolution::{Bound, CoefficientsChunk};
|
||||
use std::slice;
|
||||
|
||||
use super::Bound;
|
||||
|
||||
// This code is based on C-implementation from Pillow-SIMD package for Python
|
||||
// https://github.com/uploadcare/pillow-simd
|
||||
|
||||
@@ -87,6 +88,12 @@ pub struct NormalizerGuard {
|
||||
precision: u8,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
pub struct CoefficientsI16Chunk<'a> {
|
||||
pub start: u32,
|
||||
pub values: &'a [i16],
|
||||
}
|
||||
|
||||
impl NormalizerGuard {
|
||||
#[inline]
|
||||
pub fn new(mut values: Vec<f64>) -> Self {
|
||||
@@ -119,18 +126,18 @@ impl NormalizerGuard {
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn normalized(&self) -> &[i16] {
|
||||
pub fn normalized_i16(&self) -> &[i16] {
|
||||
let len = self.values.len();
|
||||
let ptr = self.values.as_ptr();
|
||||
unsafe { slice::from_raw_parts(ptr as *const i16, len) }
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn normalized_chunks(
|
||||
pub fn normalized_i16_chunks(
|
||||
&self,
|
||||
window_size: usize,
|
||||
bounds: &[Bound],
|
||||
) -> Vec<CoefficientsChunk> {
|
||||
) -> Vec<CoefficientsI16Chunk> {
|
||||
let len = self.values.len();
|
||||
let ptr = self.values.as_ptr();
|
||||
let mut cooefs = unsafe { slice::from_raw_parts(ptr as *const i16, len) };
|
||||
@@ -139,7 +146,7 @@ impl NormalizerGuard {
|
||||
let (left, right) = cooefs.split_at(window_size);
|
||||
cooefs = right;
|
||||
let size = bound.size as usize;
|
||||
res.push(CoefficientsChunk {
|
||||
res.push(CoefficientsI16Chunk {
|
||||
start: bound.start,
|
||||
values: &left[0..size],
|
||||
});
|
||||
@@ -0,0 +1,3 @@
|
||||
pub use u8x4::Sse4U8x4;
|
||||
|
||||
mod u8x4;
|
||||
@@ -1,9 +1,10 @@
|
||||
use std::arch::x86_64::*;
|
||||
use std::intrinsics::transmute;
|
||||
|
||||
use crate::convolution::{Bound, Coefficients, CoefficientsChunk, Convolution};
|
||||
use crate::convolution::optimisations::CoefficientsI16Chunk;
|
||||
use crate::convolution::{optimisations, Bound, Coefficients, Convolution};
|
||||
use crate::image_view::{DstImageView, FourRows, FourRowsMut, SrcImageView};
|
||||
use crate::{optimisations, simd_utils};
|
||||
use crate::simd_utils;
|
||||
|
||||
pub struct Sse4U8x4;
|
||||
|
||||
@@ -21,7 +22,7 @@ impl Sse4U8x4 {
|
||||
&self,
|
||||
src_rows: FourRows,
|
||||
dst_rows: FourRowsMut,
|
||||
coefficients_chunks: &[CoefficientsChunk],
|
||||
coefficients_chunks: &[CoefficientsI16Chunk],
|
||||
precision: u8,
|
||||
) {
|
||||
let (s_row0, s_row1, s_row2, s_row3) = src_rows;
|
||||
@@ -159,7 +160,7 @@ impl Sse4U8x4 {
|
||||
&self,
|
||||
src_row: &[u32],
|
||||
dst_row: &mut [u32],
|
||||
coefficients_chunks: &[CoefficientsChunk],
|
||||
coefficients_chunks: &[CoefficientsI16Chunk],
|
||||
precision: u8,
|
||||
) {
|
||||
let initial = _mm_set1_epi32(1 << (precision - 1));
|
||||
@@ -497,7 +498,7 @@ impl Convolution for Sse4U8x4 {
|
||||
let normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks =
|
||||
normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
normalizer_guard.normalized_i16_chunks(window_size, &bounds_per_pixel);
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
@@ -533,7 +534,7 @@ impl Convolution for Sse4U8x4 {
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coeffs_i16 = normalizer_guard.normalized();
|
||||
let coeffs_i16 = normalizer_guard.normalized_i16();
|
||||
let coeffs_chunks = coeffs_i16.chunks(window_size);
|
||||
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
+1
-1
@@ -1,4 +1,5 @@
|
||||
#![doc = include_str!("../README.md")]
|
||||
|
||||
pub use alpha::{MulDiv, MulDivImageError, MulDivImagesError};
|
||||
pub use convolution::FilterType;
|
||||
pub use errors::{CropBoxError, ImageBufferError, ImageRowsError, InvalidBufferSizeError};
|
||||
@@ -11,6 +12,5 @@ mod convolution;
|
||||
mod errors;
|
||||
mod image_data;
|
||||
mod image_view;
|
||||
mod optimisations;
|
||||
mod resizer;
|
||||
mod simd_utils;
|
||||
|
||||
Reference in New Issue
Block a user