mirror of
https://github.com/Cykooz/fast_image_resize.git
synced 2026-10-08 01:11:09 +00:00
Revert few changes
This commit is contained in:
+11
-11
@@ -82,17 +82,17 @@ fn downscale_bench(
|
||||
|
||||
pub fn resize_bench(bench_group: &mut utils::BenchGroup) {
|
||||
let pixel_types = [
|
||||
PixelType::U8,
|
||||
PixelType::U8x2,
|
||||
// PixelType::U8,
|
||||
// PixelType::U8x2,
|
||||
PixelType::U8x3,
|
||||
PixelType::U8x4,
|
||||
PixelType::U16,
|
||||
PixelType::U16x2,
|
||||
PixelType::U16x3,
|
||||
PixelType::U16x4,
|
||||
PixelType::I32,
|
||||
// PixelType::U8x4,
|
||||
// PixelType::U16,
|
||||
// PixelType::U16x2,
|
||||
// PixelType::U16x3,
|
||||
// PixelType::U16x4,
|
||||
// PixelType::I32,
|
||||
];
|
||||
let mut cpu_extensions = vec![CpuExtensions::None];
|
||||
let mut cpu_extensions = vec![]; //CpuExtensions::None];
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
{
|
||||
cpu_extensions.push(CpuExtensions::Sse4_1);
|
||||
@@ -124,8 +124,8 @@ pub fn resize_bench(bench_group: &mut utils::BenchGroup) {
|
||||
}
|
||||
}
|
||||
|
||||
native_nearest_u8x4_bench(bench_group);
|
||||
native_nearest_u8_bench(bench_group);
|
||||
// native_nearest_u8x4_bench(bench_group);
|
||||
// native_nearest_u8_bench(bench_group);
|
||||
}
|
||||
|
||||
fn main() {
|
||||
|
||||
@@ -215,12 +215,13 @@ unsafe fn horiz_convolution_8u(
|
||||
R: |-1 03| |-1 00|
|
||||
*/
|
||||
let src_width = src_row.len();
|
||||
let initial = i32x4_splat(1 << (precision - 1));
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let x_start = coeffs_chunk.start as usize;
|
||||
let mut x = x_start;
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut sss = i32x4_splat(1 << (precision - 1));
|
||||
let mut sss = initial;
|
||||
|
||||
// Next block of code will be load source pixels by 16 bytes per time.
|
||||
// We must guarantee what this process will not go beyond
|
||||
|
||||
@@ -38,7 +38,7 @@ unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
|
||||
let y_start = coeffs_chunk.start;
|
||||
let coeffs = coeffs_chunk.values;
|
||||
let max_y = y_start + coeffs.len() as u32;
|
||||
let precision = normalizer.precision() as u32;
|
||||
let precision = normalizer.precision();
|
||||
let mut dst_u8 = T::components_mut(dst_row);
|
||||
|
||||
let initial = i32x4_splat(1 << (precision - 1));
|
||||
@@ -144,14 +144,20 @@ unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
|
||||
sss7 = i32x4_add(sss7, i32x4_dot_i16x8(pix, mmk));
|
||||
}
|
||||
|
||||
sss0 = i32x4_shr(sss0, precision);
|
||||
sss1 = i32x4_shr(sss1, precision);
|
||||
sss2 = i32x4_shr(sss2, precision);
|
||||
sss3 = i32x4_shr(sss3, precision);
|
||||
sss4 = i32x4_shr(sss4, precision);
|
||||
sss5 = i32x4_shr(sss5, precision);
|
||||
sss6 = i32x4_shr(sss6, precision);
|
||||
sss7 = i32x4_shr(sss7, precision);
|
||||
// This version of code works faster.
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss0 = i32x4_shr(sss0, $imm8);
|
||||
sss1 = i32x4_shr(sss1, $imm8);
|
||||
sss2 = i32x4_shr(sss2, $imm8);
|
||||
sss3 = i32x4_shr(sss3, $imm8);
|
||||
sss4 = i32x4_shr(sss4, $imm8);
|
||||
sss5 = i32x4_shr(sss5, $imm8);
|
||||
sss6 = i32x4_shr(sss6, $imm8);
|
||||
sss7 = i32x4_shr(sss7, $imm8);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
|
||||
sss0 = i16x8_narrow_i32x4(sss0, sss1);
|
||||
sss2 = i16x8_narrow_i32x4(sss2, sss3);
|
||||
@@ -210,8 +216,13 @@ unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
|
||||
sss1 = i32x4_add(sss1, i32x4_dot_i16x8(pix, mmk));
|
||||
}
|
||||
|
||||
sss0 = i32x4_shr(sss0, precision);
|
||||
sss1 = i32x4_shr(sss1, precision);
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss0 = i32x4_shr(sss0, $imm8);
|
||||
sss1 = i32x4_shr(sss1, $imm8);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
|
||||
sss0 = i16x8_narrow_i32x4(sss0, sss1);
|
||||
sss0 = u8x16_narrow_i16x8(sss0, sss0);
|
||||
@@ -253,7 +264,13 @@ unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
|
||||
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
|
||||
}
|
||||
|
||||
sss = i32x4_shr(sss, precision);
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss = i32x4_shr(sss, $imm8);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
|
||||
sss = i16x8_narrow_i32x4(sss, sss);
|
||||
let dst_ptr = dst_chunk.as_mut_ptr() as *mut i32;
|
||||
*dst_ptr = i32x4_extract_lane::<0>(u8x16_narrow_i16x8(sss, sss));
|
||||
|
||||
+1
-1
@@ -111,7 +111,7 @@ impl<'a> Image<'a> {
|
||||
#[inline(always)]
|
||||
pub fn buffer(&self) -> &[u8] {
|
||||
match &self.buffer {
|
||||
BufferContainer::MutU8(p) => *p,
|
||||
BufferContainer::MutU8(p) => p,
|
||||
BufferContainer::VecU8(v) => v,
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user