Revert few changes

This commit is contained in:
Kirill Kuzminykh
2023-02-28 00:16:07 +03:00
parent 999c684fce
commit 1dd629eb81
4 changed files with 43 additions and 25 deletions
+11 -11
View File
@@ -82,17 +82,17 @@ fn downscale_bench(
pub fn resize_bench(bench_group: &mut utils::BenchGroup) {
let pixel_types = [
PixelType::U8,
PixelType::U8x2,
// PixelType::U8,
// PixelType::U8x2,
PixelType::U8x3,
PixelType::U8x4,
PixelType::U16,
PixelType::U16x2,
PixelType::U16x3,
PixelType::U16x4,
PixelType::I32,
// PixelType::U8x4,
// PixelType::U16,
// PixelType::U16x2,
// PixelType::U16x3,
// PixelType::U16x4,
// PixelType::I32,
];
let mut cpu_extensions = vec![CpuExtensions::None];
let mut cpu_extensions = vec![]; //CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{
cpu_extensions.push(CpuExtensions::Sse4_1);
@@ -124,8 +124,8 @@ pub fn resize_bench(bench_group: &mut utils::BenchGroup) {
}
}
native_nearest_u8x4_bench(bench_group);
native_nearest_u8_bench(bench_group);
// native_nearest_u8x4_bench(bench_group);
// native_nearest_u8_bench(bench_group);
}
fn main() {
+2 -1
View File
@@ -215,12 +215,13 @@ unsafe fn horiz_convolution_8u(
R: |-1 03| |-1 00|
*/
let src_width = src_row.len();
let initial = i32x4_splat(1 << (precision - 1));
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let x_start = coeffs_chunk.start as usize;
let mut x = x_start;
let mut coeffs = coeffs_chunk.values;
let mut sss = i32x4_splat(1 << (precision - 1));
let mut sss = initial;
// Next block of code will be load source pixels by 16 bytes per time.
// We must guarantee what this process will not go beyond
+29 -12
View File
@@ -38,7 +38,7 @@ unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
let y_start = coeffs_chunk.start;
let coeffs = coeffs_chunk.values;
let max_y = y_start + coeffs.len() as u32;
let precision = normalizer.precision() as u32;
let precision = normalizer.precision();
let mut dst_u8 = T::components_mut(dst_row);
let initial = i32x4_splat(1 << (precision - 1));
@@ -144,14 +144,20 @@ unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
sss7 = i32x4_add(sss7, i32x4_dot_i16x8(pix, mmk));
}
sss0 = i32x4_shr(sss0, precision);
sss1 = i32x4_shr(sss1, precision);
sss2 = i32x4_shr(sss2, precision);
sss3 = i32x4_shr(sss3, precision);
sss4 = i32x4_shr(sss4, precision);
sss5 = i32x4_shr(sss5, precision);
sss6 = i32x4_shr(sss6, precision);
sss7 = i32x4_shr(sss7, precision);
// This version of code works faster.
macro_rules! call {
($imm8:expr) => {{
sss0 = i32x4_shr(sss0, $imm8);
sss1 = i32x4_shr(sss1, $imm8);
sss2 = i32x4_shr(sss2, $imm8);
sss3 = i32x4_shr(sss3, $imm8);
sss4 = i32x4_shr(sss4, $imm8);
sss5 = i32x4_shr(sss5, $imm8);
sss6 = i32x4_shr(sss6, $imm8);
sss7 = i32x4_shr(sss7, $imm8);
}};
}
constify_imm8!(precision, call);
sss0 = i16x8_narrow_i32x4(sss0, sss1);
sss2 = i16x8_narrow_i32x4(sss2, sss3);
@@ -210,8 +216,13 @@ unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
sss1 = i32x4_add(sss1, i32x4_dot_i16x8(pix, mmk));
}
sss0 = i32x4_shr(sss0, precision);
sss1 = i32x4_shr(sss1, precision);
macro_rules! call {
($imm8:expr) => {{
sss0 = i32x4_shr(sss0, $imm8);
sss1 = i32x4_shr(sss1, $imm8);
}};
}
constify_imm8!(precision, call);
sss0 = i16x8_narrow_i32x4(sss0, sss1);
sss0 = u8x16_narrow_i16x8(sss0, sss0);
@@ -253,7 +264,13 @@ unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
}
sss = i32x4_shr(sss, precision);
macro_rules! call {
($imm8:expr) => {{
sss = i32x4_shr(sss, $imm8);
}};
}
constify_imm8!(precision, call);
sss = i16x8_narrow_i32x4(sss, sss);
let dst_ptr = dst_chunk.as_mut_ptr() as *mut i32;
*dst_ptr = i32x4_extract_lane::<0>(u8x16_narrow_i16x8(sss, sss));
+1 -1
View File
@@ -111,7 +111,7 @@ impl<'a> Image<'a> {
#[inline(always)]
pub fn buffer(&self) -> &[u8] {
match &self.buffer {
BufferContainer::MutU8(p) => *p,
BufferContainer::MutU8(p) => p,
BufferContainer::VecU8(v) => v,
}
}