Fixed Neon implementations.

This commit is contained in:
Kirill Kuzminykh
2024-04-18 20:28:40 +00:00
parent 39f942f4c5
commit 9d56153b2f
10 changed files with 53 additions and 42 deletions
+2 -2
View File
@@ -15,7 +15,7 @@ pub(crate) fn horiz_convolution(
let normalizer = optimisations::Normalizer32::new(coeffs);
let precision = normalizer.precision();
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_view.height().get();
let dst_height = dst_view.height();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
@@ -25,7 +25,7 @@ pub(crate) fn horiz_convolution(
}
}
let mut yy = dst_height - dst_height % 4;
let yy = dst_height - dst_height % 4;
let src_rows = src_view.iter_rows(yy + offset);
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
+2 -2
View File
@@ -15,7 +15,7 @@ pub(crate) fn horiz_convolution(
let normalizer = optimisations::Normalizer32::new(coeffs);
let precision = normalizer.precision();
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_view.height().get();
let dst_height = dst_view.height();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
@@ -25,7 +25,7 @@ pub(crate) fn horiz_convolution(
}
}
let mut yy = dst_height - dst_height % 4;
let yy = dst_height - dst_height % 4;
let src_rows = src_view.iter_rows(yy + offset);
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
+2 -2
View File
@@ -15,7 +15,7 @@ pub(crate) fn horiz_convolution(
let normalizer = optimisations::Normalizer32::new(coeffs);
let precision = normalizer.precision();
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_view.height().get();
let dst_height = dst_view.height();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
@@ -25,7 +25,7 @@ pub(crate) fn horiz_convolution(
}
}
let mut yy = dst_height - dst_height % 4;
let yy = dst_height - dst_height % 4;
let src_rows = src_view.iter_rows(yy + offset);
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
+2 -2
View File
@@ -14,10 +14,10 @@ pub(crate) fn horiz_convolution(
) {
let normalizer = optimisations::Normalizer16::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_view.height().get();
let dst_height = dst_view.height();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut(0);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
+2 -2
View File
@@ -14,7 +14,7 @@ pub(crate) fn horiz_convolution(
let normalizer = optimisations::Normalizer16::new(coeffs);
let precision = normalizer.precision();
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_view.height().get();
let dst_height = dst_view.height();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
@@ -24,7 +24,7 @@ pub(crate) fn horiz_convolution(
}
}
let mut yy = dst_height - dst_height % 4;
let yy = dst_height - dst_height % 4;
let src_rows = src_view.iter_rows(yy + offset);
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
+2 -2
View File
@@ -15,7 +15,7 @@ pub(crate) fn horiz_convolution(
let normalizer = optimisations::Normalizer16::new(coeffs);
let precision = normalizer.precision();
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_view.height().get();
let dst_height = dst_view.height();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
@@ -25,7 +25,7 @@ pub(crate) fn horiz_convolution(
}
}
let mut yy = dst_height - dst_height % 4;
let yy = dst_height - dst_height % 4;
let src_rows = src_view.iter_rows(yy + offset);
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
+2 -2
View File
@@ -15,7 +15,7 @@ pub(crate) fn horiz_convolution(
let normalizer = optimisations::Normalizer16::new(coeffs);
let precision = normalizer.precision();
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_view.height().get();
let dst_height = dst_view.height();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
@@ -25,7 +25,7 @@ pub(crate) fn horiz_convolution(
}
}
let mut yy = dst_height - dst_height % 4;
let yy = dst_height - dst_height % 4;
let src_rows = src_view.iter_rows(yy + offset);
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
+1 -1
View File
@@ -20,7 +20,7 @@ pub(crate) fn vert_convolution<T>(
let initial = 1i64 << (precision - 1);
let start_src_x = offset as usize * T::count_of_components();
let mut tmp_dst = vec![0i64; dst_view.width().get() as usize * T::count_of_components()];
let mut tmp_dst = vec![0i64; dst_view.width() as usize * T::count_of_components()];
let tmp_buf = tmp_dst.as_mut_slice();
let dst_rows = dst_view.iter_rows_mut(0);
for (dst_row, coeffs_chunk) in dst_rows.zip(coefficients_chunks) {
+1 -1
View File
@@ -20,7 +20,7 @@ pub(crate) fn vert_convolution<T>(
let initial = 1 << (precision - 1);
let start_src_x = offset as usize * T::count_of_components();
let mut tmp_dst = vec![0i32; dst_view.width().get() as usize * T::count_of_components()];
let mut tmp_dst = vec![0i32; dst_view.width() as usize * T::count_of_components()];
let tmp_buf = tmp_dst.as_mut_slice();
let dst_rows = dst_view.iter_rows_mut(0);
for (dst_row, coeffs_chunk) in dst_rows.zip(coefficients_chunks) {
+37 -26
View File
@@ -60,10 +60,10 @@ pub unsafe fn load_deintrel_u8x16x2<T>(buf: &[T], index: usize) -> uint8x16x2_t
vld2q_u8(buf.get_unchecked(index..).as_ptr() as *const u8)
}
#[inline(always)]
pub unsafe fn load_deintrel_u8x16x3<T>(buf: &[T], index: usize) -> uint8x16x3_t {
vld3q_u8(buf.get_unchecked(index..).as_ptr() as *const u8)
}
// #[inline(always)]
// pub unsafe fn load_deintrel_u8x16x3<T>(buf: &[T], index: usize) -> uint8x16x3_t {
// vld3q_u8(buf.get_unchecked(index..).as_ptr() as *const u8)
// }
#[inline(always)]
pub unsafe fn load_deintrel_u8x16x4<T>(buf: &[T], index: usize) -> uint8x16x4_t {
@@ -284,26 +284,26 @@ pub unsafe fn create_u8x16_from_one_u32<T>(buf: &[T], index: usize) -> uint8x16_
/// Multiply the packed unsigned 16-bit integers in a and b, producing
/// intermediate 32-bit integers, and store the high 16 bits of the intermediate
/// integers in dst.
#[inline(always)]
pub unsafe fn mulhi_u16x8(a: uint16x8_t, b: uint16x8_t) -> uint16x8_t {
let a3210 = vget_low_u16(a);
let b3210 = vget_low_u16(b);
let ab3210 = vmull_u16(a3210, b3210);
let ab7654 = vmull_high_u16(a, b);
vuzp2q_u16(vreinterpretq_u16_u32(ab3210), vreinterpretq_u16_u32(ab7654))
}
// #[inline(always)]
// pub unsafe fn mulhi_u16x8(a: uint16x8_t, b: uint16x8_t) -> uint16x8_t {
// let a3210 = vget_low_u16(a);
// let b3210 = vget_low_u16(b);
// let ab3210 = vmull_u16(a3210, b3210);
// let ab7654 = vmull_high_u16(a, b);
// vuzp2q_u16(vreinterpretq_u16_u32(ab3210), vreinterpretq_u16_u32(ab7654))
// }
/// Multiply the packed unsigned 32-bit integers in a and b, producing
/// intermediate 64-bit integers, and store the high 32 bits of the intermediate
/// integers in dst.
#[inline(always)]
pub unsafe fn mulhi_u32x4(a: uint32x4_t, b: uint32x4_t) -> uint32x4_t {
let a3210 = vget_low_u32(a);
let b3210 = vget_low_u32(b);
let ab3210 = vmull_u32(a3210, b3210);
let ab7654 = vmull_high_u32(a, b);
vuzp2q_u32(vreinterpretq_u32_u64(ab3210), vreinterpretq_u32_u64(ab7654))
}
// #[inline(always)]
// pub unsafe fn mulhi_u32x4(a: uint32x4_t, b: uint32x4_t) -> uint32x4_t {
// let a3210 = vget_low_u32(a);
// let b3210 = vget_low_u32(b);
// let ab3210 = vmull_u32(a3210, b3210);
// let ab7654 = vmull_high_u32(a, b);
// vuzp2q_u32(vreinterpretq_u32_u64(ab3210), vreinterpretq_u32_u64(ab7654))
// }
#[inline]
#[target_feature(enable = "neon")]
@@ -358,6 +358,14 @@ pub unsafe fn multiply_color_to_alpha_u16x4(color: uint16x4_t, alpha: uint16x4_t
vaddhn_u32(color_u32, vshrq_n_u32::<16>(color_u32))
}
#[inline]
#[target_feature(enable = "neon")]
pub unsafe fn mul_color_alpha_u16x8(color: uint16x8_t, alpha: uint16x8_t) -> uint16x8_t {
let res_color_lo_u16 = vrshrn_n_u32::<16>(vmull_u16(vget_low_u16(color), vget_low_u16(alpha)));
let res_color_hi_u16 = vrshrn_n_u32::<16>(vmull_high_u16(color, alpha));
vcombine_u16(res_color_lo_u16, res_color_hi_u16)
}
#[inline]
#[target_feature(enable = "neon")]
pub unsafe fn mul_color_recip_alpha_u8x16(
@@ -367,9 +375,11 @@ pub unsafe fn mul_color_recip_alpha_u8x16(
) -> uint8x16_t {
let color_u16_lo = vreinterpretq_u16_u8(vzip1q_u8(zero, color));
let color_u16_hi = vreinterpretq_u16_u8(vzip2q_u8(zero, color));
let res_u16_lo = multiply_color_to_alpha_u16x8(color_u16_lo, recip_alpha.0);
let res_u16_hi = multiply_color_to_alpha_u16x8(color_u16_hi, recip_alpha.1);
vcombine_u8(vmovn_u16(res_u16_lo), vmovn_u16(res_u16_hi))
let res_u16_lo = mul_color_alpha_u16x8(color_u16_lo, recip_alpha.0);
let res_u16_hi = mul_color_alpha_u16x8(color_u16_hi, recip_alpha.1);
vcombine_u8(vqmovn_u16(res_u16_lo), vqmovn_u16(res_u16_hi))
}
#[inline]
@@ -382,11 +392,12 @@ pub unsafe fn mul_color_recip_alpha_u8x8(
let color_u16_lo = vreinterpret_u16_u8(vzip1_u8(zero, color));
let color_u16_hi = vreinterpret_u16_u8(vzip2_u8(zero, color));
let color_u16 = vcombine_u16(color_u16_lo, color_u16_hi);
let res_u16 = multiply_color_to_alpha_u16x8(color_u16, recip_alpha);
vmovn_u16(res_u16)
let res_color_u16 = mul_color_alpha_u16x8(color_u16, recip_alpha);
vqmovn_u16(res_color_u16)
}
#[inline(always)]
#[inline]
#[target_feature(enable = "neon")]
pub unsafe fn mul_color_recip_alpha_u16x8(
color: uint16x8_t,