mirror of
https://github.com/Cykooz/fast_image_resize.git
synced 2026-10-08 01:11:09 +00:00
Fixed Neon implementations.
This commit is contained in:
@@ -15,7 +15,7 @@ pub(crate) fn horiz_convolution(
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height().get();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
@@ -25,7 +25,7 @@ pub(crate) fn horiz_convolution(
|
||||
}
|
||||
}
|
||||
|
||||
let mut yy = dst_height - dst_height % 4;
|
||||
let yy = dst_height - dst_height % 4;
|
||||
let src_rows = src_view.iter_rows(yy + offset);
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
|
||||
@@ -15,7 +15,7 @@ pub(crate) fn horiz_convolution(
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height().get();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
@@ -25,7 +25,7 @@ pub(crate) fn horiz_convolution(
|
||||
}
|
||||
}
|
||||
|
||||
let mut yy = dst_height - dst_height % 4;
|
||||
let yy = dst_height - dst_height % 4;
|
||||
let src_rows = src_view.iter_rows(yy + offset);
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
|
||||
@@ -15,7 +15,7 @@ pub(crate) fn horiz_convolution(
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height().get();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
@@ -25,7 +25,7 @@ pub(crate) fn horiz_convolution(
|
||||
}
|
||||
}
|
||||
|
||||
let mut yy = dst_height - dst_height % 4;
|
||||
let yy = dst_height - dst_height % 4;
|
||||
let src_rows = src_view.iter_rows(yy + offset);
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
|
||||
@@ -14,10 +14,10 @@ pub(crate) fn horiz_convolution(
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height().get();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut(0);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
|
||||
@@ -14,7 +14,7 @@ pub(crate) fn horiz_convolution(
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height().get();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
@@ -24,7 +24,7 @@ pub(crate) fn horiz_convolution(
|
||||
}
|
||||
}
|
||||
|
||||
let mut yy = dst_height - dst_height % 4;
|
||||
let yy = dst_height - dst_height % 4;
|
||||
let src_rows = src_view.iter_rows(yy + offset);
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
|
||||
@@ -15,7 +15,7 @@ pub(crate) fn horiz_convolution(
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height().get();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
@@ -25,7 +25,7 @@ pub(crate) fn horiz_convolution(
|
||||
}
|
||||
}
|
||||
|
||||
let mut yy = dst_height - dst_height % 4;
|
||||
let yy = dst_height - dst_height % 4;
|
||||
let src_rows = src_view.iter_rows(yy + offset);
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
|
||||
@@ -15,7 +15,7 @@ pub(crate) fn horiz_convolution(
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height().get();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
@@ -25,7 +25,7 @@ pub(crate) fn horiz_convolution(
|
||||
}
|
||||
}
|
||||
|
||||
let mut yy = dst_height - dst_height % 4;
|
||||
let yy = dst_height - dst_height % 4;
|
||||
let src_rows = src_view.iter_rows(yy + offset);
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
|
||||
@@ -20,7 +20,7 @@ pub(crate) fn vert_convolution<T>(
|
||||
let initial = 1i64 << (precision - 1);
|
||||
let start_src_x = offset as usize * T::count_of_components();
|
||||
|
||||
let mut tmp_dst = vec![0i64; dst_view.width().get() as usize * T::count_of_components()];
|
||||
let mut tmp_dst = vec![0i64; dst_view.width() as usize * T::count_of_components()];
|
||||
let tmp_buf = tmp_dst.as_mut_slice();
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
for (dst_row, coeffs_chunk) in dst_rows.zip(coefficients_chunks) {
|
||||
|
||||
@@ -20,7 +20,7 @@ pub(crate) fn vert_convolution<T>(
|
||||
let initial = 1 << (precision - 1);
|
||||
let start_src_x = offset as usize * T::count_of_components();
|
||||
|
||||
let mut tmp_dst = vec![0i32; dst_view.width().get() as usize * T::count_of_components()];
|
||||
let mut tmp_dst = vec![0i32; dst_view.width() as usize * T::count_of_components()];
|
||||
let tmp_buf = tmp_dst.as_mut_slice();
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
for (dst_row, coeffs_chunk) in dst_rows.zip(coefficients_chunks) {
|
||||
|
||||
+37
-26
@@ -60,10 +60,10 @@ pub unsafe fn load_deintrel_u8x16x2<T>(buf: &[T], index: usize) -> uint8x16x2_t
|
||||
vld2q_u8(buf.get_unchecked(index..).as_ptr() as *const u8)
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
pub unsafe fn load_deintrel_u8x16x3<T>(buf: &[T], index: usize) -> uint8x16x3_t {
|
||||
vld3q_u8(buf.get_unchecked(index..).as_ptr() as *const u8)
|
||||
}
|
||||
// #[inline(always)]
|
||||
// pub unsafe fn load_deintrel_u8x16x3<T>(buf: &[T], index: usize) -> uint8x16x3_t {
|
||||
// vld3q_u8(buf.get_unchecked(index..).as_ptr() as *const u8)
|
||||
// }
|
||||
|
||||
#[inline(always)]
|
||||
pub unsafe fn load_deintrel_u8x16x4<T>(buf: &[T], index: usize) -> uint8x16x4_t {
|
||||
@@ -284,26 +284,26 @@ pub unsafe fn create_u8x16_from_one_u32<T>(buf: &[T], index: usize) -> uint8x16_
|
||||
/// Multiply the packed unsigned 16-bit integers in a and b, producing
|
||||
/// intermediate 32-bit integers, and store the high 16 bits of the intermediate
|
||||
/// integers in dst.
|
||||
#[inline(always)]
|
||||
pub unsafe fn mulhi_u16x8(a: uint16x8_t, b: uint16x8_t) -> uint16x8_t {
|
||||
let a3210 = vget_low_u16(a);
|
||||
let b3210 = vget_low_u16(b);
|
||||
let ab3210 = vmull_u16(a3210, b3210);
|
||||
let ab7654 = vmull_high_u16(a, b);
|
||||
vuzp2q_u16(vreinterpretq_u16_u32(ab3210), vreinterpretq_u16_u32(ab7654))
|
||||
}
|
||||
// #[inline(always)]
|
||||
// pub unsafe fn mulhi_u16x8(a: uint16x8_t, b: uint16x8_t) -> uint16x8_t {
|
||||
// let a3210 = vget_low_u16(a);
|
||||
// let b3210 = vget_low_u16(b);
|
||||
// let ab3210 = vmull_u16(a3210, b3210);
|
||||
// let ab7654 = vmull_high_u16(a, b);
|
||||
// vuzp2q_u16(vreinterpretq_u16_u32(ab3210), vreinterpretq_u16_u32(ab7654))
|
||||
// }
|
||||
|
||||
/// Multiply the packed unsigned 32-bit integers in a and b, producing
|
||||
/// intermediate 64-bit integers, and store the high 32 bits of the intermediate
|
||||
/// integers in dst.
|
||||
#[inline(always)]
|
||||
pub unsafe fn mulhi_u32x4(a: uint32x4_t, b: uint32x4_t) -> uint32x4_t {
|
||||
let a3210 = vget_low_u32(a);
|
||||
let b3210 = vget_low_u32(b);
|
||||
let ab3210 = vmull_u32(a3210, b3210);
|
||||
let ab7654 = vmull_high_u32(a, b);
|
||||
vuzp2q_u32(vreinterpretq_u32_u64(ab3210), vreinterpretq_u32_u64(ab7654))
|
||||
}
|
||||
// #[inline(always)]
|
||||
// pub unsafe fn mulhi_u32x4(a: uint32x4_t, b: uint32x4_t) -> uint32x4_t {
|
||||
// let a3210 = vget_low_u32(a);
|
||||
// let b3210 = vget_low_u32(b);
|
||||
// let ab3210 = vmull_u32(a3210, b3210);
|
||||
// let ab7654 = vmull_high_u32(a, b);
|
||||
// vuzp2q_u32(vreinterpretq_u32_u64(ab3210), vreinterpretq_u32_u64(ab7654))
|
||||
// }
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "neon")]
|
||||
@@ -358,6 +358,14 @@ pub unsafe fn multiply_color_to_alpha_u16x4(color: uint16x4_t, alpha: uint16x4_t
|
||||
vaddhn_u32(color_u32, vshrq_n_u32::<16>(color_u32))
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "neon")]
|
||||
pub unsafe fn mul_color_alpha_u16x8(color: uint16x8_t, alpha: uint16x8_t) -> uint16x8_t {
|
||||
let res_color_lo_u16 = vrshrn_n_u32::<16>(vmull_u16(vget_low_u16(color), vget_low_u16(alpha)));
|
||||
let res_color_hi_u16 = vrshrn_n_u32::<16>(vmull_high_u16(color, alpha));
|
||||
vcombine_u16(res_color_lo_u16, res_color_hi_u16)
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "neon")]
|
||||
pub unsafe fn mul_color_recip_alpha_u8x16(
|
||||
@@ -367,9 +375,11 @@ pub unsafe fn mul_color_recip_alpha_u8x16(
|
||||
) -> uint8x16_t {
|
||||
let color_u16_lo = vreinterpretq_u16_u8(vzip1q_u8(zero, color));
|
||||
let color_u16_hi = vreinterpretq_u16_u8(vzip2q_u8(zero, color));
|
||||
let res_u16_lo = multiply_color_to_alpha_u16x8(color_u16_lo, recip_alpha.0);
|
||||
let res_u16_hi = multiply_color_to_alpha_u16x8(color_u16_hi, recip_alpha.1);
|
||||
vcombine_u8(vmovn_u16(res_u16_lo), vmovn_u16(res_u16_hi))
|
||||
|
||||
let res_u16_lo = mul_color_alpha_u16x8(color_u16_lo, recip_alpha.0);
|
||||
let res_u16_hi = mul_color_alpha_u16x8(color_u16_hi, recip_alpha.1);
|
||||
|
||||
vcombine_u8(vqmovn_u16(res_u16_lo), vqmovn_u16(res_u16_hi))
|
||||
}
|
||||
|
||||
#[inline]
|
||||
@@ -382,11 +392,12 @@ pub unsafe fn mul_color_recip_alpha_u8x8(
|
||||
let color_u16_lo = vreinterpret_u16_u8(vzip1_u8(zero, color));
|
||||
let color_u16_hi = vreinterpret_u16_u8(vzip2_u8(zero, color));
|
||||
let color_u16 = vcombine_u16(color_u16_lo, color_u16_hi);
|
||||
let res_u16 = multiply_color_to_alpha_u16x8(color_u16, recip_alpha);
|
||||
vmovn_u16(res_u16)
|
||||
|
||||
let res_color_u16 = mul_color_alpha_u16x8(color_u16, recip_alpha);
|
||||
vqmovn_u16(res_color_u16)
|
||||
}
|
||||
|
||||
#[inline(always)]
|
||||
#[inline]
|
||||
#[target_feature(enable = "neon")]
|
||||
pub unsafe fn mul_color_recip_alpha_u16x8(
|
||||
color: uint16x8_t,
|
||||
|
||||
Reference in New Issue
Block a user