Adapted code for aarch64 + Neon architecture to support ImageView and ImageViewMut traits.

This commit is contained in:
Kirill Kuzminykh
2024-04-18 18:53:01 +03:00
parent 9b9766f146
commit 39f942f4c5
15 changed files with 214 additions and 222 deletions
+18 -29
View File
@@ -8,11 +8,11 @@ use super::native;
#[target_feature(enable = "neon")]
pub(crate) unsafe fn multiply_alpha(
src_image: &ImageView<U16x2>,
dst_image: &mut ImageViewMut<U16x2>,
src_view: &impl ImageView<Pixel = U16x2>,
dst_view: &mut impl ImageViewMut<Pixel = U16x2>,
) {
let src_rows = src_image.iter_rows(0);
let dst_rows = dst_image.iter_rows_mut();
let src_rows = src_view.iter_rows(0);
let dst_rows = dst_view.iter_rows_mut(0);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
multiply_alpha_row(src_row, dst_row);
@@ -20,8 +20,8 @@ pub(crate) unsafe fn multiply_alpha(
}
#[target_feature(enable = "neon")]
pub(crate) unsafe fn multiply_alpha_inplace(image: &mut ImageViewMut<U16x2>) {
for row in image.iter_rows_mut() {
pub(crate) unsafe fn multiply_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = U16x2>) {
for row in image_view.iter_rows_mut(0) {
multiply_alpha_row_inplace(row);
}
}
@@ -91,11 +91,11 @@ unsafe fn multiply_alpha_row_inplace(row: &mut [U16x2]) {
#[target_feature(enable = "neon")]
pub(crate) unsafe fn divide_alpha(
src_image: &ImageView<U16x2>,
dst_image: &mut ImageViewMut<U16x2>,
src_view: &impl ImageView<Pixel = U16x2>,
dst_view: &mut impl ImageViewMut<Pixel = U16x2>,
) {
let src_rows = src_image.iter_rows(0);
let dst_rows = dst_image.iter_rows_mut();
let src_rows = src_view.iter_rows(0);
let dst_rows = dst_view.iter_rows_mut(0);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
divide_alpha_row(src_row, dst_row);
@@ -103,8 +103,8 @@ pub(crate) unsafe fn divide_alpha(
}
#[target_feature(enable = "neon")]
pub(crate) unsafe fn divide_alpha_inplace(image: &mut ImageViewMut<U16x2>) {
for row in image.iter_rows_mut() {
pub(crate) unsafe fn divide_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = U16x2>) {
for row in image_view.iter_rows_mut(0) {
divide_alpha_row_inplace(row);
}
}
@@ -188,23 +188,12 @@ unsafe fn divide_alpha_8_pixels(mut pixels: uint16x8x2_t) -> uint16x8x2_t {
let alpha_scaled_f32 = vcvtq_f32_u32(alpha_scaled_u32);
let recip_alpha_hi_f32 = vdivq_f32(alpha_scale, alpha_scaled_f32);
pixels.0 = mul_color_recip_alpha(pixels.0, recip_alpha_lo_f32, recip_alpha_hi_f32, zero);
pixels.0 = neon_utils::mul_color_recip_alpha_u16x8(
pixels.0,
recip_alpha_lo_f32,
recip_alpha_hi_f32,
zero,
);
pixels.0 = vandq_u16(pixels.0, nonzero_alpha_mask);
pixels
}
#[inline(always)]
unsafe fn mul_color_recip_alpha(
color: uint16x8_t,
recip_alpha_lo: float32x4_t,
recip_alpha_hi: float32x4_t,
zero: uint16x8_t,
) -> uint16x8_t {
let color_lo_f32 = vcvtq_f32_u32(vreinterpretq_u32_u16(vzip1q_u16(color, zero)));
let res_lo_u32 = vcvtaq_u32_f32(vmulq_f32(color_lo_f32, recip_alpha_lo));
let color_hi_f32 = vcvtq_f32_u32(vreinterpretq_u32_u16(vzip2q_u16(color, zero)));
let res_hi_u32 = vcvtaq_u32_f32(vmulq_f32(color_hi_f32, recip_alpha_hi));
vcombine_u16(vmovn_u32(res_lo_u32), vmovn_u32(res_hi_u32))
}
+30 -31
View File
@@ -9,11 +9,11 @@ use super::native;
#[target_feature(enable = "neon")]
pub(crate) unsafe fn multiply_alpha(
src_image: &ImageView<U16x4>,
dst_image: &mut ImageViewMut<U16x4>,
src_view: &impl ImageView<Pixel = U16x4>,
dst_view: &mut impl ImageViewMut<Pixel = U16x4>,
) {
let src_rows = src_image.iter_rows(0);
let dst_rows = dst_image.iter_rows_mut();
let src_rows = src_view.iter_rows(0);
let dst_rows = dst_view.iter_rows_mut(0);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
multiply_alpha_row(src_row, dst_row);
@@ -21,8 +21,8 @@ pub(crate) unsafe fn multiply_alpha(
}
#[target_feature(enable = "neon")]
pub(crate) unsafe fn multiply_alpha_inplace(image: &mut ImageViewMut<U16x4>) {
for row in image.iter_rows_mut() {
pub(crate) unsafe fn multiply_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = U16x4>) {
for row in image_view.iter_rows_mut(0) {
multiply_alpha_row_inplace(row);
}
}
@@ -107,11 +107,11 @@ unsafe fn multiply_alpha_row_inplace(row: &mut [U16x4]) {
#[target_feature(enable = "neon")]
pub(crate) unsafe fn divide_alpha(
src_image: &ImageView<U16x4>,
dst_image: &mut ImageViewMut<U16x4>,
src_view: &impl ImageView<Pixel = U16x4>,
dst_view: &mut impl ImageViewMut<Pixel = U16x4>,
) {
let src_rows = src_image.iter_rows(0);
let dst_rows = dst_image.iter_rows_mut();
let src_rows = src_view.iter_rows(0);
let dst_rows = dst_view.iter_rows_mut(0);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
divide_alpha_row(src_row, dst_row);
@@ -119,8 +119,8 @@ pub(crate) unsafe fn divide_alpha(
}
#[target_feature(enable = "neon")]
pub(crate) unsafe fn divide_alpha_inplace(image: &mut ImageViewMut<U16x4>) {
for row in image.iter_rows_mut() {
pub(crate) unsafe fn divide_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = U16x4>) {
for row in image_view.iter_rows_mut(0) {
divide_alpha_row_inplace(row);
}
}
@@ -215,27 +215,26 @@ unsafe fn divide_alpha_8_pixels(mut pixels: uint16x8x4_t) -> uint16x8x4_t {
let alpha_scaled_f32 = vcvtq_f32_u32(alpha_scaled_u32);
let recip_alpha_hi_f32 = vdivq_f32(alpha_scale, alpha_scaled_f32);
pixels.0 = mul_color_recip_alpha(pixels.0, recip_alpha_lo_f32, recip_alpha_hi_f32, zero);
pixels.0 = neon_utils::mul_color_recip_alpha_u16x8(
pixels.0,
recip_alpha_lo_f32,
recip_alpha_hi_f32,
zero,
);
pixels.0 = vandq_u16(pixels.0, nonzero_alpha_mask);
pixels.1 = mul_color_recip_alpha(pixels.1, recip_alpha_lo_f32, recip_alpha_hi_f32, zero);
pixels.1 = neon_utils::mul_color_recip_alpha_u16x8(
pixels.1,
recip_alpha_lo_f32,
recip_alpha_hi_f32,
zero,
);
pixels.1 = vandq_u16(pixels.1, nonzero_alpha_mask);
pixels.2 = mul_color_recip_alpha(pixels.2, recip_alpha_lo_f32, recip_alpha_hi_f32, zero);
pixels.2 = neon_utils::mul_color_recip_alpha_u16x8(
pixels.2,
recip_alpha_lo_f32,
recip_alpha_hi_f32,
zero,
);
pixels.2 = vandq_u16(pixels.2, nonzero_alpha_mask);
pixels
}
#[inline(always)]
unsafe fn mul_color_recip_alpha(
color: uint16x8_t,
recip_alpha_lo: float32x4_t,
recip_alpha_hi: float32x4_t,
zero: uint16x8_t,
) -> uint16x8_t {
let color_lo_f32 = vcvtq_f32_u32(vreinterpretq_u32_u16(vzip1q_u16(color, zero)));
let res_lo_u32 = vcvtaq_u32_f32(vmulq_f32(color_lo_f32, recip_alpha_lo));
let color_hi_f32 = vcvtq_f32_u32(vreinterpretq_u32_u16(vzip2q_u16(color, zero)));
let res_hi_u32 = vcvtaq_u32_f32(vmulq_f32(color_hi_f32, recip_alpha_hi));
vcombine_u16(vmovn_u32(res_lo_u32), vmovn_u32(res_hi_u32))
}
+18 -15
View File
@@ -9,11 +9,11 @@ use super::native;
#[target_feature(enable = "neon")]
pub(crate) unsafe fn multiply_alpha(
src_image: &ImageView<U8x2>,
dst_image: &mut ImageViewMut<U8x2>,
src_view: &impl ImageView<Pixel = U8x2>,
dst_view: &mut impl ImageViewMut<Pixel = U8x2>,
) {
let src_rows = src_image.iter_rows(0);
let dst_rows = dst_image.iter_rows_mut();
let src_rows = src_view.iter_rows(0);
let dst_rows = dst_view.iter_rows_mut(0);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
multiply_alpha_row(src_row, dst_row);
@@ -21,8 +21,8 @@ pub(crate) unsafe fn multiply_alpha(
}
#[target_feature(enable = "neon")]
pub(crate) unsafe fn multiply_alpha_inplace(image: &mut ImageViewMut<U8x2>) {
for row in image.iter_rows_mut() {
pub(crate) unsafe fn multiply_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = U8x2>) {
for row in image_view.iter_rows_mut(0) {
multiply_alpha_row_inplace(row);
}
}
@@ -172,9 +172,12 @@ unsafe fn multiplies_alpha_8_pixels(mut pixels: uint8x8x2_t) -> uint8x8x2_t {
// Divide
#[target_feature(enable = "neon")]
pub(crate) unsafe fn divide_alpha(src_image: &ImageView<U8x2>, dst_image: &mut ImageViewMut<U8x2>) {
let src_rows = src_image.iter_rows(0);
let dst_rows = dst_image.iter_rows_mut();
pub(crate) unsafe fn divide_alpha(
src_view: &impl ImageView<Pixel = U8x2>,
dst_view: &mut impl ImageViewMut<Pixel = U8x2>,
) {
let src_rows = src_view.iter_rows(0);
let dst_rows = dst_view.iter_rows_mut(0);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
divide_alpha_row(src_row, dst_row);
@@ -182,8 +185,8 @@ pub(crate) unsafe fn divide_alpha(src_image: &ImageView<U8x2>, dst_image: &mut I
}
#[target_feature(enable = "neon")]
pub(crate) unsafe fn divide_alpha_inplace(image: &mut ImageViewMut<U8x2>) {
for row in image.iter_rows_mut() {
pub(crate) unsafe fn divide_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = U8x2>) {
for row in image_view.iter_rows_mut(0) {
divide_alpha_row_inplace(row);
}
}
@@ -221,13 +224,13 @@ unsafe fn divide_alpha_row(src_row: &[U8x2], dst_row: &mut [U8x2]) {
if !src_remainder.is_empty() {
let dst_reminder = dst_chunks.into_remainder();
let mut src_pixels = [U8x2::new(0); 8];
let mut src_pixels = [U8x2::new([0; 2]); 8];
src_pixels
.iter_mut()
.zip(src_remainder)
.for_each(|(d, s)| *d = *s);
let mut dst_pixels = [U8x2::new(0); 8];
let mut dst_pixels = [U8x2::new([0; 2]); 8];
let mut pixels = neon_utils::load_deintrel_u8x8x2(&src_pixels, 0);
pixels = divide_alpha_8_pixels(pixels);
let dst_ptr = dst_pixels.as_mut_ptr() as *mut u8;
@@ -267,13 +270,13 @@ unsafe fn divide_alpha_row_inplace(row: &mut [U8x2]) {
let reminder = chunks.into_remainder();
if !reminder.is_empty() {
let mut src_pixels = [U8x2::new(0); 8];
let mut src_pixels = [U8x2::new([0; 2]); 8];
src_pixels
.iter_mut()
.zip(reminder.iter())
.for_each(|(d, s)| *d = *s);
let mut dst_pixels = [U8x2::new(0); 8];
let mut dst_pixels = [U8x2::new([0; 2]); 8];
let mut pixels = neon_utils::load_deintrel_u8x8x2(&src_pixels, 0);
pixels = divide_alpha_8_pixels(pixels);
let dst_ptr = dst_pixels.as_mut_ptr() as *mut u8;
+14 -11
View File
@@ -9,11 +9,11 @@ use super::native;
#[target_feature(enable = "neon")]
pub(crate) unsafe fn multiply_alpha(
src_image: &ImageView<U8x4>,
dst_image: &mut ImageViewMut<U8x4>,
src_view: &impl ImageView<Pixel = U8x4>,
dst_view: &mut impl ImageViewMut<Pixel = U8x4>,
) {
let src_rows = src_image.iter_rows(0);
let dst_rows = dst_image.iter_rows_mut();
let src_rows = src_view.iter_rows(0);
let dst_rows = dst_view.iter_rows_mut(0);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
multiply_alpha_row(src_row, dst_row);
@@ -21,8 +21,8 @@ pub(crate) unsafe fn multiply_alpha(
}
#[target_feature(enable = "neon")]
pub(crate) unsafe fn multiply_alpha_inplace(image: &mut ImageViewMut<U8x4>) {
for row in image.iter_rows_mut() {
pub(crate) unsafe fn multiply_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = U8x4>) {
for row in image_view.iter_rows_mut(0) {
multiply_alpha_row_inplace(row);
}
}
@@ -124,9 +124,12 @@ unsafe fn multiplies_alpha_8_pixles(mut pixels: uint8x8x4_t) -> uint8x8x4_t {
// Divide
#[target_feature(enable = "neon")]
pub(crate) unsafe fn divide_alpha(src_image: &ImageView<U8x4>, dst_image: &mut ImageViewMut<U8x4>) {
let src_rows = src_image.iter_rows(0);
let dst_rows = dst_image.iter_rows_mut();
pub(crate) unsafe fn divide_alpha(
src_view: &impl ImageView<Pixel = U8x4>,
dst_view: &mut impl ImageViewMut<Pixel = U8x4>,
) {
let src_rows = src_view.iter_rows(0);
let dst_rows = dst_view.iter_rows_mut(0);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
divide_alpha_row(src_row, dst_row);
@@ -134,8 +137,8 @@ pub(crate) unsafe fn divide_alpha(src_image: &ImageView<U8x4>, dst_image: &mut I
}
#[target_feature(enable = "neon")]
pub(crate) unsafe fn divide_alpha_inplace(image: &mut ImageViewMut<U8x4>) {
for row in image.iter_rows_mut() {
pub(crate) unsafe fn divide_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = U8x4>) {
for row in image_view.iter_rows_mut(0) {
divide_alpha_row_inline(row);
}
}
+11 -15
View File
@@ -7,18 +7,18 @@ use crate::{ImageView, ImageViewMut};
#[inline]
pub(crate) fn horiz_convolution(
src_image: &ImageView<U16>,
dst_image: &mut ImageViewMut<U16>,
src_view: &impl ImageView<Pixel = U16>,
dst_view: &mut impl ImageViewMut<Pixel = U16>,
offset: u32,
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer32::new(coeffs);
let precision = normalizer.precision();
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_image.height().get();
let dst_height = dst_view.height().get();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_image.iter_4_rows_mut();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, precision);
@@ -26,16 +26,12 @@ pub(crate) fn horiz_convolution(
}
let mut yy = dst_height - dst_height % 4;
while yy < dst_height {
let src_rows = src_view.iter_rows(yy + offset);
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_row(
src_image.get_row(yy + offset).unwrap(),
dst_image.get_row_mut(yy).unwrap(),
&coefficients_chunks,
precision,
);
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, precision);
}
yy += 1;
}
}
@@ -48,7 +44,7 @@ pub(crate) fn horiz_convolution(
#[target_feature(enable = "neon")]
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U16]; 4],
dst_rows: [&mut &mut [U16]; 4],
dst_rows: [&mut [U16]; 4],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
precision: u8,
) {
@@ -132,7 +128,7 @@ unsafe fn horiz_convolution_four_rows(
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "neon")]
unsafe fn horiz_convolution_row(
unsafe fn horiz_convolution_one_row(
src_row: &[U16],
dst_row: &mut [U16],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
+12 -16
View File
@@ -7,18 +7,18 @@ use crate::{ImageView, ImageViewMut};
#[inline]
pub(crate) fn horiz_convolution(
src_image: &ImageView<U16x2>,
dst_image: &mut ImageViewMut<U16x2>,
src_view: &impl ImageView<Pixel = U16x2>,
dst_view: &mut impl ImageViewMut<Pixel = U16x2>,
offset: u32,
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer32::new(coeffs);
let precision = normalizer.precision();
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_image.height().get();
let dst_height = dst_view.height().get();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_image.iter_4_rows_mut();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, precision);
@@ -26,16 +26,12 @@ pub(crate) fn horiz_convolution(
}
let mut yy = dst_height - dst_height % 4;
while yy < dst_height {
let src_rows = src_view.iter_rows(yy + offset);
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_row(
src_image.get_row(yy + offset).unwrap(),
dst_image.get_row_mut(yy).unwrap(),
&coefficients_chunks,
precision,
);
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, precision);
}
yy += 1;
}
}
@@ -48,12 +44,12 @@ pub(crate) fn horiz_convolution(
#[target_feature(enable = "neon")]
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U16x2]; 4],
dst_rows: [&mut &mut [U16x2]; 4],
dst_rows: [&mut [U16x2]; 4],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
precision: u8,
) {
let initial = vdupq_n_s64(1i64 << (precision - 1));
let zero_u16x8 = vdupq_n_u16(0);
// let zero_u16x8 = vdupq_n_u16(0);
let zero_u16x4 = vdup_n_u16(0);
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
@@ -119,7 +115,7 @@ unsafe fn horiz_convolution_four_rows(
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "neon")]
unsafe fn horiz_convolution_row(
unsafe fn horiz_convolution_one_row(
src_row: &[U16x2],
dst_row: &mut [U16x2],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
+6 -6
View File
@@ -7,8 +7,8 @@ use crate::{ImageView, ImageViewMut};
#[inline]
pub(crate) fn horiz_convolution(
src_image: &ImageView<U16x3>,
dst_image: &mut ImageViewMut<U16x3>,
src_view: &impl ImageView<Pixel = U16x3>,
dst_view: &mut impl ImageViewMut<Pixel = U16x3>,
offset: u32,
coeffs: Coefficients,
) {
@@ -16,11 +16,11 @@ pub(crate) fn horiz_convolution(
let precision = normalizer.precision();
let coefficients_chunks = normalizer.normalized_chunks();
let src_iter = src_image.iter_rows(offset);
let dst_iter = dst_image.iter_rows_mut();
let src_iter = src_view.iter_rows(offset);
let dst_iter = dst_view.iter_rows_mut(0);
for (src_row, dst_row) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_row(src_row, dst_row, &coefficients_chunks, precision);
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, precision);
}
}
}
@@ -31,7 +31,7 @@ pub(crate) fn horiz_convolution(
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "neon")]
unsafe fn horiz_convolution_row(
unsafe fn horiz_convolution_one_row(
src_row: &[U16x3],
dst_row: &mut [U16x3],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
+13 -17
View File
@@ -7,35 +7,31 @@ use crate::{ImageView, ImageViewMut};
#[inline]
pub(crate) fn horiz_convolution(
src_image: &ImageView<U16x4>,
dst_image: &mut ImageViewMut<U16x4>,
src_view: &impl ImageView<Pixel = U16x4>,
dst_view: &mut impl ImageViewMut<Pixel = U16x4>,
offset: u32,
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer32::new(coeffs);
let precision = normalizer.precision();
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_image.height().get();
let dst_height = dst_view.height().get();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_image.iter_4_rows_mut();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_4_rows(src_rows, dst_rows, &coefficients_chunks, precision);
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, precision);
}
}
let mut yy = dst_height - dst_height % 4;
while yy < dst_height {
let src_rows = src_view.iter_rows(yy + offset);
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_row(
src_image.get_row(yy + offset).unwrap(),
dst_image.get_row_mut(yy).unwrap(),
&coefficients_chunks,
precision,
);
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, precision);
}
yy += 1;
}
}
@@ -46,9 +42,9 @@ pub(crate) fn horiz_convolution(
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "neon")]
unsafe fn horiz_convolution_4_rows(
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U16x4]; 4],
dst_rows: [&mut &mut [U16x4]; 4],
dst_rows: [&mut [U16x4]; 4],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
precision: u8,
) {
@@ -171,7 +167,7 @@ unsafe fn horiz_convolution_4_rows(
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "neon")]
unsafe fn horiz_convolution_row(
unsafe fn horiz_convolution_one_row(
src_row: &[U16x4],
dst_row: &mut [U16x4],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
+12 -16
View File
@@ -7,34 +7,30 @@ use crate::{ImageView, ImageViewMut};
#[inline]
pub(crate) fn horiz_convolution(
src_image: &ImageView<U8>,
dst_image: &mut ImageViewMut<U8>,
src_view: &impl ImageView<Pixel = U8>,
dst_view: &mut impl ImageViewMut<Pixel = U8>,
offset: u32,
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer16::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_image.height().get();
let dst_height = dst_view.height().get();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_image.iter_4_rows_mut();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut(0);
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
}
}
let mut yy = dst_height - dst_height % 4;
while yy < dst_height {
let yy = dst_height - dst_height % 4;
let src_rows = src_view.iter_rows(yy + offset);
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_row(
src_image.get_row(yy + offset).unwrap(),
dst_image.get_row_mut(yy).unwrap(),
&coefficients_chunks,
&normalizer,
);
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
}
yy += 1;
}
}
@@ -47,7 +43,7 @@ pub(crate) fn horiz_convolution(
#[target_feature(enable = "neon")]
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U8]; 4],
dst_rows: [&mut &mut [U8]; 4],
dst_rows: [&mut [U8]; 4],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
normalizer: &optimisations::Normalizer16,
) {
@@ -154,7 +150,7 @@ unsafe fn horiz_convolution_four_rows(
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "neon")]
unsafe fn horiz_convolution_row(
unsafe fn horiz_convolution_one_row(
src_row: &[U8],
dst_row: &mut [U8],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
+7 -7
View File
@@ -14,7 +14,7 @@ pub(crate) fn horiz_convolution(
let normalizer = optimisations::Normalizer16::new(coeffs);
let precision = normalizer.precision();
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_image.height().get();
let dst_height = dst_view.height().get();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
@@ -29,7 +29,7 @@ pub(crate) fn horiz_convolution(
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, &normalizer);
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, precision);
}
}
}
@@ -117,7 +117,7 @@ unsafe fn horiz_convolution_four_rows(
let coeff0 = vzip1_s16(coeffs_i16x4, coeffs_i16x4);
let coeff1 = vzip2_s16(coeffs_i16x4, coeffs_i16x4);
let mut four_pixels = [U8x2::new(0); 4];
let mut four_pixels = [U8x2::new([0; 2]); 4];
for i in 0..4 {
four_pixels
@@ -154,7 +154,7 @@ unsafe fn horiz_convolution_four_rows(
vqmovn_s32(sss),
vreinterpret_s16_u8(zero_u8x8),
)));
dst_rows[i].get_unchecked_mut(dst_x).0 = vget_lane_u16::<0>(s);
dst_rows[i].get_unchecked_mut(dst_x).0 = vget_lane_u16::<0>(s).to_le_bytes();
}
}
}
@@ -165,7 +165,7 @@ unsafe fn horiz_convolution_four_rows(
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "neon")]
unsafe fn horiz_convolution_row(
unsafe fn horiz_convolution_one_row(
src_row: &[U8x2],
dst_row: &mut [U8x2],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
@@ -215,7 +215,7 @@ unsafe fn horiz_convolution_row(
.zip(coeffs)
.for_each(|(d, s)| *d = *s);
let mut four_pixels = [U8x2::new(0); 4];
let mut four_pixels = [U8x2::new([0; 2]); 4];
four_pixels
.iter_mut()
.zip(src_row.get_unchecked(x..))
@@ -238,7 +238,7 @@ unsafe fn horiz_convolution_row(
vqmovn_s32(sss),
vreinterpret_s16_u8(zero_u8x8),
)));
dst_row.get_unchecked_mut(dst_x).0 = vget_lane_u16::<0>(s);
dst_row.get_unchecked_mut(dst_x).0 = vget_lane_u16::<0>(s).to_le_bytes();
}
}
+11 -15
View File
@@ -7,18 +7,18 @@ use crate::{ImageView, ImageViewMut};
#[inline]
pub(crate) fn horiz_convolution(
src_image: &ImageView<U8x3>,
dst_image: &mut ImageViewMut<U8x3>,
src_view: &impl ImageView<Pixel = U8x3>,
dst_view: &mut impl ImageViewMut<Pixel = U8x3>,
offset: u32,
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer16::new(coeffs);
let precision = normalizer.precision();
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_image.height().get();
let dst_height = dst_view.height().get();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_image.iter_4_rows_mut();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, precision);
@@ -26,16 +26,12 @@ pub(crate) fn horiz_convolution(
}
let mut yy = dst_height - dst_height % 4;
while yy < dst_height {
let src_rows = src_view.iter_rows(yy + offset);
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_row(
src_image.get_row(yy + offset).unwrap(),
dst_image.get_row_mut(yy).unwrap(),
&coefficients_chunks,
precision,
);
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, precision);
}
yy += 1;
}
}
@@ -48,7 +44,7 @@ pub(crate) fn horiz_convolution(
#[target_feature(enable = "neon")]
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U8x3]; 4],
dst_rows: [&mut &mut [U8x3]; 4],
dst_rows: [&mut [U8x3]; 4],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
precision: u8,
) {
@@ -121,7 +117,7 @@ unsafe fn horiz_convolution_four_rows(
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "neon")]
unsafe fn horiz_convolution_row(
unsafe fn horiz_convolution_one_row(
src_row: &[U8x3],
dst_row: &mut [U8x3],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
+13 -17
View File
@@ -7,35 +7,31 @@ use crate::{ImageView, ImageViewMut};
#[inline]
pub(crate) fn horiz_convolution(
src_image: &ImageView<U8x4>,
dst_image: &mut ImageViewMut<U8x4>,
src_view: &impl ImageView<Pixel = U8x4>,
dst_view: &mut impl ImageViewMut<Pixel = U8x4>,
offset: u32,
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer16::new(coeffs);
let precision = normalizer.precision();
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_image.height().get();
let dst_height = dst_view.height().get();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_image.iter_4_rows_mut();
let src_iter = src_view.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_8u4x(src_rows, dst_rows, &coefficients_chunks, precision);
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, precision);
}
}
let mut yy = dst_height - dst_height % 4;
while yy < dst_height {
let src_rows = src_view.iter_rows(yy + offset);
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_8u(
src_image.get_row(yy + offset).unwrap(),
dst_image.get_row_mut(yy).unwrap(),
&coefficients_chunks,
precision,
);
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, precision);
}
yy += 1;
}
}
@@ -46,9 +42,9 @@ pub(crate) fn horiz_convolution(
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "neon")]
unsafe fn horiz_convolution_8u4x(
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U8x4]; 4],
dst_rows: [&mut &mut [U8x4]; 4],
dst_rows: [&mut [U8x4]; 4],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
precision: u8,
) {
@@ -196,7 +192,7 @@ unsafe fn horiz_convolution_8u4x(
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "neon")]
unsafe fn horiz_convolution_8u(
unsafe fn horiz_convolution_one_row(
src_row: &[U8x4],
dst_row: &mut [U8x4],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
+13 -11
View File
@@ -3,28 +3,30 @@ use std::mem::transmute;
use crate::convolution::{optimisations, Coefficients};
use crate::neon_utils;
use crate::pixels::PixelExt;
use crate::pixels::InnerPixel;
use crate::{ImageView, ImageViewMut};
pub(crate) fn vert_convolution<T: PixelExt<Component = u16>>(
src_image: &ImageView<T>,
dst_image: &mut ImageViewMut<T>,
pub(crate) fn vert_convolution<T>(
src_view: &impl ImageView<Pixel = T>,
dst_view: &mut impl ImageViewMut<Pixel = T>,
offset: u32,
coeffs: Coefficients,
) {
) where
T: InnerPixel<Component = u16>,
{
let normalizer = optimisations::Normalizer32::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let precision = normalizer.precision();
let initial = 1i64 << (precision - 1);
let start_src_x = offset as usize * T::count_of_components();
let mut tmp_dst = vec![0i64; dst_image.width().get() as usize * T::count_of_components()];
let mut tmp_dst = vec![0i64; dst_view.width().get() as usize * T::count_of_components()];
let tmp_buf = tmp_dst.as_mut_slice();
let dst_rows = dst_image.iter_rows_mut();
let dst_rows = dst_view.iter_rows_mut(0);
for (dst_row, coeffs_chunk) in dst_rows.zip(coefficients_chunks) {
tmp_buf.fill(initial);
unsafe {
vert_convolution_into_one_row_i64(src_image, tmp_buf, start_src_x, coeffs_chunk);
vert_convolution_into_one_row_i64(src_view, tmp_buf, start_src_x, coeffs_chunk);
let dst_comp = T::components_mut(dst_row);
macro_rules! call {
($imm8:expr) => {{
@@ -37,8 +39,8 @@ pub(crate) fn vert_convolution<T: PixelExt<Component = u16>>(
}
#[target_feature(enable = "neon")]
unsafe fn vert_convolution_into_one_row_i64<T: PixelExt<Component = u16>>(
src_img: &ImageView<T>,
unsafe fn vert_convolution_into_one_row_i64<T: InnerPixel<Component = u16>>(
src_view: &impl ImageView<Pixel = T>,
dst_buf: &mut [i64],
start_src_x: usize,
coeffs_chunk: optimisations::CoefficientsI32Chunk,
@@ -50,7 +52,7 @@ unsafe fn vert_convolution_into_one_row_i64<T: PixelExt<Component = u16>>(
let zero_u16x8 = vdupq_n_u16(0);
let zero_u16x4 = vdup_n_u16(0);
for (s_row, &coeff) in src_img.iter_rows(y_start).zip(coeffs) {
for (s_row, &coeff) in src_view.iter_rows(y_start).zip(coeffs) {
let components = T::components(s_row);
let coeff_i32x2 = vdup_n_s32(coeff);
+16 -12
View File
@@ -3,28 +3,30 @@ use std::mem::transmute;
use crate::convolution::{optimisations, Coefficients};
use crate::neon_utils;
use crate::pixels::PixelExt;
use crate::pixels::InnerPixel;
use crate::{ImageView, ImageViewMut};
pub(crate) fn vert_convolution<T: PixelExt<Component = u8>>(
src_image: &ImageView<T>,
dst_image: &mut ImageViewMut<T>,
pub(crate) fn vert_convolution<T>(
src_view: &impl ImageView<Pixel = T>,
dst_view: &mut impl ImageViewMut<Pixel = T>,
offset: u32,
coeffs: Coefficients,
) {
) where
T: InnerPixel<Component = u8>,
{
let normalizer = optimisations::Normalizer16::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let precision = normalizer.precision();
let initial = 1 << (precision - 1);
let start_src_x = offset as usize * T::count_of_components();
let mut tmp_dst = vec![0i32; dst_image.width().get() as usize * T::count_of_components()];
let mut tmp_dst = vec![0i32; dst_view.width().get() as usize * T::count_of_components()];
let tmp_buf = tmp_dst.as_mut_slice();
let dst_rows = dst_image.iter_rows_mut();
let dst_rows = dst_view.iter_rows_mut(0);
for (dst_row, coeffs_chunk) in dst_rows.zip(coefficients_chunks) {
tmp_buf.fill(initial);
unsafe {
vert_convolution_into_one_row_i32(src_image, tmp_buf, start_src_x, coeffs_chunk);
vert_convolution_into_one_row_i32(src_view, tmp_buf, start_src_x, coeffs_chunk);
let dst_comp = T::components_mut(dst_row);
macro_rules! call {
($imm8:expr) => {{
@@ -37,12 +39,14 @@ pub(crate) fn vert_convolution<T: PixelExt<Component = u8>>(
}
#[target_feature(enable = "neon")]
unsafe fn vert_convolution_into_one_row_i32<T: PixelExt<Component = u8>>(
src_img: &ImageView<T>,
unsafe fn vert_convolution_into_one_row_i32<T>(
src_view: &impl ImageView<Pixel = T>,
dst_buf: &mut [i32],
start_src_x: usize,
coeffs_chunk: optimisations::CoefficientsI16Chunk,
) {
) where
T: InnerPixel<Component = u8>,
{
let width = dst_buf.len();
let y_start = coeffs_chunk.start;
let coeffs = coeffs_chunk.values;
@@ -50,7 +54,7 @@ unsafe fn vert_convolution_into_one_row_i32<T: PixelExt<Component = u8>>(
let zero_u8x16 = vdupq_n_u8(0);
let zero_u8x8 = vdup_n_u8(0);
for (s_row, &coeff) in src_img.iter_rows(y_start).zip(coeffs) {
for (s_row, &coeff) in src_view.iter_rows(y_start).zip(coeffs) {
let components = T::components(s_row);
let coeff_i16x4 = vdup_n_s16(coeff);
+20 -4
View File
@@ -367,9 +367,8 @@ pub unsafe fn mul_color_recip_alpha_u8x16(
) -> uint8x16_t {
let color_u16_lo = vreinterpretq_u16_u8(vzip1q_u8(zero, color));
let color_u16_hi = vreinterpretq_u16_u8(vzip2q_u8(zero, color));
let res_u16_lo = mulhi_u16x8(color_u16_lo, recip_alpha.0);
let res_u16_hi = mulhi_u16x8(color_u16_hi, recip_alpha.1);
let res_u16_lo = multiply_color_to_alpha_u16x8(color_u16_lo, recip_alpha.0);
let res_u16_hi = multiply_color_to_alpha_u16x8(color_u16_hi, recip_alpha.1);
vcombine_u8(vmovn_u16(res_u16_lo), vmovn_u16(res_u16_hi))
}
@@ -383,6 +382,23 @@ pub unsafe fn mul_color_recip_alpha_u8x8(
let color_u16_lo = vreinterpret_u16_u8(vzip1_u8(zero, color));
let color_u16_hi = vreinterpret_u16_u8(vzip2_u8(zero, color));
let color_u16 = vcombine_u16(color_u16_lo, color_u16_hi);
let res_u16 = mulhi_u16x8(color_u16, recip_alpha);
let res_u16 = multiply_color_to_alpha_u16x8(color_u16, recip_alpha);
vmovn_u16(res_u16)
}
#[inline(always)]
#[target_feature(enable = "neon")]
pub unsafe fn mul_color_recip_alpha_u16x8(
color: uint16x8_t,
recip_alpha_lo: float32x4_t,
recip_alpha_hi: float32x4_t,
zero: uint16x8_t,
) -> uint16x8_t {
let color_lo_f32 = vcvtq_f32_u32(vreinterpretq_u32_u16(vzip1q_u16(color, zero)));
let res_lo_u32 = vcvtaq_u32_f32(vmulq_f32(color_lo_f32, recip_alpha_lo));
let color_hi_f32 = vcvtq_f32_u32(vreinterpretq_u32_u16(vzip2q_u16(color, zero)));
let res_hi_u32 = vcvtaq_u32_f32(vmulq_f32(color_hi_f32, recip_alpha_hi));
vcombine_u16(vmovn_u32(res_lo_u32), vmovn_u32(res_hi_u32))
}