Optimized Neon implementation.

This commit is contained in:
Kirill Kuzminykh
2024-04-19 18:14:29 +03:00
parent d31c9b7a49
commit 10ae414752
7 changed files with 183 additions and 158 deletions
+26 -23
View File
@@ -14,6 +14,21 @@ pub(crate) fn horiz_convolution(
) {
let normalizer = optimisations::Normalizer32::new(coeffs);
let precision = normalizer.precision();
macro_rules! call {
($imm8:expr) => {{
horiz_convolution_p::<$imm8>(src_view, dst_view, offset, normalizer);
}};
}
constify_64_imm8!(precision, call);
}
fn horiz_convolution_p<const PRECISION: i32>(
src_view: &impl ImageView<Pixel = U16>,
dst_view: &mut impl ImageViewMut<Pixel = U16>,
offset: u32,
normalizer: optimisations::Normalizer32,
) {
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_view.height();
@@ -21,7 +36,7 @@ pub(crate) fn horiz_convolution(
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, precision);
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &coefficients_chunks);
}
}
@@ -30,7 +45,7 @@ pub(crate) fn horiz_convolution(
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, precision);
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &coefficients_chunks);
}
}
}
@@ -42,13 +57,12 @@ pub(crate) fn horiz_convolution(
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "neon")]
unsafe fn horiz_convolution_four_rows(
unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
src_rows: [&[U16]; 4],
dst_rows: [&mut [U16]; 4],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
precision: u8,
) {
let initial = vdupq_n_s64(1i64 << (precision - 2));
let initial = vdupq_n_s64(1i64 << (PRECISION - 2));
let zero_u16x4 = vdup_n_u16(0);
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
@@ -105,15 +119,10 @@ unsafe fn horiz_convolution_four_rows(
vadd_s64(vget_low_s64(sss_a[2]), vget_high_s64(sss_a[2])),
vadd_s64(vget_low_s64(sss_a[3]), vget_high_s64(sss_a[3])),
];
macro_rules! call {
($imm8:expr) => {{
sss_a_i64[0] = vshr_n_s64::<$imm8>(sss_a_i64[0]);
sss_a_i64[1] = vshr_n_s64::<$imm8>(sss_a_i64[1]);
sss_a_i64[2] = vshr_n_s64::<$imm8>(sss_a_i64[2]);
sss_a_i64[3] = vshr_n_s64::<$imm8>(sss_a_i64[3]);
}};
}
constify_64_imm8!(precision, call);
sss_a_i64[0] = vshr_n_s64::<PRECISION>(sss_a_i64[0]);
sss_a_i64[1] = vshr_n_s64::<PRECISION>(sss_a_i64[1]);
sss_a_i64[2] = vshr_n_s64::<PRECISION>(sss_a_i64[2]);
sss_a_i64[3] = vshr_n_s64::<PRECISION>(sss_a_i64[3]);
for i in 0..4 {
let res = vdupd_lane_s64::<0>(sss_a_i64[i]);
@@ -128,13 +137,12 @@ unsafe fn horiz_convolution_four_rows(
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "neon")]
unsafe fn horiz_convolution_one_row(
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
src_row: &[U16],
dst_row: &mut [U16],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
precision: u8,
) {
let initial = vdupq_n_s64(1i64 << (precision - 2));
let initial = vdupq_n_s64(1i64 << (PRECISION - 2));
let zero_u16x8 = vdupq_n_u16(0);
let zero_u16x4 = vdup_n_u16(0);
@@ -192,12 +200,7 @@ unsafe fn horiz_convolution_one_row(
}
let mut sss_i64 = vadd_s64(vget_low_s64(sss), vget_high_s64(sss));
macro_rules! call {
($imm8:expr) => {{
sss_i64 = vshr_n_s64::<$imm8>(sss_i64);
}};
}
constify_64_imm8!(precision, call);
sss_i64 = vshr_n_s64::<PRECISION>(sss_i64);
let res = vdupd_lane_s64::<0>(sss_i64);
dst_row.get_unchecked_mut(dst_x).0 = vqmovns_u32(vqmovund_s64(res));
+26 -23
View File
@@ -14,6 +14,21 @@ pub(crate) fn horiz_convolution(
) {
let normalizer = optimisations::Normalizer32::new(coeffs);
let precision = normalizer.precision();
macro_rules! call {
($imm8:expr) => {{
horiz_convolution_p::<$imm8>(src_view, dst_view, offset, normalizer);
}};
}
constify_64_imm8!(precision, call);
}
fn horiz_convolution_p<const PRECISION: i32>(
src_view: &impl ImageView<Pixel = U16x2>,
dst_view: &mut impl ImageViewMut<Pixel = U16x2>,
offset: u32,
normalizer: optimisations::Normalizer32,
) {
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_view.height();
@@ -21,7 +36,7 @@ pub(crate) fn horiz_convolution(
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, precision);
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &coefficients_chunks);
}
}
@@ -30,7 +45,7 @@ pub(crate) fn horiz_convolution(
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, precision);
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &coefficients_chunks);
}
}
}
@@ -42,13 +57,12 @@ pub(crate) fn horiz_convolution(
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "neon")]
unsafe fn horiz_convolution_four_rows(
unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
src_rows: [&[U16x2]; 4],
dst_rows: [&mut [U16x2]; 4],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
precision: u8,
) {
let initial = vdupq_n_s64(1i64 << (precision - 1));
let initial = vdupq_n_s64(1i64 << (PRECISION - 1));
// let zero_u16x8 = vdupq_n_u16(0);
let zero_u16x4 = vdup_n_u16(0);
@@ -86,15 +100,10 @@ unsafe fn horiz_convolution_four_rows(
}
}
macro_rules! call {
($imm8:expr) => {{
sss_a[0] = vshrq_n_s64::<$imm8>(sss_a[0]);
sss_a[1] = vshrq_n_s64::<$imm8>(sss_a[1]);
sss_a[2] = vshrq_n_s64::<$imm8>(sss_a[2]);
sss_a[3] = vshrq_n_s64::<$imm8>(sss_a[3]);
}};
}
constify_64_imm8!(precision, call);
sss_a[0] = vshrq_n_s64::<PRECISION>(sss_a[0]);
sss_a[1] = vshrq_n_s64::<PRECISION>(sss_a[1]);
sss_a[2] = vshrq_n_s64::<PRECISION>(sss_a[2]);
sss_a[3] = vshrq_n_s64::<PRECISION>(sss_a[3]);
for i in 0..4 {
let res_u16x4 = vqmovun_s32(vcombine_s32(
@@ -115,13 +124,12 @@ unsafe fn horiz_convolution_four_rows(
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "neon")]
unsafe fn horiz_convolution_one_row(
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
src_row: &[U16x2],
dst_row: &mut [U16x2],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
precision: u8,
) {
let initial = vdupq_n_s64(1i64 << (precision - 1));
let initial = vdupq_n_s64(1i64 << (PRECISION - 1));
let zero_u16x8 = vdupq_n_u16(0);
let zero_u16x4 = vdup_n_u16(0);
@@ -173,12 +181,7 @@ unsafe fn horiz_convolution_one_row(
sss = vmlal_s32(sss, pix_i32, coeff);
}
macro_rules! call {
($imm8:expr) => {{
sss = vshrq_n_s64::<$imm8>(sss);
}};
}
constify_64_imm8!(precision, call);
sss = vshrq_n_s64::<PRECISION>(sss);
let res_u16x4 = vqmovun_s32(vcombine_s32(
vqmovn_s64(sss),
+22 -12
View File
@@ -14,13 +14,28 @@ pub(crate) fn horiz_convolution(
) {
let normalizer = optimisations::Normalizer32::new(coeffs);
let precision = normalizer.precision();
macro_rules! call {
($imm8:expr) => {{
horiz_convolution_p::<$imm8>(src_view, dst_view, offset, normalizer);
}};
}
constify_64_imm8!(precision, call);
}
fn horiz_convolution_p<const PRECISION: i32>(
src_view: &impl ImageView<Pixel = U16x3>,
dst_view: &mut impl ImageViewMut<Pixel = U16x3>,
offset: u32,
normalizer: optimisations::Normalizer32,
) {
let coefficients_chunks = normalizer.normalized_chunks();
let src_iter = src_view.iter_rows(offset);
let dst_iter = dst_view.iter_rows_mut(0);
for (src_row, dst_row) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, precision);
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &coefficients_chunks);
}
}
}
@@ -31,13 +46,12 @@ pub(crate) fn horiz_convolution(
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "neon")]
unsafe fn horiz_convolution_one_row(
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
src_row: &[U16x3],
dst_row: &mut [U16x3],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
precision: u8,
) {
let initial = vdupq_n_s64(1i64 << (precision - 2));
let initial = vdupq_n_s64(1i64 << (PRECISION - 2));
let zero_u16x8 = vdupq_n_u16(0);
let zero_u16x4 = vdup_n_u16(0);
@@ -98,14 +112,10 @@ unsafe fn horiz_convolution_one_row(
vadd_s64(vget_low_s64(sss[1]), vget_high_s64(sss[1])),
vadd_s64(vget_low_s64(sss[2]), vget_high_s64(sss[2])),
];
macro_rules! call {
($imm8:expr) => {{
sss_i64[0] = vshr_n_s64::<$imm8>(sss_i64[0]);
sss_i64[1] = vshr_n_s64::<$imm8>(sss_i64[1]);
sss_i64[2] = vshr_n_s64::<$imm8>(sss_i64[2]);
}};
}
constify_64_imm8!(precision, call);
sss_i64[0] = vshr_n_s64::<PRECISION>(sss_i64[0]);
sss_i64[1] = vshr_n_s64::<PRECISION>(sss_i64[1]);
sss_i64[2] = vshr_n_s64::<PRECISION>(sss_i64[2]);
dst_row.get_unchecked_mut(dst_x).0 = [
vqmovns_u32(vqmovund_s64(vdupd_lane_s64::<0>(sss_i64[0]))),
+31 -28
View File
@@ -14,6 +14,21 @@ pub(crate) fn horiz_convolution(
) {
let normalizer = optimisations::Normalizer32::new(coeffs);
let precision = normalizer.precision();
macro_rules! call {
($imm8:expr) => {{
horiz_convolution_p::<$imm8>(src_view, dst_view, offset, normalizer);
}};
}
constify_64_imm8!(precision, call);
}
fn horiz_convolution_p<const PRECISION: i32>(
src_view: &impl ImageView<Pixel = U16x4>,
dst_view: &mut impl ImageViewMut<Pixel = U16x4>,
offset: u32,
normalizer: optimisations::Normalizer32,
) {
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_view.height();
@@ -21,7 +36,7 @@ pub(crate) fn horiz_convolution(
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, precision);
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &coefficients_chunks);
}
}
@@ -30,7 +45,7 @@ pub(crate) fn horiz_convolution(
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, precision);
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &coefficients_chunks);
}
}
}
@@ -42,13 +57,12 @@ pub(crate) fn horiz_convolution(
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "neon")]
unsafe fn horiz_convolution_four_rows(
unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
src_rows: [&[U16x4]; 4],
dst_rows: [&mut [U16x4]; 4],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
precision: u8,
) {
let initial = vdupq_n_s64(1i64 << (precision - 1));
let initial = vdupq_n_s64(1i64 << (PRECISION - 1));
let zero_u16x8 = vdupq_n_u16(0);
let zero_u16x4 = vdup_n_u16(0);
@@ -136,19 +150,14 @@ unsafe fn horiz_convolution_four_rows(
}
}
macro_rules! call {
($imm8:expr) => {{
sss_a[0].0 = vshrq_n_s64::<$imm8>(sss_a[0].0);
sss_a[0].1 = vshrq_n_s64::<$imm8>(sss_a[0].1);
sss_a[1].0 = vshrq_n_s64::<$imm8>(sss_a[1].0);
sss_a[1].1 = vshrq_n_s64::<$imm8>(sss_a[1].1);
sss_a[2].0 = vshrq_n_s64::<$imm8>(sss_a[2].0);
sss_a[2].1 = vshrq_n_s64::<$imm8>(sss_a[2].1);
sss_a[3].0 = vshrq_n_s64::<$imm8>(sss_a[3].0);
sss_a[3].1 = vshrq_n_s64::<$imm8>(sss_a[3].1);
}};
}
constify_64_imm8!(precision as i64, call);
sss_a[0].0 = vshrq_n_s64::<PRECISION>(sss_a[0].0);
sss_a[0].1 = vshrq_n_s64::<PRECISION>(sss_a[0].1);
sss_a[1].0 = vshrq_n_s64::<PRECISION>(sss_a[1].0);
sss_a[1].1 = vshrq_n_s64::<PRECISION>(sss_a[1].1);
sss_a[2].0 = vshrq_n_s64::<PRECISION>(sss_a[2].0);
sss_a[2].1 = vshrq_n_s64::<PRECISION>(sss_a[2].1);
sss_a[3].0 = vshrq_n_s64::<PRECISION>(sss_a[3].0);
sss_a[3].1 = vshrq_n_s64::<PRECISION>(sss_a[3].1);
for i in 0..4 {
let sss = sss_a[i];
@@ -167,13 +176,12 @@ unsafe fn horiz_convolution_four_rows(
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "neon")]
unsafe fn horiz_convolution_one_row(
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
src_row: &[U16x4],
dst_row: &mut [U16x4],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
precision: u8,
) {
let initial = vdupq_n_s64(1i64 << (precision - 1));
let initial = vdupq_n_s64(1i64 << (PRECISION - 1));
let zero_u16x8 = vdupq_n_u16(0);
let zero_u16x4 = vdup_n_u16(0);
@@ -243,13 +251,8 @@ unsafe fn horiz_convolution_one_row(
sss.1 = vmlal_s32(sss.1, vget_high_s32(pix_i32), coeff);
}
macro_rules! call {
($imm8:expr) => {{
sss.0 = vshrq_n_s64::<$imm8>(sss.0);
sss.1 = vshrq_n_s64::<$imm8>(sss.1);
}};
}
constify_64_imm8!(precision as i64, call);
sss.0 = vshrq_n_s64::<PRECISION>(sss.0);
sss.1 = vshrq_n_s64::<PRECISION>(sss.1);
let sss_i32x4 = vcombine_s32(vqmovn_s64(sss.0), vqmovn_s64(sss.1));
let sss_u16x4 = vqmovun_s32(sss_i32x4);
+26 -25
View File
@@ -13,6 +13,21 @@ pub(crate) fn horiz_convolution(
) {
let normalizer = optimisations::Normalizer16::new(coeffs);
let precision = normalizer.precision();
macro_rules! call {
($imm8:expr) => {{
horiz_convolution_p::<$imm8>(src_view, dst_view, offset, normalizer);
}};
}
constify_imm8!(precision, call);
}
fn horiz_convolution_p<const PRECISION: i32>(
src_view: &impl ImageView<Pixel = U8x2>,
dst_view: &mut impl ImageViewMut<Pixel = U8x2>,
offset: u32,
normalizer: optimisations::Normalizer16,
) {
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_view.height();
@@ -20,7 +35,7 @@ pub(crate) fn horiz_convolution(
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, precision);
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &coefficients_chunks);
}
}
@@ -29,7 +44,7 @@ pub(crate) fn horiz_convolution(
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, precision);
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &coefficients_chunks);
}
}
}
@@ -41,13 +56,12 @@ pub(crate) fn horiz_convolution(
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "neon")]
unsafe fn horiz_convolution_four_rows(
unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
src_rows: [&[U8x2]; 4],
dst_rows: [&mut [U8x2]; 4],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
precision: u8,
) {
let initial = vdupq_n_s32(1 << (precision - 2));
let initial = vdupq_n_s32(1 << (PRECISION - 2));
let zero_u8x16 = vdupq_n_u8(0);
let zero_u8x8 = vdup_n_u8(0);
@@ -137,16 +151,10 @@ unsafe fn horiz_convolution_four_rows(
}
let mut res_i32x2x4 = sss_a.map(|sss| vadd_s32(vget_low_s32(sss), vget_high_s32(sss)));
macro_rules! call {
($imm8:expr) => {{
res_i32x2x4[0] = vshr_n_s32::<$imm8>(res_i32x2x4[0]);
res_i32x2x4[1] = vshr_n_s32::<$imm8>(res_i32x2x4[1]);
res_i32x2x4[2] = vshr_n_s32::<$imm8>(res_i32x2x4[2]);
res_i32x2x4[3] = vshr_n_s32::<$imm8>(res_i32x2x4[3]);
}};
}
constify_imm8!(precision, call);
res_i32x2x4[0] = vshr_n_s32::<PRECISION>(res_i32x2x4[0]);
res_i32x2x4[1] = vshr_n_s32::<PRECISION>(res_i32x2x4[1]);
res_i32x2x4[2] = vshr_n_s32::<PRECISION>(res_i32x2x4[2]);
res_i32x2x4[3] = vshr_n_s32::<PRECISION>(res_i32x2x4[3]);
for i in 0..4 {
let sss = vcombine_s32(res_i32x2x4[i], vreinterpret_s32_u8(zero_u8x8));
@@ -165,13 +173,12 @@ unsafe fn horiz_convolution_four_rows(
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "neon")]
unsafe fn horiz_convolution_one_row(
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
src_row: &[U8x2],
dst_row: &mut [U8x2],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
precision: u8,
) {
let initial = vdupq_n_s32(1 << (precision - 2));
let initial = vdupq_n_s32(1 << (PRECISION - 2));
let zero_u8x16 = vdupq_n_u8(0);
let zero_u8x8 = vdup_n_u8(0);
@@ -225,13 +232,7 @@ unsafe fn horiz_convolution_one_row(
}
let mut res_i32x2 = vadd_s32(vget_low_s32(sss), vget_high_s32(sss));
macro_rules! call {
($imm8:expr) => {{
res_i32x2 = vshr_n_s32::<$imm8>(res_i32x2);
}};
}
constify_imm8!(precision, call);
res_i32x2 = vshr_n_s32::<PRECISION>(res_i32x2);
let sss = vcombine_s32(res_i32x2, vreinterpret_s32_u8(zero_u8x8));
let s = vreinterpret_u16_u8(vqmovun_s16(vcombine_s16(
+26 -24
View File
@@ -14,6 +14,21 @@ pub(crate) fn horiz_convolution(
) {
let normalizer = optimisations::Normalizer16::new(coeffs);
let precision = normalizer.precision();
macro_rules! call {
($imm8:expr) => {{
horiz_convolution_p::<$imm8>(src_view, dst_view, offset, normalizer);
}};
}
constify_imm8!(precision, call);
}
fn horiz_convolution_p<const PRECISION: i32>(
src_view: &impl ImageView<Pixel = U8x3>,
dst_view: &mut impl ImageViewMut<Pixel = U8x3>,
offset: u32,
normalizer: optimisations::Normalizer16,
) {
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_view.height();
@@ -21,7 +36,7 @@ pub(crate) fn horiz_convolution(
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, precision);
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &coefficients_chunks);
}
}
@@ -30,7 +45,7 @@ pub(crate) fn horiz_convolution(
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, precision);
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &coefficients_chunks);
}
}
}
@@ -42,13 +57,12 @@ pub(crate) fn horiz_convolution(
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "neon")]
unsafe fn horiz_convolution_four_rows(
unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
src_rows: [&[U8x3]; 4],
dst_rows: [&mut [U8x3]; 4],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
precision: u8,
) {
let initial = vdupq_n_s32(1 << (precision - 1));
let initial = vdupq_n_s32(1 << (PRECISION - 1));
let zero_u8x8 = vdup_n_u8(0);
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
@@ -95,15 +109,10 @@ unsafe fn horiz_convolution_four_rows(
}
}
macro_rules! call {
($imm8:expr) => {{
sss_a[0] = vshrq_n_s32::<$imm8>(sss_a[0]);
sss_a[1] = vshrq_n_s32::<$imm8>(sss_a[1]);
sss_a[2] = vshrq_n_s32::<$imm8>(sss_a[2]);
sss_a[3] = vshrq_n_s32::<$imm8>(sss_a[3]);
}};
}
constify_imm8!(precision, call);
sss_a[0] = vshrq_n_s32::<PRECISION>(sss_a[0]);
sss_a[1] = vshrq_n_s32::<PRECISION>(sss_a[1]);
sss_a[2] = vshrq_n_s32::<PRECISION>(sss_a[2]);
sss_a[3] = vshrq_n_s32::<PRECISION>(sss_a[3]);
for i in 0..4 {
store_pixel(sss_a[i], dst_rows[i], dst_x, zero_u8x8);
@@ -117,13 +126,12 @@ unsafe fn horiz_convolution_four_rows(
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "neon")]
unsafe fn horiz_convolution_one_row(
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
src_row: &[U8x3],
dst_row: &mut [U8x3],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
precision: u8,
) {
let initial = vdupq_n_s32(1 << (precision - 1));
let initial = vdupq_n_s32(1 << (PRECISION - 1));
let zero_u8x8 = vdup_n_u8(0);
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
@@ -165,13 +173,7 @@ unsafe fn horiz_convolution_one_row(
sss = conv_4_pixels(sss, coeffs_i16x4, &four_pixels, 0, zero_u8x8);
}
macro_rules! call {
($imm8:expr) => {{
sss = vshrq_n_s32::<$imm8>(sss);
}};
}
constify_imm8!(precision, call);
sss = vshrq_n_s32::<PRECISION>(sss);
store_pixel(sss, dst_row, dst_x, zero_u8x8);
}
}
+26 -23
View File
@@ -14,6 +14,21 @@ pub(crate) fn horiz_convolution(
) {
let normalizer = optimisations::Normalizer16::new(coeffs);
let precision = normalizer.precision();
macro_rules! call {
($imm8:expr) => {{
horiz_convolution_p::<$imm8>(src_view, dst_view, offset, normalizer);
}};
}
constify_imm8!(precision, call);
}
fn horiz_convolution_p<const PRECISION: i32>(
src_view: &impl ImageView<Pixel = U8x4>,
dst_view: &mut impl ImageViewMut<Pixel = U8x4>,
offset: u32,
normalizer: optimisations::Normalizer16,
) {
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_view.height();
@@ -21,7 +36,7 @@ pub(crate) fn horiz_convolution(
let dst_iter = dst_view.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, precision);
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &coefficients_chunks);
}
}
@@ -30,7 +45,7 @@ pub(crate) fn horiz_convolution(
let dst_rows = dst_view.iter_rows_mut(yy);
for (src_row, dst_row) in src_rows.zip(dst_rows) {
unsafe {
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, precision);
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &coefficients_chunks);
}
}
}
@@ -42,13 +57,12 @@ pub(crate) fn horiz_convolution(
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "neon")]
unsafe fn horiz_convolution_four_rows(
unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
src_rows: [&[U8x4]; 4],
dst_rows: [&mut [U8x4]; 4],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
precision: u8,
) {
let initial = vdupq_n_s32(1 << (precision - 1));
let initial = vdupq_n_s32(1 << (PRECISION - 1));
let zero_u8x16 = vdupq_n_u8(0);
let zero_u8x8 = vdup_n_u8(0);
@@ -168,15 +182,10 @@ unsafe fn horiz_convolution_four_rows(
}
}
macro_rules! call {
($imm8:expr) => {{
sss_a[0] = vshrq_n_s32::<$imm8>(sss_a[0]);
sss_a[1] = vshrq_n_s32::<$imm8>(sss_a[1]);
sss_a[2] = vshrq_n_s32::<$imm8>(sss_a[2]);
sss_a[3] = vshrq_n_s32::<$imm8>(sss_a[3]);
}};
}
constify_imm8!(precision, call);
sss_a[0] = vshrq_n_s32::<PRECISION>(sss_a[0]);
sss_a[1] = vshrq_n_s32::<PRECISION>(sss_a[1]);
sss_a[2] = vshrq_n_s32::<PRECISION>(sss_a[2]);
sss_a[3] = vshrq_n_s32::<PRECISION>(sss_a[3]);
for i in 0..4 {
let s = vqmovun_s16(vcombine_s16(vqmovn_s32(sss_a[i]), vdup_n_s16(0)));
@@ -192,13 +201,12 @@ unsafe fn horiz_convolution_four_rows(
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
#[target_feature(enable = "neon")]
unsafe fn horiz_convolution_one_row(
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
src_row: &[U8x4],
dst_row: &mut [U8x4],
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
precision: u8,
) {
let initial = vdupq_n_s32(1 << (precision - 1));
let initial = vdupq_n_s32(1 << (PRECISION - 1));
let zero_u8x16 = vdupq_n_u8(0);
let zero_u8x8 = vdup_n_u8(0);
@@ -284,12 +292,7 @@ unsafe fn horiz_convolution_one_row(
sss = vmlal_s16(sss, pix, vdup_n_s16(k));
}
macro_rules! call {
($imm8:expr) => {{
sss = vshrq_n_s32::<$imm8>(sss);
}};
}
constify_imm8!(precision, call);
sss = vshrq_n_s32::<PRECISION>(sss);
let s = vqmovun_s16(vcombine_s16(vqmovn_s32(sss), vdup_n_s16(0)));
let s = vreinterpret_u32_u8(s);