mirror of
https://github.com/Cykooz/fast_image_resize.git
synced 2026-10-08 01:11:09 +00:00
Optimized Neon implementation.
This commit is contained in:
@@ -14,6 +14,21 @@ pub(crate) fn horiz_convolution(
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
horiz_convolution_p::<$imm8>(src_view, dst_view, offset, normalizer);
|
||||
}};
|
||||
}
|
||||
constify_64_imm8!(precision, call);
|
||||
}
|
||||
|
||||
fn horiz_convolution_p<const PRECISION: i32>(
|
||||
src_view: &impl ImageView<Pixel = U16>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16>,
|
||||
offset: u32,
|
||||
normalizer: optimisations::Normalizer32,
|
||||
) {
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
@@ -21,7 +36,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, precision);
|
||||
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &coefficients_chunks);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -30,7 +45,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, precision);
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &coefficients_chunks);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -42,13 +57,12 @@ pub(crate) fn horiz_convolution(
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[target_feature(enable = "neon")]
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
src_rows: [&[U16]; 4],
|
||||
dst_rows: [&mut [U16]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
precision: u8,
|
||||
) {
|
||||
let initial = vdupq_n_s64(1i64 << (precision - 2));
|
||||
let initial = vdupq_n_s64(1i64 << (PRECISION - 2));
|
||||
let zero_u16x4 = vdup_n_u16(0);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
@@ -105,15 +119,10 @@ unsafe fn horiz_convolution_four_rows(
|
||||
vadd_s64(vget_low_s64(sss_a[2]), vget_high_s64(sss_a[2])),
|
||||
vadd_s64(vget_low_s64(sss_a[3]), vget_high_s64(sss_a[3])),
|
||||
];
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss_a_i64[0] = vshr_n_s64::<$imm8>(sss_a_i64[0]);
|
||||
sss_a_i64[1] = vshr_n_s64::<$imm8>(sss_a_i64[1]);
|
||||
sss_a_i64[2] = vshr_n_s64::<$imm8>(sss_a_i64[2]);
|
||||
sss_a_i64[3] = vshr_n_s64::<$imm8>(sss_a_i64[3]);
|
||||
}};
|
||||
}
|
||||
constify_64_imm8!(precision, call);
|
||||
sss_a_i64[0] = vshr_n_s64::<PRECISION>(sss_a_i64[0]);
|
||||
sss_a_i64[1] = vshr_n_s64::<PRECISION>(sss_a_i64[1]);
|
||||
sss_a_i64[2] = vshr_n_s64::<PRECISION>(sss_a_i64[2]);
|
||||
sss_a_i64[3] = vshr_n_s64::<PRECISION>(sss_a_i64[3]);
|
||||
|
||||
for i in 0..4 {
|
||||
let res = vdupd_lane_s64::<0>(sss_a_i64[i]);
|
||||
@@ -128,13 +137,12 @@ unsafe fn horiz_convolution_four_rows(
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[target_feature(enable = "neon")]
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
src_row: &[U16],
|
||||
dst_row: &mut [U16],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
precision: u8,
|
||||
) {
|
||||
let initial = vdupq_n_s64(1i64 << (precision - 2));
|
||||
let initial = vdupq_n_s64(1i64 << (PRECISION - 2));
|
||||
let zero_u16x8 = vdupq_n_u16(0);
|
||||
let zero_u16x4 = vdup_n_u16(0);
|
||||
|
||||
@@ -192,12 +200,7 @@ unsafe fn horiz_convolution_one_row(
|
||||
}
|
||||
|
||||
let mut sss_i64 = vadd_s64(vget_low_s64(sss), vget_high_s64(sss));
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss_i64 = vshr_n_s64::<$imm8>(sss_i64);
|
||||
}};
|
||||
}
|
||||
constify_64_imm8!(precision, call);
|
||||
sss_i64 = vshr_n_s64::<PRECISION>(sss_i64);
|
||||
|
||||
let res = vdupd_lane_s64::<0>(sss_i64);
|
||||
dst_row.get_unchecked_mut(dst_x).0 = vqmovns_u32(vqmovund_s64(res));
|
||||
|
||||
@@ -14,6 +14,21 @@ pub(crate) fn horiz_convolution(
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
horiz_convolution_p::<$imm8>(src_view, dst_view, offset, normalizer);
|
||||
}};
|
||||
}
|
||||
constify_64_imm8!(precision, call);
|
||||
}
|
||||
|
||||
fn horiz_convolution_p<const PRECISION: i32>(
|
||||
src_view: &impl ImageView<Pixel = U16x2>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16x2>,
|
||||
offset: u32,
|
||||
normalizer: optimisations::Normalizer32,
|
||||
) {
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
@@ -21,7 +36,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, precision);
|
||||
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &coefficients_chunks);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -30,7 +45,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, precision);
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &coefficients_chunks);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -42,13 +57,12 @@ pub(crate) fn horiz_convolution(
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[target_feature(enable = "neon")]
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
src_rows: [&[U16x2]; 4],
|
||||
dst_rows: [&mut [U16x2]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
precision: u8,
|
||||
) {
|
||||
let initial = vdupq_n_s64(1i64 << (precision - 1));
|
||||
let initial = vdupq_n_s64(1i64 << (PRECISION - 1));
|
||||
// let zero_u16x8 = vdupq_n_u16(0);
|
||||
let zero_u16x4 = vdup_n_u16(0);
|
||||
|
||||
@@ -86,15 +100,10 @@ unsafe fn horiz_convolution_four_rows(
|
||||
}
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss_a[0] = vshrq_n_s64::<$imm8>(sss_a[0]);
|
||||
sss_a[1] = vshrq_n_s64::<$imm8>(sss_a[1]);
|
||||
sss_a[2] = vshrq_n_s64::<$imm8>(sss_a[2]);
|
||||
sss_a[3] = vshrq_n_s64::<$imm8>(sss_a[3]);
|
||||
}};
|
||||
}
|
||||
constify_64_imm8!(precision, call);
|
||||
sss_a[0] = vshrq_n_s64::<PRECISION>(sss_a[0]);
|
||||
sss_a[1] = vshrq_n_s64::<PRECISION>(sss_a[1]);
|
||||
sss_a[2] = vshrq_n_s64::<PRECISION>(sss_a[2]);
|
||||
sss_a[3] = vshrq_n_s64::<PRECISION>(sss_a[3]);
|
||||
|
||||
for i in 0..4 {
|
||||
let res_u16x4 = vqmovun_s32(vcombine_s32(
|
||||
@@ -115,13 +124,12 @@ unsafe fn horiz_convolution_four_rows(
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[target_feature(enable = "neon")]
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
src_row: &[U16x2],
|
||||
dst_row: &mut [U16x2],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
precision: u8,
|
||||
) {
|
||||
let initial = vdupq_n_s64(1i64 << (precision - 1));
|
||||
let initial = vdupq_n_s64(1i64 << (PRECISION - 1));
|
||||
let zero_u16x8 = vdupq_n_u16(0);
|
||||
let zero_u16x4 = vdup_n_u16(0);
|
||||
|
||||
@@ -173,12 +181,7 @@ unsafe fn horiz_convolution_one_row(
|
||||
sss = vmlal_s32(sss, pix_i32, coeff);
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss = vshrq_n_s64::<$imm8>(sss);
|
||||
}};
|
||||
}
|
||||
constify_64_imm8!(precision, call);
|
||||
sss = vshrq_n_s64::<PRECISION>(sss);
|
||||
|
||||
let res_u16x4 = vqmovun_s32(vcombine_s32(
|
||||
vqmovn_s64(sss),
|
||||
|
||||
@@ -14,13 +14,28 @@ pub(crate) fn horiz_convolution(
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
horiz_convolution_p::<$imm8>(src_view, dst_view, offset, normalizer);
|
||||
}};
|
||||
}
|
||||
constify_64_imm8!(precision, call);
|
||||
}
|
||||
|
||||
fn horiz_convolution_p<const PRECISION: i32>(
|
||||
src_view: &impl ImageView<Pixel = U16x3>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16x3>,
|
||||
offset: u32,
|
||||
normalizer: optimisations::Normalizer32,
|
||||
) {
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
|
||||
let src_iter = src_view.iter_rows(offset);
|
||||
let dst_iter = dst_view.iter_rows_mut(0);
|
||||
for (src_row, dst_row) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, precision);
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &coefficients_chunks);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -31,13 +46,12 @@ pub(crate) fn horiz_convolution(
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[target_feature(enable = "neon")]
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
src_row: &[U16x3],
|
||||
dst_row: &mut [U16x3],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
precision: u8,
|
||||
) {
|
||||
let initial = vdupq_n_s64(1i64 << (precision - 2));
|
||||
let initial = vdupq_n_s64(1i64 << (PRECISION - 2));
|
||||
let zero_u16x8 = vdupq_n_u16(0);
|
||||
let zero_u16x4 = vdup_n_u16(0);
|
||||
|
||||
@@ -98,14 +112,10 @@ unsafe fn horiz_convolution_one_row(
|
||||
vadd_s64(vget_low_s64(sss[1]), vget_high_s64(sss[1])),
|
||||
vadd_s64(vget_low_s64(sss[2]), vget_high_s64(sss[2])),
|
||||
];
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss_i64[0] = vshr_n_s64::<$imm8>(sss_i64[0]);
|
||||
sss_i64[1] = vshr_n_s64::<$imm8>(sss_i64[1]);
|
||||
sss_i64[2] = vshr_n_s64::<$imm8>(sss_i64[2]);
|
||||
}};
|
||||
}
|
||||
constify_64_imm8!(precision, call);
|
||||
|
||||
sss_i64[0] = vshr_n_s64::<PRECISION>(sss_i64[0]);
|
||||
sss_i64[1] = vshr_n_s64::<PRECISION>(sss_i64[1]);
|
||||
sss_i64[2] = vshr_n_s64::<PRECISION>(sss_i64[2]);
|
||||
|
||||
dst_row.get_unchecked_mut(dst_x).0 = [
|
||||
vqmovns_u32(vqmovund_s64(vdupd_lane_s64::<0>(sss_i64[0]))),
|
||||
|
||||
@@ -14,6 +14,21 @@ pub(crate) fn horiz_convolution(
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
horiz_convolution_p::<$imm8>(src_view, dst_view, offset, normalizer);
|
||||
}};
|
||||
}
|
||||
constify_64_imm8!(precision, call);
|
||||
}
|
||||
|
||||
fn horiz_convolution_p<const PRECISION: i32>(
|
||||
src_view: &impl ImageView<Pixel = U16x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U16x4>,
|
||||
offset: u32,
|
||||
normalizer: optimisations::Normalizer32,
|
||||
) {
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
@@ -21,7 +36,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, precision);
|
||||
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &coefficients_chunks);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -30,7 +45,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, precision);
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &coefficients_chunks);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -42,13 +57,12 @@ pub(crate) fn horiz_convolution(
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[target_feature(enable = "neon")]
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
src_rows: [&[U16x4]; 4],
|
||||
dst_rows: [&mut [U16x4]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
precision: u8,
|
||||
) {
|
||||
let initial = vdupq_n_s64(1i64 << (precision - 1));
|
||||
let initial = vdupq_n_s64(1i64 << (PRECISION - 1));
|
||||
let zero_u16x8 = vdupq_n_u16(0);
|
||||
let zero_u16x4 = vdup_n_u16(0);
|
||||
|
||||
@@ -136,19 +150,14 @@ unsafe fn horiz_convolution_four_rows(
|
||||
}
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss_a[0].0 = vshrq_n_s64::<$imm8>(sss_a[0].0);
|
||||
sss_a[0].1 = vshrq_n_s64::<$imm8>(sss_a[0].1);
|
||||
sss_a[1].0 = vshrq_n_s64::<$imm8>(sss_a[1].0);
|
||||
sss_a[1].1 = vshrq_n_s64::<$imm8>(sss_a[1].1);
|
||||
sss_a[2].0 = vshrq_n_s64::<$imm8>(sss_a[2].0);
|
||||
sss_a[2].1 = vshrq_n_s64::<$imm8>(sss_a[2].1);
|
||||
sss_a[3].0 = vshrq_n_s64::<$imm8>(sss_a[3].0);
|
||||
sss_a[3].1 = vshrq_n_s64::<$imm8>(sss_a[3].1);
|
||||
}};
|
||||
}
|
||||
constify_64_imm8!(precision as i64, call);
|
||||
sss_a[0].0 = vshrq_n_s64::<PRECISION>(sss_a[0].0);
|
||||
sss_a[0].1 = vshrq_n_s64::<PRECISION>(sss_a[0].1);
|
||||
sss_a[1].0 = vshrq_n_s64::<PRECISION>(sss_a[1].0);
|
||||
sss_a[1].1 = vshrq_n_s64::<PRECISION>(sss_a[1].1);
|
||||
sss_a[2].0 = vshrq_n_s64::<PRECISION>(sss_a[2].0);
|
||||
sss_a[2].1 = vshrq_n_s64::<PRECISION>(sss_a[2].1);
|
||||
sss_a[3].0 = vshrq_n_s64::<PRECISION>(sss_a[3].0);
|
||||
sss_a[3].1 = vshrq_n_s64::<PRECISION>(sss_a[3].1);
|
||||
|
||||
for i in 0..4 {
|
||||
let sss = sss_a[i];
|
||||
@@ -167,13 +176,12 @@ unsafe fn horiz_convolution_four_rows(
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[target_feature(enable = "neon")]
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
src_row: &[U16x4],
|
||||
dst_row: &mut [U16x4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
precision: u8,
|
||||
) {
|
||||
let initial = vdupq_n_s64(1i64 << (precision - 1));
|
||||
let initial = vdupq_n_s64(1i64 << (PRECISION - 1));
|
||||
let zero_u16x8 = vdupq_n_u16(0);
|
||||
let zero_u16x4 = vdup_n_u16(0);
|
||||
|
||||
@@ -243,13 +251,8 @@ unsafe fn horiz_convolution_one_row(
|
||||
sss.1 = vmlal_s32(sss.1, vget_high_s32(pix_i32), coeff);
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss.0 = vshrq_n_s64::<$imm8>(sss.0);
|
||||
sss.1 = vshrq_n_s64::<$imm8>(sss.1);
|
||||
}};
|
||||
}
|
||||
constify_64_imm8!(precision as i64, call);
|
||||
sss.0 = vshrq_n_s64::<PRECISION>(sss.0);
|
||||
sss.1 = vshrq_n_s64::<PRECISION>(sss.1);
|
||||
|
||||
let sss_i32x4 = vcombine_s32(vqmovn_s64(sss.0), vqmovn_s64(sss.1));
|
||||
let sss_u16x4 = vqmovun_s32(sss_i32x4);
|
||||
|
||||
@@ -13,6 +13,21 @@ pub(crate) fn horiz_convolution(
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
horiz_convolution_p::<$imm8>(src_view, dst_view, offset, normalizer);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
}
|
||||
|
||||
fn horiz_convolution_p<const PRECISION: i32>(
|
||||
src_view: &impl ImageView<Pixel = U8x2>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x2>,
|
||||
offset: u32,
|
||||
normalizer: optimisations::Normalizer16,
|
||||
) {
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
@@ -20,7 +35,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, precision);
|
||||
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &coefficients_chunks);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -29,7 +44,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, precision);
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &coefficients_chunks);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -41,13 +56,12 @@ pub(crate) fn horiz_convolution(
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[target_feature(enable = "neon")]
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
src_rows: [&[U8x2]; 4],
|
||||
dst_rows: [&mut [U8x2]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
precision: u8,
|
||||
) {
|
||||
let initial = vdupq_n_s32(1 << (precision - 2));
|
||||
let initial = vdupq_n_s32(1 << (PRECISION - 2));
|
||||
let zero_u8x16 = vdupq_n_u8(0);
|
||||
let zero_u8x8 = vdup_n_u8(0);
|
||||
|
||||
@@ -137,16 +151,10 @@ unsafe fn horiz_convolution_four_rows(
|
||||
}
|
||||
|
||||
let mut res_i32x2x4 = sss_a.map(|sss| vadd_s32(vget_low_s32(sss), vget_high_s32(sss)));
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
res_i32x2x4[0] = vshr_n_s32::<$imm8>(res_i32x2x4[0]);
|
||||
res_i32x2x4[1] = vshr_n_s32::<$imm8>(res_i32x2x4[1]);
|
||||
res_i32x2x4[2] = vshr_n_s32::<$imm8>(res_i32x2x4[2]);
|
||||
res_i32x2x4[3] = vshr_n_s32::<$imm8>(res_i32x2x4[3]);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
res_i32x2x4[0] = vshr_n_s32::<PRECISION>(res_i32x2x4[0]);
|
||||
res_i32x2x4[1] = vshr_n_s32::<PRECISION>(res_i32x2x4[1]);
|
||||
res_i32x2x4[2] = vshr_n_s32::<PRECISION>(res_i32x2x4[2]);
|
||||
res_i32x2x4[3] = vshr_n_s32::<PRECISION>(res_i32x2x4[3]);
|
||||
|
||||
for i in 0..4 {
|
||||
let sss = vcombine_s32(res_i32x2x4[i], vreinterpret_s32_u8(zero_u8x8));
|
||||
@@ -165,13 +173,12 @@ unsafe fn horiz_convolution_four_rows(
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[target_feature(enable = "neon")]
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
src_row: &[U8x2],
|
||||
dst_row: &mut [U8x2],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
precision: u8,
|
||||
) {
|
||||
let initial = vdupq_n_s32(1 << (precision - 2));
|
||||
let initial = vdupq_n_s32(1 << (PRECISION - 2));
|
||||
let zero_u8x16 = vdupq_n_u8(0);
|
||||
let zero_u8x8 = vdup_n_u8(0);
|
||||
|
||||
@@ -225,13 +232,7 @@ unsafe fn horiz_convolution_one_row(
|
||||
}
|
||||
|
||||
let mut res_i32x2 = vadd_s32(vget_low_s32(sss), vget_high_s32(sss));
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
res_i32x2 = vshr_n_s32::<$imm8>(res_i32x2);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
res_i32x2 = vshr_n_s32::<PRECISION>(res_i32x2);
|
||||
|
||||
let sss = vcombine_s32(res_i32x2, vreinterpret_s32_u8(zero_u8x8));
|
||||
let s = vreinterpret_u16_u8(vqmovun_s16(vcombine_s16(
|
||||
|
||||
@@ -14,6 +14,21 @@ pub(crate) fn horiz_convolution(
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
horiz_convolution_p::<$imm8>(src_view, dst_view, offset, normalizer);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
}
|
||||
|
||||
fn horiz_convolution_p<const PRECISION: i32>(
|
||||
src_view: &impl ImageView<Pixel = U8x3>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x3>,
|
||||
offset: u32,
|
||||
normalizer: optimisations::Normalizer16,
|
||||
) {
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
@@ -21,7 +36,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, precision);
|
||||
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &coefficients_chunks);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -30,7 +45,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, precision);
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &coefficients_chunks);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -42,13 +57,12 @@ pub(crate) fn horiz_convolution(
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[target_feature(enable = "neon")]
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
src_rows: [&[U8x3]; 4],
|
||||
dst_rows: [&mut [U8x3]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
precision: u8,
|
||||
) {
|
||||
let initial = vdupq_n_s32(1 << (precision - 1));
|
||||
let initial = vdupq_n_s32(1 << (PRECISION - 1));
|
||||
let zero_u8x8 = vdup_n_u8(0);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
@@ -95,15 +109,10 @@ unsafe fn horiz_convolution_four_rows(
|
||||
}
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss_a[0] = vshrq_n_s32::<$imm8>(sss_a[0]);
|
||||
sss_a[1] = vshrq_n_s32::<$imm8>(sss_a[1]);
|
||||
sss_a[2] = vshrq_n_s32::<$imm8>(sss_a[2]);
|
||||
sss_a[3] = vshrq_n_s32::<$imm8>(sss_a[3]);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
sss_a[0] = vshrq_n_s32::<PRECISION>(sss_a[0]);
|
||||
sss_a[1] = vshrq_n_s32::<PRECISION>(sss_a[1]);
|
||||
sss_a[2] = vshrq_n_s32::<PRECISION>(sss_a[2]);
|
||||
sss_a[3] = vshrq_n_s32::<PRECISION>(sss_a[3]);
|
||||
|
||||
for i in 0..4 {
|
||||
store_pixel(sss_a[i], dst_rows[i], dst_x, zero_u8x8);
|
||||
@@ -117,13 +126,12 @@ unsafe fn horiz_convolution_four_rows(
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[target_feature(enable = "neon")]
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
src_row: &[U8x3],
|
||||
dst_row: &mut [U8x3],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
precision: u8,
|
||||
) {
|
||||
let initial = vdupq_n_s32(1 << (precision - 1));
|
||||
let initial = vdupq_n_s32(1 << (PRECISION - 1));
|
||||
let zero_u8x8 = vdup_n_u8(0);
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
@@ -165,13 +173,7 @@ unsafe fn horiz_convolution_one_row(
|
||||
sss = conv_4_pixels(sss, coeffs_i16x4, &four_pixels, 0, zero_u8x8);
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss = vshrq_n_s32::<$imm8>(sss);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
|
||||
sss = vshrq_n_s32::<PRECISION>(sss);
|
||||
store_pixel(sss, dst_row, dst_x, zero_u8x8);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -14,6 +14,21 @@ pub(crate) fn horiz_convolution(
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
horiz_convolution_p::<$imm8>(src_view, dst_view, offset, normalizer);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
}
|
||||
|
||||
fn horiz_convolution_p<const PRECISION: i32>(
|
||||
src_view: &impl ImageView<Pixel = U8x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = U8x4>,
|
||||
offset: u32,
|
||||
normalizer: optimisations::Normalizer16,
|
||||
) {
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_view.height();
|
||||
|
||||
@@ -21,7 +36,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_iter = dst_view.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, precision);
|
||||
horiz_convolution_four_rows::<PRECISION>(src_rows, dst_rows, &coefficients_chunks);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -30,7 +45,7 @@ pub(crate) fn horiz_convolution(
|
||||
let dst_rows = dst_view.iter_rows_mut(yy);
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(src_row, dst_row, &coefficients_chunks, precision);
|
||||
horiz_convolution_one_row::<PRECISION>(src_row, dst_row, &coefficients_chunks);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -42,13 +57,12 @@ pub(crate) fn horiz_convolution(
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[target_feature(enable = "neon")]
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
unsafe fn horiz_convolution_four_rows<const PRECISION: i32>(
|
||||
src_rows: [&[U8x4]; 4],
|
||||
dst_rows: [&mut [U8x4]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
precision: u8,
|
||||
) {
|
||||
let initial = vdupq_n_s32(1 << (precision - 1));
|
||||
let initial = vdupq_n_s32(1 << (PRECISION - 1));
|
||||
let zero_u8x16 = vdupq_n_u8(0);
|
||||
let zero_u8x8 = vdup_n_u8(0);
|
||||
|
||||
@@ -168,15 +182,10 @@ unsafe fn horiz_convolution_four_rows(
|
||||
}
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss_a[0] = vshrq_n_s32::<$imm8>(sss_a[0]);
|
||||
sss_a[1] = vshrq_n_s32::<$imm8>(sss_a[1]);
|
||||
sss_a[2] = vshrq_n_s32::<$imm8>(sss_a[2]);
|
||||
sss_a[3] = vshrq_n_s32::<$imm8>(sss_a[3]);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
sss_a[0] = vshrq_n_s32::<PRECISION>(sss_a[0]);
|
||||
sss_a[1] = vshrq_n_s32::<PRECISION>(sss_a[1]);
|
||||
sss_a[2] = vshrq_n_s32::<PRECISION>(sss_a[2]);
|
||||
sss_a[3] = vshrq_n_s32::<PRECISION>(sss_a[3]);
|
||||
|
||||
for i in 0..4 {
|
||||
let s = vqmovun_s16(vcombine_s16(vqmovn_s32(sss_a[i]), vdup_n_s16(0)));
|
||||
@@ -192,13 +201,12 @@ unsafe fn horiz_convolution_four_rows(
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[target_feature(enable = "neon")]
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
unsafe fn horiz_convolution_one_row<const PRECISION: i32>(
|
||||
src_row: &[U8x4],
|
||||
dst_row: &mut [U8x4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
precision: u8,
|
||||
) {
|
||||
let initial = vdupq_n_s32(1 << (precision - 1));
|
||||
let initial = vdupq_n_s32(1 << (PRECISION - 1));
|
||||
let zero_u8x16 = vdupq_n_u8(0);
|
||||
let zero_u8x8 = vdup_n_u8(0);
|
||||
|
||||
@@ -284,12 +292,7 @@ unsafe fn horiz_convolution_one_row(
|
||||
sss = vmlal_s16(sss, pix, vdup_n_s16(k));
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss = vshrq_n_s32::<$imm8>(sss);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
sss = vshrq_n_s32::<PRECISION>(sss);
|
||||
|
||||
let s = vqmovun_s16(vcombine_s16(vqmovn_s32(sss), vdup_n_s16(0)));
|
||||
let s = vreinterpret_u32_u8(s);
|
||||
|
||||
Reference in New Issue
Block a user