Resize Complete

This commit is contained in:
Colin
2023-01-23 16:46:42 -05:00
committed by Colin Murphy
parent 9b15dcbc71
commit a8e33d213c
6 changed files with 745 additions and 2 deletions
+257
View File
@@ -0,0 +1,257 @@
use std::arch::wasm32::*;
use crate::convolution::{optimisations, Coefficients};
use crate::pixels::U16x2;
use crate::wasm32_utils;
use crate::{ImageView, ImageViewMut};
#[inline]
pub(crate) fn horiz_convolution(
src_image: &ImageView<U16x2>,
dst_image: &mut ImageViewMut<U16x2>,
offset: u32,
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer32::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_image.height().get();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_image.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
}
}
let mut yy = dst_height - dst_height % 4;
while yy < dst_height {
unsafe {
horiz_convolution_one_row(
src_image.get_row(yy + offset).unwrap(),
dst_image.get_row_mut(yy).unwrap(),
&coefficients_chunks,
&normalizer,
);
}
yy += 1;
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - length of all rows in src_rows must be equal
/// - length of all rows in dst_rows must be equal
/// - coefficients_chunks.len() == dst_rows.0.len()
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U16x2]; 4],
dst_rows: [&mut &mut [U16x2]; 4],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
normalizer: &optimisations::Normalizer32,
) {
let precision = normalizer.precision();
let half_error = 1i64 << (precision - 1);
let mut ll_buf = [0i64; 2];
/*
|L0 A0 | |L1 A1 | |L2 A2 | |L3 A3 |
|0001 0203| |0405 0607| |0809 1011| |1213 1415|
Shuffle to extract L0 and A0 as i64:
0, 1, -1, -1, -1, -1, -1, -1, 2, 3, -1, -1, -1, -1, -1, -1
Shuffle to extract L1 and A1 as i64:
4, 5, -1, -1, -1, -1, -1, -1, 6, 7, -1, -1, -1, -1, -1, -1
Shuffle to extract L2 and A2 as i64:
8, 9, -1, -1, -1, -1, -1, -1, 10, 11, -1, -1, -1, -1, -1, -1
Shuffle to extract L3 and A3 as i64:
12, 13, -1, -1, -1, -1, -1, -1, 14, 15, -1, -1, -1, -1, -1, -1
*/
let p0_shuffle = i8x16(0, 1, -1, -1, -1, -1, -1, -1, 2, 3, -1, -1, -1, -1, -1, -1);
let p1_shuffle = i8x16(4, 5, -1, -1, -1, -1, -1, -1, 6, 7, -1, -1, -1, -1, -1, -1);
let p2_shuffle = i8x16(8, 9, -1, -1, -1, -1, -1, -1, 10, 11, -1, -1, -1, -1, -1, -1);
let p3_shuffle = i8x16(
12, 13, -1, -1, -1, -1, -1, -1, 14, 15, -1, -1, -1, -1, -1, -1,
);
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
let mut ll_sum = [i64x2_splat(half_error); 4];
let mut coeffs = coeffs_chunk.values;
let coeffs_by_4 = coeffs.chunks_exact(4);
coeffs = coeffs_by_4.remainder();
for k in coeffs_by_4 {
let coeff0_i64x2 = i64x2_splat(k[0] as i64);
let coeff1_i64x2 = i64x2_splat(k[1] as i64);
let coeff2_i64x2 = i64x2_splat(k[2] as i64);
let coeff3_i64x2 = i64x2_splat(k[3] as i64);
for i in 0..4 {
let mut sum = ll_sum[i];
let source = wasm32_utils::load_v128(src_rows[i], x);
let p_i64x2 = i8x16_swizzle(source, p0_shuffle);
sum = i64x2_add(sum, i64x2_mul(p_i64x2, coeff0_i64x2));
let p_i64x2 = i8x16_swizzle(source, p1_shuffle);
sum = i64x2_add(sum, i64x2_mul(p_i64x2, coeff1_i64x2));
let p_i64x2 = i8x16_swizzle(source, p2_shuffle);
sum = i64x2_add(sum, i64x2_mul(p_i64x2, coeff2_i64x2));
let p_i64x2 = i8x16_swizzle(source, p3_shuffle);
sum = i64x2_add(sum, i64x2_mul(p_i64x2, coeff3_i64x2));
ll_sum[i] = sum;
}
x += 4;
}
let coeffs_by_2 = coeffs.chunks_exact(2);
coeffs = coeffs_by_2.remainder();
for k in coeffs_by_2 {
let coeff0_i64x2 = i64x2_splat(k[0] as i64);
let coeff1_i64x2 = i64x2_splat(k[1] as i64);
for i in 0..4 {
let mut sum = ll_sum[i];
let source = wasm32_utils::loadl_i64(src_rows[i], x);
let p_i64x2 = i8x16_swizzle(source, p0_shuffle);
sum = i64x2_add(sum, i64x2_mul(p_i64x2, coeff0_i64x2));
let p_i64x2 = i8x16_swizzle(source, p1_shuffle);
sum = i64x2_add(sum, i64x2_mul(p_i64x2, coeff1_i64x2));
ll_sum[i] = sum;
}
x += 2;
}
if let Some(&k) = coeffs.first() {
let coeff0_i64x2 = i64x2_splat(k as i64);
for i in 0..4 {
let source = wasm32_utils::loadl_i32(src_rows[i], x);
let p_i64x2 = i8x16_swizzle(source, p0_shuffle);
ll_sum[i] = i64x2_add(ll_sum[i], i64x2_mul(p_i64x2, coeff0_i64x2));
}
}
for i in 0..4 {
v128_store((&mut ll_buf).as_mut_ptr() as *mut v128, ll_sum[i]);
let dst_pixel = dst_rows[i].get_unchecked_mut(dst_x);
dst_pixel.0 = [normalizer.clip(ll_buf[0]), normalizer.clip(ll_buf[1])];
}
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - bounds.len() == dst_row.len()
/// - coeffs.len() == dst_rows.0.len() * window_size
/// - max(bound.start + bound.size for bound in bounds) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
#[inline]
unsafe fn horiz_convolution_one_row(
src_row: &[U16x2],
dst_row: &mut [U16x2],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
normalizer: &optimisations::Normalizer32,
) {
let precision = normalizer.precision();
let half_error = 1i64 << (precision - 1);
let mut ll_buf = [0i64; 2];
/*
|L0 A0 | |L1 A1 | |L2 A2 | |L3 A3 |
|0001 0203| |0405 0607| |0809 1011| |1213 1415|
Shuffle to extract L0 and A0 as i64:
0, 1, -1, -1, -1, -1, -1, -1, 2, 3, -1, -1, -1, -1, -1, -1
Shuffle to extract L1 and A1 as i64:
4, 5, -1, -1, -1, -1, -1, -1, 6, 7, -1, -1, -1, -1, -1, -1
Shuffle to extract L2 and A2 as i64:
8, 9, -1, -1, -1, -1, -1, -1, 10, 11, -1, -1, -1, -1, -1, -1
Shuffle to extract L3 and A3 as i64:
12, 13, -1, -1, -1, -1, -1, -1, 14, 15, -1, -1, -1, -1, -1, -1
*/
let p0_shuffle = i8x16(0, 1, -1, -1, -1, -1, -1, -1, 2, 3, -1, -1, -1, -1, -1, -1);
let p1_shuffle = i8x16(4, 5, -1, -1, -1, -1, -1, -1, 6, 7, -1, -1, -1, -1, -1, -1);
let p2_shuffle = i8x16(8, 9, -1, -1, -1, -1, -1, -1, 10, 11, -1, -1, -1, -1, -1, -1);
let p3_shuffle = i8x16(
12, 13, -1, -1, -1, -1, -1, -1, 14, 15, -1, -1, -1, -1, -1, -1,
);
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
let mut ll_sum = i64x2_splat(half_error);
let mut coeffs = coeffs_chunk.values;
let coeffs_by_4 = coeffs.chunks_exact(4);
coeffs = coeffs_by_4.remainder();
for k in coeffs_by_4 {
let coeff0_i64x2 = i64x2_splat(k[0] as i64);
let coeff1_i64x2 = i64x2_splat(k[1] as i64);
let coeff2_i64x2 = i64x2_splat(k[2] as i64);
let coeff3_i64x2 = i64x2_splat(k[3] as i64);
let source = wasm32_utils::load_v128(src_row, x);
let p_i64x2 = i8x16_swizzle(source, p0_shuffle);
ll_sum = i64x2_add(ll_sum, i64x2_mul(p_i64x2, coeff0_i64x2));
let p_i64x2 = i8x16_swizzle(source, p1_shuffle);
ll_sum = i64x2_add(ll_sum, i64x2_mul(p_i64x2, coeff1_i64x2));
let p_i64x2 = i8x16_swizzle(source, p2_shuffle);
ll_sum = i64x2_add(ll_sum, i64x2_mul(p_i64x2, coeff2_i64x2));
let p_i64x2 = i8x16_swizzle(source, p3_shuffle);
ll_sum = i64x2_add(ll_sum, i64x2_mul(p_i64x2, coeff3_i64x2));
x += 4;
}
let coeffs_by_2 = coeffs.chunks_exact(2);
coeffs = coeffs_by_2.remainder();
for k in coeffs_by_2 {
let coeff0_i64x2 = i64x2_splat(k[0] as i64);
let coeff1_i64x2 = i64x2_splat(k[1] as i64);
let source = wasm32_utils::loadl_i64(src_row, x);
let p_i64x2 = i8x16_swizzle(source, p0_shuffle);
ll_sum = i64x2_add(ll_sum, i64x2_mul(p_i64x2, coeff0_i64x2));
let p_i64x2 = i8x16_swizzle(source, p1_shuffle);
ll_sum = i64x2_add(ll_sum, i64x2_mul(p_i64x2, coeff1_i64x2));
x += 2;
}
if let Some(&k) = coeffs.first() {
let coeff0_i64x2 = i64x2_splat(k as i64);
let source = wasm32_utils::loadl_i32(src_row, x);
let p_i64x2 = i8x16_swizzle(source, p0_shuffle);
ll_sum = i64x2_add(ll_sum, i64x2_mul(p_i64x2, coeff0_i64x2));
}
v128_store((&mut ll_buf).as_mut_ptr() as *mut v128, ll_sum);
let dst_pixel = dst_row.get_unchecked_mut(dst_x);
dst_pixel.0 = [normalizer.clip(ll_buf[0]), normalizer.clip(ll_buf[1])];
}
}
+3 -1
View File
@@ -12,6 +12,8 @@ mod native;
mod neon;
#[cfg(target_arch = "x86_64")]
mod sse4;
#[cfg(target_arch = "wasm32")]
mod wasm32;
impl Convolution for U16x3 {
fn horiz_convolution(
@@ -30,7 +32,7 @@ impl Convolution for U16x3 {
CpuExtensions::Neon => neon::horiz_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => {
native::horiz_convolution(src_image, dst_image, offset, coeffs)
wasm32::horiz_convolution(src_image, dst_image, offset, coeffs)
}
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
}
+226
View File
@@ -0,0 +1,226 @@
use std::arch::wasm32::*;
use crate::convolution::optimisations::CoefficientsI32Chunk;
use crate::convolution::{optimisations, Coefficients};
use crate::pixels::U16x3;
use crate::wasm32_utils;
use crate::{ImageView, ImageViewMut};
#[inline]
pub(crate) fn horiz_convolution(
src_image: &ImageView<U16x3>,
dst_image: &mut ImageViewMut<U16x3>,
offset: u32,
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer32::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_image.height().get();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_image.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_8u4x(src_rows, dst_rows, &coefficients_chunks, &normalizer);
}
}
let mut yy = dst_height - dst_height % 4;
while yy < dst_height {
unsafe {
horiz_convolution_8u(
src_image.get_row(yy + offset).unwrap(),
dst_image.get_row_mut(yy).unwrap(),
&coefficients_chunks,
&normalizer,
);
}
yy += 1;
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - length of all rows in src_rows must be equal
/// - length of all rows in dst_rows must be equal
/// - coefficients_chunks.len() == dst_rows.0.len()
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
unsafe fn horiz_convolution_8u4x(
src_rows: [&[U16x3]; 4],
dst_rows: [&mut &mut [U16x3]; 4],
coefficients_chunks: &[CoefficientsI32Chunk],
normalizer: &optimisations::Normalizer32,
) {
let precision = normalizer.precision();
let half_error = 1i64 << (precision - 1);
let mut rg_buf = [0i64; 2];
let mut bb_buf = [0i64; 2];
/*
|R G B | |R G B | |R G |
|0001 0203 0405| |0607 0809 1011| |1213 1415|
Shuffle to extract RG components of first pixel as i64:
0, 1, -1, -1, -1, -1, -1, -1, 2, 3, -1, -1, -1, -1, -1, -1
Shuffle to extract RG components of second pixel as i64:
6, 7, -1, -1, -1, -1, -1, -1, 8, 9, -1, -1, -1, -1, -1, -1
Shuffle to extract B components of two pixels as i64:
4, 5, -1, -1, -1, -1, -1, -1, 10, 11, -1, -1, -1, -1, -1, -1
*/
let rg0_shuffle = i8x16(0, 1, -1, -1, -1, -1, -1, -1, 2, 3, -1, -1, -1, -1, -1, -1);
let rg1_shuffle = i8x16(6, 7, -1, -1, -1, -1, -1, -1, 8, 9, -1, -1, -1, -1, -1, -1);
let bb_shuffle = i8x16(4, 5, -1, -1, -1, -1, -1, -1, 10, 11, -1, -1, -1, -1, -1, -1);
let width = src_rows[0].len();
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
let mut rg_sum = [i8x16_splat(0); 4];
let mut bb_sum = [i8x16_splat(0); 4];
let mut coeffs = coeffs_chunk.values;
let end_x = x + coeffs.len();
if width - end_x >= 1 {
let coeffs_by_2 = coeffs.chunks_exact(2);
coeffs = coeffs_by_2.remainder();
for k in coeffs_by_2 {
let coeff0_i64x2 = i64x2_splat(k[0] as i64);
let coeff1_i64x2 = i64x2_splat(k[1] as i64);
let coeff_i64x2 = i64x2(k[0] as i64, k[1] as i64);
for i in 0..4 {
let source = wasm32_utils::load_v128(src_rows[i], x);
let rg0_i64x2 = i8x16_swizzle(source, rg0_shuffle);
rg_sum[i] = i64x2_add(rg_sum[i], i64x2_mul(rg0_i64x2, coeff0_i64x2));
let rg1_i64x2 = i8x16_swizzle(source, rg1_shuffle);
rg_sum[i] = i64x2_add(rg_sum[i], i64x2_mul(rg1_i64x2, coeff1_i64x2));
let bb_i64x2 = i8x16_swizzle(source, bb_shuffle);
bb_sum[i] = i64x2_add(bb_sum[i], i64x2_mul(bb_i64x2, coeff_i64x2));
}
x += 2;
}
}
for &k in coeffs {
let coeff_i64x2 = i64x2_splat(k as i64);
for i in 0..4 {
let &pixel = src_rows[i].get_unchecked(x);
let rg_i64x2 = i64x2(pixel.0[0] as i64, pixel.0[1] as i64);
rg_sum[i] = i64x2_add(rg_sum[i], i64x2_mul(rg_i64x2, coeff_i64x2));
let bb_i64x2 = i64x2(pixel.0[2] as i64, 0);
bb_sum[i] = i64x2_add(bb_sum[i], i64x2_mul(bb_i64x2, coeff_i64x2));
}
x += 1;
}
for i in 0..4 {
v128_store((&mut rg_buf).as_mut_ptr() as *mut v128, rg_sum[i]);
v128_store((&mut bb_buf).as_mut_ptr() as *mut v128, bb_sum[i]);
let dst_pixel = dst_rows[i].get_unchecked_mut(dst_x);
dst_pixel.0[0] = normalizer.clip(rg_buf[0] + half_error);
dst_pixel.0[1] = normalizer.clip(rg_buf[1] + half_error);
dst_pixel.0[2] = normalizer.clip(bb_buf[0] + bb_buf[1] + half_error);
}
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - bounds.len() == dst_row.len()
/// - coefficients_chunks.len() == dst_row.len()
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
unsafe fn horiz_convolution_8u(
src_row: &[U16x3],
dst_row: &mut [U16x3],
coefficients_chunks: &[CoefficientsI32Chunk],
normalizer: &optimisations::Normalizer32,
) {
let precision = normalizer.precision();
let rg_initial = i64x2_splat(1 << (precision - 1));
let bb_initial = i64x2_splat(1 << (precision - 2));
/*
|R G B | |R G B | |R G |
|0001 0203 0405| |0607 0809 1011| |1213 1415|
Shuffle to extract RG components of first pixel as i64:
0, 1, -1, -1, -1, -1, -1, -1, 2, 3, -1, -1, -1, -1, -1, -1
Shuffle to extract RG components of second pixel as i64:
6, 7, -1, -1, -1, -1, -1, -1, 8, 9, -1, -1, -1, -1, -1, -1
Shuffle to extract B components of two pixels as i64:
4, 5, -1, -1, -1, -1, -1, -1, 10, 11, -1, -1, -1, -1, -1, -1
*/
let rg0_shuffle = i8x16(0, 1, -1, -1, -1, -1, -1, -1, 2, 3, -1, -1, -1, -1, -1, -1);
let rg1_shuffle = i8x16(6, 7, -1, -1, -1, -1, -1, -1, 8, 9, -1, -1, -1, -1, -1, -1);
let bb_shuffle = i8x16(4, 5, -1, -1, -1, -1, -1, -1, 10, 11, -1, -1, -1, -1, -1, -1);
let mut rg_buf = [0i64; 2];
let mut bb_buf = [0i64; 2];
let width = src_row.len();
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
let mut rg_sum = rg_initial;
let mut bb_sum = bb_initial;
let mut coeffs = coeffs_chunk.values;
let end_x = x + coeffs.len();
if width - end_x >= 1 {
let coeffs_by_2 = coeffs.chunks_exact(2);
coeffs = coeffs_by_2.remainder();
for k in coeffs_by_2 {
let coeff0_i64x2 = i64x2_splat(k[0] as i64);
let coeff1_i64x2 = i64x2_splat(k[1] as i64);
let coeff_i64x2 = i64x2(k[0] as i64, k[1] as i64);
let source = wasm32_utils::load_v128(src_row, x);
let rg0_i64x2 = i8x16_swizzle(source, rg0_shuffle);
rg_sum = i64x2_add(rg_sum, i64x2_mul(rg0_i64x2, coeff0_i64x2));
let rg1_i64x2 = i8x16_swizzle(source, rg1_shuffle);
rg_sum = i64x2_add(rg_sum, i64x2_mul(rg1_i64x2, coeff1_i64x2));
let bb_i64x2 = i8x16_swizzle(source, bb_shuffle);
bb_sum = i64x2_add(bb_sum, i64x2_mul(bb_i64x2, coeff_i64x2));
x += 2;
}
}
for &k in coeffs {
let coeff_i64x2 = i64x2_splat(k as i64);
let &pixel = src_row.get_unchecked(x);
let rg_i64x2 = i64x2(pixel.0[0] as i64, pixel.0[1] as i64);
rg_sum = i64x2_add(rg_sum, i64x2_mul(rg_i64x2, coeff_i64x2));
let bb_i64x2 = i64x2(pixel.0[2] as i64, 0);
bb_sum = i64x2_add(bb_sum, i64x2_mul(bb_i64x2, coeff_i64x2));
x += 1;
}
v128_store((&mut rg_buf).as_mut_ptr() as *mut v128, rg_sum);
v128_store((&mut bb_buf).as_mut_ptr() as *mut v128, bb_sum);
let dst_pixel = dst_row.get_unchecked_mut(dst_x);
dst_pixel.0[0] = normalizer.clip(rg_buf[0]);
dst_pixel.0[1] = normalizer.clip(rg_buf[1]);
dst_pixel.0[2] = normalizer.clip(bb_buf[0] + bb_buf[1]);
}
}
+3 -1
View File
@@ -12,6 +12,8 @@ mod native;
mod neon;
#[cfg(target_arch = "x86_64")]
mod sse4;
#[cfg(target_arch = "wasm32")]
mod wasm32;
impl Convolution for U16x4 {
fn horiz_convolution(
@@ -30,7 +32,7 @@ impl Convolution for U16x4 {
CpuExtensions::Neon => neon::horiz_convolution(src_image, dst_image, offset, coeffs),
#[cfg(target_arch = "wasm32")]
CpuExtensions::Wasm32 => {
native::horiz_convolution(src_image, dst_image, offset, coeffs)
wasm32::horiz_convolution(src_image, dst_image, offset, coeffs)
}
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
}
+228
View File
@@ -0,0 +1,228 @@
use std::arch::wasm32::*;
use crate::convolution::{optimisations, Coefficients};
use crate::pixels::U16x4;
use crate::wasm32_utils;
use crate::{ImageView, ImageViewMut};
#[inline]
pub(crate) fn horiz_convolution(
src_image: &ImageView<U16x4>,
dst_image: &mut ImageViewMut<U16x4>,
offset: u32,
coeffs: Coefficients,
) {
let normalizer = optimisations::Normalizer32::new(coeffs);
let coefficients_chunks = normalizer.normalized_chunks();
let dst_height = dst_image.height().get();
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
let dst_iter = dst_image.iter_4_rows_mut();
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
unsafe {
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
}
}
let mut yy = dst_height - dst_height % 4;
while yy < dst_height {
unsafe {
horiz_convolution_one_row(
src_image.get_row(yy + offset).unwrap(),
dst_image.get_row_mut(yy).unwrap(),
&coefficients_chunks,
&normalizer,
);
}
yy += 1;
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - length of all rows in src_rows must be equal
/// - length of all rows in dst_rows must be equal
/// - coefficients_chunks.len() == dst_rows.0.len()
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
/// - precision <= MAX_COEFS_PRECISION
unsafe fn horiz_convolution_four_rows(
src_rows: [&[U16x4]; 4],
dst_rows: [&mut &mut [U16x4]; 4],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
normalizer: &optimisations::Normalizer32,
) {
let precision = normalizer.precision();
let half_error = 1i64 << (precision - 1);
let mut rg_buf = [0i64; 2];
let mut ba_buf = [0i64; 2];
/*
|R0 G0 B0 A0 | |R1 G1 B1 A1 |
|0001 0203 0405 0607| |0809 1011 1213 1415|
Shuffle to extract R0 and G0 as i64:
0, 1, -1, -1, -1, -1, -1, -1, 2, 3, -1, -1, -1, -1, -1, -1
Shuffle to extract R1 and G1 as i64:
8, 9, -1, -1, -1, -1, -1, -1, 10, 11, -1, -1, -1, -1, -1, -1
Shuffle to extract B0 and A0 as i64:
4, 5, -1, -1, -1, -1, -1, -1, 6, 7, -1, -1, -1, -1, -1, -1
Shuffle to extract B1 and A1 as i64:
12, 13, -1, -1, -1, -1, -1, -1, 14, 15, -1, -1, -1, -1, -1, -1
*/
let rg0_shuffle = i8x16(0, 1, -1, -1, -1, -1, -1, -1, 2, 3, -1, -1, -1, -1, -1, -1);
let rg1_shuffle = i8x16(8, 9, -1, -1, -1, -1, -1, -1, 10, 11, -1, -1, -1, -1, -1, -1);
let ba0_shuffle = i8x16(4, 5, -1, -1, -1, -1, -1, -1, 6, 7, -1, -1, -1, -1, -1, -1);
let ba1_shuffle = i8x16(
12, 13, -1, -1, -1, -1, -1, -1, 14, 15, -1, -1, -1, -1, -1, -1,
);
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
let mut rg_sum = [i64x2_splat(half_error); 4];
let mut ba_sum = [i64x2_splat(half_error); 4];
let mut coeffs = coeffs_chunk.values;
let coeffs_by_2 = coeffs.chunks_exact(2);
coeffs = coeffs_by_2.remainder();
for k in coeffs_by_2 {
let coeff0_i64x2 = i64x2_splat(k[0] as i64);
let coeff1_i64x2 = i64x2_splat(k[1] as i64);
for i in 0..4 {
let source = wasm32_utils::load_v128(src_rows[i], x);
let mut sum = rg_sum[i];
let rg_i64x2 = i8x16_swizzle(source, rg0_shuffle);
sum = i64x2_add(sum, i64x2_mul(rg_i64x2, coeff0_i64x2));
let rg_i64x2 = i8x16_swizzle(source, rg1_shuffle);
sum = i64x2_add(sum, i64x2_mul(rg_i64x2, coeff1_i64x2));
rg_sum[i] = sum;
let mut sum = ba_sum[i];
let ba_i64x2 = i8x16_swizzle(source, ba0_shuffle);
sum = i64x2_add(sum, i64x2_mul(ba_i64x2, coeff0_i64x2));
let ba_i64x2 = i8x16_swizzle(source, ba1_shuffle);
sum = i64x2_add(sum, i64x2_mul(ba_i64x2, coeff1_i64x2));
ba_sum[i] = sum;
}
x += 2;
}
if let Some(&k) = coeffs.first() {
let coeff0_i64x2 = i64x2_splat(k as i64);
for i in 0..4 {
let source = wasm32_utils::loadl_i64(src_rows[i], x);
let rg_i64x2 = i8x16_swizzle(source, rg0_shuffle);
rg_sum[i] = i64x2_add(rg_sum[i], i64x2_mul(rg_i64x2, coeff0_i64x2));
let ba_i64x2 = i8x16_swizzle(source, ba0_shuffle);
ba_sum[i] = i64x2_add(ba_sum[i], i64x2_mul(ba_i64x2, coeff0_i64x2));
}
}
for i in 0..4 {
v128_store((&mut rg_buf).as_mut_ptr() as *mut v128, rg_sum[i]);
v128_store((&mut ba_buf).as_mut_ptr() as *mut v128, ba_sum[i]);
let dst_pixel = dst_rows[i].get_unchecked_mut(dst_x);
dst_pixel.0 = [
normalizer.clip(rg_buf[0]),
normalizer.clip(rg_buf[1]),
normalizer.clip(ba_buf[0]),
normalizer.clip(ba_buf[1]),
];
}
}
}
/// For safety, it is necessary to ensure the following conditions:
/// - bounds.len() == dst_row.len()
/// - coeffs.len() == dst_rows.0.len() * window_size
/// - max(bound.start + bound.size for bound in bounds) <= src_row.len()
/// - precision <= MAX_COEFS_PRECISION
#[inline]
unsafe fn horiz_convolution_one_row(
src_row: &[U16x4],
dst_row: &mut [U16x4],
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
normalizer: &optimisations::Normalizer32,
) {
let precision = normalizer.precision();
let half_error = 1i64 << (precision - 1);
let mut rg_buf = [0i64; 2];
let mut ba_buf = [0i64; 2];
/*
|R0 G0 B0 A0 | |R1 G1 B1 A1 |
|0001 0203 0405 0607| |0809 1011 1213 1415|
Shuffle to extract R0 and G0 as i64:
0, 1, -1, -1, -1, -1, -1, -1, 2, 3, -1, -1, -1, -1, -1, -1
Shuffle to extract R1 and G1 as i64:
8, 9, -1, -1, -1, -1, -1, -1, 10, 11, -1, -1, -1, -1, -1, -1
Shuffle to extract B0 and A0 as i64:
4, 5, -1, -1, -1, -1, -1, -1, 6, 7, -1, -1, -1, -1, -1, -1
Shuffle to extract B1 and A1 as i64:
12, 13, -1, -1, -1, -1, -1, -1, 14, 15, -1, -1, -1, -1, -1, -1
*/
let rg0_shuffle = i8x16(0, 1, -1, -1, -1, -1, -1, -1, 2, 3, -1, -1, -1, -1, -1, -1);
let rg1_shuffle = i8x16(8, 9, -1, -1, -1, -1, -1, -1, 10, 11, -1, -1, -1, -1, -1, -1);
let ba0_shuffle = i8x16(4, 5, -1, -1, -1, -1, -1, -1, 6, 7, -1, -1, -1, -1, -1, -1);
let ba1_shuffle = i8x16(
12, 13, -1, -1, -1, -1, -1, -1, 14, 15, -1, -1, -1, -1, -1, -1,
);
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
let mut coeffs = coeffs_chunk.values;
let mut rg_sum = i64x2_splat(half_error);
let mut ba_sum = i64x2_splat(half_error);
let coeffs_by_2 = coeffs.chunks_exact(2);
coeffs = coeffs_by_2.remainder();
for k in coeffs_by_2 {
let coeff0_i64x2 = i64x2_splat(k[0] as i64);
let coeff1_i64x2 = i64x2_splat(k[1] as i64);
let source = wasm32_utils::load_v128(src_row, x);
let rg_i64x2 = i8x16_swizzle(source, rg0_shuffle);
rg_sum = i64x2_add(rg_sum, i64x2_mul(rg_i64x2, coeff0_i64x2));
let rg_i64x2 = i8x16_swizzle(source, rg1_shuffle);
rg_sum = i64x2_add(rg_sum, i64x2_mul(rg_i64x2, coeff1_i64x2));
let ba_i64x2 = i8x16_swizzle(source, ba0_shuffle);
ba_sum = i64x2_add(ba_sum, i64x2_mul(ba_i64x2, coeff0_i64x2));
let ba_i64x2 = i8x16_swizzle(source, ba1_shuffle);
ba_sum = i64x2_add(ba_sum, i64x2_mul(ba_i64x2, coeff1_i64x2));
x += 2;
}
if let Some(&k) = coeffs.first() {
let coeff0_i64x2 = i64x2_splat(k as i64);
let source = wasm32_utils::loadl_i64(src_row, x);
let rg_i64x2 = i8x16_swizzle(source, rg0_shuffle);
rg_sum = i64x2_add(rg_sum, i64x2_mul(rg_i64x2, coeff0_i64x2));
let ba_i64x2 = i8x16_swizzle(source, ba0_shuffle);
ba_sum = i64x2_add(ba_sum, i64x2_mul(ba_i64x2, coeff0_i64x2));
}
v128_store((&mut rg_buf).as_mut_ptr() as *mut v128, rg_sum);
v128_store((&mut ba_buf).as_mut_ptr() as *mut v128, ba_sum);
let dst_pixel = dst_row.get_unchecked_mut(dst_x);
dst_pixel.0 = [
normalizer.clip(rg_buf[0]),
normalizer.clip(rg_buf[1]),
normalizer.clip(ba_buf[0]),
normalizer.clip(ba_buf[1]),
];
}
}
+28
View File
@@ -654,6 +654,10 @@ fn upscale_u16() {
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
@@ -682,6 +686,10 @@ fn downscale_u16x2() {
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
}
for cpu_extensions in cpu_extensions_vec {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
@@ -710,6 +718,10 @@ fn upscale_u16x2() {
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
@@ -738,6 +750,10 @@ fn downscale_u16x3() {
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
}
for cpu_extensions in cpu_extensions_vec {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
@@ -766,6 +782,10 @@ fn upscale_u16x3() {
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
@@ -794,6 +814,10 @@ fn downscale_u16x4() {
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
}
for cpu_extensions in cpu_extensions_vec {
P::downscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),
@@ -822,6 +846,10 @@ fn upscale_u16x4() {
{
cpu_extensions_vec.push(CpuExtensions::Neon);
}
#[cfg(target_arch = "wasm32")]
{
cpu_extensions_vec.push(CpuExtensions::Wasm32);
}
for cpu_extensions in cpu_extensions_vec {
P::upscale_test(
ResizeAlg::Convolution(FilterType::Lanczos3),