mirror of
https://github.com/Cykooz/fast_image_resize.git
synced 2026-10-08 01:11:09 +00:00
Resize Complete
This commit is contained in:
@@ -0,0 +1,257 @@
|
||||
use std::arch::wasm32::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::pixels::U16x2;
|
||||
use crate::wasm32_utils;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn horiz_convolution(
|
||||
src_image: &ImageView<U16x2>,
|
||||
dst_image: &mut ImageViewMut<U16x2>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
let mut yy = dst_height - dst_height % 4;
|
||||
while yy < dst_height {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
&coefficients_chunks,
|
||||
&normalizer,
|
||||
);
|
||||
}
|
||||
yy += 1;
|
||||
}
|
||||
}
|
||||
|
||||
/// For safety, it is necessary to ensure the following conditions:
|
||||
/// - length of all rows in src_rows must be equal
|
||||
/// - length of all rows in dst_rows must be equal
|
||||
/// - coefficients_chunks.len() == dst_rows.0.len()
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U16x2]; 4],
|
||||
dst_rows: [&mut &mut [U16x2]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 2];
|
||||
|
||||
/*
|
||||
|L0 A0 | |L1 A1 | |L2 A2 | |L3 A3 |
|
||||
|0001 0203| |0405 0607| |0809 1011| |1213 1415|
|
||||
|
||||
Shuffle to extract L0 and A0 as i64:
|
||||
0, 1, -1, -1, -1, -1, -1, -1, 2, 3, -1, -1, -1, -1, -1, -1
|
||||
|
||||
Shuffle to extract L1 and A1 as i64:
|
||||
4, 5, -1, -1, -1, -1, -1, -1, 6, 7, -1, -1, -1, -1, -1, -1
|
||||
|
||||
Shuffle to extract L2 and A2 as i64:
|
||||
8, 9, -1, -1, -1, -1, -1, -1, 10, 11, -1, -1, -1, -1, -1, -1
|
||||
|
||||
Shuffle to extract L3 and A3 as i64:
|
||||
12, 13, -1, -1, -1, -1, -1, -1, 14, 15, -1, -1, -1, -1, -1, -1
|
||||
*/
|
||||
|
||||
let p0_shuffle = i8x16(0, 1, -1, -1, -1, -1, -1, -1, 2, 3, -1, -1, -1, -1, -1, -1);
|
||||
let p1_shuffle = i8x16(4, 5, -1, -1, -1, -1, -1, -1, 6, 7, -1, -1, -1, -1, -1, -1);
|
||||
let p2_shuffle = i8x16(8, 9, -1, -1, -1, -1, -1, -1, 10, 11, -1, -1, -1, -1, -1, -1);
|
||||
let p3_shuffle = i8x16(
|
||||
12, 13, -1, -1, -1, -1, -1, -1, 14, 15, -1, -1, -1, -1, -1, -1,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut ll_sum = [i64x2_splat(half_error); 4];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
coeffs = coeffs_by_4.remainder();
|
||||
|
||||
for k in coeffs_by_4 {
|
||||
let coeff0_i64x2 = i64x2_splat(k[0] as i64);
|
||||
let coeff1_i64x2 = i64x2_splat(k[1] as i64);
|
||||
let coeff2_i64x2 = i64x2_splat(k[2] as i64);
|
||||
let coeff3_i64x2 = i64x2_splat(k[3] as i64);
|
||||
|
||||
for i in 0..4 {
|
||||
let mut sum = ll_sum[i];
|
||||
let source = wasm32_utils::load_v128(src_rows[i], x);
|
||||
|
||||
let p_i64x2 = i8x16_swizzle(source, p0_shuffle);
|
||||
sum = i64x2_add(sum, i64x2_mul(p_i64x2, coeff0_i64x2));
|
||||
|
||||
let p_i64x2 = i8x16_swizzle(source, p1_shuffle);
|
||||
sum = i64x2_add(sum, i64x2_mul(p_i64x2, coeff1_i64x2));
|
||||
|
||||
let p_i64x2 = i8x16_swizzle(source, p2_shuffle);
|
||||
sum = i64x2_add(sum, i64x2_mul(p_i64x2, coeff2_i64x2));
|
||||
|
||||
let p_i64x2 = i8x16_swizzle(source, p3_shuffle);
|
||||
sum = i64x2_add(sum, i64x2_mul(p_i64x2, coeff3_i64x2));
|
||||
|
||||
ll_sum[i] = sum;
|
||||
}
|
||||
x += 4;
|
||||
}
|
||||
|
||||
let coeffs_by_2 = coeffs.chunks_exact(2);
|
||||
coeffs = coeffs_by_2.remainder();
|
||||
|
||||
for k in coeffs_by_2 {
|
||||
let coeff0_i64x2 = i64x2_splat(k[0] as i64);
|
||||
let coeff1_i64x2 = i64x2_splat(k[1] as i64);
|
||||
|
||||
for i in 0..4 {
|
||||
let mut sum = ll_sum[i];
|
||||
let source = wasm32_utils::loadl_i64(src_rows[i], x);
|
||||
|
||||
let p_i64x2 = i8x16_swizzle(source, p0_shuffle);
|
||||
sum = i64x2_add(sum, i64x2_mul(p_i64x2, coeff0_i64x2));
|
||||
|
||||
let p_i64x2 = i8x16_swizzle(source, p1_shuffle);
|
||||
sum = i64x2_add(sum, i64x2_mul(p_i64x2, coeff1_i64x2));
|
||||
|
||||
ll_sum[i] = sum;
|
||||
}
|
||||
x += 2;
|
||||
}
|
||||
|
||||
if let Some(&k) = coeffs.first() {
|
||||
let coeff0_i64x2 = i64x2_splat(k as i64);
|
||||
for i in 0..4 {
|
||||
let source = wasm32_utils::loadl_i32(src_rows[i], x);
|
||||
let p_i64x2 = i8x16_swizzle(source, p0_shuffle);
|
||||
ll_sum[i] = i64x2_add(ll_sum[i], i64x2_mul(p_i64x2, coeff0_i64x2));
|
||||
}
|
||||
}
|
||||
|
||||
for i in 0..4 {
|
||||
v128_store((&mut ll_buf).as_mut_ptr() as *mut v128, ll_sum[i]);
|
||||
let dst_pixel = dst_rows[i].get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = [normalizer.clip(ll_buf[0]), normalizer.clip(ll_buf[1])];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// For safety, it is necessary to ensure the following conditions:
|
||||
/// - bounds.len() == dst_row.len()
|
||||
/// - coeffs.len() == dst_rows.0.len() * window_size
|
||||
/// - max(bound.start + bound.size for bound in bounds) <= src_row.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[inline]
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16x2],
|
||||
dst_row: &mut [U16x2],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 2];
|
||||
|
||||
/*
|
||||
|L0 A0 | |L1 A1 | |L2 A2 | |L3 A3 |
|
||||
|0001 0203| |0405 0607| |0809 1011| |1213 1415|
|
||||
|
||||
Shuffle to extract L0 and A0 as i64:
|
||||
0, 1, -1, -1, -1, -1, -1, -1, 2, 3, -1, -1, -1, -1, -1, -1
|
||||
|
||||
Shuffle to extract L1 and A1 as i64:
|
||||
4, 5, -1, -1, -1, -1, -1, -1, 6, 7, -1, -1, -1, -1, -1, -1
|
||||
|
||||
Shuffle to extract L2 and A2 as i64:
|
||||
8, 9, -1, -1, -1, -1, -1, -1, 10, 11, -1, -1, -1, -1, -1, -1
|
||||
|
||||
Shuffle to extract L3 and A3 as i64:
|
||||
12, 13, -1, -1, -1, -1, -1, -1, 14, 15, -1, -1, -1, -1, -1, -1
|
||||
*/
|
||||
|
||||
let p0_shuffle = i8x16(0, 1, -1, -1, -1, -1, -1, -1, 2, 3, -1, -1, -1, -1, -1, -1);
|
||||
let p1_shuffle = i8x16(4, 5, -1, -1, -1, -1, -1, -1, 6, 7, -1, -1, -1, -1, -1, -1);
|
||||
let p2_shuffle = i8x16(8, 9, -1, -1, -1, -1, -1, -1, 10, 11, -1, -1, -1, -1, -1, -1);
|
||||
let p3_shuffle = i8x16(
|
||||
12, 13, -1, -1, -1, -1, -1, -1, 14, 15, -1, -1, -1, -1, -1, -1,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut ll_sum = i64x2_splat(half_error);
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
|
||||
let coeffs_by_4 = coeffs.chunks_exact(4);
|
||||
coeffs = coeffs_by_4.remainder();
|
||||
|
||||
for k in coeffs_by_4 {
|
||||
let coeff0_i64x2 = i64x2_splat(k[0] as i64);
|
||||
let coeff1_i64x2 = i64x2_splat(k[1] as i64);
|
||||
let coeff2_i64x2 = i64x2_splat(k[2] as i64);
|
||||
let coeff3_i64x2 = i64x2_splat(k[3] as i64);
|
||||
|
||||
let source = wasm32_utils::load_v128(src_row, x);
|
||||
|
||||
let p_i64x2 = i8x16_swizzle(source, p0_shuffle);
|
||||
ll_sum = i64x2_add(ll_sum, i64x2_mul(p_i64x2, coeff0_i64x2));
|
||||
|
||||
let p_i64x2 = i8x16_swizzle(source, p1_shuffle);
|
||||
ll_sum = i64x2_add(ll_sum, i64x2_mul(p_i64x2, coeff1_i64x2));
|
||||
|
||||
let p_i64x2 = i8x16_swizzle(source, p2_shuffle);
|
||||
ll_sum = i64x2_add(ll_sum, i64x2_mul(p_i64x2, coeff2_i64x2));
|
||||
|
||||
let p_i64x2 = i8x16_swizzle(source, p3_shuffle);
|
||||
ll_sum = i64x2_add(ll_sum, i64x2_mul(p_i64x2, coeff3_i64x2));
|
||||
|
||||
x += 4;
|
||||
}
|
||||
|
||||
let coeffs_by_2 = coeffs.chunks_exact(2);
|
||||
coeffs = coeffs_by_2.remainder();
|
||||
|
||||
for k in coeffs_by_2 {
|
||||
let coeff0_i64x2 = i64x2_splat(k[0] as i64);
|
||||
let coeff1_i64x2 = i64x2_splat(k[1] as i64);
|
||||
|
||||
let source = wasm32_utils::loadl_i64(src_row, x);
|
||||
|
||||
let p_i64x2 = i8x16_swizzle(source, p0_shuffle);
|
||||
ll_sum = i64x2_add(ll_sum, i64x2_mul(p_i64x2, coeff0_i64x2));
|
||||
|
||||
let p_i64x2 = i8x16_swizzle(source, p1_shuffle);
|
||||
ll_sum = i64x2_add(ll_sum, i64x2_mul(p_i64x2, coeff1_i64x2));
|
||||
|
||||
x += 2;
|
||||
}
|
||||
|
||||
if let Some(&k) = coeffs.first() {
|
||||
let coeff0_i64x2 = i64x2_splat(k as i64);
|
||||
let source = wasm32_utils::loadl_i32(src_row, x);
|
||||
|
||||
let p_i64x2 = i8x16_swizzle(source, p0_shuffle);
|
||||
ll_sum = i64x2_add(ll_sum, i64x2_mul(p_i64x2, coeff0_i64x2));
|
||||
}
|
||||
|
||||
v128_store((&mut ll_buf).as_mut_ptr() as *mut v128, ll_sum);
|
||||
let dst_pixel = dst_row.get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = [normalizer.clip(ll_buf[0]), normalizer.clip(ll_buf[1])];
|
||||
}
|
||||
}
|
||||
@@ -12,6 +12,8 @@ mod native;
|
||||
mod neon;
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod sse4;
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
mod wasm32;
|
||||
|
||||
impl Convolution for U16x3 {
|
||||
fn horiz_convolution(
|
||||
@@ -30,7 +32,7 @@ impl Convolution for U16x3 {
|
||||
CpuExtensions::Neon => neon::horiz_convolution(src_image, dst_image, offset, coeffs),
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Wasm32 => {
|
||||
native::horiz_convolution(src_image, dst_image, offset, coeffs)
|
||||
wasm32::horiz_convolution(src_image, dst_image, offset, coeffs)
|
||||
}
|
||||
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
|
||||
}
|
||||
|
||||
@@ -0,0 +1,226 @@
|
||||
use std::arch::wasm32::*;
|
||||
|
||||
use crate::convolution::optimisations::CoefficientsI32Chunk;
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::pixels::U16x3;
|
||||
use crate::wasm32_utils;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn horiz_convolution(
|
||||
src_image: &ImageView<U16x3>,
|
||||
dst_image: &mut ImageViewMut<U16x3>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_8u4x(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
let mut yy = dst_height - dst_height % 4;
|
||||
while yy < dst_height {
|
||||
unsafe {
|
||||
horiz_convolution_8u(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
&coefficients_chunks,
|
||||
&normalizer,
|
||||
);
|
||||
}
|
||||
yy += 1;
|
||||
}
|
||||
}
|
||||
|
||||
/// For safety, it is necessary to ensure the following conditions:
|
||||
/// - length of all rows in src_rows must be equal
|
||||
/// - length of all rows in dst_rows must be equal
|
||||
/// - coefficients_chunks.len() == dst_rows.0.len()
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
unsafe fn horiz_convolution_8u4x(
|
||||
src_rows: [&[U16x3]; 4],
|
||||
dst_rows: [&mut &mut [U16x3]; 4],
|
||||
coefficients_chunks: &[CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut rg_buf = [0i64; 2];
|
||||
let mut bb_buf = [0i64; 2];
|
||||
|
||||
/*
|
||||
|R G B | |R G B | |R G |
|
||||
|0001 0203 0405| |0607 0809 1011| |1213 1415|
|
||||
|
||||
Shuffle to extract RG components of first pixel as i64:
|
||||
0, 1, -1, -1, -1, -1, -1, -1, 2, 3, -1, -1, -1, -1, -1, -1
|
||||
|
||||
Shuffle to extract RG components of second pixel as i64:
|
||||
6, 7, -1, -1, -1, -1, -1, -1, 8, 9, -1, -1, -1, -1, -1, -1
|
||||
|
||||
Shuffle to extract B components of two pixels as i64:
|
||||
4, 5, -1, -1, -1, -1, -1, -1, 10, 11, -1, -1, -1, -1, -1, -1
|
||||
|
||||
*/
|
||||
|
||||
let rg0_shuffle = i8x16(0, 1, -1, -1, -1, -1, -1, -1, 2, 3, -1, -1, -1, -1, -1, -1);
|
||||
let rg1_shuffle = i8x16(6, 7, -1, -1, -1, -1, -1, -1, 8, 9, -1, -1, -1, -1, -1, -1);
|
||||
let bb_shuffle = i8x16(4, 5, -1, -1, -1, -1, -1, -1, 10, 11, -1, -1, -1, -1, -1, -1);
|
||||
|
||||
let width = src_rows[0].len();
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut rg_sum = [i8x16_splat(0); 4];
|
||||
let mut bb_sum = [i8x16_splat(0); 4];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let end_x = x + coeffs.len();
|
||||
|
||||
if width - end_x >= 1 {
|
||||
let coeffs_by_2 = coeffs.chunks_exact(2);
|
||||
coeffs = coeffs_by_2.remainder();
|
||||
|
||||
for k in coeffs_by_2 {
|
||||
let coeff0_i64x2 = i64x2_splat(k[0] as i64);
|
||||
let coeff1_i64x2 = i64x2_splat(k[1] as i64);
|
||||
let coeff_i64x2 = i64x2(k[0] as i64, k[1] as i64);
|
||||
|
||||
for i in 0..4 {
|
||||
let source = wasm32_utils::load_v128(src_rows[i], x);
|
||||
|
||||
let rg0_i64x2 = i8x16_swizzle(source, rg0_shuffle);
|
||||
rg_sum[i] = i64x2_add(rg_sum[i], i64x2_mul(rg0_i64x2, coeff0_i64x2));
|
||||
|
||||
let rg1_i64x2 = i8x16_swizzle(source, rg1_shuffle);
|
||||
rg_sum[i] = i64x2_add(rg_sum[i], i64x2_mul(rg1_i64x2, coeff1_i64x2));
|
||||
|
||||
let bb_i64x2 = i8x16_swizzle(source, bb_shuffle);
|
||||
bb_sum[i] = i64x2_add(bb_sum[i], i64x2_mul(bb_i64x2, coeff_i64x2));
|
||||
}
|
||||
x += 2;
|
||||
}
|
||||
}
|
||||
|
||||
for &k in coeffs {
|
||||
let coeff_i64x2 = i64x2_splat(k as i64);
|
||||
|
||||
for i in 0..4 {
|
||||
let &pixel = src_rows[i].get_unchecked(x);
|
||||
let rg_i64x2 = i64x2(pixel.0[0] as i64, pixel.0[1] as i64);
|
||||
rg_sum[i] = i64x2_add(rg_sum[i], i64x2_mul(rg_i64x2, coeff_i64x2));
|
||||
let bb_i64x2 = i64x2(pixel.0[2] as i64, 0);
|
||||
bb_sum[i] = i64x2_add(bb_sum[i], i64x2_mul(bb_i64x2, coeff_i64x2));
|
||||
}
|
||||
x += 1;
|
||||
}
|
||||
|
||||
for i in 0..4 {
|
||||
v128_store((&mut rg_buf).as_mut_ptr() as *mut v128, rg_sum[i]);
|
||||
v128_store((&mut bb_buf).as_mut_ptr() as *mut v128, bb_sum[i]);
|
||||
let dst_pixel = dst_rows[i].get_unchecked_mut(dst_x);
|
||||
dst_pixel.0[0] = normalizer.clip(rg_buf[0] + half_error);
|
||||
dst_pixel.0[1] = normalizer.clip(rg_buf[1] + half_error);
|
||||
dst_pixel.0[2] = normalizer.clip(bb_buf[0] + bb_buf[1] + half_error);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// For safety, it is necessary to ensure the following conditions:
|
||||
/// - bounds.len() == dst_row.len()
|
||||
/// - coefficients_chunks.len() == dst_row.len()
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
unsafe fn horiz_convolution_8u(
|
||||
src_row: &[U16x3],
|
||||
dst_row: &mut [U16x3],
|
||||
coefficients_chunks: &[CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let rg_initial = i64x2_splat(1 << (precision - 1));
|
||||
let bb_initial = i64x2_splat(1 << (precision - 2));
|
||||
|
||||
/*
|
||||
|R G B | |R G B | |R G |
|
||||
|0001 0203 0405| |0607 0809 1011| |1213 1415|
|
||||
|
||||
Shuffle to extract RG components of first pixel as i64:
|
||||
0, 1, -1, -1, -1, -1, -1, -1, 2, 3, -1, -1, -1, -1, -1, -1
|
||||
|
||||
Shuffle to extract RG components of second pixel as i64:
|
||||
6, 7, -1, -1, -1, -1, -1, -1, 8, 9, -1, -1, -1, -1, -1, -1
|
||||
|
||||
Shuffle to extract B components of two pixels as i64:
|
||||
4, 5, -1, -1, -1, -1, -1, -1, 10, 11, -1, -1, -1, -1, -1, -1
|
||||
|
||||
*/
|
||||
|
||||
let rg0_shuffle = i8x16(0, 1, -1, -1, -1, -1, -1, -1, 2, 3, -1, -1, -1, -1, -1, -1);
|
||||
let rg1_shuffle = i8x16(6, 7, -1, -1, -1, -1, -1, -1, 8, 9, -1, -1, -1, -1, -1, -1);
|
||||
let bb_shuffle = i8x16(4, 5, -1, -1, -1, -1, -1, -1, 10, 11, -1, -1, -1, -1, -1, -1);
|
||||
let mut rg_buf = [0i64; 2];
|
||||
let mut bb_buf = [0i64; 2];
|
||||
|
||||
let width = src_row.len();
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
|
||||
let mut rg_sum = rg_initial;
|
||||
let mut bb_sum = bb_initial;
|
||||
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let end_x = x + coeffs.len();
|
||||
|
||||
if width - end_x >= 1 {
|
||||
let coeffs_by_2 = coeffs.chunks_exact(2);
|
||||
coeffs = coeffs_by_2.remainder();
|
||||
|
||||
for k in coeffs_by_2 {
|
||||
let coeff0_i64x2 = i64x2_splat(k[0] as i64);
|
||||
let coeff1_i64x2 = i64x2_splat(k[1] as i64);
|
||||
let coeff_i64x2 = i64x2(k[0] as i64, k[1] as i64);
|
||||
|
||||
let source = wasm32_utils::load_v128(src_row, x);
|
||||
|
||||
let rg0_i64x2 = i8x16_swizzle(source, rg0_shuffle);
|
||||
rg_sum = i64x2_add(rg_sum, i64x2_mul(rg0_i64x2, coeff0_i64x2));
|
||||
|
||||
let rg1_i64x2 = i8x16_swizzle(source, rg1_shuffle);
|
||||
rg_sum = i64x2_add(rg_sum, i64x2_mul(rg1_i64x2, coeff1_i64x2));
|
||||
|
||||
let bb_i64x2 = i8x16_swizzle(source, bb_shuffle);
|
||||
bb_sum = i64x2_add(bb_sum, i64x2_mul(bb_i64x2, coeff_i64x2));
|
||||
x += 2;
|
||||
}
|
||||
}
|
||||
|
||||
for &k in coeffs {
|
||||
let coeff_i64x2 = i64x2_splat(k as i64);
|
||||
|
||||
let &pixel = src_row.get_unchecked(x);
|
||||
let rg_i64x2 = i64x2(pixel.0[0] as i64, pixel.0[1] as i64);
|
||||
rg_sum = i64x2_add(rg_sum, i64x2_mul(rg_i64x2, coeff_i64x2));
|
||||
let bb_i64x2 = i64x2(pixel.0[2] as i64, 0);
|
||||
bb_sum = i64x2_add(bb_sum, i64x2_mul(bb_i64x2, coeff_i64x2));
|
||||
|
||||
x += 1;
|
||||
}
|
||||
|
||||
v128_store((&mut rg_buf).as_mut_ptr() as *mut v128, rg_sum);
|
||||
v128_store((&mut bb_buf).as_mut_ptr() as *mut v128, bb_sum);
|
||||
let dst_pixel = dst_row.get_unchecked_mut(dst_x);
|
||||
dst_pixel.0[0] = normalizer.clip(rg_buf[0]);
|
||||
dst_pixel.0[1] = normalizer.clip(rg_buf[1]);
|
||||
dst_pixel.0[2] = normalizer.clip(bb_buf[0] + bb_buf[1]);
|
||||
}
|
||||
}
|
||||
@@ -12,6 +12,8 @@ mod native;
|
||||
mod neon;
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod sse4;
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
mod wasm32;
|
||||
|
||||
impl Convolution for U16x4 {
|
||||
fn horiz_convolution(
|
||||
@@ -30,7 +32,7 @@ impl Convolution for U16x4 {
|
||||
CpuExtensions::Neon => neon::horiz_convolution(src_image, dst_image, offset, coeffs),
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
CpuExtensions::Wasm32 => {
|
||||
native::horiz_convolution(src_image, dst_image, offset, coeffs)
|
||||
wasm32::horiz_convolution(src_image, dst_image, offset, coeffs)
|
||||
}
|
||||
_ => native::horiz_convolution(src_image, dst_image, offset, coeffs),
|
||||
}
|
||||
|
||||
@@ -0,0 +1,228 @@
|
||||
use std::arch::wasm32::*;
|
||||
|
||||
use crate::convolution::{optimisations, Coefficients};
|
||||
use crate::pixels::U16x4;
|
||||
use crate::wasm32_utils;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
#[inline]
|
||||
pub(crate) fn horiz_convolution(
|
||||
src_image: &ImageView<U16x4>,
|
||||
dst_image: &mut ImageViewMut<U16x4>,
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
let mut yy = dst_height - dst_height % 4;
|
||||
while yy < dst_height {
|
||||
unsafe {
|
||||
horiz_convolution_one_row(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
&coefficients_chunks,
|
||||
&normalizer,
|
||||
);
|
||||
}
|
||||
yy += 1;
|
||||
}
|
||||
}
|
||||
|
||||
/// For safety, it is necessary to ensure the following conditions:
|
||||
/// - length of all rows in src_rows must be equal
|
||||
/// - length of all rows in dst_rows must be equal
|
||||
/// - coefficients_chunks.len() == dst_rows.0.len()
|
||||
/// - max(chunk.start + chunk.values.len() for chunk in coefficients_chunks) <= src_row.0.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: [&[U16x4]; 4],
|
||||
dst_rows: [&mut &mut [U16x4]; 4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut rg_buf = [0i64; 2];
|
||||
let mut ba_buf = [0i64; 2];
|
||||
|
||||
/*
|
||||
|R0 G0 B0 A0 | |R1 G1 B1 A1 |
|
||||
|0001 0203 0405 0607| |0809 1011 1213 1415|
|
||||
|
||||
Shuffle to extract R0 and G0 as i64:
|
||||
0, 1, -1, -1, -1, -1, -1, -1, 2, 3, -1, -1, -1, -1, -1, -1
|
||||
|
||||
Shuffle to extract R1 and G1 as i64:
|
||||
8, 9, -1, -1, -1, -1, -1, -1, 10, 11, -1, -1, -1, -1, -1, -1
|
||||
|
||||
Shuffle to extract B0 and A0 as i64:
|
||||
4, 5, -1, -1, -1, -1, -1, -1, 6, 7, -1, -1, -1, -1, -1, -1
|
||||
|
||||
Shuffle to extract B1 and A1 as i64:
|
||||
12, 13, -1, -1, -1, -1, -1, -1, 14, 15, -1, -1, -1, -1, -1, -1
|
||||
*/
|
||||
|
||||
let rg0_shuffle = i8x16(0, 1, -1, -1, -1, -1, -1, -1, 2, 3, -1, -1, -1, -1, -1, -1);
|
||||
let rg1_shuffle = i8x16(8, 9, -1, -1, -1, -1, -1, -1, 10, 11, -1, -1, -1, -1, -1, -1);
|
||||
let ba0_shuffle = i8x16(4, 5, -1, -1, -1, -1, -1, -1, 6, 7, -1, -1, -1, -1, -1, -1);
|
||||
let ba1_shuffle = i8x16(
|
||||
12, 13, -1, -1, -1, -1, -1, -1, 14, 15, -1, -1, -1, -1, -1, -1,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut rg_sum = [i64x2_splat(half_error); 4];
|
||||
let mut ba_sum = [i64x2_splat(half_error); 4];
|
||||
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
|
||||
let coeffs_by_2 = coeffs.chunks_exact(2);
|
||||
coeffs = coeffs_by_2.remainder();
|
||||
|
||||
for k in coeffs_by_2 {
|
||||
let coeff0_i64x2 = i64x2_splat(k[0] as i64);
|
||||
let coeff1_i64x2 = i64x2_splat(k[1] as i64);
|
||||
|
||||
for i in 0..4 {
|
||||
let source = wasm32_utils::load_v128(src_rows[i], x);
|
||||
let mut sum = rg_sum[i];
|
||||
let rg_i64x2 = i8x16_swizzle(source, rg0_shuffle);
|
||||
sum = i64x2_add(sum, i64x2_mul(rg_i64x2, coeff0_i64x2));
|
||||
let rg_i64x2 = i8x16_swizzle(source, rg1_shuffle);
|
||||
sum = i64x2_add(sum, i64x2_mul(rg_i64x2, coeff1_i64x2));
|
||||
rg_sum[i] = sum;
|
||||
|
||||
let mut sum = ba_sum[i];
|
||||
let ba_i64x2 = i8x16_swizzle(source, ba0_shuffle);
|
||||
sum = i64x2_add(sum, i64x2_mul(ba_i64x2, coeff0_i64x2));
|
||||
let ba_i64x2 = i8x16_swizzle(source, ba1_shuffle);
|
||||
sum = i64x2_add(sum, i64x2_mul(ba_i64x2, coeff1_i64x2));
|
||||
ba_sum[i] = sum;
|
||||
}
|
||||
x += 2;
|
||||
}
|
||||
|
||||
if let Some(&k) = coeffs.first() {
|
||||
let coeff0_i64x2 = i64x2_splat(k as i64);
|
||||
for i in 0..4 {
|
||||
let source = wasm32_utils::loadl_i64(src_rows[i], x);
|
||||
let rg_i64x2 = i8x16_swizzle(source, rg0_shuffle);
|
||||
rg_sum[i] = i64x2_add(rg_sum[i], i64x2_mul(rg_i64x2, coeff0_i64x2));
|
||||
let ba_i64x2 = i8x16_swizzle(source, ba0_shuffle);
|
||||
ba_sum[i] = i64x2_add(ba_sum[i], i64x2_mul(ba_i64x2, coeff0_i64x2));
|
||||
}
|
||||
}
|
||||
|
||||
for i in 0..4 {
|
||||
v128_store((&mut rg_buf).as_mut_ptr() as *mut v128, rg_sum[i]);
|
||||
v128_store((&mut ba_buf).as_mut_ptr() as *mut v128, ba_sum[i]);
|
||||
let dst_pixel = dst_rows[i].get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = [
|
||||
normalizer.clip(rg_buf[0]),
|
||||
normalizer.clip(rg_buf[1]),
|
||||
normalizer.clip(ba_buf[0]),
|
||||
normalizer.clip(ba_buf[1]),
|
||||
];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// For safety, it is necessary to ensure the following conditions:
|
||||
/// - bounds.len() == dst_row.len()
|
||||
/// - coeffs.len() == dst_rows.0.len() * window_size
|
||||
/// - max(bound.start + bound.size for bound in bounds) <= src_row.len()
|
||||
/// - precision <= MAX_COEFS_PRECISION
|
||||
#[inline]
|
||||
unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16x4],
|
||||
dst_row: &mut [U16x4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut rg_buf = [0i64; 2];
|
||||
let mut ba_buf = [0i64; 2];
|
||||
|
||||
/*
|
||||
|R0 G0 B0 A0 | |R1 G1 B1 A1 |
|
||||
|0001 0203 0405 0607| |0809 1011 1213 1415|
|
||||
|
||||
Shuffle to extract R0 and G0 as i64:
|
||||
0, 1, -1, -1, -1, -1, -1, -1, 2, 3, -1, -1, -1, -1, -1, -1
|
||||
|
||||
Shuffle to extract R1 and G1 as i64:
|
||||
8, 9, -1, -1, -1, -1, -1, -1, 10, 11, -1, -1, -1, -1, -1, -1
|
||||
|
||||
Shuffle to extract B0 and A0 as i64:
|
||||
4, 5, -1, -1, -1, -1, -1, -1, 6, 7, -1, -1, -1, -1, -1, -1
|
||||
|
||||
Shuffle to extract B1 and A1 as i64:
|
||||
12, 13, -1, -1, -1, -1, -1, -1, 14, 15, -1, -1, -1, -1, -1, -1
|
||||
*/
|
||||
|
||||
let rg0_shuffle = i8x16(0, 1, -1, -1, -1, -1, -1, -1, 2, 3, -1, -1, -1, -1, -1, -1);
|
||||
let rg1_shuffle = i8x16(8, 9, -1, -1, -1, -1, -1, -1, 10, 11, -1, -1, -1, -1, -1, -1);
|
||||
let ba0_shuffle = i8x16(4, 5, -1, -1, -1, -1, -1, -1, 6, 7, -1, -1, -1, -1, -1, -1);
|
||||
let ba1_shuffle = i8x16(
|
||||
12, 13, -1, -1, -1, -1, -1, -1, 14, 15, -1, -1, -1, -1, -1, -1,
|
||||
);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x: usize = coeffs_chunk.start as usize;
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
let mut rg_sum = i64x2_splat(half_error);
|
||||
let mut ba_sum = i64x2_splat(half_error);
|
||||
|
||||
let coeffs_by_2 = coeffs.chunks_exact(2);
|
||||
coeffs = coeffs_by_2.remainder();
|
||||
|
||||
for k in coeffs_by_2 {
|
||||
let coeff0_i64x2 = i64x2_splat(k[0] as i64);
|
||||
let coeff1_i64x2 = i64x2_splat(k[1] as i64);
|
||||
|
||||
let source = wasm32_utils::load_v128(src_row, x);
|
||||
|
||||
let rg_i64x2 = i8x16_swizzle(source, rg0_shuffle);
|
||||
rg_sum = i64x2_add(rg_sum, i64x2_mul(rg_i64x2, coeff0_i64x2));
|
||||
let rg_i64x2 = i8x16_swizzle(source, rg1_shuffle);
|
||||
rg_sum = i64x2_add(rg_sum, i64x2_mul(rg_i64x2, coeff1_i64x2));
|
||||
|
||||
let ba_i64x2 = i8x16_swizzle(source, ba0_shuffle);
|
||||
ba_sum = i64x2_add(ba_sum, i64x2_mul(ba_i64x2, coeff0_i64x2));
|
||||
let ba_i64x2 = i8x16_swizzle(source, ba1_shuffle);
|
||||
ba_sum = i64x2_add(ba_sum, i64x2_mul(ba_i64x2, coeff1_i64x2));
|
||||
|
||||
x += 2;
|
||||
}
|
||||
|
||||
if let Some(&k) = coeffs.first() {
|
||||
let coeff0_i64x2 = i64x2_splat(k as i64);
|
||||
let source = wasm32_utils::loadl_i64(src_row, x);
|
||||
let rg_i64x2 = i8x16_swizzle(source, rg0_shuffle);
|
||||
rg_sum = i64x2_add(rg_sum, i64x2_mul(rg_i64x2, coeff0_i64x2));
|
||||
let ba_i64x2 = i8x16_swizzle(source, ba0_shuffle);
|
||||
ba_sum = i64x2_add(ba_sum, i64x2_mul(ba_i64x2, coeff0_i64x2));
|
||||
}
|
||||
|
||||
v128_store((&mut rg_buf).as_mut_ptr() as *mut v128, rg_sum);
|
||||
v128_store((&mut ba_buf).as_mut_ptr() as *mut v128, ba_sum);
|
||||
let dst_pixel = dst_row.get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = [
|
||||
normalizer.clip(rg_buf[0]),
|
||||
normalizer.clip(rg_buf[1]),
|
||||
normalizer.clip(ba_buf[0]),
|
||||
normalizer.clip(ba_buf[1]),
|
||||
];
|
||||
}
|
||||
}
|
||||
@@ -654,6 +654,10 @@ fn upscale_u16() {
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Neon);
|
||||
}
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Wasm32);
|
||||
}
|
||||
for cpu_extensions in cpu_extensions_vec {
|
||||
P::upscale_test(
|
||||
ResizeAlg::Convolution(FilterType::Lanczos3),
|
||||
@@ -682,6 +686,10 @@ fn downscale_u16x2() {
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Neon);
|
||||
}
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Wasm32);
|
||||
}
|
||||
for cpu_extensions in cpu_extensions_vec {
|
||||
P::downscale_test(
|
||||
ResizeAlg::Convolution(FilterType::Lanczos3),
|
||||
@@ -710,6 +718,10 @@ fn upscale_u16x2() {
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Neon);
|
||||
}
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Wasm32);
|
||||
}
|
||||
for cpu_extensions in cpu_extensions_vec {
|
||||
P::upscale_test(
|
||||
ResizeAlg::Convolution(FilterType::Lanczos3),
|
||||
@@ -738,6 +750,10 @@ fn downscale_u16x3() {
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Neon);
|
||||
}
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Wasm32);
|
||||
}
|
||||
for cpu_extensions in cpu_extensions_vec {
|
||||
P::downscale_test(
|
||||
ResizeAlg::Convolution(FilterType::Lanczos3),
|
||||
@@ -766,6 +782,10 @@ fn upscale_u16x3() {
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Neon);
|
||||
}
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Wasm32);
|
||||
}
|
||||
for cpu_extensions in cpu_extensions_vec {
|
||||
P::upscale_test(
|
||||
ResizeAlg::Convolution(FilterType::Lanczos3),
|
||||
@@ -794,6 +814,10 @@ fn downscale_u16x4() {
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Neon);
|
||||
}
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Wasm32);
|
||||
}
|
||||
for cpu_extensions in cpu_extensions_vec {
|
||||
P::downscale_test(
|
||||
ResizeAlg::Convolution(FilterType::Lanczos3),
|
||||
@@ -822,6 +846,10 @@ fn upscale_u16x4() {
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Neon);
|
||||
}
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
{
|
||||
cpu_extensions_vec.push(CpuExtensions::Wasm32);
|
||||
}
|
||||
for cpu_extensions in cpu_extensions_vec {
|
||||
P::upscale_test(
|
||||
ResizeAlg::Convolution(FilterType::Lanczos3),
|
||||
|
||||
Reference in New Issue
Block a user