Constantize v128 declarations.

Slight increase in performance by avoiding checks that values are within
bounds.
This commit is contained in:
Colin Murphy
2023-01-23 16:51:16 -05:00
parent 37a96fd79d
commit 74fd01a267
5 changed files with 51 additions and 37 deletions
+27 -19
View File
@@ -73,11 +73,14 @@ pub(crate) unsafe fn multiply_alpha_row_inplace(row: &mut [U16x2]) {
#[inline]
unsafe fn multiplies_alpha_4_pixels(pixels: v128) -> v128 {
let zero = i64x2_splat(0);
let half = i32x4_splat(0x8000);
const HALF: v128 = i32x4(0x8000, 0x8000, 0x8000, 0x8000);
const MAX_A: i32 = 0xffff0000u32 as i32;
let max_alpha = i32x4_splat(MAX_A);
const MAX_ALPHA: v128 = i32x4(
0xffff0000u32 as i32,
0xffff0000u32 as i32,
0xffff0000u32 as i32,
0xffff0000u32 as i32,
);
/*
|L0 A0 | |L1 A1 | |L2 A2 | |L3 A3 |
|0001 0203| |0405 0607| |0809 1011| |1213 1415|
@@ -85,17 +88,17 @@ unsafe fn multiplies_alpha_4_pixels(pixels: v128) -> v128 {
const FACTOR_MASK: v128 = i8x16(2, 3, 2, 3, 6, 7, 6, 7, 10, 11, 10, 11, 14, 15, 14, 15);
let factor_pixels = u8x16_swizzle(pixels, FACTOR_MASK);
let factor_pixels = v128_or(factor_pixels, max_alpha);
let factor_pixels = v128_or(factor_pixels, MAX_ALPHA);
let src_i32_lo = i16x8_shuffle::<0, 8, 1, 9, 2, 10, 3, 11>(pixels, zero);
let factors = i16x8_shuffle::<0, 8, 1, 9, 2, 10, 3, 11>(factor_pixels, zero);
let src_i32_lo = i32x4_add(i32x4_mul(src_i32_lo, factors), half);
let src_u32_lo = u32x4_extend_low_u16x8(pixels);
let factors = u32x4_extend_low_u16x8(factor_pixels);
let src_i32_lo = i32x4_add(i32x4_mul(src_u32_lo, factors), HALF);
let dst_i32_lo = i32x4_add(src_i32_lo, u32x4_shr(src_i32_lo, 16));
let dst_i32_lo = u32x4_shr(dst_i32_lo, 16);
let src_i32_hi = i16x8_shuffle::<4, 12, 5, 13, 6, 14, 7, 15>(pixels, zero);
let factors = i16x8_shuffle::<4, 12, 5, 13, 6, 14, 7, 15>(factor_pixels, zero);
let src_i32_hi = i32x4_add(i32x4_mul(src_i32_hi, factors), half);
let src_u32_hi = u32x4_extend_high_u16x8(pixels);
let factors = u32x4_extend_high_u16x8(factor_pixels);
let src_i32_hi = i32x4_add(i32x4_mul(src_u32_hi, factors), HALF);
let dst_i32_hi = i32x4_add(src_i32_hi, u32x4_shr(src_i32_hi, 16));
let dst_i32_hi = u32x4_shr(dst_i32_hi, 16);
@@ -188,10 +191,15 @@ pub(crate) unsafe fn divide_alpha_row_inplace(row: &mut [U16x2]) {
#[inline]
unsafe fn divide_alpha_4_pixels(pixels: v128) -> v128 {
let alpha_mask = i32x4_splat(0xffff0000u32 as i32);
let luma_mask = i32x4_splat(0xffff);
let alpha_max = f32x4_splat(65535.0);
let alpha_scale_max = f32x4_splat(2147483648f32);
const ALPHA_MASK: v128 = i32x4(
0xffff0000u32 as i32,
0xffff0000u32 as i32,
0xffff0000u32 as i32,
0xffff0000u32 as i32,
);
const LUMA_MASK: v128 = i32x4(0xffff, 0xffff, 0xffff, 0xffff);
const ALPHA_MAX: v128 = f32x4(65535.0, 65535.0, 65535.0, 65535.0);
const ALPHA_SCALE_MAX: v128 = f32x4(2147483648f32, 2147483648f32, 2147483648f32, 2147483648f32);
/*
|L0 A0 | |L1 A1 | |L2 A2 | |L3 A3 |
|0001 0203| |0405 0607| |0809 1011| |1213 1415|
@@ -199,14 +207,14 @@ unsafe fn divide_alpha_4_pixels(pixels: v128) -> v128 {
const ALPHA32_SH: v128 = i8x16(2, 3, -1, -1, 6, 7, -1, -1, 10, 11, -1, -1, 14, 15, -1, -1);
let alpha_f32x4 = f32x4_convert_i32x4(u8x16_swizzle(pixels, ALPHA32_SH));
let luma_f32x4 = f32x4_convert_i32x4(v128_and(pixels, luma_mask));
let scaled_luma_f32x4 = f32x4_mul(luma_f32x4, alpha_max);
let luma_f32x4 = f32x4_convert_i32x4(v128_and(pixels, LUMA_MASK));
let scaled_luma_f32x4 = f32x4_mul(luma_f32x4, ALPHA_MAX);
let divided_luma_u32x4 = u32x4_trunc_sat_f32x4(f32x4_pmin(
f32x4_div(scaled_luma_f32x4, alpha_f32x4),
alpha_scale_max,
ALPHA_SCALE_MAX,
));
let alpha = v128_and(pixels, alpha_mask);
let alpha = v128_and(pixels, ALPHA_MASK);
u8x16_shuffle::<0, 1, 18, 19, 4, 5, 22, 23, 8, 9, 26, 27, 12, 13, 30, 31>(
divided_luma_u32x4,
alpha,
+13 -5
View File
@@ -51,6 +51,7 @@ unsafe fn horiz_convolution_8u4x(
coefficients_chunks: &[CoefficientsI32Chunk],
normalizer: &optimisations::Normalizer32,
) {
const ZERO: v128 = i64x2(0, 0);
let precision = normalizer.precision();
let half_error = 1i64 << (precision - 1);
let mut rg_buf = [0i64; 2];
@@ -79,8 +80,8 @@ unsafe fn horiz_convolution_8u4x(
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let mut x: usize = coeffs_chunk.start as usize;
let mut rg_sum = [i8x16_splat(0); 4];
let mut bb_sum = [i8x16_splat(0); 4];
let mut rg_sum = [ZERO; 4];
let mut bb_sum = [ZERO; 4];
let mut coeffs = coeffs_chunk.values;
let end_x = x + coeffs.len();
@@ -98,13 +99,20 @@ unsafe fn horiz_convolution_8u4x(
let source = wasm32_utils::load_v128(src_rows[i], x);
let rg0_i64x2 = i8x16_swizzle(source, RG0_SHUFFLE);
rg_sum[i] = i64x2_add(rg_sum[i], wasm32_utils::i64x2_mul_lo(rg0_i64x2, coeff0_i64x2));
rg_sum[i] = i64x2_add(
rg_sum[i],
wasm32_utils::i64x2_mul_lo(rg0_i64x2, coeff0_i64x2),
);
let rg1_i64x2 = i8x16_swizzle(source, RG1_SHUFFLE);
rg_sum[i] = i64x2_add(rg_sum[i], wasm32_utils::i64x2_mul_lo(rg1_i64x2, coeff1_i64x2));
rg_sum[i] = i64x2_add(
rg_sum[i],
wasm32_utils::i64x2_mul_lo(rg1_i64x2, coeff1_i64x2),
);
let bb_i64x2 = i8x16_swizzle(source, BB_SHUFFLE);
bb_sum[i] = i64x2_add(bb_sum[i], wasm32_utils::i64x2_mul_lo(bb_i64x2, coeff_i64x2));
bb_sum[i] =
i64x2_add(bb_sum[i], wasm32_utils::i64x2_mul_lo(bb_i64x2, coeff_i64x2));
}
x += 2;
}
+4 -4
View File
@@ -51,14 +51,14 @@ unsafe fn horiz_convolution_four_rows(
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
normalizer: &optimisations::Normalizer16,
) {
let zero = i64x2_splat(0);
const ZERO: v128 = i64x2(0, 0);
let initial = 1 << (normalizer.precision() - 1);
let mut buf = [0, 0, 0, 0, initial];
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let coeffs = coeffs_chunk.values;
let mut x = coeffs_chunk.start as usize;
let mut result_i32x4 = [zero, zero, zero, zero];
let mut result_i32x4 = [ZERO, ZERO, ZERO, ZERO];
let coeffs_by_8 = coeffs.chunks_exact(8);
let reminder8 = coeffs_by_8.remainder();
@@ -118,14 +118,14 @@ unsafe fn horiz_convolution_row(
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
normalizer: &optimisations::Normalizer16,
) {
let zero = i64x2_splat(0);
const ZERO: v128 = i64x2(0, 0);
let initial = 1 << (normalizer.precision() - 1);
let mut buf = [0, 0, 0, 0, initial];
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
let coeffs = coeffs_chunk.values;
let mut x = coeffs_chunk.start as usize;
let mut result_i32x4 = zero;
let mut result_i32x4 = ZERO;
let coeffs_by_8 = coeffs.chunks_exact(8);
let reminder8 = coeffs_by_8.remainder();
+3 -3
View File
@@ -53,7 +53,7 @@ unsafe fn horiz_convolution_8u4x(
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
precision: u8,
) {
let zero = i64x2_splat(0);
const ZERO: v128 = i64x2(0, 0);
let initial = i32x4_splat(1 << (precision - 1));
let src_width = src_rows[0].len();
@@ -163,8 +163,8 @@ unsafe fn horiz_convolution_8u4x(
constify_imm8!(precision, call);
for i in 0..4 {
let sss = i16x8_narrow_i32x4(sss_a[i], zero);
let pixel: u32 = transmute(i32x4_extract_lane::<0>(u8x16_narrow_i16x8(sss, zero)));
let sss = i16x8_narrow_i32x4(sss_a[i], ZERO);
let pixel: u32 = transmute(i32x4_extract_lane::<0>(u8x16_narrow_i16x8(sss, ZERO)));
let bytes = pixel.to_le_bytes();
dst_rows[i].get_unchecked_mut(dst_x).0 = [bytes[0], bytes[1], bytes[2]];
}
+4 -6
View File
@@ -32,6 +32,7 @@ pub(crate) unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8
coeffs_chunk: optimisations::CoefficientsI16Chunk,
normalizer: &optimisations::Normalizer16,
) {
const ZERO: v128 = i64x2(0, 0);
let y_start = coeffs_chunk.start;
let coeffs = coeffs_chunk.values;
let max_y = y_start + coeffs.len() as u32;
@@ -111,8 +112,7 @@ pub(crate) unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8
let source1 = wasm32_utils::load_v128(components, src_x); // top line
let source = i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(
source1,
i64x2_splat(0),
source1, ZERO,
);
let pix = i16x8_extend_low_u8x16(source);
sss0 = i32x4_add(sss0, i32x4_dot_i16x8(pix, mmk));
@@ -128,8 +128,7 @@ pub(crate) unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8
let source1 = wasm32_utils::load_v128(components, src_x + 16); // top line
let source = i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(
source1,
i64x2_splat(0),
source1, ZERO,
);
let pix = i16x8_extend_low_u8x16(source);
sss4 = i32x4_add(sss4, i32x4_dot_i16x8(pix, mmk));
@@ -206,8 +205,7 @@ pub(crate) unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8
let source1 = wasm32_utils::loadl_i64(components, src_x); // top line
let source = i8x16_shuffle::<0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23>(
source1,
i64x2_splat(0),
source1, ZERO,
);
let pix = i16x8_extend_low_u8x16(source);
sss0 = i32x4_add(sss0, i32x4_dot_i16x8(pix, mmk));