v4: Fixed NEON implementation for dividing image by alpha channel for images with U16x2 and U16x4 pixels.

This commit is contained in:
Kirill Kuzminykh
2025-05-16 12:26:46 +03:00
parent 6680de23c4
commit 6ec8f5babb
2 changed files with 9 additions and 3 deletions
+4 -1
View File
@@ -2,7 +2,10 @@
### Fixed
- Fixed dividing image by alpha channel for images with `U16x2` pixels.
- Fixed `SSE4.1` and `AVX2` implementation fo dividing image by
alpha channel for images with `U16x2` pixels.
- Fixed `NEON` implementation for dividing image by
alpha channel for images with `U16x2` and `U16x4` pixels.
## [4.2.2] - 2025-04-06
+5 -2
View File
@@ -400,11 +400,14 @@ pub unsafe fn mul_color_recip_alpha_u16x8(
recip_alpha_hi: float32x4_t,
zero: uint16x8_t,
) -> uint16x8_t {
let max_value = vdupq_n_u32(0xffff);
let color_lo_f32 = vcvtq_f32_u32(vreinterpretq_u32_u16(vzip1q_u16(color, zero)));
let res_lo_u32 = vcvtaq_u32_f32(vmulq_f32(color_lo_f32, recip_alpha_lo));
let mut res_lo_u32 = vcvtaq_u32_f32(vmulq_f32(color_lo_f32, recip_alpha_lo));
res_lo_u32 = vminq_u32(res_lo_u32, max_value);
let color_hi_f32 = vcvtq_f32_u32(vreinterpretq_u32_u16(vzip2q_u16(color, zero)));
let res_hi_u32 = vcvtaq_u32_f32(vmulq_f32(color_hi_f32, recip_alpha_hi));
let mut res_hi_u32 = vcvtaq_u32_f32(vmulq_f32(color_hi_f32, recip_alpha_hi));
res_hi_u32 = vminq_u32(res_hi_u32, max_value);
vcombine_u16(vmovn_u32(res_lo_u32), vmovn_u32(res_hi_u32))
}