Merge branch 'v4'

# Conflicts:
#	CHANGELOG.md
#	Cargo.lock
#	Cargo.toml
#	README.md
#	resizer/src/main.rs
This commit is contained in:
Kirill Kuzminykh
2025-05-16 12:27:35 +03:00
7 changed files with 29 additions and 13 deletions
+9
View File
@@ -57,6 +57,15 @@
- Optimized convolution algorythm by deleting zero coefficients from start and
end of bounds.
## [Unreleased] - ReleaseDate
### Fixed
- Fixed `SSE4.1` and `AVX2` implementation fo dividing image by
alpha channel for images with `U16x2` pixels.
- Fixed `NEON` implementation for dividing image by
alpha channel for images with `U16x2` and `U16x4` pixels.
## [4.2.2] - 2025-04-06
## Fixed
+4 -3
View File
@@ -1,11 +1,10 @@
use std::arch::x86_64::*;
use super::sse4;
use crate::pixels::U16x2;
use crate::utils::foreach_with_pre_reading;
use crate::{ImageView, ImageViewMut};
use super::sse4;
#[target_feature(enable = "avx2")]
pub(crate) unsafe fn multiply_alpha(
src_view: &impl ImageView<Pixel = U16x2>,
@@ -220,7 +219,9 @@ unsafe fn divide_alpha_8_pixels(pixels: __m256i) -> __m256i {
let luma_f32x8 = _mm256_cvtepi32_ps(luma_i32x8);
let scaled_luma_f32x8 = _mm256_mul_ps(luma_f32x8, alpha_max);
let divided_luma_f32x8 = _mm256_div_ps(scaled_luma_f32x8, alpha_f32x8);
let divided_luma_i32x8 = _mm256_cvtps_epi32(divided_luma_f32x8);
let mut divided_luma_i32x8 = _mm256_cvtps_epi32(divided_luma_f32x8);
// Clamp result to [0..0xffff]
divided_luma_i32x8 = _mm256_min_epi32(divided_luma_i32x8, luma_mask);
let alpha = _mm256_and_si256(pixels, alpha_mask);
_mm256_blendv_epi8(divided_luma_i32x8, alpha, alpha_mask)
+4 -3
View File
@@ -1,11 +1,10 @@
use std::arch::x86_64::*;
use super::native;
use crate::pixels::U16x2;
use crate::utils::foreach_with_pre_reading;
use crate::{ImageView, ImageViewMut};
use super::native;
#[target_feature(enable = "sse4.1")]
pub(crate) unsafe fn multiply_alpha(
src_view: &impl ImageView<Pixel = U16x2>,
@@ -210,7 +209,9 @@ unsafe fn divide_alpha_4_pixels(pixels: __m128i) -> __m128i {
let alpha_f32x4 = _mm_cvtepi32_ps(_mm_shuffle_epi8(pixels, alpha32_sh));
let luma_f32x4 = _mm_cvtepi32_ps(_mm_and_si128(pixels, luma_mask));
let scaled_luma_f32x4 = _mm_mul_ps(luma_f32x4, alpha_max);
let divided_luma_i32x4 = _mm_cvtps_epi32(_mm_div_ps(scaled_luma_f32x4, alpha_f32x4));
let mut divided_luma_i32x4 = _mm_cvtps_epi32(_mm_div_ps(scaled_luma_f32x4, alpha_f32x4));
// Clamp result to [0..0xffff]
divided_luma_i32x4 = _mm_min_epi32(divided_luma_i32x4, luma_mask);
let alpha = _mm_and_si128(pixels, alpha_mask);
_mm_blendv_epi8(divided_luma_i32x4, alpha, alpha_mask)
+2 -2
View File
@@ -1,11 +1,10 @@
use std::arch::x86_64::*;
use super::sse4;
use crate::pixels::U16x4;
use crate::utils::foreach_with_pre_reading;
use crate::{ImageView, ImageViewMut};
use super::sse4;
#[target_feature(enable = "avx2")]
pub(crate) unsafe fn multiply_alpha(
src_view: &impl ImageView<Pixel = U16x4>,
@@ -238,6 +237,7 @@ unsafe fn divide_alpha_4_pixels(pixels: __m256i) -> __m256i {
let divided_pix0_i32x8 = _mm256_cvtps_epi32(_mm256_div_ps(scaled_pix0_f32x8, alpha0_f32x8));
let divided_pix1_i32x8 = _mm256_cvtps_epi32(_mm256_div_ps(scaled_pix1_f32x8, alpha1_f32x8));
// All negative values will be stored as 0.
let two_pixels_i16x16 = _mm256_packus_epi32(divided_pix0_i32x8, divided_pix1_i32x8);
let alpha = _mm256_and_si256(pixels, alpha_mask);
_mm256_blendv_epi8(two_pixels_i16x16, alpha, alpha_mask)
+2 -3
View File
@@ -1,11 +1,10 @@
use std::arch::x86_64::*;
use super::native;
use crate::pixels::U8x2;
use crate::utils::foreach_with_pre_reading;
use crate::{ImageView, ImageViewMut};
use super::native;
#[target_feature(enable = "sse4.1")]
pub(crate) unsafe fn multiply_alpha(
src_view: &impl ImageView<Pixel = U8x2>,
@@ -212,7 +211,7 @@ unsafe fn divide_alpha_8_pixels(pixels: __m128i) -> __m128i {
let scaled_alpha_lo_i32 = _mm_cvtps_epi32(_mm_div_ps(alpha_scale, alpha_lo_f32));
let alpha_hi_f32 = _mm_cvtepi32_ps(_mm_shuffle_epi8(pixels, alpha32_sh_hi));
let scaled_alpha_hi_i32 = _mm_cvtps_epi32(_mm_div_ps(alpha_scale, alpha_hi_f32));
// All negative values will stored as 0.
// All negative values will be stored as 0.
let scaled_alpha_i16 = _mm_packus_epi32(scaled_alpha_lo_i32, scaled_alpha_hi_i32);
let luma_i16 = _mm_and_si128(pixels, luma_mask);
+5 -2
View File
@@ -400,11 +400,14 @@ pub unsafe fn mul_color_recip_alpha_u16x8(
recip_alpha_hi: float32x4_t,
zero: uint16x8_t,
) -> uint16x8_t {
let max_value = vdupq_n_u32(0xffff);
let color_lo_f32 = vcvtq_f32_u32(vreinterpretq_u32_u16(vzip1q_u16(color, zero)));
let res_lo_u32 = vcvtaq_u32_f32(vmulq_f32(color_lo_f32, recip_alpha_lo));
let mut res_lo_u32 = vcvtaq_u32_f32(vmulq_f32(color_lo_f32, recip_alpha_lo));
res_lo_u32 = vminq_u32(res_lo_u32, max_value);
let color_hi_f32 = vcvtq_f32_u32(vreinterpretq_u32_u16(vzip2q_u16(color, zero)));
let res_hi_u32 = vcvtaq_u32_f32(vmulq_f32(color_hi_f32, recip_alpha_hi));
let mut res_hi_u32 = vcvtaq_u32_f32(vmulq_f32(color_hi_f32, recip_alpha_hi));
res_hi_u32 = vminq_u32(res_hi_u32, max_value);
vcombine_u16(vmovn_u32(res_lo_u32), vmovn_u32(res_hi_u32))
}
+3
View File
@@ -330,6 +330,7 @@ mod u16_tests {
new_case_16(0xffff, 0, 0),
new_case_16(0x8000, 0, 0),
new_case_16(0, 0, 0),
new_case_16(0xffff, 0xc0c0, 0xffff),
];
let mut scr_pixels = vec![];
let mut expected_pixels = vec![];
@@ -456,6 +457,8 @@ mod f32_tests {
new_case_f32(1., 0., 0.),
new_case_f32(0.5, 0., 0.),
new_case_f32(0., 0., 0.),
// f32 can afford to have a value greater than 1.0
new_case_f32(1., 0.7, 1. / 0.7),
];
let mut scr_pixels = vec![];
let mut expected_pixels = vec![];