mirror of
https://github.com/Cykooz/fast_image_resize.git
synced 2026-10-08 01:11:09 +00:00
Merge branch 'main' into dev
# Conflicts: # src/alpha/u16x2/avx2.rs # src/alpha/u16x2/sse4.rs
This commit is contained in:
@@ -66,6 +66,15 @@
|
||||
- Optimized convolution algorythm by deleting zero coefficients from start and
|
||||
end of bounds.
|
||||
|
||||
## [Unreleased] - ReleaseDate
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed `SSE4.1` and `AVX2` implementation fo dividing image by
|
||||
alpha channel for images with `U16x2` pixels.
|
||||
- Fixed `NEON` implementation for dividing image by
|
||||
alpha channel for images with `U16x2` and `U16x4` pixels.
|
||||
|
||||
## [4.2.2] - 2025-04-06
|
||||
|
||||
## Fixed
|
||||
|
||||
@@ -221,7 +221,6 @@ unsafe fn divide_alpha_8_pixels(pixels: __m256i) -> __m256i {
|
||||
let divided_luma_f32x8 = _mm256_div_ps(scaled_luma_f32x8, alpha_f32x8);
|
||||
let mut divided_luma_i32x8 = _mm256_cvtps_epi32(divided_luma_f32x8);
|
||||
// Clamp result to [0..0xffff]
|
||||
divided_luma_i32x8 = _mm256_max_epi32(divided_luma_i32x8, _mm256_setzero_si256());
|
||||
divided_luma_i32x8 = _mm256_min_epi32(divided_luma_i32x8, luma_mask);
|
||||
|
||||
let alpha = _mm256_and_si256(pixels, alpha_mask);
|
||||
|
||||
@@ -211,7 +211,6 @@ unsafe fn divide_alpha_4_pixels(pixels: __m128i) -> __m128i {
|
||||
let scaled_luma_f32x4 = _mm_mul_ps(luma_f32x4, alpha_max);
|
||||
let mut divided_luma_i32x4 = _mm_cvtps_epi32(_mm_div_ps(scaled_luma_f32x4, alpha_f32x4));
|
||||
// Clamp result to [0..0xffff]
|
||||
divided_luma_i32x4 = _mm_max_epi32(divided_luma_i32x4, _mm_setzero_si128());
|
||||
divided_luma_i32x4 = _mm_min_epi32(divided_luma_i32x4, luma_mask);
|
||||
|
||||
let alpha = _mm_and_si128(pixels, alpha_mask);
|
||||
|
||||
Reference in New Issue
Block a user