v4: Fixed dividing image by alpha channel for images with U16x2 pixels.

This commit is contained in:
Kirill Kuzminykh
2025-05-16 12:25:01 +03:00
parent a40d5de798
commit 6680de23c4
12 changed files with 504 additions and 350 deletions
+6
View File
@@ -1,3 +1,9 @@
## [Unreleased] - ReleaseDate
### Fixed
- Fixed dividing image by alpha channel for images with `U16x2` pixels.
## [4.2.2] - 2025-04-06
## Fixed
Generated
+449 -307
View File
File diff suppressed because it is too large Load Diff
+7 -7
View File
@@ -23,10 +23,10 @@ exclude = ["/data"]
cfg-if = "1.0"
num-traits = "0.2.19"
thiserror = "1.0"
document-features = "0.2.10"
document-features = "0.2.11"
# Optional dependencies
image = { version = "0.25.1", optional = true, default-features = false }
bytemuck = { version = "1.16", optional = true }
image = { version = "0.25.6", optional = true, default-features = false }
bytemuck = { version = "1.23", optional = true }
[features]
## Enable this feature to implement traits [IntoImageView](crate::IntoImageView) and
@@ -40,20 +40,20 @@ only_u8x4 = ["testing/only_u8x4"] # This can be used to experiment with the cra
[dev-dependencies]
fast_image_resize = { path = ".", features = ["for_testing"] }
resize = { version = "0.8.4", default-features = false, features = ["std"] }
rgb = "0.8.45"
resize = { version = "0.8.8", default-features = false, features = ["std"] }
rgb = "0.8.50"
png = "0.17.13"
serde = { version = "1.0", features = ["serde_derive"] }
serde_json = "1.0"
walkdir = "2.5"
itertools = "0.13.0"
itertools = "0.14.0"
criterion = { version = "0.5.1", default-features = false, features = ["cargo_bench_support"] }
tera = "1.20"
testing = { path = "testing" }
[target.'cfg(not(target_arch = "wasm32"))'.dev-dependencies]
nix = { version = "0.29.0", default-features = false, features = ["sched"] }
nix = { version = "0.30.1", default-features = false, features = ["sched"] }
[target.'cfg(all(not(target_arch = "wasm32"), not(target_os = "windows")))'.dev-dependencies]
+2 -2
View File
@@ -137,7 +137,7 @@ Otherwise, you have to convert such images into supported by the crate image typ
use std::io::BufWriter;
use image::codecs::png::PngEncoder;
use image::io::Reader as ImageReader;
use image::ImageReader;
use image::{ExtendedColorType, ImageEncoder};
use fast_image_resize::{IntoImageView, Resizer};
@@ -181,7 +181,7 @@ fn main() {
```rust
use image::codecs::png::PngEncoder;
use image::io::Reader as ImageReader;
use image::ImageReader;
use image::{ColorType, GenericImageView};
use fast_image_resize::{IntoImageView, Resizer, ResizeOptions};
+3 -5
View File
@@ -3,14 +3,12 @@ use std::path::PathBuf;
use anyhow::{anyhow, Context, Result};
use clap::Parser;
use image::io::Reader as ImageReader;
use image::ColorType;
use log::debug;
use once_cell::sync::Lazy;
use fast_image_resize as fr;
use fast_image_resize::images::Image;
use fast_image_resize::ResizeOptions;
use image::{ColorType, ImageReader};
use log::debug;
use once_cell::sync::Lazy;
mod structs;
+4
View File
@@ -0,0 +1,4 @@
unstable_features = true
imports_granularity = "Module"
group_imports = "StdExternalCrate"
+4 -3
View File
@@ -1,11 +1,10 @@
use std::arch::x86_64::*;
use super::sse4;
use crate::pixels::U16x2;
use crate::utils::foreach_with_pre_reading;
use crate::{ImageView, ImageViewMut};
use super::sse4;
#[target_feature(enable = "avx2")]
pub(crate) unsafe fn multiply_alpha(
src_view: &impl ImageView<Pixel = U16x2>,
@@ -220,7 +219,9 @@ unsafe fn divide_alpha_8_pixels(pixels: __m256i) -> __m256i {
let luma_f32x8 = _mm256_cvtepi32_ps(luma_i32x8);
let scaled_luma_f32x8 = _mm256_mul_ps(luma_f32x8, alpha_max);
let divided_luma_f32x8 = _mm256_div_ps(scaled_luma_f32x8, alpha_f32x8);
let divided_luma_i32x8 = _mm256_cvtps_epi32(divided_luma_f32x8);
let mut divided_luma_i32x8 = _mm256_cvtps_epi32(divided_luma_f32x8);
// Clamp result to [0..0xffff]
divided_luma_i32x8 = _mm256_min_epi32(divided_luma_i32x8, luma_mask);
let alpha = _mm256_and_si256(pixels, alpha_mask);
_mm256_blendv_epi8(divided_luma_i32x8, alpha, alpha_mask)
+4 -3
View File
@@ -1,11 +1,10 @@
use std::arch::x86_64::*;
use super::native;
use crate::pixels::U16x2;
use crate::utils::foreach_with_pre_reading;
use crate::{ImageView, ImageViewMut};
use super::native;
#[target_feature(enable = "sse4.1")]
pub(crate) unsafe fn multiply_alpha(
src_view: &impl ImageView<Pixel = U16x2>,
@@ -210,7 +209,9 @@ unsafe fn divide_alpha_4_pixels(pixels: __m128i) -> __m128i {
let alpha_f32x4 = _mm_cvtepi32_ps(_mm_shuffle_epi8(pixels, alpha32_sh));
let luma_f32x4 = _mm_cvtepi32_ps(_mm_and_si128(pixels, luma_mask));
let scaled_luma_f32x4 = _mm_mul_ps(luma_f32x4, alpha_max);
let divided_luma_i32x4 = _mm_cvtps_epi32(_mm_div_ps(scaled_luma_f32x4, alpha_f32x4));
let mut divided_luma_i32x4 = _mm_cvtps_epi32(_mm_div_ps(scaled_luma_f32x4, alpha_f32x4));
// Clamp result to [0..0xffff]
divided_luma_i32x4 = _mm_min_epi32(divided_luma_i32x4, luma_mask);
let alpha = _mm_and_si128(pixels, alpha_mask);
_mm_blendv_epi8(divided_luma_i32x4, alpha, alpha_mask)
+2 -2
View File
@@ -1,11 +1,10 @@
use std::arch::x86_64::*;
use super::sse4;
use crate::pixels::U16x4;
use crate::utils::foreach_with_pre_reading;
use crate::{ImageView, ImageViewMut};
use super::sse4;
#[target_feature(enable = "avx2")]
pub(crate) unsafe fn multiply_alpha(
src_view: &impl ImageView<Pixel = U16x4>,
@@ -238,6 +237,7 @@ unsafe fn divide_alpha_4_pixels(pixels: __m256i) -> __m256i {
let divided_pix0_i32x8 = _mm256_cvtps_epi32(_mm256_div_ps(scaled_pix0_f32x8, alpha0_f32x8));
let divided_pix1_i32x8 = _mm256_cvtps_epi32(_mm256_div_ps(scaled_pix1_f32x8, alpha1_f32x8));
// All negative values will be stored as 0.
let two_pixels_i16x16 = _mm256_packus_epi32(divided_pix0_i32x8, divided_pix1_i32x8);
let alpha = _mm256_and_si256(pixels, alpha_mask);
_mm256_blendv_epi8(two_pixels_i16x16, alpha, alpha_mask)
+2 -3
View File
@@ -1,11 +1,10 @@
use std::arch::x86_64::*;
use super::native;
use crate::pixels::U8x2;
use crate::utils::foreach_with_pre_reading;
use crate::{ImageView, ImageViewMut};
use super::native;
#[target_feature(enable = "sse4.1")]
pub(crate) unsafe fn multiply_alpha(
src_view: &impl ImageView<Pixel = U8x2>,
@@ -212,7 +211,7 @@ unsafe fn divide_alpha_8_pixels(pixels: __m128i) -> __m128i {
let scaled_alpha_lo_i32 = _mm_cvtps_epi32(_mm_div_ps(alpha_scale, alpha_lo_f32));
let alpha_hi_f32 = _mm_cvtepi32_ps(_mm_shuffle_epi8(pixels, alpha32_sh_hi));
let scaled_alpha_hi_i32 = _mm_cvtps_epi32(_mm_div_ps(alpha_scale, alpha_hi_f32));
// All negative values will stored as 0.
// All negative values will be stored as 0.
let scaled_alpha_i16 = _mm_packus_epi32(scaled_alpha_lo_i32, scaled_alpha_hi_i32);
let luma_i16 = _mm_and_si128(pixels, luma_mask);
+3
View File
@@ -329,6 +329,7 @@ mod u16_tests {
new_case_16(0xffff, 0, 0),
new_case_16(0x8000, 0, 0),
new_case_16(0, 0, 0),
new_case_16(0xffff, 0xc0c0, 0xffff),
];
let mut scr_pixels = vec![];
let mut expected_pixels = vec![];
@@ -455,6 +456,8 @@ mod f32_tests {
new_case_f32(1., 0., 0.),
new_case_f32(0.5, 0., 0.),
new_case_f32(0., 0., 0.),
// f32 can afford to have a value greater than 1.0
new_case_f32(1., 0.7, 1. / 0.7),
];
let mut scr_pixels = vec![];
let mut expected_pixels = vec![];
+18 -18
View File
@@ -94,7 +94,7 @@ fn resize_to_same_size_after_cropping() {
}
/// In this test, we check that resizer won't use horizontal convolution
/// if width of destination image is equal to width of cropped source image.
/// if the width of destination image is equal to the width of cropped source image.
fn resize_to_same_width<const C: usize>(
pixel_type: PixelType,
cpu_extensions: CpuExtensions,
@@ -127,6 +127,13 @@ fn resize_to_same_width<const C: usize>(
)
.unwrap();
assert!(fr_testing::logs_contain(
"compute vertical convolution coefficients"
));
assert!(!fr_testing::logs_contain(
"compute horizontal convolution coefficients"
));
let expected_result: Vec<u8> = (0..8000u32)
.flat_map(|v| create_pixel((10 + v % 100) as u8))
.collect();
@@ -137,17 +144,10 @@ fn resize_to_same_width<const C: usize>(
pixel_type,
cpu_extensions
);
assert!(fr_testing::logs_contain(
"compute vertical convolution coefficients"
));
assert!(!fr_testing::logs_contain(
"compute horizontal convolution coefficients"
));
}
/// In this test, we check that resizer won't use vertical convolution
/// if height of destination image is equal to height of cropped source image.
/// if the height of destination image is equal to the height of cropped source image.
fn resize_to_same_height<const C: usize>(
pixel_type: PixelType,
cpu_extensions: CpuExtensions,
@@ -163,8 +163,8 @@ fn resize_to_same_height<const C: usize>(
.flat_map(|v| create_pixel((v / 120) as u8))
.collect();
let src_image = Image::from_vec_u8(src_width, src_height, buffer, pixel_type).unwrap();
let mut dst_image = Image::new(width, height, pixel_type);
let mut resizer = Resizer::new();
unsafe {
resizer.set_cpu_extensions(cpu_extensions);
@@ -179,6 +179,13 @@ fn resize_to_same_height<const C: usize>(
)
.unwrap();
assert!(!fr_testing::logs_contain(
"compute vertical convolution coefficients"
));
assert!(fr_testing::logs_contain(
"compute horizontal convolution coefficients"
));
let expected_result: Vec<u8> = (0..8000u32)
.flat_map(|v| create_pixel((10 + v / 100) as u8))
.collect();
@@ -189,17 +196,10 @@ fn resize_to_same_height<const C: usize>(
pixel_type,
cpu_extensions
);
assert!(!fr_testing::logs_contain(
"compute vertical convolution coefficients"
));
assert!(fr_testing::logs_contain(
"compute horizontal convolution coefficients"
));
}
#[test]
fn resize_to_same_width_after_cropping() {
fn resize_to_same_width_or_height_after_cropping() {
let mut cpu_extensions_vec = vec![CpuExtensions::None];
#[cfg(target_arch = "x86_64")]
{