mirror of
https://github.com/Cykooz/fast_image_resize.git
synced 2026-10-08 01:11:09 +00:00
Added support for optimization with help of SSE4.1 and AVX2 to multiple and divide by alpha channel for the F32x4 pixel type (#30).
This commit is contained in:
+3
-3
@@ -24,7 +24,7 @@ cfg-if = "1.0"
|
||||
num-traits = "0.2.19"
|
||||
thiserror = "1.0"
|
||||
document-features = "0.2.8"
|
||||
|
||||
# Optional dependencies
|
||||
image = { version = "0.25.1", optional = true }
|
||||
bytemuck = { version = "1.16", optional = true }
|
||||
|
||||
@@ -34,12 +34,12 @@ bytemuck = { version = "1.16", optional = true }
|
||||
## [DynamicImage](https://docs.rs/image/latest/image/enum.DynamicImage.html)
|
||||
## type from the `image` crate.
|
||||
image = ["dep:image", "dep:bytemuck"]
|
||||
for_test = ["image"]
|
||||
for_testing = ["image"]
|
||||
only_u8x4 = [] # This can be used to experiment with the crate's code.
|
||||
|
||||
|
||||
[dev-dependencies]
|
||||
fast_image_resize = { path = ".", features = ["for_test"] }
|
||||
fast_image_resize = { path = ".", features = ["for_testing"] }
|
||||
resize = "0.8.4"
|
||||
rgb = "0.8.37"
|
||||
png = "0.17.13"
|
||||
|
||||
@@ -27,12 +27,17 @@ fn multiplies_alpha(
|
||||
let width = 4096;
|
||||
let height = 2048;
|
||||
let f32x2_bytes: Vec<u8> = [1.0, 0.5].iter().flat_map(|v| v.to_le_bytes()).collect();
|
||||
let f32x4_bytes: Vec<u8> = [1.0, 0.5, 0., 0.5]
|
||||
.iter()
|
||||
.flat_map(|v| v.to_le_bytes())
|
||||
.collect();
|
||||
let pixel: &[u8] = match pixel_type {
|
||||
PixelType::U8x4 => &[255, 128, 0, 128],
|
||||
PixelType::U8x2 => &[255, 128],
|
||||
PixelType::U16x2 => &[255, 255, 0, 128],
|
||||
PixelType::U16x4 => &[0, 255, 0, 128, 0, 0, 0, 128],
|
||||
PixelType::F32x2 => &f32x2_bytes,
|
||||
PixelType::F32x4 => &f32x4_bytes,
|
||||
_ => unreachable!(),
|
||||
};
|
||||
let src_data = get_src_image(width, height, pixel_type, pixel);
|
||||
@@ -80,12 +85,17 @@ fn divides_alpha(
|
||||
let width = 4095;
|
||||
let height = 2048;
|
||||
let f32x2_bytes: Vec<u8> = [0.5, 0.5].iter().flat_map(|v| v.to_le_bytes()).collect();
|
||||
let f32x4_bytes: Vec<u8> = [0.5, 0.25, 0., 0.5]
|
||||
.iter()
|
||||
.flat_map(|v| v.to_le_bytes())
|
||||
.collect();
|
||||
let pixel: &[u8] = match pixel_type {
|
||||
PixelType::U8x4 => &[128, 64, 0, 128],
|
||||
PixelType::U8x2 => &[128, 128],
|
||||
PixelType::U16x2 => &[0, 128, 0, 128],
|
||||
PixelType::U16x4 => &[0, 128, 0, 64, 0, 0, 0, 128],
|
||||
PixelType::F32x2 => &f32x2_bytes,
|
||||
PixelType::F32x4 => &f32x4_bytes,
|
||||
_ => unreachable!(),
|
||||
};
|
||||
let src_data = get_src_image(width, height, pixel_type, pixel);
|
||||
@@ -131,6 +141,7 @@ fn bench_alpha(bench_group: &mut utils::BenchGroup) {
|
||||
PixelType::U16x2,
|
||||
PixelType::U16x4,
|
||||
PixelType::F32x2,
|
||||
PixelType::F32x4,
|
||||
];
|
||||
let mut cpu_extensions = vec![CpuExtensions::None];
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
|
||||
@@ -0,0 +1,186 @@
|
||||
use std::arch::x86_64::*;
|
||||
|
||||
use crate::pixels::F32x4;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
use super::native;
|
||||
|
||||
#[target_feature(enable = "avx2")]
|
||||
pub(crate) unsafe fn multiply_alpha(
|
||||
src_view: &impl ImageView<Pixel = F32x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = F32x4>,
|
||||
) {
|
||||
let src_rows = src_view.iter_rows(0);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
multiply_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "avx2")]
|
||||
pub(crate) unsafe fn multiply_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = F32x4>) {
|
||||
for row in image_view.iter_rows_mut(0) {
|
||||
multiply_alpha_row_inplace(row);
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "avx2")]
|
||||
pub(crate) unsafe fn multiply_alpha_row(src_row: &[F32x4], dst_row: &mut [F32x4]) {
|
||||
let src_chunks = src_row.chunks_exact(8);
|
||||
let src_remainder = src_chunks.remainder();
|
||||
let mut dst_chunks = dst_row.chunks_exact_mut(8);
|
||||
for (src_chunk, dst_chunk) in src_chunks.zip(&mut dst_chunks) {
|
||||
let src_pixels = load_8_pixels(src_chunk);
|
||||
multiply_alpha_8_pixels(src_pixels, dst_chunk);
|
||||
}
|
||||
|
||||
if !src_remainder.is_empty() {
|
||||
let dst_reminder = dst_chunks.into_remainder();
|
||||
native::multiply_alpha_row(src_remainder, dst_reminder);
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "avx2")]
|
||||
pub(crate) unsafe fn multiply_alpha_row_inplace(row: &mut [F32x4]) {
|
||||
let mut chunks = row.chunks_exact_mut(8);
|
||||
for chunk in &mut chunks {
|
||||
let src_pixels = load_8_pixels(chunk);
|
||||
multiply_alpha_8_pixels(src_pixels, chunk);
|
||||
}
|
||||
|
||||
let reminder = chunks.into_remainder();
|
||||
if !reminder.is_empty() {
|
||||
native::multiply_alpha_row_inplace(reminder);
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "avx2")]
|
||||
unsafe fn multiply_alpha_8_pixels(pixels: [__m256; 4], dst_chunk: &mut [F32x4]) {
|
||||
let r_f32x8 = _mm256_mul_ps(pixels[0], pixels[3]);
|
||||
let g_f32x8 = _mm256_mul_ps(pixels[1], pixels[3]);
|
||||
let b_f32x8 = _mm256_mul_ps(pixels[2], pixels[3]);
|
||||
store_8_pixels([r_f32x8, g_f32x8, b_f32x8, pixels[3]], dst_chunk);
|
||||
}
|
||||
|
||||
// Divide
|
||||
|
||||
#[target_feature(enable = "avx2")]
|
||||
pub(crate) unsafe fn divide_alpha(
|
||||
src_view: &impl ImageView<Pixel = F32x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = F32x4>,
|
||||
) {
|
||||
let src_rows = src_view.iter_rows(0);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
divide_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "avx2")]
|
||||
pub(crate) unsafe fn divide_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = F32x4>) {
|
||||
for row in image_view.iter_rows_mut(0) {
|
||||
divide_alpha_row_inplace(row);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "avx2")]
|
||||
pub(crate) unsafe fn divide_alpha_row(src_row: &[F32x4], dst_row: &mut [F32x4]) {
|
||||
let src_chunks = src_row.chunks_exact(8);
|
||||
let src_remainder = src_chunks.remainder();
|
||||
let mut dst_chunks = dst_row.chunks_exact_mut(8);
|
||||
|
||||
for (src_chunk, dst_chunk) in src_chunks.zip(&mut dst_chunks) {
|
||||
let src_pixels = load_8_pixels(src_chunk);
|
||||
divide_alpha_8_pixels(src_pixels, dst_chunk);
|
||||
}
|
||||
|
||||
if !src_remainder.is_empty() {
|
||||
let dst_reminder = dst_chunks.into_remainder();
|
||||
native::divide_alpha_row(src_remainder, dst_reminder);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "avx2")]
|
||||
pub(crate) unsafe fn divide_alpha_row_inplace(row: &mut [F32x4]) {
|
||||
let mut chunks = row.chunks_exact_mut(8);
|
||||
for chunk in &mut chunks {
|
||||
let src_pixels = load_8_pixels(chunk);
|
||||
divide_alpha_8_pixels(src_pixels, chunk);
|
||||
}
|
||||
|
||||
let reminder = chunks.into_remainder();
|
||||
if !reminder.is_empty() {
|
||||
native::divide_alpha_row_inplace(reminder);
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "avx2")]
|
||||
unsafe fn divide_alpha_8_pixels(pixels: [__m256; 4], dst_chunk: &mut [F32x4]) {
|
||||
let mut r_f32x8 = _mm256_div_ps(pixels[0], pixels[3]);
|
||||
let mut g_f32x8 = _mm256_div_ps(pixels[1], pixels[3]);
|
||||
let mut b_f32x8 = _mm256_div_ps(pixels[2], pixels[3]);
|
||||
|
||||
let zero = _mm256_setzero_ps();
|
||||
let mask_zero = _mm256_cmp_ps::<_CMP_NEQ_UQ>(pixels[3], zero);
|
||||
r_f32x8 = _mm256_and_ps(mask_zero, r_f32x8);
|
||||
g_f32x8 = _mm256_and_ps(mask_zero, g_f32x8);
|
||||
b_f32x8 = _mm256_and_ps(mask_zero, b_f32x8);
|
||||
|
||||
store_8_pixels([r_f32x8, g_f32x8, b_f32x8, pixels[3]], dst_chunk);
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "avx2")]
|
||||
unsafe fn load_8_pixels(pixels: &[F32x4]) -> [__m256; 4] {
|
||||
let ptr = pixels.as_ptr() as *const f32;
|
||||
cols_into_rows([
|
||||
_mm256_loadu_ps(ptr),
|
||||
_mm256_loadu_ps(ptr.add(8)),
|
||||
_mm256_loadu_ps(ptr.add(16)),
|
||||
_mm256_loadu_ps(ptr.add(24)),
|
||||
])
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "avx2")]
|
||||
unsafe fn store_8_pixels(pixels: [__m256; 4], dst_chunk: &mut [F32x4]) {
|
||||
let pixels = cols_into_rows(pixels);
|
||||
let mut dst_ptr = dst_chunk.as_mut_ptr() as *mut f32;
|
||||
for rgba in pixels {
|
||||
_mm256_storeu_ps(dst_ptr, rgba);
|
||||
dst_ptr = dst_ptr.add(8)
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "avx2")]
|
||||
unsafe fn cols_into_rows(pixels: [__m256; 4]) -> [__m256; 4] {
|
||||
let rrgg02_rrgg13 = _mm256_unpacklo_ps(pixels[0], pixels[1]);
|
||||
let rrgg46_rrgg57 = _mm256_unpacklo_ps(pixels[2], pixels[3]);
|
||||
let r0246_r1357 = _mm256_castsi256_ps(_mm256_unpacklo_epi64(
|
||||
_mm256_castps_si256(rrgg02_rrgg13),
|
||||
_mm256_castps_si256(rrgg46_rrgg57),
|
||||
));
|
||||
let g0246_g1357 = _mm256_castsi256_ps(_mm256_unpackhi_epi64(
|
||||
_mm256_castps_si256(rrgg02_rrgg13),
|
||||
_mm256_castps_si256(rrgg46_rrgg57),
|
||||
));
|
||||
|
||||
let bbaa02_bbaa13 = _mm256_unpackhi_ps(pixels[0], pixels[1]);
|
||||
let bbaa46_bbaa57 = _mm256_unpackhi_ps(pixels[2], pixels[3]);
|
||||
let b0246_b1357 = _mm256_castsi256_ps(_mm256_unpacklo_epi64(
|
||||
_mm256_castps_si256(bbaa02_bbaa13),
|
||||
_mm256_castps_si256(bbaa46_bbaa57),
|
||||
));
|
||||
let a0246_a1357 = _mm256_castsi256_ps(_mm256_unpackhi_epi64(
|
||||
_mm256_castps_si256(bbaa02_bbaa13),
|
||||
_mm256_castps_si256(bbaa46_bbaa57),
|
||||
));
|
||||
[r0246_r1357, g0246_g1357, b0246_b1357, a0246_a1357]
|
||||
}
|
||||
+20
-20
@@ -4,11 +4,11 @@ use crate::{ImageError, ImageView, ImageViewMut};
|
||||
|
||||
use super::AlphaMulDiv;
|
||||
|
||||
// #[cfg(target_arch = "x86_64")]
|
||||
// mod avx2;
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod avx2;
|
||||
mod native;
|
||||
// #[cfg(target_arch = "x86_64")]
|
||||
// mod sse4;
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod sse4;
|
||||
|
||||
impl AlphaMulDiv for F32x4 {
|
||||
fn multiply_alpha(
|
||||
@@ -17,10 +17,10 @@ impl AlphaMulDiv for F32x4 {
|
||||
cpu_extensions: CpuExtensions,
|
||||
) -> Result<(), ImageError> {
|
||||
match cpu_extensions {
|
||||
// #[cfg(target_arch = "x86_64")]
|
||||
// CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha(src_view, dst_view) },
|
||||
// #[cfg(target_arch = "x86_64")]
|
||||
// CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha(src_view, dst_view) },
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => unsafe { neon::multiply_alpha(src_view, dst_view) },
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
@@ -35,10 +35,10 @@ impl AlphaMulDiv for F32x4 {
|
||||
cpu_extensions: CpuExtensions,
|
||||
) -> Result<(), ImageError> {
|
||||
match cpu_extensions {
|
||||
// #[cfg(target_arch = "x86_64")]
|
||||
// CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha_inplace(image_view) },
|
||||
// #[cfg(target_arch = "x86_64")]
|
||||
// CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::multiply_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::multiply_alpha_inplace(image_view) },
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => unsafe { neon::multiply_alpha_inplace(image_view) },
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
@@ -54,10 +54,10 @@ impl AlphaMulDiv for F32x4 {
|
||||
cpu_extensions: CpuExtensions,
|
||||
) -> Result<(), ImageError> {
|
||||
match cpu_extensions {
|
||||
// #[cfg(target_arch = "x86_64")]
|
||||
// CpuExtensions::Avx2 => unsafe { avx2::divide_alpha(src_view, dst_view) },
|
||||
// #[cfg(target_arch = "x86_64")]
|
||||
// CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::divide_alpha(src_view, dst_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha(src_view, dst_view) },
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => unsafe { crate::alpha::u16x2::neon::divide_alpha(src_view, dst_view) },
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
@@ -72,10 +72,10 @@ impl AlphaMulDiv for F32x4 {
|
||||
cpu_extensions: CpuExtensions,
|
||||
) -> Result<(), ImageError> {
|
||||
match cpu_extensions {
|
||||
// #[cfg(target_arch = "x86_64")]
|
||||
// CpuExtensions::Avx2 => unsafe { avx2::divide_alpha_inplace(image_view) },
|
||||
// #[cfg(target_arch = "x86_64")]
|
||||
// CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Avx2 => unsafe { avx2::divide_alpha_inplace(image_view) },
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
CpuExtensions::Sse4_1 => unsafe { sse4::divide_alpha_inplace(image_view) },
|
||||
// #[cfg(target_arch = "aarch64")]
|
||||
// CpuExtensions::Neon => unsafe { crate::alpha::u16x2::neon::divide_alpha_inplace(image_view) },
|
||||
// #[cfg(target_arch = "wasm32")]
|
||||
|
||||
@@ -0,0 +1,185 @@
|
||||
use std::arch::x86_64::*;
|
||||
|
||||
use crate::pixels::F32x4;
|
||||
use crate::{ImageView, ImageViewMut};
|
||||
|
||||
use super::native;
|
||||
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
pub(crate) unsafe fn multiply_alpha(
|
||||
src_view: &impl ImageView<Pixel = F32x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = F32x4>,
|
||||
) {
|
||||
let src_rows = src_view.iter_rows(0);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
multiply_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
pub(crate) unsafe fn multiply_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = F32x4>) {
|
||||
for row in image_view.iter_rows_mut(0) {
|
||||
multiply_alpha_row_inplace(row);
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
pub(crate) unsafe fn multiply_alpha_row(src_row: &[F32x4], dst_row: &mut [F32x4]) {
|
||||
let src_chunks = src_row.chunks_exact(4);
|
||||
let src_remainder = src_chunks.remainder();
|
||||
let mut dst_chunks = dst_row.chunks_exact_mut(4);
|
||||
for (src_chunk, dst_chunk) in src_chunks.zip(&mut dst_chunks) {
|
||||
let src_pixels = load_4_pixels(src_chunk);
|
||||
multiply_alpha_4_pixels(src_pixels, dst_chunk);
|
||||
}
|
||||
|
||||
if !src_remainder.is_empty() {
|
||||
let dst_reminder = dst_chunks.into_remainder();
|
||||
native::multiply_alpha_row(src_remainder, dst_reminder);
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
pub(crate) unsafe fn multiply_alpha_row_inplace(row: &mut [F32x4]) {
|
||||
let mut chunks = row.chunks_exact_mut(4);
|
||||
for chunk in &mut chunks {
|
||||
let src_pixels = load_4_pixels(chunk);
|
||||
multiply_alpha_4_pixels(src_pixels, chunk);
|
||||
}
|
||||
|
||||
let reminder = chunks.into_remainder();
|
||||
if !reminder.is_empty() {
|
||||
native::multiply_alpha_row_inplace(reminder);
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
unsafe fn multiply_alpha_4_pixels(pixels: [__m128; 4], dst_chunk: &mut [F32x4]) {
|
||||
let r_f32x4 = _mm_mul_ps(pixels[0], pixels[3]);
|
||||
let g_f32x4 = _mm_mul_ps(pixels[1], pixels[3]);
|
||||
let b_f32x4 = _mm_mul_ps(pixels[2], pixels[3]);
|
||||
store_4_pixels([r_f32x4, g_f32x4, b_f32x4, pixels[3]], dst_chunk);
|
||||
}
|
||||
|
||||
// Divide
|
||||
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
pub(crate) unsafe fn divide_alpha(
|
||||
src_view: &impl ImageView<Pixel = F32x4>,
|
||||
dst_view: &mut impl ImageViewMut<Pixel = F32x4>,
|
||||
) {
|
||||
let src_rows = src_view.iter_rows(0);
|
||||
let dst_rows = dst_view.iter_rows_mut(0);
|
||||
|
||||
for (src_row, dst_row) in src_rows.zip(dst_rows) {
|
||||
divide_alpha_row(src_row, dst_row);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
pub(crate) unsafe fn divide_alpha_inplace(image_view: &mut impl ImageViewMut<Pixel = F32x4>) {
|
||||
for row in image_view.iter_rows_mut(0) {
|
||||
divide_alpha_row_inplace(row);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
pub(crate) unsafe fn divide_alpha_row(src_row: &[F32x4], dst_row: &mut [F32x4]) {
|
||||
let src_chunks = src_row.chunks_exact(4);
|
||||
let src_remainder = src_chunks.remainder();
|
||||
let mut dst_chunks = dst_row.chunks_exact_mut(4);
|
||||
|
||||
for (src_chunk, dst_chunk) in src_chunks.zip(&mut dst_chunks) {
|
||||
let src_pixels = load_4_pixels(src_chunk);
|
||||
divide_alpha_4_pixels(src_pixels, dst_chunk);
|
||||
}
|
||||
|
||||
if !src_remainder.is_empty() {
|
||||
let dst_reminder = dst_chunks.into_remainder();
|
||||
native::divide_alpha_row(src_remainder, dst_reminder);
|
||||
}
|
||||
}
|
||||
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
pub(crate) unsafe fn divide_alpha_row_inplace(row: &mut [F32x4]) {
|
||||
let mut chunks = row.chunks_exact_mut(4);
|
||||
for chunk in &mut chunks {
|
||||
let src_pixels = load_4_pixels(chunk);
|
||||
divide_alpha_4_pixels(src_pixels, chunk);
|
||||
}
|
||||
|
||||
let reminder = chunks.into_remainder();
|
||||
if !reminder.is_empty() {
|
||||
native::divide_alpha_row_inplace(reminder);
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
unsafe fn divide_alpha_4_pixels(pixels: [__m128; 4], dst_chunk: &mut [F32x4]) {
|
||||
let mut r_f32x4 = _mm_div_ps(pixels[0], pixels[3]);
|
||||
let mut g_f32x4 = _mm_div_ps(pixels[1], pixels[3]);
|
||||
let mut b_f32x4 = _mm_div_ps(pixels[2], pixels[3]);
|
||||
let zero = _mm_setzero_ps();
|
||||
let mask_zero = _mm_cmpneq_ps(pixels[3], zero);
|
||||
r_f32x4 = _mm_and_ps(mask_zero, r_f32x4);
|
||||
g_f32x4 = _mm_and_ps(mask_zero, g_f32x4);
|
||||
b_f32x4 = _mm_and_ps(mask_zero, b_f32x4);
|
||||
|
||||
store_4_pixels([r_f32x4, g_f32x4, b_f32x4, pixels[3]], dst_chunk);
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
unsafe fn load_4_pixels(pixels: &[F32x4]) -> [__m128; 4] {
|
||||
let ptr = pixels.as_ptr() as *const f32;
|
||||
cols_into_rows([
|
||||
_mm_loadu_ps(ptr),
|
||||
_mm_loadu_ps(ptr.add(4)),
|
||||
_mm_loadu_ps(ptr.add(8)),
|
||||
_mm_loadu_ps(ptr.add(12)),
|
||||
])
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
unsafe fn store_4_pixels(pixels: [__m128; 4], dst_chunk: &mut [F32x4]) {
|
||||
let pixels = cols_into_rows(pixels);
|
||||
let mut dst_ptr = dst_chunk.as_mut_ptr() as *mut f32;
|
||||
for rgba in pixels {
|
||||
_mm_storeu_ps(dst_ptr, rgba);
|
||||
dst_ptr = dst_ptr.add(4)
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
unsafe fn cols_into_rows(pixels: [__m128; 4]) -> [__m128; 4] {
|
||||
let rrgg01 = _mm_unpacklo_ps(pixels[0], pixels[1]);
|
||||
let rrgg23 = _mm_unpacklo_ps(pixels[2], pixels[3]);
|
||||
let r0123 = _mm_castsi128_ps(_mm_unpacklo_epi64(
|
||||
_mm_castps_si128(rrgg01),
|
||||
_mm_castps_si128(rrgg23),
|
||||
));
|
||||
let g0123 = _mm_castsi128_ps(_mm_unpackhi_epi64(
|
||||
_mm_castps_si128(rrgg01),
|
||||
_mm_castps_si128(rrgg23),
|
||||
));
|
||||
|
||||
let bbaa01 = _mm_unpackhi_ps(pixels[0], pixels[1]);
|
||||
let bbaa23 = _mm_unpackhi_ps(pixels[2], pixels[3]);
|
||||
let b0123 = _mm_castsi128_ps(_mm_unpacklo_epi64(
|
||||
_mm_castps_si128(bbaa01),
|
||||
_mm_castps_si128(bbaa23),
|
||||
));
|
||||
let a0123 = _mm_castsi128_ps(_mm_unpackhi_epi64(
|
||||
_mm_castps_si128(bbaa01),
|
||||
_mm_castps_si128(bbaa23),
|
||||
));
|
||||
[r0123, g0123, b0123, a0123]
|
||||
}
|
||||
@@ -21,8 +21,8 @@ const fn get_clip_table() -> [u8; 1280] {
|
||||
// Handles values form -640 to 639.
|
||||
static CLIP8_LOOKUPS: [u8; 1280] = get_clip_table();
|
||||
|
||||
// 8 bits for result. Filter can have negative areas.
|
||||
// In one cases the sum of the coefficients will be negative,
|
||||
// 8 bits for a result. Filter can have negative areas.
|
||||
// In one case, the sum of the coefficients will be negative,
|
||||
// in the other it will be more than 1.0. That is why we need
|
||||
// two extra bits for overflow and i32 type.
|
||||
const PRECISION_BITS: u8 = 32 - 8 - 2;
|
||||
@@ -57,8 +57,8 @@ impl Normalizer16 {
|
||||
for cur_precision in 0..PRECISION_BITS {
|
||||
precision = cur_precision;
|
||||
let next_value: i32 = (max_weight * (1 << (precision + 1)) as f64).round() as i32;
|
||||
// The next value will be outside the range, so just stop
|
||||
if next_value >= (1 << MAX_COEFFS_PRECISION) {
|
||||
// The next value will be outside the range, so stop
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
+1
-1
@@ -39,7 +39,7 @@ pub mod pixels;
|
||||
mod resizer;
|
||||
#[cfg(target_arch = "x86_64")]
|
||||
mod simd_utils;
|
||||
#[cfg(feature = "for_test")]
|
||||
#[cfg(feature = "for_testing")]
|
||||
pub mod testing;
|
||||
#[cfg(target_arch = "wasm32")]
|
||||
mod wasm32_utils;
|
||||
|
||||
+1
-1
@@ -19,7 +19,7 @@ pub(crate) fn foreach_with_pre_reading<D, I>(
|
||||
|
||||
macro_rules! test_log {
|
||||
($s:expr) => {
|
||||
#[cfg(feature = "for_test")]
|
||||
#[cfg(feature = "for_testing")]
|
||||
{
|
||||
use crate::testing::log_message;
|
||||
log_message($s);
|
||||
|
||||
+1
-1
@@ -5,7 +5,7 @@ edition = "2021"
|
||||
|
||||
|
||||
[dependencies]
|
||||
fast_image_resize = { path = "..", features = ["image"] }
|
||||
fast_image_resize = { path = "..", features = ["for_testing"] }
|
||||
image = "0.25.1"
|
||||
|
||||
|
||||
|
||||
@@ -58,6 +58,7 @@ fn mul_div_alpha_test<P: PixelTrait>(
|
||||
"divide"
|
||||
};
|
||||
|
||||
let cpu_ext_str = cpu_ext_into_str(cpu_extensions);
|
||||
let expected_pixels: Vec<P> = expected_pixels_tpl
|
||||
.iter()
|
||||
.copied()
|
||||
@@ -71,7 +72,8 @@ fn mul_div_alpha_test<P: PixelTrait>(
|
||||
{
|
||||
assert_eq!(
|
||||
r, *e,
|
||||
"failed test for {oper_str} alpha: src={s:?}, result={r:?}, expected_result={e:?}",
|
||||
"failed test for {oper_str} alpha with '{cpu_ext_str}' CPU extensions: \
|
||||
src={s:?}, result={r:?}, expected_result={e:?}",
|
||||
);
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user