Added using of (read|write)_unaligned for unaligned pointers on arm64 and wasm32 architecture (#15).

This commit is contained in:
Kirill Kuzminykh
2023-05-04 23:42:48 +03:00
parent 07778909a7
commit 8ff91df590
5 changed files with 60 additions and 34 deletions
+2 -1
View File
@@ -2,7 +2,8 @@
### Crate
- Added using of (read|write)_unaligned for unaligned pointers on `arm64` architecture.
- Added using of (read|write)_unaligned for unaligned pointers
on `arm64` and `wasm32` architecture.
([#15](https://github.com/Cykooz/fast_image_resize/issues/15)).
## [2.7.1] - 2023-04-28
+3 -1
View File
@@ -48,8 +48,10 @@ pub(crate) fn convolution_by_u16<T: PixelExt<Component = u16>>(
let mut ss = initial;
let src_rows = src_image.iter_rows(first_y_src);
for (&k, src_row) in ks.iter().zip(src_rows) {
// SAFETY: Alignment of src_row is greater or equal than alignment u16
// because one component of pixel type T is u16.
let src_ptr = src_row.as_ptr() as *const u16;
let src_component = unsafe { *src_ptr.add(x_src as usize) };
let src_component = unsafe { *src_ptr.add(x_src) };
ss += src_component as i64 * (k as i64);
}
*dst_component = normalizer.clip(ss);
+3 -3
View File
@@ -226,8 +226,8 @@ unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
sss0 = i16x8_narrow_i32x4(sss0, sss1);
sss0 = u8x16_narrow_i16x8(sss0, sss0);
let dst_ptr = dst_chunk.as_mut_ptr() as *mut [i64; 2];
(*dst_ptr)[0] = i64x2_extract_lane::<0>(sss0);
let dst_ptr = dst_chunk.as_mut_ptr() as *mut i64;
dst_ptr.write_unaligned(i64x2_extract_lane::<0>(sss0));
src_x += 8;
}
@@ -273,7 +273,7 @@ unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
sss = i16x8_narrow_i32x4(sss, sss);
let dst_ptr = dst_chunk.as_mut_ptr() as *mut i32;
*dst_ptr = i32x4_extract_lane::<0>(u8x16_narrow_i16x8(sss, sss));
dst_ptr.write_unaligned(i32x4_extract_lane::<0>(u8x16_narrow_i16x8(sss, sss)));
src_x += 4;
}
+38 -12
View File
@@ -1,5 +1,7 @@
use std::arch::aarch64::*;
use crate::pixels::PixelExt;
#[inline(always)]
pub unsafe fn load_u8x1<T>(buf: &[T], index: usize) -> uint8x8_t {
let ptr = buf.get_unchecked(index..).as_ptr() as *const u8;
@@ -101,50 +103,74 @@ pub unsafe fn load_u16x8x4<T>(buf: &[T], index: usize) -> uint16x8x4_t {
}
#[inline(always)]
pub unsafe fn load_deintrel_u16x1x3<T>(buf: &[T], index: usize) -> uint16x4x3_t {
pub unsafe fn load_deintrel_u16x1x3<T: PixelExt<Component = u16>>(
buf: &[T],
index: usize,
) -> uint16x4x3_t {
let mut arr = [0u16; 12];
let src_ptr = buf.get_unchecked(index..).as_ptr() as *const u16;
let src_slice = std::slice::from_raw_parts(src_ptr, 3);
arr[0..3].copy_from_slice(src_slice);
let dst_ptr = arr.as_mut_ptr();
std::ptr::copy_nonoverlapping(src_ptr, dst_ptr, 3);
vld3_u16(arr.as_ptr())
}
#[inline(always)]
pub unsafe fn load_deintrel_u16x2x3<T>(buf: &[T], index: usize) -> uint16x4x3_t {
pub unsafe fn load_deintrel_u16x2x3<T: PixelExt<Component = u16>>(
buf: &[T],
index: usize,
) -> uint16x4x3_t {
let mut arr = [0u16; 12];
let src_ptr = buf.get_unchecked(index..).as_ptr() as *const u16;
let src_slice = std::slice::from_raw_parts(src_ptr, 6);
arr[0..6].copy_from_slice(src_slice);
let dst_ptr = arr.as_mut_ptr();
std::ptr::copy_nonoverlapping(src_ptr, dst_ptr, 6);
vld3_u16(arr.as_ptr())
}
#[inline(always)]
pub unsafe fn load_deintrel_u16x4x3<T>(buf: &[T], index: usize) -> uint16x4x3_t {
pub unsafe fn load_deintrel_u16x4x3<T: PixelExt<Component = u16>>(
buf: &[T],
index: usize,
) -> uint16x4x3_t {
vld3_u16(buf.get_unchecked(index..).as_ptr() as *const u16)
}
#[inline(always)]
pub unsafe fn load_deintrel_u16x4x4<T>(buf: &[T], index: usize) -> uint16x4x4_t {
pub unsafe fn load_deintrel_u16x4x4<T: PixelExt<Component = u16>>(
buf: &[T],
index: usize,
) -> uint16x4x4_t {
vld4_u16(buf.get_unchecked(index..).as_ptr() as *const u16)
}
#[inline(always)]
pub unsafe fn load_deintrel_u16x4x2<T>(buf: &[T], index: usize) -> uint16x4x2_t {
pub unsafe fn load_deintrel_u16x4x2<T: PixelExt<Component = u16>>(
buf: &[T],
index: usize,
) -> uint16x4x2_t {
vld2_u16(buf.get_unchecked(index..).as_ptr() as *const u16)
}
#[inline(always)]
pub unsafe fn load_deintrel_u16x8x2<T>(buf: &[T], index: usize) -> uint16x8x2_t {
pub unsafe fn load_deintrel_u16x8x2<T: PixelExt<Component = u16>>(
buf: &[T],
index: usize,
) -> uint16x8x2_t {
vld2q_u16(buf.get_unchecked(index..).as_ptr() as *const u16)
}
#[inline(always)]
pub unsafe fn load_deintrel_u16x8x3<T>(buf: &[T], index: usize) -> uint16x8x3_t {
pub unsafe fn load_deintrel_u16x8x3<T: PixelExt<Component = u16>>(
buf: &[T],
index: usize,
) -> uint16x8x3_t {
vld3q_u16(buf.get_unchecked(index..).as_ptr() as *const u16)
}
#[inline(always)]
pub unsafe fn load_deintrel_u16x8x4<T>(buf: &[T], index: usize) -> uint16x8x4_t {
pub unsafe fn load_deintrel_u16x8x4<T: PixelExt<Component = u16>>(
buf: &[T],
index: usize,
) -> uint16x8x4_t {
vld4q_u16(buf.get_unchecked(index..).as_ptr() as *const u16)
}
+14 -17
View File
@@ -1,5 +1,4 @@
use std::arch::wasm32::*;
use std::ptr;
use crate::pixels::{U8x3, U8x4};
@@ -12,45 +11,43 @@ pub(crate) unsafe fn load_v128<T>(buf: &[T], index: usize) -> v128 {
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn loadl_i64<T>(buf: &[T], index: usize) -> v128 {
let i = buf.get_unchecked(index..).as_ptr() as *const i64;
i64x2(ptr::read_unaligned(i), 0)
let p = buf.get_unchecked(index..).as_ptr() as *const i64;
i64x2(p.read_unaligned(), 0)
}
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn loadl_i32<T>(buf: &[T], index: usize) -> v128 {
let i = buf.get_unchecked(index..).as_ptr() as *const i32;
i32x4(ptr::read_unaligned(i), 0, 0, 0)
let p = buf.get_unchecked(index..).as_ptr() as *const i32;
i32x4(p.read_unaligned(), 0, 0, 0)
}
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn loadl_i16<T>(buf: &[T], index: usize) -> v128 {
let i = buf.get_unchecked(index..).as_ptr() as *const i16;
i16x8(ptr::read_unaligned(i), 0, 0, 0, 0, 0, 0, 0)
let p = buf.get_unchecked(index..).as_ptr() as *const i16;
i16x8(p.read_unaligned(), 0, 0, 0, 0, 0, 0, 0)
}
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn ptr_i16_to_set1_i64(buf: &[i16], index: usize) -> v128 {
i64x2_splat(ptr::read_unaligned(
buf.get_unchecked(index..).as_ptr() as *const i64
))
let p = buf.get_unchecked(index..).as_ptr() as *const i64;
i64x2_splat(p.read_unaligned())
}
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn ptr_i16_to_set1_i32(buf: &[i16], index: usize) -> v128 {
i32x4_splat(ptr::read_unaligned(
buf.get_unchecked(index..).as_ptr() as *const i32
))
let p = buf.get_unchecked(index..).as_ptr() as *const i32;
i32x4_splat(p.read_unaligned())
}
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn i32x4_extend_low_ptr_u8(buf: &[u8], index: usize) -> v128 {
let ptr = buf.get_unchecked(index..).as_ptr() as *const v128;
u32x4_extend_low_u16x8(i16x8_extend_low_u8x16(v128_load(ptr)))
let p = buf.get_unchecked(index..).as_ptr() as *const v128;
u32x4_extend_low_u16x8(i16x8_extend_low_u8x16(v128_load(p)))
}
#[inline]
@@ -70,8 +67,8 @@ pub(crate) unsafe fn i32x4_extend_low_ptr_u8x3(buf: &[U8x3], index: usize) -> v1
#[inline]
#[target_feature(enable = "simd128")]
pub(crate) unsafe fn i32x4_v128_from_u8(buf: &[u8], index: usize) -> v128 {
let ptr = buf.get_unchecked(index..).as_ptr() as *const i32;
i32x4(*ptr, 0, 0, 0)
let p = buf.get_unchecked(index..).as_ptr() as *const i32;
i32x4(p.read_unaligned(), 0, 0, 0)
}
#[inline]