mirror of
https://github.com/Cykooz/fast_image_resize.git
synced 2026-10-08 01:11:09 +00:00
- Renamed NormalizerGuard into Normalizer.
- Removed unsafe blocks from implementation of Normalizer.
This commit is contained in:
+34
-34
@@ -32,11 +32,11 @@ Pipeline:
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|------------|:-------:|:--------:|:----------:|:--------:|
|
||||
| image | 19.24 | 82.52 | 152.17 | 207.63 |
|
||||
| resize | - | 52.19 | 103.40 | 154.15 |
|
||||
| fir rust | 0.28 | 40.88 | 69.39 | 101.53 |
|
||||
| fir sse4.1 | 0.28 | 28.21 | 43.03 | 59.46 |
|
||||
| fir avx2 | 0.28 | 7.33 | 9.47 | 13.59 |
|
||||
| image | 19.50 | 83.55 | 142.66 | 202.49 |
|
||||
| resize | - | 52.12 | 102.98 | 153.42 |
|
||||
| fir rust | 0.28 | 40.94 | 69.96 | 100.86 |
|
||||
| fir sse4.1 | 0.28 | 28.10 | 43.06 | 58.05 |
|
||||
| fir avx2 | 0.28 | 7.24 | 9.49 | 13.61 |
|
||||
|
||||
### Resize RGBA8 image (U8x4) 4928x3279 => 852x567
|
||||
|
||||
@@ -51,10 +51,10 @@ Pipeline:
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|------------|:-------:|:--------:|:----------:|:--------:|
|
||||
| resize | - | 61.93 | 122.10 | 182.55 |
|
||||
| fir rust | 0.18 | 36.57 | 52.28 | 74.14 |
|
||||
| fir sse4.1 | 0.18 | 13.14 | 17.21 | 22.44 |
|
||||
| fir avx2 | 0.18 | 9.69 | 11.99 | 16.23 |
|
||||
| resize | - | 61.96 | 122.09 | 182.24 |
|
||||
| fir rust | 0.19 | 36.38 | 52.06 | 74.03 |
|
||||
| fir sse4.1 | 0.19 | 13.37 | 17.44 | 22.63 |
|
||||
| fir avx2 | 0.19 | 9.76 | 12.18 | 16.30 |
|
||||
|
||||
### Resize L8 (luma) image (U8) 4928x3279 => 852x567
|
||||
|
||||
@@ -68,11 +68,11 @@ Pipeline:
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|------------|:-------:|:--------:|:----------:|:--------:|
|
||||
| image | 15.86 | 47.17 | 74.46 | 102.53 |
|
||||
| resize | - | 17.30 | 35.92 | 61.52 |
|
||||
| fir rust | 0.15 | 14.10 | 16.20 | 24.12 |
|
||||
| fir sse4.1 | 0.15 | 11.93 | 12.13 | 18.20 |
|
||||
| fir avx2 | 0.15 | 6.30 | 4.71 | 7.62 |
|
||||
| image | 15.95 | 47.09 | 74.65 | 103.92 |
|
||||
| resize | - | 17.28 | 35.54 | 61.15 |
|
||||
| fir rust | 0.16 | 14.33 | 16.25 | 24.35 |
|
||||
| fir sse4.1 | 0.16 | 12.17 | 12.18 | 18.41 |
|
||||
| fir avx2 | 0.16 | 6.31 | 4.66 | 7.97 |
|
||||
|
||||
### Resize LA8 (luma with alpha channel) image (U8x2) 4928x3279 => 852x567
|
||||
|
||||
@@ -89,9 +89,9 @@ Pipeline:
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|------------|:-------:|:--------:|:----------:|:--------:|
|
||||
| fir rust | 0.17 | 25.73 | 30.75 | 42.34 |
|
||||
| fir sse4.1 | 0.17 | 12.81 | 14.64 | 18.06 |
|
||||
| fir avx2 | 0.17 | 11.26 | 12.42 | 15.46 |
|
||||
| fir rust | 0.18 | 25.79 | 30.79 | 42.38 |
|
||||
| fir sse4.1 | 0.17 | 12.71 | 14.68 | 18.16 |
|
||||
| fir avx2 | 0.17 | 11.25 | 12.53 | 15.53 |
|
||||
|
||||
### Resize RGB16 image (U16x3) 4928x3279 => 852x567
|
||||
|
||||
@@ -105,11 +105,11 @@ Pipeline:
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|------------|:-------:|:--------:|:----------:|:--------:|
|
||||
| image | 18.58 | 76.20 | 138.35 | 193.78 |
|
||||
| resize | - | 54.72 | 106.58 | 158.30 |
|
||||
| fir rust | 0.33 | 43.80 | 80.11 | 116.95 |
|
||||
| fir sse4.1 | 0.33 | 24.40 | 39.44 | 55.86 |
|
||||
| fir avx2 | 0.33 | 20.51 | 30.34 | 35.88 |
|
||||
| image | 19.09 | 76.86 | 140.93 | 191.14 |
|
||||
| resize | - | 55.19 | 107.76 | 159.66 |
|
||||
| fir rust | 0.33 | 43.89 | 80.03 | 117.89 |
|
||||
| fir sse4.1 | 0.33 | 24.46 | 39.45 | 55.91 |
|
||||
| fir avx2 | 0.33 | 21.01 | 31.07 | 36.95 |
|
||||
|
||||
### Resize RGBA16 image (U16x4) 4928x3279 => 852x567
|
||||
|
||||
@@ -124,10 +124,10 @@ Pipeline:
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|------------|:-------:|:--------:|:----------:|:--------:|
|
||||
| resize | - | 63.81 | 127.53 | 191.08 |
|
||||
| fir rust | 0.37 | 80.36 | 118.89 | 159.05 |
|
||||
| fir sse4.1 | 0.37 | 42.70 | 63.96 | 86.08 |
|
||||
| fir avx2 | 0.37 | 25.40 | 36.62 | 47.99 |
|
||||
| resize | - | 62.92 | 124.11 | 185.29 |
|
||||
| fir rust | 0.38 | 84.98 | 123.53 | 163.92 |
|
||||
| fir sse4.1 | 0.38 | 42.35 | 63.57 | 85.62 |
|
||||
| fir avx2 | 0.38 | 23.60 | 34.29 | 45.60 |
|
||||
|
||||
### Resize L16 image (U16) 4928x3279 => 852x567
|
||||
|
||||
@@ -141,11 +141,11 @@ Pipeline:
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|------------|:-------:|:--------:|:----------:|:--------:|
|
||||
| image | 16.37 | 47.13 | 74.89 | 104.83 |
|
||||
| resize | - | 15.35 | 31.91 | 57.04 |
|
||||
| fir rust | 0.17 | 19.05 | 28.02 | 37.48 |
|
||||
| fir sse4.1 | 0.17 | 7.80 | 13.16 | 19.13 |
|
||||
| fir avx2 | 0.17 | 7.07 | 9.48 | 14.80 |
|
||||
| image | 16.43 | 48.12 | 75.85 | 104.63 |
|
||||
| resize | - | 15.48 | 32.05 | 57.16 |
|
||||
| fir rust | 0.17 | 19.31 | 26.93 | 37.76 |
|
||||
| fir sse4.1 | 0.18 | 7.87 | 13.22 | 19.26 |
|
||||
| fir avx2 | 0.18 | 7.17 | 9.62 | 14.73 |
|
||||
|
||||
### Resize LA16 (luma with alpha channel) image (U16x2) 4928x3279 => 852x567
|
||||
|
||||
@@ -162,6 +162,6 @@ Pipeline:
|
||||
|
||||
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|
||||
|------------|:-------:|:--------:|:----------:|:--------:|
|
||||
| fir rust | 0.19 | 33.44 | 53.17 | 72.06 |
|
||||
| fir sse4.1 | 0.19 | 21.89 | 33.99 | 46.56 |
|
||||
| fir avx2 | 0.19 | 15.22 | 21.95 | 28.99 |
|
||||
| fir rust | 0.19 | 34.70 | 54.59 | 74.64 |
|
||||
| fir sse4.1 | 0.19 | 22.71 | 34.82 | 47.46 |
|
||||
| fir avx2 | 0.20 | 15.68 | 22.42 | 29.13 |
|
||||
|
||||
@@ -1,6 +1,5 @@
|
||||
use std::slice;
|
||||
|
||||
use super::Bound;
|
||||
use crate::convolution::Coefficients;
|
||||
|
||||
// This code is based on C-implementation from Pillow-SIMD package for Python
|
||||
// https://github.com/uploadcare/pillow-simd
|
||||
@@ -28,25 +27,27 @@ const CLIP8_LOOKUPS: [u8; 1280] = get_clip_table();
|
||||
// two extra bits for overflow and i32 type.
|
||||
const PRECISION_BITS: u8 = 32 - 8 - 2;
|
||||
// We use i16 type to store coefficients.
|
||||
const MAX_COEFS_PRECISION: u8 = 16 - 1;
|
||||
const MAX_COEFFS_PRECISION: u8 = 16 - 1;
|
||||
|
||||
/// Converts `Vec<f64>` into `&[i16]` without additional memory allocations.
|
||||
/// The memory buffer from `Vec<f64>` uses as `[i16]` .
|
||||
pub struct NormalizerGuard16 {
|
||||
values: Vec<f64>,
|
||||
/// Converts `Vec<f64>` into `Vec<i16>`.
|
||||
pub(crate) struct Normalizer16 {
|
||||
values: Vec<i16>,
|
||||
precision: u8,
|
||||
window_size: usize,
|
||||
bounds: Vec<Bound>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
pub struct CoefficientsI16Chunk<'a> {
|
||||
pub(crate) struct CoefficientsI16Chunk<'a> {
|
||||
pub start: u32,
|
||||
pub values: &'a [i16],
|
||||
}
|
||||
|
||||
impl NormalizerGuard16 {
|
||||
impl Normalizer16 {
|
||||
#[inline]
|
||||
pub fn new(mut values: Vec<f64>) -> Self {
|
||||
let max_weight = values
|
||||
pub fn new(coefficients: Coefficients) -> Self {
|
||||
let max_weight = coefficients
|
||||
.values
|
||||
.iter()
|
||||
.max_by(|&x, &y| x.partial_cmp(y).unwrap())
|
||||
.unwrap_or(&0.0)
|
||||
@@ -57,36 +58,32 @@ impl NormalizerGuard16 {
|
||||
precision = cur_precision;
|
||||
let next_value: i32 = (max_weight * (1 << (precision + 1)) as f64).round() as i32;
|
||||
// The next value will be outside the range, so just stop
|
||||
if next_value >= (1 << MAX_COEFS_PRECISION) {
|
||||
if next_value >= (1 << MAX_COEFFS_PRECISION) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
debug_assert!(precision >= 4); // required for some SIMD optimisations
|
||||
|
||||
let len = values.len();
|
||||
let ptr = values.as_mut_ptr();
|
||||
// Size of `[i16]` always will be not greater than `[f64]` with same number of items
|
||||
let values_i16 = unsafe { slice::from_raw_parts_mut(ptr as *mut i16, len) };
|
||||
let mut values_i16 = Vec::with_capacity(coefficients.values.len());
|
||||
|
||||
let scale = (1 << precision) as f64;
|
||||
for (&src, dst) in values.iter().zip(values_i16.iter_mut()) {
|
||||
*dst = (src * scale).round() as i16;
|
||||
for src in coefficients.values.iter().copied() {
|
||||
values_i16.push((src * scale).round() as i16);
|
||||
}
|
||||
Self {
|
||||
values: values_i16,
|
||||
precision,
|
||||
window_size: coefficients.window_size,
|
||||
bounds: coefficients.bounds,
|
||||
}
|
||||
Self { values, precision }
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn normalized_chunks(
|
||||
&self,
|
||||
window_size: usize,
|
||||
bounds: &[Bound],
|
||||
) -> Vec<CoefficientsI16Chunk> {
|
||||
let len = self.values.len();
|
||||
let ptr = self.values.as_ptr();
|
||||
let mut cooefs = unsafe { slice::from_raw_parts(ptr as *const i16, len) };
|
||||
let mut res = Vec::with_capacity(bounds.len());
|
||||
for bound in bounds {
|
||||
let (left, right) = cooefs.split_at(window_size);
|
||||
pub fn normalized_chunks(&self) -> Vec<CoefficientsI16Chunk> {
|
||||
let mut cooefs = self.values.as_slice();
|
||||
let mut res = Vec::with_capacity(self.bounds.len());
|
||||
for bound in self.bounds.iter() {
|
||||
let (left, right) = cooefs.split_at(self.window_size);
|
||||
cooefs = right;
|
||||
let size = bound.size as usize;
|
||||
res.push(CoefficientsI16Chunk {
|
||||
@@ -121,25 +118,27 @@ impl NormalizerGuard16 {
|
||||
// two extra bits for overflow and i64 type.
|
||||
const PRECISION16_BITS: u8 = 64 - 16 - 2;
|
||||
// We use i32 type to store coefficients.
|
||||
const MAX_COEFS_PRECISION16: u8 = 32 - 1;
|
||||
const MAX_COEFFS_PRECISION16: u8 = 32 - 1;
|
||||
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
pub struct CoefficientsI32Chunk<'a> {
|
||||
pub(crate) struct CoefficientsI32Chunk<'a> {
|
||||
pub start: u32,
|
||||
pub values: &'a [i32],
|
||||
}
|
||||
|
||||
/// Converts `Vec<f64>` into `&[i32]` without additional memory allocations.
|
||||
/// The memory buffer from `Vec<f64>` uses as `[i32]` .
|
||||
pub struct NormalizerGuard32 {
|
||||
values: Vec<f64>,
|
||||
/// Converts `Vec<f64>` into `Vec<i32>`.
|
||||
pub(crate) struct Normalizer32 {
|
||||
values: Vec<i32>,
|
||||
precision: u8,
|
||||
window_size: usize,
|
||||
bounds: Vec<Bound>,
|
||||
}
|
||||
|
||||
impl NormalizerGuard32 {
|
||||
impl Normalizer32 {
|
||||
#[inline]
|
||||
pub fn new(mut values: Vec<f64>) -> Self {
|
||||
let max_weight = values
|
||||
pub fn new(coefficients: Coefficients) -> Self {
|
||||
let max_weight = coefficients
|
||||
.values
|
||||
.iter()
|
||||
.max_by(|&x, &y| x.partial_cmp(y).unwrap())
|
||||
.unwrap_or(&0.0)
|
||||
@@ -150,36 +149,32 @@ impl NormalizerGuard32 {
|
||||
precision = cur_precision;
|
||||
let next_value: i64 = (max_weight * (1i64 << (precision + 1)) as f64).round() as i64;
|
||||
// The next value will be outside the range, so just stop
|
||||
if next_value >= (1i64 << MAX_COEFS_PRECISION16) {
|
||||
if next_value >= (1i64 << MAX_COEFFS_PRECISION16) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
debug_assert!(precision >= 4); // required for some SIMD optimisations
|
||||
|
||||
let len = values.len();
|
||||
let ptr = values.as_mut_ptr();
|
||||
// Size of `[i32]` always will be not greater than `[f64]` with same number of items
|
||||
let values_i32 = unsafe { slice::from_raw_parts_mut(ptr as *mut i32, len) };
|
||||
let mut values_i32 = Vec::with_capacity(coefficients.values.len());
|
||||
|
||||
let scale = (1i64 << precision) as f64;
|
||||
for (&src, dst) in values.iter().zip(values_i32.iter_mut()) {
|
||||
*dst = (src * scale).round() as i32;
|
||||
for src in coefficients.values.iter().copied() {
|
||||
values_i32.push((src * scale).round() as i32);
|
||||
}
|
||||
Self {
|
||||
values: values_i32,
|
||||
precision,
|
||||
window_size: coefficients.window_size,
|
||||
bounds: coefficients.bounds,
|
||||
}
|
||||
Self { values, precision }
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn normalized_chunks(
|
||||
&self,
|
||||
window_size: usize,
|
||||
bounds: &[Bound],
|
||||
) -> Vec<CoefficientsI32Chunk> {
|
||||
let len = self.values.len();
|
||||
let ptr = self.values.as_ptr();
|
||||
let mut cooefs = unsafe { slice::from_raw_parts(ptr as *const i32, len) };
|
||||
let mut res = Vec::with_capacity(bounds.len());
|
||||
for bound in bounds {
|
||||
let (left, right) = cooefs.split_at(window_size);
|
||||
pub fn normalized_chunks(&self) -> Vec<CoefficientsI32Chunk> {
|
||||
let mut cooefs = self.values.as_slice();
|
||||
let mut res = Vec::with_capacity(self.bounds.len());
|
||||
for bound in self.bounds.iter() {
|
||||
let (left, right) = cooefs.split_at(self.window_size);
|
||||
cooefs = right;
|
||||
let size = bound.size as usize;
|
||||
res.push(CoefficientsI32Chunk {
|
||||
@@ -205,12 +200,20 @@ impl NormalizerGuard32 {
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn get_coefficients(value: f64) -> Coefficients {
|
||||
Coefficients {
|
||||
values: vec![value],
|
||||
window_size: 0,
|
||||
bounds: vec![],
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_minimal_precision() {
|
||||
// required for some SIMD optimisations
|
||||
assert!(NormalizerGuard16::new(vec![0.0]).precision() >= 4);
|
||||
assert!(NormalizerGuard16::new(vec![2.0]).precision() >= 4);
|
||||
assert!(NormalizerGuard32::new(vec![0.0]).precision() >= 4);
|
||||
assert!(NormalizerGuard32::new(vec![2.0]).precision() >= 4);
|
||||
assert!(Normalizer16::new(get_coefficients(0.0)).precision() >= 4);
|
||||
assert!(Normalizer16::new(get_coefficients(2.0)).precision() >= 4);
|
||||
assert!(Normalizer32::new(get_coefficients(0.0)).precision() >= 4);
|
||||
assert!(Normalizer32::new(get_coefficients(2.0)).precision() >= 4);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -12,23 +12,15 @@ pub(crate) fn horiz_convolution(
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard32::new(values);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(
|
||||
src_rows,
|
||||
dst_rows,
|
||||
&coefficients_chunks,
|
||||
&normalizer_guard,
|
||||
);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -39,7 +31,7 @@ pub(crate) fn horiz_convolution(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
&coefficients_chunks,
|
||||
&normalizer_guard,
|
||||
&normalizer,
|
||||
);
|
||||
}
|
||||
yy += 1;
|
||||
@@ -56,13 +48,13 @@ unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: FourRows<U16>,
|
||||
dst_rows: FourRowsMut<U16>,
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer_guard: &optimisations::NormalizerGuard32,
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let (s_row0, s_row1, s_row2, s_row3) = src_rows;
|
||||
let s_rows = [s_row0, s_row1, s_row2, s_row3];
|
||||
let (d_row0, d_row1, d_row2, d_row3) = dst_rows;
|
||||
let d_rows = [d_row0, d_row1, d_row2, d_row3];
|
||||
let precision = normalizer_guard.precision();
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 4];
|
||||
|
||||
@@ -202,10 +194,10 @@ unsafe fn horiz_convolution_four_rows(
|
||||
for (i, &ll) in ll_sum.iter().enumerate() {
|
||||
_mm256_storeu_si256((&mut ll_buf).as_mut_ptr() as *mut __m256i, ll);
|
||||
let dst_pixel = d_rows[i * 2].get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = normalizer_guard.clip(ll_buf[0] + ll_buf[1] + half_error);
|
||||
dst_pixel.0 = normalizer.clip(ll_buf[0] + ll_buf[1] + half_error);
|
||||
|
||||
let dst_pixel = d_rows[i * 2 + 1].get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = normalizer_guard.clip(ll_buf[2] + ll_buf[3] + half_error);
|
||||
dst_pixel.0 = normalizer.clip(ll_buf[2] + ll_buf[3] + half_error);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -219,9 +211,9 @@ unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16],
|
||||
dst_row: &mut [U16],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer_guard: &optimisations::NormalizerGuard32,
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let precision = normalizer_guard.precision();
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 4];
|
||||
|
||||
@@ -354,6 +346,6 @@ unsafe fn horiz_convolution_one_row(
|
||||
|
||||
_mm256_storeu_si256((&mut ll_buf).as_mut_ptr() as *mut __m256i, ll_sum);
|
||||
let dst_pixel = dst_row.get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = normalizer_guard.clip(ll_buf.iter().sum::<i64>() + half_error);
|
||||
dst_pixel.0 = normalizer.clip(ll_buf.iter().sum::<i64>() + half_error);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -9,11 +9,9 @@ pub(crate) fn horiz_convolution(
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard32::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let initial: i64 = 1 << (precision - 1);
|
||||
|
||||
let src_rows = src_image.iter_rows(offset);
|
||||
@@ -26,7 +24,7 @@ pub(crate) fn horiz_convolution(
|
||||
for (&k, src_pixel) in coeffs_chunk.values.iter().zip(src_pixels) {
|
||||
sum += src_pixel.0 as i64 * (k as i64);
|
||||
}
|
||||
dst_pixel.0 = normalizer_guard.clip(sum);
|
||||
dst_pixel.0 = normalizer.clip(sum);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -12,23 +12,15 @@ pub(crate) fn horiz_convolution(
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard32::new(values);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(
|
||||
src_rows,
|
||||
dst_rows,
|
||||
&coefficients_chunks,
|
||||
&normalizer_guard,
|
||||
);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -39,7 +31,7 @@ pub(crate) fn horiz_convolution(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
&coefficients_chunks,
|
||||
&normalizer_guard,
|
||||
&normalizer,
|
||||
);
|
||||
}
|
||||
yy += 1;
|
||||
@@ -56,13 +48,13 @@ unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: FourRows<U16>,
|
||||
dst_rows: FourRowsMut<U16>,
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer_guard: &optimisations::NormalizerGuard32,
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let (s_row0, s_row1, s_row2, s_row3) = src_rows;
|
||||
let s_rows = [s_row0, s_row1, s_row2, s_row3];
|
||||
let (d_row0, d_row1, d_row2, d_row3) = dst_rows;
|
||||
let d_rows = [d_row0, d_row1, d_row2, d_row3];
|
||||
let precision = normalizer_guard.precision();
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 2];
|
||||
|
||||
@@ -173,7 +165,7 @@ unsafe fn horiz_convolution_four_rows(
|
||||
for i in 0..4 {
|
||||
_mm_storeu_si128((&mut ll_buf).as_mut_ptr() as *mut __m128i, ll_sum[i]);
|
||||
let dst_pixel = d_rows[i].get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = normalizer_guard.clip(ll_buf.iter().sum::<i64>() + half_error);
|
||||
dst_pixel.0 = normalizer.clip(ll_buf.iter().sum::<i64>() + half_error);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -187,9 +179,9 @@ unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16],
|
||||
dst_row: &mut [U16],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer_guard: &optimisations::NormalizerGuard32,
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let precision = normalizer_guard.precision();
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 2];
|
||||
|
||||
@@ -288,6 +280,6 @@ unsafe fn horiz_convolution_one_row(
|
||||
|
||||
_mm_storeu_si128((&mut ll_buf).as_mut_ptr() as *mut __m128i, ll_sum);
|
||||
let dst_pixel = dst_row.get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = normalizer_guard.clip(ll_buf[0] + ll_buf[1] + half_error);
|
||||
dst_pixel.0 = normalizer.clip(ll_buf[0] + ll_buf[1] + half_error);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -12,23 +12,15 @@ pub(crate) fn horiz_convolution(
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard32::new(values);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(
|
||||
src_rows,
|
||||
dst_rows,
|
||||
&coefficients_chunks,
|
||||
&normalizer_guard,
|
||||
);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -39,7 +31,7 @@ pub(crate) fn horiz_convolution(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
&coefficients_chunks,
|
||||
&normalizer_guard,
|
||||
&normalizer,
|
||||
);
|
||||
}
|
||||
yy += 1;
|
||||
@@ -56,13 +48,13 @@ unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: FourRows<U16x2>,
|
||||
dst_rows: FourRowsMut<U16x2>,
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer_guard: &optimisations::NormalizerGuard32,
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let (s_row0, s_row1, s_row2, s_row3) = src_rows;
|
||||
let s_rows = [s_row0, s_row1, s_row2, s_row3];
|
||||
let (d_row0, d_row1, d_row2, d_row3) = dst_rows;
|
||||
let d_rows = [d_row0, d_row1, d_row2, d_row3];
|
||||
let precision = normalizer_guard.precision();
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 4];
|
||||
|
||||
@@ -179,16 +171,10 @@ unsafe fn horiz_convolution_four_rows(
|
||||
_mm256_storeu_si256((&mut ll_buf).as_mut_ptr() as *mut __m256i, ll);
|
||||
let dst_pixel = d_rows[i * 2].get_unchecked_mut(dst_x);
|
||||
|
||||
dst_pixel.0 = [
|
||||
normalizer_guard.clip(ll_buf[0]),
|
||||
normalizer_guard.clip(ll_buf[1]),
|
||||
];
|
||||
dst_pixel.0 = [normalizer.clip(ll_buf[0]), normalizer.clip(ll_buf[1])];
|
||||
|
||||
let dst_pixel = d_rows[i * 2 + 1].get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = [
|
||||
normalizer_guard.clip(ll_buf[2]),
|
||||
normalizer_guard.clip(ll_buf[3]),
|
||||
];
|
||||
dst_pixel.0 = [normalizer.clip(ll_buf[2]), normalizer.clip(ll_buf[3])];
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -202,9 +188,9 @@ unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16x2],
|
||||
dst_row: &mut [U16x2],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer_guard: &optimisations::NormalizerGuard32,
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let precision = normalizer_guard.precision();
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 4];
|
||||
|
||||
@@ -329,8 +315,8 @@ unsafe fn horiz_convolution_one_row(
|
||||
_mm256_storeu_si256((&mut ll_buf).as_mut_ptr() as *mut __m256i, ll_sum);
|
||||
let dst_pixel = dst_row.get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = [
|
||||
normalizer_guard.clip(ll_buf[0] + ll_buf[2] + half_error),
|
||||
normalizer_guard.clip(ll_buf[1] + ll_buf[3] + half_error),
|
||||
normalizer.clip(ll_buf[0] + ll_buf[2] + half_error),
|
||||
normalizer.clip(ll_buf[1] + ll_buf[3] + half_error),
|
||||
];
|
||||
}
|
||||
}
|
||||
|
||||
@@ -9,11 +9,9 @@ pub(crate) fn horiz_convolution(
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard32::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let initial: i64 = 1 << (precision - 1);
|
||||
|
||||
let src_rows = src_image.iter_rows(offset);
|
||||
@@ -29,7 +27,7 @@ pub(crate) fn horiz_convolution(
|
||||
}
|
||||
}
|
||||
for (i, s) in ss.iter().copied().enumerate() {
|
||||
dst_pixel.0[i] = normalizer_guard.clip(s);
|
||||
dst_pixel.0[i] = normalizer.clip(s);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -12,23 +12,15 @@ pub(crate) fn horiz_convolution(
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard32::new(values);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(
|
||||
src_rows,
|
||||
dst_rows,
|
||||
&coefficients_chunks,
|
||||
&normalizer_guard,
|
||||
);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -39,7 +31,7 @@ pub(crate) fn horiz_convolution(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
&coefficients_chunks,
|
||||
&normalizer_guard,
|
||||
&normalizer,
|
||||
);
|
||||
}
|
||||
yy += 1;
|
||||
@@ -57,13 +49,13 @@ unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: FourRows<U16x2>,
|
||||
dst_rows: FourRowsMut<U16x2>,
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer_guard: &optimisations::NormalizerGuard32,
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let (s_row0, s_row1, s_row2, s_row3) = src_rows;
|
||||
let s_rows = [s_row0, s_row1, s_row2, s_row3];
|
||||
let (d_row0, d_row1, d_row2, d_row3) = dst_rows;
|
||||
let d_rows = [d_row0, d_row1, d_row2, d_row3];
|
||||
let precision = normalizer_guard.precision();
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 2];
|
||||
|
||||
@@ -161,10 +153,7 @@ unsafe fn horiz_convolution_four_rows(
|
||||
for i in 0..4 {
|
||||
_mm_storeu_si128((&mut ll_buf).as_mut_ptr() as *mut __m128i, ll_sum[i]);
|
||||
let dst_pixel = d_rows[i].get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = [
|
||||
normalizer_guard.clip(ll_buf[0]),
|
||||
normalizer_guard.clip(ll_buf[1]),
|
||||
];
|
||||
dst_pixel.0 = [normalizer.clip(ll_buf[0]), normalizer.clip(ll_buf[1])];
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -180,9 +169,9 @@ unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16x2],
|
||||
dst_row: &mut [U16x2],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer_guard: &optimisations::NormalizerGuard32,
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let precision = normalizer_guard.precision();
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut ll_buf = [0i64; 2];
|
||||
|
||||
@@ -269,9 +258,6 @@ unsafe fn horiz_convolution_one_row(
|
||||
|
||||
_mm_storeu_si128((&mut ll_buf).as_mut_ptr() as *mut __m128i, ll_sum);
|
||||
let dst_pixel = dst_row.get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = [
|
||||
normalizer_guard.clip(ll_buf[0]),
|
||||
normalizer_guard.clip(ll_buf[1]),
|
||||
];
|
||||
dst_pixel.0 = [normalizer.clip(ll_buf[0]), normalizer.clip(ll_buf[1])];
|
||||
}
|
||||
}
|
||||
|
||||
@@ -12,23 +12,15 @@ pub(crate) fn horiz_convolution(
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard32::new(values);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(
|
||||
src_rows,
|
||||
dst_rows,
|
||||
&coefficients_chunks,
|
||||
&normalizer_guard,
|
||||
);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -39,7 +31,7 @@ pub(crate) fn horiz_convolution(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
&coefficients_chunks,
|
||||
&normalizer_guard,
|
||||
&normalizer,
|
||||
);
|
||||
}
|
||||
yy += 1;
|
||||
@@ -56,13 +48,13 @@ unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: FourRows<U16x3>,
|
||||
dst_rows: FourRowsMut<U16x3>,
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer_guard: &optimisations::NormalizerGuard32,
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let (s_row0, s_row1, s_row2, s_row3) = src_rows;
|
||||
let s_rows = [s_row0, s_row1, s_row2, s_row3];
|
||||
let (d_row0, d_row1, d_row2, d_row3) = dst_rows;
|
||||
let d_rows = [d_row0, d_row1, d_row2, d_row3];
|
||||
let precision = normalizer_guard.precision();
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut rg_buf = [0i64; 4];
|
||||
let mut rg_bb_buf = [0i64; 4];
|
||||
@@ -177,11 +169,9 @@ unsafe fn horiz_convolution_four_rows(
|
||||
_mm256_storeu_si256((&mut rg_bb_buf).as_mut_ptr() as *mut __m256i, rg_bb_sum[i]);
|
||||
_mm256_storeu_si256((&mut bbb_buf).as_mut_ptr() as *mut __m256i, bbb_sum[i]);
|
||||
let dst_pixel = d_rows[i].get_unchecked_mut(dst_x);
|
||||
dst_pixel.0[0] =
|
||||
normalizer_guard.clip(rg_buf[0] + rg_buf[2] + rg_bb_buf[0] + half_error);
|
||||
dst_pixel.0[1] =
|
||||
normalizer_guard.clip(rg_buf[1] + rg_buf[3] + rg_bb_buf[1] + half_error);
|
||||
dst_pixel.0[2] = normalizer_guard.clip(
|
||||
dst_pixel.0[0] = normalizer.clip(rg_buf[0] + rg_buf[2] + rg_bb_buf[0] + half_error);
|
||||
dst_pixel.0[1] = normalizer.clip(rg_buf[1] + rg_buf[3] + rg_bb_buf[1] + half_error);
|
||||
dst_pixel.0[2] = normalizer.clip(
|
||||
rg_bb_buf[2] + rg_bb_buf[3] + bbb_buf[0] + bbb_buf[1] + bbb_buf[2] + half_error,
|
||||
);
|
||||
}
|
||||
@@ -197,9 +187,9 @@ unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16x3],
|
||||
dst_row: &mut [U16x3],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer_guard: &optimisations::NormalizerGuard32,
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let precision = normalizer_guard.precision();
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut rg_buf = [0i64; 4];
|
||||
let mut rg_bb_buf = [0i64; 4];
|
||||
@@ -306,9 +296,9 @@ unsafe fn horiz_convolution_one_row(
|
||||
_mm256_storeu_si256((&mut rg_bb_buf).as_mut_ptr() as *mut __m256i, rg_bb_sum);
|
||||
_mm256_storeu_si256((&mut bbb_buf).as_mut_ptr() as *mut __m256i, bbb_sum);
|
||||
let dst_pixel = dst_row.get_unchecked_mut(dst_x);
|
||||
dst_pixel.0[0] = normalizer_guard.clip(rg_buf[0] + rg_buf[2] + rg_bb_buf[0] + half_error);
|
||||
dst_pixel.0[1] = normalizer_guard.clip(rg_buf[1] + rg_buf[3] + rg_bb_buf[1] + half_error);
|
||||
dst_pixel.0[2] = normalizer_guard
|
||||
dst_pixel.0[0] = normalizer.clip(rg_buf[0] + rg_buf[2] + rg_bb_buf[0] + half_error);
|
||||
dst_pixel.0[1] = normalizer.clip(rg_buf[1] + rg_buf[3] + rg_bb_buf[1] + half_error);
|
||||
dst_pixel.0[2] = normalizer
|
||||
.clip(rg_bb_buf[2] + rg_bb_buf[3] + bbb_buf[0] + bbb_buf[1] + bbb_buf[2] + half_error);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -9,11 +9,9 @@ pub(crate) fn horiz_convolution(
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard32::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let initial: i64 = 1 << (precision - 1);
|
||||
|
||||
let src_rows = src_image.iter_rows(offset);
|
||||
@@ -29,7 +27,7 @@ pub(crate) fn horiz_convolution(
|
||||
}
|
||||
}
|
||||
for (i, s) in ss.iter().copied().enumerate() {
|
||||
dst_pixel.0[i] = normalizer_guard.clip(s);
|
||||
dst_pixel.0[i] = normalizer.clip(s);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -13,18 +13,15 @@ pub(crate) fn horiz_convolution(
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard32::new(values);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_8u4x(src_rows, dst_rows, &coefficients_chunks, &normalizer_guard);
|
||||
horiz_convolution_8u4x(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -35,7 +32,7 @@ pub(crate) fn horiz_convolution(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
&coefficients_chunks,
|
||||
&normalizer_guard,
|
||||
&normalizer,
|
||||
);
|
||||
}
|
||||
yy += 1;
|
||||
@@ -53,13 +50,13 @@ unsafe fn horiz_convolution_8u4x(
|
||||
src_rows: FourRows<U16x3>,
|
||||
dst_rows: FourRowsMut<U16x3>,
|
||||
coefficients_chunks: &[CoefficientsI32Chunk],
|
||||
normalizer_guard: &optimisations::NormalizerGuard32,
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let (s_row0, s_row1, s_row2, s_row3) = src_rows;
|
||||
let s_rows = [s_row0, s_row1, s_row2, s_row3];
|
||||
let (d_row0, d_row1, d_row2, d_row3) = dst_rows;
|
||||
let d_rows = [d_row0, d_row1, d_row2, d_row3];
|
||||
let precision = normalizer_guard.precision();
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut rg_buf = [0i64; 2];
|
||||
let mut bb_buf = [0i64; 2];
|
||||
@@ -135,9 +132,9 @@ unsafe fn horiz_convolution_8u4x(
|
||||
_mm_storeu_si128((&mut rg_buf).as_mut_ptr() as *mut __m128i, rg_sum[i]);
|
||||
_mm_storeu_si128((&mut bb_buf).as_mut_ptr() as *mut __m128i, bb_sum[i]);
|
||||
let dst_pixel = d_rows[i].get_unchecked_mut(dst_x);
|
||||
dst_pixel.0[0] = normalizer_guard.clip(rg_buf[0] + half_error);
|
||||
dst_pixel.0[1] = normalizer_guard.clip(rg_buf[1] + half_error);
|
||||
dst_pixel.0[2] = normalizer_guard.clip(bb_buf[0] + bb_buf[1] + half_error);
|
||||
dst_pixel.0[0] = normalizer.clip(rg_buf[0] + half_error);
|
||||
dst_pixel.0[1] = normalizer.clip(rg_buf[1] + half_error);
|
||||
dst_pixel.0[2] = normalizer.clip(bb_buf[0] + bb_buf[1] + half_error);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -152,9 +149,9 @@ unsafe fn horiz_convolution_8u(
|
||||
src_row: &[U16x3],
|
||||
dst_row: &mut [U16x3],
|
||||
coefficients_chunks: &[CoefficientsI32Chunk],
|
||||
normalizer_guard: &optimisations::NormalizerGuard32,
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let precision = normalizer_guard.precision();
|
||||
let precision = normalizer.precision();
|
||||
let rg_initial = _mm_set1_epi64x(1 << (precision - 1));
|
||||
let bb_initial = _mm_set1_epi64x(1 << (precision - 2));
|
||||
|
||||
@@ -228,8 +225,8 @@ unsafe fn horiz_convolution_8u(
|
||||
_mm_storeu_si128((&mut rg_buf).as_mut_ptr() as *mut __m128i, rg_sum);
|
||||
_mm_storeu_si128((&mut bb_buf).as_mut_ptr() as *mut __m128i, bb_sum);
|
||||
let dst_pixel = dst_row.get_unchecked_mut(dst_x);
|
||||
dst_pixel.0[0] = normalizer_guard.clip(rg_buf[0]);
|
||||
dst_pixel.0[1] = normalizer_guard.clip(rg_buf[1]);
|
||||
dst_pixel.0[2] = normalizer_guard.clip(bb_buf[0] + bb_buf[1]);
|
||||
dst_pixel.0[0] = normalizer.clip(rg_buf[0]);
|
||||
dst_pixel.0[1] = normalizer.clip(rg_buf[1]);
|
||||
dst_pixel.0[2] = normalizer.clip(bb_buf[0] + bb_buf[1]);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -12,23 +12,15 @@ pub(crate) fn horiz_convolution(
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard32::new(values);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(
|
||||
src_rows,
|
||||
dst_rows,
|
||||
&coefficients_chunks,
|
||||
&normalizer_guard,
|
||||
);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -39,7 +31,7 @@ pub(crate) fn horiz_convolution(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
&coefficients_chunks,
|
||||
&normalizer_guard,
|
||||
&normalizer,
|
||||
);
|
||||
}
|
||||
yy += 1;
|
||||
@@ -57,13 +49,13 @@ unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: FourRows<U16x4>,
|
||||
dst_rows: FourRowsMut<U16x4>,
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer_guard: &optimisations::NormalizerGuard32,
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let (s_row0, s_row1, s_row2, s_row3) = src_rows;
|
||||
let s_rows = [s_row0, s_row1, s_row2, s_row3];
|
||||
let (d_row0, d_row1, d_row2, d_row3) = dst_rows;
|
||||
let d_rows = [d_row0, d_row1, d_row2, d_row3];
|
||||
let precision = normalizer_guard.precision();
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut rg_buf = [0i64; 4];
|
||||
let mut ba_buf = [0i64; 4];
|
||||
@@ -169,18 +161,18 @@ unsafe fn horiz_convolution_four_rows(
|
||||
|
||||
let dst_pixel = d_rows[i * 2].get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = [
|
||||
normalizer_guard.clip(rg_buf[0]),
|
||||
normalizer_guard.clip(rg_buf[1]),
|
||||
normalizer_guard.clip(ba_buf[0]),
|
||||
normalizer_guard.clip(ba_buf[1]),
|
||||
normalizer.clip(rg_buf[0]),
|
||||
normalizer.clip(rg_buf[1]),
|
||||
normalizer.clip(ba_buf[0]),
|
||||
normalizer.clip(ba_buf[1]),
|
||||
];
|
||||
|
||||
let dst_pixel = d_rows[i * 2 + 1].get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = [
|
||||
normalizer_guard.clip(rg_buf[2]),
|
||||
normalizer_guard.clip(rg_buf[3]),
|
||||
normalizer_guard.clip(ba_buf[2]),
|
||||
normalizer_guard.clip(ba_buf[3]),
|
||||
normalizer.clip(rg_buf[2]),
|
||||
normalizer.clip(rg_buf[3]),
|
||||
normalizer.clip(ba_buf[2]),
|
||||
normalizer.clip(ba_buf[3]),
|
||||
];
|
||||
}
|
||||
}
|
||||
@@ -197,9 +189,9 @@ unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16x4],
|
||||
dst_row: &mut [U16x4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer_guard: &optimisations::NormalizerGuard32,
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let precision = normalizer_guard.precision();
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut rg_buf = [0i64; 4];
|
||||
let mut ba_buf = [0i64; 4];
|
||||
@@ -304,10 +296,10 @@ unsafe fn horiz_convolution_one_row(
|
||||
_mm256_storeu_si256((&mut ba_buf).as_mut_ptr() as *mut __m256i, ba_sum);
|
||||
let dst_pixel = dst_row.get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = [
|
||||
normalizer_guard.clip(rg_buf[0] + rg_buf[2] + half_error),
|
||||
normalizer_guard.clip(rg_buf[1] + rg_buf[3] + half_error),
|
||||
normalizer_guard.clip(ba_buf[0] + ba_buf[2] + half_error),
|
||||
normalizer_guard.clip(ba_buf[1] + ba_buf[3] + half_error),
|
||||
normalizer.clip(rg_buf[0] + rg_buf[2] + half_error),
|
||||
normalizer.clip(rg_buf[1] + rg_buf[3] + half_error),
|
||||
normalizer.clip(ba_buf[0] + ba_buf[2] + half_error),
|
||||
normalizer.clip(ba_buf[1] + ba_buf[3] + half_error),
|
||||
];
|
||||
}
|
||||
}
|
||||
|
||||
@@ -9,11 +9,9 @@ pub(crate) fn horiz_convolution(
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard32::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let initial: i64 = 1 << (precision - 1);
|
||||
|
||||
let src_rows = src_image.iter_rows(offset);
|
||||
@@ -29,7 +27,7 @@ pub(crate) fn horiz_convolution(
|
||||
}
|
||||
}
|
||||
for (i, s) in ss.iter().copied().enumerate() {
|
||||
dst_pixel.0[i] = normalizer_guard.clip(s);
|
||||
dst_pixel.0[i] = normalizer.clip(s);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -12,23 +12,15 @@ pub(crate) fn horiz_convolution(
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard32::new(values);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(
|
||||
src_rows,
|
||||
dst_rows,
|
||||
&coefficients_chunks,
|
||||
&normalizer_guard,
|
||||
);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -39,7 +31,7 @@ pub(crate) fn horiz_convolution(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
&coefficients_chunks,
|
||||
&normalizer_guard,
|
||||
&normalizer,
|
||||
);
|
||||
}
|
||||
yy += 1;
|
||||
@@ -57,13 +49,13 @@ unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: FourRows<U16x4>,
|
||||
dst_rows: FourRowsMut<U16x4>,
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer_guard: &optimisations::NormalizerGuard32,
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let (s_row0, s_row1, s_row2, s_row3) = src_rows;
|
||||
let s_rows = [s_row0, s_row1, s_row2, s_row3];
|
||||
let (d_row0, d_row1, d_row2, d_row3) = dst_rows;
|
||||
let d_rows = [d_row0, d_row1, d_row2, d_row3];
|
||||
let precision = normalizer_guard.precision();
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut rg_buf = [0i64; 2];
|
||||
let mut ba_buf = [0i64; 2];
|
||||
@@ -141,10 +133,10 @@ unsafe fn horiz_convolution_four_rows(
|
||||
_mm_storeu_si128((&mut ba_buf).as_mut_ptr() as *mut __m128i, ba_sum[i]);
|
||||
let dst_pixel = d_rows[i].get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = [
|
||||
normalizer_guard.clip(rg_buf[0]),
|
||||
normalizer_guard.clip(rg_buf[1]),
|
||||
normalizer_guard.clip(ba_buf[0]),
|
||||
normalizer_guard.clip(ba_buf[1]),
|
||||
normalizer.clip(rg_buf[0]),
|
||||
normalizer.clip(rg_buf[1]),
|
||||
normalizer.clip(ba_buf[0]),
|
||||
normalizer.clip(ba_buf[1]),
|
||||
];
|
||||
}
|
||||
}
|
||||
@@ -161,9 +153,9 @@ unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U16x4],
|
||||
dst_row: &mut [U16x4],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI32Chunk],
|
||||
normalizer_guard: &optimisations::NormalizerGuard32,
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let precision = normalizer_guard.precision();
|
||||
let precision = normalizer.precision();
|
||||
let half_error = 1i64 << (precision - 1);
|
||||
let mut rg_buf = [0i64; 2];
|
||||
let mut ba_buf = [0i64; 2];
|
||||
@@ -233,10 +225,10 @@ unsafe fn horiz_convolution_one_row(
|
||||
_mm_storeu_si128((&mut ba_buf).as_mut_ptr() as *mut __m128i, ba_sum);
|
||||
let dst_pixel = dst_row.get_unchecked_mut(dst_x);
|
||||
dst_pixel.0 = [
|
||||
normalizer_guard.clip(rg_buf[0]),
|
||||
normalizer_guard.clip(rg_buf[1]),
|
||||
normalizer_guard.clip(ba_buf[0]),
|
||||
normalizer_guard.clip(ba_buf[1]),
|
||||
normalizer.clip(rg_buf[0]),
|
||||
normalizer.clip(rg_buf[1]),
|
||||
normalizer.clip(ba_buf[0]),
|
||||
normalizer.clip(ba_buf[1]),
|
||||
];
|
||||
}
|
||||
}
|
||||
|
||||
@@ -12,18 +12,15 @@ pub(crate) fn horiz_convolution(
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_8u4x(src_rows, dst_rows, &coefficients_chunks, &normalizer_guard);
|
||||
horiz_convolution_8u4x(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -34,7 +31,7 @@ pub(crate) fn horiz_convolution(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
&coefficients_chunks,
|
||||
&normalizer_guard,
|
||||
&normalizer,
|
||||
);
|
||||
}
|
||||
yy += 1;
|
||||
@@ -53,13 +50,13 @@ unsafe fn horiz_convolution_8u4x(
|
||||
src_rows: FourRows<U8>,
|
||||
dst_rows: FourRowsMut<U8>,
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer_guard: &optimisations::NormalizerGuard16,
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
let s_rows = [src_rows.0, src_rows.1, src_rows.2, src_rows.3];
|
||||
let d_rows = [dst_rows.0, dst_rows.1, dst_rows.2, dst_rows.3];
|
||||
let zero = _mm_setzero_si128();
|
||||
// 8 components will be added, use only 1/8 of the error
|
||||
let initial = _mm256_set1_epi32(1 << (normalizer_guard.precision() - 4));
|
||||
let initial = _mm256_set1_epi32(1 << (normalizer.precision() - 4));
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let coeffs = coeffs_chunk.values;
|
||||
@@ -106,7 +103,7 @@ unsafe fn horiz_convolution_8u4x(
|
||||
x += 1;
|
||||
}
|
||||
|
||||
let result_u8x4 = result_i32x4.map(|v| normalizer_guard.clip(v));
|
||||
let result_u8x4 = result_i32x4.map(|v| normalizer.clip(v));
|
||||
for i in 0..4 {
|
||||
d_rows[i].get_unchecked_mut(dst_x).0 = result_u8x4[i];
|
||||
}
|
||||
@@ -124,11 +121,11 @@ unsafe fn horiz_convolution_8u(
|
||||
src_row: &[U8],
|
||||
dst_row: &mut [U8],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer_guard: &optimisations::NormalizerGuard16,
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
let zero = _mm_setzero_si128();
|
||||
// 8 components will be added, use only 1/8 of the error
|
||||
let initial = _mm256_set1_epi32(1 << (normalizer_guard.precision() - 4));
|
||||
let initial = _mm256_set1_epi32(1 << (normalizer.precision() - 4));
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let coeffs = coeffs_chunk.values;
|
||||
@@ -170,7 +167,7 @@ unsafe fn horiz_convolution_8u(
|
||||
x += 1;
|
||||
}
|
||||
|
||||
dst_row.get_unchecked_mut(dst_x).0 = normalizer_guard.clip(result_i32);
|
||||
dst_row.get_unchecked_mut(dst_x).0 = normalizer.clip(result_i32);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -9,11 +9,9 @@ pub(crate) fn horiz_convolution(
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let initial = 1 << (precision - 1);
|
||||
|
||||
let src_rows = src_image.iter_rows(offset);
|
||||
@@ -28,7 +26,7 @@ pub(crate) fn horiz_convolution(
|
||||
for (&k, &src_pixel) in ks.iter().zip(src_pixels) {
|
||||
ss += src_pixel.0 as i32 * (k as i32);
|
||||
}
|
||||
dst_pixel.0 = unsafe { normalizer_guard.clip(ss) };
|
||||
dst_pixel.0 = unsafe { normalizer.clip(ss) };
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -12,11 +12,8 @@ pub(crate) fn horiz_convolution(
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
@@ -27,7 +24,7 @@ pub(crate) fn horiz_convolution(
|
||||
src_rows,
|
||||
dst_rows,
|
||||
&coefficients_chunks,
|
||||
&normalizer_guard,
|
||||
&normalizer,
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -39,7 +36,7 @@ pub(crate) fn horiz_convolution(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
&coefficients_chunks,
|
||||
&normalizer_guard,
|
||||
&normalizer,
|
||||
);
|
||||
}
|
||||
yy += 1;
|
||||
@@ -58,11 +55,11 @@ unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: FourRows<U8x2>,
|
||||
dst_rows: FourRowsMut<U8x2>,
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer_guard: &optimisations::NormalizerGuard16,
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
let (s_row0, s_row1, s_row2, s_row3) = src_rows;
|
||||
let (d_row0, d_row1, d_row2, d_row3) = dst_rows;
|
||||
let precision = normalizer_guard.precision();
|
||||
let precision = normalizer.precision();
|
||||
let initial = _mm256_set1_epi32(1 << (precision - 2));
|
||||
|
||||
/*
|
||||
@@ -192,13 +189,13 @@ unsafe fn horiz_convolution_four_rows(
|
||||
|
||||
let lo128 = _mm256_extracti128_si256::<0>(sss0);
|
||||
let hi128 = _mm256_extracti128_si256::<1>(sss0);
|
||||
set_dst_pixel(lo128, d_row0, dst_x, normalizer_guard);
|
||||
set_dst_pixel(hi128, d_row1, dst_x, normalizer_guard);
|
||||
set_dst_pixel(lo128, d_row0, dst_x, normalizer);
|
||||
set_dst_pixel(hi128, d_row1, dst_x, normalizer);
|
||||
|
||||
let lo128 = _mm256_extracti128_si256::<0>(sss1);
|
||||
let hi128 = _mm256_extracti128_si256::<1>(sss1);
|
||||
set_dst_pixel(lo128, d_row2, dst_x, normalizer_guard);
|
||||
set_dst_pixel(hi128, d_row3, dst_x, normalizer_guard);
|
||||
set_dst_pixel(lo128, d_row2, dst_x, normalizer);
|
||||
set_dst_pixel(hi128, d_row3, dst_x, normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -208,14 +205,14 @@ unsafe fn set_dst_pixel(
|
||||
raw: __m128i,
|
||||
d_row: &mut &mut [U8x2],
|
||||
dst_x: usize,
|
||||
normalizer_guard: &optimisations::NormalizerGuard16,
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
let l32x2 = _mm_extract_epi64::<0>(raw);
|
||||
let a32x2 = _mm_extract_epi64::<1>(raw);
|
||||
let l32 = ((l32x2 >> 32) as i32).saturating_add((l32x2 & 0xffffffff) as i32);
|
||||
let a32 = ((a32x2 >> 32) as i32).saturating_add((a32x2 & 0xffffffff) as i32);
|
||||
let l8 = normalizer_guard.clip(l32);
|
||||
let a8 = normalizer_guard.clip(a32);
|
||||
let l8 = normalizer.clip(l32);
|
||||
let a8 = normalizer.clip(a32);
|
||||
d_row.get_unchecked_mut(dst_x).0 = u16::from_le_bytes([l8, a8]);
|
||||
}
|
||||
|
||||
@@ -230,9 +227,9 @@ unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U8x2],
|
||||
dst_row: &mut [U8x2],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer_guard: &optimisations::NormalizerGuard16,
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
let precision = normalizer_guard.precision();
|
||||
let precision = normalizer.precision();
|
||||
/*
|
||||
|L A | |L A | |L A | |L A | |L A | |L A | |L A | |L A |
|
||||
|00 01| |02 03| |04 05| |06 07| |08 09| |10 11| |12 13| |14 15|
|
||||
@@ -421,8 +418,8 @@ unsafe fn horiz_convolution_one_row(
|
||||
|
||||
let a32 = ((lo >> 32) as i32).saturating_add((hi >> 32) as i32);
|
||||
let l32 = ((lo & 0xffffffff) as i32).saturating_add((hi & 0xffffffff) as i32);
|
||||
let a8 = normalizer_guard.clip(a32);
|
||||
let l8 = normalizer_guard.clip(l32);
|
||||
let a8 = normalizer.clip(a32);
|
||||
let l8 = normalizer.clip(l32);
|
||||
dst_row.get_unchecked_mut(dst_x).0 = u16::from_le_bytes([l8, a8]);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -8,11 +8,9 @@ pub(crate) fn horiz_convolution(
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let initial = 1 << (precision - 1);
|
||||
|
||||
let src_rows = src_image.iter_rows(offset);
|
||||
@@ -29,7 +27,7 @@ pub(crate) fn horiz_convolution(
|
||||
*s += components[i] as i32 * (k as i32);
|
||||
}
|
||||
}
|
||||
dst_pixel.0 = u16::from_le_bytes(ss.map(|v| unsafe { normalizer_guard.clip(v) }));
|
||||
dst_pixel.0 = u16::from_le_bytes(ss.map(|v| unsafe { normalizer.clip(v) }));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -12,23 +12,15 @@ pub(crate) fn horiz_convolution(
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
let dst_iter = dst_image.iter_4_rows_mut();
|
||||
for (src_rows, dst_rows) in src_iter.zip(dst_iter) {
|
||||
unsafe {
|
||||
horiz_convolution_four_rows(
|
||||
src_rows,
|
||||
dst_rows,
|
||||
&coefficients_chunks,
|
||||
&normalizer_guard,
|
||||
);
|
||||
horiz_convolution_four_rows(src_rows, dst_rows, &coefficients_chunks, &normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -39,7 +31,7 @@ pub(crate) fn horiz_convolution(
|
||||
src_image.get_row(yy + offset).unwrap(),
|
||||
dst_image.get_row_mut(yy).unwrap(),
|
||||
&coefficients_chunks,
|
||||
&normalizer_guard,
|
||||
&normalizer,
|
||||
);
|
||||
}
|
||||
yy += 1;
|
||||
@@ -58,13 +50,13 @@ unsafe fn horiz_convolution_four_rows(
|
||||
src_rows: FourRows<U8x2>,
|
||||
dst_rows: FourRowsMut<U8x2>,
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer_guard: &optimisations::NormalizerGuard16,
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
let (s_row0, s_row1, s_row2, s_row3) = src_rows;
|
||||
let s_rows = [s_row0, s_row1, s_row2, s_row3];
|
||||
let (d_row0, d_row1, d_row2, d_row3) = dst_rows;
|
||||
let d_rows = [d_row0, d_row1, d_row2, d_row3];
|
||||
let precision = normalizer_guard.precision();
|
||||
let precision = normalizer.precision();
|
||||
let initial = _mm_set1_epi32(1 << (precision - 2));
|
||||
|
||||
/*
|
||||
@@ -151,7 +143,7 @@ unsafe fn horiz_convolution_four_rows(
|
||||
}
|
||||
|
||||
for i in 0..4 {
|
||||
set_dst_pixel(sss[i], d_rows[i], dst_x, normalizer_guard);
|
||||
set_dst_pixel(sss[i], d_rows[i], dst_x, normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -162,14 +154,14 @@ unsafe fn set_dst_pixel(
|
||||
raw: __m128i,
|
||||
d_row: &mut &mut [U8x2],
|
||||
dst_x: usize,
|
||||
normalizer_guard: &optimisations::NormalizerGuard16,
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
let l32x2 = _mm_extract_epi64::<0>(raw);
|
||||
let a32x2 = _mm_extract_epi64::<1>(raw);
|
||||
let l32 = ((l32x2 >> 32) as i32).saturating_add((l32x2 & 0xffffffff) as i32);
|
||||
let a32 = ((a32x2 >> 32) as i32).saturating_add((a32x2 & 0xffffffff) as i32);
|
||||
let l8 = normalizer_guard.clip(l32);
|
||||
let a8 = normalizer_guard.clip(a32);
|
||||
let l8 = normalizer.clip(l32);
|
||||
let a8 = normalizer.clip(a32);
|
||||
d_row.get_unchecked_mut(dst_x).0 = u16::from_le_bytes([l8, a8]);
|
||||
}
|
||||
|
||||
@@ -184,9 +176,9 @@ unsafe fn horiz_convolution_one_row(
|
||||
src_row: &[U8x2],
|
||||
dst_row: &mut [U8x2],
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer_guard: &optimisations::NormalizerGuard16,
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
let precision = normalizer_guard.precision();
|
||||
let precision = normalizer.precision();
|
||||
/*
|
||||
|L A | |L A | |L A | |L A | |L A | |L A | |L A | |L A |
|
||||
|00 01| |02 03| |04 05| |06 07| |08 09| |10 11| |12 13| |14 15|
|
||||
@@ -325,8 +317,8 @@ unsafe fn horiz_convolution_one_row(
|
||||
|
||||
let a32 = ((lo >> 32) as i32).saturating_add((hi >> 32) as i32);
|
||||
let l32 = ((lo & 0xffffffff) as i32).saturating_add((hi & 0xffffffff) as i32);
|
||||
let a8 = normalizer_guard.clip(a32);
|
||||
let l8 = normalizer_guard.clip(l32);
|
||||
let a8 = normalizer.clip(a32);
|
||||
let l8 = normalizer.clip(l32);
|
||||
dst_row.get_unchecked_mut(dst_x).0 = u16::from_le_bytes([l8, a8]);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -13,12 +13,9 @@ pub(crate) fn horiz_convolution(
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
|
||||
@@ -9,11 +9,9 @@ pub(crate) fn horiz_convolution(
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let initial = 1 << (precision - 1);
|
||||
|
||||
let src_rows = src_image.iter_rows(offset);
|
||||
@@ -29,7 +27,7 @@ pub(crate) fn horiz_convolution(
|
||||
}
|
||||
}
|
||||
for (i, s) in ss.iter().copied().enumerate() {
|
||||
dst_pixel.0[i] = unsafe { normalizer_guard.clip(s) };
|
||||
dst_pixel.0[i] = unsafe { normalizer.clip(s) };
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -16,12 +16,9 @@ pub(crate) fn horiz_convolution(
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
|
||||
@@ -8,11 +8,9 @@ pub(crate) fn horiz_convolution(
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let initial = 1 << (precision - 1);
|
||||
|
||||
let src_rows = src_image.iter_rows(offset);
|
||||
@@ -29,7 +27,7 @@ pub(crate) fn horiz_convolution(
|
||||
*s += components[i] as i32 * (k as i32);
|
||||
}
|
||||
}
|
||||
dst_pixel.0 = u32::from_le_bytes(ss.map(|v| unsafe { normalizer_guard.clip(v) }));
|
||||
dst_pixel.0 = u32::from_le_bytes(ss.map(|v| unsafe { normalizer.clip(v) }));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -16,12 +16,9 @@ pub(crate) fn horiz_convolution(
|
||||
offset: u32,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
|
||||
let precision = normalizer_guard.precision();
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let precision = normalizer.precision();
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let dst_height = dst_image.height().get();
|
||||
|
||||
let src_iter = src_image.iter_4_rows(offset, dst_height + offset);
|
||||
|
||||
@@ -12,17 +12,13 @@ pub(crate) fn vert_convolution<T>(
|
||||
) where
|
||||
T: Pixel<Component = u16>,
|
||||
{
|
||||
// native::vert_convolution(src_image, dst_image, coeffs);
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard32::new(values);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for (dst_row, coeffs_chunk) in dst_rows.zip(coefficients_chunks) {
|
||||
unsafe {
|
||||
vert_convolution_into_one_row_u16(&src_image, dst_row, coeffs_chunk, &normalizer_guard);
|
||||
vert_convolution_into_one_row_u16(&src_image, dst_row, coeffs_chunk, &normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -32,7 +28,7 @@ pub(crate) unsafe fn vert_convolution_into_one_row_u16<T>(
|
||||
src_img: &TypedImageView<T>,
|
||||
dst_row: &mut [T],
|
||||
coeffs_chunk: optimisations::CoefficientsI32Chunk,
|
||||
normalizer_guard: &optimisations::NormalizerGuard32,
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) where
|
||||
T: Pixel<Component = u16>,
|
||||
{
|
||||
@@ -86,7 +82,7 @@ pub(crate) unsafe fn vert_convolution_into_one_row_u16<T>(
|
||||
),
|
||||
];
|
||||
|
||||
let precision = normalizer_guard.precision();
|
||||
let precision = normalizer.precision();
|
||||
let initial = _mm256_set1_epi64x(1 << (precision - 1));
|
||||
let mut comp_buf = [0i64; 4];
|
||||
|
||||
@@ -108,13 +104,13 @@ pub(crate) unsafe fn vert_convolution_into_one_row_u16<T>(
|
||||
for i in 0..4 {
|
||||
_mm256_storeu_si256((&mut comp_buf).as_mut_ptr() as *mut __m256i, sum[i]);
|
||||
let component = dst_components.get_unchecked_mut(xx + i * 2);
|
||||
*component = normalizer_guard.clip(comp_buf[0]);
|
||||
*component = normalizer.clip(comp_buf[0]);
|
||||
let component = dst_components.get_unchecked_mut(xx + i * 2 + 1);
|
||||
*component = normalizer_guard.clip(comp_buf[1]);
|
||||
*component = normalizer.clip(comp_buf[1]);
|
||||
let component = dst_components.get_unchecked_mut(xx + i * 2 + 8);
|
||||
*component = normalizer_guard.clip(comp_buf[2]);
|
||||
*component = normalizer.clip(comp_buf[2]);
|
||||
let component = dst_components.get_unchecked_mut(xx + i * 2 + 9);
|
||||
*component = normalizer_guard.clip(comp_buf[3]);
|
||||
*component = normalizer.clip(comp_buf[3]);
|
||||
}
|
||||
|
||||
xx += 16;
|
||||
@@ -141,13 +137,13 @@ pub(crate) unsafe fn vert_convolution_into_one_row_u16<T>(
|
||||
for i in 0..4 {
|
||||
_mm256_storeu_si256((&mut comp_buf).as_mut_ptr() as *mut __m256i, sum[i]);
|
||||
let component = buf.get_unchecked_mut(i * 2);
|
||||
*component = normalizer_guard.clip(comp_buf[0]);
|
||||
*component = normalizer.clip(comp_buf[0]);
|
||||
let component = buf.get_unchecked_mut(i * 2 + 1);
|
||||
*component = normalizer_guard.clip(comp_buf[1]);
|
||||
*component = normalizer.clip(comp_buf[1]);
|
||||
let component = buf.get_unchecked_mut(i * 2 + 8);
|
||||
*component = normalizer_guard.clip(comp_buf[2]);
|
||||
*component = normalizer.clip(comp_buf[2]);
|
||||
let component = buf.get_unchecked_mut(i * 2 + 9);
|
||||
*component = normalizer_guard.clip(comp_buf[3]);
|
||||
*component = normalizer.clip(comp_buf[3]);
|
||||
}
|
||||
for (i, v) in dst_components
|
||||
.get_unchecked_mut(xx..)
|
||||
|
||||
@@ -12,10 +12,9 @@ pub(crate) fn vert_convolution<T: Pixel<Component = u16>>(
|
||||
debug_assert_eq!(src_image.width(), dst_image.width());
|
||||
debug_assert_eq!(coeffs.bounds.len(), dst_image.height().get() as usize);
|
||||
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
let normalizer_guard = optimisations::NormalizerGuard32::new(values);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
|
||||
let precision = normalizer_guard.precision();
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let precision = normalizer.precision();
|
||||
let initial: i64 = 1 << (precision - 1);
|
||||
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
@@ -27,7 +26,7 @@ pub(crate) fn vert_convolution<T: Pixel<Component = u16>>(
|
||||
|
||||
convolution_by_u16(
|
||||
&src_image,
|
||||
&normalizer_guard,
|
||||
&normalizer,
|
||||
initial,
|
||||
dst_components,
|
||||
0,
|
||||
@@ -40,7 +39,7 @@ pub(crate) fn vert_convolution<T: Pixel<Component = u16>>(
|
||||
#[inline(always)]
|
||||
pub(crate) fn convolution_by_u16<T: Pixel<Component = u16>>(
|
||||
src_image: &TypedImageView<T>,
|
||||
normalizer_guard: &optimisations::NormalizerGuard32,
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
initial: i64,
|
||||
dst_components: &mut [u16],
|
||||
mut x_src: usize,
|
||||
@@ -55,7 +54,7 @@ pub(crate) fn convolution_by_u16<T: Pixel<Component = u16>>(
|
||||
let src_component = unsafe { *src_ptr.add(x_src as usize) };
|
||||
ss += src_component as i64 * (k as i64);
|
||||
}
|
||||
*dst_component = normalizer_guard.clip(ss);
|
||||
*dst_component = normalizer.clip(ss);
|
||||
x_src += 1
|
||||
}
|
||||
x_src
|
||||
|
||||
@@ -12,17 +12,13 @@ pub(crate) fn vert_convolution<T: Pixel<Component = u16>>(
|
||||
mut dst_image: TypedImageViewMut<T>,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
// native::vert_convolution(src_image, dst_image, coeffs);
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard32::new(values);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
let normalizer = optimisations::Normalizer32::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for (dst_row, coeffs_chunk) in dst_rows.zip(coefficients_chunks) {
|
||||
unsafe {
|
||||
vert_convolution_into_one_row_u16(&src_image, dst_row, coeffs_chunk, &normalizer_guard);
|
||||
vert_convolution_into_one_row_u16(&src_image, dst_row, coeffs_chunk, &normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -32,7 +28,7 @@ unsafe fn vert_convolution_into_one_row_u16<T: Pixel<Component = u16>>(
|
||||
src_img: &TypedImageView<T>,
|
||||
dst_row: &mut [T],
|
||||
coeffs_chunk: CoefficientsI32Chunk,
|
||||
normalizer_guard: &optimisations::NormalizerGuard32,
|
||||
normalizer: &optimisations::Normalizer32,
|
||||
) {
|
||||
let mut xx: usize = 0;
|
||||
let src_width = src_img.width().get() as usize * T::components_count();
|
||||
@@ -69,7 +65,7 @@ unsafe fn vert_convolution_into_one_row_u16<T: Pixel<Component = u16>>(
|
||||
),
|
||||
];
|
||||
|
||||
let precision = normalizer_guard.precision();
|
||||
let precision = normalizer.precision();
|
||||
let initial = _mm_set1_epi64x(1 << (precision - 1));
|
||||
let mut c_buf = [0i64; 2];
|
||||
|
||||
@@ -114,9 +110,9 @@ unsafe fn vert_convolution_into_one_row_u16<T: Pixel<Component = u16>>(
|
||||
for x in 0..2 {
|
||||
for sum in sums {
|
||||
_mm_storeu_si128((&mut c_buf).as_mut_ptr() as *mut __m128i, sum[x]);
|
||||
*dst_ptr_u16 = normalizer_guard.clip(c_buf[0]);
|
||||
*dst_ptr_u16 = normalizer.clip(c_buf[0]);
|
||||
dst_ptr_u16 = dst_ptr_u16.add(1);
|
||||
*dst_ptr_u16 = normalizer_guard.clip(c_buf[1]);
|
||||
*dst_ptr_u16 = normalizer.clip(c_buf[1]);
|
||||
dst_ptr_u16 = dst_ptr_u16.add(1);
|
||||
}
|
||||
}
|
||||
@@ -166,9 +162,9 @@ unsafe fn vert_convolution_into_one_row_u16<T: Pixel<Component = u16>>(
|
||||
// sums[i] = _mm_srl_epi64(sums[i] , precision_i64);
|
||||
// _mm_packus_epi32(sums[i] , sums[i] );
|
||||
_mm_storeu_si128((&mut c_buf).as_mut_ptr() as *mut __m128i, sum);
|
||||
*dst_ptr_u16 = normalizer_guard.clip(c_buf[0]);
|
||||
*dst_ptr_u16 = normalizer.clip(c_buf[0]);
|
||||
dst_ptr_u16 = dst_ptr_u16.add(1);
|
||||
*dst_ptr_u16 = normalizer_guard.clip(c_buf[1]);
|
||||
*dst_ptr_u16 = normalizer.clip(c_buf[1]);
|
||||
dst_ptr_u16 = dst_ptr_u16.add(1);
|
||||
}
|
||||
|
||||
@@ -212,14 +208,14 @@ unsafe fn vert_convolution_into_one_row_u16<T: Pixel<Component = u16>>(
|
||||
}
|
||||
|
||||
_mm_storeu_si128((&mut c_buf).as_mut_ptr() as *mut __m128i, c01);
|
||||
*dst_ptr_u16 = normalizer_guard.clip(c_buf[0]);
|
||||
*dst_ptr_u16 = normalizer.clip(c_buf[0]);
|
||||
dst_ptr_u16 = dst_ptr_u16.add(1);
|
||||
*dst_ptr_u16 = normalizer_guard.clip(c_buf[1]);
|
||||
*dst_ptr_u16 = normalizer.clip(c_buf[1]);
|
||||
dst_ptr_u16 = dst_ptr_u16.add(1);
|
||||
_mm_storeu_si128((&mut c_buf).as_mut_ptr() as *mut __m128i, c23);
|
||||
*dst_ptr_u16 = normalizer_guard.clip(c_buf[0]);
|
||||
*dst_ptr_u16 = normalizer.clip(c_buf[0]);
|
||||
dst_ptr_u16 = dst_ptr_u16.add(1);
|
||||
*dst_ptr_u16 = normalizer_guard.clip(c_buf[1]);
|
||||
*dst_ptr_u16 = normalizer.clip(c_buf[1]);
|
||||
dst_ptr_u16 = dst_ptr_u16.add(1);
|
||||
|
||||
xx += 4;
|
||||
@@ -229,7 +225,7 @@ unsafe fn vert_convolution_into_one_row_u16<T: Pixel<Component = u16>>(
|
||||
let initial = 1 << (precision - 1);
|
||||
convolution_by_u16(
|
||||
src_img,
|
||||
normalizer_guard,
|
||||
normalizer,
|
||||
initial,
|
||||
dst_components,
|
||||
xx,
|
||||
|
||||
@@ -13,16 +13,13 @@ pub(crate) fn vert_convolution<T>(
|
||||
) where
|
||||
T: Pixel<Component = u8>,
|
||||
{
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for (dst_row, coeffs_chunk) in dst_rows.zip(coefficients_chunks) {
|
||||
unsafe {
|
||||
vert_convolution_into_one_row_u8(&src_image, dst_row, coeffs_chunk, &normalizer_guard);
|
||||
vert_convolution_into_one_row_u8(&src_image, dst_row, coeffs_chunk, &normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -33,7 +30,7 @@ unsafe fn vert_convolution_into_one_row_u8<T>(
|
||||
src_img: &TypedImageView<T>,
|
||||
dst_row: &mut [T],
|
||||
coeffs_chunk: optimisations::CoefficientsI16Chunk,
|
||||
normalizer_guard: &optimisations::NormalizerGuard16,
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) where
|
||||
T: Pixel<Component = u8>,
|
||||
{
|
||||
@@ -41,7 +38,7 @@ unsafe fn vert_convolution_into_one_row_u8<T>(
|
||||
let y_start = coeffs_chunk.start;
|
||||
let coeffs = coeffs_chunk.values;
|
||||
let max_y = y_start + coeffs.len() as u32;
|
||||
let precision = normalizer_guard.precision();
|
||||
let precision = normalizer.precision();
|
||||
|
||||
let initial = _mm_set1_epi32(1 << (precision - 1));
|
||||
let initial_256 = _mm256_set1_epi32(1 << (precision - 1));
|
||||
@@ -234,7 +231,7 @@ unsafe fn vert_convolution_into_one_row_u8<T>(
|
||||
ss0 += src_component as i32 * (k as i32);
|
||||
}
|
||||
}
|
||||
*dst_pixel = normalizer_guard.clip(ss0);
|
||||
*dst_pixel = normalizer.clip(ss0);
|
||||
x_in_bytes += 1;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -14,10 +14,9 @@ pub(crate) fn vert_convolution<T>(
|
||||
debug_assert_eq!(src_image.width(), dst_image.width());
|
||||
debug_assert_eq!(coeffs.bounds.len(), dst_image.height().get() as usize);
|
||||
|
||||
let (values, window_size, bounds) = (coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds);
|
||||
let precision = normalizer_guard.precision();
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
let precision = normalizer.precision();
|
||||
let initial = 1 << (precision - 1);
|
||||
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
@@ -32,7 +31,7 @@ pub(crate) fn vert_convolution<T>(
|
||||
if !head.is_empty() {
|
||||
x_src = convolution_by_u8(
|
||||
&src_image,
|
||||
&normalizer_guard,
|
||||
&normalizer,
|
||||
initial,
|
||||
head,
|
||||
x_src,
|
||||
@@ -57,14 +56,14 @@ pub(crate) fn vert_convolution<T>(
|
||||
*s += c as i32 * (k as i32);
|
||||
}
|
||||
}
|
||||
*dst_chunk = u32::from_le_bytes(ss.map(|v| unsafe { normalizer_guard.clip(v) }));
|
||||
*dst_chunk = u32::from_le_bytes(ss.map(|v| unsafe { normalizer.clip(v) }));
|
||||
x_src += 4;
|
||||
}
|
||||
|
||||
if !tail.is_empty() {
|
||||
convolution_by_u8(
|
||||
&src_image,
|
||||
&normalizer_guard,
|
||||
&normalizer,
|
||||
initial,
|
||||
tail,
|
||||
x_src,
|
||||
@@ -78,7 +77,7 @@ pub(crate) fn vert_convolution<T>(
|
||||
#[inline(always)]
|
||||
fn convolution_by_u8<T>(
|
||||
src_image: &TypedImageView<T>,
|
||||
normalizer_guard: &optimisations::NormalizerGuard16,
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
initial: i32,
|
||||
dst_components: &mut [u8],
|
||||
mut x_src: usize,
|
||||
@@ -96,7 +95,7 @@ where
|
||||
let src_component = unsafe { *src_ptr.add(x_src as usize) };
|
||||
ss += src_component as i32 * (k as i32);
|
||||
}
|
||||
*dst_component = unsafe { normalizer_guard.clip(ss) };
|
||||
*dst_component = unsafe { normalizer.clip(ss) };
|
||||
x_src += 1
|
||||
}
|
||||
x_src
|
||||
|
||||
@@ -11,16 +11,13 @@ pub(crate) fn vert_convolution<T: Pixel<Component = u8>>(
|
||||
mut dst_image: TypedImageViewMut<T>,
|
||||
coeffs: Coefficients,
|
||||
) {
|
||||
let (values, window_size, bounds_per_pixel) =
|
||||
(coeffs.values, coeffs.window_size, coeffs.bounds);
|
||||
|
||||
let normalizer_guard = optimisations::NormalizerGuard16::new(values);
|
||||
let coefficients_chunks = normalizer_guard.normalized_chunks(window_size, &bounds_per_pixel);
|
||||
let normalizer = optimisations::Normalizer16::new(coeffs);
|
||||
let coefficients_chunks = normalizer.normalized_chunks();
|
||||
|
||||
let dst_rows = dst_image.iter_rows_mut();
|
||||
for (dst_row, coeffs_chunk) in dst_rows.zip(coefficients_chunks) {
|
||||
unsafe {
|
||||
vert_convolution_into_one_row_u8(&src_image, dst_row, coeffs_chunk, &normalizer_guard);
|
||||
vert_convolution_into_one_row_u8(&src_image, dst_row, coeffs_chunk, &normalizer);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -30,14 +27,14 @@ pub(crate) unsafe fn vert_convolution_into_one_row_u8<T: Pixel<Component = u8>>(
|
||||
src_img: &TypedImageView<T>,
|
||||
dst_row: &mut [T],
|
||||
coeffs_chunk: optimisations::CoefficientsI16Chunk,
|
||||
normalizer_guard: &optimisations::NormalizerGuard16,
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
let mut xx: usize = 0;
|
||||
let src_width = src_img.width().get() as usize * T::components_count();
|
||||
let y_start = coeffs_chunk.start;
|
||||
let coeffs = coeffs_chunk.values;
|
||||
let max_y = y_start + coeffs.len() as u32;
|
||||
let precision = normalizer_guard.precision();
|
||||
let precision = normalizer.precision();
|
||||
let dst_ptr_u8 = T::components_mut(dst_row).as_mut_ptr() as *mut u8;
|
||||
|
||||
let initial = _mm_set1_epi32(1 << (precision - 1));
|
||||
@@ -264,7 +261,7 @@ pub(crate) unsafe fn vert_convolution_into_one_row_u8<T: Pixel<Component = u8>>(
|
||||
ss0 += src_component as i32 * (k as i32);
|
||||
}
|
||||
}
|
||||
*dst_pixel = normalizer_guard.clip(ss0);
|
||||
*dst_pixel = normalizer.clip(ss0);
|
||||
xx += 1;
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user