mirror of
https://github.com/Cykooz/fast_image_resize.git
synced 2026-10-08 01:11:09 +00:00
Slightly improved speed of Convolution implementation for U8x2 images and Wasm32 SIMD128 instructions.
This commit is contained in:
@@ -1,3 +1,8 @@
|
||||
## [Unreleased] - ReleaseDate
|
||||
|
||||
- Slightly improved speed of `Convolution` implementation for `U8x2` images
|
||||
and `Wasm32 SIMD128` instructions.
|
||||
|
||||
## [2.5.0] - 2023-01-29
|
||||
|
||||
### Crate
|
||||
|
||||
+5
-1
@@ -106,10 +106,14 @@ opt-level = 3
|
||||
opt-level = 3
|
||||
#incremental = true
|
||||
lto = true
|
||||
codegen-units = 1
|
||||
#codegen-units = 1
|
||||
strip = true
|
||||
|
||||
|
||||
[profile.release.package.fast_image_resize]
|
||||
codegen-units = 1
|
||||
|
||||
|
||||
[profile.test]
|
||||
opt-level = 3
|
||||
|
||||
|
||||
@@ -27,7 +27,7 @@ Supported pixel formats and available optimisations:
|
||||
|
||||
Resizer from this crate does not convert image into linear colorspace
|
||||
during resize process. If it is important for you to resize images with a
|
||||
non-linear color space (e.g. sRGB) correctly, then you hove to convert
|
||||
non-linear color space (e.g. sRGB) correctly, then you have to convert
|
||||
it to a linear color space before resizing and convert back to the color space of
|
||||
result image. [Read more](https://legacy.imagemagick.org/Usage/resize/#resize_colorspace)
|
||||
about resizing with respect to color space.
|
||||
|
||||
@@ -5,8 +5,8 @@ Environment:
|
||||
- CPU: AMD Ryzen 9 5950X
|
||||
- RAM: DDR4 3800 MHz
|
||||
- Ubuntu 22.04 (linux 5.15.0)
|
||||
- Rust 1.67
|
||||
- wasmtime = "5.0.0"
|
||||
- Rust 1.67.1
|
||||
- wasmtime = "6.0.0"
|
||||
- criterion = "0.4"
|
||||
- fast_image_resize = "2.5.0"
|
||||
|
||||
|
||||
+53
-139
@@ -64,18 +64,12 @@ unsafe fn horiz_convolution_four_rows(
|
||||
A: |-1 07| |-1 05| |-1 03| |-1 01|
|
||||
L: |-1 06| |-1 04| |-1 02| |-1 00|
|
||||
*/
|
||||
#[rustfmt::skip]
|
||||
const SH1: v128 = i8x16(
|
||||
0, -1, 2, -1, 4, -1, 6, -1, 1, -1, 3, -1, 5, -1, 7, -1
|
||||
);
|
||||
const SH1: v128 = i8x16(0, -1, 2, -1, 4, -1, 6, -1, 1, -1, 3, -1, 5, -1, 7, -1);
|
||||
/*
|
||||
A: |-1 15| |-1 13| |-1 11| |-1 09|
|
||||
L: |-1 14| |-1 12| |-1 10| |-1 08|
|
||||
*/
|
||||
#[rustfmt::skip]
|
||||
const SH2: v128 = i8x16(
|
||||
8, -1, 10, -1, 12, -1, 14, -1, 9, -1, 11, -1, 13, -1, 15, -1
|
||||
);
|
||||
const SH2: v128 = i8x16(8, -1, 10, -1, 12, -1, 14, -1, 9, -1, 11, -1, 13, -1, 15, -1);
|
||||
|
||||
for (dst_x, coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x = coeffs_chunk.start as usize;
|
||||
@@ -102,7 +96,6 @@ unsafe fn horiz_convolution_four_rows(
|
||||
|
||||
let coeffs_by_4 = reminder.chunks_exact(4);
|
||||
let reminder = coeffs_by_4.remainder();
|
||||
|
||||
for k in coeffs_by_4 {
|
||||
let mmk = wasm32_utils::ptr_i16_to_set1_i64(k, 0);
|
||||
|
||||
@@ -116,7 +109,6 @@ unsafe fn horiz_convolution_four_rows(
|
||||
|
||||
let coeffs_by_2 = reminder.chunks_exact(2);
|
||||
let reminder = coeffs_by_2.remainder();
|
||||
|
||||
for k in coeffs_by_2 {
|
||||
let mmk = wasm32_utils::ptr_i16_to_set1_i32(k, 0);
|
||||
|
||||
@@ -144,23 +136,6 @@ unsafe fn horiz_convolution_four_rows(
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn set_dst_pixel(
|
||||
raw: v128,
|
||||
d_row: &mut &mut [U8x2],
|
||||
dst_x: usize,
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
let l32x2 = i64x2_extract_lane::<0>(raw);
|
||||
let a32x2 = i64x2_extract_lane::<1>(raw);
|
||||
let l32 = ((l32x2 >> 32) as i32).saturating_add((l32x2 & 0xffffffff) as i32);
|
||||
let a32 = ((a32x2 >> 32) as i32).saturating_add((a32x2 & 0xffffffff) as i32);
|
||||
let l8 = normalizer.clip(l32);
|
||||
let a8 = normalizer.clip(a32);
|
||||
d_row.get_unchecked_mut(dst_x).0 = u16::from_le_bytes([l8, a8]);
|
||||
}
|
||||
|
||||
/// For safety, it is necessary to ensure the following conditions:
|
||||
/// - bounds.len() == dst_row.len()
|
||||
/// - coeffs.len() == dst_rows.0.len() * window_size
|
||||
@@ -174,103 +149,35 @@ unsafe fn horiz_convolution_one_row(
|
||||
coefficients_chunks: &[optimisations::CoefficientsI16Chunk],
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
const SH1: v128 = i8x16(0, -1, 2, -1, 4, -1, 6, -1, 1, -1, 3, -1, 5, -1, 7, -1);
|
||||
/*
|
||||
A: |-1 15| |-1 13| |-1 11| |-1 09|
|
||||
L: |-1 14| |-1 12| |-1 10| |-1 08|
|
||||
*/
|
||||
const SH2: v128 = i8x16(8, -1, 10, -1, 12, -1, 14, -1, 9, -1, 11, -1, 13, -1, 15, -1);
|
||||
|
||||
// Lower part will be added to higher, use only half of the error
|
||||
let precision = normalizer.precision();
|
||||
/*
|
||||
|L A | |L A | |L A | |L A | |L A | |L A | |L A | |L A |
|
||||
|00 01| |02 03| |04 05| |06 07| |08 09| |10 11| |12 13| |14 15|
|
||||
|
||||
Scale first four pixels into i16:
|
||||
|
||||
A: |-1 07| |-1 05|
|
||||
L: |-1 06| |-1 04|
|
||||
A: |-1 03| |-1 01|
|
||||
L: |-1 02| |-1 00|
|
||||
*/
|
||||
#[rustfmt::skip]
|
||||
const PIX_SH1: v128 = i8x16(
|
||||
0, -1, 2, -1, 1, -1, 3, -1, 4, -1, 6, -1, 5, -1, 7, -1
|
||||
);
|
||||
/*
|
||||
|C0 | |C1 | |C2 | |C3 | |C4 | |C5 | |C6 | |C7 |
|
||||
|00 01| |02 03| |04 05| |06 07| |08 09| |10 11| |12 13| |14 15|
|
||||
|
||||
Duplicate first four coefficients for A and L components of pixels:
|
||||
|
||||
CA: |07 06| |05 04|
|
||||
CL: |07 06| |05 04|
|
||||
CA: |03 02| |01 00|
|
||||
CL: |03 02| |01 00|
|
||||
*/
|
||||
#[rustfmt::skip]
|
||||
const COEFF_SH1: v128 = i8x16(
|
||||
0, 1, 2, 3, 0, 1, 2, 3, 4, 5, 6, 7, 4, 5, 6, 7
|
||||
);
|
||||
|
||||
/*
|
||||
|L A | |L A | |L A | |L A | |L A | |L A | |L A | |L A |
|
||||
|00 01| |02 03| |04 05| |06 07| |08 09| |10 11| |12 13| |14 15|
|
||||
|
||||
Scale second four pixels into i16:
|
||||
|
||||
A: |-1 15| |-1 13|
|
||||
L: |-1 14| |-1 12|
|
||||
A: |-1 11| |-1 09|
|
||||
L: |-1 10| |-1 08|
|
||||
*/
|
||||
#[rustfmt::skip]
|
||||
const PIX_SH2: v128 = i8x16(
|
||||
8, -1, 10, -1, 9, -1, 11, -1, 12, -1, 14, -1, 13, -1, 15, -1
|
||||
);
|
||||
/*
|
||||
|C0 | |C1 | |C2 | |C3 | |C4 | |C5 | |C6 | |C7 |
|
||||
|00 01| |02 03| |04 05| |06 07| |08 09| |10 11| |12 13| |14 15|
|
||||
|
||||
Duplicate second four coefficients for A and L components of pixels:
|
||||
|
||||
CA: |15 14| |13 12|
|
||||
CL: |15 14| |13 12|
|
||||
CA: |11 10| |09 08|
|
||||
CL: |11 10| |09 08|
|
||||
*/
|
||||
#[rustfmt::skip]
|
||||
const COEFF_SH2: v128 = i8x16(
|
||||
8, 9, 10, 11, 8, 9, 10, 11, 12, 13, 14, 15, 12, 13, 14, 15
|
||||
);
|
||||
|
||||
/*
|
||||
|L A | |L A | |L A | |L A |
|
||||
|00 01| |02 03| |04 05| |06 07| |08 09| |10 11| |12 13| |14 15|
|
||||
|
||||
Scale four pixels into i16:
|
||||
|
||||
A: |-1 07| |-1 05|
|
||||
L: |-1 06| |-1 04|
|
||||
A: |-1 03| |-1 01|
|
||||
L: |-1 02| |-1 00|
|
||||
*/
|
||||
const PIX_SH3: v128 = i8x16(0, -1, 2, -1, 1, -1, 3, -1, 4, -1, 6, -1, 5, -1, 7, -1);
|
||||
let initial = i32x4_splat(1 << (precision - 2));
|
||||
|
||||
for (dst_x, &coeffs_chunk) in coefficients_chunks.iter().enumerate() {
|
||||
let mut x = coeffs_chunk.start as usize;
|
||||
let mut coeffs = coeffs_chunk.values;
|
||||
|
||||
// Lower part will be added to higher, use only half of the error
|
||||
let mut sss = i32x4_splat(1 << (precision - 2));
|
||||
let mut sss = initial;
|
||||
|
||||
let coeffs_by_8 = coeffs.chunks_exact(8);
|
||||
coeffs = coeffs_by_8.remainder();
|
||||
|
||||
for k in coeffs_by_8 {
|
||||
let ksource = wasm32_utils::load_v128(k, 0);
|
||||
let mmk0 = wasm32_utils::ptr_i16_to_set1_i64(k, 0);
|
||||
let mmk1 = wasm32_utils::ptr_i16_to_set1_i64(k, 4);
|
||||
|
||||
let source = wasm32_utils::load_v128(src_row, x);
|
||||
|
||||
let pix = i8x16_swizzle(source, PIX_SH1);
|
||||
let mmk = i8x16_swizzle(ksource, COEFF_SH1);
|
||||
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
let pix = i8x16_swizzle(source, PIX_SH2);
|
||||
let mmk = i8x16_swizzle(ksource, COEFF_SH2);
|
||||
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
|
||||
let pix = i8x16_swizzle(source, SH1);
|
||||
let tmp_sum = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk0));
|
||||
let pix = i8x16_swizzle(source, SH2);
|
||||
sss = i32x4_add(tmp_sum, i32x4_dot_i16x8(pix, mmk1));
|
||||
|
||||
x += 8;
|
||||
}
|
||||
@@ -279,42 +186,49 @@ unsafe fn horiz_convolution_one_row(
|
||||
let reminder1 = coeffs_by_4.remainder();
|
||||
|
||||
for k in coeffs_by_4 {
|
||||
let mmk = i16x8(k[0], k[1], k[0], k[1], k[2], k[3], k[2], k[3]);
|
||||
let mmk = wasm32_utils::ptr_i16_to_set1_i64(k, 0);
|
||||
let source = wasm32_utils::loadl_i64(src_row, x);
|
||||
let pix = i8x16_swizzle(source, PIX_SH3);
|
||||
let pix = i8x16_swizzle(source, SH1);
|
||||
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
x += 4
|
||||
}
|
||||
|
||||
if !reminder1.is_empty() {
|
||||
let mut pixels: [i16; 6] = [0; 6];
|
||||
let mut coeffs: [i16; 3] = [0; 3];
|
||||
for (i, &coeff) in reminder1.iter().enumerate() {
|
||||
coeffs[i] = coeff;
|
||||
let pixel: [u8; 2] = (*src_row.get_unchecked(x)).0.to_le_bytes();
|
||||
pixels[i * 2] = pixel[0] as i16;
|
||||
pixels[i * 2 + 1] = pixel[1] as i16;
|
||||
x += 1;
|
||||
}
|
||||
let coeffs_by_2 = reminder1.chunks_exact(2);
|
||||
let reminder = coeffs_by_2.remainder();
|
||||
for k in coeffs_by_2 {
|
||||
let mmk = wasm32_utils::ptr_i16_to_set1_i32(k, 0);
|
||||
|
||||
let pix = i16x8(
|
||||
pixels[0], pixels[2], pixels[1], pixels[3], pixels[4], 0, pixels[5], 0,
|
||||
);
|
||||
let mmk = i16x8(
|
||||
coeffs[0], coeffs[1], coeffs[0], coeffs[1], coeffs[2], 0, coeffs[2], 0,
|
||||
);
|
||||
let source = wasm32_utils::loadl_i32(src_row, x);
|
||||
let pix = i8x16_swizzle(source, SH1);
|
||||
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
|
||||
|
||||
x += 2;
|
||||
}
|
||||
|
||||
if let Some(&k) = reminder.first() {
|
||||
let mmk = i32x4_splat(k as i32);
|
||||
let source = wasm32_utils::loadl_i16(src_row, x);
|
||||
let pix = i8x16_swizzle(source, SH1);
|
||||
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
|
||||
}
|
||||
|
||||
let lo = i64x2_extract_lane::<0>(sss);
|
||||
let hi = i64x2_extract_lane::<1>(sss);
|
||||
|
||||
let a32 = ((lo >> 32) as i32).saturating_add((hi >> 32) as i32);
|
||||
let l32 = ((lo & 0xffffffff) as i32).saturating_add((hi & 0xffffffff) as i32);
|
||||
let a8 = normalizer.clip(a32);
|
||||
let l8 = normalizer.clip(l32);
|
||||
dst_row.get_unchecked_mut(dst_x).0 = u16::from_le_bytes([l8, a8]);
|
||||
set_dst_pixel(sss, dst_row, dst_x, normalizer);
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "simd128")]
|
||||
unsafe fn set_dst_pixel(
|
||||
raw: v128,
|
||||
d_row: &mut [U8x2],
|
||||
dst_x: usize,
|
||||
normalizer: &optimisations::Normalizer16,
|
||||
) {
|
||||
let mut buf = [0i32; 4];
|
||||
v128_store(buf.as_mut_ptr() as _, raw);
|
||||
let l32 = buf[0].saturating_add(buf[1]);
|
||||
let a32 = buf[2].saturating_add(buf[3]);
|
||||
let l8 = normalizer.clip(l32);
|
||||
let a8 = normalizer.clip(a32);
|
||||
d_row.get_unchecked_mut(dst_x).0 = u16::from_le_bytes([l8, a8]);
|
||||
}
|
||||
|
||||
@@ -26,7 +26,7 @@ pub(crate) fn vert_convolution<T: PixelExt<Component = u8>>(
|
||||
}
|
||||
|
||||
#[target_feature(enable = "sse4.1")]
|
||||
pub(crate) unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
|
||||
unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
|
||||
src_img: &ImageView<T>,
|
||||
dst_row: &mut [T],
|
||||
mut src_x: usize,
|
||||
|
||||
@@ -25,8 +25,9 @@ pub(crate) fn vert_convolution<T: PixelExt<Component = u8>>(
|
||||
}
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "simd128")]
|
||||
pub(crate) unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
|
||||
unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8>>(
|
||||
src_img: &ImageView<T>,
|
||||
dst_row: &mut [T],
|
||||
mut src_x: usize,
|
||||
@@ -37,7 +38,7 @@ pub(crate) unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8
|
||||
let y_start = coeffs_chunk.start;
|
||||
let coeffs = coeffs_chunk.values;
|
||||
let max_y = y_start + coeffs.len() as u32;
|
||||
let precision = normalizer.precision();
|
||||
let precision = normalizer.precision() as u32;
|
||||
let mut dst_u8 = T::components_mut(dst_row);
|
||||
|
||||
let initial = i32x4_splat(1 << (precision - 1));
|
||||
@@ -143,19 +144,14 @@ pub(crate) unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8
|
||||
sss7 = i32x4_add(sss7, i32x4_dot_i16x8(pix, mmk));
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss0 = i32x4_shr(sss0, $imm8);
|
||||
sss1 = i32x4_shr(sss1, $imm8);
|
||||
sss2 = i32x4_shr(sss2, $imm8);
|
||||
sss3 = i32x4_shr(sss3, $imm8);
|
||||
sss4 = i32x4_shr(sss4, $imm8);
|
||||
sss5 = i32x4_shr(sss5, $imm8);
|
||||
sss6 = i32x4_shr(sss6, $imm8);
|
||||
sss7 = i32x4_shr(sss7, $imm8);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
sss0 = i32x4_shr(sss0, precision);
|
||||
sss1 = i32x4_shr(sss1, precision);
|
||||
sss2 = i32x4_shr(sss2, precision);
|
||||
sss3 = i32x4_shr(sss3, precision);
|
||||
sss4 = i32x4_shr(sss4, precision);
|
||||
sss5 = i32x4_shr(sss5, precision);
|
||||
sss6 = i32x4_shr(sss6, precision);
|
||||
sss7 = i32x4_shr(sss7, precision);
|
||||
|
||||
sss0 = i16x8_narrow_i32x4(sss0, sss1);
|
||||
sss2 = i16x8_narrow_i32x4(sss2, sss3);
|
||||
@@ -214,13 +210,8 @@ pub(crate) unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8
|
||||
sss1 = i32x4_add(sss1, i32x4_dot_i16x8(pix, mmk));
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss0 = i32x4_shr(sss0, $imm8);
|
||||
sss1 = i32x4_shr(sss1, $imm8);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
sss0 = i32x4_shr(sss0, precision);
|
||||
sss1 = i32x4_shr(sss1, precision);
|
||||
|
||||
sss0 = i16x8_narrow_i32x4(sss0, sss1);
|
||||
sss0 = u8x16_narrow_i16x8(sss0, sss0);
|
||||
@@ -262,13 +253,7 @@ pub(crate) unsafe fn vert_convolution_into_one_row_u8<T: PixelExt<Component = u8
|
||||
sss = i32x4_add(sss, i32x4_dot_i16x8(pix, mmk));
|
||||
}
|
||||
|
||||
macro_rules! call {
|
||||
($imm8:expr) => {{
|
||||
sss = i32x4_shr(sss, $imm8);
|
||||
}};
|
||||
}
|
||||
constify_imm8!(precision, call);
|
||||
|
||||
sss = i32x4_shr(sss, precision);
|
||||
sss = i16x8_narrow_i32x4(sss, sss);
|
||||
let dst_ptr = dst_chunk.as_mut_ptr() as *mut i32;
|
||||
*dst_ptr = i32x4_extract_lane::<0>(u8x16_narrow_i16x8(sss, sss));
|
||||
|
||||
+10
-5
@@ -1,4 +1,5 @@
|
||||
use std::arch::wasm32::*;
|
||||
use std::ptr;
|
||||
|
||||
use crate::pixels::{U8x3, U8x4};
|
||||
|
||||
@@ -12,33 +13,37 @@ pub(crate) unsafe fn load_v128<T>(buf: &[T], index: usize) -> v128 {
|
||||
#[target_feature(enable = "simd128")]
|
||||
pub(crate) unsafe fn loadl_i64<T>(buf: &[T], index: usize) -> v128 {
|
||||
let i = buf.get_unchecked(index..).as_ptr() as *const i64;
|
||||
i64x2(*i, 0)
|
||||
i64x2(ptr::read_unaligned(i), 0)
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "simd128")]
|
||||
pub(crate) unsafe fn loadl_i32<T>(buf: &[T], index: usize) -> v128 {
|
||||
let i = buf.get_unchecked(index..).as_ptr() as *const i32;
|
||||
i32x4(*i, 0, 0, 0)
|
||||
i32x4(ptr::read_unaligned(i), 0, 0, 0)
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "simd128")]
|
||||
pub(crate) unsafe fn loadl_i16<T>(buf: &[T], index: usize) -> v128 {
|
||||
let i = buf.get_unchecked(index..).as_ptr() as *const i16;
|
||||
i16x8(*i, 0, 0, 0, 0, 0, 0, 0)
|
||||
i16x8(ptr::read_unaligned(i), 0, 0, 0, 0, 0, 0, 0)
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "simd128")]
|
||||
pub(crate) unsafe fn ptr_i16_to_set1_i64(buf: &[i16], index: usize) -> v128 {
|
||||
i64x2_splat(*(buf.get_unchecked(index..).as_ptr() as *const i64))
|
||||
i64x2_splat(ptr::read_unaligned(
|
||||
buf.get_unchecked(index..).as_ptr() as *const i64
|
||||
))
|
||||
}
|
||||
|
||||
#[inline]
|
||||
#[target_feature(enable = "simd128")]
|
||||
pub(crate) unsafe fn ptr_i16_to_set1_i32(buf: &[i16], index: usize) -> v128 {
|
||||
i32x4_splat(*(buf.get_unchecked(index..).as_ptr() as *const i32))
|
||||
i32x4_splat(ptr::read_unaligned(
|
||||
buf.get_unchecked(index..).as_ptr() as *const i32
|
||||
))
|
||||
}
|
||||
|
||||
#[inline]
|
||||
|
||||
@@ -303,8 +303,7 @@ where
|
||||
assert_eq!(
|
||||
testing::image_checksum::<Self, CC>(&result),
|
||||
checksum,
|
||||
"Error in checksum for {:?}",
|
||||
cpu_extensions
|
||||
"Error in checksum for {cpu_extensions:?}",
|
||||
);
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user