- Added benchmarks for libvips.

- Added Tera templates to build benchmark reports.
This commit is contained in:
Kirill Kuzminykh
2023-11-11 14:50:09 +03:00
parent 24edd65eef
commit 2ad679e044
29 changed files with 1106 additions and 500 deletions
Generated
+689 -390
View File
File diff suppressed because it is too large Load Diff
+9 -4
View File
@@ -23,24 +23,29 @@ exclude = ["/data"]
num-traits = "0.2"
thiserror = "1.0"
[features]
for_test = []
[dev-dependencies]
fast_image_resize = { path = ".", features = ["for_test"] }
image = "0.24"
resize = "0.7.4"
resize = "0.8"
rgb = "0.8"
png = "0.17"
serde = { version = "1.0", features = ["serde_derive"] }
serde_json = "1"
walkdir = "2"
itertools = "0.10"
criterion = { version = "0.4", default-features = false, features = ["cargo_bench_support"] }
itertools = "0.11"
criterion = { version = "0.5", default-features = false, features = ["cargo_bench_support"] }
tera = "1"
testing = { path = "testing" }
[target.'cfg(not(target_arch = "wasm32"))'.dev-dependencies]
nix = { version = "0.26", default-features = false, features = ["sched"] }
nix = { version = "0.27", default-features = false, features = ["sched"] }
libvips = "=1.4.3"
[[bench]]
+16 -10
View File
@@ -6,6 +6,11 @@
Rust library for fast image resizing with using of SIMD instructions.
| | Nearest | Box | Linear | Cubic | Lanczos3 |
|----------|:-------:|:-----:|:------:|:-----:|:--------:|
| libvips | 8.42 | 36.42 | 12.94 | 18.04 | 22.13 |
| fir avx2 | 0.86 | 19.53 | 21.28 | 27.35 | 38.65 |
[CHANGELOG](https://github.com/Cykooz/fast_image_resize/blob/main/CHANGELOG.md)
Supported pixel formats and available optimisations:
@@ -54,6 +59,7 @@ Rust libraries used to compare of resizing speed:
- image (<https://crates.io/crates/image>)
- resize (<https://crates.io/crates/resize>)
<!-- bench_compare_rgb start -->
### Resize RGB8 image (U8x3) 4928x3279 => 852x567
Pipeline:
@@ -63,16 +69,17 @@ Pipeline:
- Source image [nasa-4928x3279.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279.png)
- Numbers in table is mean duration of image resizing in milliseconds.
<!-- bench_compare_rgb start -->
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|------------|:-------:|:--------:|:----------:|:--------:|
| image | 19.37 | 81.71 | 149.63 | 205.67 |
| resize | - | 48.65 | 97.50 | 145.90 |
| fir rust | 0.28 | 38.21 | 65.73 | 96.99 |
| fir sse4.1 | - | 9.78 | 14.26 | 20.00 |
| fir avx2 | - | 7.85 | 9.96 | 14.75 |
| | Nearest | Box | Bilinear | Bicubic | Lanczos3 |
|------------|:-------:|:------:|:--------:|:-------:|:--------:|
| image | 100.67 | - | 247.87 | 410.70 | 579.80 |
| resize | - | 75.18 | 124.37 | 223.05 | 320.85 |
| libvips | 20.02 | 191.59 | 48.95 | 75.14 | 101.29 |
| fir rust | 0.85 | 58.66 | 98.47 | 166.57 | 229.08 |
| fir sse4.1 | - | 22.75 | 26.92 | 39.61 | 56.12 |
| fir avx2 | - | 20.40 | 21.61 | 28.31 | 40.84 |
<!-- bench_compare_rgb end -->
<!-- bench_compare_rgba start -->
### Resize RGBA8 image (U8x4) 4928x3279 => 852x567
Pipeline:
@@ -84,7 +91,6 @@ Pipeline:
- Numbers in table is mean duration of image resizing in milliseconds.
- The `image` crate does not support multiplying and dividing by alpha channel.
<!-- bench_compare_rgba start -->
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|------------|:-------:|:--------:|:----------:|:--------:|
| resize | - | 74.59 | 138.44 | 202.18 |
@@ -93,6 +99,7 @@ Pipeline:
| fir avx2 | - | 9.77 | 12.20 | 16.56 |
<!-- bench_compare_rgba end -->
<!-- bench_compare_l start -->
### Resize L8 image (U8) 4928x3279 => 852x567
Pipeline:
@@ -103,7 +110,6 @@ Pipeline:
has converted into grayscale image with one byte per pixel.
- Numbers in table is mean duration of image resizing in milliseconds.
<!-- bench_compare_l start -->
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|------------|:-------:|:--------:|:----------:|:--------:|
| image | 16.55 | 48.08 | 75.52 | 103.29 |
+1
View File
@@ -17,6 +17,7 @@ pub fn bench_compare_l(bench_group: &mut utils::BenchGroup) {
src_image.width(),
src_image.height(),
);
utils::libvips_resize::<P>(bench_group, false);
utils::fir_resize::<P>(bench_group);
}
+1
View File
@@ -17,6 +17,7 @@ pub fn bench_downscale_l16(bench_group: &mut utils::BenchGroup) {
src_image.width(),
src_image.height(),
);
utils::libvips_resize::<P>(bench_group, false);
utils::fir_resize::<P>(bench_group);
}
+3 -1
View File
@@ -3,7 +3,9 @@ use fast_image_resize::pixels::U8x2;
mod utils;
pub fn bench_downscale_la(bench_group: &mut utils::BenchGroup) {
utils::fir_resize_with_alpha::<U8x2>(bench_group);
type P = U8x2;
utils::libvips_resize::<P>(bench_group, true);
utils::fir_resize_with_alpha::<P>(bench_group);
}
fn main() {
+3 -1
View File
@@ -3,7 +3,9 @@ use fast_image_resize::pixels::U16x2;
mod utils;
pub fn bench_downscale_la16(bench_group: &mut utils::BenchGroup) {
utils::fir_resize_with_alpha::<U16x2>(bench_group);
type P = U16x2;
utils::libvips_resize::<P>(bench_group, true);
utils::fir_resize_with_alpha::<P>(bench_group);
}
fn main() {
+1
View File
@@ -17,6 +17,7 @@ pub fn bench_downscale_rgb(bench_group: &mut utils::BenchGroup) {
src_image.width(),
src_image.height(),
);
utils::libvips_resize::<P>(bench_group, false);
utils::fir_resize::<P>(bench_group);
}
+1
View File
@@ -17,6 +17,7 @@ pub fn bench_downscale_rgb16(bench_group: &mut utils::BenchGroup) {
src_image.width(),
src_image.height(),
);
utils::libvips_resize::<P>(bench_group, false);
utils::fir_resize::<P>(bench_group);
}
+1
View File
@@ -16,6 +16,7 @@ pub fn bench_downscale_rgba(bench_group: &mut utils::BenchGroup) {
src_image.width(),
src_image.height(),
);
utils::libvips_resize::<P>(bench_group, true);
utils::fir_resize_with_alpha::<P>(bench_group);
}
+1
View File
@@ -16,6 +16,7 @@ pub fn bench_downscale_rgba16(bench_group: &mut utils::BenchGroup) {
src_image.width(),
src_image.height(),
);
utils::libvips_resize::<P>(bench_group, true);
utils::fir_resize_with_alpha::<P>(bench_group);
}
+11
View File
@@ -0,0 +1,11 @@
### Resize L8 image (U8) 4928x3279 => 852x567
Pipeline:
`src_image => resize => dst_image`
- Source image [nasa-4928x3279.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279.png)
has converted into grayscale image with one byte per pixel.
- Numbers in table is mean duration of image resizing in milliseconds.
{{ compare_results -}}
@@ -0,0 +1,11 @@
### Resize L16 image (U16) 4928x3279 => 852x567
Pipeline:
`src_image => resize => dst_image`
- Source image [nasa-4928x3279.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279.png)
has converted into grayscale image with two bytes per pixel.
- Numbers in table is mean duration of image resizing in milliseconds.
{{ compare_results -}}
@@ -0,0 +1,14 @@
### Resize LA8 image (U8x2) 4928x3279 => 852x567
Pipeline:
`src_image => multiply by alpha => resize => divide by alpha => dst_image`
- Source image
[nasa-4928x3279-rgba.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279-rgba.png)
has converted into grayscale image with alpha channel (two bytes per pixel).
- Numbers in table is mean duration of image resizing in milliseconds.
- The `image` crate does not support multiplying and dividing by alpha channel.
- The `resize` crate does not support this pixel format.
{{ compare_results -}}
@@ -0,0 +1,14 @@
### Resize LA16 (luma with alpha channel) image (U16x2) 4928x3279 => 852x567
Pipeline:
`src_image => multiply by alpha => resize => divide by alpha => dst_image`
- Source image
[nasa-4928x3279-rgba.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279-rgba.png)
has converted into grayscale image with alpha channel (four bytes per pixel).
- Numbers in table is mean duration of image resizing in milliseconds.
- The `image` crate does not support multiplying and dividing by alpha channel.
- The `resize` crate does not support this pixel format.
{{ compare_results -}}
@@ -0,0 +1,10 @@
### Resize RGB8 image (U8x3) 4928x3279 => 852x567
Pipeline:
`src_image => resize => dst_image`
- Source image [nasa-4928x3279.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279.png)
- Numbers in table is mean duration of image resizing in milliseconds.
{{ compare_results -}}
@@ -0,0 +1,11 @@
### Resize RGB16 image (U16x3) 4928x3279 => 852x567
Pipeline:
`src_image => resize => dst_image`
- Source image [nasa-4928x3279.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279.png)
has converted into RGB16 image.
- Numbers in table is mean duration of image resizing in milliseconds.
{{ compare_results -}}
@@ -0,0 +1,12 @@
### Resize RGBA8 image (U8x4) 4928x3279 => 852x567
Pipeline:
`src_image => multiply by alpha => resize => divide by alpha => dst_image`
- Source image
[nasa-4928x3279-rgba.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279-rgba.png)
- Numbers in table is mean duration of image resizing in milliseconds.
- The `image` crate does not support multiplying and dividing by alpha channel.
{{ compare_results -}}
@@ -0,0 +1,12 @@
### Resize RGBA16 image (U16x4) 4928x3279 => 852x567
Pipeline:
`src_image => multiply by alpha => resize => divide by alpha => dst_image`
- Source image
[nasa-4928x3279-rgba.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279-rgba.png)
- Numbers in table is mean duration of image resizing in milliseconds.
- The `image` crate does not support multiplying and dividing by alpha channel.
{{ compare_results -}}
+33
View File
@@ -0,0 +1,33 @@
## Benchmarks of fast_image_resize crate for {{ arch_name }} architecture
Environment:
{% if arch_id == "arm64" -%}
- CPU: Neoverse-N1 2GHz (Oracle Cloud Compute, VM.Standard.A1.Flex)
{% else -%}
- CPU: AMD Ryzen 9 5950X
- RAM: DDR4 3800 MHz
{% endif -%}
- Ubuntu 22.04 (linux 6.2.0)
- Rust 1.73.0
- criterion = "0.5.1"
- fast_image_resize = "2.7.1"
{% if arch_id == "wasm32" -%}
- wasmtime = "8.0.0"
{% endif %}
Other libraries used to compare of resizing speed:
- image = "0.24.7" (<https://crates.io/crates/image>)
- resize = "0.8.2" (<https://crates.io/crates/resize>)
{% if arch_id != "wasm32" -%}
- libvips = "8.12.1" (single-threaded mode, cache disabled)
{% endif %}
Resize algorithms:
- Nearest
- Box - convolution with adaptive kernel size, min 1x1 px
- Bilinear - convolution with adaptive kernel size, min 2x2 px
- Bicubic (CatmullRom) - convolution with adaptive kernel size, min 4x4 px
- Lanczos3 -convolution with adaptive kernel size, min 6x6 px
+3 -3
View File
@@ -5,7 +5,7 @@ use std::time::SystemTime;
use criterion::measurement::WallTime;
use criterion::{Bencher, BenchmarkGroup, BenchmarkId, Criterion};
use super::{cargo_target_directory, get_arch_name, get_results, BenchResult};
use super::{cargo_target_directory, get_arch_id_and_name, get_results, BenchResult};
pub type BenchGroup<'a> = BenchmarkGroup<'a, WallTime>;
@@ -15,8 +15,8 @@ where
{
pin_process_to_cpu0();
let arch_name = get_arch_name();
let output_dir = criterion_output_directory().join(arch_name);
let arch_id = get_arch_id_and_name().0;
let output_dir = criterion_output_directory().join(arch_id);
let mut criterion = Criterion::default()
.output_directory(&output_dir)
.configure_from_args();
+5 -5
View File
@@ -12,19 +12,19 @@ mod bencher;
mod resize_functions;
mod results;
const fn get_arch_name() -> &'static str {
const fn get_arch_id_and_name() -> (&'static str, &'static str) {
#[cfg(target_arch = "x86_64")]
return "x86_64";
return ("x86_64", "x86_64");
#[cfg(target_arch = "aarch64")]
return "arm64";
return ("arm64", "arm64");
#[cfg(target_arch = "wasm32")]
return "wasm32";
return ("wasm32", "Wasm32");
#[cfg(not(any(
target_arch = "x86_64",
target_arch = "aarch64",
target_arch = "wasm32"
)))]
return "unknown";
return ("unknown", "Unknown");
}
/// Returns the Cargo target directory, possibly calling `cargo metadata` to
+120 -7
View File
@@ -1,12 +1,14 @@
use std::ops::Deref;
use criterion::black_box;
use image::{imageops, ImageBuffer};
use crate::utils::bencher::{bench, BenchGroup};
use fast_image_resize::{CpuExtensions, FilterType, Image, MulDiv, ResizeAlg, Resizer};
use testing::{cpu_ext_into_str, nonzero, PixelTestingExt};
const ALG_NAMES: [&str; 4] = ["Nearest", "Bilinear", "CatmullRom", "Lanczos3"];
use crate::utils::bencher::{bench, BenchGroup};
const ALG_NAMES: [&str; 5] = ["Nearest", "Box", "Bilinear", "Bicubic", "Lanczos3"];
const NEW_WIDTH: u32 = 852;
const NEW_HEIGHT: u32 = 567;
@@ -20,7 +22,7 @@ where
let (filter, sample_size) = match alg_name {
"Nearest" => (imageops::Nearest, 80),
"Bilinear" => (imageops::Triangle, 50),
"CatmullRom" => (imageops::CatmullRom, 30),
"Bicubic" => (imageops::CatmullRom, 30),
"Lanczos3" => (imageops::Lanczos3, 20),
_ => continue,
};
@@ -43,19 +45,25 @@ pub fn resize_resize<Format, Out>(
Out: Clone,
Format: resize::PixelFormat<OutputPixel = Out> + Copy,
{
fn box_kernel(_: f32) -> f32 {
1.0
}
for alg_name in ALG_NAMES {
if alg_name == "Nearest" {
// "resize" doesn't support "nearest" algorithm
continue;
}
let mut dst =
vec![pixel_format.into_pixel(Format::new()); (NEW_WIDTH * NEW_HEIGHT) as usize];
let sample_size = if alg_name == "Lanczos3" { 60 } else { 100 };
bench(bench_group, sample_size, "resize", alg_name, |bencher| {
let filter = match alg_name {
"Box" => resize::Type::Custom(resize::Filter::new(Box::new(box_kernel), 0.5)),
"Bilinear" => resize::Type::Triangle,
"CatmullRom" => resize::Type::Catrom,
"Bicubic" => resize::Type::Catrom,
"Lanczos3" => resize::Type::Lanczos3,
_ => return,
};
@@ -75,6 +83,109 @@ pub fn resize_resize<Format, Out>(
}
}
/// Resize image with help of "libvips" crate (https://crates.io/crates/libvips)
pub fn libvips_resize<P: PixelTestingExt>(bench_group: &mut BenchGroup, has_alpha: bool) {
#[cfg(not(target_arch = "wasm32"))]
vips::libvips_resize_inner::<P>(bench_group, has_alpha);
}
#[cfg(not(target_arch = "wasm32"))]
mod vips {
use libvips::ops::{self, BandFormat, Kernel, ReduceOptions};
use libvips::{VipsApp, VipsImage};
use super::*;
const SAMPLE_SIZE: usize = 100;
#[cfg(not(target_arch = "wasm32"))]
pub(crate) fn libvips_resize_inner<P: PixelTestingExt>(
bench_group: &mut BenchGroup,
has_alpha: bool,
) {
let app = VipsApp::new("Test Libvips", false).expect("Cannot initialize libvips");
app.concurrency_set(1);
app.cache_set_max(0);
app.cache_set_max_mem(0);
let src_image_data = P::load_big_src_image();
let src_width = src_image_data.width().get() as i32;
let src_height = src_image_data.height().get() as i32;
let band_format = if P::count_of_component_values() > 256 {
BandFormat::Ushort
} else {
BandFormat::Uchar
};
let src_vips_image = VipsImage::new_from_memory(
src_image_data.buffer(),
src_width,
src_height,
P::count_of_components() as i32,
band_format,
)
.unwrap();
let hshrink = src_width as f64 / NEW_WIDTH as f64;
let vshrink = src_height as f64 / NEW_HEIGHT as f64;
for alg_name in ALG_NAMES {
let kernel = match alg_name {
"Nearest" => Kernel::Nearest,
"Box" => {
bench(bench_group, SAMPLE_SIZE, "libvips", alg_name, |bencher| {
if has_alpha {
bencher.iter(|| {
let premultiplied = ops::premultiply(&src_vips_image).unwrap();
let resized =
ops::shrink(&premultiplied, hshrink, vshrink).unwrap();
let result = ops::unpremultiply(&resized).unwrap();
let result = ops::cast(&result, band_format).unwrap();
let res_bytes = result.image_write_to_memory();
black_box(&res_bytes);
})
} else {
bencher.iter(|| {
let result =
ops::shrink(&src_vips_image, hshrink, vshrink).unwrap();
let res_bytes = result.image_write_to_memory();
black_box(&res_bytes);
})
}
});
continue;
}
"Bilinear" => Kernel::Linear,
"Bicubic" => Kernel::Cubic,
"Lanczos3" => Kernel::Lanczos3,
_ => continue,
};
let options = ReduceOptions { kernel };
bench(bench_group, SAMPLE_SIZE, "libvips", alg_name, |bencher| {
if has_alpha {
bencher.iter(|| {
let premultiplied = ops::premultiply(&src_vips_image).unwrap();
let resized =
ops::reduce_with_opts(&premultiplied, hshrink, vshrink, &options)
.unwrap();
let result = ops::unpremultiply(&resized).unwrap();
let result = ops::cast(&result, band_format).unwrap();
let res_bytes = result.image_write_to_memory();
black_box(&res_bytes);
})
} else {
bencher.iter(|| {
let result =
ops::reduce_with_opts(&src_vips_image, hshrink, vshrink, &options)
.unwrap();
let res_bytes = result.image_write_to_memory();
black_box(&res_bytes);
})
}
});
}
}
}
/// Resize image with help of "fast_imager_resize" crate
pub fn fir_resize<P: PixelTestingExt>(bench_group: &mut BenchGroup) {
let src_image_data = P::load_big_src_image();
@@ -110,8 +221,9 @@ pub fn fir_resize<P: PixelTestingExt>(bench_group: &mut BenchGroup) {
}
ResizeAlg::Nearest
}
"Box" => ResizeAlg::Convolution(FilterType::Box),
"Bilinear" => ResizeAlg::Convolution(FilterType::Bilinear),
"CatmullRom" => ResizeAlg::Convolution(FilterType::CatmullRom),
"Bicubic" => ResizeAlg::Convolution(FilterType::CatmullRom),
"Lanczos3" => ResizeAlg::Convolution(FilterType::Lanczos3),
_ => continue,
};
@@ -175,8 +287,9 @@ pub fn fir_resize_with_alpha<P: PixelTestingExt>(bench_group: &mut BenchGroup) {
}
ResizeAlg::Nearest
}
"Box" => ResizeAlg::Convolution(FilterType::Box),
"Bilinear" => ResizeAlg::Convolution(FilterType::Bilinear),
"CatmullRom" => ResizeAlg::Convolution(FilterType::CatmullRom),
"Bicubic" => ResizeAlg::Convolution(FilterType::CatmullRom),
"Lanczos3" => ResizeAlg::Convolution(FilterType::Lanczos3),
_ => return,
};
@@ -212,7 +325,7 @@ pub fn fir_resize_with_alpha<P: PixelTestingExt>(bench_group: &mut BenchGroup) {
mul_div.divide_alpha_inplace(&mut dst_view).unwrap();
});
}
}
};
},
);
}
+40 -18
View File
@@ -8,7 +8,7 @@ use itertools::Itertools;
use serde::Deserialize;
use walkdir::WalkDir;
use super::get_arch_name;
use super::get_arch_id_and_name;
#[derive(Debug)]
pub struct BenchResult {
@@ -96,7 +96,7 @@ pub fn get_results(parent_dir: &PathBuf, modified_after: &SystemTime) -> Vec<Ben
result
}
static COL_ORDER: [&str; 4] = ["Nearest", "Bilinear", "CatmullRom", "Lanczos3"];
static COL_ORDER: [&str; 5] = ["Nearest", "Box", "Bilinear", "Bicubic", "Lanczos3"];
pub fn build_md_table(bench_results: &[BenchResult]) -> String {
let mut row_names: Vec<String> = Vec::new();
@@ -189,10 +189,9 @@ pub fn build_md_table(bench_results: &[BenchResult]) -> String {
fn table_row(buffer: &mut Vec<String>, widths: &[usize], values: &[String]) {
for (i, (&width, value)) in widths.iter().zip(values).enumerate() {
if i == 0 {
buffer.push(format!("| {:width$} ", value, width = width));
} else {
buffer.push(format!("| {:^width$} ", value, width = width));
match i {
0 => buffer.push(format!("| {:width$} ", value, width = width)),
_ => buffer.push(format!("| {:^width$} ", value, width = width)),
}
}
buffer.push("|\n".to_string());
@@ -200,10 +199,9 @@ fn table_row(buffer: &mut Vec<String>, widths: &[usize], values: &[String]) {
fn table_header_underline(buffer: &mut Vec<String>, widths: &[usize]) {
for (i, &width) in widths.iter().enumerate() {
if i == 0 {
buffer.push(format!("|{:-<width$}", "", width = width + 2));
} else {
buffer.push(format!("|:{:-<width$}:", "", width = width));
match i {
0 => buffer.push(format!("|{:-<width$}", "", width = width + 2)),
_ => buffer.push(format!("|:{:-<width$}:", "", width = width)),
}
}
buffer.push("|\n".to_string());
@@ -239,20 +237,44 @@ fn insert_string_into_file(path: &Path, placeholder_name: &str, string: &str) {
}
fn write_bench_results_into_file(md_table: &str) {
let file_name = format!("benchmarks-{}.md", get_arch_name());
let file_path = PathBuf::from(file_name);
if !file_path.is_file() {
panic!("Can't find file {:?} in current directory", file_path);
let (arch_id, arch_name) = get_arch_id_and_name();
let file_name = format!("benchmarks-{}.md", arch_id);
let file_path_buf = PathBuf::from(file_name);
if !file_path_buf.is_file() {
panic!("Can't find file {:?} in current directory", file_path_buf);
}
let crate_name = env!("CARGO_CRATE_NAME");
insert_string_into_file(file_path.as_path(), crate_name, md_table);
let file_path = file_path_buf.as_path();
if get_arch_name() == "x86_64" {
let tera_engine = match tera::Tera::new("benches/templates/**/*.tera") {
Ok(t) => t,
Err(e) => {
println!("Parsing error(s): {}", e);
::std::process::exit(1);
}
};
let mut context = tera::Context::new();
context.insert("arch_id", &arch_id);
context.insert("arch_name", &arch_name);
// Update introduction text
let introduction = tera_engine
.render("introduction.md.tera", &context)
.unwrap();
insert_string_into_file(file_path, "introduction", &introduction);
// Update benchmark results
let crate_name = env!("CARGO_CRATE_NAME");
context.insert("compare_results", md_table);
let tpl_file_name = format!("{}.md.tera", crate_name);
let results_block = tera_engine.render(&tpl_file_name, &context).unwrap();
insert_string_into_file(file_path, crate_name, &results_block);
if arch_id == "x86_64" {
let file_path = PathBuf::from("README.md");
if !file_path.is_file() {
panic!("Can't find file {:?} in current directory", file_path);
}
insert_string_into_file(file_path.as_path(), crate_name, md_table);
insert_string_into_file(file_path.as_path(), crate_name, &results_block);
}
}
+19 -15
View File
@@ -1,4 +1,5 @@
## Benchmarks of fast_image_resize crate for arm64 architecture
<!-- introduction start -->
## Benchmarks of fast_image_resize crate for x86_64 architecture
Environment:
@@ -8,18 +9,22 @@ Environment:
- criterion = "0.4"
- fast_image_resize = "2.7.1"
Other Rust libraries used to compare of resizing speed:
Other libraries used to compare of resizing speed:
- image = "0.24.6" (<https://crates.io/crates/image>)
- resize = "0.7.4" (<https://crates.io/crates/resize>)
- image = "0.24.7" (<https://crates.io/crates/image>)
- resize = "0.8.2" (<https://crates.io/crates/resize>)
- libvips = "8.12.1" (single-threaded mode, cache disabled)
Resize algorithms:
- Nearest
- Convolution with Bilinear filter
- Convolution with CatmullRom filter
- Convolution with Lanczos3 filter
- Box - convolution with adaptive kernel size, min 1x1 px
- Bilinear - convolution with adaptive kernel size, min 2x2 px
- Becubic (CatmullRom) - convolution with adaptive kernel size, min 4x4 px
- Lanczos3 -convolution with adaptive kernel size, min 6x6 px
<!-- introduction end -->
<!-- bench_compare_rgb start -->
### Resize RGB8 image (U8x3) 4928x3279 => 852x567
Pipeline:
@@ -29,7 +34,6 @@ Pipeline:
- Source image [nasa-4928x3279.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279.png)
- Numbers in table is mean duration of image resizing in milliseconds.
<!-- bench_compare_rgb start -->
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|----------|:-------:|:--------:|:----------:|:--------:|
| image | 85.83 | 172.77 | 327.49 | 461.80 |
@@ -38,6 +42,7 @@ Pipeline:
| fir neon | - | 43.11 | 57.52 | 81.95 |
<!-- bench_compare_rgb end -->
<!-- bench_compare_rgba start -->
### Resize RGBA8 image (U8x4) 4928x3279 => 852x567
Pipeline:
@@ -49,7 +54,6 @@ Pipeline:
- Numbers in table is mean duration of image resizing in milliseconds.
- The `image` crate does not support multiplying and dividing by alpha channel.
<!-- bench_compare_rgba start -->
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|----------|:-------:|:--------:|:----------:|:--------:|
| resize | - | 115.37 | 199.42 | 294.89 |
@@ -57,6 +61,7 @@ Pipeline:
| fir neon | - | 44.97 | 62.11 | 84.34 |
<!-- bench_compare_rgba end -->
<!-- bench_compare_l start -->
### Resize L8 image (U8) 4928x3279 => 852x567
Pipeline:
@@ -67,7 +72,6 @@ Pipeline:
has converted into grayscale image with one byte per pixel.
- Numbers in table is mean duration of image resizing in milliseconds.
<!-- bench_compare_l start -->
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|----------|:-------:|:--------:|:----------:|:--------:|
| image | 76.70 | 105.97 | 177.56 | 246.03 |
@@ -76,6 +80,7 @@ Pipeline:
| fir neon | - | 15.13 | 20.21 | 28.06 |
<!-- bench_compare_l end -->
<!-- bench_compare_la start -->
### Resize LA8 image (U8x2) 4928x3279 => 852x567
Pipeline:
@@ -89,13 +94,13 @@ Pipeline:
- The `image` crate does not support multiplying and dividing by alpha channel.
- The `resize` crate does not support this pixel format.
<!-- bench_compare_la start -->
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|----------|:-------:|:--------:|:----------:|:--------:|
| fir rust | 0.69 | 60.27 | 75.29 | 73.66 |
| fir neon | - | 35.24 | 39.87 | 55.15 |
<!-- bench_compare_la end -->
<!-- bench_compare_rgb16 start -->
### Resize RGB16 image (U16x3) 4928x3279 => 852x567
Pipeline:
@@ -106,7 +111,6 @@ Pipeline:
has converted into RGB16 image.
- Numbers in table is mean duration of image resizing in milliseconds.
<!-- bench_compare_rgb16 start -->
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|----------|:-------:|:--------:|:----------:|:--------:|
| image | 86.76 | 167.74 | 333.95 | 483.54 |
@@ -115,6 +119,7 @@ Pipeline:
| fir neon | - | 74.86 | 93.61 | 129.32 |
<!-- bench_compare_rgb16 end -->
<!-- bench_compare_rgba16 start -->
### Resize RGBA16 image (U16x4) 4928x3279 => 852x567
Pipeline:
@@ -126,7 +131,6 @@ Pipeline:
- Numbers in table is mean duration of image resizing in milliseconds.
- The `image` crate does not support multiplying and dividing by alpha channel.
<!-- bench_compare_rgba16 start -->
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|----------|:-------:|:--------:|:----------:|:--------:|
| resize | - | 120.12 | 205.34 | 304.43 |
@@ -134,6 +138,7 @@ Pipeline:
| fir neon | - | 79.02 | 119.04 | 161.24 |
<!-- bench_compare_rgba16 end -->
<!-- bench_compare_l16 start -->
### Resize L16 image (U16) 4928x3279 => 852x567
Pipeline:
@@ -144,7 +149,6 @@ Pipeline:
has converted into grayscale image with two bytes per pixel.
- Numbers in table is mean duration of image resizing in milliseconds.
<!-- bench_compare_l16 start -->
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|----------|:-------:|:--------:|:----------:|:--------:|
| image | 77.69 | 111.50 | 189.42 | 262.56 |
@@ -153,6 +157,7 @@ Pipeline:
| fir neon | - | 17.54 | 26.77 | 37.92 |
<!-- bench_compare_l16 end -->
<!-- bench_compare_la16 start -->
### Resize LA16 (luma with alpha channel) image (U16x2) 4928x3279 => 852x567
Pipeline:
@@ -166,7 +171,6 @@ Pipeline:
- The `image` crate does not support multiplying and dividing by alpha channel.
- The `resize` crate does not support this pixel format.
<!-- bench_compare_la16 start -->
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|----------|:-------:|:--------:|:----------:|:--------:|
| fir rust | 1.06 | 106.70 | 193.68 | 272.60 |
+24 -20
View File
@@ -1,27 +1,32 @@
## Benchmarks of fast_image_resize crate for Wasm32 architecture
<!-- introduction start -->
## Benchmarks of fast_image_resize crate for x86_64 architecture
Environment:
- CPU: AMD Ryzen 9 5950X
- RAM: DDR4 3800 MHz
- Ubuntu 22.04 (linux 5.19.0)
- Rust 1.69.0
- wasmtime = "8.0.0"
- criterion = "0.4"
- Ubuntu 22.04 (linux 6.2.0)
- Rust 1.73.0
- criterion = "0.5.1"
- fast_image_resize = "2.7.1"
- wasmtime = "8.0.0"
Other Rust libraries used to compare of resizing speed:
Other libraries used to compare of resizing speed:
- image = "0.24.6" (<https://crates.io/crates/image>)
- resize = "0.7.4" (<https://crates.io/crates/resize>)
- image = "0.24.7" (<https://crates.io/crates/image>)
- resize = "0.8.2" (<https://crates.io/crates/resize>)
- libvips = "8.12.1" (single-threaded mode, cache disabled)
Resize algorithms:
- Nearest
- Convolution with Bilinear filter
- Convolution with CatmullRom filter
- Convolution with Lanczos3 filter
- Box - convolution with adaptive kernel size, min 1x1 px
- Bilinear - convolution with adaptive kernel size, min 2x2 px
- Becubic (CatmullRom) - convolution with adaptive kernel size, min 4x4 px
- Lanczos3 -convolution with adaptive kernel size, min 6x6 px
<!-- introduction end -->
<!-- bench_compare_rgb start -->
### Resize RGB8 image (U8x3) 4928x3279 => 852x567
Pipeline:
@@ -31,7 +36,6 @@ Pipeline:
- Source image [nasa-4928x3279.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279.png)
- Numbers in table is mean duration of image resizing in milliseconds.
<!-- bench_compare_rgb start -->
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|-------------|:-------:|:--------:|:----------:|:--------:|
| image | 23.16 | 99.13 | 173.28 | 247.24 |
@@ -40,6 +44,7 @@ Pipeline:
| fir simd128 | - | 15.76 | 21.15 | 29.93 |
<!-- bench_compare_rgb end -->
<!-- bench_compare_rgba start -->
### Resize RGBA8 image (U8x4) 4928x3279 => 852x567
Pipeline:
@@ -51,7 +56,6 @@ Pipeline:
- Numbers in table is mean duration of image resizing in milliseconds.
- The `image` crate does not support multiplying and dividing by alpha channel.
<!-- bench_compare_rgba start -->
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|-------------|:-------:|:--------:|:----------:|:--------:|
| resize | - | 70.26 | 135.86 | 201.70 |
@@ -59,6 +63,7 @@ Pipeline:
| fir simd128 | - | 18.99 | 24.47 | 32.11 |
<!-- bench_compare_rgba end -->
<!-- bench_compare_l start -->
### Resize L8 image (U8) 4928x3279 => 852x567
Pipeline:
@@ -69,7 +74,6 @@ Pipeline:
has converted into grayscale image with one byte per pixel.
- Numbers in table is mean duration of image resizing in milliseconds.
<!-- bench_compare_l start -->
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|-------------|:-------:|:--------:|:----------:|:--------:|
| image | 19.97 | 76.69 | 130.46 | 183.81 |
@@ -78,6 +82,7 @@ Pipeline:
| fir simd128 | - | 8.62 | 8.73 | 13.31 |
<!-- bench_compare_l end -->
<!-- bench_compare_la start -->
### Resize LA8 image (U8x2) 4928x3279 => 852x567
Pipeline:
@@ -91,13 +96,13 @@ Pipeline:
- The `image` crate does not support multiplying and dividing by alpha channel.
- The `resize` crate does not support this pixel format.
<!-- bench_compare_la start -->
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|-------------|:-------:|:--------:|:----------:|:--------:|
| fir rust | 0.22 | 61.12 | 94.63 | 129.43 |
| fir simd128 | - | 20.28 | 21.32 | 27.25 |
<!-- bench_compare_la end -->
<!-- bench_compare_rgb16 start -->
### Resize RGB16 image (U16x3) 4928x3279 => 852x567
Pipeline:
@@ -108,7 +113,6 @@ Pipeline:
has converted into RGB16 image.
- Numbers in table is mean duration of image resizing in milliseconds.
<!-- bench_compare_rgb16 start -->
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|-------------|:-------:|:--------:|:----------:|:--------:|
| image | 23.09 | 104.82 | 196.11 | 285.95 |
@@ -117,6 +121,7 @@ Pipeline:
| fir simd128 | - | 53.94 | 91.66 | 131.08 |
<!-- bench_compare_rgb16 end -->
<!-- bench_compare_rgba16 start -->
### Resize RGBA16 image (U16x4) 4928x3279 => 852x567
Pipeline:
@@ -128,7 +133,6 @@ Pipeline:
- Numbers in table is mean duration of image resizing in milliseconds.
- The `image` crate does not support multiplying and dividing by alpha channel.
<!-- bench_compare_rgba16 start -->
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|-------------|:-------:|:--------:|:----------:|:--------:|
| resize | - | 73.08 | 139.06 | 205.89 |
@@ -136,6 +140,7 @@ Pipeline:
| fir simd128 | - | 76.39 | 124.16 | 173.10 |
<!-- bench_compare_rgba16 end -->
<!-- bench_compare_l16 start -->
### Resize L16 image (U16) 4928x3279 => 852x567
Pipeline:
@@ -146,7 +151,6 @@ Pipeline:
has converted into grayscale image with two bytes per pixel.
- Numbers in table is mean duration of image resizing in milliseconds.
<!-- bench_compare_l16 start -->
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|-------------|:-------:|:--------:|:----------:|:--------:|
| image | 21.17 | 77.63 | 131.44 | 185.26 |
@@ -155,6 +159,7 @@ Pipeline:
| fir simd128 | - | 19.42 | 30.96 | 44.40 |
<!-- bench_compare_l16 end -->
<!-- bench_compare_la16 start -->
### Resize LA16 (luma with alpha channel) image (U16x2) 4928x3279 => 852x567
Pipeline:
@@ -168,9 +173,8 @@ Pipeline:
- The `image` crate does not support multiplying and dividing by alpha channel.
- The `resize` crate does not support this pixel format.
<!-- bench_compare_la16 start -->
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|-------------|:-------:|:--------:|:----------:|:--------:|
| fir rust | 0.29 | 77.94 | 118.99 | 159.79 |
| fir simd128 | - | 40.72 | 65.98 | 92.33 |
<!-- bench_compare_la16 end -->
<!-- bench_compare_la16 end -->
+24 -18
View File
@@ -1,26 +1,33 @@
<!-- introduction start -->
## Benchmarks of fast_image_resize crate for x86_64 architecture
Environment:
- CPU: AMD Ryzen 9 5950X
- RAM: DDR4 3800 MHz
- Ubuntu 22.04 (linux 5.19.0)
- Rust 1.69.0
- criterion = "0.4"
- Ubuntu 22.04 (linux 6.2.0)
- Rust 1.73.0
- criterion = "0.5.1"
- fast_image_resize = "2.7.1"
Other Rust libraries used to compare of resizing speed:
- image = "0.24.6" (<https://crates.io/crates/image>)
- resize = "0.7.4" (<https://crates.io/crates/resize>)
Other libraries used to compare of resizing speed:
- image = "0.24.7" (<https://crates.io/crates/image>)
- resize = "0.8.2" (<https://crates.io/crates/resize>)
- libvips = "8.12.1" (single-threaded mode, cache disabled)
Resize algorithms:
- Nearest
- Convolution with Bilinear filter
- Convolution with CatmullRom filter
- Convolution with Lanczos3 filter
- Box - convolution with adaptive kernel size, min 1x1 px
- Bilinear - convolution with adaptive kernel size, min 2x2 px
- Bicubic (CatmullRom) - convolution with adaptive kernel size, min 4x4 px
- Lanczos3 -convolution with adaptive kernel size, min 6x6 px
<!-- introduction end -->
<!-- bench_compare_rgb start -->
### Resize RGB8 image (U8x3) 4928x3279 => 852x567
Pipeline:
@@ -30,7 +37,6 @@ Pipeline:
- Source image [nasa-4928x3279.png](https://github.com/Cykooz/fast_image_resize/blob/main/data/nasa-4928x3279.png)
- Numbers in table is mean duration of image resizing in milliseconds.
<!-- bench_compare_rgb start -->
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|------------|:-------:|:--------:|:----------:|:--------:|
| image | 19.39 | 81.40 | 138.60 | 196.45 |
@@ -40,6 +46,7 @@ Pipeline:
| fir avx2 | - | 7.91 | 9.89 | 14.84 |
<!-- bench_compare_rgb end -->
<!-- bench_compare_rgba start -->
### Resize RGBA8 image (U8x4) 4928x3279 => 852x567
Pipeline:
@@ -51,7 +58,6 @@ Pipeline:
- Numbers in table is mean duration of image resizing in milliseconds.
- The `image` crate does not support multiplying and dividing by alpha channel.
<!-- bench_compare_rgba start -->
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|------------|:-------:|:--------:|:----------:|:--------:|
| resize | - | 73.12 | 137.57 | 202.70 |
@@ -60,6 +66,7 @@ Pipeline:
| fir avx2 | - | 9.53 | 12.06 | 16.49 |
<!-- bench_compare_rgba end -->
<!-- bench_compare_l start -->
### Resize L8 image (U8) 4928x3279 => 852x567
Pipeline:
@@ -70,7 +77,6 @@ Pipeline:
has converted into grayscale image with one byte per pixel.
- Numbers in table is mean duration of image resizing in milliseconds.
<!-- bench_compare_l start -->
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|------------|:-------:|:--------:|:----------:|:--------:|
| image | 16.29 | 47.78 | 75.65 | 103.63 |
@@ -80,6 +86,7 @@ Pipeline:
| fir avx2 | - | 6.66 | 4.97 | 8.14 |
<!-- bench_compare_l end -->
<!-- bench_compare_la start -->
### Resize LA8 image (U8x2) 4928x3279 => 852x567
Pipeline:
@@ -93,7 +100,6 @@ Pipeline:
- The `image` crate does not support multiplying and dividing by alpha channel.
- The `resize` crate does not support this pixel format.
<!-- bench_compare_la start -->
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|------------|:-------:|:--------:|:----------:|:--------:|
| fir rust | 0.17 | 24.81 | 30.32 | 42.56 |
@@ -101,6 +107,7 @@ Pipeline:
| fir avx2 | - | 8.73 | 9.59 | 12.41 |
<!-- bench_compare_la end -->
<!-- bench_compare_rgb16 start -->
### Resize RGB16 image (U16x3) 4928x3279 => 852x567
Pipeline:
@@ -111,7 +118,6 @@ Pipeline:
has converted into RGB16 image.
- Numbers in table is mean duration of image resizing in milliseconds.
<!-- bench_compare_rgb16 start -->
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|------------|:-------:|:--------:|:----------:|:--------:|
| image | 19.01 | 73.92 | 133.43 | 183.58 |
@@ -121,6 +127,7 @@ Pipeline:
| fir avx2 | - | 19.87 | 29.82 | 36.09 |
<!-- bench_compare_rgb16 end -->
<!-- bench_compare_rgba16 start -->
### Resize RGBA16 image (U16x4) 4928x3279 => 852x567
Pipeline:
@@ -132,7 +139,6 @@ Pipeline:
- Numbers in table is mean duration of image resizing in milliseconds.
- The `image` crate does not support multiplying and dividing by alpha channel.
<!-- bench_compare_rgba16 start -->
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|------------|:-------:|:--------:|:----------:|:--------:|
| resize | - | 71.67 | 134.02 | 195.58 |
@@ -141,6 +147,7 @@ Pipeline:
| fir avx2 | - | 25.64 | 36.42 | 47.76 |
<!-- bench_compare_rgba16 end -->
<!-- bench_compare_l16 start -->
### Resize L16 image (U16) 4928x3279 => 852x567
Pipeline:
@@ -151,7 +158,6 @@ Pipeline:
has converted into grayscale image with two bytes per pixel.
- Numbers in table is mean duration of image resizing in milliseconds.
<!-- bench_compare_l16 start -->
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|------------|:-------:|:--------:|:----------:|:--------:|
| image | 16.34 | 48.68 | 76.27 | 105.09 |
@@ -161,6 +167,7 @@ Pipeline:
| fir avx2 | - | 6.66 | 8.71 | 13.97 |
<!-- bench_compare_l16 end -->
<!-- bench_compare_la16 start -->
### Resize LA16 (luma with alpha channel) image (U16x2) 4928x3279 => 852x567
Pipeline:
@@ -174,10 +181,9 @@ Pipeline:
- The `image` crate does not support multiplying and dividing by alpha channel.
- The `resize` crate does not support this pixel format.
<!-- bench_compare_la16 start -->
| | Nearest | Bilinear | CatmullRom | Lanczos3 |
|------------|:-------:|:--------:|:----------:|:--------:|
| fir rust | 0.20 | 33.72 | 53.68 | 72.66 |
| fir sse4.1 | - | 21.98 | 34.05 | 46.60 |
| fir avx2 | - | 15.26 | 22.01 | 29.30 |
<!-- bench_compare_la16 end -->
<!-- bench_compare_la16 end -->
+15 -8
View File
@@ -2,42 +2,49 @@ use std::f64::consts::PI;
pub type FilterFn<'a> = &'a dyn Fn(f64) -> f64;
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
#[derive(Default, Clone, Copy, Debug, PartialEq, Eq)]
#[non_exhaustive]
pub enum FilterType {
/// Each pixel of source image contributes to one pixel of the
/// destination image with identical weights. For upscaling is equivalent
/// of `Nearest` resize algorithm.
/// of `Nearest` resize algorithm.
///
/// Minimal kernel size 1x1 px.
Box,
/// Bilinear filter calculate the output pixel value using linear
/// interpolation on all pixels that may contribute to the output value.
///
/// Minimal kernel size 2x2 px.
Bilinear,
/// Hamming filter has the same performance as `Bilinear` filter while
/// providing the image downscaling quality comparable to bicubic
/// (`CatmulRom` or `Mitchell`). Produces a sharper image than `Bilinear`,
/// doesn't have dislocations on local level like with `Box`.
/// The filter don’t show good quality for the image upscaling.
///
/// Minimal kernel size 2x2 px.
Hamming,
/// Catmull-Rom bicubic filter calculate the output pixel value using
/// cubic interpolation on all pixels that may contribute to the output
/// value.
///
/// Minimal kernel size 4x4 px.
CatmullRom,
/// Mitchell–Netravali bicubic filter calculate the output pixel value
/// using cubic interpolation on all pixels that may contribute to the
/// output value.
///
/// Minimal kernel size 4x4 px.
Mitchell,
/// Lanczos3 filter calculate the output pixel value using a high-quality
/// Lanczos filter (a truncated sinc) on all pixels that may contribute
/// to the output value.
///
/// Minimal kernel size 6x6 px.
#[default]
Lanczos3,
}
impl Default for FilterType {
fn default() -> Self {
FilterType::Lanczos3
}
}
/// Returns reference to filter function and value of `filter_support`.
#[inline]
pub fn get_filter_func(filter_type: FilterType) -> (FilterFn<'static>, f64) {
+2
View File
@@ -132,6 +132,8 @@ pub fn precompute_coefficients(
ww += w;
}
if ww != 0.0 {
// Normalise values of coefficients.
// Sum of coefficients must be equal to 1.0.
coeffs[cur_index..].iter_mut().for_each(|w| *w /= ww);
}
// Remaining values should stay empty if they are used despite x_max.