diff --git a/.idea/av-denoise.iml b/.idea/av-denoise.iml
index b294e21..c17e1b4 100644
--- a/.idea/av-denoise.iml
+++ b/.idea/av-denoise.iml
@@ -12,10 +12,13 @@
+
+
+
diff --git a/Cargo.lock b/Cargo.lock
index d516805..db6018b 100644
--- a/Cargo.lock
+++ b/Cargo.lock
@@ -187,19 +187,25 @@ dependencies = [
"av-decoders",
"av-denoise-core",
"av-scenechange",
+ "bon",
"bytesize",
"clap",
"crossbeam-channel",
+ "cubecl",
+ "etcetera",
"ffms2-sys",
"indicatif",
"mimalloc",
"pkg-config",
"strum",
"strum_macros",
+ "tempfile",
+ "thiserror",
"tracing",
"tracing-indicatif",
"tracing-subscriber",
"v_frame",
+ "wgpu",
"y4m",
]
@@ -211,11 +217,9 @@ dependencies = [
"bon",
"clap",
"cubecl",
- "etcetera",
"futures",
"strum",
"strum_macros",
- "tempfile",
"thiserror",
"tracing",
]
@@ -225,7 +229,7 @@ name = "av-denoise-vs"
version = "0.5.0-alpha3"
dependencies = [
"anyhow",
- "av-denoise-core",
+ "av-denoise",
"tracing",
"tracing-subscriber",
"vapoursynth",
diff --git a/Cargo.toml b/Cargo.toml
index 5fa9ade..34f5ec9 100644
--- a/Cargo.toml
+++ b/Cargo.toml
@@ -20,6 +20,7 @@ strum_macros = "0.28"
bon = "3"
cubecl = { version = "0.10", features = ["std"] }
av-denoise-core = { path = "av-denoise-core", version = "0.5.0-alpha3", default-features = false }
+av-denoise = { path = "av-denoise", version = "0.5.0-alpha3", default-features = false }
[profile.test]
opt-level = 3
diff --git a/Justfile b/Justfile
index 0a99e14..fad4048 100755
--- a/Justfile
+++ b/Justfile
@@ -66,6 +66,10 @@ build-static features=static_features:
bench *ARGS:
cargo bench -p av-denoise-core {{ARGS}}
+# Benchmarks the host layer, end to end through `HostDenoiser` and `PlanarDenoiser`.
+bench-host *ARGS:
+ cargo bench -p av-denoise --features vulkan {{ARGS}}
+
build-vs *ARGS:
cargo build -p av-denoise-vs --release {{ARGS}}
@@ -84,6 +88,7 @@ test-rust:
cargo nextest run --release -p av-denoise --features vulkan,binary
cargo nextest run --release -p av-denoise-vs --features vulkan
cargo test --doc -p av-denoise-core --features vulkan
+ cargo test --doc -p av-denoise --features vulkan
cargo check --workspace
# Every Python test, including the ones that render on the GPU.
diff --git a/README.md b/README.md
index bfd49d9..1eea8df 100644
--- a/README.md
+++ b/README.md
@@ -171,8 +171,8 @@ without use.
If the cache directory cannot be created, `av-denoise` logs a warning and carries on without a cache.
-Library users can call `av_denoise::install_compilation_cache()` before `Denoiser::create` to get the same
-behaviour in their own binary. It has to run before the first `Denoiser` exists, because building a CubeCL client
+Library users can call `av_denoise::install_compilation_cache()` before `HostDenoiser::create` to get the same
+behaviour in their own binary. It has to run before the first denoiser exists, because building a CubeCL client
locks the global config. An embedder that wants to choose the cache directory itself can call
`av_denoise::default_cache_dir()` to get the same default this crate uses, and
`av_denoise::install_compilation_cache_at()` to install it, or any other directory, directly.
diff --git a/av-denoise-core/Cargo.toml b/av-denoise-core/Cargo.toml
index cc3129f..bf73faa 100644
--- a/av-denoise-core/Cargo.toml
+++ b/av-denoise-core/Cargo.toml
@@ -4,7 +4,7 @@ version.workspace = true
edition.workspace = true
license.workspace = true
repository.workspace = true
-description = "Core kernels and types for av-denoise (Do not use directly)"
+description = "GPU denoising engines that read and write CubeCL handles"
categories = ["multimedia::video"]
keywords = ["cubecl", "denoise", "gpu"]
readme = "README.md"
@@ -28,14 +28,12 @@ tracing = "0.1"
strum_macros = "0.28"
strum = "0.28"
bon = "3"
-etcetera = "0.11"
cubecl = { version = "0.10", features = ["std"] }
[dev-dependencies]
futures = "0.3"
clap = { version = "4", features = ["derive"] }
-tempfile = "3"
[features]
vulkan = ["cubecl/vulkan"]
@@ -52,26 +50,14 @@ harness = false
name = "bench_kernels"
harness = false
-[[bench]]
-name = "denoise"
-harness = false
-
[[bench]]
name = "motion"
harness = false
-[[bench]]
-name = "convert"
-harness = false
-
[[bench]]
name = "nl4d_ablation"
harness = false
-[[bench]]
-name = "reseed"
-harness = false
-
[[bench]]
name = "mc_accuracy"
harness = false
diff --git a/av-denoise-core/README.md b/av-denoise-core/README.md
index d3873e4..18c76c2 100644
--- a/av-denoise-core/README.md
+++ b/av-denoise-core/README.md
@@ -1,5 +1,44 @@
# av-denoise-core
-The core library and kernels for the av-denoise system.
+GPU denoising engines that read and write CubeCL plane handles.
-You should use either the `av-denoise` library, binary or VapourSynth plugin, not this crate directly.
+`Nlmeans` and `Nl4d` implement `Engine`. Push one frame of planes at a time. Each push returns how many
+frames are ready, and every ready frame is written into planes you own with `Engine::emit_into` before the
+next push. At the end of a stream, `Engine::finish` returns the tail count.
+
+`Nl4dOptions::default()` uses the base preset's `temporal_radius` and calibrated defaults for the rest.
+
+```rust,no_run
+use av_denoise_core::{ChannelMode, DevicePlane, Engine, Geometry, Nl4d, Nl4dOptions, SampleFormat};
+use cubecl::Runtime;
+use cubecl::wgpu::WgpuRuntime;
+
+# fn main() -> Result<(), av_denoise_core::Error> {
+let device = ::Device::default();
+let client = WgpuRuntime::client(&device);
+let geometry = Geometry {
+ width: 1920,
+ height: 1080,
+ channels: ChannelMode::Luma,
+ input: SampleFormat::U16 { depth: 10 },
+ output: SampleFormat::U16 { depth: 10 },
+};
+let mut engine = Nl4d::new(&client, Nl4dOptions::default(), geometry)?;
+
+let input = client.empty(1920 * 1080 * 2);
+let output = client.empty(1920 * 1080 * 2);
+let input_planes = [DevicePlane::new(&input, 1920, 1080)];
+let output_planes = [DevicePlane::new(&output, 1920, 1080)];
+
+let ready = engine.push(&input_planes)?;
+for _ in 0..ready {
+ engine.emit_into(&output_planes)?;
+}
+
+let tail = engine.finish()?;
+for _ in 0..tail {
+ engine.emit_into(&output_planes)?;
+}
+# Ok(())
+# }
+```
diff --git a/av-denoise-core/benches/bench_kernels.rs b/av-denoise-core/benches/bench_kernels.rs
index 0c4b131..44bf077 100644
--- a/av-denoise-core/benches/bench_kernels.rs
+++ b/av-denoise-core/benches/bench_kernels.rs
@@ -1,8 +1,8 @@
-use av_denoise_core::Depth;
-use cubecl::prelude::*;
-
mod kernels;
+use av_denoise_core::bench_api::Device;
+use clap::Parser;
+use cubecl::prelude::*;
use kernels::accumulate::AccumulateBench;
use kernels::bilateral::BilateralBench;
use kernels::collab_aggregate::{CollabNormaliseBench, CollabZeroAccumBench};
@@ -14,6 +14,7 @@ use kernels::distance::DistanceBench;
use kernels::distance_pair::DistancePairBench;
use kernels::distance_pair_ref::DistancePairRefBench;
use kernels::distance_ref::DistanceRefBench;
+use kernels::egress::{EgressBench, EgressFormat};
use kernels::finish::FinishBench;
use kernels::fused_window::{
FusedPairWindowBench,
@@ -24,6 +25,7 @@ use kernels::fused_window::{
use kernels::grain::{GRAIN_SIZES, GrainMeasureBench, GrainReducePartialsBench, GrainSaveVectorsBench};
use kernels::horizontal_sum::HSumBench;
use kernels::horizontal_sum_pair::HSumPairBench;
+use kernels::ingest::{IngestBench, IngestFormat};
use kernels::mc_block_match_coarse::BlockMatchCoarseBench;
use kernels::mc_block_match_fine::BlockMatchFineBench;
use kernels::mc_chain_compose::ChainComposeBench;
@@ -32,9 +34,7 @@ use kernels::mc_downscale::DownscaleBench;
use kernels::mc_warp::WarpBench;
use kernels::mv_regularise::MvRegulariseBench;
use kernels::noise_partial::NoisePartialBench;
-use kernels::pack_wire::PackWireBench;
use kernels::temporal_noise_stats::TemporalNoiseStatsBench;
-use kernels::unpack_wire::UnpackWireBench;
use kernels::vertical_weight::VWeightBench;
use kernels::vweight_pair_accumulate::VWeightPairAccBench;
use kernels::zero::ZeroBench;
@@ -47,110 +47,133 @@ fn run_all(backend: &str, device: &R::Device) {
println!("--- {backend} ---");
print_header();
- for &(ch, ch_name) in CHANNELS {
+ for &(channels, channel_name) in CHANNELS {
run(CopyBench {
client: client.clone(),
- ch,
- ch_name,
+ channels,
+ channel_name,
});
}
- for &(ch, ch_name) in CHANNELS {
- for depth in [Depth::Eight, Depth::Ten] {
- run(PackWireBench {
- client: client.clone(),
- ch,
- ch_name,
- depth,
- });
- }
+
+ let ingest_cases = [
+ (1920, 1080, 1, "luma", IngestFormat::U8),
+ (960, 540, 2, "chroma", IngestFormat::U16Ten),
+ (1920, 1080, 3, "yuv", IngestFormat::U16Ten),
+ (1920, 1080, 1, "luma", IngestFormat::F32),
+ ];
+ for (width, height, channels, channel_name, format) in ingest_cases {
+ run(IngestBench {
+ client: client.clone(),
+ width,
+ height,
+ channels,
+ channel_name,
+ format,
+ });
}
- for &(ch, ch_name) in CHANNELS {
- for depth in [Depth::Eight, Depth::Ten] {
- run(UnpackWireBench {
- client: client.clone(),
- ch,
- ch_name,
- depth,
- });
- }
+
+ let egress_cases = [
+ (1920, 1080, 1, "luma", EgressFormat::U8),
+ (960, 540, 2, "chroma", EgressFormat::U16Ten),
+ (1920, 1080, 3, "yuv", EgressFormat::U16Ten),
+ (1920, 1080, 1, "luma", EgressFormat::F32),
+ ];
+ for (width, height, channels, channel_name, format) in egress_cases {
+ run(EgressBench {
+ client: client.clone(),
+ width,
+ height,
+ channels,
+ channel_name,
+ format,
+ });
}
- for &(ch, ch_name) in CHANNELS {
+
+ for &(channels, channel_name) in CHANNELS {
run(ZeroBench {
client: client.clone(),
- ch,
- ch_name,
+ channels,
+ channel_name,
});
}
- for &(ch, ch_name) in CHANNELS {
+ for &(channels, channel_name) in CHANNELS {
run(DistWeightBench {
client: client.clone(),
- ch,
- ch_name,
+ channels,
+ channel_name,
});
}
- for &(ch, ch_name) in CHANNELS {
+
+ for &(channels, channel_name) in CHANNELS {
run(DistWeightRefBench {
client: client.clone(),
- ch,
- ch_name,
+ channels,
+ channel_name,
});
}
- for &(ch, ch_name) in CHANNELS {
+
+ for &(channels, channel_name) in CHANNELS {
run(FusedSingleWindowBench {
client: client.clone(),
- ch,
- ch_name,
+ channels,
+ channel_name,
});
}
- for &(ch, ch_name) in CHANNELS {
+
+ for &(channels, channel_name) in CHANNELS {
run(FusedPairWindowBench {
client: client.clone(),
- ch,
- ch_name,
+ channels,
+ channel_name,
});
}
- for &(ch, ch_name) in CHANNELS {
+
+ for &(channels, channel_name) in CHANNELS {
run(FusedSingleWindowRefBench {
client: client.clone(),
- ch,
- ch_name,
+ channels,
+ channel_name,
});
}
- for &(ch, ch_name) in CHANNELS {
+
+ for &(channels, channel_name) in CHANNELS {
run(FusedPairWindowRefBench {
client: client.clone(),
- ch,
- ch_name,
+ channels,
+ channel_name,
});
}
- for &(ch, ch_name) in CHANNELS {
+ for &(channels, channel_name) in CHANNELS {
run(DistanceBench {
client: client.clone(),
- ch,
- ch_name,
+ channels,
+ channel_name,
});
}
- for &(ch, ch_name) in CHANNELS {
+
+ for &(channels, channel_name) in CHANNELS {
run(DistanceRefBench {
client: client.clone(),
- ch,
- ch_name,
+ channels,
+ channel_name,
});
}
- for &(ch, ch_name) in CHANNELS {
+
+ for &(channels, channel_name) in CHANNELS {
run(DistancePairBench {
client: client.clone(),
- ch,
- ch_name,
+ channels,
+ channel_name,
});
}
- for &(ch, ch_name) in CHANNELS {
+
+ for &(channels, channel_name) in CHANNELS {
run(DistancePairRefBench {
client: client.clone(),
- ch,
- ch_name,
+ channels,
+ channel_name,
});
}
@@ -163,45 +186,44 @@ fn run_all(backend: &str, device: &R::Device) {
run(VWeightBench {
client: client.clone(),
});
- for &(ch, ch_name) in CHANNELS {
+
+ for &(channels, channel_name) in CHANNELS {
run(VWeightPairAccBench {
client: client.clone(),
- ch,
- ch_name,
+ channels,
+ channel_name,
});
}
- for &(ch, ch_name) in CHANNELS {
+ for &(channels, channel_name) in CHANNELS {
run(AccumulateBench {
client: client.clone(),
- ch,
- ch_name,
+ channels,
+ channel_name,
});
}
- for &(ch, ch_name) in CHANNELS {
+
+ for &(channels, channel_name) in CHANNELS {
run(FinishBench {
client: client.clone(),
- ch,
- ch_name,
+ channels,
+ channel_name,
});
}
- for &(ch, ch_name) in CHANNELS {
+ for &(channels, channel_name) in CHANNELS {
run(BilateralBench {
client: client.clone(),
- ch,
- ch_name,
+ channels,
+ channel_name,
});
}
- // Immerkær noise estimate (stage 1 + stage 2 back-to-back).
run(NoisePartialBench {
client: client.clone(),
});
- // Temporal-residual noise-stats kernel, the default `luma_fields`
- // off variant every caller but nl4d's luma pass uses, and the on
- // variant that pass will enable.
+ // `luma_fields` off as nlmeans runs it, and on as nl4d runs it with the noise map.
run(TemporalNoiseStatsBench {
client: client.clone(),
luma_fields: false,
@@ -211,9 +233,8 @@ fn run_all(backend: &str, device: &R::Device) {
luma_fields: true,
});
- // Motion-compensation kernels. Pyramid build and analyse are
- // luma-only (ME doesn't look at chroma); warp runs per channel
- // mode because its memory traffic scales with `stored_ch`.
+ // Pyramid build and block matching are luma-only since motion estimation ignores chroma. Warp
+ // runs per channel mode because its memory traffic scales with `stored_ch`.
run(DownscaleBench {
client: client.clone(),
});
@@ -232,6 +253,7 @@ fn run_all(backend: &str, device: &R::Device) {
run(MvRegulariseBench {
client: client.clone(),
});
+
for &size in GRAIN_SIZES {
run(GrainSaveVectorsBench {
client: client.clone(),
@@ -246,21 +268,23 @@ fn run_all(backend: &str, device: &R::Device) {
size,
});
}
- for &(ch, ch_name) in CHANNELS {
+
+ for &(channels, channel_name) in CHANNELS {
run(CollabFusedBench {
client: client.clone(),
- ch,
- ch_name,
+ channels,
+ channel_name,
split_mv: false,
noise_curve: false,
strength_map: false,
pooled: false,
});
}
+
run(CollabFusedBench {
client: client.clone(),
- ch: 1,
- ch_name: "luma",
+ channels: 1,
+ channel_name: "luma",
split_mv: false,
noise_curve: true,
strength_map: false,
@@ -268,8 +292,8 @@ fn run_all(backend: &str, device: &R::Device) {
});
run(CollabFusedBench {
client: client.clone(),
- ch: 1,
- ch_name: "luma",
+ channels: 1,
+ channel_name: "luma",
split_mv: false,
noise_curve: true,
strength_map: true,
@@ -277,43 +301,47 @@ fn run_all(backend: &str, device: &R::Device) {
});
run(CollabFusedBench {
client: client.clone(),
- ch: 1,
- ch_name: "luma",
+ channels: 1,
+ channel_name: "luma",
split_mv: false,
noise_curve: true,
strength_map: true,
pooled: true,
});
- for &(ch, ch_name) in CHANNELS {
+
+ for &(channels, channel_name) in CHANNELS {
run(CollabFusedBench {
client: client.clone(),
- ch,
- ch_name,
+ channels,
+ channel_name,
split_mv: true,
noise_curve: false,
strength_map: false,
pooled: false,
});
}
- for &(ch, ch_name) in CHANNELS {
+
+ for &(channels, channel_name) in CHANNELS {
run(CollabNormaliseBench {
client: client.clone(),
- ch,
- ch_name,
+ channels,
+ channel_name,
});
}
- for &(ch, ch_name) in CHANNELS {
+
+ for &(channels, channel_name) in CHANNELS {
run(CollabZeroAccumBench {
client: client.clone(),
- ch,
- ch_name,
+ channels,
+ channel_name,
});
}
- for &(ch, ch_name) in CHANNELS {
+
+ for &(channels, channel_name) in CHANNELS {
run(WarpBench {
client: client.clone(),
- ch,
- ch_name,
+ channels,
+ channel_name,
});
}
@@ -323,18 +351,16 @@ fn run_all(backend: &str, device: &R::Device) {
#[derive(clap::Parser, Debug)]
#[command(about = "NLMeans per-kernel benchmarks", long_about = None)]
struct Cli {
- /// GPU device to bind to. Format: `default`, `discrete[:N]`,
- /// `integrated[:N]`, `virtual[:N]`, or `cpu`.
+ /// GPU device to bind to, one of `default`, `discrete[:N]`, `integrated[:N]`, `virtual[:N]` or `cpu`.
#[arg(long, default_value = "default")]
- device: av_denoise_core::Device,
+ device: Device,
- /// Swallowed: cargo passes this when invoking the bench binary.
+ /// Swallowed, since cargo passes this when invoking the bench binary.
#[arg(long, hide = true)]
bench: bool,
}
fn main() {
- use clap::Parser;
let cli = Cli::parse();
println!("NLMeans Per-Kernel Benchmarks - 1920x1080 (TimingMethod::Device)");
diff --git a/av-denoise-core/benches/convert.rs b/av-denoise-core/benches/convert.rs
deleted file mode 100644
index c8daf68..0000000
--- a/av-denoise-core/benches/convert.rs
+++ /dev/null
@@ -1,150 +0,0 @@
-use std::time::Instant;
-
-const W: usize = 1920;
-const H: usize = 1080;
-/// 4:2:0 sample count for one frame.
-const SAMPLES: usize = W * H + 2 * ((W / 2) * (H / 2));
-
-const WARMUP: usize = 5;
-const ITERS: usize = 200;
-
-#[derive(clap::Parser, Debug)]
-#[command(about = "Sample <-> f32 conversion benchmark", long_about = None)]
-struct Cli {
- /// Swallowed: cargo passes this when invoking the bench binary.
- #[arg(long, hide = true)]
- bench: bool,
-}
-
-fn time(label: &str, mut f: impl FnMut() -> usize) {
- for _ in 0..WARMUP {
- std::hint::black_box(f());
- }
-
- let t = Instant::now();
- for _ in 0..ITERS {
- std::hint::black_box(f());
- }
- let per_ms = t.elapsed().as_secs_f64() / ITERS as f64 * 1000.0;
-
- println!("{label:<44} {per_ms:>8.3} ms/frame");
-}
-
-fn main() {
- let _cli: Cli = clap::Parser::parse();
-
- let normalized: Vec = (0..SAMPLES).map(|i| (i % 1024) as f32 / 1023.0).collect();
-
- // Flat luma plane, the simplest read path.
- let wire8: Vec = (0..SAMPLES).map(|i| (i % 256) as u8).collect();
- let wire10: Vec = (0..SAMPLES)
- .flat_map(|i| ((i % 1024) as u16).to_le_bytes())
- .collect();
-
- // Equal-length YUV444 planes for the interleaving path.
- let yuv_pixels = SAMPLES / 3;
- let plane8: Vec = (0..yuv_pixels).map(|i| (i % 256) as u8).collect();
-
- println!("{SAMPLES} samples/frame (1080p 4:2:0), {ITERS} iters");
-
- println!("-- output --");
- time("f32 -> 8-bit plane", || {
- quantise_plane_narrow(&normalized, 255.0).len()
- });
- time("f32 -> 10-bit plane", || {
- quantise_plane_wide(&normalized, 1023.0).len()
- });
-
- println!("-- input --");
- time("8-bit plane -> f32", || read_plane_narrow(&wire8, 255.0).len());
- time("10-bit plane -> f32", || read_plane_wide(&wire10, 1023.0).len());
-
- // The fused YUV444 path reads three planes per pixel by index. The
- // reviewer of the converter task flagged that indexed reads may not
- // vectorise as well as the zip form it replaced, because the bounds
- // on the second and third planes cannot be proven. These two rows
- // are what decides whether that concern is real.
- println!("-- interleave (fused YUV444) --");
- time("8-bit YUV planes -> interleaved f32", || {
- interleave_yuv_narrow(&plane8, &plane8, &plane8, 255.0).len()
- });
- time("8-bit YUV planes -> interleaved f32 (sliced)", || {
- interleave_yuv_narrow_sliced(&plane8, &plane8, &plane8, 255.0).len()
- });
-}
-
-/// Mirrors `plane_to_f32`'s narrow arm.
-fn read_plane_narrow(plane: &[u8], max: f32) -> Vec {
- let out: Vec = (0..plane.len()).map(|i| plane[i] as f32 / max).collect();
- std::hint::black_box(&out);
- out
-}
-
-/// Mirrors `plane_to_f32`'s wide arm.
-fn read_plane_wide(plane: &[u8], max: f32) -> Vec {
- let samples = plane.len() / 2;
- let out: Vec = (0..samples)
- .map(|i| u16::from_le_bytes([plane[2 * i], plane[2 * i + 1]]) as f32 / max)
- .collect();
- std::hint::black_box(&out);
- out
-}
-
-/// Mirrors `interleave_yuv_to_f32`'s narrow arm exactly as shipped.
-fn interleave_yuv_narrow(y: &[u8], u: &[u8], v: &[u8], max: f32) -> Vec {
- let pixels = y.len();
- let mut out = Vec::with_capacity(pixels * 3);
-
- for i in 0..pixels {
- out.push(y[i] as f32 / max);
- out.push(u[i] as f32 / max);
- out.push(v[i] as f32 / max);
- }
-
- std::hint::black_box(&out);
- out
-}
-
-/// The same loop with all three planes pre-sliced to `pixels`, which lets
-/// the compiler drop the per-pixel bounds checks on `u` and `v`.
-fn interleave_yuv_narrow_sliced(y: &[u8], u: &[u8], v: &[u8], max: f32) -> Vec {
- let pixels = y.len();
- let (y, u, v) = (&y[..pixels], &u[..pixels], &v[..pixels]);
- let mut out = Vec::with_capacity(pixels * 3);
-
- for i in 0..pixels {
- out.push(y[i] as f32 / max);
- out.push(u[i] as f32 / max);
- out.push(v[i] as f32 / max);
- }
-
- std::hint::black_box(&out);
- out
-}
-
-/// Mirrors `f32_to_plane`'s narrow arm. The indexed write matches
-/// `Narrow::write` rather than a `zip`, because a bounds-check-free
-/// iterator idiom would measure a loop the binary never runs.
-fn quantise_plane_narrow(plane: &[f32], max: f32) -> Vec {
- let mut out = vec![0u8; plane.len()];
- for (i, &v) in plane.iter().enumerate() {
- out[i] = quantise(v, max) as u8;
- }
- std::hint::black_box(&out);
- out
-}
-
-/// Mirrors `f32_to_plane`'s wide arm, indexed to match `Wide::write`.
-fn quantise_plane_wide(plane: &[f32], max: f32) -> Vec {
- let mut out = vec![0u8; plane.len() * 2];
- for (i, &v) in plane.iter().enumerate() {
- out[2 * i..2 * i + 2].copy_from_slice(&quantise(v, max).to_le_bytes());
- }
- std::hint::black_box(&out);
- out
-}
-
-#[inline(always)]
-fn quantise(v: f32, max: f32) -> u16 {
- (v.clamp(0.0, 1.0) * max + 0.5) as u16
-}
diff --git a/av-denoise-core/benches/kernels/accumulate.rs b/av-denoise-core/benches/kernels/accumulate.rs
index e774ef5..f61128c 100644
--- a/av-denoise-core/benches/kernels/accumulate.rs
+++ b/av-denoise-core/benches/kernels/accumulate.rs
@@ -1,18 +1,18 @@
-use av_denoise_core::nlmeans::kernels::nlm_accumulate;
+use av_denoise_core::bench_api::kernels::nlm_accumulate;
use cubecl::benchmark::Benchmark;
use cubecl::prelude::*;
use cubecl::server::Handle;
use super::{
- H,
+ HEIGHT,
Q_X,
Q_Y,
- W,
+ WIDTH,
block_sync,
cube_count_2d,
cube_dim_2d,
make_padded_frame,
- shapes_with_ch,
+ shapes_with_channels,
stored_channels,
};
@@ -28,8 +28,8 @@ pub struct AccumulateInput {
pub struct AccumulateBench {
pub client: ComputeClient,
- pub ch: u32,
- pub ch_name: &'static str,
+ pub channels: u32,
+ pub channel_name: &'static str,
}
impl Benchmark for AccumulateBench {
@@ -37,15 +37,19 @@ impl Benchmark for AccumulateBench {
type Output = ();
fn prepare(&self) -> Self::Input {
- let pixels = (W * H) as usize;
- let stored = stored_channels(self.ch) as usize;
- let frame = make_padded_frame(W, H, self.ch);
- let input = self.client.create_from_slice(f32::as_bytes(&frame));
+ let pixels = (WIDTH * HEIGHT) as usize;
+ let stored_ch = stored_channels(self.channels) as usize;
+ let frame = make_padded_frame(WIDTH, HEIGHT, self.channels);
+ let frame_bytes = f32::as_bytes(&frame);
+ let input = self.client.create_from_slice(frame_bytes);
+
let weights_data = vec![0.5f32; pixels];
- let weights = self.client.create_from_slice(f32::as_bytes(&weights_data));
- let accum = self.client.empty(pixels * stored * size_of::());
+ let weights_bytes = f32::as_bytes(&weights_data);
+ let weights = self.client.create_from_slice(weights_bytes);
+ let accum = self.client.empty(pixels * stored_ch * size_of::());
let weight_sum = self.client.empty(pixels * size_of::());
let max_weight = self.client.empty(pixels * size_of::());
+
AccumulateInput {
input,
accum,
@@ -57,16 +61,19 @@ impl Benchmark for AccumulateBench {
}
fn execute(&self, args: Self::Input) -> Result<(), String> {
- let pixels = (W * H) as usize;
- let stored = stored_channels(self.ch) as usize;
+ let pixels = (WIDTH * HEIGHT) as usize;
+ let stored_ch = stored_channels(self.channels) as usize;
+ let cube_count = cube_count_2d();
+ let cube_dim = cube_dim_2d();
+
unsafe {
nlm_accumulate::launch_unchecked::(
&self.client,
- cube_count_2d(),
- cube_dim_2d(),
- stored,
+ cube_count,
+ cube_dim,
+ stored_ch,
ArrayArg::from_raw_parts(args.input.clone(), args.frame_len),
- ArrayArg::from_raw_parts(args.accum.clone(), pixels * stored),
+ ArrayArg::from_raw_parts(args.accum.clone(), pixels * stored_ch),
ArrayArg::from_raw_parts(args.weight_sum.clone(), pixels),
ArrayArg::from_raw_parts(args.weights.clone(), pixels),
ArrayArg::from_raw_parts(args.weights.clone(), pixels),
@@ -75,20 +82,23 @@ impl Benchmark for AccumulateBench {
0u32,
Q_X,
Q_Y,
- W,
- H,
+ WIDTH,
+ HEIGHT,
);
}
+
Ok(())
}
fn name(&self) -> String {
- format!("accumulate_1080p_{}", self.ch_name)
+ format!("accumulate_1080p_{}", self.channel_name)
}
+
fn sync(&self) {
block_sync(&self.client);
}
+
fn shapes(&self) -> Vec> {
- shapes_with_ch(self.ch)
+ shapes_with_channels(self.channels)
}
}
diff --git a/av-denoise-core/benches/kernels/bilateral.rs b/av-denoise-core/benches/kernels/bilateral.rs
index 3d06d50..82508a8 100644
--- a/av-denoise-core/benches/kernels/bilateral.rs
+++ b/av-denoise-core/benches/kernels/bilateral.rs
@@ -1,5 +1,5 @@
-use av_denoise_core::nlmeans::kernels::nlm_bilateral;
-use av_denoise_core::nlmeans::prefilter::bilateral_radius;
+use av_denoise_core::bench_api::kernels::nlm_bilateral;
+use av_denoise_core::bench_api::prefilter::bilateral_radius;
use cubecl::benchmark::Benchmark;
use cubecl::prelude::*;
@@ -8,21 +8,21 @@ use super::{
BILATERAL_SIGMA_S,
BLOCK_X,
BLOCK_Y,
- H,
+ HEIGHT,
InputOutput,
- W,
+ WIDTH,
block_sync,
cube_count_2d,
cube_dim_2d,
make_padded_frame,
- shapes_with_ch,
+ shapes_with_channels,
stored_channels,
};
pub struct BilateralBench {
pub client: ComputeClient,
- pub ch: u32,
- pub ch_name: &'static str,
+ pub channels: u32,
+ pub channel_name: &'static str,
}
impl Benchmark for BilateralBench {
@@ -30,11 +30,13 @@ impl Benchmark for BilateralBench {
type Output = ();
fn prepare(&self) -> Self::Input {
- let pixels = (W * H) as usize;
- let stored = stored_channels(self.ch) as usize;
- let frame = make_padded_frame(W, H, self.ch);
- let input = self.client.create_from_slice(f32::as_bytes(&frame));
- let output = self.client.empty(pixels * stored * size_of::());
+ let pixels = (WIDTH * HEIGHT) as usize;
+ let stored_ch = stored_channels(self.channels) as usize;
+ let frame = make_padded_frame(WIDTH, HEIGHT, self.channels);
+ let frame_bytes = f32::as_bytes(&frame);
+ let input = self.client.create_from_slice(frame_bytes);
+ let output = self.client.empty(pixels * stored_ch * size_of::());
+
InputOutput {
input,
output,
@@ -43,38 +45,44 @@ impl Benchmark for BilateralBench {
}
fn execute(&self, args: Self::Input) -> Result<(), String> {
- let pixels = (W * H) as usize;
- let stored = stored_channels(self.ch) as usize;
+ let pixels = (WIDTH * HEIGHT) as usize;
+ let stored_ch = stored_channels(self.channels) as usize;
let radius = bilateral_radius(BILATERAL_SIGMA_S);
+ let cube_count = cube_count_2d();
+ let cube_dim = cube_dim_2d();
+
unsafe {
nlm_bilateral::launch_unchecked::(
&self.client,
- cube_count_2d(),
- cube_dim_2d(),
- stored,
+ cube_count,
+ cube_dim,
+ stored_ch,
ArrayArg::from_raw_parts(args.input.clone(), args.frame_len),
- ArrayArg::from_raw_parts(args.output.clone(), pixels * stored),
+ ArrayArg::from_raw_parts(args.output.clone(), pixels * stored_ch),
0u32,
1.0 / (2.0 * BILATERAL_SIGMA_S * BILATERAL_SIGMA_S),
1.0 / (2.0 * BILATERAL_SIGMA_R * BILATERAL_SIGMA_R),
- W,
- H,
- self.ch,
+ WIDTH,
+ HEIGHT,
+ self.channels,
radius,
BLOCK_X,
BLOCK_Y,
);
}
+
Ok(())
}
fn name(&self) -> String {
- format!("bilateral_1080p_{}", self.ch_name)
+ format!("bilateral_1080p_{}", self.channel_name)
}
+
fn sync(&self) {
block_sync(&self.client);
}
+
fn shapes(&self) -> Vec> {
- shapes_with_ch(self.ch)
+ shapes_with_channels(self.channels)
}
}
diff --git a/av-denoise-core/benches/kernels/collab_aggregate.rs b/av-denoise-core/benches/kernels/collab_aggregate.rs
index a91a1c4..12047b9 100644
--- a/av-denoise-core/benches/kernels/collab_aggregate.rs
+++ b/av-denoise-core/benches/kernels/collab_aggregate.rs
@@ -1,18 +1,22 @@
-use av_denoise_core::collab::kernels::aggregate::{collab_normalise, collab_zero_accum};
+use av_denoise_core::bench_api::collab::kernels::aggregate::{collab_normalise, collab_zero_accum};
use cubecl::benchmark::Benchmark;
use cubecl::prelude::*;
use cubecl::server::Handle;
-use super::{BLOCK_X, BLOCK_Y, H, W, block_sync, stored_channels};
+use super::{BLOCK_X, BLOCK_Y, HEIGHT, WIDTH, block_sync, stored_channels};
-/// Divides the fixed-point accumulators the filters scattered into back
-/// out to a finished 1080p frame plane, for each channel mode. Cost
-/// scales with `stored_ch`, since one accumulator slot per channel is
-/// read for every pixel.
+/// The 65,535 workgroups per dimension GPU limit the library clamps to.
+///
+/// The library's own constant is crate-private, so a bench target spells it out.
+const MAX_GRID_1D: u32 = 65_535;
+
+/// Divides the fixed-point accumulators back out to a finished 1080p frame plane.
+///
+/// Cost scales with `stored_ch`, since one accumulator slot per channel is read for every pixel.
pub struct CollabNormaliseBench {
pub client: ComputeClient,
- pub ch: u32,
- pub ch_name: &'static str,
+ pub channels: u32,
+ pub channel_name: &'static str,
}
#[derive(Clone)]
@@ -27,48 +31,56 @@ impl Benchmark for CollabNormaliseBench {
type Output = ();
fn prepare(&self) -> Self::Input {
- let stored = stored_channels(self.ch) as usize;
- let pixels = (W * H) as usize;
-
- // Shaped like a real pass: a few dozen contributions per pixel,
- // each already scaled into the accumulator's fixed point.
- let accum_data: Vec = (0..pixels * stored).map(|i| (i % 97) as i32 * 8192).collect();
- let wsum_data: Vec = (0..pixels).map(|i| ((i % 31) + 20) as i32 * 8192).collect();
-
- CollabNormaliseInput {
- accum: self.client.create_from_slice(i32::as_bytes(&accum_data)),
- wsum: self.client.create_from_slice(i32::as_bytes(&wsum_data)),
- output: self.client.empty(pixels * stored * size_of::()),
- }
+ let stored_ch = stored_channels(self.channels) as usize;
+ let pixels = (WIDTH * HEIGHT) as usize;
+
+ // Shaped like a real pass, with a few dozen contributions per pixel, each already scaled
+ // into the accumulator's fixed point.
+ let accum_data: Vec = (0..pixels * stored_ch)
+ .map(|index| (index % 97) as i32 * 8192)
+ .collect();
+ let wsum_data: Vec = (0..pixels)
+ .map(|index| ((index % 31) + 20) as i32 * 8192)
+ .collect();
+
+ let accum_bytes = i32::as_bytes(&accum_data);
+ let accum = self.client.create_from_slice(accum_bytes);
+ let wsum_bytes = i32::as_bytes(&wsum_data);
+ let wsum = self.client.create_from_slice(wsum_bytes);
+ let output = self.client.empty(pixels * stored_ch * size_of::());
+
+ CollabNormaliseInput { accum, wsum, output }
}
fn execute(&self, args: Self::Input) -> Result<(), String> {
- let stored = stored_channels(self.ch) as usize;
- let pixels = (W * H) as usize;
+ let stored_ch = stored_channels(self.channels) as usize;
+ let pixels = (WIDTH * HEIGHT) as usize;
+ let cubes_x = WIDTH.div_ceil(BLOCK_X);
+ let cubes_y = HEIGHT.div_ceil(BLOCK_Y);
unsafe {
collab_normalise::launch_unchecked::(
&self.client,
- CubeCount::new_2d(W.div_ceil(BLOCK_X), H.div_ceil(BLOCK_Y)),
+ CubeCount::new_2d(cubes_x, cubes_y),
CubeDim::new_2d(BLOCK_X, BLOCK_Y),
- stored,
- ArrayArg::from_raw_parts(args.accum.clone(), pixels * stored),
+ stored_ch,
+ ArrayArg::from_raw_parts(args.accum.clone(), pixels * stored_ch),
ArrayArg::from_raw_parts(args.wsum.clone(), pixels),
- ArrayArg::from_raw_parts(args.output.clone(), pixels * stored),
- // A single-frame region. This bench measures one frame's
- // worth of normalisation, not a cross-frame ring.
+ ArrayArg::from_raw_parts(args.output.clone(), pixels * stored_ch),
+ // One frame's region, since this bench measures a single frame's normalisation.
0u32,
- W,
- H,
- self.ch,
- stored as u32,
+ WIDTH,
+ HEIGHT,
+ self.channels,
+ stored_ch as u32,
);
}
+
Ok(())
}
fn name(&self) -> String {
- format!("collab_normalise_1080p_{}", self.ch_name)
+ format!("collab_normalise_1080p_{}", self.channel_name)
}
fn sync(&self) {
@@ -76,15 +88,15 @@ impl Benchmark for CollabNormaliseBench {
}
fn shapes(&self) -> Vec> {
- vec![vec![W as usize, H as usize, self.ch as usize]]
+ vec![vec![WIDTH as usize, HEIGHT as usize, self.channels as usize]]
}
}
/// Clears both accumulators, which runs once before every filter pass.
pub struct CollabZeroAccumBench {
pub client: ComputeClient,
- pub ch: u32,
- pub ch_name: &'static str,
+ pub channels: u32,
+ pub channel_name: &'static str,
}
impl Benchmark for CollabZeroAccumBench {
@@ -92,46 +104,43 @@ impl Benchmark for CollabZeroAccumBench {
type Output = ();
fn prepare(&self) -> Self::Input {
- let stored = stored_channels(self.ch) as usize;
- let pixels = (W * H) as usize;
- CollabNormaliseInput {
- accum: self.client.empty(pixels * stored * size_of::()),
- wsum: self.client.empty(pixels * size_of::()),
- output: self.client.empty(size_of::()),
- }
+ let stored_ch = stored_channels(self.channels) as usize;
+ let pixels = (WIDTH * HEIGHT) as usize;
+ let accum = self.client.empty(pixels * stored_ch * size_of::());
+ let wsum = self.client.empty(pixels * size_of::());
+ let output = self.client.empty(size_of::());
+
+ CollabNormaliseInput { accum, wsum, output }
}
fn execute(&self, args: Self::Input) -> Result<(), String> {
- let stored = stored_channels(self.ch) as usize;
- let pixels = (W * H) as usize;
+ let stored_ch = stored_channels(self.channels) as usize;
+ let pixels = (WIDTH * HEIGHT) as usize;
let dim = 256u32;
- // The same 65,535-workgroups-per-dimension GPU limit the library
- // clamps to, spelled out here because its own constant is
- // crate-private and a bench is a separate crate target.
- // `collab_zero_accum` strides, so the clamp still reaches every
- // slot.
- const MAX_GRID_1D: u32 = 65_535;
- let grid = ((pixels * stored) as u32).div_ceil(dim).min(MAX_GRID_1D);
+
+ // `collab_zero_accum` strides, so the clamped grid still reaches every slot.
+ let grid = ((pixels * stored_ch) as u32).div_ceil(dim).min(MAX_GRID_1D);
unsafe {
collab_zero_accum::launch_unchecked::(
&self.client,
CubeCount::new_1d(grid),
CubeDim::new_1d(dim),
- ArrayArg::from_raw_parts(args.accum.clone(), pixels * stored),
+ ArrayArg::from_raw_parts(args.accum.clone(), pixels * stored_ch),
ArrayArg::from_raw_parts(args.wsum.clone(), pixels),
// A single-frame region, as a single-frame caller passes.
0u32,
pixels as u32,
- stored as u32,
+ stored_ch as u32,
grid * dim,
);
}
+
Ok(())
}
fn name(&self) -> String {
- format!("collab_zero_accum_1080p_{}", self.ch_name)
+ format!("collab_zero_accum_1080p_{}", self.channel_name)
}
fn sync(&self) {
@@ -139,6 +148,6 @@ impl Benchmark for CollabZeroAccumBench {
}
fn shapes(&self) -> Vec> {
- vec![vec![W as usize, H as usize, self.ch as usize]]
+ vec![vec![WIDTH as usize, HEIGHT as usize, self.channels as usize]]
}
}
diff --git a/av-denoise-core/benches/kernels/collab_fused.rs b/av-denoise-core/benches/kernels/collab_fused.rs
index 9021d7a..97898a0 100644
--- a/av-denoise-core/benches/kernels/collab_fused.rs
+++ b/av-denoise-core/benches/kernels/collab_fused.rs
@@ -1,9 +1,13 @@
-use av_denoise_core::collab::geometry::{fused_cubes_x, ref_count, refs_along, strength_map_dims};
-use av_denoise_core::collab::kernels::aggregate::{cross_frame_accum_scale, kaiser_window, weight_scale};
-use av_denoise_core::collab::kernels::fused::{STRENGTH_MAP_LUMA, STRENGTH_MAP_OFF, collab_fused};
-use av_denoise_core::collab::kernels::transforms::dct_noise_profile;
-use av_denoise_core::collab::{PATCH_SIZE, grid_frames, needs_warp_uniform_search};
-use av_denoise_core::nlmeans::NOISE_CURVE_BINS;
+use av_denoise_core::bench_api::NOISE_CURVE_BINS;
+use av_denoise_core::bench_api::collab::geometry::{fused_cubes_x, ref_count, refs_along, strength_map_dims};
+use av_denoise_core::bench_api::collab::kernels::aggregate::{
+ cross_frame_accum_scale,
+ kaiser_window,
+ weight_scale,
+};
+use av_denoise_core::bench_api::collab::kernels::fused::{STRENGTH_MAP_LUMA, STRENGTH_MAP_OFF, collab_fused};
+use av_denoise_core::bench_api::collab::kernels::transforms::dct_noise_profile;
+use av_denoise_core::bench_api::collab::{PATCH_SIZE, grid_frames, needs_warp_uniform_search};
use cubecl::benchmark::Benchmark;
use cubecl::prelude::*;
use cubecl::server::Handle;
@@ -23,40 +27,29 @@ use super::nl4d_geometry::{
conf_stride,
mv_stride,
};
-use super::{H, W, block_sync, make_padded_frame, shapes_with_ch, stored_channels};
+use super::{HEIGHT, WIDTH, block_sync, make_padded_frame, shapes_with_channels, stored_channels};
-/// The fused collaborative kernel at the library's default search
-/// geometry, over a 1080p frame ring. One 64-lane cube carries eight
-/// reference patches, eight lanes each, so the grid is an eighth as wide
-/// along x as the reference count.
-///
-/// This is the whole collaborative stage in one launch, matching,
-/// filtering and scatter.
+/// The fused collaborative kernel at the library's default search geometry over a 1080p frame ring.
///
-/// Confidence is uniformly 1.0, so no neighbour block is gated and every
-/// candidate the kernel finds runs the full patch comparison. Gating a
-/// block skips its comparisons entirely, so leaving it always open here
-/// measures the worst case. A bench that gates freely would report a
-/// time well under the real one.
+/// This is the whole collaborative stage in one launch, matching, filtering and scatter. One 64-lane
+/// cube carries eight reference patches of eight lanes each, so the grid is an eighth as wide along
+/// x as the reference count.
///
-/// `split_mv` picks which of the two motion fields the arm runs on, and
-/// the two bracket the real cost of the covering-block search.
+/// Confidence is uniformly 1.0, so no neighbour block is gated and every candidate runs the full
+/// patch comparison. Gating a block skips its comparisons entirely, so this measures the worst case
+/// and a bench that gates freely would report a time well under the real one.
///
-/// `false` gives a zeroed field. Every block covering a patch then
-/// predicts the same position, all four rectangles coincide, three of
-/// them are dropped by the duplicate check and no extra pixel
-/// comparison runs. That arm measures the duplicate check on its own.
-///
-/// `true` gives each block a vector from its own grid parity, spaced
-/// eight pixels apart, which is further than the refine window is wide.
-/// The four rectangles covering a patch are then disjoint, nothing
-/// deduplicates, and the neighbour search scores four times the
-/// positions. That arm is the worst case, and a real motion field lands
-/// between the two.
+/// `split_mv` picks one of two motion fields that bracket the real cost of the covering-block
+/// search. `false` gives a zeroed field. Every block covering a patch then predicts the same
+/// position, the duplicate check drops three of the four rectangles and no extra pixel comparison
+/// runs, so that arm measures the duplicate check on its own. `true` gives each block a vector from
+/// its own grid parity, eight pixels apart, which is further than the refine window is wide. The
+/// four rectangles are then disjoint, nothing deduplicates and the neighbour search scores four
+/// times the positions. That arm is the worst case, and a real motion field lands between the two.
pub struct CollabFusedBench {
pub client: ComputeClient,
- pub ch: u32,
- pub ch_name: &'static str,
+ pub channels: u32,
+ pub channel_name: &'static str,
pub split_mv: bool,
/// Launches with a stepped noise curve and `curve_valid = 1`.
pub noise_curve: bool,
@@ -69,21 +62,18 @@ pub struct CollabFusedBench {
/// The luma pool ratio at the default lambda. The kernel's cost does not depend on its value.
const POOL_RATIO: f32 = 2.2 / 3.78;
-/// A curve that doubles the luma threshold in the darker half and halves it
-/// in the brighter one.
+/// A curve that doubles the luma threshold in the darker half and halves it in the brighter one.
fn stepped_curve() -> [f32; NOISE_CURVE_BINS] {
let mut curve = [2.0f32; NOISE_CURVE_BINS];
curve[NOISE_CURVE_BINS / 2..].fill(0.5);
curve
}
-/// How far apart two neighbouring blocks' vectors sit in the split
-/// field, in pixels.
+/// How far apart two neighbouring blocks' vectors sit in the split field, in pixels.
///
-/// `REFINE` is the rectangle's half-width, so two rectangles stay
-/// disjoint once their centres are more than `2 * REFINE` apart. Eight
-/// clears that with room and keeps every predicted position well inside
-/// a 1080p frame.
+/// `REFINE` is the rectangle's half-width, so two rectangles stay disjoint once their centres are
+/// more than `2 * REFINE` apart. Eight clears that with room and keeps every predicted position
+/// well inside a 1080p frame.
const SPLIT_MV_SPACING: i32 = 8;
#[derive(Clone)]
@@ -94,8 +84,9 @@ pub struct CollabFusedInput {
pub neighbour_slots: Handle,
pub sigma: Handle,
pub dct_profile: Handle,
- /// The uniform aggregation window. The bench measures the kernel's
- /// own cost, and a taper changes none of the work it does.
+ /// The uniform aggregation window.
+ ///
+ /// The bench measures the kernel's own cost, and a taper changes none of the work it does.
pub kaiser: Handle,
pub accum: Handle,
pub wsum: Handle,
@@ -112,65 +103,76 @@ impl Benchmark for CollabFusedBench {
type Output = ();
fn prepare(&self) -> Self::Input {
- let stored_ch = stored_channels(self.ch);
- let pixels = (W * H) as usize;
+ let stored_ch = stored_channels(self.channels);
+ let pixels = (WIDTH * HEIGHT) as usize;
let frame_len = pixels * stored_ch as usize;
let mut ring_data = Vec::new();
for _ in 0..N_FRAMES {
- ring_data.extend(make_padded_frame(W, H, self.ch));
+ let frame = make_padded_frame(WIDTH, HEIGHT, self.channels);
+ ring_data.extend(frame);
}
- let ring = self.client.create_from_slice(f32::as_bytes(&ring_data));
- let blocks_x = W.div_ceil(BLK_STEP);
- let blocks_y = H.div_ceil(BLK_STEP);
+ let ring_bytes = f32::as_bytes(&ring_data);
+ let ring = self.client.create_from_slice(ring_bytes);
+
+ let blocks_x = WIDTH.div_ceil(BLK_STEP);
+ let blocks_y = HEIGHT.div_ceil(BLK_STEP);
let align = self.client.properties().memory.alignment;
- let mv_stride = mv_stride(blocks_x, blocks_y, align);
- let conf_stride = conf_stride(blocks_x, blocks_y, align);
+ let neighbour_mv_stride = mv_stride(blocks_x, blocks_y, align);
+ let neighbour_conf_stride = conf_stride(blocks_x, blocks_y, align);
- let mut mv_data = vec![0i32; (2 * RADIUS * mv_stride) as usize];
+ let mut mv_data = vec![0i32; (2 * RADIUS * neighbour_mv_stride) as usize];
if self.split_mv {
- for t in 0..2 * RADIUS {
- for by in 0..blocks_y {
- for bx in 0..blocks_x {
- let base = (t * mv_stride + (by * blocks_x + bx) * 2) as usize;
- mv_data[base] = (bx % 2) as i32 * SPLIT_MV_SPACING;
- mv_data[base + 1] = (by % 2) as i32 * SPLIT_MV_SPACING;
+ for slot in 0..2 * RADIUS {
+ for block_y in 0..blocks_y {
+ for block_x in 0..blocks_x {
+ let base = (slot * neighbour_mv_stride + (block_y * blocks_x + block_x) * 2) as usize;
+ mv_data[base] = (block_x % 2) as i32 * SPLIT_MV_SPACING;
+ mv_data[base + 1] = (block_y % 2) as i32 * SPLIT_MV_SPACING;
}
}
}
}
- let mv_field = self.client.create_from_slice(i32::as_bytes(&mv_data));
- let conf_data = vec![1.0f32; (2 * RADIUS * conf_stride) as usize];
- let confidence = self.client.create_from_slice(f32::as_bytes(&conf_data));
- let neighbour_slots = self.client.create_from_slice(u32::as_bytes(&NEIGHBOUR_SLOTS));
- // Sized for the stored lane count and filled for the logical
- // ones, matching what `Nl4dDenoiser` uploads each pass.
+ let mv_bytes = i32::as_bytes(&mv_data);
+ let mv_field = self.client.create_from_slice(mv_bytes);
+ let conf_data = vec![1.0f32; (2 * RADIUS * neighbour_conf_stride) as usize];
+ let conf_bytes = f32::as_bytes(&conf_data);
+ let confidence = self.client.create_from_slice(conf_bytes);
+ let slots_bytes = u32::as_bytes(&NEIGHBOUR_SLOTS);
+ let neighbour_slots = self.client.create_from_slice(slots_bytes);
+
+ // Sized for the stored lane count and filled for the logical ones, matching what
+ // `Nl4dDenoiser` uploads each pass.
let mut sigma_host = vec![0.0f32; stored_ch as usize];
- sigma_host[..self.ch as usize].fill(SIGMA);
- let sigma = self.client.create_from_slice(f32::as_bytes(&sigma_host));
- let dct_profile = self
- .client
- .create_from_slice(f32::as_bytes(&dct_noise_profile(0.0)));
- let kaiser = self.client.create_from_slice(f32::as_bytes(&kaiser_window(0.0)));
+ sigma_host[..self.channels as usize].fill(SIGMA);
+ let sigma_bytes = f32::as_bytes(&sigma_host);
+ let sigma = self.client.create_from_slice(sigma_bytes);
+ let profile_host = dct_noise_profile(0.0);
+ let profile_bytes = f32::as_bytes(&profile_host);
+ let dct_profile = self.client.create_from_slice(profile_bytes);
+ let kaiser_host = kaiser_window(0.0);
+ let kaiser_bytes = f32::as_bytes(&kaiser_host);
+ let kaiser = self.client.create_from_slice(kaiser_bytes);
- // One accumulator region per ring slot, the same shape
- // `Nl4dDenoiser` allocates, so the scatter crosses the same
- // address range it does in the pipeline.
+ // One accumulator region per ring slot, the same shape `Nl4dDenoiser` allocates, so the
+ // scatter crosses the same address range it does in the pipeline.
let accum = self
.client
.empty(frame_len * N_FRAMES as usize * size_of::());
let wsum = self.client.empty(pixels * N_FRAMES as usize * size_of::());
- let group_weight = self.client.empty(ref_count(W, H) * size_of::());
+ let refs = ref_count(WIDTH, HEIGHT);
+ let group_weight = self.client.empty(refs * size_of::());
let curve_host = if self.noise_curve {
stepped_curve()
} else {
[0.0f32; NOISE_CURVE_BINS]
};
- let noise_curve = self.client.create_from_slice(f32::as_bytes(&curve_host));
+ let curve_bytes = f32::as_bytes(&curve_host);
+ let noise_curve = self.client.create_from_slice(curve_bytes);
- let (map_cols, map_rows) = strength_map_dims(W, H);
+ let (map_cols, map_rows) = strength_map_dims(WIDTH, HEIGHT);
let map_len = (map_cols * map_rows) as usize;
let map_host: Vec = if self.strength_map {
(0..map_len)
@@ -179,7 +181,8 @@ impl Benchmark for CollabFusedBench {
} else {
vec![1.0f32; map_len]
};
- let strength_map = self.client.create_from_slice(f32::as_bytes(&map_host));
+ let map_bytes = f32::as_bytes(&map_host);
+ let strength_map = self.client.create_from_slice(map_bytes);
CollabFusedInput {
ring,
@@ -200,27 +203,35 @@ impl Benchmark for CollabFusedBench {
}
fn execute(&self, args: Self::Input) -> Result<(), String> {
- let stored_ch = stored_channels(self.ch);
- let pixels = (W * H) as usize;
+ let stored_ch = stored_channels(self.channels);
+ let pixels = (WIDTH * HEIGHT) as usize;
let frame_len = pixels * stored_ch as usize;
- let refs = ref_count(W, H);
- let refs_x = refs_along(W);
- let refs_y = refs_along(H);
+ let refs = ref_count(WIDTH, HEIGHT);
+ let refs_x = refs_along(WIDTH);
+ let refs_y = refs_along(HEIGHT);
- let blocks_x = W.div_ceil(BLK_STEP);
- let blocks_y = H.div_ceil(BLK_STEP);
+ let blocks_x = WIDTH.div_ceil(BLK_STEP);
+ let blocks_y = HEIGHT.div_ceil(BLK_STEP);
let align = self.client.properties().memory.alignment;
- let mv_stride = mv_stride(blocks_x, blocks_y, align);
- let conf_stride = conf_stride(blocks_x, blocks_y, align);
+ let neighbour_mv_stride = mv_stride(blocks_x, blocks_y, align);
+ let neighbour_conf_stride = conf_stride(blocks_x, blocks_y, align);
- let (map_cols, map_rows) = strength_map_dims(W, H);
+ let (map_cols, map_rows) = strength_map_dims(WIDTH, HEIGHT);
let map_mode = if self.strength_map {
STRENGTH_MAP_LUMA
} else {
STRENGTH_MAP_OFF
};
- let grid = CubeCount::new_2d(fused_cubes_x(W), refs_y);
+ let curve_valid = u32::from(self.noise_curve);
+ let dct_profile = dct_noise_profile(0.0);
+ let group_weight_scale = weight_scale(SIGMA, &dct_profile);
+ let accum_scale = cross_frame_accum_scale(SPATIAL_RADIUS, RADIUS);
+ let uniform_search = needs_warp_uniform_search(&self.client);
+ let frames_per_volume = grid_frames(RADIUS);
+
+ let cubes_x = fused_cubes_x(WIDTH);
+ let grid = CubeCount::new_2d(cubes_x, refs_y);
let dim = CubeDim::new_1d(64);
unsafe {
@@ -230,8 +241,11 @@ impl Benchmark for CollabFusedBench {
dim,
stored_ch as usize,
ArrayArg::from_raw_parts(args.ring.clone(), args.ring_len),
- ArrayArg::from_raw_parts(args.mv_field.clone(), (2 * RADIUS * mv_stride) as usize),
- ArrayArg::from_raw_parts(args.confidence.clone(), (2 * RADIUS * conf_stride) as usize),
+ ArrayArg::from_raw_parts(args.mv_field.clone(), (2 * RADIUS * neighbour_mv_stride) as usize),
+ ArrayArg::from_raw_parts(
+ args.confidence.clone(),
+ (2 * RADIUS * neighbour_conf_stride) as usize,
+ ),
ArrayArg::from_raw_parts(args.neighbour_slots.clone(), NEIGHBOUR_SLOTS.len()),
ArrayArg::from_raw_parts(args.sigma.clone(), stored_ch as usize),
ArrayArg::from_raw_parts(args.noise_curve.clone(), NOISE_CURVE_BINS),
@@ -244,23 +258,23 @@ impl Benchmark for CollabFusedBench {
CENTRE_SLOT,
0.0f32,
LAMBDA_HT,
- u32::from(self.noise_curve),
+ curve_valid,
map_mode,
- weight_scale(SIGMA, &dct_noise_profile(0.0)),
- cross_frame_accum_scale(SPATIAL_RADIUS, RADIUS),
- needs_warp_uniform_search(&self.client),
+ group_weight_scale,
+ accum_scale,
+ uniform_search,
RADIUS,
- grid_frames(RADIUS),
+ frames_per_volume,
REFINE,
- mv_stride,
- conf_stride,
+ neighbour_mv_stride,
+ neighbour_conf_stride,
BLK_STEP,
BLKSIZE,
blocks_x,
blocks_y,
- W,
- H,
- self.ch,
+ WIDTH,
+ HEIGHT,
+ self.channels,
K_MAX,
stored_ch,
SPATIAL_RADIUS,
@@ -271,6 +285,7 @@ impl Benchmark for CollabFusedBench {
self.pooled,
);
}
+
Ok(())
}
@@ -279,7 +294,10 @@ impl Benchmark for CollabFusedBench {
let curve = if self.noise_curve { "_noise_curve" } else { "" };
let map = if self.strength_map { "_strength_map" } else { "" };
let pool = if self.pooled { "_pooled" } else { "" };
- format!("collab_fused_1080p_{}{field}{curve}{map}{pool}", self.ch_name)
+ format!(
+ "collab_fused_1080p_{}{field}{curve}{map}{pool}",
+ self.channel_name
+ )
}
fn sync(&self) {
@@ -287,6 +305,6 @@ impl Benchmark for CollabFusedBench {
}
fn shapes(&self) -> Vec> {
- shapes_with_ch(self.ch)
+ shapes_with_channels(self.channels)
}
}
diff --git a/av-denoise-core/benches/kernels/copy.rs b/av-denoise-core/benches/kernels/copy.rs
index 0caa155..6804720 100644
--- a/av-denoise-core/benches/kernels/copy.rs
+++ b/av-denoise-core/benches/kernels/copy.rs
@@ -1,9 +1,18 @@
-use av_denoise_core::nlmeans::kernels::gpu_copy;
+use av_denoise_core::bench_api::kernels::gpu_copy;
use cubecl::benchmark::Benchmark;
use cubecl::prelude::*;
use cubecl::server::Handle;
-use super::{BLOCK_1D, COPY_GRID_1D, H, W, block_sync, make_padded_frame, shapes_with_ch, stored_channels};
+use super::{
+ BLOCK_1D,
+ COPY_GRID_1D,
+ HEIGHT,
+ WIDTH,
+ block_sync,
+ make_padded_frame,
+ shapes_with_channels,
+ stored_channels,
+};
#[derive(Clone)]
pub struct CopyInput {
@@ -13,8 +22,8 @@ pub struct CopyInput {
pub struct CopyBench {
pub client: ComputeClient,
- pub ch: u32,
- pub ch_name: &'static str,
+ pub channels: u32,
+ pub channel_name: &'static str,
}
impl Benchmark for CopyBench {
@@ -22,15 +31,19 @@ impl Benchmark for CopyBench {
type Output = ();
fn prepare(&self) -> Self::Input {
- let frame = make_padded_frame(W, H, self.ch);
- let src = self.client.create_from_slice(f32::as_bytes(&frame));
+ let frame = make_padded_frame(WIDTH, HEIGHT, self.channels);
+ let frame_bytes = f32::as_bytes(&frame);
+ let src = self.client.create_from_slice(frame_bytes);
let dst = self.client.empty(frame.len() * size_of::());
+
CopyInput { src, dst }
}
fn execute(&self, args: Self::Input) -> Result<(), String> {
- let len = (W * H) as usize * stored_channels(self.ch) as usize;
+ let stored_ch = stored_channels(self.channels) as usize;
+ let len = (WIDTH * HEIGHT) as usize * stored_ch;
let total_threads = COPY_GRID_1D * BLOCK_1D;
+
unsafe {
gpu_copy::launch_unchecked::(
&self.client,
@@ -44,16 +57,19 @@ impl Benchmark for CopyBench {
total_threads,
);
}
+
Ok(())
}
fn name(&self) -> String {
- format!("gpu_copy_1080p_{}", self.ch_name)
+ format!("gpu_copy_1080p_{}", self.channel_name)
}
+
fn sync(&self) {
block_sync(&self.client);
}
+
fn shapes(&self) -> Vec> {
- shapes_with_ch(self.ch)
+ shapes_with_channels(self.channels)
}
}
diff --git a/av-denoise-core/benches/kernels/dist_2d_weight.rs b/av-denoise-core/benches/kernels/dist_2d_weight.rs
index 2544b9b..a7d4852 100644
--- a/av-denoise-core/benches/kernels/dist_2d_weight.rs
+++ b/av-denoise-core/benches/kernels/dist_2d_weight.rs
@@ -1,29 +1,29 @@
-use av_denoise_core::nlmeans::kernels::nlm_dist_2d_weight;
+use av_denoise_core::bench_api::kernels::nlm_dist_2d_weight;
use cubecl::benchmark::Benchmark;
use cubecl::prelude::*;
use super::{
BLOCK_X,
BLOCK_Y,
- H,
+ HEIGHT,
InputOutput,
PATCH_RADIUS,
Q_X,
Q_Y,
- W,
+ WIDTH,
block_sync,
cube_count_2d,
cube_dim_2d,
h2_inv_norm,
make_padded_frame,
- shapes_with_ch,
+ shapes_with_channels,
stored_channels,
};
pub struct DistWeightBench {
pub client: ComputeClient,
- pub ch: u32,
- pub ch_name: &'static str,
+ pub channels: u32,
+ pub channel_name: &'static str,
}
impl Benchmark for DistWeightBench {
@@ -31,10 +31,12 @@ impl Benchmark for DistWeightBench {
type Output = ();
fn prepare(&self) -> Self::Input {
- let pixels = (W * H) as usize;
- let frame = make_padded_frame(W, H, self.ch);
- let input = self.client.create_from_slice(f32::as_bytes(&frame));
+ let pixels = (WIDTH * HEIGHT) as usize;
+ let frame = make_padded_frame(WIDTH, HEIGHT, self.channels);
+ let frame_bytes = f32::as_bytes(&frame);
+ let input = self.client.create_from_slice(frame_bytes);
let output = self.client.empty(pixels * size_of::());
+
InputOutput {
input,
output,
@@ -43,40 +45,47 @@ impl Benchmark for DistWeightBench {
}
fn execute(&self, args: Self::Input) -> Result<(), String> {
- let pixels = (W * H) as usize;
- let stored = stored_channels(self.ch) as usize;
+ let pixels = (WIDTH * HEIGHT) as usize;
+ let stored_ch = stored_channels(self.channels) as usize;
+ let cube_count = cube_count_2d();
+ let cube_dim = cube_dim_2d();
+ let inv_norm = h2_inv_norm();
+
unsafe {
nlm_dist_2d_weight::launch_unchecked::(
&self.client,
- cube_count_2d(),
- cube_dim_2d(),
- stored,
+ cube_count,
+ cube_dim,
+ stored_ch,
ArrayArg::from_raw_parts(args.input.clone(), args.frame_len),
ArrayArg::from_raw_parts(args.output.clone(), pixels),
0u32,
0u32,
Q_X,
Q_Y,
- h2_inv_norm(),
+ inv_norm,
0.0f32,
- W,
- H,
- self.ch,
+ WIDTH,
+ HEIGHT,
+ self.channels,
PATCH_RADIUS,
BLOCK_X,
BLOCK_Y,
);
}
+
Ok(())
}
fn name(&self) -> String {
- format!("dist_2d_weight_1080p_{}", self.ch_name)
+ format!("dist_2d_weight_1080p_{}", self.channel_name)
}
+
fn sync(&self) {
block_sync(&self.client);
}
+
fn shapes(&self) -> Vec> {
- shapes_with_ch(self.ch)
+ shapes_with_channels(self.channels)
}
}
diff --git a/av-denoise-core/benches/kernels/dist_2d_weight_ref.rs b/av-denoise-core/benches/kernels/dist_2d_weight_ref.rs
index 52c6650..6dcb8e3 100644
--- a/av-denoise-core/benches/kernels/dist_2d_weight_ref.rs
+++ b/av-denoise-core/benches/kernels/dist_2d_weight_ref.rs
@@ -1,29 +1,29 @@
-use av_denoise_core::nlmeans::kernels::nlm_dist_2d_weight_ref;
+use av_denoise_core::bench_api::kernels::nlm_dist_2d_weight_ref;
use cubecl::benchmark::Benchmark;
use cubecl::prelude::*;
use super::{
BLOCK_X,
BLOCK_Y,
- H,
+ HEIGHT,
InputOutput,
PATCH_RADIUS,
Q_X,
Q_Y,
- W,
+ WIDTH,
block_sync,
cube_count_2d,
cube_dim_2d,
h2_inv_norm,
make_padded_frame,
- shapes_with_ch,
+ shapes_with_channels,
stored_channels,
};
pub struct DistWeightRefBench {
pub client: ComputeClient,
- pub ch: u32,
- pub ch_name: &'static str,
+ pub channels: u32,
+ pub channel_name: &'static str,
}
impl Benchmark for DistWeightRefBench {
@@ -31,10 +31,12 @@ impl Benchmark for DistWeightRefBench {
type Output = ();
fn prepare(&self) -> Self::Input {
- let pixels = (W * H) as usize;
- let frame = make_padded_frame(W, H, self.ch);
- let input = self.client.create_from_slice(f32::as_bytes(&frame));
+ let pixels = (WIDTH * HEIGHT) as usize;
+ let frame = make_padded_frame(WIDTH, HEIGHT, self.channels);
+ let frame_bytes = f32::as_bytes(&frame);
+ let input = self.client.create_from_slice(frame_bytes);
let output = self.client.empty(pixels * size_of::());
+
InputOutput {
input,
output,
@@ -43,40 +45,47 @@ impl Benchmark for DistWeightRefBench {
}
fn execute(&self, args: Self::Input) -> Result<(), String> {
- let pixels = (W * H) as usize;
- let stored = stored_channels(self.ch) as usize;
+ let pixels = (WIDTH * HEIGHT) as usize;
+ let stored_ch = stored_channels(self.channels) as usize;
+ let cube_count = cube_count_2d();
+ let cube_dim = cube_dim_2d();
+ let inv_norm = h2_inv_norm();
+
unsafe {
nlm_dist_2d_weight_ref::launch_unchecked::(
&self.client,
- cube_count_2d(),
- cube_dim_2d(),
- stored,
+ cube_count,
+ cube_dim,
+ stored_ch,
ArrayArg::from_raw_parts(args.input.clone(), args.frame_len),
ArrayArg::from_raw_parts(args.output.clone(), pixels),
0u32,
0u32,
Q_X,
Q_Y,
- h2_inv_norm(),
+ inv_norm,
0.0f32,
- W,
- H,
- self.ch,
+ WIDTH,
+ HEIGHT,
+ self.channels,
PATCH_RADIUS,
BLOCK_X,
BLOCK_Y,
);
}
+
Ok(())
}
fn name(&self) -> String {
- format!("dist_2d_weight_ref_1080p_{}", self.ch_name)
+ format!("dist_2d_weight_ref_1080p_{}", self.channel_name)
}
+
fn sync(&self) {
block_sync(&self.client);
}
+
fn shapes(&self) -> Vec> {
- shapes_with_ch(self.ch)
+ shapes_with_channels(self.channels)
}
}
diff --git a/av-denoise-core/benches/kernels/distance.rs b/av-denoise-core/benches/kernels/distance.rs
index 1bef8a2..ce0f852 100644
--- a/av-denoise-core/benches/kernels/distance.rs
+++ b/av-denoise-core/benches/kernels/distance.rs
@@ -1,18 +1,18 @@
-use av_denoise_core::nlmeans::kernels::nlm_distance;
+use av_denoise_core::bench_api::kernels::nlm_distance;
use cubecl::benchmark::Benchmark;
use cubecl::prelude::*;
use cubecl::server::Handle;
use super::{
- H,
+ HEIGHT,
Q_X,
Q_Y,
- W,
+ WIDTH,
block_sync,
cube_count_2d,
cube_dim_2d,
make_padded_frame,
- shapes_with_ch,
+ shapes_with_channels,
stored_channels,
};
@@ -25,8 +25,8 @@ pub struct DistanceInput {
pub struct DistanceBench {
pub client: ComputeClient,
- pub ch: u32,
- pub ch_name: &'static str,
+ pub channels: u32,
+ pub channel_name: &'static str,
}
impl Benchmark for DistanceBench {
@@ -34,10 +34,12 @@ impl Benchmark for DistanceBench {
type Output = ();
fn prepare(&self) -> Self::Input {
- let pixels = (W * H) as usize;
- let frame = make_padded_frame(W, H, self.ch);
- let input = self.client.create_from_slice(f32::as_bytes(&frame));
+ let pixels = (WIDTH * HEIGHT) as usize;
+ let frame = make_padded_frame(WIDTH, HEIGHT, self.channels);
+ let frame_bytes = f32::as_bytes(&frame);
+ let input = self.client.create_from_slice(frame_bytes);
let dist = self.client.empty(pixels * size_of::());
+
DistanceInput {
input,
dist,
@@ -46,35 +48,41 @@ impl Benchmark for DistanceBench {
}
fn execute(&self, args: Self::Input) -> Result<(), String> {
- let pixels = (W * H) as usize;
- let stored = stored_channels(self.ch) as usize;
+ let pixels = (WIDTH * HEIGHT) as usize;
+ let stored_ch = stored_channels(self.channels) as usize;
+ let cube_count = cube_count_2d();
+ let cube_dim = cube_dim_2d();
+
unsafe {
nlm_distance::launch_unchecked::(
&self.client,
- cube_count_2d(),
- cube_dim_2d(),
- stored,
+ cube_count,
+ cube_dim,
+ stored_ch,
ArrayArg::from_raw_parts(args.input.clone(), args.frame_len),
ArrayArg::from_raw_parts(args.dist.clone(), pixels),
0u32,
0u32,
Q_X,
Q_Y,
- W,
- H,
- self.ch,
+ WIDTH,
+ HEIGHT,
+ self.channels,
);
}
+
Ok(())
}
fn name(&self) -> String {
- format!("distance_1080p_{}", self.ch_name)
+ format!("distance_1080p_{}", self.channel_name)
}
+
fn sync(&self) {
block_sync(&self.client);
}
+
fn shapes(&self) -> Vec> {
- shapes_with_ch(self.ch)
+ shapes_with_channels(self.channels)
}
}
diff --git a/av-denoise-core/benches/kernels/distance_pair.rs b/av-denoise-core/benches/kernels/distance_pair.rs
index 0c1a9da..dddc3cf 100644
--- a/av-denoise-core/benches/kernels/distance_pair.rs
+++ b/av-denoise-core/benches/kernels/distance_pair.rs
@@ -1,18 +1,18 @@
-use av_denoise_core::nlmeans::kernels::nlm_distance_pair;
+use av_denoise_core::bench_api::kernels::nlm_distance_pair;
use cubecl::benchmark::Benchmark;
use cubecl::prelude::*;
use cubecl::server::Handle;
use super::{
- H,
+ HEIGHT,
Q_X,
Q_Y,
- W,
+ WIDTH,
block_sync,
cube_count_2d,
cube_dim_2d,
make_padded_frame,
- shapes_with_ch,
+ shapes_with_channels,
stored_channels,
};
@@ -26,8 +26,8 @@ pub struct DistancePairInput {
pub struct DistancePairBench {
pub client: ComputeClient,
- pub ch: u32,
- pub ch_name: &'static str,
+ pub channels: u32,
+ pub channel_name: &'static str,
}
impl Benchmark for DistancePairBench {
@@ -35,11 +35,13 @@ impl Benchmark for DistancePairBench {
type Output = ();
fn prepare(&self) -> Self::Input {
- let pixels = (W * H) as usize;
- let frame = make_padded_frame(W, H, self.ch);
- let input = self.client.create_from_slice(f32::as_bytes(&frame));
+ let pixels = (WIDTH * HEIGHT) as usize;
+ let frame = make_padded_frame(WIDTH, HEIGHT, self.channels);
+ let frame_bytes = f32::as_bytes(&frame);
+ let input = self.client.create_from_slice(frame_bytes);
let dist_fwd = self.client.empty(pixels * size_of::());
let dist_bwd = self.client.empty(pixels * size_of::());
+
DistancePairInput {
input,
dist_fwd,
@@ -49,14 +51,17 @@ impl Benchmark for DistancePairBench {
}
fn execute(&self, args: Self::Input) -> Result<(), String> {
- let pixels = (W * H) as usize;
- let stored = stored_channels(self.ch) as usize;
+ let pixels = (WIDTH * HEIGHT) as usize;
+ let stored_ch = stored_channels(self.channels) as usize;
+ let cube_count = cube_count_2d();
+ let cube_dim = cube_dim_2d();
+
unsafe {
nlm_distance_pair::launch_unchecked::(
&self.client,
- cube_count_2d(),
- cube_dim_2d(),
- stored,
+ cube_count,
+ cube_dim,
+ stored_ch,
ArrayArg::from_raw_parts(args.input.clone(), args.frame_len),
ArrayArg::from_raw_parts(args.dist_fwd.clone(), pixels),
ArrayArg::from_raw_parts(args.dist_bwd.clone(), pixels),
@@ -65,21 +70,24 @@ impl Benchmark for DistancePairBench {
0u32,
Q_X,
Q_Y,
- W,
- H,
- self.ch,
+ WIDTH,
+ HEIGHT,
+ self.channels,
);
}
+
Ok(())
}
fn name(&self) -> String {
- format!("distance_pair_1080p_{}", self.ch_name)
+ format!("distance_pair_1080p_{}", self.channel_name)
}
+
fn sync(&self) {
block_sync(&self.client);
}
+
fn shapes(&self) -> Vec> {
- shapes_with_ch(self.ch)
+ shapes_with_channels(self.channels)
}
}
diff --git a/av-denoise-core/benches/kernels/distance_pair_ref.rs b/av-denoise-core/benches/kernels/distance_pair_ref.rs
index 445b4c9..f75d7c1 100644
--- a/av-denoise-core/benches/kernels/distance_pair_ref.rs
+++ b/av-denoise-core/benches/kernels/distance_pair_ref.rs
@@ -1,25 +1,25 @@
-use av_denoise_core::nlmeans::kernels::nlm_distance_pair_ref;
+use av_denoise_core::bench_api::kernels::nlm_distance_pair_ref;
use cubecl::benchmark::Benchmark;
use cubecl::prelude::*;
use super::distance_pair::DistancePairInput;
use super::{
- H,
+ HEIGHT,
Q_X,
Q_Y,
- W,
+ WIDTH,
block_sync,
cube_count_2d,
cube_dim_2d,
make_padded_frame,
- shapes_with_ch,
+ shapes_with_channels,
stored_channels,
};
pub struct DistancePairRefBench {
pub client: ComputeClient,
- pub ch: u32,
- pub ch_name: &'static str,
+ pub channels: u32,
+ pub channel_name: &'static str,
}
impl Benchmark for DistancePairRefBench {
@@ -27,11 +27,13 @@ impl Benchmark for DistancePairRefBench {
type Output = ();
fn prepare(&self) -> Self::Input {
- let pixels = (W * H) as usize;
- let frame = make_padded_frame(W, H, self.ch);
- let input = self.client.create_from_slice(f32::as_bytes(&frame));
+ let pixels = (WIDTH * HEIGHT) as usize;
+ let frame = make_padded_frame(WIDTH, HEIGHT, self.channels);
+ let frame_bytes = f32::as_bytes(&frame);
+ let input = self.client.create_from_slice(frame_bytes);
let dist_fwd = self.client.empty(pixels * size_of::());
let dist_bwd = self.client.empty(pixels * size_of::());
+
DistancePairInput {
input,
dist_fwd,
@@ -41,14 +43,17 @@ impl Benchmark for DistancePairRefBench {
}
fn execute(&self, args: Self::Input) -> Result<(), String> {
- let pixels = (W * H) as usize;
- let stored = stored_channels(self.ch) as usize;
+ let pixels = (WIDTH * HEIGHT) as usize;
+ let stored_ch = stored_channels(self.channels) as usize;
+ let cube_count = cube_count_2d();
+ let cube_dim = cube_dim_2d();
+
unsafe {
nlm_distance_pair_ref::launch_unchecked::(
&self.client,
- cube_count_2d(),
- cube_dim_2d(),
- stored,
+ cube_count,
+ cube_dim,
+ stored_ch,
ArrayArg::from_raw_parts(args.input.clone(), args.frame_len),
ArrayArg::from_raw_parts(args.dist_fwd.clone(), pixels),
ArrayArg::from_raw_parts(args.dist_bwd.clone(), pixels),
@@ -57,21 +62,24 @@ impl Benchmark for DistancePairRefBench {
0u32,
Q_X,
Q_Y,
- W,
- H,
- self.ch,
+ WIDTH,
+ HEIGHT,
+ self.channels,
);
}
+
Ok(())
}
fn name(&self) -> String {
- format!("distance_pair_ref_1080p_{}", self.ch_name)
+ format!("distance_pair_ref_1080p_{}", self.channel_name)
}
+
fn sync(&self) {
block_sync(&self.client);
}
+
fn shapes(&self) -> Vec> {
- shapes_with_ch(self.ch)
+ shapes_with_channels(self.channels)
}
}
diff --git a/av-denoise-core/benches/kernels/distance_ref.rs b/av-denoise-core/benches/kernels/distance_ref.rs
index 3b03f51..fa9e993 100644
--- a/av-denoise-core/benches/kernels/distance_ref.rs
+++ b/av-denoise-core/benches/kernels/distance_ref.rs
@@ -1,25 +1,25 @@
-use av_denoise_core::nlmeans::kernels::nlm_distance_ref;
+use av_denoise_core::bench_api::kernels::nlm_distance_ref;
use cubecl::benchmark::Benchmark;
use cubecl::prelude::*;
use super::distance::DistanceInput;
use super::{
- H,
+ HEIGHT,
Q_X,
Q_Y,
- W,
+ WIDTH,
block_sync,
cube_count_2d,
cube_dim_2d,
make_padded_frame,
- shapes_with_ch,
+ shapes_with_channels,
stored_channels,
};
pub struct DistanceRefBench {
pub client: ComputeClient,
- pub ch: u32,
- pub ch_name: &'static str,
+ pub channels: u32,
+ pub channel_name: &'static str,
}
impl Benchmark for DistanceRefBench {
@@ -27,10 +27,12 @@ impl Benchmark for DistanceRefBench {
type Output = ();
fn prepare(&self) -> Self::Input {
- let pixels = (W * H) as usize;
- let frame = make_padded_frame(W, H, self.ch);
- let input = self.client.create_from_slice(f32::as_bytes(&frame));
+ let pixels = (WIDTH * HEIGHT) as usize;
+ let frame = make_padded_frame(WIDTH, HEIGHT, self.channels);
+ let frame_bytes = f32::as_bytes(&frame);
+ let input = self.client.create_from_slice(frame_bytes);
let dist = self.client.empty(pixels * size_of::());
+
DistanceInput {
input,
dist,
@@ -39,35 +41,41 @@ impl Benchmark for DistanceRefBench {
}
fn execute(&self, args: Self::Input) -> Result<(), String> {
- let pixels = (W * H) as usize;
- let stored = stored_channels(self.ch) as usize;
+ let pixels = (WIDTH * HEIGHT) as usize;
+ let stored_ch = stored_channels(self.channels) as usize;
+ let cube_count = cube_count_2d();
+ let cube_dim = cube_dim_2d();
+
unsafe {
nlm_distance_ref::launch_unchecked::(
&self.client,
- cube_count_2d(),
- cube_dim_2d(),
- stored,
+ cube_count,
+ cube_dim,
+ stored_ch,
ArrayArg::from_raw_parts(args.input.clone(), args.frame_len),
ArrayArg::from_raw_parts(args.dist.clone(), pixels),
0u32,
0u32,
Q_X,
Q_Y,
- W,
- H,
- self.ch,
+ WIDTH,
+ HEIGHT,
+ self.channels,
);
}
+
Ok(())
}
fn name(&self) -> String {
- format!("distance_ref_1080p_{}", self.ch_name)
+ format!("distance_ref_1080p_{}", self.channel_name)
}
+
fn sync(&self) {
block_sync(&self.client);
}
+
fn shapes(&self) -> Vec> {
- shapes_with_ch(self.ch)
+ shapes_with_channels(self.channels)
}
}
diff --git a/av-denoise-core/benches/kernels/egress.rs b/av-denoise-core/benches/kernels/egress.rs
new file mode 100644
index 0000000..7bc2825
--- /dev/null
+++ b/av-denoise-core/benches/kernels/egress.rs
@@ -0,0 +1,165 @@
+use av_denoise_core::bench_api::engine_kernels::{egress_f32, egress_words};
+use cubecl::benchmark::Benchmark;
+use cubecl::prelude::*;
+use cubecl::server::Handle;
+
+use super::{BLOCK_1D, block_sync, make_padded_frame, stored_channels};
+
+/// How the planes under test are stored.
+#[derive(Clone, Copy, Debug)]
+pub enum EgressFormat {
+ U8,
+ U16Ten,
+ F32,
+}
+
+impl EgressFormat {
+ fn samples_per_word(self) -> u32 {
+ match self {
+ EgressFormat::U8 => 4,
+ EgressFormat::U16Ten => 2,
+ EgressFormat::F32 => 1,
+ }
+ }
+
+ fn max(self) -> f32 {
+ match self {
+ EgressFormat::U8 => 255.0,
+ EgressFormat::U16Ten => 1023.0,
+ EgressFormat::F32 => 1.0,
+ }
+ }
+}
+
+#[derive(Clone)]
+pub struct EgressInput {
+ frame: Handle,
+ planes: Vec,
+ placeholder: Handle,
+}
+
+pub struct EgressBench {
+ pub client: ComputeClient,
+ pub width: u32,
+ pub height: u32,
+ pub channels: u32,
+ pub channel_name: &'static str,
+ pub format: EgressFormat,
+}
+
+impl EgressBench {
+ fn pixels(&self) -> u32 {
+ self.width * self.height
+ }
+
+ fn words(&self) -> u32 {
+ let samples_per_word = self.format.samples_per_word();
+ self.pixels().div_ceil(samples_per_word)
+ }
+
+ fn frame_len(&self) -> usize {
+ let stored_ch = stored_channels(self.channels);
+ self.pixels() as usize * stored_ch as usize
+ }
+}
+
+impl Benchmark for EgressBench {
+ type Input = EgressInput;
+ type Output = ();
+
+ fn prepare(&self) -> Self::Input {
+ let frame_data = make_padded_frame(self.width, self.height, self.channels);
+ let frame_bytes = f32::as_bytes(&frame_data);
+ let frame = self.client.create_from_slice(frame_bytes);
+ let plane_bytes = self.words() as usize * size_of::();
+ let planes = (0..self.channels)
+ .map(|_| self.client.empty(plane_bytes))
+ .collect();
+ let placeholder = self.client.empty(size_of::());
+
+ EgressInput {
+ frame,
+ planes,
+ placeholder,
+ }
+ }
+
+ fn execute(&self, args: Self::Input) -> Result<(), String> {
+ let pixels = self.pixels();
+ let stored_ch = stored_channels(self.channels);
+ let frame_len = self.frame_len();
+ let samples_per_word = self.format.samples_per_word();
+ let max = self.format.max();
+ let words = self.words();
+ let plane_len = match self.format {
+ EgressFormat::F32 => pixels as usize,
+ _ => words as usize,
+ };
+ let threads = match self.format {
+ EgressFormat::F32 => pixels,
+ _ => words,
+ };
+ let groups = threads.div_ceil(BLOCK_1D).clamp(1, 65535);
+ let total_threads = groups * BLOCK_1D;
+
+ // Planes past `channels` bind a distinct placeholder the kernel never writes.
+ let plane_0 = args.planes[0].clone();
+ let plane_1 = args.planes.get(1).unwrap_or(&args.placeholder).clone();
+ let plane_2 = args.planes.get(2).unwrap_or(&args.placeholder).clone();
+
+ unsafe {
+ match self.format {
+ EgressFormat::F32 => egress_f32::launch_unchecked::(
+ &self.client,
+ CubeCount::new_1d(groups),
+ CubeDim::new_1d(BLOCK_1D),
+ ArrayArg::from_raw_parts(args.frame.clone(), frame_len),
+ ArrayArg::from_raw_parts(plane_0, plane_len),
+ ArrayArg::from_raw_parts(plane_1, plane_len),
+ ArrayArg::from_raw_parts(plane_2, plane_len),
+ pixels,
+ self.channels,
+ stored_ch,
+ total_threads,
+ ),
+ EgressFormat::U8 | EgressFormat::U16Ten => egress_words::launch_unchecked::(
+ &self.client,
+ CubeCount::new_1d(groups),
+ CubeDim::new_1d(BLOCK_1D),
+ ArrayArg::from_raw_parts(args.frame.clone(), frame_len),
+ ArrayArg::from_raw_parts(plane_0, plane_len),
+ ArrayArg::from_raw_parts(plane_1, plane_len),
+ ArrayArg::from_raw_parts(plane_2, plane_len),
+ max,
+ pixels,
+ self.channels,
+ stored_ch,
+ samples_per_word,
+ words,
+ total_threads,
+ ),
+ }
+ }
+
+ Ok(())
+ }
+
+ fn name(&self) -> String {
+ format!(
+ "egress_{}x{}_{:?}_{}",
+ self.width, self.height, self.format, self.channel_name
+ )
+ }
+
+ fn sync(&self) {
+ block_sync(&self.client);
+ }
+
+ fn shapes(&self) -> Vec> {
+ vec![vec![
+ self.width as usize,
+ self.height as usize,
+ self.channels as usize,
+ ]]
+ }
+}
diff --git a/av-denoise-core/benches/kernels/finish.rs b/av-denoise-core/benches/kernels/finish.rs
index 5a90688..56408c5 100644
--- a/av-denoise-core/benches/kernels/finish.rs
+++ b/av-denoise-core/benches/kernels/finish.rs
@@ -1,16 +1,16 @@
-use av_denoise_core::nlmeans::kernels::nlm_finish;
+use av_denoise_core::bench_api::kernels::nlm_finish;
use cubecl::benchmark::Benchmark;
use cubecl::prelude::*;
use cubecl::server::Handle;
use super::{
- H,
- W,
+ HEIGHT,
+ WIDTH,
block_sync,
cube_count_2d,
cube_dim_2d,
make_padded_frame,
- shapes_with_ch,
+ shapes_with_channels,
stored_channels,
};
@@ -26,8 +26,8 @@ pub struct FinishInput {
pub struct FinishBench {
pub client: ComputeClient,
- pub ch: u32,
- pub ch_name: &'static str,
+ pub channels: u32,
+ pub channel_name: &'static str,
}
impl Benchmark for FinishBench {
@@ -35,17 +35,26 @@ impl Benchmark for FinishBench {
type Output = ();
fn prepare(&self) -> Self::Input {
- let pixels = (W * H) as usize;
- let stored = stored_channels(self.ch) as usize;
- let frame = make_padded_frame(W, H, self.ch);
- let input = self.client.create_from_slice(f32::as_bytes(&frame));
- let accum_data = vec![0.25f32; pixels * stored];
- let accum = self.client.create_from_slice(f32::as_bytes(&accum_data));
- let ws_data = vec![1.0f32; pixels];
- let weight_sum = self.client.create_from_slice(f32::as_bytes(&ws_data));
- let mw_data = vec![0.8f32; pixels];
- let max_weight = self.client.create_from_slice(f32::as_bytes(&mw_data));
- let output = self.client.empty(pixels * stored * size_of::());
+ let pixels = (WIDTH * HEIGHT) as usize;
+ let stored_ch = stored_channels(self.channels) as usize;
+ let frame = make_padded_frame(WIDTH, HEIGHT, self.channels);
+ let frame_bytes = f32::as_bytes(&frame);
+ let input = self.client.create_from_slice(frame_bytes);
+
+ let accum_data = vec![0.25f32; pixels * stored_ch];
+ let accum_bytes = f32::as_bytes(&accum_data);
+ let accum = self.client.create_from_slice(accum_bytes);
+
+ let weight_sum_data = vec![1.0f32; pixels];
+ let weight_sum_bytes = f32::as_bytes(&weight_sum_data);
+ let weight_sum = self.client.create_from_slice(weight_sum_bytes);
+
+ let max_weight_data = vec![0.8f32; pixels];
+ let max_weight_bytes = f32::as_bytes(&max_weight_data);
+ let max_weight = self.client.create_from_slice(max_weight_bytes);
+
+ let output = self.client.empty(pixels * stored_ch * size_of::());
+
FinishInput {
input,
output,
@@ -57,37 +66,43 @@ impl Benchmark for FinishBench {
}
fn execute(&self, args: Self::Input) -> Result<(), String> {
- let pixels = (W * H) as usize;
- let stored = stored_channels(self.ch) as usize;
+ let pixels = (WIDTH * HEIGHT) as usize;
+ let stored_ch = stored_channels(self.channels) as usize;
+ let cube_count = cube_count_2d();
+ let cube_dim = cube_dim_2d();
+
unsafe {
nlm_finish::launch_unchecked::(
&self.client,
- cube_count_2d(),
- cube_dim_2d(),
- stored,
+ cube_count,
+ cube_dim,
+ stored_ch,
ArrayArg::from_raw_parts(args.input.clone(), args.frame_len),
- ArrayArg::from_raw_parts(args.output.clone(), pixels * stored),
- ArrayArg::from_raw_parts(args.accum.clone(), pixels * stored),
+ ArrayArg::from_raw_parts(args.output.clone(), pixels * stored_ch),
+ ArrayArg::from_raw_parts(args.accum.clone(), pixels * stored_ch),
ArrayArg::from_raw_parts(args.weight_sum.clone(), pixels),
ArrayArg::from_raw_parts(args.max_weight.clone(), pixels),
0u32,
0u32,
1.0f32,
- W,
- H,
- self.ch,
+ WIDTH,
+ HEIGHT,
+ self.channels,
);
}
+
Ok(())
}
fn name(&self) -> String {
- format!("finish_1080p_{}", self.ch_name)
+ format!("finish_1080p_{}", self.channel_name)
}
+
fn sync(&self) {
block_sync(&self.client);
}
+
fn shapes(&self) -> Vec> {
- shapes_with_ch(self.ch)
+ shapes_with_channels(self.channels)
}
}
diff --git a/av-denoise-core/benches/kernels/fused_window.rs b/av-denoise-core/benches/kernels/fused_window.rs
index ef514eb..961f3c5 100644
--- a/av-denoise-core/benches/kernels/fused_window.rs
+++ b/av-denoise-core/benches/kernels/fused_window.rs
@@ -1,4 +1,4 @@
-use av_denoise_core::nlmeans::kernels::{
+use av_denoise_core::bench_api::kernels::{
nlm_fused_pair_accumulate_window,
nlm_fused_pair_accumulate_window_ref,
nlm_fused_single_window,
@@ -11,27 +11,28 @@ use cubecl::server::Handle;
use super::{
BLOCK_X,
BLOCK_Y,
- H,
+ HEIGHT,
PATCH_RADIUS,
SEARCH_RADIUS,
- W,
+ WIDTH,
block_sync,
cube_count_2d,
cube_dim_2d,
h2_inv_norm,
make_padded_frame,
- shapes_with_ch,
+ shapes_with_channels,
stored_channels,
};
-/// Zero-filled spatial-offset LUT for `SEARCH_RADIUS`. Zero everywhere
-/// reproduces the old flat `noise_offset = 0.0` the single-window
-/// benches measured before the LUT replaced that scalar, so the
-/// timing stays comparable.
+/// A zero-filled spatial-offset LUT for `SEARCH_RADIUS`.
+///
+/// Zero everywhere applies no noise offset, matching the `0.0` `noise_offset` the pair-window rows pass.
fn zero_spatial_offset_lut(client: &ComputeClient) -> (Handle, usize) {
let side = (2 * SEARCH_RADIUS + 1) as usize;
let lut = vec![0.0f32; side * side];
- let handle = client.create_from_slice(f32::as_bytes(&lut));
+ let lut_bytes = f32::as_bytes(&lut);
+ let handle = client.create_from_slice(lut_bytes);
+
(handle, lut.len())
}
@@ -60,16 +61,18 @@ pub struct WindowRefInput {
frame_len: usize,
}
-fn prepare_window(client: &ComputeClient, ch: u32) -> WindowInput {
- let pixels = (W * H) as usize;
- let stored = stored_channels(ch) as usize;
- let frame = make_padded_frame(W, H, ch);
- let input = client.create_from_slice(f32::as_bytes(&frame));
- let accum = client.empty(pixels * stored * size_of::());
+fn prepare_window(client: &ComputeClient, channels: u32) -> WindowInput {
+ let pixels = (WIDTH * HEIGHT) as usize;
+ let stored_ch = stored_channels(channels) as usize;
+ let frame = make_padded_frame(WIDTH, HEIGHT, channels);
+ let frame_bytes = f32::as_bytes(&frame);
+ let input = client.create_from_slice(frame_bytes);
+ let accum = client.empty(pixels * stored_ch * size_of::());
let weight_sum = client.empty(pixels * size_of::());
let max_weight = client.empty(pixels * size_of::());
let confidence_dummy = client.empty(size_of::());
let (spatial_offset_lut, spatial_offset_lut_len) = zero_spatial_offset_lut(client);
+
WindowInput {
input,
accum,
@@ -82,17 +85,19 @@ fn prepare_window(client: &ComputeClient, ch: u32) -> WindowInput
}
}
-fn prepare_window_ref(client: &ComputeClient, ch: u32) -> WindowRefInput {
- let pixels = (W * H) as usize;
- let stored = stored_channels(ch) as usize;
- let frame = make_padded_frame(W, H, ch);
- let input = client.create_from_slice(f32::as_bytes(&frame));
- let reference = client.create_from_slice(f32::as_bytes(&frame));
- let accum = client.empty(pixels * stored * size_of::());
+fn prepare_window_ref(client: &ComputeClient, channels: u32) -> WindowRefInput {
+ let pixels = (WIDTH * HEIGHT) as usize;
+ let stored_ch = stored_channels(channels) as usize;
+ let frame = make_padded_frame(WIDTH, HEIGHT, channels);
+ let frame_bytes = f32::as_bytes(&frame);
+ let input = client.create_from_slice(frame_bytes);
+ let reference = client.create_from_slice(frame_bytes);
+ let accum = client.empty(pixels * stored_ch * size_of::());
let weight_sum = client.empty(pixels * size_of::());
let max_weight = client.empty(pixels * size_of::());
let confidence_dummy = client.empty(size_of::());
let (spatial_offset_lut, spatial_offset_lut_len) = zero_spatial_offset_lut(client);
+
WindowRefInput {
input,
reference,
@@ -108,8 +113,8 @@ fn prepare_window_ref(client: &ComputeClient, ch: u32) -> WindowR
pub struct FusedPairWindowBench {
pub client: ComputeClient,
- pub ch: u32,
- pub ch_name: &'static str,
+ pub channels: u32,
+ pub channel_name: &'static str,
}
impl Benchmark for FusedPairWindowBench {
@@ -117,20 +122,24 @@ impl Benchmark for FusedPairWindowBench {
type Output = ();
fn prepare(&self) -> Self::Input {
- prepare_window(&self.client, self.ch)
+ prepare_window(&self.client, self.channels)
}
fn execute(&self, args: Self::Input) -> Result<(), String> {
- let pixels = (W * H) as usize;
- let stored = stored_channels(self.ch) as usize;
+ let pixels = (WIDTH * HEIGHT) as usize;
+ let stored_ch = stored_channels(self.channels) as usize;
+ let cube_count = cube_count_2d();
+ let cube_dim = cube_dim_2d();
+ let inv_norm = h2_inv_norm();
+
unsafe {
nlm_fused_pair_accumulate_window::launch_unchecked::(
&self.client,
- cube_count_2d(),
- cube_dim_2d(),
- stored,
+ cube_count,
+ cube_dim,
+ stored_ch,
ArrayArg::from_raw_parts(args.input.clone(), args.frame_len),
- ArrayArg::from_raw_parts(args.accum.clone(), pixels * stored),
+ ArrayArg::from_raw_parts(args.accum.clone(), pixels * stored_ch),
ArrayArg::from_raw_parts(args.weight_sum.clone(), pixels),
ArrayArg::from_raw_parts(args.max_weight.clone(), pixels),
ArrayArg::from_raw_parts(args.confidence_dummy.clone(), 1),
@@ -139,11 +148,11 @@ impl Benchmark for FusedPairWindowBench {
0u32,
0u32,
0u32,
- h2_inv_norm(),
+ inv_norm,
0.0f32,
- W,
- H,
- self.ch,
+ WIDTH,
+ HEIGHT,
+ self.channels,
PATCH_RADIUS,
SEARCH_RADIUS,
BLOCK_X,
@@ -153,24 +162,27 @@ impl Benchmark for FusedPairWindowBench {
1u32,
);
}
+
Ok(())
}
fn name(&self) -> String {
- format!("fused_pair_accumulate_window_1080p_{}", self.ch_name)
+ format!("fused_pair_accumulate_window_1080p_{}", self.channel_name)
}
+
fn sync(&self) {
block_sync(&self.client);
}
+
fn shapes(&self) -> Vec> {
- shapes_with_ch(self.ch)
+ shapes_with_channels(self.channels)
}
}
pub struct FusedSingleWindowBench {
pub client: ComputeClient,
- pub ch: u32,
- pub ch_name: &'static str,
+ pub channels: u32,
+ pub channel_name: &'static str,
}
impl Benchmark for FusedSingleWindowBench {
@@ -178,52 +190,59 @@ impl Benchmark for FusedSingleWindowBench {
type Output = ();
fn prepare(&self) -> Self::Input {
- prepare_window(&self.client, self.ch)
+ prepare_window(&self.client, self.channels)
}
fn execute(&self, args: Self::Input) -> Result<(), String> {
- let pixels = (W * H) as usize;
- let stored = stored_channels(self.ch) as usize;
+ let pixels = (WIDTH * HEIGHT) as usize;
+ let stored_ch = stored_channels(self.channels) as usize;
+ let cube_count = cube_count_2d();
+ let cube_dim = cube_dim_2d();
+ let inv_norm = h2_inv_norm();
+
unsafe {
nlm_fused_single_window::launch_unchecked::(
&self.client,
- cube_count_2d(),
- cube_dim_2d(),
- stored,
+ cube_count,
+ cube_dim,
+ stored_ch,
ArrayArg::from_raw_parts(args.input.clone(), args.frame_len),
- ArrayArg::from_raw_parts(args.accum.clone(), pixels * stored),
+ ArrayArg::from_raw_parts(args.accum.clone(), pixels * stored_ch),
ArrayArg::from_raw_parts(args.weight_sum.clone(), pixels),
ArrayArg::from_raw_parts(args.max_weight.clone(), pixels),
0u32,
- h2_inv_norm(),
+ inv_norm,
ArrayArg::from_raw_parts(args.spatial_offset_lut.clone(), args.spatial_offset_lut_len),
- W,
- H,
- self.ch,
+ WIDTH,
+ HEIGHT,
+ self.channels,
PATCH_RADIUS,
SEARCH_RADIUS,
BLOCK_X,
BLOCK_Y,
);
}
+
Ok(())
}
fn name(&self) -> String {
- format!("fused_single_window_1080p_{}", self.ch_name)
+ format!("fused_single_window_1080p_{}", self.channel_name)
}
+
fn sync(&self) {
block_sync(&self.client);
}
+
fn shapes(&self) -> Vec