diff --git a/Cargo.lock b/Cargo.lock index 275db7ef8ee..fd12441ad3e 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -11068,6 +11068,7 @@ dependencies = [ "rstest", "vortex-alp", "vortex-array", + "vortex-bench-support", "vortex-buffer", "vortex-error", "vortex-fastlanes", diff --git a/encodings/fastlanes/Cargo.toml b/encodings/fastlanes/Cargo.toml index 9085390b67b..dfe2fb7ea70 100644 --- a/encodings/fastlanes/Cargo.toml +++ b/encodings/fastlanes/Cargo.toml @@ -39,6 +39,7 @@ rand = { workspace = true } rstest = { workspace = true } vortex-alp = { path = "../alp" } vortex-array = { workspace = true, features = ["_test-harness"] } +vortex-bench-support = { workspace = true } vortex-fastlanes = { path = ".", features = ["_test-harness"] } [features] @@ -48,6 +49,10 @@ _test-harness = ["dep:rand"] name = "bitpacking_take" harness = false +[[bench]] +name = "bitpacking_filter" +harness = false + [[bench]] name = "canonicalize_bench" harness = false diff --git a/encodings/fastlanes/benches/bitpacking_filter.rs b/encodings/fastlanes/benches/bitpacking_filter.rs new file mode 100644 index 00000000000..298d6f52182 --- /dev/null +++ b/encodings/fastlanes/benches/bitpacking_filter.rs @@ -0,0 +1,116 @@ +// SPDX-License-Identifier: Apache-2.0 +// SPDX-FileCopyrightText: Copyright the Vortex contributors + +//! Measures selective filtering around the sparse extraction thresholds. + +#![expect(clippy::cast_possible_truncation)] +#![expect(clippy::unwrap_used)] + +use std::sync::LazyLock; + +use divan::Bencher; +use divan::counter::ItemsCount; +use vortex_array::ArrayRef; +use vortex_array::IntoArray; +use vortex_array::VortexSessionExecute; +use vortex_array::arrays::PrimitiveArray; +use vortex_array::dtype::NativePType; +use vortex_array::validity::Validity; +use vortex_buffer::BufferMut; +use vortex_fastlanes::BitPackedData; +use vortex_mask::Mask; +use vortex_session::VortexSession; + +fn main() { + divan::main(); +} + +static SESSION: LazyLock = LazyLock::new(|| { + let session = vortex_array::array_session(); + vortex_fastlanes::initialize(&session); + session +}); + +const NUM_ARRAY_CHUNKS: usize = 64; +// Keep the array density below the outer full-decode policy. +const NUM_SELECTED_CHUNKS: usize = 8; +const CHUNK_SIZE: usize = 1_024; +const LEN: usize = NUM_ARRAY_CHUNKS * CHUNK_SIZE; + +trait BenchInt: NativePType { + fn from_counter(value: u64) -> Self; +} + +macro_rules! impl_bench_int { + ($($T:ty),+) => { + $(impl BenchInt for $T { + fn from_counter(value: u64) -> Self { + value as $T + } + })+ + }; +} + +impl_bench_int!(u8, u16, u32, u64); + +fn fixture(bit_width: usize, selected_per_chunk: usize) -> (ArrayRef, Mask) { + let limit = if bit_width == 64 { + u64::MAX + } else { + 1_u64 << bit_width + }; + let values: BufferMut = (0..LEN) + .map(|index| T::from_counter(index as u64 % limit)) + .collect(); + let packed = BitPackedData::encode( + &PrimitiveArray::new(values.freeze(), Validity::NonNullable).into_array(), + bit_width as u8, + &mut SESSION.create_execution_ctx(), + ) + .unwrap() + .into_array(); + let indices = (0..NUM_SELECTED_CHUNKS).flat_map(|chunk| { + (0..selected_per_chunk) + .map(move |index| chunk * CHUNK_SIZE + index * CHUNK_SIZE / selected_per_chunk) + }); + (packed, Mask::from_indices(LEN, indices)) +} + +macro_rules! bench_width { + ($module:ident, $T:ty, $bit_width:expr, [$($selected:expr),+ $(,)?]) => { + mod $module { + use super::*; + + #[vortex_bench_support::cpu_features] + #[divan::bench(args = [$($selected),+])] + fn filter(bencher: Bencher, selected_per_chunk: usize) { + let (packed, mask) = fixture::<$T>($bit_width, selected_per_chunk); + bencher + .counter(ItemsCount::new(LEN)) + .with_inputs(|| (mask.clone(), SESSION.create_execution_ctx())) + .bench_refs(|(mask, ctx)| { + packed + .filter(mask.clone()) + .unwrap() + .execute::(ctx) + .unwrap() + }); + } + } + }; +} + +macro_rules! bench_type { + ($module:ident, $T:ty, [$(($width_module:ident, $bit_width:expr)),+ $(,)?], $selected:tt) => { + mod $module { + use super::*; + + $(bench_width!($width_module, $T, $bit_width, $selected);)+ + } + }; +} + +bench_type!(u8, u8, [(width4, 4)], [8, 16, 24]); +bench_type!(u16, u16, [(width8, 8)], [8, 32, 48]); +bench_type!(u32, u32, [(width16, 16)], [8, 64, 80]); +bench_type!(u64, u64, [(width32, 32)], [8, 160, 192]); diff --git a/encodings/fastlanes/benches/bitpacking_take.rs b/encodings/fastlanes/benches/bitpacking_take.rs index eb072017ae3..345127da247 100644 --- a/encodings/fastlanes/benches/bitpacking_take.rs +++ b/encodings/fastlanes/benches/bitpacking_take.rs @@ -7,18 +7,23 @@ use std::sync::LazyLock; use divan::Bencher; +use divan::counter::ItemsCount; use rand::RngExt; use rand::SeedableRng; use rand::distr::Uniform; use rand::prelude::StdRng; +use vortex_array::ArrayRef; use vortex_array::IntoArray as _; use vortex_array::RecursiveCanonical; use vortex_array::VortexSessionExecute; use vortex_array::arrays::PrimitiveArray; +use vortex_array::dtype::NativePType; use vortex_array::validity::Validity; use vortex_buffer::Buffer; +use vortex_buffer::BufferMut; use vortex_buffer::buffer; use vortex_fastlanes::BitPackedArrayExt; +use vortex_fastlanes::BitPackedData; use vortex_fastlanes::bitpack_compress::bitpack_to_best_bit_width; use vortex_session::VortexSession; @@ -32,6 +37,94 @@ static SESSION: LazyLock = LazyLock::new(|| { session }); +const NUM_ARRAY_CHUNKS: usize = 64; +// Keep the selected count below the outer full-decode policy. +const NUM_SELECTED_CHUNKS: usize = 8; +const CHUNK_SIZE: usize = 1_024; +const THRESHOLD_FIXTURE_LEN: usize = NUM_ARRAY_CHUNKS * CHUNK_SIZE; + +trait BenchInt: NativePType { + fn from_counter(value: u64) -> Self; +} + +macro_rules! impl_bench_int { + ($($T:ty),+) => { + $(impl BenchInt for $T { + fn from_counter(value: u64) -> Self { + value as $T + } + })+ + }; +} + +impl_bench_int!(u8, u16, u32, u64); + +fn threshold_fixture( + bit_width: usize, + selected_per_chunk: usize, +) -> (ArrayRef, ArrayRef) { + let limit = if bit_width == 64 { + u64::MAX + } else { + 1_u64 << bit_width + }; + let values: BufferMut = (0..THRESHOLD_FIXTURE_LEN) + .map(|index| T::from_counter(index as u64 % limit)) + .collect(); + let packed = BitPackedData::encode( + &PrimitiveArray::new(values.freeze(), Validity::NonNullable).into_array(), + bit_width as u8, + &mut SESSION.create_execution_ctx(), + ) + .unwrap() + .into_array(); + let indices = PrimitiveArray::from_iter((0..NUM_SELECTED_CHUNKS).flat_map(|chunk| { + (0..selected_per_chunk) + .map(move |index| (chunk * CHUNK_SIZE + index * CHUNK_SIZE / selected_per_chunk) as u32) + })) + .into_array(); + (packed, indices) +} + +macro_rules! bench_width { + ($module:ident, $T:ty, $bit_width:expr, [$($selected:expr),+ $(,)?]) => { + mod $module { + use super::*; + + #[vortex_bench_support::cpu_features] + #[divan::bench(args = [$($selected),+])] + fn threshold(bencher: Bencher, selected_per_chunk: usize) { + let (packed, indices) = threshold_fixture::<$T>($bit_width, selected_per_chunk); + bencher + .counter(ItemsCount::new(indices.len())) + .with_inputs(|| (indices.clone(), SESSION.create_execution_ctx())) + .bench_refs(|(indices, ctx)| { + packed + .take(indices.clone()) + .unwrap() + .execute::(ctx) + .unwrap() + }); + } + } + }; +} + +macro_rules! bench_type { + ($module:ident, $T:ty, [$(($width_module:ident, $bit_width:expr)),+ $(,)?], $selected:tt) => { + mod $module { + use super::*; + + $(bench_width!($width_module, $T, $bit_width, $selected);)+ + } + }; +} + +bench_type!(u8, u8, [(width4, 4)], [8, 16, 24]); +bench_type!(u16, u16, [(width8, 8)], [8, 32, 48]); +bench_type!(u32, u32, [(width16, 16)], [8, 64, 80]); +bench_type!(u64, u64, [(width32, 32)], [8, 160, 192]); + #[divan::bench] fn take_10_stratified(bencher: Bencher) { let values = fixture(65_536, 8);